From 89318bd9e431baa234248dd95bd410811662bb44 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Fri, 11 Sep 2026 12:51:55 -0400 Subject: [PATCH 1/5] feat(cli): say which storage DDL is still outstanding, and converge it MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Adds `schemabot storage diff` and `schemabot storage apply` for SchemaBot's own storage database. The diff reads the live database and compares it against the embedded schema files of the binary that answers, never against a version pin: a release tag says what that release would converge to, not what the storage converged to, and the two diverge exactly when a deploy has failed. `--deployment -e ` routes through the control plane to the data plane's own gRPC endpoint, because a data plane owns its storage database and generally sits where a workstation cannot reach it. The direct `--dsn` / `--config` path stays for when the server is down, including down because its own bootstrap is failing. Nothing falls back between the two: an unreachable deployment is an error naming it, never a report about another database. Apply is the startup bootstrap called unchanged — same differ, same destructive-statement refusal, same advisory lock — so it extends AV-9 to an operator-invoked convergence rather than deciding policy again. Both routes are admin-only and both sit at the write tier, the read-only diff included (AZ-2). Co-Authored-By: Claude Opus 5 --- docs/auth.md | 8 +- docs/configuration.md | 100 ++++ docs/invariants.md | 28 +- pkg/api/ensure_schema.go | 26 +- pkg/api/route_authorization_sweep_test.go | 7 + pkg/api/service.go | 14 + pkg/api/storage_schema.go | 483 +++++++++++++++ pkg/api/storage_schema_handlers.go | 289 +++++++++ pkg/api/storage_schema_handlers_test.go | 399 ++++++++++++ pkg/api/storage_schema_integration_test.go | 291 +++++++++ pkg/api/storage_schema_test.go | 176 ++++++ pkg/apitypes/storage_schema.go | 77 +++ pkg/auth/tiers.go | 30 +- pkg/auth/tiers_test.go | 20 + pkg/cmd/client/client.go | 45 ++ pkg/cmd/client/request.go | 24 +- pkg/cmd/commands/exit_code.go | 59 ++ pkg/cmd/commands/storage.go | 148 ++++- pkg/cmd/commands/storage_schema.go | 461 ++++++++++++++ pkg/cmd/commands/storage_schema_test.go | 420 +++++++++++++ pkg/cmd/main.go | 9 +- pkg/proto/tern.proto | 123 ++++ pkg/proto/ternv1/tern.pb.go | 666 +++++++++++++++++---- pkg/proto/ternv1/tern.pb.gw.go | 184 +++++- pkg/proto/ternv1/tern_grpc.pb.go | 150 ++++- pkg/serve/serve.go | 28 +- pkg/serve/serve_build_test.go | 4 +- pkg/serve/storage_schema.go | 136 +++++ pkg/serve/storage_schema_test.go | 119 ++++ pkg/tern/grpc_client.go | 28 + pkg/tern/server.go | 67 ++- pkg/tern/storage_schema.go | 43 ++ pkg/testutil/postgres.go | 15 + 33 files changed, 4445 insertions(+), 232 deletions(-) create mode 100644 pkg/api/storage_schema.go create mode 100644 pkg/api/storage_schema_handlers.go create mode 100644 pkg/api/storage_schema_handlers_test.go create mode 100644 pkg/api/storage_schema_integration_test.go create mode 100644 pkg/api/storage_schema_test.go create mode 100644 pkg/apitypes/storage_schema.go create mode 100644 pkg/cmd/commands/exit_code.go create mode 100644 pkg/cmd/commands/storage_schema.go create mode 100644 pkg/cmd/commands/storage_schema_test.go create mode 100644 pkg/serve/storage_schema.go create mode 100644 pkg/serve/storage_schema_test.go create mode 100644 pkg/tern/storage_schema.go diff --git a/docs/auth.md b/docs/auth.md index 757c86ae2..6c8b137bc 100644 --- a/docs/auth.md +++ b/docs/auth.md @@ -638,7 +638,7 @@ permission levels, called **tiers** in configuration and logs: | Access | Operations | |---|---| | Read | List databases, pull live schemas, view stored plans, status, progress, logs, history, and locks | -| Write | Create plans, apply changes, stop or resume work, cut over, cancel, revert, skip revert, roll back, acquire or release locks, change settings, and run check or webhook maintenance | +| Write | Create plans, apply changes, stop or resume work, cut over, cancel, revert, skip revert, roll back, acquire or release locks, change settings, run check or webhook maintenance, and inspect or converge SchemaBot's own storage schema | Creating a plan requires write access because it stages a change. Reading a plan that already exists requires only read access. @@ -650,6 +650,12 @@ to writes under `forward_auth`. In the route rules, `GET` and `HEAD` requests are reads, as is `POST /api/pull`. Other requests require write access by default. +`GET /api/storage/schema/diff` is the exception in the other direction: it +reads, but it requires write access. It reports the internal shape of +SchemaBot's own bookkeeping database, and its sibling route converges that +database, so both belong to the people who operate the server rather than to +everyone who can see the schema changes it runs. + ### Grant a team access to its database diff --git a/docs/configuration.md b/docs/configuration.md index 2b65f4728..93c55570a 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1042,6 +1042,106 @@ storage: Leave the flag false during normal operation and revert it after the removal converges. +### Ask what storage DDL is outstanding + +A deploy that did not converge leaves one question open: which storage DDL is +still outstanding. Two commands answer it, and both read the live storage +database. Neither takes a version, because a release tag says what that +release would converge to, not what the storage converged to, and the two +answers differ exactly when a deploy has failed. + +`storage diff` is read-only. It takes no lock and holds no transaction, so it +is safe at any time, including against production during an incident. + +```console +$ schemabot storage diff +schemabot (mysql) needs 3 statements: 3 outstanding, against the schema embedded in v1.2.3. + +Outstanding, and run automatically on the next boot or apply (3): + +ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; +ALTER TABLE `checks` ADD COLUMN `blocked_reason` varchar(64) NOT NULL DEFAULT '' AFTER `state`; +CREATE TABLE `check_gate_audit` ( + `id` BIGINT UNSIGNED AUTO_INCREMENT, + `check_id` BIGINT UNSIGNED NOT NULL, + PRIMARY KEY (`id`) +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci; + +Converge it with: schemabot storage apply +``` + +The statements are printed bare and one per line so a whole section can be +pasted into a client as it stands. The exit status is the machine-readable half +of the answer: `0` when the storage needs nothing, `2` when statements are +outstanding, and `1` when the read itself failed. A pre-deploy gate needs those +three apart, since "converged" and "unreachable" call for opposite decisions. + +`storage apply` converges the database by running the same bootstrap the next +boot would run: the same differ, the same refusal of destructive statements, +and the same advisory lock, so two operators running it at once serialize the +way two booting pods do. It previews the statements and prompts before running +them; `--auto-approve` (`-y`) skips the prompt for scripted maintenance. + +```console +$ schemabot storage apply +schemabot (mysql) needs 1 statement: 1 outstanding, against the schema embedded in v1.2.3. + +Outstanding, and run automatically on the next boot or apply (1): + +ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; + +Run these statements against schemabot (mysql)? Only 'yes' will be accepted: yes +Ran 1 statement against schemabot (mysql). +schemabot (mysql) is converged. +``` + +Destructive statements are refused here exactly as they are at startup, and for +the same reason: a binary older than the storage sees the newer schema's tables +as surplus. A refusal is reported rather than silently dropped, and +`--allow-destructive` opts in per invocation, widening the deployment's standing +`allow_destructive_schema_changes` policy without ever narrowing it. + +```console +$ schemabot storage diff +schemabot (mysql) needs 1 statement: 1 destructive, against the schema embedded in v1.2.3. + +Destructive, and refused; surplus state stays in place (1): + +-- check_gate_audit: DROP TABLE destroys data +DROP TABLE `check_gate_audit`; + +Converge it with: schemabot storage apply +``` + +### Reach the right storage database + +Both commands take a target, and which path applies is stated rather than +discovered. Nothing falls back from one to the other: a deployment that cannot +be reached through the API is an error naming the deployment, never a report +about a different database that happened to be reachable. + +| Target | Reads | +|---|---| +| no flags | the storage of the server the CLI is pointed at | +| `--deployment -e ` | that data plane's own storage, over the gRPC connection that already exists between the two | +| `--dsn ` or `--config ` | the storage database this workstation opens itself | + +A data plane owns its storage database and generally sits where a workstation +cannot dial it, so `--deployment` routes through the control plane: the control +plane asks the data plane, and the data plane reads its own storage with its own +embedded schema files. That is also what makes the answer trustworthy, since the +binary that reports the diff is the binary whose next boot would run it. + +The direct path exists for when the server is down, including when it is down +because its own schema bootstrap is failing. It reads the storage with *this +CLI's* embedded schema files, so run a binary of the release you are deploying. +`--dialect` states the storage family when a DSN's form does not say; it applies +only to a direct connection. + +Both routes are admin-only and both sit at the write tier, the read-only diff +included, because the diff exposes the internal shape of SchemaBot's bookkeeping +database. See [Authentication and authorization](auth.md#what-read-and-write-access-include). + ## Support Channel SchemaBot can add an opt-in support link to GitHub PR comments so authors know diff --git a/docs/invariants.md b/docs/invariants.md index 012d70489..a90d0abfd 100644 --- a/docs/invariants.md +++ b/docs/invariants.md @@ -312,17 +312,18 @@ clamp (`pkg/webhook/plan_drift.go`); the request body limit (`pkg/webhook/handle ### AV-9: SchemaBot never destroys its own storage to start -The startup schema bootstrap converges SchemaBot's own storage additively, and decides before it -writes. On MySQL a destructive statement (a `DROP TABLE`, or an `ALTER TABLE` carrying a `DROP -COLUMN`) is refused unless destructive storage changes are explicitly allowed, and a statement -whose destructive clauses cannot be partitioned out is refused *whole*. Refusing the whole -statement runs strictly less than any split of it, so the fallback can never widen what the -bootstrap executes, and startup continues on the safe remainder. On PostgreSQL the convergence is -additive-only and gates on the entire drift set before touching anything, so a change needing -manual remediation aborts the pass rather than leaving storage half-converged. *Breaks if -violated:* the first instance of a rolling deploy drops state the rest of the fleet is still -reading. *Enforced:* the per-dialect bootstrappers (`pkg/api/ensure_schema.go`, -`pkg/api/ensure_schema_postgres.go`). +Every convergence of SchemaBot's own storage — at startup, or on an operator's command — is +additive, and decides before it writes. On MySQL a destructive statement (a `DROP TABLE`, or an +`ALTER TABLE` carrying a `DROP COLUMN`) is refused unless destructive storage changes are +explicitly allowed, and a statement whose destructive clauses cannot be partitioned out is refused +*whole*. Refusing the whole statement runs strictly less than any split of it, so the fallback can +never widen what the bootstrap executes, and startup continues on the safe remainder. On +PostgreSQL the convergence is additive-only and gates on the entire drift set before touching +anything, so a change needing manual remediation aborts the pass rather than leaving storage +half-converged. *Breaks if violated:* the first instance of a rolling deploy drops state the rest +of the fleet is still reading. *Enforced:* the per-dialect bootstrappers +(`pkg/api/ensure_schema.go`, `pkg/api/ensure_schema_postgres.go`), which the operator-facing +storage schema surface calls rather than reimplements (`pkg/api/storage_schema.go`). ### AV-10: Anything the PR can do, the CLI can do @@ -1326,8 +1327,9 @@ to another. *Enforced:* server-side routing (`pkg/tern/target_router.go`) and so An API route not classified as a read is treated as a write. Planning counts as a write, since it stages a change. A target that cannot be resolved never authorizes. A configured grant that could never match any request is a startup error rather than silent dead config. And a new mutating -endpoint cannot ship without a test proving it denies unauthorized callers. *Enforced:* route -classification with a structural sweep test over the route table (`pkg/api/service.go`). +endpoint cannot ship without a test proving it denies unauthorized callers. *Enforced:* the tier +classification (`pkg/auth/tiers.go`), with a structural sweep test over the route table +(`pkg/api/service.go`). ### AZ-3: Identity comes from a verified lane diff --git a/pkg/api/ensure_schema.go b/pkg/api/ensure_schema.go index 1c8af8095..921d1a1f4 100644 --- a/pkg/api/ensure_schema.go +++ b/pkg/api/ensure_schema.go @@ -106,13 +106,7 @@ func WithPostgresStatementTimeout(d time.Duration) EnsureSchemaOption { // adding a dialect means adding a bootstrapper here, not threading // dialect-conditionals through the MySQL flow. func EnsureSchema(dsn string, logger *slog.Logger, opts ...EnsureSchemaOption) error { - o := ensureSchemaOptions{ - dialect: schema.DialectMySQL, - postgresStatementTimeout: DefaultPostgresStatementTimeout, - } - for _, opt := range opts { - opt(&o) - } + o := newEnsureSchemaOptions(opts...) switch o.dialect { case schema.DialectMySQL: return ensureMySQLSchema(dsn, logger, o, namedlock.MySQL{}) @@ -189,7 +183,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l // Fast path: plan without a lock. If no changes, return immediately. // This is the common case (99% of deploys) and avoids lock overhead. planResult, err := eng.Plan(ctx, &engine.PlanRequest{ - Database: "schemabot", + Database: storageSchemaNamespace, SchemaFiles: schemaFiles, Credentials: &engine.Credentials{DSN: dsn}, }) @@ -247,7 +241,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l // removed above. eng = spirit.New(spirit.Config{Logger: spiritLogger}) planResult, err = eng.Plan(ctx, &engine.PlanRequest{ - Database: "schemabot", + Database: storageSchemaNamespace, SchemaFiles: schemaFiles, Credentials: &engine.Credentials{DSN: dsn}, }) @@ -272,7 +266,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l } if len(allowed) == 0 { logger.Warn("all planned storage schema changes are destructive and refused; storage schema left unchanged", - "database", "schemabot", + "database", storageSchemaNamespace, "refused_count", len(refused), ) return nil @@ -293,7 +287,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l // Apply all DDL via Spirit (starts async schema change) applyStart := time.Now() _, err = eng.Apply(ctx, &engine.ApplyRequest{ - Database: "schemabot", + Database: storageSchemaNamespace, Changes: changes, Credentials: &engine.Credentials{DSN: dsn}, }) @@ -308,7 +302,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l for { progress, err := eng.Progress(ctx, &engine.ProgressRequest{ - Database: "schemabot", + Database: storageSchemaNamespace, Credentials: &engine.Credentials{DSN: dsn}, }) if err != nil { @@ -327,7 +321,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l // log search. Include the DDL count and the underlying message so a // failed bootstrap is triageable from the message line alone. logger.Error("storage schema change failed; SchemaBot storage will not initialize", - "database", "schemabot", + "database", storageSchemaNamespace, "ddl_count", len(tableChanges), "error", progress.ErrorMessage, ) @@ -359,7 +353,7 @@ func ensureMySQLSchema(dsn string, logger *slog.Logger, o ensureSchemaOptions, l // online DDL) instead of surfacing a bare "context canceled" from the driver. func ensureSchemaTimeoutError(ctx context.Context, ddlCount int, logger *slog.Logger) error { logger.Error("storage schema change did not complete before EnsureSchemaTimeout; SchemaBot storage will not initialize", - "database", "schemabot", + "database", storageSchemaNamespace, "timeout", EnsureSchemaTimeout, "ddl_count", ddlCount, ) @@ -388,7 +382,7 @@ type refusedStorageChange struct { // unsplittable ALTER carries the error that prevented the split. func (r refusedStorageChange) refusalTelemetry() (scope, message string, attrs []any) { attrs = []any{ - "database", "schemabot", + "database", storageSchemaNamespace, "table", r.change.Table, "operation", ddl.StatementTypeToOp(r.change.Operation), "reason", r.reason, @@ -547,7 +541,7 @@ func readEmbeddedSchemaFiles() (schema.SchemaFiles, error) { } return schema.SchemaFiles{ - "schemabot": &schema.Namespace{Files: files}, + storageSchemaNamespace: &schema.Namespace{Files: files}, }, nil } diff --git a/pkg/api/route_authorization_sweep_test.go b/pkg/api/route_authorization_sweep_test.go index 510ee5246..b521f1abf 100644 --- a/pkg/api/route_authorization_sweep_test.go +++ b/pkg/api/route_authorization_sweep_test.go @@ -111,6 +111,13 @@ func TestMutatingRoutesDenyScopedOperatorByDefault(t *testing.T) { "POST /api/checks/synthesize": `{}`, "POST /api/checks/repos": `{}`, "POST /api/webhooks/redrive": `{}`, + // The storage schema routes take no database — they are about + // SchemaBot's own bookkeeping database — so a scoped operator is denied + // on the admin requirement itself rather than on a target outside their + // grant. The diff route is a GET and still appears here, because + // auth.TierForRequest admits it at the write tier. + "GET /api/storage/schema/diff": ``, + "POST /api/storage/schema/apply": `{}`, } svc := New(st, scopedWriteConfig(), nil, slog.New(slog.DiscardHandler)) diff --git a/pkg/api/service.go b/pkg/api/service.go index f3e342831..a4b0ee151 100644 --- a/pkg/api/service.go +++ b/pkg/api/service.go @@ -217,6 +217,14 @@ type Service struct { pendingObserverMu sync.Mutex pendingObservers map[pendingObserverKey]tern.ProgressObserver + + // storageSchemaService answers the storage schema routes for this server's + // own storage database. An embedder registers it with + // SetStorageSchemaService once it has resolved the storage DSN and dialect + // it booted with; it is nil in builds that never resolve one, and the + // routes refuse rather than guess at a database. + storageSchemaMu sync.RWMutex + storageSchemaService tern.StorageSchemaService } // SetApplyObserver sets a progress observer on the tern client for an apply. @@ -839,6 +847,12 @@ func (s *Service) apiRoutes() []apiRoute { {"GET /api/locks/{database}/{dbtype}", s.handleLockGet}, {"GET /api/locks", s.handleLockList}, + // Storage schema API (SchemaBot's own bookkeeping database). Both + // routes are admin-only and both are admitted at the write tier — + // including the read-only diff, see auth.TierForRequest. + {"GET /api/storage/schema/diff", s.handleStorageSchemaDiff}, + {"POST /api/storage/schema/apply", s.handleStorageSchemaApply}, + // Settings API {"GET /api/settings", s.handleSettingsList}, {"GET /api/settings/{key}", s.handleSettingsGet}, diff --git a/pkg/api/storage_schema.go b/pkg/api/storage_schema.go new file mode 100644 index 000000000..826c90d25 --- /dev/null +++ b/pkg/api/storage_schema.go @@ -0,0 +1,483 @@ +package api + +import ( + "context" + "fmt" + "log/slog" + "time" + + "github.com/block/schemabot/pkg/apitypes" + "github.com/block/schemabot/pkg/ddl" + "github.com/block/schemabot/pkg/engine" + "github.com/block/schemabot/pkg/engine/spirit" + "github.com/block/schemabot/pkg/postgresconn" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/schema" + "github.com/block/spirit/pkg/utils" +) + +// Storage-schema inspection answers one question an operator has during a +// deploy that did not converge: which storage DDL is still outstanding, right +// now, on this database. +// +// It answers from the live database and from the embedded schema files of the +// binary that serves the request — never from a version pin. A consumer's +// go.mod pin says which release a host binary *was built against*; it says +// nothing about what the storage it talks to has actually converged to, and the +// two diverge exactly when a deploy has failed to converge. Since that is +// precisely when someone computes this diff, the pin is the one input that +// cannot be trusted, so no input to this package is a version. +// +// The diff is the bootstrap's own diff, not a second implementation of it. On +// MySQL that is Spirit's differ over readEmbeddedSchemaFiles; on PostgreSQL it +// is postgresSchemaDriftFor over the embedded PostgreSQL files. A statement +// this reports is a statement a boot of this binary would plan, because it came +// from the same call. + +// StorageSchemaStatement is one outstanding storage-schema statement, with the +// classification the bootstrap would apply to it. +type StorageSchemaStatement struct { + // Table is the storage table the statement acts on. + Table string + // Operation is the statement's kind, in the vocabulary the storage layer + // uses for it (create_table, alter_table, add_column, create_index, ...). + Operation string + // DDL is the statement itself, runnable as printed. + DDL string + // Reason is why the statement is classified destructive, or why it needs + // manual remediation. Empty for a statement that runs automatically. + Reason string +} + +// StorageSchemaReport is what the storage schema of one database needs in order +// to match the embedded schema of the binary that produced the report. +// +// The three statement sets are disjoint and have different dispositions, so an +// operator reading the report never has to work out which statements would +// actually run: +// +// Outstanding runs on the next boot, and on apply +// Destructive refused unless destructive changes are explicitly allowed +// Manual blocks the whole convergence until an operator resolves it +type StorageSchemaReport struct { + // Dialect is the storage database's family. + Dialect schema.Dialect + // Database is the live database the diff read, as the server reports it — + // so a report cannot be misread as being about a different database. + Database string + // Version is the SchemaBot version of the binary whose embedded schema + // files produced the diff. It is reported, never consumed: the diff is + // computed from the files themselves, and this only says whose files they + // were. + Version string + // Outstanding lists the statements that converge the schema and run + // automatically, in the order the convergence would run them. + Outstanding []StorageSchemaStatement + // Destructive lists the statements the bootstrap classifies as destroying + // data. They are refused unless destructive changes are explicitly + // allowed; DestructiveAllowed says which of the two this database is in. + Destructive []StorageSchemaStatement + // DestructiveAllowed reports whether the destructive statements would + // actually run. It reflects the effective policy for the request — the + // storage config's allowance, or an explicit per-request opt-in. + DestructiveAllowed bool + // Manual lists changes that cannot run automatically, each naming the + // situation and the remediation. Any entry aborts convergence before a + // single statement executes, so an apply is refused while one is present. + Manual []StorageSchemaStatement +} + +// Converged reports whether the storage schema needs nothing at all. A report +// with only refused destructive statements is not converged: the surplus state +// stays in place deliberately (AV-9), and saying otherwise would tell an +// operator the database matches this binary's schema when it does not. +func (r *StorageSchemaReport) Converged() bool { + return len(r.Outstanding) == 0 && len(r.Destructive) == 0 && len(r.Manual) == 0 +} + +// StorageSchemaDiffTimeout bounds a read-only storage-schema diff. It is far +// below EnsureSchemaTimeout because a diff does no DDL: it reads the live +// catalog and compares. Bounding it separately keeps an unreachable storage +// database from holding an operator's request — or a control-plane request +// thread — for the whole bootstrap budget. +const StorageSchemaDiffTimeout = 30 * time.Second + +// DiffStorageSchema reports the storage DDL outstanding between the embedded +// schema files of this binary and the live storage database at dsn. It is +// strictly read-only: it opens connections, reads the catalog, and computes a +// diff. It executes no DDL, takes no advisory lock, and writes nothing, so it +// is safe to run at any time, including against a database an apply is +// converging right now. +// +// The dialect selects the differ, mirroring EnsureSchema's dispatch, and fails +// closed for a dialect without one rather than running another family's +// catalog queries. Pass the same options EnsureSchema is wired with so the +// report describes what a boot would decide; +// WithAllowDestructiveSchemaChanges only labels the report here, since a diff +// executes nothing either way. +func DiffStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, opts ...EnsureSchemaOption) (*StorageSchemaReport, error) { + o := newEnsureSchemaOptions(opts...) + switch o.dialect { + case schema.DialectMySQL: + return diffMySQLStorageSchema(ctx, dsn, o) + case schema.DialectPostgres: + return diffPostgresStorageSchema(ctx, dsn, o) + default: + return nil, fmt.Errorf("no storage schema differ for storage dialect %q (supported: %q, %q)", o.dialect, schema.DialectMySQL, schema.DialectPostgres) + } +} + +// ApplyStorageSchema converges the storage database at dsn and reports what it +// found and what it left behind. +// +// The convergence is the startup bootstrap, called unchanged: the same differ, +// the same destructive-change refusal, the same advisory lock serializing it +// against every other instance and against another operator running this at +// the same time. Nothing about the policy is re-decided here — an operator +// command that converged storage differently from a boot would be a second +// implementation of the one path that must not have two. +// +// Two reports bracket the run, because "what happened" and "what is left" are +// different questions and an operator mid-incident needs both: +// +// planned what was outstanding before, including refusals +// remaining what is still outstanding after +// +// remaining is empty on a clean convergence. It carries the refused +// destructive statements when the database holds state this binary's schema +// does not declare, which is the expected steady state during a rollback +// (AV-9) rather than a failure. +func ApplyStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, opts ...EnsureSchemaOption) (planned, remaining *StorageSchemaReport, err error) { + planned, err = DiffStorageSchema(ctx, dsn, logger, opts...) + if err != nil { + return nil, nil, fmt.Errorf("diff storage schema before converging it: %w", err) + } + if len(planned.Manual) > 0 { + // The convergence would refuse the whole drift set anyway. Returning + // here reports every problem at once with its remediation, instead of + // surfacing the bootstrap's joined error string as an opaque failure. + logger.Warn("refusing to converge storage schema because changes need manual remediation", + "dialect", planned.Dialect, + "database", planned.Database, + "manual_count", len(planned.Manual), + "outstanding_count", len(planned.Outstanding), + ) + return planned, planned, nil + } + if planned.Converged() { + logger.Info("storage schema already converged; nothing to apply", + "dialect", planned.Dialect, "database", planned.Database) + return planned, planned, nil + } + + logger.Info("converging storage schema on operator request", + "dialect", planned.Dialect, + "database", planned.Database, + "outstanding_count", len(planned.Outstanding), + "destructive_count", len(planned.Destructive), + "destructive_allowed", planned.DestructiveAllowed, + ) + if err := EnsureSchema(dsn, logger, opts...); err != nil { + return planned, nil, fmt.Errorf("converge storage schema on database %q (%s): %w", planned.Database, planned.Dialect, err) + } + + remaining, err = DiffStorageSchema(ctx, dsn, logger, opts...) + if err != nil { + // The convergence succeeded; only the confirming read failed. Report + // that distinctly — an operator must not read a failed verification as + // a failed apply and run it again looking for a different answer. + return planned, nil, fmt.Errorf("storage schema converged on database %q (%s) but re-reading it to confirm failed: %w", planned.Database, planned.Dialect, err) + } + logger.Info("storage schema convergence complete", + "dialect", remaining.Dialect, + "database", remaining.Database, + "applied_count", len(planned.Outstanding), + "remaining_count", len(remaining.Outstanding)+len(remaining.Destructive), + ) + return planned, remaining, nil +} + +// diffMySQLStorageSchema diffs the embedded MySQL schema files against the live +// storage database with Spirit's differ — the same Plan call ensureMySQLSchema +// makes, so the two cannot disagree about what a boot would run. Spirit emits +// one combined ALTER per table, and partitionDestructiveChanges splits it the +// way the bootstrap would, so a mixed ALTER is reported as the additive clauses +// that run plus the destructive clauses that are refused, not as one statement +// whose disposition an operator has to guess. +func diffMySQLStorageSchema(ctx context.Context, dsn string, o ensureSchemaOptions) (*StorageSchemaReport, error) { + report := &StorageSchemaReport{Dialect: schema.DialectMySQL, DestructiveAllowed: o.allowDestructive} + + // The database name is what makes the report readable as being about one + // database. Unlike the bootstrap's preamble, a failure here is fatal: the + // bootstrap can converge without knowing the name, but a report that cannot + // say which database it read is one an operator cannot act on. + diag, err := diagnoseStorageTarget(ctx, dsn) + if err != nil { + return nil, fmt.Errorf("read storage target identity: %w", err) + } + report.Database = diag.database + + schemaFiles, err := readEmbeddedSchemaFiles() + if err != nil { + return nil, err + } + + eng := spirit.New(spirit.Config{Logger: slog.New(slog.DiscardHandler)}) + planResult, err := eng.Plan(ctx, &engine.PlanRequest{ + Database: storageSchemaNamespace, + SchemaFiles: schemaFiles, + Credentials: &engine.Credentials{DSN: dsn}, + }) + if err != nil { + return nil, fmt.Errorf("plan storage schema against database %q: %w", diag.database, err) + } + if planResult.NoChanges { + return report, nil + } + + allowed, refused, err := partitionDestructiveChanges(planResult.Changes) + if err != nil { + return nil, fmt.Errorf("classify storage schema changes on database %q: %w", diag.database, err) + } + for _, tc := range flatTableChanges(allowed) { + operation, err := storageSchemaOperation(tc.Operation) + if err != nil { + return nil, fmt.Errorf("classify storage schema statement on table %q of database %q: %w", tc.Table, diag.database, err) + } + report.Outstanding = append(report.Outstanding, StorageSchemaStatement{ + Table: tc.Table, + Operation: operation, + DDL: tc.DDL, + }) + } + for _, r := range refused { + operation, err := storageSchemaOperation(r.change.Operation) + if err != nil { + return nil, fmt.Errorf("classify refused storage schema statement on table %q of database %q: %w", r.change.Table, diag.database, err) + } + report.Destructive = append(report.Destructive, StorageSchemaStatement{ + Table: r.change.Table, + Operation: operation, + DDL: r.change.DDL, + Reason: r.reportedReason(), + }) + } + return report, nil +} + +// A report names a statement's kind in one vocabulary whichever dialect the +// storage runs on, so an operator reading two deployments' reports — or a job +// parsing them — does not have to learn two. The PostgreSQL convergence +// already speaks it (postgresOpCreateTable and its siblings); the MySQL differ +// speaks in statement types, and storageSchemaOperation translates. +const ( + storageSchemaOpCreateTable = postgresOpCreateTable + storageSchemaOpAlterTable = "alter_table" + storageSchemaOpDropTable = "drop_table" +) + +// storageSchemaOperation names a MySQL statement type in the report's +// vocabulary. A type the storage schema cannot contain is an error rather than +// an "unknown" label: the storage schema is a fixed set of tables, so a +// RENAME or a view arriving here means the differ saw something this package +// does not understand, and labelling it would hide that in a report an +// operator is about to act on. +func storageSchemaOperation(t ddl.StatementType) (string, error) { + switch t { + case ddl.StatementCreateTable: + return storageSchemaOpCreateTable, nil + case ddl.StatementAlterTable: + return storageSchemaOpAlterTable, nil + case ddl.StatementDropTable: + return storageSchemaOpDropTable, nil + default: + return "", fmt.Errorf("unexpected statement type %q in a storage schema diff; the storage schema is converged with CREATE TABLE, ALTER TABLE and DROP TABLE only", t) + } +} + +// storageSchemaNamespace is the namespace the storage schema files declare. +// EnsureSchema passes the same value to the engine; naming it once keeps the +// diff and the bootstrap from drifting apart on a literal. +const storageSchemaNamespace = "schemabot" + +// diffPostgresStorageSchema diffs the embedded PostgreSQL schema files against +// the live storage database with the additive convergence's own drift scan, so +// the report is exactly what ensurePostgresSchema would decide. The convergence +// never drops or alters an existing object, so the report has no destructive +// set; what it does have is the manual-remediation set, whose entries abort a +// whole convergence pass rather than being skipped. +func diffPostgresStorageSchema(ctx context.Context, dsn string, o ensureSchemaOptions) (*StorageSchemaReport, error) { + report := &StorageSchemaReport{Dialect: schema.DialectPostgres} + + tables, files, err := readEmbeddedPostgresSchemaFiles() + if err != nil { + return nil, err + } + + db, err := postgresconn.Open(dsn, postgresconn.WithStatementTimeout(o.postgresStatementTimeout)) + if err != nil { + return nil, fmt.Errorf("open storage database: %w", err) + } + defer utils.CloseAndLog(db) + if err := db.PingContext(ctx); err != nil { + return nil, fmt.Errorf("ping storage database: %w", err) + } + if err := db.QueryRowContext(ctx, "SELECT current_database()").Scan(&report.Database); err != nil { + return nil, fmt.Errorf("read storage target identity: %w", err) + } + + drift, err := postgresSchemaDriftFor(ctx, db, tables, files) + if err != nil { + return nil, fmt.Errorf("inspect storage schema on database %q: %w", report.Database, err) + } + // Walk tables rather than the drift map so the statements come out in the + // order the convergence would run them, which is the order they are safe + // to paste and run by hand. + for _, table := range tables { + for _, change := range drift[table] { + statement := StorageSchemaStatement{ + Table: table, + Operation: change.operation, + DDL: change.ddl, + } + if change.manualReason == "" { + report.Outstanding = append(report.Outstanding, statement) + continue + } + statement.Reason = postgresManualProblem(table, change) + report.Manual = append(report.Manual, statement) + } + } + return report, nil +} + +// reportedReason is the refusal reason for an operator-facing report. A split +// refusal and a whole refusal both carry Spirit's classification; a refusal +// that happened because the clauses could not be partitioned says so, because +// that is the difference between "these clauses are refused" and "none of this +// statement ran". +func (r refusedStorageChange) reportedReason() string { + if r.splitErr != nil { + return fmt.Sprintf("%s; refused whole because its clauses could not be partitioned: %v", r.reason, r.splitErr) + } + return r.reason +} + +// APIType converts a report to the HTTP response shape, which the CLI renders +// from. Deployment and Environment are left to the caller: only whoever routed +// the request knows which storage was asked for, and a report has to name the +// storage the operator meant rather than whichever one answered. +func (r *StorageSchemaReport) APIType() *apitypes.StorageSchemaReport { + if r == nil { + return nil + } + return &apitypes.StorageSchemaReport{ + Dialect: string(r.Dialect), + Database: r.Database, + Version: r.Version, + Converged: r.Converged(), + Outstanding: storageSchemaStatementsAPIType(r.Outstanding), + Destructive: storageSchemaStatementsAPIType(r.Destructive), + DestructiveAllowed: r.DestructiveAllowed, + Manual: storageSchemaStatementsAPIType(r.Manual), + } +} + +func storageSchemaStatementsAPIType(statements []StorageSchemaStatement) []apitypes.StorageSchemaStatement { + if len(statements) == 0 { + return nil + } + out := make([]apitypes.StorageSchemaStatement, 0, len(statements)) + for _, s := range statements { + out = append(out, apitypes.StorageSchemaStatement{ + Table: s.Table, + Operation: s.Operation, + DDL: s.DDL, + Reason: s.Reason, + }) + } + return out +} + +// StorageSchemaReportProto converts a report to its wire form, for a data +// plane answering the control plane's storage-schema RPC. +func StorageSchemaReportProto(r *StorageSchemaReport) *ternv1.StorageSchemaReport { + if r == nil { + return nil + } + return &ternv1.StorageSchemaReport{ + Dialect: string(r.Dialect), + Database: r.Database, + Version: r.Version, + Outstanding: storageSchemaStatementsProto(r.Outstanding), + Destructive: storageSchemaStatementsProto(r.Destructive), + DestructiveAllowed: r.DestructiveAllowed, + Manual: storageSchemaStatementsProto(r.Manual), + } +} + +// StorageSchemaReportFromProto converts a report back from its wire form, for +// a control plane rendering what a data plane reported. A nil message yields a +// nil report: an RPC that answered with no report at all is a different +// condition from one that reported convergence, and the caller decides which +// error to raise rather than having "converged" invented here. +func StorageSchemaReportFromProto(p *ternv1.StorageSchemaReport) *StorageSchemaReport { + if p == nil { + return nil + } + return &StorageSchemaReport{ + Dialect: schema.Dialect(p.GetDialect()), + Database: p.GetDatabase(), + Version: p.GetVersion(), + Outstanding: storageSchemaStatementsFromProto(p.GetOutstanding()), + Destructive: storageSchemaStatementsFromProto(p.GetDestructive()), + DestructiveAllowed: p.GetDestructiveAllowed(), + Manual: storageSchemaStatementsFromProto(p.GetManual()), + } +} + +func storageSchemaStatementsProto(statements []StorageSchemaStatement) []*ternv1.StorageSchemaStatement { + if len(statements) == 0 { + return nil + } + out := make([]*ternv1.StorageSchemaStatement, 0, len(statements)) + for _, s := range statements { + out = append(out, &ternv1.StorageSchemaStatement{ + Table: s.Table, + Operation: s.Operation, + Ddl: s.DDL, + Reason: s.Reason, + }) + } + return out +} + +func storageSchemaStatementsFromProto(statements []*ternv1.StorageSchemaStatement) []StorageSchemaStatement { + if len(statements) == 0 { + return nil + } + out := make([]StorageSchemaStatement, 0, len(statements)) + for _, s := range statements { + out = append(out, StorageSchemaStatement{ + Table: s.GetTable(), + Operation: s.GetOperation(), + DDL: s.GetDdl(), + Reason: s.GetReason(), + }) + } + return out +} + +// newEnsureSchemaOptions applies opts over the defaults every entry point +// shares, so DiffStorageSchema and EnsureSchema start from the same policy for +// an option a caller did not set. +func newEnsureSchemaOptions(opts ...EnsureSchemaOption) ensureSchemaOptions { + o := ensureSchemaOptions{ + dialect: schema.DialectMySQL, + postgresStatementTimeout: DefaultPostgresStatementTimeout, + } + for _, opt := range opts { + opt(&o) + } + return o +} diff --git a/pkg/api/storage_schema_handlers.go b/pkg/api/storage_schema_handlers.go new file mode 100644 index 000000000..031b1f9fe --- /dev/null +++ b/pkg/api/storage_schema_handlers.go @@ -0,0 +1,289 @@ +// storage_schema_handlers.go answers "which storage DDL is outstanding, right +// now" for SchemaBot's own storage — the control plane's, or a data plane's. +// +// The question turns up during a deploy that did not converge, and there the +// obvious way to answer it is wrong. Reading a consumer's go.mod pin, checking +// out that release tag, and diffing its embedded schema files tells you what +// that release *would* converge to; it does not tell you what the storage +// actually converged to, and the two answers differ exactly when a deploy has +// failed. Since that is the only time anyone asks, a version is never an input +// here: the diff is computed by the binary that is running, against the live +// catalog, in one call. +// +// That is also why a data plane's storage is reached through the data plane +// rather than dialed from the control plane. Its storage database usually sits +// where neither an operator's workstation nor the control plane can open a +// connection, and the binary that owns it is the only one that can say what its +// own embedded schema declares. The gRPC link that already exists between the +// planes carries the question to it. +package api + +import ( + "encoding/json" + "errors" + "fmt" + "io" + "net/http" + "strings" + + "github.com/block/schemabot/pkg/apitypes" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/tern" +) + +// SetStorageSchemaService registers the answerer for this server's own storage +// schema. An embedder supplies one bound to the storage DSN and dialect it +// booted with — the same adapter its gRPC endpoint serves — so the answer a +// local request gets and the answer a peer control plane gets over gRPC come +// from one implementation rather than two that can drift. +// +// Without one the storage schema routes refuse rather than guessing at a DSN, +// because a guess here reads the wrong database and reports it as the right +// one. +func (s *Service) SetStorageSchemaService(service tern.StorageSchemaService) { + s.storageSchemaMu.Lock() + defer s.storageSchemaMu.Unlock() + s.storageSchemaService = service +} + +func (s *Service) localStorageSchemaService() tern.StorageSchemaService { + s.storageSchemaMu.RLock() + defer s.storageSchemaMu.RUnlock() + return s.storageSchemaService +} + +// storageSchemaTarget is the resolved answer to "whose storage is this request +// about". Resolution is explicit and total: every request either names one +// instance's storage or is refused. Nothing here falls back from an +// unreachable deployment to a reachable database — reporting the control +// plane's storage to an operator who asked about a data plane's would be a +// wrong answer dressed as a right one, and during an incident it is the kind +// of wrong answer that gets acted on. +type storageSchemaTarget struct { + // deployment and environment are empty for the control plane's own + // storage, and set for a data plane's. + deployment string + environment string + service tern.StorageSchemaService +} + +// resolveStorageSchemaTarget decides which instance answers for the request. +// +// An empty deployment means the server the request was made to: its storage is +// the one storage the request can reach without routing, and asking for it +// needs no deployment because there is nothing to choose between. +// +// A named deployment must have a configured gRPC endpoint for the environment. +// The endpoint is checked before the client is resolved, and deliberately not +// inferred from the client that comes back: TernClient prefers an in-process +// client whenever the name also matches a locally configured database, so +// resolving the client first and then asking what it turned out to be is how a +// request for a data plane's storage silently becomes a report about the +// control plane's. +func (s *Service) resolveStorageSchemaTarget(deployment, environment string) (*storageSchemaTarget, error) { + deployment = strings.TrimSpace(deployment) + environment = strings.TrimSpace(environment) + + if deployment == "" { + if environment != "" { + return nil, fmt.Errorf("environment %q was given without a deployment: an environment selects which of a deployment's endpoints to reach, so name the deployment too, or omit both to read this server's own storage", environment) + } + service := s.localStorageSchemaService() + if service == nil { + return nil, fmt.Errorf("this server does not expose its own storage schema: it was built without a storage schema service, so there is no storage it can name; name a deployment to read a data plane's storage instead") + } + return &storageSchemaTarget{service: service}, nil + } + if environment == "" { + return nil, fmt.Errorf("deployment %q needs an environment: a deployment serves one endpoint per environment, so there is no single storage to read without one", deployment) + } + + if _, err := s.config.TernDeployments.Endpoint(deployment, environment); err != nil { + return nil, fmt.Errorf("no data plane configured for deployment %q in environment %q: the storage schema of a data plane is read through its gRPC endpoint, so add it under tern_deployments.%s.%s in the server config, or omit the deployment to read this server's own storage: %w", + deployment, environment, deployment, environment, err) + } + client, err := s.TernClient(deployment, environment) + if err != nil { + return nil, fmt.Errorf("resolve data plane client for deployment %q environment %q: %w", deployment, environment, err) + } + service, ok := client.(tern.StorageSchemaService) + if !ok { + // A configured endpoint that resolves to an in-process client means + // the routing config is ambiguous, not that the storage is local. + // Saying so beats reporting this server's storage under the + // deployment's name. + return nil, fmt.Errorf("deployment %q environment %q has a configured data plane endpoint but resolves to an in-process client (%T), so its storage cannot be read remotely; check that no locally configured database shares the name %q", + deployment, environment, client, deployment) + } + return &storageSchemaTarget{deployment: deployment, environment: environment, service: service}, nil +} + +// handleStorageSchemaDiff is the HTTP handler for +// GET /api/storage/schema/diff. +// +// It is a GET because it only reads: it plans nothing, stores nothing, and +// takes no lock, so it is safe to call repeatedly against production while an +// incident is in progress. It is nonetheless admitted at the write tier and +// gated on admin membership — see storageSchemaOperation below — because what +// it returns is the internal shape of SchemaBot's own bookkeeping database, +// and because its sibling route converges that database. +func (s *Service) handleStorageSchemaDiff(w http.ResponseWriter, r *http.Request) { + query := r.URL.Query() + deployment := query.Get("deployment") + environment := query.Get("environment") + allowDestructive := query.Get("allow_destructive") == "true" + + if !s.authorizeStorageSchemaOperation(w, r, storageSchemaDiffOperation) { + return + } + target, err := s.resolveStorageSchemaTarget(deployment, environment) + if err != nil { + s.logger.Warn("rejecting storage schema diff because its target could not be resolved", + "deployment", deployment, "environment", environment, "error", err) + s.writeError(w, http.StatusBadRequest, err.Error()) + return + } + + resp, err := target.service.StorageSchemaDiff(r.Context(), &ternv1.StorageSchemaDiffRequest{ + AllowDestructive: allowDestructive, + }) + if err != nil { + s.logger.Error("storage schema diff failed", + "deployment", target.deployment, "environment", target.environment, "error", err) + s.writeError(w, http.StatusInternalServerError, fmt.Sprintf("storage schema diff failed: %v", err)) + return + } + report := storageSchemaReportResponse(target, resp.GetReport()) + if report == nil { + s.logger.Error("storage schema diff returned no report", + "deployment", target.deployment, "environment", target.environment) + s.writeError(w, http.StatusInternalServerError, "storage schema diff returned no report") + return + } + s.writeJSON(w, http.StatusOK, apitypes.StorageSchemaDiffResponse{Report: report}) +} + +// handleStorageSchemaApply is the HTTP handler for +// POST /api/storage/schema/apply. It runs the target instance's startup +// bootstrap, under the advisory lock that bootstrap already takes, so two +// operators running it at once serialize the same way two booting pods do. +func (s *Service) handleStorageSchemaApply(w http.ResponseWriter, r *http.Request) { + req, err := decodeStorageSchemaApplyRequest(r) + if err != nil { + s.writeBodyDecodeError(w, err) + return + } + if !s.authorizeStorageSchemaOperation(w, r, storageSchemaApplyOperation) { + return + } + target, err := s.resolveStorageSchemaTarget(req.Deployment, req.Environment) + if err != nil { + s.logger.Warn("rejecting storage schema apply because its target could not be resolved", + "deployment", req.Deployment, "environment", req.Environment, "error", err) + s.writeError(w, http.StatusBadRequest, err.Error()) + return + } + + operator := resolveCaller(r.Context(), req.Caller) + s.logger.Info("converging storage schema on operator request", + "deployment", target.deployment, + "environment", target.environment, + "allow_destructive", req.AllowDestructive, + "caller", operator) + + resp, err := target.service.StorageSchemaApply(r.Context(), &ternv1.StorageSchemaApplyRequest{ + AllowDestructive: req.AllowDestructive, + Caller: operator, + }) + if err != nil { + s.logger.Error("storage schema apply failed", + "deployment", target.deployment, + "environment", target.environment, + "allow_destructive", req.AllowDestructive, + "caller", operator, + "error", err) + s.writeError(w, http.StatusInternalServerError, fmt.Sprintf("storage schema apply failed: %v", err)) + return + } + planned := storageSchemaReportResponse(target, resp.GetPlanned()) + remaining := storageSchemaReportResponse(target, resp.GetRemaining()) + if planned == nil || remaining == nil { + // Both halves are required to say what happened: without the pair, an + // operator cannot tell a convergence that finished from one that left + // statements behind, which is the only thing this answer is for. + s.logger.Error("storage schema apply returned an incomplete result", + "deployment", target.deployment, + "environment", target.environment, + "has_planned", planned != nil, + "has_remaining", remaining != nil) + s.writeError(w, http.StatusInternalServerError, "storage schema apply returned an incomplete result; check the target's logs for whether it converged") + return + } + s.writeJSON(w, http.StatusOK, apitypes.StorageSchemaApplyResponse{Planned: planned, Remaining: remaining}) +} + +// storageSchemaReportResponse converts one wire report to its HTTP form, +// stamping the target it describes. Stamping here rather than at the source is +// deliberate: only the control plane knows which route it asked, and an +// operator reading a report needs it to name the storage they meant, not just +// the storage that answered. +func storageSchemaReportResponse(target *storageSchemaTarget, report *ternv1.StorageSchemaReport) *apitypes.StorageSchemaReport { + response := StorageSchemaReportFromProto(report).APIType() + if response == nil { + return nil + } + response.Deployment = target.deployment + response.Environment = target.environment + return response +} + +// The storage schema operations' names in the authorization decision metric +// and denial logs. +const ( + storageSchemaDiffOperation = "storage_schema_diff" + storageSchemaApplyOperation = "storage_schema_apply" +) + +// authorizeStorageSchemaOperation gates both storage schema routes on admin +// membership and reports whether the request may proceed. +// +// Admin-only, with no scoped lane, because there is nothing to scope to. A +// database operator grant authorizes an operator for their own database's +// schema changes; SchemaBot's storage database is not any team's database, it +// is the instance's own bookkeeping — the same class of operation as changing +// deployment settings or redriving webhooks, which are admin-only for the same +// reason. +// +// This is the second of two gates and it is not the one that usually bites. +// The first is the tier the auth middleware admits the route at, and both +// routes are classified write there (auth.TierForRequest), including the +// read-only diff. That classification is what makes the admin requirement real +// on a deployment whose whole authorization model is read groups and write +// groups: the handler-level scoped-write decision is a pass-through until some +// database configures operator_groups, so a route left on the read tier would +// be readable by every reader no matter what this function said. +func (s *Service) authorizeStorageSchemaOperation(w http.ResponseWriter, r *http.Request, operation string) bool { + return s.authorizeDirectAdminWrite(w, r, operation) +} + +// decodeStorageSchemaApplyRequest decodes the apply body, tolerating an empty +// one. Every field is optional — the defaults name this server's own storage +// and refuse destructive statements — so a caller sending no body at all gets +// the safe defaults rather than a decode error. Unknown fields are still +// rejected, so a misspelled "deployment" cannot quietly become a convergence +// of the wrong storage. +func decodeStorageSchemaApplyRequest(r *http.Request) (apitypes.StorageSchemaApplyRequest, error) { + var req apitypes.StorageSchemaApplyRequest + if r.Body == nil { + return req, nil + } + decoder := json.NewDecoder(r.Body) + decoder.DisallowUnknownFields() + if err := decoder.Decode(&req); err != nil { + if errors.Is(err, io.EOF) { + return apitypes.StorageSchemaApplyRequest{}, nil + } + return apitypes.StorageSchemaApplyRequest{}, err + } + return req, nil +} diff --git a/pkg/api/storage_schema_handlers_test.go b/pkg/api/storage_schema_handlers_test.go new file mode 100644 index 000000000..2926a4efc --- /dev/null +++ b/pkg/api/storage_schema_handlers_test.go @@ -0,0 +1,399 @@ +package api + +import ( + "context" + "encoding/json" + "log/slog" + "net/http" + "net/http/httptest" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/apitypes" + "github.com/block/schemabot/pkg/auth" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/schema" + "github.com/block/schemabot/pkg/tern" +) + +// fakeStorageSchemaService answers the storage schema RPCs for one instance's +// storage, capturing what it was asked so a test can assert the request the +// control plane forwarded rather than only the response it rendered. +type fakeStorageSchemaService struct { + diffReq *ternv1.StorageSchemaDiffRequest + diffResp *ternv1.StorageSchemaDiffResponse + diffErr error + + applyReq *ternv1.StorageSchemaApplyRequest + applyResp *ternv1.StorageSchemaApplyResponse + applyErr error +} + +func (f *fakeStorageSchemaService) StorageSchemaDiff(_ context.Context, req *ternv1.StorageSchemaDiffRequest) (*ternv1.StorageSchemaDiffResponse, error) { + f.diffReq = req + return f.diffResp, f.diffErr +} + +func (f *fakeStorageSchemaService) StorageSchemaApply(_ context.Context, req *ternv1.StorageSchemaApplyRequest) (*ternv1.StorageSchemaApplyResponse, error) { + f.applyReq = req + return f.applyResp, f.applyErr +} + +// storageSchemaTernClient is a tern client that also answers the storage schema +// RPCs, the way a GRPCClient to a data plane does. +type storageSchemaTernClient struct { + *mockTernClient + *fakeStorageSchemaService +} + +var ( + _ tern.StorageSchemaService = (*fakeStorageSchemaService)(nil) + _ tern.Client = (*storageSchemaTernClient)(nil) + _ tern.StorageSchemaService = (*storageSchemaTernClient)(nil) +) + +func storageSchemaReportMessage(database string, outstanding ...string) *ternv1.StorageSchemaReport { + report := &ternv1.StorageSchemaReport{ + Dialect: string(schema.DialectMySQL), + Database: database, + Version: "v0.1.0", + } + for _, table := range outstanding { + report.Outstanding = append(report.Outstanding, &ternv1.StorageSchemaStatement{ + Table: table, + Operation: storageSchemaOpAlterTable, + Ddl: "ALTER TABLE `" + table + "` ADD COLUMN `superseded_by` varchar(255) NOT NULL DEFAULT ''", + }) + } + return report +} + +// storageSchemaRoutingConfig configures one data plane endpoint, so a request +// naming it routes rather than being refused as unconfigured. +func storageSchemaRoutingConfig() *ServerConfig { + return &ServerConfig{ + TernDeployments: TernConfig{ + "west": TernEndpoints{"production": "west.example:9090"}, + }, + } +} + +func newStorageSchemaService(t *testing.T, cfg *ServerConfig) *Service { + t.Helper() + return New(nil, cfg, nil, slog.New(slog.DiscardHandler)) +} + +func storageSchemaDiffRequest(t *testing.T, svc *Service, query string) *httptest.ResponseRecorder { + t.Helper() + mux := http.NewServeMux() + svc.ConfigureRoutes(mux) + path := "/api/storage/schema/diff" + if query != "" { + path += "?" + query + } + req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, path, nil) + rec := httptest.NewRecorder() + mux.ServeHTTP(rec, req) + return rec +} + +func storageSchemaApplyRequest(t *testing.T, svc *Service, body string) *httptest.ResponseRecorder { + t.Helper() + mux := http.NewServeMux() + svc.ConfigureRoutes(mux) + req := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/api/storage/schema/apply", strings.NewReader(body)) + rec := httptest.NewRecorder() + mux.ServeHTTP(rec, req) + return rec +} + +func decodeDiffResponse(t *testing.T, rec *httptest.ResponseRecorder) apitypes.StorageSchemaDiffResponse { + t.Helper() + var response apitypes.StorageSchemaDiffResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &response), "body: %s", rec.Body.String()) + require.NotNil(t, response.Report) + return response +} + +// A request that names no deployment reads the storage of the server it was +// made to, and the report says which database that was. +func TestHandleStorageSchemaDiff_ReadsThisServersStorage(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + local := &fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("schemabot_storage", "applies")}, + } + svc.SetStorageSchemaService(local) + + rec := storageSchemaDiffRequest(t, svc, "") + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + report := decodeDiffResponse(t, rec).Report + assert.Empty(t, report.Deployment, "this server's own storage is not a deployment") + assert.Empty(t, report.Environment) + assert.Equal(t, "schemabot_storage", report.Database) + assert.Equal(t, string(schema.DialectMySQL), report.Dialect) + assert.False(t, report.Converged) + require.Len(t, report.Outstanding, 1) + assert.Equal(t, "applies", report.Outstanding[0].Table) + assert.Contains(t, report.Outstanding[0].DDL, "ADD COLUMN") + assert.False(t, local.diffReq.GetAllowDestructive(), "the default must not opt into destructive statements") +} + +// A converged storage database reports converged, which is the answer a +// pre-deploy check is looking for. +func TestHandleStorageSchemaDiff_ReportsConverged(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + svc.SetStorageSchemaService(&fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("schemabot_storage")}, + }) + + rec := storageSchemaDiffRequest(t, svc, "") + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + report := decodeDiffResponse(t, rec).Report + assert.True(t, report.Converged) + assert.Empty(t, report.Outstanding) +} + +// allow_destructive reaches the instance that answers, so the report says what +// an apply with the same flag would do rather than what the default would. +func TestHandleStorageSchemaDiff_ForwardsAllowDestructive(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + local := &fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("schemabot_storage")}, + } + svc.SetStorageSchemaService(local) + + rec := storageSchemaDiffRequest(t, svc, "allow_destructive=true") + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.True(t, local.diffReq.GetAllowDestructive()) +} + +// A server with no storage schema service refuses rather than guessing at a +// storage DSN: a guess reads the wrong database and reports it as the right +// one. +func TestHandleStorageSchemaDiff_RefusesWithoutLocalService(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + + rec := storageSchemaDiffRequest(t, svc, "") + require.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "does not expose its own storage schema") + assert.Contains(t, rec.Body.String(), "name a deployment") +} + +// A named deployment is read through its own endpoint, and the report is +// stamped with the deployment the operator asked about — not only with whatever +// database answered. +func TestHandleStorageSchemaDiff_ReadsDataPlaneStorage(t *testing.T) { + svc := newStorageSchemaService(t, storageSchemaRoutingConfig()) + svc.SetStorageSchemaService(&fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("control_plane_storage")}, + }) + remote := &fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("west_storage", "apply_operations")}, + } + svc.RegisterTernClient("west", "production", &storageSchemaTernClient{ + mockTernClient: &mockTernClient{isRemote: true}, + fakeStorageSchemaService: remote, + }) + + rec := storageSchemaDiffRequest(t, svc, "deployment=west&environment=production") + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + report := decodeDiffResponse(t, rec).Report + assert.Equal(t, "west", report.Deployment) + assert.Equal(t, "production", report.Environment) + assert.Equal(t, "west_storage", report.Database, "the data plane's storage, not the control plane's") + require.Len(t, report.Outstanding, 1) + assert.Equal(t, "apply_operations", report.Outstanding[0].Table) + assert.NotNil(t, remote.diffReq, "the data plane must be the one asked") +} + +// A deployment with no configured endpoint is an error naming what was looked +// under. It never falls back to the storage of the server that took the +// request: reporting a converged control plane to an operator asking about an +// unreachable data plane is a wrong answer that reads like a right one. +func TestHandleStorageSchemaDiff_UnconfiguredDeploymentDoesNotFallBack(t *testing.T) { + svc := newStorageSchemaService(t, storageSchemaRoutingConfig()) + local := &fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("control_plane_storage")}, + } + svc.SetStorageSchemaService(local) + + rec := storageSchemaDiffRequest(t, svc, "deployment=east&environment=production") + require.Equal(t, http.StatusBadRequest, rec.Code) + body := rec.Body.String() + assert.Contains(t, body, "no data plane configured for deployment") + assert.Contains(t, body, "east") + assert.Contains(t, body, "production") + assert.Contains(t, body, "tern_deployments.east.production") + assert.Nil(t, local.diffReq, "the control plane's own storage must not answer for a data plane") +} + +// A deployment whose endpoint is configured but whose client resolves +// in-process is a routing ambiguity, not a local read. Saying so beats +// reporting this server's storage under the deployment's name. +func TestHandleStorageSchemaDiff_RefusesNonRoutableDeployment(t *testing.T) { + svc := newStorageSchemaService(t, storageSchemaRoutingConfig()) + svc.SetStorageSchemaService(&fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("control_plane_storage")}, + }) + svc.RegisterTernClient("west", "production", &mockTernClient{}) + + rec := storageSchemaDiffRequest(t, svc, "deployment=west&environment=production") + require.Equal(t, http.StatusBadRequest, rec.Code) + assert.Contains(t, rec.Body.String(), "in-process client") + assert.Contains(t, rec.Body.String(), "west") +} + +// Naming half a target is refused with the half that is missing, on both +// halves: a deployment with no environment has no single endpoint, and an +// environment with no deployment selects nothing. +func TestHandleStorageSchemaDiff_RefusesHalfNamedTarget(t *testing.T) { + svc := newStorageSchemaService(t, storageSchemaRoutingConfig()) + svc.SetStorageSchemaService(&fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("control_plane_storage")}, + }) + + deploymentOnly := storageSchemaDiffRequest(t, svc, "deployment=west") + require.Equal(t, http.StatusBadRequest, deploymentOnly.Code) + assert.Contains(t, deploymentOnly.Body.String(), "needs an environment") + + environmentOnly := storageSchemaDiffRequest(t, svc, "environment=production") + require.Equal(t, http.StatusBadRequest, environmentOnly.Code) + assert.Contains(t, environmentOnly.Body.String(), "without a deployment") +} + +// A convergence returns both halves — what was outstanding and what is left — +// and attributes the run to the caller, so the target's own logs name a person. +func TestHandleStorageSchemaApply_ConvergesThisServersStorage(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + local := &fakeStorageSchemaService{ + applyResp: &ternv1.StorageSchemaApplyResponse{ + Planned: storageSchemaReportMessage("schemabot_storage", "applies"), + Remaining: storageSchemaReportMessage("schemabot_storage"), + }, + } + svc.SetStorageSchemaService(local) + + rec := storageSchemaApplyRequest(t, svc, `{"caller":"cli:operator@workstation"}`) + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + + var response apitypes.StorageSchemaApplyResponse + require.NoError(t, json.Unmarshal(rec.Body.Bytes(), &response)) + require.NotNil(t, response.Planned) + require.NotNil(t, response.Remaining) + assert.Len(t, response.Planned.Outstanding, 1) + assert.True(t, response.Remaining.Converged) + assert.Equal(t, "cli:operator@workstation", local.applyReq.GetCaller()) + assert.False(t, local.applyReq.GetAllowDestructive()) +} + +// An empty body converges the storage of the server the request was made to, +// with destructive statements refused: every field is optional and the defaults +// are the safe ones. +func TestHandleStorageSchemaApply_EmptyBodyUsesSafeDefaults(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + local := &fakeStorageSchemaService{ + applyResp: &ternv1.StorageSchemaApplyResponse{ + Planned: storageSchemaReportMessage("schemabot_storage"), + Remaining: storageSchemaReportMessage("schemabot_storage"), + }, + } + svc.SetStorageSchemaService(local) + + rec := storageSchemaApplyRequest(t, svc, "") + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + require.NotNil(t, local.applyReq) + assert.False(t, local.applyReq.GetAllowDestructive()) +} + +// A misspelled field is refused rather than ignored: a dropped "deployment" +// would converge the wrong storage database. +func TestHandleStorageSchemaApply_RefusesUnknownField(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + local := &fakeStorageSchemaService{} + svc.SetStorageSchemaService(local) + + rec := storageSchemaApplyRequest(t, svc, `{"deploymnet":"west"}`) + require.Equal(t, http.StatusBadRequest, rec.Code) + assert.Nil(t, local.applyReq, "nothing may converge on a request that was not understood") +} + +// An incomplete answer from the target is an error, not a convergence: without +// both halves there is no way to tell a run that finished from one that left +// statements behind. +func TestHandleStorageSchemaApply_RefusesIncompleteResult(t *testing.T) { + svc := newStorageSchemaService(t, &ServerConfig{}) + svc.SetStorageSchemaService(&fakeStorageSchemaService{ + applyResp: &ternv1.StorageSchemaApplyResponse{ + Planned: storageSchemaReportMessage("schemabot_storage", "applies"), + }, + }) + + rec := storageSchemaApplyRequest(t, svc, `{}`) + require.Equal(t, http.StatusInternalServerError, rec.Code) + assert.Contains(t, rec.Body.String(), "incomplete result") +} + +// Both routes are admin-only, and the read-only diff is no exception. A scoped +// database operator is denied on it even though it is a GET, because the tier +// rule admits it as a write — the storage database is SchemaBot's own +// bookkeeping, not any team's database, so there is nothing for a per-database +// grant to scope to. +func TestStorageSchemaRoutes_DenyScopedOperator(t *testing.T) { + svc := newStorageSchemaService(t, scopedWriteConfig()) + local := &fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("schemabot_storage")}, + applyResp: &ternv1.StorageSchemaApplyResponse{ + Planned: storageSchemaReportMessage("schemabot_storage"), + Remaining: storageSchemaReportMessage("schemabot_storage"), + }, + } + svc.SetStorageSchemaService(local) + + mux := http.NewServeMux() + svc.ConfigureRoutes(mux) + operator := auth.WithUser(t.Context(), &auth.User{Subject: "bob", Groups: []string{"payments-team"}}) + + for _, route := range []struct { + method string + path string + body string + }{ + {http.MethodGet, "/api/storage/schema/diff", ""}, + {http.MethodPost, "/api/storage/schema/apply", `{}`}, + } { + t.Run(route.method+" "+route.path, func(t *testing.T) { + req := httptest.NewRequestWithContext(operator, route.method, route.path, strings.NewReader(route.body)) + rec := httptest.NewRecorder() + mux.ServeHTTP(rec, req) + assert.Equal(t, http.StatusForbidden, rec.Code, rec.Body.String()) + assert.Contains(t, rec.Body.String(), "write group") + }) + } + assert.Nil(t, local.diffReq, "a denied request must not read the storage database") + assert.Nil(t, local.applyReq, "a denied request must not converge the storage database") +} + +// An admin is allowed on both routes under the same configuration that denies +// the scoped operator, so the denial above is the grant working rather than the +// routes being closed to everyone. +func TestStorageSchemaRoutes_AllowAdmin(t *testing.T) { + svc := newStorageSchemaService(t, scopedWriteConfig()) + svc.SetStorageSchemaService(&fakeStorageSchemaService{ + diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("schemabot_storage")}, + }) + + mux := http.NewServeMux() + svc.ConfigureRoutes(mux) + admin := auth.WithUser(t.Context(), &auth.User{Subject: "alice", Groups: []string{"schema-admins"}}) + req := httptest.NewRequestWithContext(admin, http.MethodGet, "/api/storage/schema/diff", nil) + rec := httptest.NewRecorder() + mux.ServeHTTP(rec, req) + + require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) + assert.True(t, decodeDiffResponse(t, rec).Report.Converged) +} diff --git a/pkg/api/storage_schema_integration_test.go b/pkg/api/storage_schema_integration_test.go new file mode 100644 index 000000000..cc560790c --- /dev/null +++ b/pkg/api/storage_schema_integration_test.go @@ -0,0 +1,291 @@ +//go:build integration + +package api + +import ( + "database/sql" + "log/slog" + "os" + "strings" + "testing" + + _ "github.com/block/mysql" + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/schema" + "github.com/block/schemabot/pkg/testutil" +) + +func storageSchemaTestLogger() *slog.Logger { + return slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelDebug})) +} + +// statementFor returns the statement a report holds for one table, so an +// assertion names the table it is about instead of indexing into a slice whose +// order would change with the schema. +func statementFor(t *testing.T, statements []StorageSchemaStatement, table string) StorageSchemaStatement { + t.Helper() + for _, statement := range statements { + if statement.Table == table { + return statement + } + } + require.Failf(t, "no statement for table", "table %q not among %d statements: %v", table, len(statements), statementTables(statements)) + return StorageSchemaStatement{} +} + +func statementTables(statements []StorageSchemaStatement) []string { + tables := make([]string, 0, len(statements)) + for _, statement := range statements { + tables = append(tables, statement.Table) + } + return tables +} + +// An empty MySQL storage database needs its whole schema, and the diff says so +// without touching it: every embedded table is reported as a CREATE TABLE, and +// the database is still empty afterwards. This is the read an operator runs +// first during a deploy that did not converge, so it has to be safe to run +// against a database in any state. +func TestDiffStorageSchemaMySQL_EmptyDatabaseNeedsEveryTable(t *testing.T) { + sdb, db := openEnsureSchemaDatabase(t) + + report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + + assert.Equal(t, schema.DialectMySQL, report.Dialect) + assert.Equal(t, sdb.Name, report.Database) + assert.False(t, report.Converged(), "an empty storage database cannot be converged") + assert.Empty(t, report.Destructive) + assert.Empty(t, report.Manual) + + files, err := readEmbeddedSchemaFiles() + require.NoError(t, err) + require.Contains(t, files, storageSchemaNamespace) + assert.Len(t, report.Outstanding, len(files[storageSchemaNamespace].Files), + "every embedded schema file's table should be outstanding on an empty database") + + applies := statementFor(t, report.Outstanding, "applies") + assert.Equal(t, "create_table", applies.Operation) + assert.Contains(t, applies.DDL, "CREATE TABLE") + assert.Contains(t, applies.DDL, "`applies`") + assert.Empty(t, applies.Reason, "a CREATE TABLE is neither destructive nor manual") + + // The diff executed nothing: the tables it reported are still absent. + assert.False(t, testutil.TableExists(t, db, sdb.Name, "applies"), + "a diff must not create the tables it reports") +} + +// Converging a storage database with the storage apply command is the startup +// bootstrap: afterwards the database matches the embedded schema, the report +// names what ran, and a second diff finds nothing. This is the pre-deploy +// convergence step, so it has to leave the database in exactly the state a +// boot would. +func TestApplyStorageSchemaMySQL_ConvergesEmptyDatabase(t *testing.T) { + sdb, db := openEnsureSchemaDatabase(t) + + planned, remaining, err := ApplyStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + + assert.NotEmpty(t, planned.Outstanding, "an empty database has statements to run") + assert.True(t, remaining.Converged(), "nothing should be outstanding after a convergence: %v", statementTables(remaining.Outstanding)) + assert.Equal(t, sdb.Name, remaining.Database) + + for _, table := range []string{"applies", "tasks", "plans", "locks", "checks", "settings", "apply_operations"} { + assert.True(t, testutil.TableExists(t, db, sdb.Name, table), "table %s missing after convergence", table) + } + + // The confirming read is the same read an operator would run next. + after, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + assert.True(t, after.Converged()) + assert.Empty(t, after.Outstanding) +} + +// A storage database that a deploy left one column short reports exactly that +// one statement, naming the column. This is the incident the command exists +// for: the answer has to come from the live catalog, because a release tag +// would say the column is present. +func TestDiffStorageSchemaMySQL_ReportsMissingColumn(t *testing.T) { + sdb, db := openEnsureSchemaDatabase(t) + require.NoError(t, EnsureSchema(sdb.DSN, storageSchemaTestLogger())) + + // Put the database back into the state a partly converged deploy leaves: + // the table is there, one column the embedded schema declares is not. + _, err := db.ExecContext(t.Context(), "ALTER TABLE `applies` DROP COLUMN `deployment`") + require.NoError(t, err, "drop a column the embedded schema declares") + require.False(t, testutil.ColumnExists(t, db, sdb.Name, "applies", "deployment")) + + report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + + assert.False(t, report.Converged()) + assert.Empty(t, report.Destructive, "restoring a missing column destroys nothing") + assert.Empty(t, report.Manual) + require.Len(t, report.Outstanding, 1, "only the one table diverges: %v", statementTables(report.Outstanding)) + + statement := report.Outstanding[0] + assert.Equal(t, "applies", statement.Table) + assert.Equal(t, "alter_table", statement.Operation) + assert.Contains(t, statement.DDL, "ADD COLUMN") + assert.Contains(t, statement.DDL, "`deployment`") + + // Applying the reported statement is what converges it, and nothing else + // is left behind. + _, remaining, err := ApplyStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + assert.True(t, remaining.Converged(), "the reported statement should be the whole difference") + assert.True(t, testutil.ColumnExists(t, db, sdb.Name, "applies", "deployment")) +} + +// A storage database holding a table the embedded schema does not declare — +// what an older binary sees after a rollback — reports the DROP TABLE as +// refused, not as outstanding, and leaves the table in place. Converging under +// that refusal is a success: the surplus state is deliberate, and destroying it +// would take newer schema state away from the release that is about to be +// rolled forward again (AV-9). +func TestDiffStorageSchemaMySQL_RefusesSurplusTable(t *testing.T) { + sdb, db := openEnsureSchemaDatabase(t) + require.NoError(t, EnsureSchema(sdb.DSN, storageSchemaTestLogger())) + + _, err := db.ExecContext(t.Context(), + "CREATE TABLE `newer_release_state` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci") + require.NoError(t, err, "create a table a newer release would own") + + report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + + assert.False(t, report.Converged(), "a refused statement is not convergence") + assert.False(t, report.DestructiveAllowed) + assert.Empty(t, report.Outstanding) + require.Len(t, report.Destructive, 1) + statement := report.Destructive[0] + assert.Equal(t, "newer_release_state", statement.Table) + assert.Equal(t, "drop_table", statement.Operation) + assert.Contains(t, statement.DDL, "DROP TABLE") + assert.NotEmpty(t, statement.Reason, "a refusal has to say why it was refused") + + planned, remaining, err := ApplyStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err, "a convergence that refuses destructive statements still succeeds") + assert.Len(t, planned.Destructive, 1) + assert.Len(t, remaining.Destructive, 1, "the refused statement is still outstanding afterwards") + assert.True(t, testutil.TableExists(t, db, sdb.Name, "newer_release_state"), + "a refused DROP TABLE must leave the table in place") + + // The same report with destructive changes allowed says the statement would + // run, which is what the operator opting in is asking to be told. + allowed, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger(), WithAllowDestructiveSchemaChanges(true)) + require.NoError(t, err) + assert.True(t, allowed.DestructiveAllowed) + require.Len(t, allowed.Destructive, 1) + assert.Equal(t, "newer_release_state", allowed.Destructive[0].Table) +} + +// A diff is safe to run against a storage database another process is +// converging: it takes no advisory lock, so it answers while the bootstrap +// holds one rather than blocking behind it. During an incident that is the +// difference between reading the state and waiting on it. +func TestDiffStorageSchemaMySQL_ReadsWhileBootstrapHoldsLock(t *testing.T) { + sdb, db := openEnsureSchemaDatabase(t) + require.NoError(t, EnsureSchema(sdb.DSN, storageSchemaTestLogger())) + + // Hold the bootstrap's advisory lock on a dedicated session, the way a + // converging pod does. + conn, err := db.Conn(t.Context()) + require.NoError(t, err) + defer func() { _ = conn.Close() }() + var acquired int + require.NoError(t, conn.QueryRowContext(t.Context(), "SELECT GET_LOCK(?, 5)", ensureSchemaLockName).Scan(&acquired)) + require.Equal(t, 1, acquired, "hold the bootstrap lock") + defer func() { + var released sql.NullInt64 + _ = conn.QueryRowContext(t.Context(), "SELECT RELEASE_LOCK(?)", ensureSchemaLockName).Scan(&released) + }() + + report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err, "a diff must not wait on the bootstrap lock") + assert.True(t, report.Converged()) +} + +// An empty PostgreSQL storage database reports every embedded table as a +// CREATE TABLE, and converging it leaves nothing outstanding. The additive +// convergence never drops anything, so the report has no destructive set at +// all on this dialect. +func TestDiffStorageSchemaPostgres_ConvergesEmptyDatabase(t *testing.T) { + dsn, db := startPostgresStorage(t) + logger := storageSchemaTestLogger() + postgres := WithDialect(schema.DialectPostgres) + + report, err := DiffStorageSchema(t.Context(), dsn, logger, postgres) + require.NoError(t, err) + assert.Equal(t, schema.DialectPostgres, report.Dialect) + assert.Equal(t, "schemabot", report.Database) + assert.False(t, report.Converged()) + assert.Empty(t, report.Destructive, "the PostgreSQL convergence is additive-only") + assert.Empty(t, report.Manual) + + tables, _, err := readEmbeddedPostgresSchemaFiles() + require.NoError(t, err) + assert.Len(t, report.Outstanding, len(tables)) + applies := statementFor(t, report.Outstanding, "applies") + assert.Equal(t, postgresOpCreateTable, applies.Operation) + assert.Contains(t, strings.ToUpper(applies.DDL), "CREATE TABLE") + + planned, remaining, err := ApplyStorageSchema(t.Context(), dsn, logger, postgres) + require.NoError(t, err) + assert.Len(t, planned.Outstanding, len(tables)) + assert.True(t, remaining.Converged(), "outstanding after convergence: %v", statementTables(remaining.Outstanding)) + requireStorageTables(t, db) +} + +// A PostgreSQL storage database missing one column reports everything that +// restores it — the ADD COLUMN, and the index that went with the column — each +// naming what it acts on, and the convergence runs exactly those. +func TestDiffStorageSchemaPostgres_ReportsMissingColumn(t *testing.T) { + dsn, db := startPostgresStorage(t) + logger := storageSchemaTestLogger() + postgres := WithDialect(schema.DialectPostgres) + require.NoError(t, EnsureSchema(dsn, logger, postgres)) + + // Dropping the column takes the index that covers it with it, which is the + // state a partly converged deploy leaves: both are outstanding, and the + // report has to name both or the convergence looks incomplete afterwards. + _, err := db.ExecContext(t.Context(), `ALTER TABLE "applies" DROP COLUMN "deployment"`) + require.NoError(t, err, "drop a column the embedded schema declares") + + report, err := DiffStorageSchema(t.Context(), dsn, logger, postgres) + require.NoError(t, err) + assert.False(t, report.Converged()) + assert.Empty(t, report.Manual) + require.Len(t, report.Outstanding, 2, "the column and its index diverge: %v", report.Outstanding) + + column := report.Outstanding[0] + assert.Equal(t, "applies", column.Table) + assert.Equal(t, postgresOpAddColumn, column.Operation) + assert.Contains(t, strings.ToUpper(column.DDL), "ADD COLUMN") + assert.Contains(t, column.DDL, "deployment") + assert.Empty(t, column.Reason, "a metadata-only column converges automatically") + + index := report.Outstanding[1] + assert.Equal(t, "applies", index.Table) + assert.Equal(t, postgresOpCreateIndex, index.Operation) + assert.Contains(t, strings.ToUpper(index.DDL), "CREATE INDEX") + assert.Contains(t, index.DDL, "deployment") + + _, remaining, err := ApplyStorageSchema(t.Context(), dsn, logger, postgres) + require.NoError(t, err) + assert.True(t, remaining.Converged(), "outstanding after convergence: %v", remaining.Outstanding) + assert.True(t, testutil.PostgresColumnExists(t, db, "public", "applies", "deployment")) +} + +// A storage dialect with no differ fails closed rather than running another +// family's catalog queries against it, and the refusal names both the dialect +// asked for and the ones that exist. +func TestDiffStorageSchema_UnsupportedDialectFailsClosed(t *testing.T) { + _, err := DiffStorageSchema(t.Context(), "unused", storageSchemaTestLogger(), WithDialect(schema.Dialect("sqlite"))) + require.Error(t, err) + assert.Contains(t, err.Error(), "sqlite") + assert.Contains(t, err.Error(), string(schema.DialectMySQL)) + assert.Contains(t, err.Error(), string(schema.DialectPostgres)) +} diff --git a/pkg/api/storage_schema_test.go b/pkg/api/storage_schema_test.go new file mode 100644 index 000000000..35ab3528a --- /dev/null +++ b/pkg/api/storage_schema_test.go @@ -0,0 +1,176 @@ +package api + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/ddl" + "github.com/block/schemabot/pkg/engine" + "github.com/block/schemabot/pkg/schema" +) + +// A report is converged only when nothing at all is outstanding. Refused +// destructive statements and manual entries both mean the database does not +// match this binary's schema, and reporting convergence there would tell an +// operator the opposite of what is true. +func TestStorageSchemaReport_Converged(t *testing.T) { + statement := StorageSchemaStatement{Table: "applies", Operation: storageSchemaOpAlterTable, DDL: "ALTER TABLE `applies` ADD COLUMN `caller` varchar(255) NOT NULL DEFAULT ''"} + + tests := []struct { + name string + report StorageSchemaReport + converged bool + }{ + {"nothing outstanding", StorageSchemaReport{Database: "schemabot"}, true}, + {"outstanding statement", StorageSchemaReport{Outstanding: []StorageSchemaStatement{statement}}, false}, + {"refused destructive statement", StorageSchemaReport{Destructive: []StorageSchemaStatement{statement}}, false}, + {"destructive statement that would run", StorageSchemaReport{Destructive: []StorageSchemaStatement{statement}, DestructiveAllowed: true}, false}, + {"manual remediation", StorageSchemaReport{Manual: []StorageSchemaStatement{statement}}, false}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + assert.Equal(t, tc.converged, tc.report.Converged()) + }) + } +} + +// A report crosses the wire to a control plane and back without losing +// anything: the control plane renders what the data plane decided, so a field +// dropped in conversion would be a statement an operator never sees. +func TestStorageSchemaReport_ProtoRoundTrip(t *testing.T) { + report := &StorageSchemaReport{ + Dialect: schema.DialectPostgres, + Database: "schemabot", + Version: "v0.1.67", + Outstanding: []StorageSchemaStatement{{ + Table: "apply_operations", + Operation: postgresOpAddColumn, + DDL: `ALTER TABLE "apply_operations" ADD COLUMN "operation_kind" varchar(32) NOT NULL DEFAULT 'work'`, + }}, + Destructive: []StorageSchemaStatement{{ + Table: "vitess_tasks", + Operation: storageSchemaOpDropTable, + DDL: "DROP TABLE `vitess_tasks`", + Reason: "DROP TABLE destroys data", + }}, + DestructiveAllowed: true, + Manual: []StorageSchemaStatement{{ + Table: "checks", + Operation: postgresOpAddColumn, + DDL: `ALTER TABLE "checks" ADD COLUMN "head_sha" varchar(64) NOT NULL`, + Reason: "column is NOT NULL without a DEFAULT", + }}, + } + + round := StorageSchemaReportFromProto(StorageSchemaReportProto(report)) + require.NotNil(t, round) + assert.Equal(t, report, round) +} + +// A nil wire report converts to a nil report rather than to an empty one. An +// RPC that answered with no report at all is a different condition from one +// that reported convergence, and inventing convergence here would turn a +// broken answer into a green light. +func TestStorageSchemaReport_NilProtoIsNotConvergence(t *testing.T) { + assert.Nil(t, StorageSchemaReportProto(nil)) + assert.Nil(t, StorageSchemaReportFromProto(nil)) + + var report *StorageSchemaReport + assert.Nil(t, report.APIType()) +} + +// The HTTP shape carries the statements and the convergence verdict, so the +// CLI renders the same answer the differ produced without recomputing it. +func TestStorageSchemaReport_APIType(t *testing.T) { + report := &StorageSchemaReport{ + Dialect: schema.DialectMySQL, + Database: "schemabot_storage", + Version: "v0.1.67", + Destructive: []StorageSchemaStatement{{ + Table: "newer_release_state", + Operation: storageSchemaOpDropTable, + DDL: "DROP TABLE `newer_release_state`", + Reason: "DROP TABLE destroys data", + }}, + } + + converted := report.APIType() + require.NotNil(t, converted) + assert.Equal(t, "mysql", converted.Dialect) + assert.Equal(t, "schemabot_storage", converted.Database) + assert.Equal(t, "v0.1.67", converted.Version) + assert.False(t, converted.Converged, "a refused destructive statement is not convergence") + assert.False(t, converted.DestructiveAllowed) + assert.Empty(t, converted.Outstanding) + require.Len(t, converted.Destructive, 1) + assert.Equal(t, "newer_release_state", converted.Destructive[0].Table) + assert.Equal(t, storageSchemaOpDropTable, converted.Destructive[0].Operation) + assert.Equal(t, "DROP TABLE `newer_release_state`", converted.Destructive[0].DDL) + assert.Equal(t, "DROP TABLE destroys data", converted.Destructive[0].Reason) +} + +// A statement's kind is named in one vocabulary whichever dialect produced it, +// so two deployments' reports read and parse alike. +func TestStorageSchemaOperation(t *testing.T) { + create, err := storageSchemaOperation(ddl.StatementCreateTable) + require.NoError(t, err) + assert.Equal(t, postgresOpCreateTable, create, "both dialects name a new table the same way") + + alter, err := storageSchemaOperation(ddl.StatementAlterTable) + require.NoError(t, err) + assert.Equal(t, "alter_table", alter) + + drop, err := storageSchemaOperation(ddl.StatementDropTable) + require.NoError(t, err) + assert.Equal(t, "drop_table", drop) +} + +// A statement type the storage schema cannot contain is an error, not a label. +// Labelling it would hide a differ result this package does not understand +// inside a report an operator is about to act on. +func TestStorageSchemaOperation_UnexpectedTypeIsAnError(t *testing.T) { + _, err := storageSchemaOperation(ddl.StatementRenameTable) + require.Error(t, err) + assert.Contains(t, err.Error(), "unexpected statement type") + assert.Contains(t, err.Error(), "CREATE TABLE") +} + +// A refusal reason says whether the whole statement was refused because its +// clauses could not be partitioned, which is the difference between "these +// clauses are refused" and "none of this ran". +func TestRefusedStorageChange_ReportedReason(t *testing.T) { + classified := refusedStorageChange{ + change: engine.TableChange{Table: "applies"}, + reason: "DROP COLUMN destroys data", + } + assert.Equal(t, "DROP COLUMN destroys data", classified.reportedReason()) + + unsplittable := refusedStorageChange{ + change: engine.TableChange{Table: "applies"}, + reason: "DROP COLUMN destroys data", + splitErr: assert.AnError, + } + assert.Contains(t, unsplittable.reportedReason(), "DROP COLUMN destroys data") + assert.Contains(t, unsplittable.reportedReason(), "refused whole") + assert.Contains(t, unsplittable.reportedReason(), assert.AnError.Error()) +} + +// The diff and the bootstrap start from one set of option defaults, so a +// report cannot describe a policy the next boot would not apply. +func TestNewEnsureSchemaOptions_Defaults(t *testing.T) { + defaults := newEnsureSchemaOptions() + assert.Equal(t, schema.DialectMySQL, defaults.dialect) + assert.Equal(t, DefaultPostgresStatementTimeout, defaults.postgresStatementTimeout) + assert.False(t, defaults.allowDestructive, "destructive storage changes are refused unless asked for") + + configured := newEnsureSchemaOptions( + WithDialect(schema.DialectPostgres), + WithAllowDestructiveSchemaChanges(true), + WithPostgresStatementTimeout(0), + ) + assert.Equal(t, schema.DialectPostgres, configured.dialect) + assert.True(t, configured.allowDestructive) + assert.Zero(t, configured.postgresStatementTimeout, "zero disables the statement budget explicitly") +} diff --git a/pkg/apitypes/storage_schema.go b/pkg/apitypes/storage_schema.go new file mode 100644 index 000000000..76931f971 --- /dev/null +++ b/pkg/apitypes/storage_schema.go @@ -0,0 +1,77 @@ +package apitypes + +// StorageSchemaStatement is one outstanding statement in a storage schema +// report. The DDL is the statement itself, runnable as printed: an operator +// mid-incident copies these out and runs them by hand. +type StorageSchemaStatement struct { + Table string `json:"table"` + Operation string `json:"operation"` + DDL string `json:"ddl"` + // Reason is why the statement is destructive, or why it needs manual + // remediation. Empty for a statement that runs automatically. + Reason string `json:"reason,omitempty"` +} + +// StorageSchemaReport is what one SchemaBot instance's storage database needs +// in order to match that instance's embedded schema. +// +// The three statement lists are disjoint and have different dispositions: +// Outstanding runs, Destructive is refused unless destructive changes are +// allowed, and any Manual entry blocks the whole convergence. +type StorageSchemaReport struct { + // Deployment is the data plane whose storage this describes; empty means + // the control plane's own storage — the server the request was made to. + Deployment string `json:"deployment,omitempty"` + // Environment is the deployment's environment; empty alongside an empty + // Deployment. + Environment string `json:"environment,omitempty"` + // Dialect is the storage database's family: "mysql" or "postgres". + Dialect string `json:"dialect"` + // Database is the live database that was read, as its server reports it. + // It is what makes a report unmistakably about one database. + Database string `json:"database"` + // Version is the SchemaBot version of the binary whose embedded schema + // produced the diff. It is attribution, not an input: the diff came from + // the files themselves. + Version string `json:"version,omitempty"` + // Converged reports that the storage schema needs nothing at all. A report + // carrying only refused destructive statements is not converged. + Converged bool `json:"converged"` + Outstanding []StorageSchemaStatement `json:"outstanding,omitempty"` + Destructive []StorageSchemaStatement `json:"destructive,omitempty"` + DestructiveAllowed bool `json:"destructive_allowed"` + Manual []StorageSchemaStatement `json:"manual,omitempty"` +} + +// StorageSchemaDiffResponse is the HTTP response for +// GET /api/storage/schema/diff. +type StorageSchemaDiffResponse struct { + Report *StorageSchemaReport `json:"report"` +} + +// StorageSchemaApplyRequest is the HTTP request for +// POST /api/storage/schema/apply. +type StorageSchemaApplyRequest struct { + // Deployment names the data plane whose storage to converge. Empty + // converges the storage of the server the request is made to. + Deployment string `json:"deployment,omitempty"` + // Environment is required alongside Deployment, since a deployment serves + // one gRPC endpoint per environment. + Environment string `json:"environment,omitempty"` + // AllowDestructive permits the destructive statements the convergence + // would otherwise refuse. It widens the target's standing storage policy + // and never narrows it. + AllowDestructive bool `json:"allow_destructive,omitempty"` + // Caller identifies the operator, so the converging instance's logs + // attribute the convergence to a person. An authenticated identity + // overrides it. + Caller string `json:"caller,omitempty"` +} + +// StorageSchemaApplyResponse is the HTTP response for +// POST /api/storage/schema/apply. It brackets the convergence: Planned is what +// was outstanding before it ran, Remaining is what is still outstanding after. +type StorageSchemaApplyResponse struct { + Planned *StorageSchemaReport `json:"planned"` + Remaining *StorageSchemaReport `json:"remaining"` +} diff --git a/pkg/auth/tiers.go b/pkg/auth/tiers.go index 30b7b8aa8..918ba9a51 100644 --- a/pkg/auth/tiers.go +++ b/pkg/auth/tiers.go @@ -45,13 +45,33 @@ var readPaths = map[string]bool{ "/api/pull": true, } +// writePaths are GET endpoints that nonetheless require write access. They read +// SchemaBot's own storage schema — the internal shape of its bookkeeping +// database — rather than anything about a user's database, and they are the +// read half of a pair whose other half converges that database. An operator +// who may inspect the surplus and missing objects in SchemaBot's own storage is +// the operator who may converge it, so both halves are admitted at one tier. +// +// This is the tier that actually decides the question on a deployment +// configured with nothing but read groups and write groups, which is the common +// case: the handler-level scoped-write gate is a pass-through until some +// database configures operator groups, so a route left on the read tier here +// would be open to every reader regardless of what its handler checked. +var writePaths = map[string]bool{ + "/api/storage/schema/diff": true, +} + // TierForRequest classifies an API request into the access tier it requires. -// GET/HEAD requests and the explicit read-only endpoints are read; everything -// else is write, so a newly added mutating-looking endpoint fails closed -// (requires authorization) until it is classified here. Exported so the -// route authorization sweep test enforces per-database scoping against the -// same rule the middleware admits with. +// GET/HEAD requests are read unless listed in writePaths, the explicit +// read-only endpoints are read, and everything else is write — so a newly +// added mutating-looking endpoint fails closed (requires authorization) until +// it is classified here. Exported so the route authorization sweep test +// enforces per-database scoping against the same rule the middleware admits +// with. func TierForRequest(method, path string) Tier { + if writePaths[path] { + return TierWrite + } if readPaths[path] { return TierRead } diff --git a/pkg/auth/tiers_test.go b/pkg/auth/tiers_test.go index cea1c6948..bacc9f44b 100644 --- a/pkg/auth/tiers_test.go +++ b/pkg/auth/tiers_test.go @@ -26,12 +26,32 @@ func TestTierForRequest(t *testing.T) { {http.MethodPost, "/api/checks/synthesize", TierWrite}, {http.MethodPost, "/api/settings", TierWrite}, {http.MethodDelete, "/api/locks", TierWrite}, + // SchemaBot's own storage schema is admin territory on both halves: + // the diff exposes the internal shape of its bookkeeping database, and + // its sibling route converges it. On a deployment configured with only + // read and write groups this tier is the whole admin decision, so the + // GET must not sit at the read tier. + {http.MethodGet, "/api/storage/schema/diff", TierWrite}, + {http.MethodPost, "/api/storage/schema/apply", TierWrite}, } for _, c := range cases { assert.Equalf(t, c.want, TierForRequest(c.method, c.path), "%s %s", c.method, c.path) } } +// Every write-tier GET is listed in writePaths, and a GET outside that list +// stays a read. The list is the only thing standing between a read-tier +// classification and an endpoint everyone with read access can call, so it has +// to be exact rather than approximate. +func TestWritePathsCoverEveryWriteTierGet(t *testing.T) { + for path := range writePaths { + assert.Equalf(t, TierWrite, TierForRequest(http.MethodGet, path), "GET %s", path) + assert.Equalf(t, TierWrite, TierForRequest(http.MethodHead, path), "HEAD %s", path) + } + assert.Equal(t, TierRead, TierForRequest(http.MethodGet, "/api/storage/schema"), + "a prefix of a write path is not itself a write path") +} + func TestMatchesAnyGroup(t *testing.T) { admin := []string{"octocat/schema-admins"} diff --git a/pkg/cmd/client/client.go b/pkg/cmd/client/client.go index 4af5698a2..1456891e3 100644 --- a/pkg/cmd/client/client.go +++ b/pkg/cmd/client/client.go @@ -131,6 +131,51 @@ func ChecksRepos(ctx context.Context, endpoint string) (*apitypes.ChecksReposRes return &result, nil } +// StorageSchemaDiff reads which storage DDL is outstanding on a SchemaBot +// instance's own storage database — the server addressed by endpoint, or a data +// plane reached through it when deployment is set. +// +// The server computes the answer from the embedded schema files of the binary +// that is running, against the live catalog. No version is sent, because a +// version is the wrong input: it says what a release would converge to, not +// what the storage actually converged to, and the two differ exactly when a +// deploy has failed to converge. +func StorageSchemaDiff(ctx context.Context, endpoint, deployment, environment string, allowDestructive bool) (*apitypes.StorageSchemaDiffResponse, error) { + values := url.Values{} + if deployment != "" { + values.Set("deployment", deployment) + } + if environment != "" { + values.Set("environment", environment) + } + if allowDestructive { + values.Set("allow_destructive", "true") + } + requestPath := "/api/storage/schema/diff" + if encoded := values.Encode(); encoded != "" { + requestPath += "?" + encoded + } + var result apitypes.StorageSchemaDiffResponse + if err := doSlowGetIntoCtx(ctx, endpoint, requestPath, &result); err != nil { + return nil, err + } + return &result, nil +} + +// StorageSchemaApply converges a SchemaBot instance's own storage database by +// running the startup bootstrap that instance would run on its next boot, +// under the same advisory lock. +func StorageSchemaApply(ctx context.Context, endpoint string, req apitypes.StorageSchemaApplyRequest) (*apitypes.StorageSchemaApplyResponse, error) { + if req.Caller == "" { + req.Caller = GenerateCLIOwner() + } + var result apitypes.StorageSchemaApplyResponse + if err := doSlowPostIntoCtx(ctx, endpoint, "/api/storage/schema/apply", req, &result); err != nil { + return nil, err + } + return &result, nil +} + // PullSchemaOptions controls optional live schema pull request fields. type PullSchemaOptions struct { Namespaces []string diff --git a/pkg/cmd/client/request.go b/pkg/cmd/client/request.go index 7cd3c1f24..6b71535df 100644 --- a/pkg/cmd/client/request.go +++ b/pkg/cmd/client/request.go @@ -27,11 +27,12 @@ var authTransport = &bearerTransport{base: http.DefaultTransport} // Uses a 30s timeout to avoid hanging indefinitely on network stalls. var httpClient = &http.Client{Timeout: 30 * time.Second, Transport: authTransport} -// webhookOpsHTTPClient serves the webhook operator endpoints, which crawl -// GitHub delivery history or every open PR server-side and routinely need far -// longer than the default client timeout. Matches the server's own budget -// for these routes. -var webhookOpsHTTPClient = &http.Client{Timeout: 15 * time.Minute, Transport: authTransport} +// operatorHTTPClient serves the operator endpoints whose server-side work +// routinely needs far longer than the default client timeout: the webhook +// operations that crawl GitHub delivery history or every open PR, and the +// storage schema convergence that runs a bootstrap against the storage +// database. Matches the server's own budget for these routes. +var operatorHTTPClient = &http.Client{Timeout: 15 * time.Minute, Transport: authTransport} // SetAuthToken configures the Bearer token attached to every CLI request. An // empty token leaves requests unauthenticated, which is correct against a @@ -130,11 +131,12 @@ func doGetIntoCtx(ctx context.Context, endpoint, path string, result any) error return doGetIntoWithClient(ctx, httpClient, endpoint, path, result) } -// doSlowGetIntoCtx is doGetIntoCtx with the long-running webhook ops client, -// for read endpoints whose server-side work calls out to GitHub and so shares -// the operator budget rather than the default request timeout. +// doSlowGetIntoCtx is doGetIntoCtx with the long-running operator client, for +// read endpoints whose server-side work calls out to GitHub or reads a storage +// catalog, and so shares the operator budget rather than the default request +// timeout. func doSlowGetIntoCtx(ctx context.Context, endpoint, path string, result any) error { - return doGetIntoWithClient(ctx, webhookOpsHTTPClient, endpoint, path, result) + return doGetIntoWithClient(ctx, operatorHTTPClient, endpoint, path, result) } func doGetIntoWithClient(ctx context.Context, client *http.Client, endpoint, path string, result any) error { @@ -203,12 +205,12 @@ func doPostInto(endpoint, path string, body any, result any) error { return doPostIntoWithClient(context.Background(), httpClient, endpoint, path, body, result) } -// doSlowPostIntoCtx is doPostInto with the long-running webhook ops client and +// doSlowPostIntoCtx is doPostInto with the long-running operator client and // a caller-supplied context, for operator endpoints whose server-side work // legitimately outlives the default timeout and that must stop promptly when // the operator cancels (Ctrl+C). func doSlowPostIntoCtx(ctx context.Context, endpoint, path string, body any, result any) error { - return doPostIntoWithClient(ctx, webhookOpsHTTPClient, endpoint, path, body, result) + return doPostIntoWithClient(ctx, operatorHTTPClient, endpoint, path, body, result) } func doPostIntoWithClient(ctx context.Context, client *http.Client, endpoint, path string, body any, result any) error { diff --git a/pkg/cmd/commands/exit_code.go b/pkg/cmd/commands/exit_code.go new file mode 100644 index 000000000..16f00f745 --- /dev/null +++ b/pkg/cmd/commands/exit_code.go @@ -0,0 +1,59 @@ +package commands + +import ( + "errors" + "fmt" +) + +// Process exit statuses beyond the CLI's usual 0 for success and 1 for +// failure. A status is a distinct number only when a script would branch on it: +// an operator reading the output can tell a converged database from an +// outstanding one, but a pre-deploy job or a monitor cannot, and telling them +// apart is the whole reason to run the command unattended. +const ( + // ExitStorageSchemaOutstanding says a storage schema diff succeeded and + // found work: statements are outstanding, refused, or waiting on manual + // remediation. The read itself worked, so it is not a failure — a caller + // that treated it as one could not distinguish it from an unreachable + // database, which is the distinction that matters when deciding whether to + // proceed with a deploy. + ExitStorageSchemaOutstanding = 2 +) + +// ExitCodeError carries the process exit status a command wants, alongside its +// error. main honors the status and prints the error unless it is silent, so a +// command chooses its status without reaching for os.Exit and skipping the +// signal teardown around the run. +type ExitCodeError struct { + Code int + Err error +} + +func (e *ExitCodeError) Error() string { + if e.Err == nil { + return fmt.Sprintf("exit status %d", e.Code) + } + return e.Err.Error() +} + +func (e *ExitCodeError) Unwrap() error { return e.Err } + +// ExitCodeFor returns the process exit status an error asks for, defaulting to +// 1 for an error that asks for nothing in particular. A nil error is status 0. +func ExitCodeFor(err error) int { + if err == nil { + return 0 + } + var coded *ExitCodeError + if errors.As(err, &coded) && coded.Code != 0 { + return coded.Code + } + return 1 +} + +// exitStorageSchemaOutstanding asks for that status without printing anything +// further: the diff has already rendered everything the operator needs to see, +// and an "Error:" line under a report that is not an error would misread it. +func exitStorageSchemaOutstanding() error { + return &ExitCodeError{Code: ExitStorageSchemaOutstanding, Err: ErrSilent} +} diff --git a/pkg/cmd/commands/storage.go b/pkg/cmd/commands/storage.go index 669a6cac3..1be75742e 100644 --- a/pkg/cmd/commands/storage.go +++ b/pkg/cmd/commands/storage.go @@ -11,15 +11,23 @@ import ( "github.com/block/spirit/pkg/utils" "github.com/block/schemabot/pkg/api" + "github.com/block/schemabot/pkg/mysqlconn" "github.com/block/schemabot/pkg/postgresconn" "github.com/block/schemabot/pkg/schema" ) -// StorageCmd groups operator commands that act directly on SchemaBot's own -// storage database. Unlike the API-client commands, these connect to storage -// themselves and work while the server is down — they exist for maintenance -// windows such as a cross-dialect data move or a restore from a dump. +// StorageCmd groups operator commands that act on SchemaBot's own storage +// database rather than on a user's database. +// +// The maintenance commands connect to storage themselves and work while the +// server is down, for windows such as a cross-dialect data move or a restore +// from a dump. The schema commands do both: they read a storage database +// through the API by default, so an operator can reach a data plane's storage +// that no workstation can dial, and connect directly when the server is down — +// including when it is down because its own schema bootstrap is failing. type StorageCmd struct { + Diff StorageDiffCmd `cmd:"" help:"Show the storage DDL outstanding between a SchemaBot instance's embedded schema and its live storage database; read-only, safe at any time. Exits 0 when converged and 2 when statements are outstanding."` + Apply StorageApplyCmd `cmd:"" help:"Converge a SchemaBot instance's storage database by running the schema bootstrap it would run on its next boot, under the same advisory lock and the same destructive-statement refusal."` ResyncIdentitySequences ResyncIdentitySequencesCmd `cmd:"" name:"resync-identity-sequences" help:"Advance PostgreSQL identity sequences on storage tables past their columns' stored maxima after an explicit-id bulk load; run after the load has fully committed and before default inserts resume — advance-only and safe to rerun."` CanonicalizeIdentityKeys CanonicalizeIdentityKeysCmd `cmd:"" name:"canonicalize-identity-keys" help:"Fold stored identity strings (repository, database, environment, deployment, lock owner) on PostgreSQL storage tables to canonical lowercase; run once, in a quiesced maintenance window, after every writer runs a release that folds identity strings at the write boundaries. The rewrite is one-way — original spellings are not recorded — so the command prompts unless --auto-approve is set; it only rewrites non-canonical rows, safe to rerun."` } @@ -159,29 +167,66 @@ func (cmd *CanonicalizeIdentityKeysCmd) Run(ctx context.Context, g *Globals) err } // resolveStorageDSN returns the storage DSN and a loggable description of -// where it came from. A direct --dsn is used after verifying it parses as a -// PostgreSQL DSN; otherwise the server config (--config, then -// $SCHEMABOT_CONFIG_FILE) is loaded and its resolved storage DSN is used, -// failing closed when the configured storage dialect is not postgres — -// purpose names the operation in these errors. The source never contains -// the DSN itself, which may embed credentials. +// where it came from, for the commands that only apply to PostgreSQL storage — +// purpose names the operation in the refusal. The source never contains the +// DSN itself, which may embed credentials. func resolveStorageDSN(dsnFlag, configFlag, purpose string) (string, string, error) { + // A direct DSN is checked against the PostgreSQL grammar rather than handed + // to the generic family inference: these commands take no --dialect, so + // there is nothing for an ambiguous DSN to be resolved by, and the refusal + // that helps is the one naming the operation that does not apply to it. + if directDSN := strings.TrimSpace(dsnFlag); directDSN != "" && configFlag == "" { + if _, err := postgresconn.ConnectionDSN(directDSN); err != nil { + return "", "", fmt.Errorf("storage DSN from --dsn is not a PostgreSQL DSN; %s only applies to %q storage: %w", purpose, schema.DialectPostgres, err) + } + return directDSN, "--dsn flag", nil + } + target, err := resolveStorageTarget(dsnFlag, configFlag, "") + if err != nil { + return "", "", err + } + if target.dialect != schema.DialectPostgres { + return "", "", fmt.Errorf("storage dialect in %s is %q; %s only applies to %q storage", target.source, target.dialect, purpose, schema.DialectPostgres) + } + return target.dsn, target.source, nil +} + +// storageTarget is a resolved direct connection to a storage database: where to +// connect, which family the database belongs to, and a loggable description of +// where the DSN came from. The description never contains the DSN itself, which +// may embed credentials. +type storageTarget struct { + dsn string + dialect schema.Dialect + source string +} + +// resolveStorageTarget resolves a storage database to connect to directly. +// +// A direct --dsn is taken as given, with its dialect from --dialect or inferred +// from the DSN's own form. Otherwise the server config (--config, then +// $SCHEMABOT_CONFIG_FILE) is loaded and both the DSN and the dialect come from +// it, which is the path that needs no inference at all: a server config states +// its storage dialect. +// +// dialectFlag is the operator's assertion and wins over inference, but it is +// only consulted on the direct path. A config that says postgres and a flag +// that says mysql is a contradiction, not a preference, so the config's own +// dialect stands and the mismatch is refused rather than resolved. +func resolveStorageTarget(dsnFlag, configFlag, dialectFlag string) (*storageTarget, error) { directDSN := strings.TrimSpace(dsnFlag) if directDSN != "" && configFlag != "" { - return "", "", fmt.Errorf("--dsn and --config are mutually exclusive; pass the storage DSN directly or resolve it from a server config, not both") + return nil, fmt.Errorf("--dsn and --config are mutually exclusive; pass the storage DSN directly or resolve it from a server config, not both") } if dsnFlag != "" { if directDSN == "" { - return "", "", fmt.Errorf("storage DSN not configured: --dsn contains only whitespace") + return nil, fmt.Errorf("storage DSN not configured: --dsn contains only whitespace") } - // The config path refuses non-postgres storage via the configured - // dialect; the direct path has no dialect field, so refuse any DSN - // that does not parse as PostgreSQL instead of failing later with an - // opaque connection error — or worse, against the wrong server. - if _, err := postgresconn.ConnectionDSN(directDSN); err != nil { - return "", "", fmt.Errorf("storage DSN from --dsn is not a PostgreSQL DSN; %s only applies to %q storage: %w", purpose, schema.DialectPostgres, err) + dialect, err := directStorageDialect(directDSN, dialectFlag) + if err != nil { + return nil, err } - return directDSN, "--dsn flag", nil + return &storageTarget{dsn: directDSN, dialect: dialect, source: "--dsn flag"}, nil } configPath := configFlag @@ -189,7 +234,7 @@ func resolveStorageDSN(dsnFlag, configFlag, purpose string) (string, string, err if configPath == "" { configPath = os.Getenv("SCHEMABOT_CONFIG_FILE") if configPath == "" { - return "", "", fmt.Errorf("no storage DSN source: set --dsn, --config, or the SCHEMABOT_CONFIG_FILE environment variable") + return nil, fmt.Errorf("no storage DSN source: set --dsn, --config, or the SCHEMABOT_CONFIG_FILE environment variable") } source = fmt.Sprintf("server config %s ($SCHEMABOT_CONFIG_FILE)", configPath) } @@ -202,24 +247,24 @@ func resolveStorageDSN(dsnFlag, configFlag, purpose string) (string, string, err cfg, err = api.LoadServerConfigFromFile(configPath) } if err != nil { - return "", "", fmt.Errorf("load %s: %w", source, err) + return nil, fmt.Errorf("load %s: %w", source, err) } dialect, err := cfg.Storage.ResolveDialect() if err != nil { - return "", "", fmt.Errorf("resolve storage dialect from %s: %w", source, err) + return nil, fmt.Errorf("resolve storage dialect from %s: %w", source, err) } - if dialect != schema.DialectPostgres { - return "", "", fmt.Errorf("storage dialect in %s is %q; %s only applies to %q storage", source, dialect, purpose, schema.DialectPostgres) + if asserted := strings.TrimSpace(strings.ToLower(dialectFlag)); asserted != "" && schema.Dialect(asserted) != dialect { + return nil, fmt.Errorf("--dialect says %q but %s configures %q storage; drop --dialect, which only applies to a DSN passed with --dsn", asserted, source, dialect) } dsn, err := cfg.StorageDSN() if err != nil { - return "", "", fmt.Errorf("resolve storage DSN from %s: %w", source, err) + return nil, fmt.Errorf("resolve storage DSN from %s: %w", source, err) } dsn = strings.TrimSpace(dsn) if dsn == "" { - return "", "", fmt.Errorf("storage DSN not configured (set --dsn, config storage.dsn or storage.dsn_from, STORAGE_DSN, or MYSQL_DSN)") + return nil, fmt.Errorf("storage DSN not configured (set --dsn, config storage.dsn or storage.dsn_from, STORAGE_DSN, or MYSQL_DSN)") } if cfg.Storage.DSN == "" && cfg.Storage.DSNFrom == nil { if strings.TrimSpace(os.Getenv("STORAGE_DSN")) != "" { @@ -228,5 +273,54 @@ func resolveStorageDSN(dsnFlag, configFlag, purpose string) (string, string, err source = "MYSQL_DSN environment variable" } } - return dsn, source, nil + return &storageTarget{dsn: dsn, dialect: dialect, source: source}, nil +} + +// directStorageDialect decides which database family a DSN passed straight on +// the command line addresses. +// +// An explicit --dialect settles it. Without one, the DSN's own form does: a +// PostgreSQL connection string is a URL with a postgres scheme or a +// keyword/value string, and the Go MySQL driver's DSN format is neither, so the +// two are told apart by reading the string rather than by guessing. A DSN that +// matches neither shape is refused naming --dialect: connecting a MySQL driver +// to a PostgreSQL port fails with a protocol error that says nothing about the +// real problem, and the operator reaching for this is usually mid-incident. +func directStorageDialect(dsn, dialectFlag string) (schema.Dialect, error) { + switch asserted := schema.Dialect(strings.TrimSpace(strings.ToLower(dialectFlag))); asserted { + case schema.DialectMySQL, schema.DialectPostgres: + return asserted, nil + case "": + default: + return "", fmt.Errorf("unsupported storage dialect %q; SchemaBot stores its own state on %q or %q", asserted, schema.DialectMySQL, schema.DialectPostgres) + } + + if looksLikePostgresDSN(dsn) { + if _, err := postgresconn.ConnectionDSN(dsn); err != nil { + return "", fmt.Errorf("--dsn looks like a PostgreSQL connection string but does not parse as one: %w", err) + } + return schema.DialectPostgres, nil + } + if _, err := mysqlconn.ConnectionDSN(dsn); err == nil { + return schema.DialectMySQL, nil + } + return "", fmt.Errorf("cannot tell which database family --dsn addresses: it is neither a PostgreSQL connection string (postgres://user@host:5432/db) nor a Go MySQL driver DSN (user:pass@tcp(host:3306)/db); pass --dialect %s or --dialect %s to say which", + schema.DialectMySQL, schema.DialectPostgres) +} + +// looksLikePostgresDSN reports whether a DSN is written in one of PostgreSQL's +// two connection-string forms: a URL with a postgres scheme, or a libpq +// keyword/value string. Both are unambiguous — the Go MySQL driver's format has +// no scheme and no keywords — so this classifies without connecting. +func looksLikePostgresDSN(dsn string) bool { + lowered := strings.ToLower(strings.TrimSpace(dsn)) + if strings.HasPrefix(lowered, "postgres://") || strings.HasPrefix(lowered, "postgresql://") { + return true + } + for _, keyword := range []string{"host=", "hostaddr=", "dbname=", "port=", "user=", "sslmode="} { + if strings.Contains(lowered, keyword) { + return true + } + } + return false } diff --git a/pkg/cmd/commands/storage_schema.go b/pkg/cmd/commands/storage_schema.go new file mode 100644 index 000000000..fcd245dc1 --- /dev/null +++ b/pkg/cmd/commands/storage_schema.go @@ -0,0 +1,461 @@ +package commands + +import ( + "context" + "encoding/json" + "fmt" + "io" + "log/slog" + "os" + "strings" + + "github.com/block/schemabot/pkg/api" + "github.com/block/schemabot/pkg/apitypes" + cmdclient "github.com/block/schemabot/pkg/cmd/client" + "github.com/block/schemabot/pkg/cmd/cliname" +) + +// Storage schema commands answer, and then close, the one question a deploy +// that did not converge leaves open: which storage DDL is still outstanding on +// this instance's own storage database. +// +// Both read the live database. Neither takes a version: a release tag says what +// that release would converge to, not what the storage converged to, and the +// two answers differ exactly when a deploy has failed — which is the only time +// anyone runs these. +// +// There are two ways to reach a storage database, and which one applies is the +// operator's to state rather than the CLI's to discover: +// +// through the API the server reads its own storage, or asks a data +// plane to read its own over the connection that +// already exists between them +// directly, by DSN this workstation opens the storage database itself, +// for when the server is down — including when it is +// down because its schema bootstrap is failing +// +// Nothing falls back from one to the other. A deployment that cannot be +// reached through the API is an error naming the deployment, never a report +// about a different database that happened to be reachable. + +// storageSchemaTargetFlags selects which storage database the command acts on. +// The two paths are mutually exclusive and the flags say which is in use, so a +// command never has to infer the operator's intent from what happened to +// resolve. +type storageSchemaTargetFlags struct { + Deployment string `help:"Data plane whose storage to read, as named under tern_deployments in the server config; omit for the storage of the server the CLI is pointed at"` + Environment string `short:"e" help:"Environment of the deployment's endpoint; required with --deployment"` + DSN string `help:"Connect to the storage database directly with this DSN instead of going through the API; for when the server is down"` + Config string `help:"Server config file to resolve the storage DSN from, connecting directly instead of going through the API; defaults to $SCHEMABOT_CONFIG_FILE when --dsn is not set"` + Dialect string `help:"Storage database family of --dsn (mysql or postgres) when its form does not say"` +} + +// direct reports whether the operator asked for a direct connection. Naming a +// DSN source is the whole signal: both flags exist only on that path. +func (f *storageSchemaTargetFlags) direct() bool { + return strings.TrimSpace(f.DSN) != "" || strings.TrimSpace(f.Config) != "" +} + +// validate refuses flag combinations that mix the two paths, instead of +// silently honoring one and dropping the other. Dropping a --deployment would +// report the wrong database under the right name, which is worse than any +// error message. +func (f *storageSchemaTargetFlags) validate() error { + if !f.direct() { + if strings.TrimSpace(f.Dialect) != "" { + return fmt.Errorf("--dialect only applies to a direct connection: through the API the server reports its own storage dialect; pass --dsn or --config to connect directly") + } + if f.Deployment == "" && f.Environment != "" { + return fmt.Errorf("-e %s was given without --deployment: an environment selects which of a deployment's endpoints to reach, so name the deployment too, or omit both to read the storage of the server the CLI is pointed at", f.Environment) + } + if f.Deployment != "" && f.Environment == "" { + return fmt.Errorf("--deployment %s needs -e : a deployment serves one endpoint per environment, so there is no single storage to read without one", f.Deployment) + } + return nil + } + if f.Deployment != "" { + return fmt.Errorf("--deployment cannot be combined with a direct connection: a DSN addresses one storage database, and reaching a data plane's storage goes through its own endpoint; drop --dsn/--config to route through the API, or drop --deployment to read the database the DSN names") + } + if f.Environment != "" { + return fmt.Errorf("-e cannot be combined with a direct connection: a DSN already names one database, so there is no endpoint to select") + } + return nil +} + +// StorageDiffCmd reports the storage DDL outstanding between a SchemaBot +// instance's embedded schema files and its live storage database. +// +// It is strictly read-only — it reads the catalog, computes a diff, and takes +// no lock — so it is safe to run at any time, including against production +// while an apply is in flight or an incident is open. +// +// The exit status is the machine-readable half of the answer: 0 when the +// storage needs nothing, 2 when statements are outstanding, and 1 when the +// read itself failed. A pre-deploy gate needs those three apart, because +// "converged" and "unreachable" call for opposite decisions. +type StorageDiffCmd struct { + storageSchemaTargetFlags `embed:""` + AllowDestructive bool `help:"Report destructive statements as ones that would run, matching what an apply with the same flag would do" name:"allow-destructive"` + JSON bool `help:"Output as JSON"` +} + +func (cmd *StorageDiffCmd) Run(ctx context.Context, g *Globals) error { + if err := cmd.validate(); err != nil { + return err + } + report, err := cmd.read(ctx, g) + if err != nil { + return err + } + if cmd.JSON { + encoder := json.NewEncoder(os.Stdout) + encoder.SetIndent("", " ") + if err := encoder.Encode(apitypes.StorageSchemaDiffResponse{Report: report}); err != nil { + return fmt.Errorf("encode storage schema report: %w", err) + } + } else if err := renderStorageSchemaReport(os.Stdout, report, storageSchemaDiffHints(cmd)); err != nil { + return err + } + if report.Converged { + return nil + } + return exitStorageSchemaOutstanding() +} + +// read fetches the report over whichever path the flags selected. +func (cmd *StorageDiffCmd) read(ctx context.Context, g *Globals) (*apitypes.StorageSchemaReport, error) { + if cmd.direct() { + target, err := resolveStorageTarget(cmd.DSN, cmd.Config, cmd.Dialect) + if err != nil { + return nil, err + } + logger := storageSchemaLogger(g) + logger.Info("reading storage schema directly", + "source", target.source, "dialect", target.dialect) + report, err := api.DiffStorageSchema(ctx, target.dsn, logger, + api.WithDialect(target.dialect), + api.WithAllowDestructiveSchemaChanges(cmd.AllowDestructive)) + if err != nil { + return nil, fmt.Errorf("diff storage schema on the database from %s: %w", target.source, err) + } + // The version is this CLI's, because on this path the embedded schema + // files that produced the diff are this binary's own. + report.Version = g.Version + return report.APIType(), nil + } + + endpoint, err := g.Resolve() + if err != nil { + return nil, err + } + response, err := cmdclient.StorageSchemaDiff(ctx, endpoint, cmd.Deployment, cmd.Environment, cmd.AllowDestructive) + if err != nil { + return nil, fmt.Errorf("diff storage schema%s: %w", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment), err) + } + if response.Report == nil { + return nil, fmt.Errorf("storage schema diff%s returned no report", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment)) + } + return response.Report, nil +} + +// StorageApplyCmd converges a SchemaBot instance's storage database by running +// the schema bootstrap that instance would run on its next boot. +// +// It is the bootstrap, not a second implementation of it: the same differ, the +// same refusal of destructive statements, and the same advisory lock — so two +// operators running this at once serialize exactly the way two booting pods +// do, and a pre-deploy convergence step is this command with nothing added. +type StorageApplyCmd struct { + storageSchemaTargetFlags `embed:""` + AllowDestructive bool `help:"Permit the destructive statements the convergence would otherwise refuse; it widens the target's standing storage policy and never narrows it" name:"allow-destructive"` + AutoApprove bool `short:"y" help:"Skip confirmation prompt" name:"auto-approve"` + JSON bool `help:"Output as JSON"` +} + +func (cmd *StorageApplyCmd) Run(ctx context.Context, g *Globals) error { + if err := cmd.validate(); err != nil { + return err + } + + // Confirm against a fresh read rather than against a description of the + // command: an operator approving DDL on SchemaBot's own storage should see + // the statements, and the read is free of side effects. With --auto-approve + // the convergence's own planned report says what ran, so there is nothing + // this preview would add. + if !cmd.AutoApprove { + preview := &StorageDiffCmd{ + storageSchemaTargetFlags: cmd.storageSchemaTargetFlags, + AllowDestructive: cmd.AllowDestructive, + } + report, err := preview.read(ctx, g) + if err != nil { + return err + } + if err := renderStorageSchemaReport(os.Stdout, report, nil); err != nil { + return err + } + if report.Converged { + // Nothing to converge and nothing to approve. Returning success is + // the honest answer: the storage already matches. + return nil + } + if len(report.Manual) > 0 { + return fmt.Errorf("refusing to converge storage schema on %s: %d change(s) need manual remediation first (listed above)", storageSchemaDatabaseLabel(report), len(report.Manual)) + } + confirmed, err := confirmAction( + fmt.Sprintf("\nRun these statements against %s? Only 'yes' will be accepted: ", storageSchemaDatabaseLabel(report)), + "\nConvergence aborted.", + ) + if err != nil { + return err + } + if !confirmed { + return nil + } + } + + planned, remaining, err := cmd.converge(ctx, g) + if err != nil { + return err + } + if cmd.JSON { + encoder := json.NewEncoder(os.Stdout) + encoder.SetIndent("", " ") + if err := encoder.Encode(apitypes.StorageSchemaApplyResponse{Planned: planned, Remaining: remaining}); err != nil { + return fmt.Errorf("encode storage schema convergence: %w", err) + } + return nil + } + return renderStorageSchemaConvergence(os.Stdout, planned, remaining) +} + +// converge runs the convergence over whichever path the flags selected. +func (cmd *StorageApplyCmd) converge(ctx context.Context, g *Globals) (planned, remaining *apitypes.StorageSchemaReport, err error) { + if cmd.direct() { + target, err := resolveStorageTarget(cmd.DSN, cmd.Config, cmd.Dialect) + if err != nil { + return nil, nil, err + } + logger := storageSchemaLogger(g) + logger.Info("converging storage schema directly", + "source", target.source, "dialect", target.dialect, "allow_destructive", cmd.AllowDestructive) + plannedReport, remainingReport, err := api.ApplyStorageSchema(ctx, target.dsn, logger, + api.WithDialect(target.dialect), + api.WithAllowDestructiveSchemaChanges(cmd.AllowDestructive)) + if err != nil { + return nil, nil, fmt.Errorf("converge storage schema on the database from %s: %w", target.source, err) + } + plannedReport.Version = g.Version + remainingReport.Version = g.Version + return plannedReport.APIType(), remainingReport.APIType(), nil + } + + endpoint, err := g.Resolve() + if err != nil { + return nil, nil, err + } + response, err := cmdclient.StorageSchemaApply(ctx, endpoint, apitypes.StorageSchemaApplyRequest{ + Deployment: cmd.Deployment, + Environment: cmd.Environment, + AllowDestructive: cmd.AllowDestructive, + }) + if err != nil { + return nil, nil, fmt.Errorf("converge storage schema%s: %w", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment), err) + } + if response.Planned == nil || response.Remaining == nil { + // Both halves are required to say what happened; without the pair there + // is no way to tell a convergence that finished from one that left + // statements behind. + return nil, nil, fmt.Errorf("storage schema convergence%s returned an incomplete result; check the target's logs for whether it converged", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment)) + } + return response.Planned, response.Remaining, nil +} + +// storageSchemaLogger builds the diagnostics logger for a direct connection. A +// text handler on stderr, deliberately: this is a one-shot command read at a +// terminal, and keeping diagnostics off stdout leaves the report itself +// pipeable. +func storageSchemaLogger(g *Globals) *slog.Logger { + return slog.New(slog.NewTextHandler(os.Stderr, &slog.HandlerOptions{ + Level: logLevel(), + })).With("schemabot_version", g.Version) +} + +// storageSchemaTargetSuffix names the target in an error, so a failure says +// which storage was being read rather than only that a read failed. +func storageSchemaTargetSuffix(deployment, environment string) string { + if deployment == "" { + return "" + } + return fmt.Sprintf(" for deployment %s in %s", deployment, environment) +} + +// storageSchemaDatabaseLabel names the database a report is about, as an +// operator would say it: the database, its dialect, and the deployment it +// belongs to when the report came from one. +func storageSchemaDatabaseLabel(report *apitypes.StorageSchemaReport) string { + label := report.Database + if label == "" { + label = "the storage database" + } + if report.Dialect != "" { + label += fmt.Sprintf(" (%s)", report.Dialect) + } + if report.Deployment != "" { + label += fmt.Sprintf(" on deployment %s in %s", report.Deployment, report.Environment) + } + return label +} + +// renderStorageSchemaReport prints a report as the statements themselves, +// runnable as pasted. +// +// The layout is chosen for what an operator does with it at three in the +// morning: the statements are bare and one per line, with their classification +// in a heading above rather than a prefix beside, so a whole section can be +// selected and pasted into a client without stripping anything. hints are +// printed under the report when there is a next step to name. +func renderStorageSchemaReport(w io.Writer, report *apitypes.StorageSchemaReport, hints []string) error { + if _, err := fmt.Fprintf(w, "%s\n", storageSchemaHeadline(report)); err != nil { + return err + } + if report.Converged { + return nil + } + sections := []struct { + title string + statements []apitypes.StorageSchemaStatement + }{ + {storageSchemaOutstandingTitle, report.Outstanding}, + {storageSchemaDestructiveTitle(report), report.Destructive}, + {"Needs manual remediation before anything converges", report.Manual}, + } + for _, section := range sections { + if len(section.statements) == 0 { + continue + } + if _, err := fmt.Fprintf(w, "\n%s (%d):\n\n", section.title, len(section.statements)); err != nil { + return err + } + for _, statement := range section.statements { + if statement.Reason != "" { + if _, err := fmt.Fprintf(w, "-- %s: %s\n", statement.Table, statement.Reason); err != nil { + return err + } + } + if _, err := fmt.Fprintf(w, "%s\n", storageSchemaRunnableDDL(statement.DDL)); err != nil { + return err + } + } + } + for _, hint := range hints { + if _, err := fmt.Fprintf(w, "\n%s\n", hint); err != nil { + return err + } + } + return nil +} + +// storageSchemaHeadline is the one line an operator reads first: which database +// was read, and whether it needs anything. +func storageSchemaHeadline(report *apitypes.StorageSchemaReport) string { + version := "" + if report.Version != "" { + version = fmt.Sprintf(", against the schema embedded in %s", report.Version) + } + if report.Converged { + return fmt.Sprintf("%s is converged%s.", storageSchemaDatabaseLabel(report), version) + } + counts := make([]string, 0, 3) + if n := len(report.Outstanding); n > 0 { + counts = append(counts, fmt.Sprintf("%d outstanding", n)) + } + if n := len(report.Destructive); n > 0 { + counts = append(counts, fmt.Sprintf("%d destructive", n)) + } + if n := len(report.Manual); n > 0 { + counts = append(counts, fmt.Sprintf("%d needing manual remediation", n)) + } + total := len(report.Outstanding) + len(report.Destructive) + len(report.Manual) + return fmt.Sprintf("%s needs %d %s: %s%s.", + storageSchemaDatabaseLabel(report), total, pluralStatements(total), strings.Join(counts, ", "), version) +} + +// storageSchemaOutstandingTitle says what the statements in the section are: +// whatever else an operator does with them, these are what the next boot of the +// target's own binary will run. +const storageSchemaOutstandingTitle = "Outstanding, and run automatically on the next boot or apply" + +// storageSchemaDestructiveTitle distinguishes destructive statements that would +// run from ones that are refused. The difference is the whole disposition of +// the section, so it goes in the heading rather than in a footnote: a refused +// statement is surplus state left in place on purpose, not a pending change. +func storageSchemaDestructiveTitle(report *apitypes.StorageSchemaReport) string { + if report.DestructiveAllowed { + return "Destructive, and permitted to run because destructive changes are allowed" + } + return "Destructive, and refused; surplus state stays in place" +} + +// storageSchemaDiffHints names the next step for the report a diff just +// printed, in the command form the operator invoked the CLI as. +func storageSchemaDiffHints(cmd *StorageDiffCmd) []string { + if cmd.direct() { + return []string{fmt.Sprintf("Converge it with: %s storage apply %s", cliname.Name(), storageSchemaDirectFlagHint(cmd.DSN, cmd.Config))} + } + if cmd.Deployment != "" { + return []string{fmt.Sprintf("Converge it with: %s storage apply --deployment %s -e %s", cliname.Name(), cmd.Deployment, cmd.Environment)} + } + return []string{fmt.Sprintf("Converge it with: %s storage apply", cliname.Name())} +} + +// storageSchemaDirectFlagHint renders the flag that selected the direct path, +// without echoing a DSN that may carry credentials. +func storageSchemaDirectFlagHint(dsn, config string) string { + if strings.TrimSpace(config) != "" { + return "--config " + config + } + if dsn != "" { + return "--dsn " + } + return "" +} + +// renderStorageSchemaConvergence prints what a convergence ran and what it left +// behind. Both halves are printed even on a clean run, because "nothing is +// left" is the fact an operator is looking for and an absent section does not +// state it. +func renderStorageSchemaConvergence(w io.Writer, planned, remaining *apitypes.StorageSchemaReport) error { + applied := len(planned.Outstanding) + if _, err := fmt.Fprintf(w, "Ran %d %s against %s.\n", applied, pluralStatements(applied), storageSchemaDatabaseLabel(planned)); err != nil { + return err + } + if remaining.Converged { + _, err := fmt.Fprintf(w, "%s is converged.\n", storageSchemaDatabaseLabel(remaining)) + return err + } + if _, err := fmt.Fprintln(w); err != nil { + return err + } + return renderStorageSchemaReport(w, remaining, []string{ + "These were not applied. A destructive statement is refused unless --allow-destructive is passed; a manual entry has to be resolved by hand before anything else converges.", + }) +} + +// storageSchemaRunnableDDL terminates a statement so a pasted section runs as +// one script. The differ emits statements without a terminator, and a section +// pasted into a client without them runs as a single malformed statement. +func storageSchemaRunnableDDL(ddl string) string { + trimmed := strings.TrimRight(ddl, "; \t\r\n") + if trimmed == "" { + return ddl + } + return strings.TrimSpace(trimmed) + ";" +} + +func pluralStatements(n int) string { + if n == 1 { + return "statement" + } + return "statements" +} diff --git a/pkg/cmd/commands/storage_schema_test.go b/pkg/cmd/commands/storage_schema_test.go new file mode 100644 index 000000000..0de84f5ff --- /dev/null +++ b/pkg/cmd/commands/storage_schema_test.go @@ -0,0 +1,420 @@ +package commands + +import ( + "bytes" + "errors" + "fmt" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/apitypes" + "github.com/block/schemabot/pkg/schema" +) + +// A target is either the API path or a direct connection, never a blend of the +// two. Every refusal below is a combination that would otherwise have read a +// database the operator did not name — the failure mode these flags exist to +// prevent. +func TestStorageSchemaTargetFlags_Validate(t *testing.T) { + tests := []struct { + name string + flags storageSchemaTargetFlags + wantErr string + }{ + { + name: "no flags reads the storage of the server the CLI is pointed at", + flags: storageSchemaTargetFlags{}, + }, + { + name: "deployment with its environment", + flags: storageSchemaTargetFlags{Deployment: "west", Environment: "production"}, + }, + { + name: "a DSN on its own", + flags: storageSchemaTargetFlags{DSN: "root@tcp(127.0.0.1:3306)/schemabot"}, + }, + { + name: "a config file on its own", + flags: storageSchemaTargetFlags{Config: "/etc/schemabot/config.yaml"}, + }, + { + name: "a deployment without an environment names no single endpoint", + flags: storageSchemaTargetFlags{Deployment: "west"}, + wantErr: "needs -e ", + }, + { + name: "an environment without a deployment selects nothing", + flags: storageSchemaTargetFlags{Environment: "production"}, + wantErr: "without --deployment", + }, + { + name: "a deployment cannot be read through a direct DSN", + flags: storageSchemaTargetFlags{Deployment: "west", Environment: "production", DSN: "root@tcp(127.0.0.1:3306)/schemabot"}, + wantErr: "--deployment cannot be combined with a direct connection", + }, + { + name: "an environment cannot be combined with a direct DSN", + flags: storageSchemaTargetFlags{Environment: "production", Config: "/etc/schemabot/config.yaml"}, + wantErr: "-e cannot be combined with a direct connection", + }, + { + name: "a dialect assertion has nothing to apply to through the API", + flags: storageSchemaTargetFlags{Dialect: "postgres"}, + wantErr: "--dialect only applies to a direct connection", + }, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + err := tc.flags.validate() + if tc.wantErr == "" { + require.NoError(t, err) + return + } + require.Error(t, err) + assert.Contains(t, err.Error(), tc.wantErr) + }) + } +} + +// Naming a DSN source is what selects the direct path, so a command can tell +// which path it is on without inspecting what happened to resolve. +func TestStorageSchemaTargetFlags_Direct(t *testing.T) { + assert.False(t, (&storageSchemaTargetFlags{}).direct()) + assert.False(t, (&storageSchemaTargetFlags{Deployment: "west", Environment: "production"}).direct()) + assert.True(t, (&storageSchemaTargetFlags{DSN: "root@tcp(127.0.0.1:3306)/schemabot"}).direct()) + assert.True(t, (&storageSchemaTargetFlags{Config: "/etc/schemabot/config.yaml"}).direct()) + assert.False(t, (&storageSchemaTargetFlags{DSN: " "}).direct(), + "whitespace names no DSN, and treating it as one would read a database from nowhere") +} + +// A direct DSN's database family is inferred from its form, and an explicit +// --dialect wins over the inference. Guessing wrong here would diff a +// PostgreSQL database against MySQL schema files, so an unrecognizable DSN is +// an error naming the flag that resolves it. +func TestDirectStorageDialect(t *testing.T) { + tests := []struct { + name string + dsn string + dialectFlag string + want schema.Dialect + wantErr string + }{ + { + name: "Go MySQL driver DSN", + dsn: "root:secret@tcp(127.0.0.1:3306)/schemabot", + want: schema.DialectMySQL, + }, + { + name: "PostgreSQL URL", + dsn: "postgres://schemabot@db.example:5432/schemabot?sslmode=require", + want: schema.DialectPostgres, + }, + { + name: "libpq keyword string", + dsn: "host=db.example port=5432 dbname=schemabot sslmode=require", + want: schema.DialectPostgres, + }, + { + name: "--dialect settles a DSN whose form does not say", + dsn: "schemabot", + dialectFlag: "postgres", + want: schema.DialectPostgres, + }, + { + name: "--dialect is case-insensitive and tolerates padding", + dsn: "root@tcp(127.0.0.1:3306)/schemabot", + dialectFlag: " MySQL ", + want: schema.DialectMySQL, + }, + { + name: "a dialect SchemaBot does not store its state on", + dsn: "root@tcp(127.0.0.1:3306)/schemabot", + dialectFlag: "sqlite", + wantErr: "unsupported storage dialect", + }, + { + name: "a PostgreSQL connection string that does not parse", + dsn: "postgres://schemabot@db.example:notaport/schemabot", + wantErr: "does not parse as one", + }, + { + name: "a DSN in neither family", + dsn: "jdbc:mysql://db.example:3306/schemabot", + wantErr: "cannot tell which database family --dsn addresses", + }, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + dialect, err := directStorageDialect(tc.dsn, tc.dialectFlag) + if tc.wantErr != "" { + require.Error(t, err) + assert.Contains(t, err.Error(), tc.wantErr) + return + } + require.NoError(t, err) + assert.Equal(t, tc.want, dialect) + }) + } +} + +// A whitespace-only --dsn is refused rather than resolved from the environment +// behind the operator's back: they named a direct connection, so falling back +// to a config file would read a different database than the one they asked for. +func TestResolveStorageTarget_RefusesBlankAndConflictingSources(t *testing.T) { + _, err := resolveStorageTarget(" ", "", "") + require.Error(t, err) + assert.Contains(t, err.Error(), "--dsn contains only whitespace") + + _, err = resolveStorageTarget("root@tcp(127.0.0.1:3306)/schemabot", "/etc/schemabot/config.yaml", "") + require.Error(t, err) + assert.Contains(t, err.Error(), "mutually exclusive") +} + +// A direct DSN resolves to the database it names, with its family inferred from +// its form and the flag that supplied it recorded as the source — the source is +// what an error message names, so it must not be the DSN itself. +func TestResolveStorageTarget_DirectDSN(t *testing.T) { + target, err := resolveStorageTarget("postgres://schemabot@db.example:5432/schemabot", "", "") + require.NoError(t, err) + assert.Equal(t, "postgres://schemabot@db.example:5432/schemabot", target.dsn) + assert.Equal(t, schema.DialectPostgres, target.dialect) + assert.Equal(t, "--dsn flag", target.source) +} + +// A pasted section has to run as a script, so every statement is terminated +// exactly once whatever the differ emitted. +func TestStorageSchemaRunnableDDL(t *testing.T) { + assert.Equal(t, "DROP TABLE `stale_state`;", storageSchemaRunnableDDL("DROP TABLE `stale_state`")) + assert.Equal(t, "DROP TABLE `stale_state`;", storageSchemaRunnableDDL("DROP TABLE `stale_state`;")) + assert.Equal(t, "DROP TABLE `stale_state`;", storageSchemaRunnableDDL(" DROP TABLE `stale_state` ;; ")) + assert.Equal(t, "", storageSchemaRunnableDDL(""), "nothing to terminate stays untouched") +} + +// The headline is the line an operator reads first, so it names the database +// and the binary whose embedded schema produced the diff. The version matters: +// the answer is only meaningful against the schema files it was compared with. +func TestStorageSchemaHeadline(t *testing.T) { + converged := storageSchemaHeadline(&apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + Version: "v1.2.3", + Converged: true, + }) + assert.Equal(t, "schemabot (mysql) is converged, against the schema embedded in v1.2.3.", converged) + + outstanding := storageSchemaHeadline(&apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "postgres", + Deployment: "west", + Environment: "production", + Version: "v1.2.3", + Outstanding: []apitypes.StorageSchemaStatement{{Table: "applies"}, {Table: "checks"}}, + Destructive: []apitypes.StorageSchemaStatement{{Table: "stale_state"}}, + Manual: []apitypes.StorageSchemaStatement{{Table: "plans"}}, + }) + assert.Equal(t, + "schemabot (postgres) on deployment west in production needs 4 statements: "+ + "2 outstanding, 1 destructive, 1 needing manual remediation, against the schema embedded in v1.2.3.", + outstanding) + + assert.Equal(t, "the storage database needs 1 statement: 1 outstanding.", + storageSchemaHeadline(&apitypes.StorageSchemaReport{ + Outstanding: []apitypes.StorageSchemaStatement{{Table: "applies"}}, + }), "an unnamed database and a single statement both still read as a sentence") +} + +// A converged report prints one line and nothing else: there is no section to +// read and no next step to take. +func TestRenderStorageSchemaReport_Converged(t *testing.T) { + var out bytes.Buffer + require.NoError(t, renderStorageSchemaReport(&out, &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + Converged: true, + }, []string{"this hint has nothing to hint at"})) + + assert.Equal(t, "schemabot (mysql) is converged.\n", out.String()) +} + +// An outstanding report prints the statements bare and one per line, grouped +// under headings that say what will happen to each group, so a whole section +// can be selected and pasted into a client as it stands. +func TestRenderStorageSchemaReport_Sections(t *testing.T) { + var out bytes.Buffer + require.NoError(t, renderStorageSchemaReport(&out, &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + Outstanding: []apitypes.StorageSchemaStatement{{ + Table: "applies", + Operation: "alter_table", + DDL: "ALTER TABLE `applies` ADD COLUMN `caller` varchar(255) NOT NULL DEFAULT ''", + }}, + Destructive: []apitypes.StorageSchemaStatement{{ + Table: "stale_state", + Operation: "drop_table", + DDL: "DROP TABLE `stale_state`", + Reason: "DROP TABLE destroys data", + }}, + }, []string{"Converge it with: schemabot storage apply"})) + + assert.Equal(t, ""+ + "schemabot (mysql) needs 2 statements: 1 outstanding, 1 destructive.\n"+ + "\n"+storageSchemaOutstandingTitle+" (1):\n"+ + "\n"+ + "ALTER TABLE `applies` ADD COLUMN `caller` varchar(255) NOT NULL DEFAULT '';\n"+ + "\nDestructive, and refused; surplus state stays in place (1):\n"+ + "\n"+ + "-- stale_state: DROP TABLE destroys data\n"+ + "DROP TABLE `stale_state`;\n"+ + "\nConverge it with: schemabot storage apply\n", + out.String()) +} + +// The destructive heading says whether those statements would run, because that +// is the whole disposition of the section: refused means surplus state stays in +// place on purpose, permitted means the next convergence destroys it. +func TestStorageSchemaDestructiveTitle(t *testing.T) { + assert.Contains(t, storageSchemaDestructiveTitle(&apitypes.StorageSchemaReport{}), "refused") + assert.Contains(t, storageSchemaDestructiveTitle(&apitypes.StorageSchemaReport{DestructiveAllowed: true}), "permitted to run") +} + +// A manual entry blocks everything behind it, so it is named as its own section +// rather than folded in with statements that will run on their own. +func TestRenderStorageSchemaReport_ManualSection(t *testing.T) { + var out bytes.Buffer + require.NoError(t, renderStorageSchemaReport(&out, &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "postgres", + Manual: []apitypes.StorageSchemaStatement{{ + Table: "checks", + Operation: "add_column", + DDL: `ALTER TABLE "checks" ADD COLUMN "head_sha" varchar(64) NOT NULL`, + Reason: "column is NOT NULL without a DEFAULT", + }}, + }, nil)) + + assert.Contains(t, out.String(), "Needs manual remediation before anything converges (1):") + assert.Contains(t, out.String(), "-- checks: column is NOT NULL without a DEFAULT") + assert.Contains(t, out.String(), `ALTER TABLE "checks" ADD COLUMN "head_sha" varchar(64) NOT NULL;`) +} + +// A diff names the next step in the command form the operator invoked, and the +// direct form never echoes the DSN back — it may carry a password, and the +// operator already has it. +func TestStorageSchemaDiffHints(t *testing.T) { + local := storageSchemaDiffHints(&StorageDiffCmd{}) + require.Len(t, local, 1) + assert.Contains(t, local[0], "storage apply") + assert.NotContains(t, local[0], "--deployment") + + deployment := storageSchemaDiffHints(&StorageDiffCmd{ + storageSchemaTargetFlags: storageSchemaTargetFlags{Deployment: "west", Environment: "production"}, + }) + require.Len(t, deployment, 1) + assert.Contains(t, deployment[0], "storage apply --deployment west -e production") + + const secret = "root:hunter2@tcp(db.example:3306)/schemabot" + dsn := storageSchemaDiffHints(&StorageDiffCmd{storageSchemaTargetFlags: storageSchemaTargetFlags{DSN: secret}}) + require.Len(t, dsn, 1) + assert.Contains(t, dsn[0], "--dsn ") + assert.NotContains(t, dsn[0], "hunter2") + + config := storageSchemaDiffHints(&StorageDiffCmd{ + storageSchemaTargetFlags: storageSchemaTargetFlags{Config: "/etc/schemabot/config.yaml"}, + }) + require.Len(t, config, 1) + assert.Contains(t, config[0], "--config /etc/schemabot/config.yaml") +} + +// A clean convergence states that nothing is left, rather than leaving an +// operator to read the absence of a section as good news. +func TestRenderStorageSchemaConvergence_Clean(t *testing.T) { + var out bytes.Buffer + planned := &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + Outstanding: []apitypes.StorageSchemaStatement{ + {Table: "applies", DDL: "ALTER TABLE `applies` ADD COLUMN `caller` varchar(255) NOT NULL DEFAULT ''"}, + {Table: "checks", DDL: "CREATE TABLE `checks` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)"}, + }, + } + remaining := &apitypes.StorageSchemaReport{Database: "schemabot", Dialect: "mysql", Converged: true} + + require.NoError(t, renderStorageSchemaConvergence(&out, planned, remaining)) + assert.Equal(t, ""+ + "Ran 2 statements against schemabot (mysql).\n"+ + "schemabot (mysql) is converged.\n", + out.String()) +} + +// A convergence that left statements behind prints them with the reason they +// did not run, so the operator's next move is on the screen rather than in the +// documentation. +func TestRenderStorageSchemaConvergence_LeftBehind(t *testing.T) { + var out bytes.Buffer + planned := &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + Outstanding: []apitypes.StorageSchemaStatement{{Table: "applies", DDL: "ALTER TABLE `applies` ADD COLUMN `caller` varchar(255) NOT NULL DEFAULT ''"}}, + } + remaining := &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + Destructive: []apitypes.StorageSchemaStatement{{ + Table: "stale_state", + DDL: "DROP TABLE `stale_state`", + Reason: "DROP TABLE destroys data", + }}, + } + + require.NoError(t, renderStorageSchemaConvergence(&out, planned, remaining)) + assert.Contains(t, out.String(), "Ran 1 statement against schemabot (mysql).") + assert.Contains(t, out.String(), "DROP TABLE `stale_state`;") + assert.Contains(t, out.String(), "--allow-destructive") + assert.NotContains(t, out.String(), "is converged") +} + +// A deployment's report is labelled with the deployment it came from, so an +// operator reading a control plane's answer can tell whose storage it describes. +func TestStorageSchemaDatabaseLabel(t *testing.T) { + assert.Equal(t, "schemabot (mysql)", storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{ + Database: "schemabot", Dialect: "mysql", + })) + assert.Equal(t, "schemabot (postgres) on deployment west in production", + storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{ + Database: "schemabot", Dialect: "postgres", Deployment: "west", Environment: "production", + })) + assert.Equal(t, "the storage database", storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{}), + "a report with no database name still reads as a sentence") +} + +// An error names the storage that was being read. A failure that said only +// that a read failed would leave an operator unable to tell whether they had +// reached the deployment they asked for. +func TestStorageSchemaTargetSuffix(t *testing.T) { + assert.Empty(t, storageSchemaTargetSuffix("", "")) + assert.Equal(t, " for deployment west in production", storageSchemaTargetSuffix("west", "production")) +} + +// The exit status is the machine-readable half of a diff's answer: a +// pre-deploy gate has to tell "converged" from "statements outstanding" from +// "the read failed", and two of those three are not failures of the command. +func TestExitCodeFor(t *testing.T) { + assert.Equal(t, 0, ExitCodeFor(nil)) + assert.Equal(t, 1, ExitCodeFor(errors.New("storage unreachable"))) + assert.Equal(t, ExitStorageSchemaOutstanding, ExitCodeFor(exitStorageSchemaOutstanding())) + assert.Equal(t, ExitStorageSchemaOutstanding, + ExitCodeFor(fmt.Errorf("wrapped: %w", exitStorageSchemaOutstanding())), + "a wrapped request for a status is still honored") + assert.Equal(t, 1, ExitCodeFor(&ExitCodeError{Err: errors.New("no status asked for")})) +} + +// A status request prints nothing further: the report is already on the screen, +// and an "Error:" line under it would read as a failure of the read. +func TestSilentExitIsSilent(t *testing.T) { + err := exitStorageSchemaOutstanding() + require.Error(t, err) + assert.ErrorIs(t, err, ErrSilent) +} diff --git a/pkg/cmd/main.go b/pkg/cmd/main.go index aac0d7100..8c99fff19 100644 --- a/pkg/cmd/main.go +++ b/pkg/cmd/main.go @@ -58,7 +58,7 @@ type CLI struct { Settings commands.SettingsCmd `cmd:"" help:"View or update schema change settings"` Webhooks commands.WebhooksCmd `cmd:"" help:"Manage GitHub App webhook deliveries"` Checks commands.ChecksCmd `cmd:"" help:"Manage SchemaBot Check Runs on PRs"` - Storage commands.StorageCmd `cmd:"" help:"Operate directly on SchemaBot's storage database"` + Storage commands.StorageCmd `cmd:"" help:"Inspect and maintain SchemaBot's own storage database"` Local commands.LocalCmd `cmd:"" hidden:"" help:"Internal local runtime host"` Serve commands.ServeCmd `cmd:"" help:"Start the SchemaBot HTTP API server"` } @@ -164,11 +164,14 @@ func main() { signal.Stop(sigCh) cancelRun() if err != nil { - // ErrSilent means the error was already displayed - just exit with code 1 + // ErrSilent means the error was already displayed - just exit. if !errors.Is(err, commands.ErrSilent) { fmt.Fprintf(os.Stderr, "\033[31mError: %v\033[0m\n", err) } - os.Exit(1) + // A command may ask for its own status when a caller scripting it needs + // to tell two successful-but-different outcomes apart; anything else + // exits 1. + os.Exit(commands.ExitCodeFor(err)) } } diff --git a/pkg/proto/tern.proto b/pkg/proto/tern.proto index e1eb2a497..0a5805d05 100644 --- a/pkg/proto/tern.proto +++ b/pkg/proto/tern.proto @@ -216,6 +216,44 @@ service Tern { body: "*" }; } + + // StorageSchemaDiff reports the DDL outstanding between the embedded storage + // schema of the binary serving this RPC and the live storage database that + // binary uses. Unlike every other RPC here, it is about the serving + // instance's own bookkeeping database rather than a target database an + // operator asked to change. + // + // It is strictly read-only: it reads the live catalog and computes a diff, + // executing no DDL and taking no lock, so it is safe to call at any time, + // including while the same storage is being converged. + // + // The answer comes from the serving binary's own embedded files, which is the + // whole point of routing it here rather than computing it from a release pin + // on the caller's side: a caller's pin says which release it was built + // against, not what this storage converged to, and those diverge exactly when + // a deploy has failed to converge. + rpc StorageSchemaDiff(StorageSchemaDiffRequest) returns (StorageSchemaDiffResponse) { + option (google.api.http) = { + post: "/v1/storage-schema/diff" + body: "*" + }; + } + + // StorageSchemaApply converges the serving instance's own storage schema and + // reports what it found and what it left behind. + // + // The convergence is the serving binary's startup bootstrap, called + // unchanged: the same differ, the same destructive-statement refusal, and the + // same advisory lock that serializes it against every other instance of that + // deployment. Destructive statements are refused unless the serving + // instance's storage config allows them or allow_destructive opts in for this + // call. + rpc StorageSchemaApply(StorageSchemaApplyRequest) returns (StorageSchemaApplyResponse) { + option (google.api.http) = { + post: "/v1/storage-schema/apply" + body: "*" + }; + } } // Engine represents the schema change engine type. @@ -923,3 +961,88 @@ message StartResponse { // Number of tasks skipped (not in stopped state). int64 skipped_count = 4; } + +// StorageSchemaDiffRequest asks the serving instance for the DDL outstanding +// on its own storage database. It deliberately carries no target: the instance +// answers for the storage it uses, and nothing a caller sends can point it at +// a different database. +message StorageSchemaDiffRequest { + // AllowDestructive reports the destructive statements as allowed rather than + // refused, matching what an apply carrying the same flag would run. It does + // not execute anything — a diff never does — and it does not change which + // statements are classified destructive, only whether the report says they + // would run. + bool allow_destructive = 1; +} + +// StorageSchemaStatement is one outstanding storage-schema statement. +message StorageSchemaStatement { + // Storage table the statement acts on. + string table = 1; + // Statement kind in the storage layer's own vocabulary (create_table, + // alter_table, add_column, create_index, ...). + string operation = 2; + // The statement itself, runnable as printed. + string ddl = 3; + // Why the statement is classified destructive, or why it needs manual + // remediation. Empty for a statement that runs automatically. + string reason = 4; +} + +// StorageSchemaReport is what one storage database needs to match the embedded +// schema of the binary that produced the report. The three statement sets are +// disjoint and have different dispositions, so a caller never has to work out +// which statements would actually run. +message StorageSchemaReport { + // Storage database family: "mysql" or "postgres". + string dialect = 1; + // The live database the diff read, as the server reports it. + string database = 2; + // SchemaBot version of the binary whose embedded schema produced the diff. + // Reported for attribution only; the diff itself is computed from the files. + string version = 3; + // Statements that converge the schema and run automatically, in the order + // the convergence would run them. + repeated StorageSchemaStatement outstanding = 4; + // Statements classified as destroying data. Refused unless destructive + // changes are allowed. + repeated StorageSchemaStatement destructive = 5; + // Whether the destructive statements would actually run. + bool destructive_allowed = 6; + // Changes that cannot run automatically, each naming the situation and its + // remediation. Any entry aborts the whole convergence before a single + // statement executes. + repeated StorageSchemaStatement manual = 7; +} + +// StorageSchemaDiffResponse carries the outstanding storage DDL. +message StorageSchemaDiffResponse { + StorageSchemaReport report = 1; +} + +// StorageSchemaApplyRequest converges the serving instance's own storage +// schema. +message StorageSchemaApplyRequest { + // AllowDestructive permits the destructive statements this call would + // otherwise refuse. It is an addition to the serving instance's storage + // config, never a restriction of it: a deployment whose config already + // allows destructive storage changes converges the same way a boot would, + // with or without this flag. + bool allow_destructive = 1; + // Caller identifies the operator who issued the command, as resolved by the + // plane that accepted it, so the data plane's logs attribute the + // convergence to a person rather than to a control plane. + string caller = 2; +} + +// StorageSchemaApplyResponse brackets the convergence with the report from +// before it and the report from after it. +message StorageSchemaApplyResponse { + // What was outstanding before the convergence ran. + StorageSchemaReport planned = 1; + // What is still outstanding after it. Empty on a clean convergence; it + // carries the refused destructive statements when the live database holds + // state the serving binary's schema does not declare, which is the expected + // steady state during a rollback rather than a failure. + StorageSchemaReport remaining = 2; +} diff --git a/pkg/proto/ternv1/tern.pb.go b/pkg/proto/ternv1/tern.pb.go index 3179d88fb..c929a7b8c 100644 --- a/pkg/proto/ternv1/tern.pb.go +++ b/pkg/proto/ternv1/tern.pb.go @@ -3920,6 +3920,408 @@ func (x *StartResponse) GetSkippedCount() int64 { return 0 } +// StorageSchemaDiffRequest asks the serving instance for the DDL outstanding +// on its own storage database. It deliberately carries no target: the instance +// answers for the storage it uses, and nothing a caller sends can point it at +// a different database. +type StorageSchemaDiffRequest struct { + state protoimpl.MessageState `protogen:"open.v1"` + // AllowDestructive reports the destructive statements as allowed rather than + // refused, matching what an apply carrying the same flag would run. It does + // not execute anything — a diff never does — and it does not change which + // statements are classified destructive, only whether the report says they + // would run. + AllowDestructive bool `protobuf:"varint,1,opt,name=allow_destructive,json=allowDestructive,proto3" json:"allow_destructive,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *StorageSchemaDiffRequest) Reset() { + *x = StorageSchemaDiffRequest{} + mi := &file_tern_proto_msgTypes[43] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *StorageSchemaDiffRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StorageSchemaDiffRequest) ProtoMessage() {} + +func (x *StorageSchemaDiffRequest) ProtoReflect() protoreflect.Message { + mi := &file_tern_proto_msgTypes[43] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StorageSchemaDiffRequest.ProtoReflect.Descriptor instead. +func (*StorageSchemaDiffRequest) Descriptor() ([]byte, []int) { + return file_tern_proto_rawDescGZIP(), []int{43} +} + +func (x *StorageSchemaDiffRequest) GetAllowDestructive() bool { + if x != nil { + return x.AllowDestructive + } + return false +} + +// StorageSchemaStatement is one outstanding storage-schema statement. +type StorageSchemaStatement struct { + state protoimpl.MessageState `protogen:"open.v1"` + // Storage table the statement acts on. + Table string `protobuf:"bytes,1,opt,name=table,proto3" json:"table,omitempty"` + // Statement kind in the storage layer's own vocabulary (create_table, + // alter_table, add_column, create_index, ...). + Operation string `protobuf:"bytes,2,opt,name=operation,proto3" json:"operation,omitempty"` + // The statement itself, runnable as printed. + Ddl string `protobuf:"bytes,3,opt,name=ddl,proto3" json:"ddl,omitempty"` + // Why the statement is classified destructive, or why it needs manual + // remediation. Empty for a statement that runs automatically. + Reason string `protobuf:"bytes,4,opt,name=reason,proto3" json:"reason,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *StorageSchemaStatement) Reset() { + *x = StorageSchemaStatement{} + mi := &file_tern_proto_msgTypes[44] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *StorageSchemaStatement) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StorageSchemaStatement) ProtoMessage() {} + +func (x *StorageSchemaStatement) ProtoReflect() protoreflect.Message { + mi := &file_tern_proto_msgTypes[44] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StorageSchemaStatement.ProtoReflect.Descriptor instead. +func (*StorageSchemaStatement) Descriptor() ([]byte, []int) { + return file_tern_proto_rawDescGZIP(), []int{44} +} + +func (x *StorageSchemaStatement) GetTable() string { + if x != nil { + return x.Table + } + return "" +} + +func (x *StorageSchemaStatement) GetOperation() string { + if x != nil { + return x.Operation + } + return "" +} + +func (x *StorageSchemaStatement) GetDdl() string { + if x != nil { + return x.Ddl + } + return "" +} + +func (x *StorageSchemaStatement) GetReason() string { + if x != nil { + return x.Reason + } + return "" +} + +// StorageSchemaReport is what one storage database needs to match the embedded +// schema of the binary that produced the report. The three statement sets are +// disjoint and have different dispositions, so a caller never has to work out +// which statements would actually run. +type StorageSchemaReport struct { + state protoimpl.MessageState `protogen:"open.v1"` + // Storage database family: "mysql" or "postgres". + Dialect string `protobuf:"bytes,1,opt,name=dialect,proto3" json:"dialect,omitempty"` + // The live database the diff read, as the server reports it. + Database string `protobuf:"bytes,2,opt,name=database,proto3" json:"database,omitempty"` + // SchemaBot version of the binary whose embedded schema produced the diff. + // Reported for attribution only; the diff itself is computed from the files. + Version string `protobuf:"bytes,3,opt,name=version,proto3" json:"version,omitempty"` + // Statements that converge the schema and run automatically, in the order + // the convergence would run them. + Outstanding []*StorageSchemaStatement `protobuf:"bytes,4,rep,name=outstanding,proto3" json:"outstanding,omitempty"` + // Statements classified as destroying data. Refused unless destructive + // changes are allowed. + Destructive []*StorageSchemaStatement `protobuf:"bytes,5,rep,name=destructive,proto3" json:"destructive,omitempty"` + // Whether the destructive statements would actually run. + DestructiveAllowed bool `protobuf:"varint,6,opt,name=destructive_allowed,json=destructiveAllowed,proto3" json:"destructive_allowed,omitempty"` + // Changes that cannot run automatically, each naming the situation and its + // remediation. Any entry aborts the whole convergence before a single + // statement executes. + Manual []*StorageSchemaStatement `protobuf:"bytes,7,rep,name=manual,proto3" json:"manual,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *StorageSchemaReport) Reset() { + *x = StorageSchemaReport{} + mi := &file_tern_proto_msgTypes[45] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *StorageSchemaReport) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StorageSchemaReport) ProtoMessage() {} + +func (x *StorageSchemaReport) ProtoReflect() protoreflect.Message { + mi := &file_tern_proto_msgTypes[45] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StorageSchemaReport.ProtoReflect.Descriptor instead. +func (*StorageSchemaReport) Descriptor() ([]byte, []int) { + return file_tern_proto_rawDescGZIP(), []int{45} +} + +func (x *StorageSchemaReport) GetDialect() string { + if x != nil { + return x.Dialect + } + return "" +} + +func (x *StorageSchemaReport) GetDatabase() string { + if x != nil { + return x.Database + } + return "" +} + +func (x *StorageSchemaReport) GetVersion() string { + if x != nil { + return x.Version + } + return "" +} + +func (x *StorageSchemaReport) GetOutstanding() []*StorageSchemaStatement { + if x != nil { + return x.Outstanding + } + return nil +} + +func (x *StorageSchemaReport) GetDestructive() []*StorageSchemaStatement { + if x != nil { + return x.Destructive + } + return nil +} + +func (x *StorageSchemaReport) GetDestructiveAllowed() bool { + if x != nil { + return x.DestructiveAllowed + } + return false +} + +func (x *StorageSchemaReport) GetManual() []*StorageSchemaStatement { + if x != nil { + return x.Manual + } + return nil +} + +// StorageSchemaDiffResponse carries the outstanding storage DDL. +type StorageSchemaDiffResponse struct { + state protoimpl.MessageState `protogen:"open.v1"` + Report *StorageSchemaReport `protobuf:"bytes,1,opt,name=report,proto3" json:"report,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *StorageSchemaDiffResponse) Reset() { + *x = StorageSchemaDiffResponse{} + mi := &file_tern_proto_msgTypes[46] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *StorageSchemaDiffResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StorageSchemaDiffResponse) ProtoMessage() {} + +func (x *StorageSchemaDiffResponse) ProtoReflect() protoreflect.Message { + mi := &file_tern_proto_msgTypes[46] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StorageSchemaDiffResponse.ProtoReflect.Descriptor instead. +func (*StorageSchemaDiffResponse) Descriptor() ([]byte, []int) { + return file_tern_proto_rawDescGZIP(), []int{46} +} + +func (x *StorageSchemaDiffResponse) GetReport() *StorageSchemaReport { + if x != nil { + return x.Report + } + return nil +} + +// StorageSchemaApplyRequest converges the serving instance's own storage +// schema. +type StorageSchemaApplyRequest struct { + state protoimpl.MessageState `protogen:"open.v1"` + // AllowDestructive permits the destructive statements this call would + // otherwise refuse. It is an addition to the serving instance's storage + // config, never a restriction of it: a deployment whose config already + // allows destructive storage changes converges the same way a boot would, + // with or without this flag. + AllowDestructive bool `protobuf:"varint,1,opt,name=allow_destructive,json=allowDestructive,proto3" json:"allow_destructive,omitempty"` + // Caller identifies the operator who issued the command, as resolved by the + // plane that accepted it, so the data plane's logs attribute the + // convergence to a person rather than to a control plane. + Caller string `protobuf:"bytes,2,opt,name=caller,proto3" json:"caller,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *StorageSchemaApplyRequest) Reset() { + *x = StorageSchemaApplyRequest{} + mi := &file_tern_proto_msgTypes[47] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *StorageSchemaApplyRequest) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StorageSchemaApplyRequest) ProtoMessage() {} + +func (x *StorageSchemaApplyRequest) ProtoReflect() protoreflect.Message { + mi := &file_tern_proto_msgTypes[47] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StorageSchemaApplyRequest.ProtoReflect.Descriptor instead. +func (*StorageSchemaApplyRequest) Descriptor() ([]byte, []int) { + return file_tern_proto_rawDescGZIP(), []int{47} +} + +func (x *StorageSchemaApplyRequest) GetAllowDestructive() bool { + if x != nil { + return x.AllowDestructive + } + return false +} + +func (x *StorageSchemaApplyRequest) GetCaller() string { + if x != nil { + return x.Caller + } + return "" +} + +// StorageSchemaApplyResponse brackets the convergence with the report from +// before it and the report from after it. +type StorageSchemaApplyResponse struct { + state protoimpl.MessageState `protogen:"open.v1"` + // What was outstanding before the convergence ran. + Planned *StorageSchemaReport `protobuf:"bytes,1,opt,name=planned,proto3" json:"planned,omitempty"` + // What is still outstanding after it. Empty on a clean convergence; it + // carries the refused destructive statements when the live database holds + // state the serving binary's schema does not declare, which is the expected + // steady state during a rollback rather than a failure. + Remaining *StorageSchemaReport `protobuf:"bytes,2,opt,name=remaining,proto3" json:"remaining,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache +} + +func (x *StorageSchemaApplyResponse) Reset() { + *x = StorageSchemaApplyResponse{} + mi := &file_tern_proto_msgTypes[48] + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + ms.StoreMessageInfo(mi) +} + +func (x *StorageSchemaApplyResponse) String() string { + return protoimpl.X.MessageStringOf(x) +} + +func (*StorageSchemaApplyResponse) ProtoMessage() {} + +func (x *StorageSchemaApplyResponse) ProtoReflect() protoreflect.Message { + mi := &file_tern_proto_msgTypes[48] + if x != nil { + ms := protoimpl.X.MessageStateOf(protoimpl.Pointer(x)) + if ms.LoadMessageInfo() == nil { + ms.StoreMessageInfo(mi) + } + return ms + } + return mi.MessageOf(x) +} + +// Deprecated: Use StorageSchemaApplyResponse.ProtoReflect.Descriptor instead. +func (*StorageSchemaApplyResponse) Descriptor() ([]byte, []int) { + return file_tern_proto_rawDescGZIP(), []int{48} +} + +func (x *StorageSchemaApplyResponse) GetPlanned() *StorageSchemaReport { + if x != nil { + return x.Planned + } + return nil +} + +func (x *StorageSchemaApplyResponse) GetRemaining() *StorageSchemaReport { + if x != nil { + return x.Remaining + } + return nil +} + var File_tern_proto protoreflect.FileDescriptor const file_tern_proto_rawDesc = "" + @@ -4263,7 +4665,30 @@ const file_tern_proto_rawDesc = "" + "\baccepted\x18\x01 \x01(\bR\baccepted\x12#\n" + "\rerror_message\x18\x02 \x01(\tR\ferrorMessage\x12#\n" + "\rstarted_count\x18\x03 \x01(\x03R\fstartedCount\x12#\n" + - "\rskipped_count\x18\x04 \x01(\x03R\fskippedCount*[\n" + + "\rskipped_count\x18\x04 \x01(\x03R\fskippedCount\"G\n" + + "\x18StorageSchemaDiffRequest\x12+\n" + + "\x11allow_destructive\x18\x01 \x01(\bR\x10allowDestructive\"v\n" + + "\x16StorageSchemaStatement\x12\x14\n" + + "\x05table\x18\x01 \x01(\tR\x05table\x12\x1c\n" + + "\toperation\x18\x02 \x01(\tR\toperation\x12\x10\n" + + "\x03ddl\x18\x03 \x01(\tR\x03ddl\x12\x16\n" + + "\x06reason\x18\x04 \x01(\tR\x06reason\"\xd5\x02\n" + + "\x13StorageSchemaReport\x12\x18\n" + + "\adialect\x18\x01 \x01(\tR\adialect\x12\x1a\n" + + "\bdatabase\x18\x02 \x01(\tR\bdatabase\x12\x18\n" + + "\aversion\x18\x03 \x01(\tR\aversion\x12A\n" + + "\voutstanding\x18\x04 \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\voutstanding\x12A\n" + + "\vdestructive\x18\x05 \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\vdestructive\x12/\n" + + "\x13destructive_allowed\x18\x06 \x01(\bR\x12destructiveAllowed\x127\n" + + "\x06manual\x18\a \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\x06manual\"Q\n" + + "\x19StorageSchemaDiffResponse\x124\n" + + "\x06report\x18\x01 \x01(\v2\x1c.tern.v1.StorageSchemaReportR\x06report\"`\n" + + "\x19StorageSchemaApplyRequest\x12+\n" + + "\x11allow_destructive\x18\x01 \x01(\bR\x10allowDestructive\x12\x16\n" + + "\x06caller\x18\x02 \x01(\tR\x06caller\"\x90\x01\n" + + "\x1aStorageSchemaApplyResponse\x126\n" + + "\aplanned\x18\x01 \x01(\v2\x1c.tern.v1.StorageSchemaReportR\aplanned\x12:\n" + + "\tremaining\x18\x02 \x01(\v2\x1c.tern.v1.StorageSchemaReportR\tremaining*[\n" + "\x06Engine\x12\x11\n" + "\rENGINE_SPIRIT\x10\x00\x12\x16\n" + "\x12ENGINE_PLANETSCALE\x10\x01\x12\x11\n" + @@ -4308,7 +4733,8 @@ const file_tern_proto_rawDesc = "" + "\x17CHANGE_TYPE_CREATE_VIEW\x10\t*T\n" + "\x11PullCatalogDetail\x12\x1d\n" + "\x19PULL_CATALOG_DETAIL_BASIC\x10\x00\x12 \n" + - "\x1cPULL_CATALOG_DETAIL_DETAILED\x10\x012\xc0\b\n" + + "\x1cPULL_CATALOG_DETAIL_DETAILED\x10\x012\xc5\n" + + "\n" + "\x04Tern\x12a\n" + "\n" + "PullSchema\x12\x1a.tern.v1.PullSchemaRequest\x1a\x1b.tern.v1.PullSchemaResponse\"\x1a\x82\xd3\xe4\x93\x02\x14:\x01*\"\x0f/v1/pull-schema\x12H\n" + @@ -4327,7 +4753,9 @@ const file_tern_proto_rawDesc = "" + "\x04Stop\x12\x14.tern.v1.StopRequest\x1a\x15.tern.v1.StopResponse\"\x13\x82\xd3\xe4\x93\x02\r:\x01*\"\b/v1/stop\x12P\n" + "\x06Cancel\x12\x16.tern.v1.CancelRequest\x1a\x17.tern.v1.CancelResponse\"\x15\x82\xd3\xe4\x93\x02\x0f:\x01*\"\n" + "/v1/cancel\x12L\n" + - "\x05Start\x12\x15.tern.v1.StartRequest\x1a\x16.tern.v1.StartResponse\"\x14\x82\xd3\xe4\x93\x02\x0e:\x01*\"\t/v1/startB-Z+github.com/block/schemabot/pkg/proto/ternv1b\x06proto3" + "\x05Start\x12\x15.tern.v1.StartRequest\x1a\x16.tern.v1.StartResponse\"\x14\x82\xd3\xe4\x93\x02\x0e:\x01*\"\t/v1/start\x12~\n" + + "\x11StorageSchemaDiff\x12!.tern.v1.StorageSchemaDiffRequest\x1a\".tern.v1.StorageSchemaDiffResponse\"\"\x82\xd3\xe4\x93\x02\x1c:\x01*\"\x17/v1/storage-schema/diff\x12\x82\x01\n" + + "\x12StorageSchemaApply\x12\".tern.v1.StorageSchemaApplyRequest\x1a#.tern.v1.StorageSchemaApplyResponse\"#\x82\xd3\xe4\x93\x02\x1d:\x01*\"\x18/v1/storage-schema/applyB-Z+github.com/block/schemabot/pkg/proto/ternv1b\x06proto3" var ( file_tern_proto_rawDescOnce sync.Once @@ -4342,85 +4770,91 @@ func file_tern_proto_rawDescGZIP() []byte { } var file_tern_proto_enumTypes = make([]protoimpl.EnumInfo, 4) -var file_tern_proto_msgTypes = make([]protoimpl.MessageInfo, 55) +var file_tern_proto_msgTypes = make([]protoimpl.MessageInfo, 61) var file_tern_proto_goTypes = []any{ - (Engine)(0), // 0: tern.v1.Engine - (State)(0), // 1: tern.v1.State - (ChangeType)(0), // 2: tern.v1.ChangeType - (PullCatalogDetail)(0), // 3: tern.v1.PullCatalogDetail - (*SchemaFiles)(nil), // 4: tern.v1.SchemaFiles - (*PullSchemaRequest)(nil), // 5: tern.v1.PullSchemaRequest - (*PulledNamespace)(nil), // 6: tern.v1.PulledNamespace - (*NamespaceCatalog)(nil), // 7: tern.v1.NamespaceCatalog - (*TableCatalog)(nil), // 8: tern.v1.TableCatalog - (*ColumnCatalog)(nil), // 9: tern.v1.ColumnCatalog - (*IndexCatalog)(nil), // 10: tern.v1.IndexCatalog - (*ForeignKeyCatalog)(nil), // 11: tern.v1.ForeignKeyCatalog - (*PullSchemaResponse)(nil), // 12: tern.v1.PullSchemaResponse - (*PlanRequest)(nil), // 13: tern.v1.PlanRequest - (*TableChange)(nil), // 14: tern.v1.TableChange - (*SchemaChange)(nil), // 15: tern.v1.SchemaChange - (*LintViolation)(nil), // 16: tern.v1.LintViolation - (*ExistingCopy)(nil), // 17: tern.v1.ExistingCopy - (*ExemptTables)(nil), // 18: tern.v1.ExemptTables - (*ShardPlan)(nil), // 19: tern.v1.ShardPlan - (*PlanResponse)(nil), // 20: tern.v1.PlanResponse - (*PlanDiffResponse)(nil), // 21: tern.v1.PlanDiffResponse - (*ApplyRequest)(nil), // 22: tern.v1.ApplyRequest - (*ApplyConflict)(nil), // 23: tern.v1.ApplyConflict - (*ApplyResponse)(nil), // 24: tern.v1.ApplyResponse - (*ProgressRequest)(nil), // 25: tern.v1.ProgressRequest - (*LogsRequest)(nil), // 26: tern.v1.LogsRequest - (*ApplyLog)(nil), // 27: tern.v1.ApplyLog - (*LogsResponse)(nil), // 28: tern.v1.LogsResponse - (*ShardProgress)(nil), // 29: tern.v1.ShardProgress - (*TableProgress)(nil), // 30: tern.v1.TableProgress - (*SettledControlRequest)(nil), // 31: tern.v1.SettledControlRequest - (*ProgressResponse)(nil), // 32: tern.v1.ProgressResponse - (*CutoverRequest)(nil), // 33: tern.v1.CutoverRequest - (*CutoverResponse)(nil), // 34: tern.v1.CutoverResponse - (*RevertRequest)(nil), // 35: tern.v1.RevertRequest - (*RevertResponse)(nil), // 36: tern.v1.RevertResponse - (*SkipRevertRequest)(nil), // 37: tern.v1.SkipRevertRequest - (*SkipRevertResponse)(nil), // 38: tern.v1.SkipRevertResponse - (*HealthRequest)(nil), // 39: tern.v1.HealthRequest - (*HealthResponse)(nil), // 40: tern.v1.HealthResponse - (*StopRequest)(nil), // 41: tern.v1.StopRequest - (*StopResponse)(nil), // 42: tern.v1.StopResponse - (*CancelRequest)(nil), // 43: tern.v1.CancelRequest - (*CancelResponse)(nil), // 44: tern.v1.CancelResponse - (*StartRequest)(nil), // 45: tern.v1.StartRequest - (*StartResponse)(nil), // 46: tern.v1.StartResponse - nil, // 47: tern.v1.SchemaFiles.FilesEntry - nil, // 48: tern.v1.PulledNamespace.TablesEntry - nil, // 49: tern.v1.PulledNamespace.ArtifactsEntry - nil, // 50: tern.v1.PulledNamespace.TableCatalogEntry - nil, // 51: tern.v1.PullSchemaResponse.NamespacesEntry - nil, // 52: tern.v1.PlanRequest.SchemaFilesEntry - nil, // 53: tern.v1.TableChange.MetadataEntry - nil, // 54: tern.v1.SchemaChange.MetadataEntry - nil, // 55: tern.v1.SchemaChange.OriginalFilesEntry - nil, // 56: tern.v1.ApplyRequest.OptionsEntry - nil, // 57: tern.v1.ApplyRequest.SchemaFilesEntry - nil, // 58: tern.v1.ProgressResponse.MetadataEntry + (Engine)(0), // 0: tern.v1.Engine + (State)(0), // 1: tern.v1.State + (ChangeType)(0), // 2: tern.v1.ChangeType + (PullCatalogDetail)(0), // 3: tern.v1.PullCatalogDetail + (*SchemaFiles)(nil), // 4: tern.v1.SchemaFiles + (*PullSchemaRequest)(nil), // 5: tern.v1.PullSchemaRequest + (*PulledNamespace)(nil), // 6: tern.v1.PulledNamespace + (*NamespaceCatalog)(nil), // 7: tern.v1.NamespaceCatalog + (*TableCatalog)(nil), // 8: tern.v1.TableCatalog + (*ColumnCatalog)(nil), // 9: tern.v1.ColumnCatalog + (*IndexCatalog)(nil), // 10: tern.v1.IndexCatalog + (*ForeignKeyCatalog)(nil), // 11: tern.v1.ForeignKeyCatalog + (*PullSchemaResponse)(nil), // 12: tern.v1.PullSchemaResponse + (*PlanRequest)(nil), // 13: tern.v1.PlanRequest + (*TableChange)(nil), // 14: tern.v1.TableChange + (*SchemaChange)(nil), // 15: tern.v1.SchemaChange + (*LintViolation)(nil), // 16: tern.v1.LintViolation + (*ExistingCopy)(nil), // 17: tern.v1.ExistingCopy + (*ExemptTables)(nil), // 18: tern.v1.ExemptTables + (*ShardPlan)(nil), // 19: tern.v1.ShardPlan + (*PlanResponse)(nil), // 20: tern.v1.PlanResponse + (*PlanDiffResponse)(nil), // 21: tern.v1.PlanDiffResponse + (*ApplyRequest)(nil), // 22: tern.v1.ApplyRequest + (*ApplyConflict)(nil), // 23: tern.v1.ApplyConflict + (*ApplyResponse)(nil), // 24: tern.v1.ApplyResponse + (*ProgressRequest)(nil), // 25: tern.v1.ProgressRequest + (*LogsRequest)(nil), // 26: tern.v1.LogsRequest + (*ApplyLog)(nil), // 27: tern.v1.ApplyLog + (*LogsResponse)(nil), // 28: tern.v1.LogsResponse + (*ShardProgress)(nil), // 29: tern.v1.ShardProgress + (*TableProgress)(nil), // 30: tern.v1.TableProgress + (*SettledControlRequest)(nil), // 31: tern.v1.SettledControlRequest + (*ProgressResponse)(nil), // 32: tern.v1.ProgressResponse + (*CutoverRequest)(nil), // 33: tern.v1.CutoverRequest + (*CutoverResponse)(nil), // 34: tern.v1.CutoverResponse + (*RevertRequest)(nil), // 35: tern.v1.RevertRequest + (*RevertResponse)(nil), // 36: tern.v1.RevertResponse + (*SkipRevertRequest)(nil), // 37: tern.v1.SkipRevertRequest + (*SkipRevertResponse)(nil), // 38: tern.v1.SkipRevertResponse + (*HealthRequest)(nil), // 39: tern.v1.HealthRequest + (*HealthResponse)(nil), // 40: tern.v1.HealthResponse + (*StopRequest)(nil), // 41: tern.v1.StopRequest + (*StopResponse)(nil), // 42: tern.v1.StopResponse + (*CancelRequest)(nil), // 43: tern.v1.CancelRequest + (*CancelResponse)(nil), // 44: tern.v1.CancelResponse + (*StartRequest)(nil), // 45: tern.v1.StartRequest + (*StartResponse)(nil), // 46: tern.v1.StartResponse + (*StorageSchemaDiffRequest)(nil), // 47: tern.v1.StorageSchemaDiffRequest + (*StorageSchemaStatement)(nil), // 48: tern.v1.StorageSchemaStatement + (*StorageSchemaReport)(nil), // 49: tern.v1.StorageSchemaReport + (*StorageSchemaDiffResponse)(nil), // 50: tern.v1.StorageSchemaDiffResponse + (*StorageSchemaApplyRequest)(nil), // 51: tern.v1.StorageSchemaApplyRequest + (*StorageSchemaApplyResponse)(nil), // 52: tern.v1.StorageSchemaApplyResponse + nil, // 53: tern.v1.SchemaFiles.FilesEntry + nil, // 54: tern.v1.PulledNamespace.TablesEntry + nil, // 55: tern.v1.PulledNamespace.ArtifactsEntry + nil, // 56: tern.v1.PulledNamespace.TableCatalogEntry + nil, // 57: tern.v1.PullSchemaResponse.NamespacesEntry + nil, // 58: tern.v1.PlanRequest.SchemaFilesEntry + nil, // 59: tern.v1.TableChange.MetadataEntry + nil, // 60: tern.v1.SchemaChange.MetadataEntry + nil, // 61: tern.v1.SchemaChange.OriginalFilesEntry + nil, // 62: tern.v1.ApplyRequest.OptionsEntry + nil, // 63: tern.v1.ApplyRequest.SchemaFilesEntry + nil, // 64: tern.v1.ProgressResponse.MetadataEntry } var file_tern_proto_depIdxs = []int32{ - 47, // 0: tern.v1.SchemaFiles.files:type_name -> tern.v1.SchemaFiles.FilesEntry + 53, // 0: tern.v1.SchemaFiles.files:type_name -> tern.v1.SchemaFiles.FilesEntry 3, // 1: tern.v1.PullSchemaRequest.catalog_detail:type_name -> tern.v1.PullCatalogDetail - 48, // 2: tern.v1.PulledNamespace.tables:type_name -> tern.v1.PulledNamespace.TablesEntry - 49, // 3: tern.v1.PulledNamespace.artifacts:type_name -> tern.v1.PulledNamespace.ArtifactsEntry + 54, // 2: tern.v1.PulledNamespace.tables:type_name -> tern.v1.PulledNamespace.TablesEntry + 55, // 3: tern.v1.PulledNamespace.artifacts:type_name -> tern.v1.PulledNamespace.ArtifactsEntry 7, // 4: tern.v1.PulledNamespace.namespace_catalog:type_name -> tern.v1.NamespaceCatalog - 50, // 5: tern.v1.PulledNamespace.table_catalog:type_name -> tern.v1.PulledNamespace.TableCatalogEntry + 56, // 5: tern.v1.PulledNamespace.table_catalog:type_name -> tern.v1.PulledNamespace.TableCatalogEntry 9, // 6: tern.v1.TableCatalog.columns:type_name -> tern.v1.ColumnCatalog 10, // 7: tern.v1.TableCatalog.indexes:type_name -> tern.v1.IndexCatalog 11, // 8: tern.v1.TableCatalog.foreign_keys:type_name -> tern.v1.ForeignKeyCatalog - 51, // 9: tern.v1.PullSchemaResponse.namespaces:type_name -> tern.v1.PullSchemaResponse.NamespacesEntry - 52, // 10: tern.v1.PlanRequest.schema_files:type_name -> tern.v1.PlanRequest.SchemaFilesEntry + 57, // 9: tern.v1.PullSchemaResponse.namespaces:type_name -> tern.v1.PullSchemaResponse.NamespacesEntry + 58, // 10: tern.v1.PlanRequest.schema_files:type_name -> tern.v1.PlanRequest.SchemaFilesEntry 2, // 11: tern.v1.TableChange.change_type:type_name -> tern.v1.ChangeType - 53, // 12: tern.v1.TableChange.metadata:type_name -> tern.v1.TableChange.MetadataEntry + 59, // 12: tern.v1.TableChange.metadata:type_name -> tern.v1.TableChange.MetadataEntry 14, // 13: tern.v1.SchemaChange.table_changes:type_name -> tern.v1.TableChange - 54, // 14: tern.v1.SchemaChange.metadata:type_name -> tern.v1.SchemaChange.MetadataEntry - 55, // 15: tern.v1.SchemaChange.original_files:type_name -> tern.v1.SchemaChange.OriginalFilesEntry + 60, // 14: tern.v1.SchemaChange.metadata:type_name -> tern.v1.SchemaChange.MetadataEntry + 61, // 15: tern.v1.SchemaChange.original_files:type_name -> tern.v1.SchemaChange.OriginalFilesEntry 14, // 16: tern.v1.ShardPlan.changes:type_name -> tern.v1.TableChange 0, // 17: tern.v1.PlanResponse.engine:type_name -> tern.v1.Engine 15, // 18: tern.v1.PlanResponse.changes:type_name -> tern.v1.SchemaChange @@ -4432,8 +4866,8 @@ var file_tern_proto_depIdxs = []int32{ 15, // 24: tern.v1.PlanDiffResponse.changes:type_name -> tern.v1.SchemaChange 16, // 25: tern.v1.PlanDiffResponse.lint_violations:type_name -> tern.v1.LintViolation 19, // 26: tern.v1.PlanDiffResponse.shards:type_name -> tern.v1.ShardPlan - 56, // 27: tern.v1.ApplyRequest.options:type_name -> tern.v1.ApplyRequest.OptionsEntry - 57, // 28: tern.v1.ApplyRequest.schema_files:type_name -> tern.v1.ApplyRequest.SchemaFilesEntry + 62, // 27: tern.v1.ApplyRequest.options:type_name -> tern.v1.ApplyRequest.OptionsEntry + 63, // 28: tern.v1.ApplyRequest.schema_files:type_name -> tern.v1.ApplyRequest.SchemaFilesEntry 14, // 29: tern.v1.ApplyRequest.ddl_changes:type_name -> tern.v1.TableChange 23, // 30: tern.v1.ApplyResponse.conflict:type_name -> tern.v1.ApplyConflict 27, // 31: tern.v1.LogsResponse.logs:type_name -> tern.v1.ApplyLog @@ -4442,43 +4876,53 @@ var file_tern_proto_depIdxs = []int32{ 1, // 34: tern.v1.ProgressResponse.state:type_name -> tern.v1.State 0, // 35: tern.v1.ProgressResponse.engine:type_name -> tern.v1.Engine 30, // 36: tern.v1.ProgressResponse.tables:type_name -> tern.v1.TableProgress - 58, // 37: tern.v1.ProgressResponse.metadata:type_name -> tern.v1.ProgressResponse.MetadataEntry + 64, // 37: tern.v1.ProgressResponse.metadata:type_name -> tern.v1.ProgressResponse.MetadataEntry 31, // 38: tern.v1.ProgressResponse.settled_control_requests:type_name -> tern.v1.SettledControlRequest - 8, // 39: tern.v1.PulledNamespace.TableCatalogEntry.value:type_name -> tern.v1.TableCatalog - 6, // 40: tern.v1.PullSchemaResponse.NamespacesEntry.value:type_name -> tern.v1.PulledNamespace - 4, // 41: tern.v1.PlanRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles - 4, // 42: tern.v1.ApplyRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles - 5, // 43: tern.v1.Tern.PullSchema:input_type -> tern.v1.PullSchemaRequest - 13, // 44: tern.v1.Tern.Plan:input_type -> tern.v1.PlanRequest - 13, // 45: tern.v1.Tern.PlanDiff:input_type -> tern.v1.PlanRequest - 22, // 46: tern.v1.Tern.Apply:input_type -> tern.v1.ApplyRequest - 25, // 47: tern.v1.Tern.Progress:input_type -> tern.v1.ProgressRequest - 26, // 48: tern.v1.Tern.Logs:input_type -> tern.v1.LogsRequest - 33, // 49: tern.v1.Tern.Cutover:input_type -> tern.v1.CutoverRequest - 35, // 50: tern.v1.Tern.Revert:input_type -> tern.v1.RevertRequest - 37, // 51: tern.v1.Tern.SkipRevert:input_type -> tern.v1.SkipRevertRequest - 39, // 52: tern.v1.Tern.Health:input_type -> tern.v1.HealthRequest - 41, // 53: tern.v1.Tern.Stop:input_type -> tern.v1.StopRequest - 43, // 54: tern.v1.Tern.Cancel:input_type -> tern.v1.CancelRequest - 45, // 55: tern.v1.Tern.Start:input_type -> tern.v1.StartRequest - 12, // 56: tern.v1.Tern.PullSchema:output_type -> tern.v1.PullSchemaResponse - 20, // 57: tern.v1.Tern.Plan:output_type -> tern.v1.PlanResponse - 21, // 58: tern.v1.Tern.PlanDiff:output_type -> tern.v1.PlanDiffResponse - 24, // 59: tern.v1.Tern.Apply:output_type -> tern.v1.ApplyResponse - 32, // 60: tern.v1.Tern.Progress:output_type -> tern.v1.ProgressResponse - 28, // 61: tern.v1.Tern.Logs:output_type -> tern.v1.LogsResponse - 34, // 62: tern.v1.Tern.Cutover:output_type -> tern.v1.CutoverResponse - 36, // 63: tern.v1.Tern.Revert:output_type -> tern.v1.RevertResponse - 38, // 64: tern.v1.Tern.SkipRevert:output_type -> tern.v1.SkipRevertResponse - 40, // 65: tern.v1.Tern.Health:output_type -> tern.v1.HealthResponse - 42, // 66: tern.v1.Tern.Stop:output_type -> tern.v1.StopResponse - 44, // 67: tern.v1.Tern.Cancel:output_type -> tern.v1.CancelResponse - 46, // 68: tern.v1.Tern.Start:output_type -> tern.v1.StartResponse - 56, // [56:69] is the sub-list for method output_type - 43, // [43:56] is the sub-list for method input_type - 43, // [43:43] is the sub-list for extension type_name - 43, // [43:43] is the sub-list for extension extendee - 0, // [0:43] is the sub-list for field type_name + 48, // 39: tern.v1.StorageSchemaReport.outstanding:type_name -> tern.v1.StorageSchemaStatement + 48, // 40: tern.v1.StorageSchemaReport.destructive:type_name -> tern.v1.StorageSchemaStatement + 48, // 41: tern.v1.StorageSchemaReport.manual:type_name -> tern.v1.StorageSchemaStatement + 49, // 42: tern.v1.StorageSchemaDiffResponse.report:type_name -> tern.v1.StorageSchemaReport + 49, // 43: tern.v1.StorageSchemaApplyResponse.planned:type_name -> tern.v1.StorageSchemaReport + 49, // 44: tern.v1.StorageSchemaApplyResponse.remaining:type_name -> tern.v1.StorageSchemaReport + 8, // 45: tern.v1.PulledNamespace.TableCatalogEntry.value:type_name -> tern.v1.TableCatalog + 6, // 46: tern.v1.PullSchemaResponse.NamespacesEntry.value:type_name -> tern.v1.PulledNamespace + 4, // 47: tern.v1.PlanRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles + 4, // 48: tern.v1.ApplyRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles + 5, // 49: tern.v1.Tern.PullSchema:input_type -> tern.v1.PullSchemaRequest + 13, // 50: tern.v1.Tern.Plan:input_type -> tern.v1.PlanRequest + 13, // 51: tern.v1.Tern.PlanDiff:input_type -> tern.v1.PlanRequest + 22, // 52: tern.v1.Tern.Apply:input_type -> tern.v1.ApplyRequest + 25, // 53: tern.v1.Tern.Progress:input_type -> tern.v1.ProgressRequest + 26, // 54: tern.v1.Tern.Logs:input_type -> tern.v1.LogsRequest + 33, // 55: tern.v1.Tern.Cutover:input_type -> tern.v1.CutoverRequest + 35, // 56: tern.v1.Tern.Revert:input_type -> tern.v1.RevertRequest + 37, // 57: tern.v1.Tern.SkipRevert:input_type -> tern.v1.SkipRevertRequest + 39, // 58: tern.v1.Tern.Health:input_type -> tern.v1.HealthRequest + 41, // 59: tern.v1.Tern.Stop:input_type -> tern.v1.StopRequest + 43, // 60: tern.v1.Tern.Cancel:input_type -> tern.v1.CancelRequest + 45, // 61: tern.v1.Tern.Start:input_type -> tern.v1.StartRequest + 47, // 62: tern.v1.Tern.StorageSchemaDiff:input_type -> tern.v1.StorageSchemaDiffRequest + 51, // 63: tern.v1.Tern.StorageSchemaApply:input_type -> tern.v1.StorageSchemaApplyRequest + 12, // 64: tern.v1.Tern.PullSchema:output_type -> tern.v1.PullSchemaResponse + 20, // 65: tern.v1.Tern.Plan:output_type -> tern.v1.PlanResponse + 21, // 66: tern.v1.Tern.PlanDiff:output_type -> tern.v1.PlanDiffResponse + 24, // 67: tern.v1.Tern.Apply:output_type -> tern.v1.ApplyResponse + 32, // 68: tern.v1.Tern.Progress:output_type -> tern.v1.ProgressResponse + 28, // 69: tern.v1.Tern.Logs:output_type -> tern.v1.LogsResponse + 34, // 70: tern.v1.Tern.Cutover:output_type -> tern.v1.CutoverResponse + 36, // 71: tern.v1.Tern.Revert:output_type -> tern.v1.RevertResponse + 38, // 72: tern.v1.Tern.SkipRevert:output_type -> tern.v1.SkipRevertResponse + 40, // 73: tern.v1.Tern.Health:output_type -> tern.v1.HealthResponse + 42, // 74: tern.v1.Tern.Stop:output_type -> tern.v1.StopResponse + 44, // 75: tern.v1.Tern.Cancel:output_type -> tern.v1.CancelResponse + 46, // 76: tern.v1.Tern.Start:output_type -> tern.v1.StartResponse + 50, // 77: tern.v1.Tern.StorageSchemaDiff:output_type -> tern.v1.StorageSchemaDiffResponse + 52, // 78: tern.v1.Tern.StorageSchemaApply:output_type -> tern.v1.StorageSchemaApplyResponse + 64, // [64:79] is the sub-list for method output_type + 49, // [49:64] is the sub-list for method input_type + 49, // [49:49] is the sub-list for extension type_name + 49, // [49:49] is the sub-list for extension extendee + 0, // [0:49] is the sub-list for field type_name } func init() { file_tern_proto_init() } @@ -4494,7 +4938,7 @@ func file_tern_proto_init() { GoPackagePath: reflect.TypeOf(x{}).PkgPath(), RawDescriptor: unsafe.Slice(unsafe.StringData(file_tern_proto_rawDesc), len(file_tern_proto_rawDesc)), NumEnums: 4, - NumMessages: 55, + NumMessages: 61, NumExtensions: 0, NumServices: 1, }, diff --git a/pkg/proto/ternv1/tern.pb.gw.go b/pkg/proto/ternv1/tern.pb.gw.go index eed2028b1..2f14a24db 100644 --- a/pkg/proto/ternv1/tern.pb.gw.go +++ b/pkg/proto/ternv1/tern.pb.gw.go @@ -380,6 +380,60 @@ func local_request_Tern_Start_0(ctx context.Context, marshaler runtime.Marshaler return msg, metadata, err } +func request_Tern_StorageSchemaDiff_0(ctx context.Context, marshaler runtime.Marshaler, client TernClient, req *http.Request, pathParams map[string]string) (proto.Message, runtime.ServerMetadata, error) { + var ( + protoReq StorageSchemaDiffRequest + metadata runtime.ServerMetadata + ) + if err := marshaler.NewDecoder(req.Body).Decode(&protoReq); err != nil && !errors.Is(err, io.EOF) { + return nil, metadata, status.Errorf(codes.InvalidArgument, "%v", err) + } + if req.Body != nil { + _, _ = io.Copy(io.Discard, req.Body) + } + msg, err := client.StorageSchemaDiff(ctx, &protoReq, grpc.Header(&metadata.HeaderMD), grpc.Trailer(&metadata.TrailerMD)) + return msg, metadata, err +} + +func local_request_Tern_StorageSchemaDiff_0(ctx context.Context, marshaler runtime.Marshaler, server TernServer, req *http.Request, pathParams map[string]string) (proto.Message, runtime.ServerMetadata, error) { + var ( + protoReq StorageSchemaDiffRequest + metadata runtime.ServerMetadata + ) + if err := marshaler.NewDecoder(req.Body).Decode(&protoReq); err != nil && !errors.Is(err, io.EOF) { + return nil, metadata, status.Errorf(codes.InvalidArgument, "%v", err) + } + msg, err := server.StorageSchemaDiff(ctx, &protoReq) + return msg, metadata, err +} + +func request_Tern_StorageSchemaApply_0(ctx context.Context, marshaler runtime.Marshaler, client TernClient, req *http.Request, pathParams map[string]string) (proto.Message, runtime.ServerMetadata, error) { + var ( + protoReq StorageSchemaApplyRequest + metadata runtime.ServerMetadata + ) + if err := marshaler.NewDecoder(req.Body).Decode(&protoReq); err != nil && !errors.Is(err, io.EOF) { + return nil, metadata, status.Errorf(codes.InvalidArgument, "%v", err) + } + if req.Body != nil { + _, _ = io.Copy(io.Discard, req.Body) + } + msg, err := client.StorageSchemaApply(ctx, &protoReq, grpc.Header(&metadata.HeaderMD), grpc.Trailer(&metadata.TrailerMD)) + return msg, metadata, err +} + +func local_request_Tern_StorageSchemaApply_0(ctx context.Context, marshaler runtime.Marshaler, server TernServer, req *http.Request, pathParams map[string]string) (proto.Message, runtime.ServerMetadata, error) { + var ( + protoReq StorageSchemaApplyRequest + metadata runtime.ServerMetadata + ) + if err := marshaler.NewDecoder(req.Body).Decode(&protoReq); err != nil && !errors.Is(err, io.EOF) { + return nil, metadata, status.Errorf(codes.InvalidArgument, "%v", err) + } + msg, err := server.StorageSchemaApply(ctx, &protoReq) + return msg, metadata, err +} + // RegisterTernHandlerServer registers the http handlers for service Tern to "mux". // UnaryRPC :call TernServer directly. // StreamingRPC :currently unsupported pending https://github.com/grpc/grpc-go/issues/906. @@ -646,6 +700,46 @@ func RegisterTernHandlerServer(ctx context.Context, mux *runtime.ServeMux, serve } forward_Tern_Start_0(annotatedContext, mux, outboundMarshaler, w, req, resp, mux.GetForwardResponseOptions()...) }) + mux.Handle(http.MethodPost, pattern_Tern_StorageSchemaDiff_0, func(w http.ResponseWriter, req *http.Request, pathParams map[string]string) { + ctx, cancel := context.WithCancel(req.Context()) + defer cancel() + var stream runtime.ServerTransportStream + ctx = grpc.NewContextWithServerTransportStream(ctx, &stream) + inboundMarshaler, outboundMarshaler := runtime.MarshalerForRequest(mux, req) + annotatedContext, err := runtime.AnnotateIncomingContext(ctx, mux, req, "/tern.v1.Tern/StorageSchemaDiff", runtime.WithHTTPPathPattern("/v1/storage-schema/diff")) + if err != nil { + runtime.HTTPError(ctx, mux, outboundMarshaler, w, req, err) + return + } + resp, md, err := local_request_Tern_StorageSchemaDiff_0(annotatedContext, inboundMarshaler, server, req, pathParams) + md.HeaderMD, md.TrailerMD = metadata.Join(md.HeaderMD, stream.Header()), metadata.Join(md.TrailerMD, stream.Trailer()) + annotatedContext = runtime.NewServerMetadataContext(annotatedContext, md) + if err != nil { + runtime.HTTPError(annotatedContext, mux, outboundMarshaler, w, req, err) + return + } + forward_Tern_StorageSchemaDiff_0(annotatedContext, mux, outboundMarshaler, w, req, resp, mux.GetForwardResponseOptions()...) + }) + mux.Handle(http.MethodPost, pattern_Tern_StorageSchemaApply_0, func(w http.ResponseWriter, req *http.Request, pathParams map[string]string) { + ctx, cancel := context.WithCancel(req.Context()) + defer cancel() + var stream runtime.ServerTransportStream + ctx = grpc.NewContextWithServerTransportStream(ctx, &stream) + inboundMarshaler, outboundMarshaler := runtime.MarshalerForRequest(mux, req) + annotatedContext, err := runtime.AnnotateIncomingContext(ctx, mux, req, "/tern.v1.Tern/StorageSchemaApply", runtime.WithHTTPPathPattern("/v1/storage-schema/apply")) + if err != nil { + runtime.HTTPError(ctx, mux, outboundMarshaler, w, req, err) + return + } + resp, md, err := local_request_Tern_StorageSchemaApply_0(annotatedContext, inboundMarshaler, server, req, pathParams) + md.HeaderMD, md.TrailerMD = metadata.Join(md.HeaderMD, stream.Header()), metadata.Join(md.TrailerMD, stream.Trailer()) + annotatedContext = runtime.NewServerMetadataContext(annotatedContext, md) + if err != nil { + runtime.HTTPError(annotatedContext, mux, outboundMarshaler, w, req, err) + return + } + forward_Tern_StorageSchemaApply_0(annotatedContext, mux, outboundMarshaler, w, req, resp, mux.GetForwardResponseOptions()...) + }) return nil } @@ -907,37 +1001,75 @@ func RegisterTernHandlerClient(ctx context.Context, mux *runtime.ServeMux, clien } forward_Tern_Start_0(annotatedContext, mux, outboundMarshaler, w, req, resp, mux.GetForwardResponseOptions()...) }) + mux.Handle(http.MethodPost, pattern_Tern_StorageSchemaDiff_0, func(w http.ResponseWriter, req *http.Request, pathParams map[string]string) { + ctx, cancel := context.WithCancel(req.Context()) + defer cancel() + inboundMarshaler, outboundMarshaler := runtime.MarshalerForRequest(mux, req) + annotatedContext, err := runtime.AnnotateContext(ctx, mux, req, "/tern.v1.Tern/StorageSchemaDiff", runtime.WithHTTPPathPattern("/v1/storage-schema/diff")) + if err != nil { + runtime.HTTPError(ctx, mux, outboundMarshaler, w, req, err) + return + } + resp, md, err := request_Tern_StorageSchemaDiff_0(annotatedContext, inboundMarshaler, client, req, pathParams) + annotatedContext = runtime.NewServerMetadataContext(annotatedContext, md) + if err != nil { + runtime.HTTPError(annotatedContext, mux, outboundMarshaler, w, req, err) + return + } + forward_Tern_StorageSchemaDiff_0(annotatedContext, mux, outboundMarshaler, w, req, resp, mux.GetForwardResponseOptions()...) + }) + mux.Handle(http.MethodPost, pattern_Tern_StorageSchemaApply_0, func(w http.ResponseWriter, req *http.Request, pathParams map[string]string) { + ctx, cancel := context.WithCancel(req.Context()) + defer cancel() + inboundMarshaler, outboundMarshaler := runtime.MarshalerForRequest(mux, req) + annotatedContext, err := runtime.AnnotateContext(ctx, mux, req, "/tern.v1.Tern/StorageSchemaApply", runtime.WithHTTPPathPattern("/v1/storage-schema/apply")) + if err != nil { + runtime.HTTPError(ctx, mux, outboundMarshaler, w, req, err) + return + } + resp, md, err := request_Tern_StorageSchemaApply_0(annotatedContext, inboundMarshaler, client, req, pathParams) + annotatedContext = runtime.NewServerMetadataContext(annotatedContext, md) + if err != nil { + runtime.HTTPError(annotatedContext, mux, outboundMarshaler, w, req, err) + return + } + forward_Tern_StorageSchemaApply_0(annotatedContext, mux, outboundMarshaler, w, req, resp, mux.GetForwardResponseOptions()...) + }) return nil } var ( - pattern_Tern_PullSchema_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "pull-schema"}, "")) - pattern_Tern_Plan_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "plan"}, "")) - pattern_Tern_PlanDiff_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "plan-diff"}, "")) - pattern_Tern_Apply_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "apply"}, "")) - pattern_Tern_Progress_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "progress"}, "")) - pattern_Tern_Logs_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "logs"}, "")) - pattern_Tern_Cutover_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "cutover"}, "")) - pattern_Tern_Revert_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "revert"}, "")) - pattern_Tern_SkipRevert_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "skip-revert"}, "")) - pattern_Tern_Health_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "health"}, "")) - pattern_Tern_Stop_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "stop"}, "")) - pattern_Tern_Cancel_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "cancel"}, "")) - pattern_Tern_Start_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "start"}, "")) + pattern_Tern_PullSchema_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "pull-schema"}, "")) + pattern_Tern_Plan_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "plan"}, "")) + pattern_Tern_PlanDiff_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "plan-diff"}, "")) + pattern_Tern_Apply_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "apply"}, "")) + pattern_Tern_Progress_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "progress"}, "")) + pattern_Tern_Logs_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "logs"}, "")) + pattern_Tern_Cutover_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "cutover"}, "")) + pattern_Tern_Revert_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "revert"}, "")) + pattern_Tern_SkipRevert_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "skip-revert"}, "")) + pattern_Tern_Health_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "health"}, "")) + pattern_Tern_Stop_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "stop"}, "")) + pattern_Tern_Cancel_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "cancel"}, "")) + pattern_Tern_Start_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1}, []string{"v1", "start"}, "")) + pattern_Tern_StorageSchemaDiff_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1, 2, 2}, []string{"v1", "storage-schema", "diff"}, "")) + pattern_Tern_StorageSchemaApply_0 = runtime.MustPattern(runtime.NewPattern(1, []int{2, 0, 2, 1, 2, 2}, []string{"v1", "storage-schema", "apply"}, "")) ) var ( - forward_Tern_PullSchema_0 = runtime.ForwardResponseMessage - forward_Tern_Plan_0 = runtime.ForwardResponseMessage - forward_Tern_PlanDiff_0 = runtime.ForwardResponseMessage - forward_Tern_Apply_0 = runtime.ForwardResponseMessage - forward_Tern_Progress_0 = runtime.ForwardResponseMessage - forward_Tern_Logs_0 = runtime.ForwardResponseMessage - forward_Tern_Cutover_0 = runtime.ForwardResponseMessage - forward_Tern_Revert_0 = runtime.ForwardResponseMessage - forward_Tern_SkipRevert_0 = runtime.ForwardResponseMessage - forward_Tern_Health_0 = runtime.ForwardResponseMessage - forward_Tern_Stop_0 = runtime.ForwardResponseMessage - forward_Tern_Cancel_0 = runtime.ForwardResponseMessage - forward_Tern_Start_0 = runtime.ForwardResponseMessage + forward_Tern_PullSchema_0 = runtime.ForwardResponseMessage + forward_Tern_Plan_0 = runtime.ForwardResponseMessage + forward_Tern_PlanDiff_0 = runtime.ForwardResponseMessage + forward_Tern_Apply_0 = runtime.ForwardResponseMessage + forward_Tern_Progress_0 = runtime.ForwardResponseMessage + forward_Tern_Logs_0 = runtime.ForwardResponseMessage + forward_Tern_Cutover_0 = runtime.ForwardResponseMessage + forward_Tern_Revert_0 = runtime.ForwardResponseMessage + forward_Tern_SkipRevert_0 = runtime.ForwardResponseMessage + forward_Tern_Health_0 = runtime.ForwardResponseMessage + forward_Tern_Stop_0 = runtime.ForwardResponseMessage + forward_Tern_Cancel_0 = runtime.ForwardResponseMessage + forward_Tern_Start_0 = runtime.ForwardResponseMessage + forward_Tern_StorageSchemaDiff_0 = runtime.ForwardResponseMessage + forward_Tern_StorageSchemaApply_0 = runtime.ForwardResponseMessage ) diff --git a/pkg/proto/ternv1/tern_grpc.pb.go b/pkg/proto/ternv1/tern_grpc.pb.go index 59cf3d755..d1a3ea939 100644 --- a/pkg/proto/ternv1/tern_grpc.pb.go +++ b/pkg/proto/ternv1/tern_grpc.pb.go @@ -19,19 +19,21 @@ import ( const _ = grpc.SupportPackageIsVersion9 const ( - Tern_PullSchema_FullMethodName = "/tern.v1.Tern/PullSchema" - Tern_Plan_FullMethodName = "/tern.v1.Tern/Plan" - Tern_PlanDiff_FullMethodName = "/tern.v1.Tern/PlanDiff" - Tern_Apply_FullMethodName = "/tern.v1.Tern/Apply" - Tern_Progress_FullMethodName = "/tern.v1.Tern/Progress" - Tern_Logs_FullMethodName = "/tern.v1.Tern/Logs" - Tern_Cutover_FullMethodName = "/tern.v1.Tern/Cutover" - Tern_Revert_FullMethodName = "/tern.v1.Tern/Revert" - Tern_SkipRevert_FullMethodName = "/tern.v1.Tern/SkipRevert" - Tern_Health_FullMethodName = "/tern.v1.Tern/Health" - Tern_Stop_FullMethodName = "/tern.v1.Tern/Stop" - Tern_Cancel_FullMethodName = "/tern.v1.Tern/Cancel" - Tern_Start_FullMethodName = "/tern.v1.Tern/Start" + Tern_PullSchema_FullMethodName = "/tern.v1.Tern/PullSchema" + Tern_Plan_FullMethodName = "/tern.v1.Tern/Plan" + Tern_PlanDiff_FullMethodName = "/tern.v1.Tern/PlanDiff" + Tern_Apply_FullMethodName = "/tern.v1.Tern/Apply" + Tern_Progress_FullMethodName = "/tern.v1.Tern/Progress" + Tern_Logs_FullMethodName = "/tern.v1.Tern/Logs" + Tern_Cutover_FullMethodName = "/tern.v1.Tern/Cutover" + Tern_Revert_FullMethodName = "/tern.v1.Tern/Revert" + Tern_SkipRevert_FullMethodName = "/tern.v1.Tern/SkipRevert" + Tern_Health_FullMethodName = "/tern.v1.Tern/Health" + Tern_Stop_FullMethodName = "/tern.v1.Tern/Stop" + Tern_Cancel_FullMethodName = "/tern.v1.Tern/Cancel" + Tern_Start_FullMethodName = "/tern.v1.Tern/Start" + Tern_StorageSchemaDiff_FullMethodName = "/tern.v1.Tern/StorageSchemaDiff" + Tern_StorageSchemaApply_FullMethodName = "/tern.v1.Tern/StorageSchemaApply" ) // TernClient is the client API for Tern service. @@ -174,6 +176,32 @@ type TernClient interface { // // Execution resumes asynchronously (same as Apply). Use Progress to track. Start(ctx context.Context, in *StartRequest, opts ...grpc.CallOption) (*StartResponse, error) + // StorageSchemaDiff reports the DDL outstanding between the embedded storage + // schema of the binary serving this RPC and the live storage database that + // binary uses. Unlike every other RPC here, it is about the serving + // instance's own bookkeeping database rather than a target database an + // operator asked to change. + // + // It is strictly read-only: it reads the live catalog and computes a diff, + // executing no DDL and taking no lock, so it is safe to call at any time, + // including while the same storage is being converged. + // + // The answer comes from the serving binary's own embedded files, which is the + // whole point of routing it here rather than computing it from a release pin + // on the caller's side: a caller's pin says which release it was built + // against, not what this storage converged to, and those diverge exactly when + // a deploy has failed to converge. + StorageSchemaDiff(ctx context.Context, in *StorageSchemaDiffRequest, opts ...grpc.CallOption) (*StorageSchemaDiffResponse, error) + // StorageSchemaApply converges the serving instance's own storage schema and + // reports what it found and what it left behind. + // + // The convergence is the serving binary's startup bootstrap, called + // unchanged: the same differ, the same destructive-statement refusal, and the + // same advisory lock that serializes it against every other instance of that + // deployment. Destructive statements are refused unless the serving + // instance's storage config allows them or allow_destructive opts in for this + // call. + StorageSchemaApply(ctx context.Context, in *StorageSchemaApplyRequest, opts ...grpc.CallOption) (*StorageSchemaApplyResponse, error) } type ternClient struct { @@ -314,6 +342,26 @@ func (c *ternClient) Start(ctx context.Context, in *StartRequest, opts ...grpc.C return out, nil } +func (c *ternClient) StorageSchemaDiff(ctx context.Context, in *StorageSchemaDiffRequest, opts ...grpc.CallOption) (*StorageSchemaDiffResponse, error) { + cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...) + out := new(StorageSchemaDiffResponse) + err := c.cc.Invoke(ctx, Tern_StorageSchemaDiff_FullMethodName, in, out, cOpts...) + if err != nil { + return nil, err + } + return out, nil +} + +func (c *ternClient) StorageSchemaApply(ctx context.Context, in *StorageSchemaApplyRequest, opts ...grpc.CallOption) (*StorageSchemaApplyResponse, error) { + cOpts := append([]grpc.CallOption{grpc.StaticMethod()}, opts...) + out := new(StorageSchemaApplyResponse) + err := c.cc.Invoke(ctx, Tern_StorageSchemaApply_FullMethodName, in, out, cOpts...) + if err != nil { + return nil, err + } + return out, nil +} + // TernServer is the server API for Tern service. // All implementations should embed UnimplementedTernServer // for forward compatibility. @@ -454,6 +502,32 @@ type TernServer interface { // // Execution resumes asynchronously (same as Apply). Use Progress to track. Start(context.Context, *StartRequest) (*StartResponse, error) + // StorageSchemaDiff reports the DDL outstanding between the embedded storage + // schema of the binary serving this RPC and the live storage database that + // binary uses. Unlike every other RPC here, it is about the serving + // instance's own bookkeeping database rather than a target database an + // operator asked to change. + // + // It is strictly read-only: it reads the live catalog and computes a diff, + // executing no DDL and taking no lock, so it is safe to call at any time, + // including while the same storage is being converged. + // + // The answer comes from the serving binary's own embedded files, which is the + // whole point of routing it here rather than computing it from a release pin + // on the caller's side: a caller's pin says which release it was built + // against, not what this storage converged to, and those diverge exactly when + // a deploy has failed to converge. + StorageSchemaDiff(context.Context, *StorageSchemaDiffRequest) (*StorageSchemaDiffResponse, error) + // StorageSchemaApply converges the serving instance's own storage schema and + // reports what it found and what it left behind. + // + // The convergence is the serving binary's startup bootstrap, called + // unchanged: the same differ, the same destructive-statement refusal, and the + // same advisory lock that serializes it against every other instance of that + // deployment. Destructive statements are refused unless the serving + // instance's storage config allows them or allow_destructive opts in for this + // call. + StorageSchemaApply(context.Context, *StorageSchemaApplyRequest) (*StorageSchemaApplyResponse, error) } // UnimplementedTernServer should be embedded to have @@ -502,6 +576,12 @@ func (UnimplementedTernServer) Cancel(context.Context, *CancelRequest) (*CancelR func (UnimplementedTernServer) Start(context.Context, *StartRequest) (*StartResponse, error) { return nil, status.Error(codes.Unimplemented, "method Start not implemented") } +func (UnimplementedTernServer) StorageSchemaDiff(context.Context, *StorageSchemaDiffRequest) (*StorageSchemaDiffResponse, error) { + return nil, status.Error(codes.Unimplemented, "method StorageSchemaDiff not implemented") +} +func (UnimplementedTernServer) StorageSchemaApply(context.Context, *StorageSchemaApplyRequest) (*StorageSchemaApplyResponse, error) { + return nil, status.Error(codes.Unimplemented, "method StorageSchemaApply not implemented") +} func (UnimplementedTernServer) testEmbeddedByValue() {} // UnsafeTernServer may be embedded to opt out of forward compatibility for this service. @@ -756,6 +836,42 @@ func _Tern_Start_Handler(srv interface{}, ctx context.Context, dec func(interfac return interceptor(ctx, in, info, handler) } +func _Tern_StorageSchemaDiff_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(StorageSchemaDiffRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(TernServer).StorageSchemaDiff(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: Tern_StorageSchemaDiff_FullMethodName, + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(TernServer).StorageSchemaDiff(ctx, req.(*StorageSchemaDiffRequest)) + } + return interceptor(ctx, in, info, handler) +} + +func _Tern_StorageSchemaApply_Handler(srv interface{}, ctx context.Context, dec func(interface{}) error, interceptor grpc.UnaryServerInterceptor) (interface{}, error) { + in := new(StorageSchemaApplyRequest) + if err := dec(in); err != nil { + return nil, err + } + if interceptor == nil { + return srv.(TernServer).StorageSchemaApply(ctx, in) + } + info := &grpc.UnaryServerInfo{ + Server: srv, + FullMethod: Tern_StorageSchemaApply_FullMethodName, + } + handler := func(ctx context.Context, req interface{}) (interface{}, error) { + return srv.(TernServer).StorageSchemaApply(ctx, req.(*StorageSchemaApplyRequest)) + } + return interceptor(ctx, in, info, handler) +} + // Tern_ServiceDesc is the grpc.ServiceDesc for Tern service. // It's only intended for direct use with grpc.RegisterService, // and not to be introspected or modified (even as a copy) @@ -815,6 +931,14 @@ var Tern_ServiceDesc = grpc.ServiceDesc{ MethodName: "Start", Handler: _Tern_Start_Handler, }, + { + MethodName: "StorageSchemaDiff", + Handler: _Tern_StorageSchemaDiff_Handler, + }, + { + MethodName: "StorageSchemaApply", + Handler: _Tern_StorageSchemaApply_Handler, + }, }, Streams: []grpc.StreamDesc{}, Metadata: "tern.proto", diff --git a/pkg/serve/serve.go b/pkg/serve/serve.go index 221eb0816..c2859ec47 100644 --- a/pkg/serve/serve.go +++ b/pkg/serve/serve.go @@ -252,6 +252,14 @@ type Server struct { telemetry *api.Telemetry authz auth.Authorizer engines map[string]tern.EngineFactory + // dialect is the storage database's family, resolved once at build time so + // every later storage operation — including the operator-facing storage + // schema surface — routes to the same family the bootstrap converged. + dialect schema.Dialect + // version is the build's SchemaBot version. Storage schema reports carry + // it for attribution: the diff itself is computed from this binary's + // embedded files, and the version only says whose files they were. + version string } // registerPlanetScaleMTLS registers the configured planetscale.mtls @@ -447,7 +455,7 @@ func Build(ctx context.Context, cfg *api.ServerConfig, opts ...Option) (*Server, } success = true - return &Server{ + srv := &Server{ cfg: cfg, svc: svc, storage: store, @@ -457,7 +465,17 @@ func Build(ctx context.Context, cfg *api.ServerConfig, opts ...Option) (*Server, telemetry: telemetry, authz: authz, engines: o.engines, - }, nil + dialect: dialect, + version: o.version, + } + + // The HTTP storage schema routes and the gRPC storage schema service answer + // from one adapter bound to the storage this server booted with, so an + // operator asking this server directly and a control plane asking it over + // gRPC read the same database with the same embedded schema files. + svc.SetStorageSchemaService(srv.storageSchemaService()) + + return srv, nil } // Storage boot retry policy. The budget is sized so that even a final attempt @@ -608,7 +626,11 @@ func (s *Server) RegisterGRPC(ctx context.Context, gs *grpc.Server) error { s.svc.SetDefaultTernClient(built) client = built } - tern.NewServer(client, s.logger).Register(gs) + // The storage-schema service answers for this instance's own storage + // database, which is the only way a control plane can read it: a data + // plane's storage is reachable from the data plane, and the gRPC endpoint + // is the connection that already exists between the two. + tern.NewServer(client, s.logger, tern.WithStorageSchemaService(s.storageSchemaService())).Register(gs) return nil } diff --git a/pkg/serve/serve_build_test.go b/pkg/serve/serve_build_test.go index 03b3b1bd9..ce57a10de 100644 --- a/pkg/serve/serve_build_test.go +++ b/pkg/serve/serve_build_test.go @@ -28,9 +28,11 @@ type stubTernClient struct{ tern.Client } // Run own the listener. func TestServerRegisterGRPCRegistersTernService(t *testing.T) { logger := slog.New(slog.DiscardHandler) + cfg := &api.ServerConfig{} srv := &Server{ + cfg: cfg, dataPlaneClient: stubTernClient{}, - svc: api.New(mysqlstore.New(nil), &api.ServerConfig{}, nil, logger), + svc: api.New(mysqlstore.New(nil), cfg, nil, logger), logger: logger, } diff --git a/pkg/serve/storage_schema.go b/pkg/serve/storage_schema.go new file mode 100644 index 000000000..518a9ba52 --- /dev/null +++ b/pkg/serve/storage_schema.go @@ -0,0 +1,136 @@ +package serve + +import ( + "context" + "fmt" + "log/slog" + "time" + + "github.com/block/schemabot/pkg/api" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/schema" +) + +// storageSchemaAdapter answers the storage-schema RPCs for the instance it was +// built on: it reads that instance's own storage database, with that +// instance's own embedded schema files, and converges it by running that +// instance's own startup bootstrap. +// +// The bindings are what make the answer trustworthy, so they are fixed at +// construction and nothing on the wire can move them. There is no target in +// the request, so no caller — not even the control plane — can point this at a +// different database. The one thing a caller may influence is whether +// destructive statements run, and that only ever widens what the local config +// already allows (see effectiveAllowDestructive). +type storageSchemaAdapter struct { + // resolveDSN re-resolves the storage DSN per call rather than capturing a + // string, so a credential rotated since startup is picked up the same way + // the storage pool picks it up. + resolveDSN func() (string, error) + dialect schema.Dialect + version string + // configAllowsDestructive is the deployment's standing storage policy + // (storage.allow_destructive_schema_changes). A boot converges under it, so + // an operator convergence must too — otherwise "apply is what a boot does" + // would stop being true on exactly the deployments that opted in. + configAllowsDestructive bool + // postgresStatementTimeout is the budget the bootstrap runs its catalog + // reads under, so a diff cannot succeed under a budget the convergence + // would then fail on, or the reverse. It carries the PostgreSQL + // statement-budget convention: zero disables the budget explicitly, + // negative leaves the platform's ambient value alone. + postgresStatementTimeout time.Duration + logger *slog.Logger +} + +// storageSchemaService builds the adapter for this server's own storage. It is +// registered on the gRPC endpoint so a control plane can reach the storage of +// a data plane an operator's workstation cannot dial directly. +func (s *Server) storageSchemaService() *storageSchemaAdapter { + return &storageSchemaAdapter{ + resolveDSN: s.cfg.StorageDSN, + dialect: s.dialect, + version: s.version, + configAllowsDestructive: s.cfg.Storage.AllowDestructiveSchemaChanges, + postgresStatementTimeout: s.cfg.Postgres.StatementTimeoutOrDefault(), + logger: s.logger, + } +} + +func (a *storageSchemaAdapter) StorageSchemaDiff(ctx context.Context, req *ternv1.StorageSchemaDiffRequest) (*ternv1.StorageSchemaDiffResponse, error) { + dsn, opts, err := a.target(req.GetAllowDestructive()) + if err != nil { + return nil, err + } + diffCtx, cancel := context.WithTimeout(ctx, api.StorageSchemaDiffTimeout) + defer cancel() + + report, err := api.DiffStorageSchema(diffCtx, dsn, a.logger, opts...) + if err != nil { + return nil, fmt.Errorf("diff storage schema (dialect %s): %w", a.dialect, err) + } + report.Version = a.version + return &ternv1.StorageSchemaDiffResponse{Report: api.StorageSchemaReportProto(report)}, nil +} + +func (a *storageSchemaAdapter) StorageSchemaApply(ctx context.Context, req *ternv1.StorageSchemaApplyRequest) (*ternv1.StorageSchemaApplyResponse, error) { + dsn, opts, err := a.target(req.GetAllowDestructive()) + if err != nil { + return nil, err + } + // No timeout of its own: the convergence is the startup bootstrap, which + // bounds itself with EnsureSchemaTimeout, and wrapping a shorter deadline + // around it would cancel a legitimate online DDL mid-copy. The caller's + // context still cuts it short if the control plane hangs up. + planned, remaining, err := api.ApplyStorageSchema(ctx, dsn, a.logger, opts...) + if err != nil { + return nil, fmt.Errorf("converge storage schema (dialect %s): %w", a.dialect, err) + } + planned.Version = a.version + remaining.Version = a.version + a.logger.InfoContext(ctx, "storage schema convergence answered", + "dialect", a.dialect, + "database", remaining.Database, + "caller", req.GetCaller(), + "planned_count", len(planned.Outstanding), + "remaining_count", len(remaining.Outstanding), + "refused_count", len(remaining.Destructive), + "manual_count", len(remaining.Manual), + ) + return &ternv1.StorageSchemaApplyResponse{ + Planned: api.StorageSchemaReportProto(planned), + Remaining: api.StorageSchemaReportProto(remaining), + }, nil +} + +// target resolves the storage DSN and the bootstrap options for one call. +func (a *storageSchemaAdapter) target(requestAllowsDestructive bool) (string, []api.EnsureSchemaOption, error) { + dsn, err := a.resolveDSN() + if err != nil { + return "", nil, fmt.Errorf("resolve storage DSN for dialect %s: %w", a.dialect, err) + } + if dsn == "" { + return "", nil, fmt.Errorf("storage DSN not configured for dialect %s; the storage schema surface has no database to read", a.dialect) + } + return dsn, []api.EnsureSchemaOption{ + api.WithDialect(a.dialect), + api.WithAllowDestructiveSchemaChanges(a.effectiveAllowDestructive(requestAllowsDestructive)), + api.WithPostgresStatementTimeout(a.postgresStatementTimeout), + }, nil +} + +// effectiveAllowDestructive is the local policy widened by an explicit +// per-request opt-in, never narrowed by its absence. +// +// The two directions are not symmetric and the asymmetry is the point. A +// deployment that configured allow_destructive_schema_changes has already made +// the decision for every boot; a convergence that ignored it would run less +// than the next boot runs, so "apply is what a boot does" — the property that +// makes this usable as a pre-deploy convergence step — would quietly stop +// holding there. In the other direction, a request opting in is exactly the +// explicit operator consent AV-9 asks for before surplus storage state is +// destroyed, arriving through a command an admin had to issue rather than +// through a config file nobody re-read. +func (a *storageSchemaAdapter) effectiveAllowDestructive(requestAllowsDestructive bool) bool { + return a.configAllowsDestructive || requestAllowsDestructive +} diff --git a/pkg/serve/storage_schema_test.go b/pkg/serve/storage_schema_test.go new file mode 100644 index 000000000..6287d957b --- /dev/null +++ b/pkg/serve/storage_schema_test.go @@ -0,0 +1,119 @@ +package serve + +import ( + "fmt" + "log/slog" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/schema" +) + +// A request may widen the deployment's destructive-statement policy and can +// never narrow it. +// +// The asymmetry is the point. A deployment that configured +// allow_destructive_schema_changes has already decided for every boot, so a +// convergence that ignored it would run less than the next boot runs — and +// "apply is what a boot does", the property that makes this usable as a +// pre-deploy step, would quietly stop holding there. In the other direction, a +// request opting in is the explicit operator consent required before surplus +// storage state is destroyed. +func TestStorageSchemaAdapter_EffectiveAllowDestructive(t *testing.T) { + tests := []struct { + name string + configAllows bool + requestAllows bool + want bool + }{ + {"neither", false, false, false}, + {"request opts in", false, true, true}, + {"config already opted in", true, false, true}, + {"both", true, true, true}, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + adapter := &storageSchemaAdapter{configAllowsDestructive: tc.configAllows} + assert.Equal(t, tc.want, adapter.effectiveAllowDestructive(tc.requestAllows)) + }) + } +} + +// The adapter's storage is fixed at construction and nothing on the wire moves +// it: a request carries no target, so the DSN comes from the server's own +// config on every call. +func TestStorageSchemaAdapter_TargetResolvesTheServersOwnDSN(t *testing.T) { + const dsn = "root@tcp(127.0.0.1:3306)/schemabot" + adapter := &storageSchemaAdapter{ + resolveDSN: func() (string, error) { return dsn, nil }, + dialect: schema.DialectPostgres, + logger: slog.New(slog.DiscardHandler), + } + + resolved, opts, err := adapter.target(false) + require.NoError(t, err) + assert.Equal(t, dsn, resolved) + assert.Len(t, opts, 3, "dialect, destructive policy and the PostgreSQL statement budget") +} + +// The DSN is re-resolved per call rather than captured once, so a credential +// rotated since startup is picked up the same way the storage pool picks it up. +func TestStorageSchemaAdapter_TargetRereadsTheDSN(t *testing.T) { + calls := 0 + adapter := &storageSchemaAdapter{ + resolveDSN: func() (string, error) { + calls++ + return fmt.Sprintf("root:rotated%d@tcp(127.0.0.1:3306)/schemabot", calls), nil + }, + dialect: schema.DialectMySQL, + logger: slog.New(slog.DiscardHandler), + } + + first, _, err := adapter.target(false) + require.NoError(t, err) + second, _, err := adapter.target(false) + require.NoError(t, err) + assert.NotEqual(t, first, second, "each call resolves the DSN again") + assert.Equal(t, 2, calls) +} + +// A server with no storage DSN configured refuses to answer rather than +// reporting on a database it guessed at, and says which dialect it was trying +// to read. +func TestStorageSchemaAdapter_RefusesWithoutStorageDSN(t *testing.T) { + adapter := &storageSchemaAdapter{ + resolveDSN: func() (string, error) { return "", nil }, + dialect: schema.DialectMySQL, + logger: slog.New(slog.DiscardHandler), + } + + _, _, err := adapter.target(false) + require.Error(t, err) + assert.Contains(t, err.Error(), "storage DSN not configured") + assert.Contains(t, err.Error(), string(schema.DialectMySQL)) + + _, err = adapter.StorageSchemaDiff(t.Context(), &ternv1.StorageSchemaDiffRequest{}) + require.Error(t, err, "a diff must not proceed without a database to read") + + _, err = adapter.StorageSchemaApply(t.Context(), &ternv1.StorageSchemaApplyRequest{}) + require.Error(t, err, "a convergence must not proceed without a database to converge") +} + +// A DSN the server cannot resolve — an unreadable credential file, say — +// surfaces as an error naming what was being resolved, not as a report about +// an empty database. +func TestStorageSchemaAdapter_SurfacesDSNResolutionFailure(t *testing.T) { + adapter := &storageSchemaAdapter{ + resolveDSN: func() (string, error) { return "", assert.AnError }, + dialect: schema.DialectPostgres, + logger: slog.New(slog.DiscardHandler), + } + + _, _, err := adapter.target(false) + require.Error(t, err) + assert.Contains(t, err.Error(), "resolve storage DSN") + assert.ErrorIs(t, err, assert.AnError) +} diff --git a/pkg/tern/grpc_client.go b/pkg/tern/grpc_client.go index c205a8351..0360b9f48 100644 --- a/pkg/tern/grpc_client.go +++ b/pkg/tern/grpc_client.go @@ -576,6 +576,34 @@ func (c *GRPCClient) Cutover(ctx context.Context, req *ternv1.CutoverRequest) (* return c.client.Cutover(ctx, req) } +// StorageSchemaDiff forwards the storage-schema diff to the data plane this +// client dials, so the answer is computed by that deployment's own binary +// against that deployment's own storage database. +// +// A data plane running a release from before the RPC existed answers +// Unimplemented. Naming the upgrade is the whole response an operator needs: +// there is no second way to read that storage from here, and silently +// answering from the control plane's storage instead would report the wrong +// database as if it were the right one. +func (c *GRPCClient) StorageSchemaDiff(ctx context.Context, req *ternv1.StorageSchemaDiffRequest) (*ternv1.StorageSchemaDiffResponse, error) { + resp, err := c.client.StorageSchemaDiff(ctx, req) + if status.Code(err) == codes.Unimplemented { + return nil, fmt.Errorf("selected data plane does not support storage schema reads; upgrade that data plane: %w", err) + } + return resp, err +} + +// StorageSchemaApply forwards the convergence to the data plane this client +// dials, which runs its own startup bootstrap against its own storage under +// its own advisory lock. See StorageSchemaDiff for the Unimplemented case. +func (c *GRPCClient) StorageSchemaApply(ctx context.Context, req *ternv1.StorageSchemaApplyRequest) (*ternv1.StorageSchemaApplyResponse, error) { + resp, err := c.client.StorageSchemaApply(ctx, req) + if status.Code(err) == codes.Unimplemented { + return nil, fmt.Errorf("selected data plane does not support storage schema convergence; upgrade that data plane: %w", err) + } + return resp, err +} + func (c *GRPCClient) processPendingCutoverControlRequest(ctx context.Context, apply *storage.Apply, scope applyTaskScope) error { controlReq, err := pendingControlRequest(ctx, c.storage, apply, storage.ControlOperationCutover) if err != nil { diff --git a/pkg/tern/server.go b/pkg/tern/server.go index 9f1865a8a..916a1d8b0 100644 --- a/pkg/tern/server.go +++ b/pkg/tern/server.go @@ -19,22 +19,85 @@ import ( type Server struct { client Client logger *slog.Logger + // storageSchema answers the storage-schema RPCs for this instance's own + // storage database. It is nil unless the embedder supplies one with + // WithStorageSchemaService, in which case those RPCs are refused rather + // than answered against something else. + storageSchema StorageSchemaService } var _ ternv1.TernServer = (*Server)(nil) +// ServerOption customizes the gRPC server's capabilities. +type ServerOption func(*Server) + +// WithStorageSchemaService lets this gRPC endpoint answer for the storage +// database of the instance serving it. An embedder supplies an adapter bound +// to its own storage DSN and dialect; without one, the storage-schema RPCs +// report Unimplemented, which is what tells a caller to upgrade this data +// plane rather than leaving it to read some other database's schema. +func WithStorageSchemaService(service StorageSchemaService) ServerOption { + return func(s *Server) { s.storageSchema = service } +} + // NewServer creates a gRPC server wrapping a Client. The logger carries the // health-check causes the RPC sanitizes out of its response, so an embedder // that wires its own logger sees them with the rest of its structured output // and can attach its deployment identifiers to them. A nil logger falls back // to the default rather than dropping the only record of the cause. -func NewServer(client Client, logger *slog.Logger) *Server { +func NewServer(client Client, logger *slog.Logger, opts ...ServerOption) *Server { if logger == nil { logger = slog.Default() } - return &Server{client: client, logger: logger} + s := &Server{client: client, logger: logger} + for _, opt := range opts { + opt(s) + } + return s } +// StorageSchemaDiff reports the DDL outstanding on this instance's own storage +// database. Read-only, and correspondingly cheap to serve: no lock, no DDL. +func (s *Server) StorageSchemaDiff(ctx context.Context, req *ternv1.StorageSchemaDiffRequest) (*ternv1.StorageSchemaDiffResponse, error) { + if s.storageSchema == nil { + return nil, errStorageSchemaUnsupported + } + resp, err := s.storageSchema.StorageSchemaDiff(ctx, req) + if err != nil { + // The caller sees a sanitized message, so this log is the only place + // the cause survives — and a diff that cannot be computed is exactly + // what an operator is trying to see during a failed deploy. + s.logger.ErrorContext(ctx, "storage schema diff failed", "error", err) + return nil, status.Error(codes.Internal, "storage schema diff failed; see data plane logs") + } + return resp, nil +} + +// StorageSchemaApply converges this instance's own storage schema by running +// its startup bootstrap, under the same advisory lock a boot takes. +func (s *Server) StorageSchemaApply(ctx context.Context, req *ternv1.StorageSchemaApplyRequest) (*ternv1.StorageSchemaApplyResponse, error) { + if s.storageSchema == nil { + return nil, errStorageSchemaUnsupported + } + s.logger.InfoContext(ctx, "converging storage schema on control plane request", + "allow_destructive", req.GetAllowDestructive(), "caller", req.GetCaller()) + resp, err := s.storageSchema.StorageSchemaApply(ctx, req) + if err != nil { + s.logger.ErrorContext(ctx, "storage schema convergence failed", + "allow_destructive", req.GetAllowDestructive(), "caller", req.GetCaller(), "error", err) + return nil, status.Error(codes.Internal, "storage schema convergence failed; see data plane logs") + } + return resp, nil +} + +// errStorageSchemaUnsupported is what an endpoint without a storage-schema +// adapter answers. Unimplemented rather than Internal or FailedPrecondition: +// the caller's own Unimplemented branch names the upgrade, and an embedder +// that never wired the adapter is in exactly the same position as one running +// a release from before the RPC existed. +var errStorageSchemaUnsupported = status.Error(codes.Unimplemented, + "this deployment does not serve storage schema requests; it is running a release that predates them, or its embedder did not register the storage schema service") + // Register registers the server on the given grpc.Server. func (s *Server) Register(srv *grpc.Server) { ternv1.RegisterTernServer(srv, s) diff --git a/pkg/tern/storage_schema.go b/pkg/tern/storage_schema.go new file mode 100644 index 000000000..1414f38b8 --- /dev/null +++ b/pkg/tern/storage_schema.go @@ -0,0 +1,43 @@ +package tern + +import ( + "context" + + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" +) + +// StorageSchemaService answers for the storage schema of one SchemaBot +// instance: which storage DDL is outstanding on the database that instance +// keeps its bookkeeping in, and converging it. +// +// One interface serves both ends of the wire, because both ends do the same +// thing to the same question: +// +// serving side an adapter the embedder supplies, bound to its own storage +// DSN and dialect, which reads that database and converges it +// caller side *GRPCClient, which forwards the call to the data plane whose +// endpoint it dials +// +// The distinction that matters is that it is never about a target database. A +// storage schema belongs to the instance that runs on it, so there is no +// database, environment, or route to name in the request: an instance answers +// for its own storage, and no field a caller can set points it elsewhere. +// That is what makes the answer trustworthy — the diff comes from the embedded +// schema files of the binary that read the live catalog, in the same call, so +// it cannot be a claim about a release pin somewhere else. +type StorageSchemaService interface { + StorageSchemaDiff(ctx context.Context, req *ternv1.StorageSchemaDiffRequest) (*ternv1.StorageSchemaDiffResponse, error) + StorageSchemaApply(ctx context.Context, req *ternv1.StorageSchemaApplyRequest) (*ternv1.StorageSchemaApplyResponse, error) +} + +// StorageSchemaService is deliberately not part of Client. Client is the +// schema change surface — plan, apply, and control a change on a *target* +// database — and every implementation of it must be able to serve all of it. +// A storage schema is a property of a deployment's own installation, so a +// Client that happens to run in the caller's process has nothing to forward +// and nothing separate to report. Keeping the two apart lets a caller ask +// "can this route reach a data plane's storage?" as a type assertion instead +// of by testing a method for a not-implemented error. +var ( + _ StorageSchemaService = (*GRPCClient)(nil) +) diff --git a/pkg/testutil/postgres.go b/pkg/testutil/postgres.go index dedf5175e..f274d7f74 100644 --- a/pkg/testutil/postgres.go +++ b/pkg/testutil/postgres.go @@ -57,3 +57,18 @@ func PostgresTableExists(t *testing.T, db *sql.DB, schemaName, tableName string) require.NoError(t, err) return count > 0 } + +// PostgresColumnExists reports whether columnName exists on schemaName.tableName +// on a PostgreSQL connection. It mirrors ColumnExists, whose `?` placeholders +// only bind on MySQL. +func PostgresColumnExists(t *testing.T, db *sql.DB, schemaName, tableName, columnName string) bool { + t.Helper() + + var count int + err := db.QueryRowContext(t.Context(), + "SELECT COUNT(*) FROM information_schema.columns WHERE table_schema = $1 AND table_name = $2 AND column_name = $3", + schemaName, tableName, columnName, + ).Scan(&count) + require.NoError(t, err) + return count > 0 +} From e6504e14806c1ed1100143f2be3a62d882e0898d Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Fri, 11 Sep 2026 13:27:22 -0400 Subject: [PATCH 2/5] docs(storage): say which binary's schema a storage diff answers from The commands take no schema directory and no flag to override one: the files are embedded in whichever binary answers the request. That is the thing to keep straight when deploying, because through the API the answer describes the release currently running, which before a roll is the old one. Documents the deploy sequence around a storage schema change, and the asymmetry an operator pre-creating an index has to plan around: a table or column created ahead of the roll survives a boot of the earlier release because dropping one is refused as destructive, while an index does not, since dropping an index destroys no data and falls outside that refusal. A storage schema integration test pins it. Co-Authored-By: Claude Opus 5 --- docs/configuration.md | 79 ++++++++++++++++++++++ docs/release.md | 14 ++++ pkg/api/storage_schema_integration_test.go | 40 +++++++++++ pkg/testutil/container.go | 14 ++++ 4 files changed, 147 insertions(+) diff --git a/docs/configuration.md b/docs/configuration.md index 93c55570a..ccd1d5946 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1142,6 +1142,85 @@ Both routes are admin-only and both sit at the write tier, the read-only diff included, because the diff exposes the internal shape of SchemaBot's bookkeeping database. See [Authentication and authorization](auth.md#what-read-and-write-access-include). +### Which binary's schema you are asking about + +There is no schema directory to point these commands at, and no flag to +override one. The schema files are compiled into the binary (`go:embed` over +`pkg/schema/mysql/` and `pkg/schema/postgres/`), and the diff always uses the +files of the binary that *answers* the request: + +| Path | Whose embedded schema | +|---|---| +| through the API, no flags | the server the CLI is pointed at | +| `--deployment -e ` | that data plane | +| `--dsn` / `--config` | the CLI binary you are running | + +That is deliberate, and it is the one thing to keep straight when deploying. +Through the API, the answer describes the release that is **currently running**. +Before a roll that is the old release, so the API can report convergence while +the release you are about to deploy still has work to do. To ask what the next +release will run, run the command from a binary of that release: the direct +path with the new CLI, or a one-shot job on the new image. + +### Deploying a release that changes the storage schema + +Every startup converges the storage schema on its own, so the routine case +needs none of this. Reach for the commands when the release notes name a +storage schema change, when the tables involved carry a long history, or when a +pod is not starting. + +1. **Before the roll, ask the new release what it will run.** Use a binary of + the release being deployed, not the running one. + + ```bash + schemabot storage diff --dsn "$STORAGE_DSN" # run from the new release's binary + ``` + + Exit status 0 means its boot has nothing to do and the rest of this does not + apply. + +2. **Decide whether the boot should do it.** Additive DDL inside the + five-minute startup budget is fine when the tables are small. It is not fine + when they are not: on MySQL an index added to an existing storage table runs + as Spirit online DDL, a table copy whose cost grows with row count, and every + pod in the roll pays it. Converge once, ahead of the roll, instead: + + ```bash + schemabot storage apply --dsn "$STORAGE_DSN" # from the new release's binary + ``` + +3. **If you converged ahead of the roll, re-check right before it.** A table or + a column you created early survives a boot of the current release, because + dropping one is destructive and is refused. **An index does not.** Dropping + an index destroys no data, so it falls outside that refusal, and any boot of + the still-running older release converges the new index away without + comment — a pod restart, a scale-up, a health-check replacement. Re-run the + diff from the new release's binary immediately before rolling, and treat a + long gap between pre-creating an index and deploying as a gap the index + probably did not survive. + +4. **After the roll, confirm through the API, per deployment.** + + ```bash + schemabot storage diff # this server's storage + schemabot storage diff --deployment west -e production # a data plane's storage + ``` + + Now the running binary is the new release, so exit status 0 is the + confirmation that its storage converged. + +5. **If a pod is crashlooping, ask directly.** A failed storage bootstrap keeps + the server from accepting traffic at all, so the API cannot answer for it. + The direct path can, with the same binary the pod runs, and the statements it + prints are the ones the pod is failing on. `storage apply` from there clears + it under the same advisory lock the pods are contending for. + +During a rollback window the diff reports the newer release's tables and columns +as refused destructive statements and exits 2. That is the expected steady state +rather than drift: the surplus state is deliberate, and it is what lets the +release be rolled forward again. A pre-deploy gate keyed on exit status 0 will +flag it, which is the correct signal to pause on. + ## Support Channel SchemaBot can add an opt-in support link to GitHub PR comments so authors know diff --git a/docs/release.md b/docs/release.md index 0ea0f4c3b..ae75f9ce6 100644 --- a/docs/release.md +++ b/docs/release.md @@ -271,6 +271,20 @@ statement (see [configuration.md](./configuration.md)). A destructive change is a coordinated operation and belongs in the release notes with instructions, not in a routine patch release. +The diff does not have to be read out of the files by hand. A binary of the +release being deployed answers it against the live database: + +```bash +schemabot storage diff --dsn "$STORAGE_DSN" # run from the new release's binary +``` + +Run it from the *new* release's binary, not the running one: through the API the +answer describes whatever release is currently serving, which before a roll is +the old one. Operators pre-creating an index ahead of the roll should also read +[Deploying a release that changes the storage schema](./configuration.md#deploying-a-release-that-changes-the-storage-schema) +— a pre-created index is removed again by any boot of the still-running earlier +release, because dropping an index is not destructive and so is not refused. + ### 3. The public Go API SchemaBot is importable as a Go library, not only runnable as a binary, so its diff --git a/pkg/api/storage_schema_integration_test.go b/pkg/api/storage_schema_integration_test.go index cc560790c..6c26f1d24 100644 --- a/pkg/api/storage_schema_integration_test.go +++ b/pkg/api/storage_schema_integration_test.go @@ -182,6 +182,46 @@ func TestDiffStorageSchemaMySQL_RefusesSurplusTable(t *testing.T) { assert.Equal(t, "newer_release_state", allowed.Destructive[0].Table) } +// A surplus index is reported as a statement that runs, not as a refused one — +// unlike a surplus table or column. Dropping an index destroys no data, so it +// is outside the destructive set that protects newer storage state from an +// older binary. +// +// The asymmetry is what an operator pre-creating an index ahead of a release +// has to plan around: the index survives only as long as no instance of the +// earlier release boots, because that boot converges it away. A diff run with +// the earlier release's binary is what says so before it happens. +func TestDiffStorageSchemaMySQL_SurplusIndexIsNotProtected(t *testing.T) { + sdb, db := openEnsureSchemaDatabase(t) + require.NoError(t, EnsureSchema(sdb.DSN, storageSchemaTestLogger())) + + // An index a later release declares, created ahead of that release the way + // an operator pre-creates one to keep the startup budget clear. + _, err := db.ExecContext(t.Context(), "CREATE INDEX `idx_applies_caller` ON `applies` (`caller`)") + require.NoError(t, err, "pre-create an index the embedded schema does not declare") + require.True(t, testutil.IndexExists(t, db, sdb.Name, "applies", "idx_applies_caller")) + + report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + + assert.False(t, report.Converged()) + assert.Empty(t, report.Destructive, "dropping an index destroys no data, so it is not refused") + assert.Empty(t, report.Manual) + require.Len(t, report.Outstanding, 1, "only the one table diverges: %v", statementTables(report.Outstanding)) + statement := report.Outstanding[0] + assert.Equal(t, "applies", statement.Table) + assert.Equal(t, "alter_table", statement.Operation) + assert.Contains(t, statement.DDL, "idx_applies_caller") + + // And a convergence removes it, which is exactly what a boot of this binary + // would do to an index a later release owns. + _, remaining, err := ApplyStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + assert.True(t, remaining.Converged()) + assert.False(t, testutil.IndexExists(t, db, sdb.Name, "applies", "idx_applies_caller"), + "a surplus index is converged away, unlike a surplus table or column") +} + // A diff is safe to run against a storage database another process is // converging: it takes no advisory lock, so it answers while the bootstrap // holds one rather than blocking behind it. During an incident that is the diff --git a/pkg/testutil/container.go b/pkg/testutil/container.go index 88b3af510..3a7d597c9 100644 --- a/pkg/testutil/container.go +++ b/pkg/testutil/container.go @@ -83,6 +83,20 @@ func ColumnExists(t *testing.T, db *sql.DB, schemaName, tableName, columnName st return count > 0 } +// IndexExists reports whether indexName exists on schemaName.tableName. +func IndexExists(t *testing.T, db *sql.DB, schemaName, tableName, indexName string) bool { + t.Helper() + + var count int + err := db.QueryRowContext(t.Context(), + `SELECT COUNT(*) FROM information_schema.statistics + WHERE table_schema = ? AND table_name = ? AND index_name = ?`, + schemaName, tableName, indexName, + ).Scan(&count) + require.NoError(t, err) + return count > 0 +} + func retryContainerOp(ctx context.Context, opName string, op func() (string, error)) (string, error) { var lastErr error delay := initialDelay From 25761bc3e9972f9856aa35927359c01bcc7726b0 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Fri, 11 Sep 2026 14:24:23 -0400 Subject: [PATCH 3/5] feat(cli): diff a storage database against the release about to roll MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A deploy asks whether the storage is ready for the release being rolled, which the running binary cannot answer from files it does not carry. The diff's desired side is now selectable: --schema-dir reads a checkout, --release fetches a tag's schema files for the dialect the live storage runs. The live side is still always a read of the database. The report names both ends — the database and host it read, and the schema it compared against — so two correct reports about the same database are tellable apart, and a caller-supplied schema is never relabelled as the answering binary's own. `storage apply` takes neither selector: a convergence runs the schema embedded in the binary running it, so it does exactly what that binary's next boot would do (AV-9). The flags are refused with the two real ways to converge a release rather than with an unknown-flag error. The diff route moves to POST so the schema files can travel on it, which also retires the write-tier GET exception in the tier classification — both storage routes now take the write tier by the default rule (AZ-2). Co-Authored-By: Claude Opus 5 --- docs/auth.md | 4 +- docs/configuration.md | 136 +++++-- docs/release.md | 8 +- pkg/api/route_authorization_sweep_test.go | 5 +- pkg/api/service.go | 8 +- pkg/api/storage_schema.go | 118 ++++-- pkg/api/storage_schema_handlers.go | 112 ++++-- pkg/api/storage_schema_handlers_test.go | 24 +- pkg/api/storage_schema_integration_test.go | 91 ++++- pkg/api/storage_schema_source.go | 174 +++++++++ pkg/api/storage_schema_source_test.go | 156 ++++++++ pkg/apitypes/storage_schema.go | 42 ++- pkg/auth/tiers.go | 30 +- pkg/auth/tiers_test.go | 24 +- pkg/cmd/client/client.go | 25 +- pkg/cmd/commands/storage_schema.go | 168 +++++++-- pkg/cmd/commands/storage_schema_source.go | 345 ++++++++++++++++++ .../commands/storage_schema_source_test.go | 239 ++++++++++++ pkg/cmd/commands/storage_schema_test.go | 84 +++-- pkg/proto/tern.proto | 30 +- pkg/proto/ternv1/tern.pb.go | 175 ++++++--- pkg/serve/storage_schema.go | 50 ++- pkg/serve/storage_schema_test.go | 41 +++ 23 files changed, 1757 insertions(+), 332 deletions(-) create mode 100644 pkg/api/storage_schema_source.go create mode 100644 pkg/api/storage_schema_source_test.go create mode 100644 pkg/cmd/commands/storage_schema_source.go create mode 100644 pkg/cmd/commands/storage_schema_source_test.go diff --git a/docs/auth.md b/docs/auth.md index 6c8b137bc..1b1d96e1a 100644 --- a/docs/auth.md +++ b/docs/auth.md @@ -650,8 +650,8 @@ to writes under `forward_auth`. In the route rules, `GET` and `HEAD` requests are reads, as is `POST /api/pull`. Other requests require write access by default. -`GET /api/storage/schema/diff` is the exception in the other direction: it -reads, but it requires write access. It reports the internal shape of +`POST /api/storage/schema/diff` reads without changing anything, and still +requires write access under that default. It reports the internal shape of SchemaBot's own bookkeeping database, and its sibling route converges that database, so both belong to the people who operate the server rather than to everyone who can see the schema changes it runs. diff --git a/docs/configuration.md b/docs/configuration.md index ccd1d5946..6cb0653b4 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -1046,16 +1046,16 @@ converges. A deploy that did not converge leaves one question open: which storage DDL is still outstanding. Two commands answer it, and both read the live storage -database. Neither takes a version, because a release tag says what that -release would converge to, not what the storage converged to, and the two -answers differ exactly when a deploy has failed. +database. A release tag never stands in for that read: what a release would +converge to and what the storage actually converged to differ exactly when a +deploy has failed. `storage diff` is read-only. It takes no lock and holds no transaction, so it is safe at any time, including against production during an incident. ```console $ schemabot storage diff -schemabot (mysql) needs 3 statements: 3 outstanding, against the schema embedded in v1.2.3. +schemabot on db-1.example (mysql) needs 3 statements: 3 outstanding, against the schema embedded in v1.2.3. Outstanding, and run automatically on the next boot or apply (3): @@ -1076,6 +1076,13 @@ of the answer: `0` when the storage needs nothing, `2` when statements are outstanding, and `1` when the read itself failed. A pre-deploy gate needs those three apart, since "converged" and "unreachable" call for opposite decisions. +The headline answers the two questions that decide what the rest of the output +means. `schemabot on db-1.example (mysql)` is the database that was read — +reported by whoever read it, so it is not re-derived from a DSN, a config file, +or a deployment name. `against the schema embedded in v1.2.3` is the schema it +was compared against; [Which schema you are asking +about](#which-schema-you-are-asking-about) is how to change that. + `storage apply` converges the database by running the same bootstrap the next boot would run: the same differ, the same refusal of destructive statements, and the same advisory lock, so two operators running it at once serialize the @@ -1084,15 +1091,15 @@ them; `--auto-approve` (`-y`) skips the prompt for scripted maintenance. ```console $ schemabot storage apply -schemabot (mysql) needs 1 statement: 1 outstanding, against the schema embedded in v1.2.3. +schemabot on db-1.example (mysql) needs 1 statement: 1 outstanding, against the schema embedded in v1.2.3. Outstanding, and run automatically on the next boot or apply (1): ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; -Run these statements against schemabot (mysql)? Only 'yes' will be accepted: yes -Ran 1 statement against schemabot (mysql). -schemabot (mysql) is converged. +Run these statements against schemabot on db-1.example (mysql)? Only 'yes' will be accepted: yes +Ran 1 statement against schemabot on db-1.example (mysql). +schemabot on db-1.example (mysql) is converged. ``` Destructive statements are refused here exactly as they are at startup, and for @@ -1103,7 +1110,7 @@ as surplus. A refusal is reported rather than silently dropped, and ```console $ schemabot storage diff -schemabot (mysql) needs 1 statement: 1 destructive, against the schema embedded in v1.2.3. +schemabot on db-1.example (mysql) needs 1 statement: 1 destructive, against the schema embedded in v1.2.3. Destructive, and refused; surplus state stays in place (1): @@ -1133,34 +1140,79 @@ embedded schema files. That is also what makes the answer trustworthy, since the binary that reports the diff is the binary whose next boot would run it. The direct path exists for when the server is down, including when it is down -because its own schema bootstrap is failing. It reads the storage with *this -CLI's* embedded schema files, so run a binary of the release you are deploying. -`--dialect` states the storage family when a DSN's form does not say; it applies -only to a direct connection. +because its own schema bootstrap is failing. `--dialect` states the storage +family when a DSN's form does not say; it applies only to a direct connection. Both routes are admin-only and both sit at the write tier, the read-only diff included, because the diff exposes the internal shape of SchemaBot's bookkeeping -database. See [Authentication and authorization](auth.md#what-read-and-write-access-include). +database. Both are `POST` requests, so they take the write tier by the default +rule rather than by an exception. See [Authentication and +authorization](auth.md#what-read-and-write-access-include). -### Which binary's schema you are asking about +### Which schema you are asking about -There is no schema directory to point these commands at, and no flag to -override one. The schema files are compiled into the binary (`go:embed` over -`pkg/schema/mysql/` and `pkg/schema/postgres/`), and the diff always uses the -files of the binary that *answers* the request: +The live side of the diff is always a read of the database. The desired side is +schema *files*, and three things can supply them: -| Path | Whose embedded schema | +| Desired schema | Where the files come from | |---|---| -| through the API, no flags | the server the CLI is pointed at | -| `--deployment -e ` | that data plane | -| `--dsn` / `--config` | the CLI binary you are running | +| no flag | the embedded files of the binary that answers the request | +| `--schema-dir ` | that directory's `.sql` files, read by the CLI | +| `--release ` | that tag's `pkg/schema//` files, fetched by the CLI | + +With no flag, the answer describes the release that is **currently running**: +the files are compiled in (`go:embed` over `pkg/schema/mysql/` and +`pkg/schema/postgres/`), so through the API it is the server or data plane that +answered, and on the direct path it is the CLI binary you are running. That is +the right default — it is what the next boot would converge — and it is the +wrong question before a roll, when the running release reports convergence while +the release about to deploy still has work to do. -That is deliberate, and it is the one thing to keep straight when deploying. -Through the API, the answer describes the release that is **currently running**. -Before a roll that is the old release, so the API can report convergence while -the release you are about to deploy still has work to do. To ask what the next -release will run, run the command from a binary of that release: the direct -path with the new CLI, or a one-shot job on the new image. +`--release` asks that question without a binary of that release: + +```console +$ schemabot storage diff --deployment west -e production --release v1.4.0 +schemabot on db-1.example (mysql), deployment west in production needs 1 statement: 1 outstanding, against the schema files of release v1.4.0 in block/schemabot. + +Outstanding, and run automatically on the next boot or apply (1): + +ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; + +These are what schemabot on db-1.example (mysql), deployment west in production needs in order to match the schema files of release v1.4.0 in block/schemabot, not what its own next boot would run. To converge them, run that release's binary against this database — its container image is that release — or let the release's first boot converge them. +``` + +Because the storage schema is declarative, one diff against the release you are +rolling to covers however many releases lie between; there is nothing to step +through. + +The report names the schema it used, always, and never relabels a schema you +supplied as the answering binary's own. That line is the difference between two +correct reports about the same database, so read it before acting on the +statements. + +Details of the two selectors: + +- **`--schema-dir `** reads `*.sql` directly from a checkout or an + extracted image layer, one file per storage table. Point it at the dialect + directory (`pkg/schema/mysql`), not at its parent. A directory with no `.sql` + files is an error naming the path — a diff against an empty schema would + report every existing table as surplus. +- **`--release `** fetches the files over the repository's contents API at + that tag. It reads the schema directory for the dialect the *live storage* + runs, which it learns by first asking the target — one extra read-only diff, + paid only by this flag. `--release-repo` points at a fork or mirror + (`block/schemabot` by default), `GITHUB_API_URL` at a different API host, and + `GITHUB_TOKEN` or `GH_TOKEN` authorizes the fetch. A repository the CLI cannot + read is an error naming the token to set and `--schema-dir` as the offline + alternative. Naming both selectors is refused rather than resolved by + precedence. + +`storage apply` has neither flag. A convergence runs the schema embedded in the +binary running it, so that it does exactly what that binary's next boot would +do — the property that makes it usable as a pre-deploy step at all, and the one +that keeps an older binary from being handed newer schema to destroy. Passing +either flag to `apply` is refused with the two real ways to converge a release: +run that release's binary, or let its first boot do it. ### Deploying a release that changes the storage schema @@ -1169,15 +1221,20 @@ needs none of this. Reach for the commands when the release notes name a storage schema change, when the tables involved carry a long history, or when a pod is not starting. -1. **Before the roll, ask the new release what it will run.** Use a binary of - the release being deployed, not the running one. +1. **Before the roll, ask the new release what it will run.** Name the release + being deployed, from whatever CLI you have to hand: ```bash - schemabot storage diff --dsn "$STORAGE_DSN" # run from the new release's binary + schemabot storage diff --deployment west -e production --release v1.4.0 ``` - Exit status 0 means its boot has nothing to do and the rest of this does not - apply. + Exit status 0 means that release's boot has nothing to do and the rest of + this does not apply. A binary of the new release answers the same question + with no flag, which is what to use where the tag cannot be fetched: + + ```bash + schemabot storage diff --dsn "$STORAGE_DSN" # run from the new release's binary + ``` 2. **Decide whether the boot should do it.** Additive DDL inside the five-minute startup budget is fine when the tables are small. It is not fine @@ -1189,15 +1246,20 @@ pod is not starting. schemabot storage apply --dsn "$STORAGE_DSN" # from the new release's binary ``` + The convergence has to come from a binary of the new release: `apply` runs the + schema embedded in whatever binary runs it, and there is no flag that points + it at a release's files. Use that release's container image as a one-shot job + if there is no binary to hand. + 3. **If you converged ahead of the roll, re-check right before it.** A table or a column you created early survives a boot of the current release, because dropping one is destructive and is refused. **An index does not.** Dropping an index destroys no data, so it falls outside that refusal, and any boot of the still-running older release converges the new index away without comment — a pod restart, a scale-up, a health-check replacement. Re-run the - diff from the new release's binary immediately before rolling, and treat a - long gap between pre-creating an index and deploying as a gap the index - probably did not survive. + step 1 diff immediately before rolling, and treat a long gap between + pre-creating an index and deploying as a gap the index probably did not + survive. 4. **After the roll, confirm through the API, per deployment.** diff --git a/docs/release.md b/docs/release.md index ae75f9ce6..5907f9549 100644 --- a/docs/release.md +++ b/docs/release.md @@ -271,14 +271,14 @@ statement (see [configuration.md](./configuration.md)). A destructive change is a coordinated operation and belongs in the release notes with instructions, not in a routine patch release. -The diff does not have to be read out of the files by hand. A binary of the -release being deployed answers it against the live database: +The diff does not have to be read out of the files by hand. Name the release +being tested and the command answers it against the live database: ```bash -schemabot storage diff --dsn "$STORAGE_DSN" # run from the new release's binary +schemabot storage diff --deployment west -e production --release v1.4.0 ``` -Run it from the *new* release's binary, not the running one: through the API the +Name the release, or run the command from a binary of it — with neither, the answer describes whatever release is currently serving, which before a roll is the old one. Operators pre-creating an index ahead of the roll should also read [Deploying a release that changes the storage schema](./configuration.md#deploying-a-release-that-changes-the-storage-schema) diff --git a/pkg/api/route_authorization_sweep_test.go b/pkg/api/route_authorization_sweep_test.go index b521f1abf..c66d067d9 100644 --- a/pkg/api/route_authorization_sweep_test.go +++ b/pkg/api/route_authorization_sweep_test.go @@ -114,9 +114,8 @@ func TestMutatingRoutesDenyScopedOperatorByDefault(t *testing.T) { // The storage schema routes take no database — they are about // SchemaBot's own bookkeeping database — so a scoped operator is denied // on the admin requirement itself rather than on a target outside their - // grant. The diff route is a GET and still appears here, because - // auth.TierForRequest admits it at the write tier. - "GET /api/storage/schema/diff": ``, + // grant. + "POST /api/storage/schema/diff": `{}`, "POST /api/storage/schema/apply": `{}`, } diff --git a/pkg/api/service.go b/pkg/api/service.go index a4b0ee151..27358213a 100644 --- a/pkg/api/service.go +++ b/pkg/api/service.go @@ -848,9 +848,11 @@ func (s *Service) apiRoutes() []apiRoute { {"GET /api/locks", s.handleLockList}, // Storage schema API (SchemaBot's own bookkeeping database). Both - // routes are admin-only and both are admitted at the write tier — - // including the read-only diff, see auth.TierForRequest. - {"GET /api/storage/schema/diff", s.handleStorageSchemaDiff}, + // routes are admin-only, and both are POSTs so both are admitted at the + // write tier by auth.TierForRequest's default rule. The diff reads and + // nothing else; it carries a body because the schema to diff against + // can come from the caller. + {"POST /api/storage/schema/diff", s.handleStorageSchemaDiff}, {"POST /api/storage/schema/apply", s.handleStorageSchemaApply}, // Settings API diff --git a/pkg/api/storage_schema.go b/pkg/api/storage_schema.go index 826c90d25..d73ee9089 100644 --- a/pkg/api/storage_schema.go +++ b/pkg/api/storage_schema.go @@ -20,19 +20,20 @@ import ( // deploy that did not converge: which storage DDL is still outstanding, right // now, on this database. // -// It answers from the live database and from the embedded schema files of the -// binary that serves the request — never from a version pin. A consumer's -// go.mod pin says which release a host binary *was built against*; it says -// nothing about what the storage it talks to has actually converged to, and the -// two diverge exactly when a deploy has failed to converge. Since that is -// precisely when someone computes this diff, the pin is the one input that -// cannot be trusted, so no input to this package is a version. +// It answers from the live database and from schema files — never from a +// version pin. A consumer's go.mod pin says which release a host binary *was +// built against*; it says nothing about what the storage it talks to has +// actually converged to, and the two diverge exactly when a deploy has failed +// to converge. Since that is precisely when someone computes this diff, the pin +// is the one input that cannot be trusted, so no input to this package is a +// version: the desired side is always files, and the report names which ones +// (see StorageSchemaSource). // // The diff is the bootstrap's own diff, not a second implementation of it. On -// MySQL that is Spirit's differ over readEmbeddedSchemaFiles; on PostgreSQL it -// is postgresSchemaDriftFor over the embedded PostgreSQL files. A statement -// this reports is a statement a boot of this binary would plan, because it came -// from the same call. +// MySQL that is Spirit's differ; on PostgreSQL it is postgresSchemaDriftFor. +// A statement this reports against this binary's own embedded schema is a +// statement a boot of this binary would plan, because it came from the same +// call. // StorageSchemaStatement is one outstanding storage-schema statement, with the // classification the bootstrap would apply to it. @@ -50,7 +51,8 @@ type StorageSchemaStatement struct { } // StorageSchemaReport is what the storage schema of one database needs in order -// to match the embedded schema of the binary that produced the report. +// to match a desired schema — the embedded schema of the binary that produced +// the report, unless SchemaSource names another. // // The three statement sets are disjoint and have different dispositions, so an // operator reading the report never has to work out which statements would @@ -65,10 +67,19 @@ type StorageSchemaReport struct { // Database is the live database the diff read, as the server reports it — // so a report cannot be misread as being about a different database. Database string - // Version is the SchemaBot version of the binary whose embedded schema - // files produced the diff. It is reported, never consumed: the diff is - // computed from the files themselves, and this only says whose files they - // were. + // Host is the database server as it names itself, so a report names the + // machine as well as the database on it. Empty when the server does not + // report one. + Host string + // SchemaSource says where the desired side of the diff came from, in the + // words StorageSchemaSource attributed it with. It is always set, because + // one live database yields different answers against different releases and + // a report that did not say which one it used could be read as either. + SchemaSource string + // Version is the SchemaBot version of the binary that produced the report. + // It is reported, never consumed: the diff is computed from schema files, + // and this says which binary read them — which is the same release that + // embedded them unless SchemaSource says otherwise. Version string // Outstanding lists the statements that converge the schema and run // automatically, in the order the convergence would run them. @@ -102,12 +113,16 @@ func (r *StorageSchemaReport) Converged() bool { // thread — for the whole bootstrap budget. const StorageSchemaDiffTimeout = 30 * time.Second -// DiffStorageSchema reports the storage DDL outstanding between the embedded -// schema files of this binary and the live storage database at dsn. It is -// strictly read-only: it opens connections, reads the catalog, and computes a -// diff. It executes no DDL, takes no advisory lock, and writes nothing, so it -// is safe to run at any time, including against a database an apply is -// converging right now. +// DiffStorageSchema reports the storage DDL outstanding between a desired +// schema and the live storage database at dsn. It is strictly read-only: it +// opens connections, reads the catalog, and computes a diff. It executes no +// DDL, takes no advisory lock, and writes nothing, so it is safe to run at any +// time, including against a database an apply is converging right now. +// +// desired is the schema to compare against; nil is the embedded schema of this +// binary, which is what a boot would converge to. A caller deploying a later +// release supplies that release's files instead (see StorageSchemaSource), and +// the report attributes the answer to whichever was used. // // The dialect selects the differ, mirroring EnsureSchema's dispatch, and fails // closed for a dialect without one rather than running another family's @@ -115,13 +130,13 @@ const StorageSchemaDiffTimeout = 30 * time.Second // report describes what a boot would decide; // WithAllowDestructiveSchemaChanges only labels the report here, since a diff // executes nothing either way. -func DiffStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, opts ...EnsureSchemaOption) (*StorageSchemaReport, error) { +func DiffStorageSchema(ctx context.Context, dsn string, desired *StorageSchemaSource, logger *slog.Logger, opts ...EnsureSchemaOption) (*StorageSchemaReport, error) { o := newEnsureSchemaOptions(opts...) switch o.dialect { case schema.DialectMySQL: - return diffMySQLStorageSchema(ctx, dsn, o) + return diffMySQLStorageSchema(ctx, dsn, desired, o) case schema.DialectPostgres: - return diffPostgresStorageSchema(ctx, dsn, o) + return diffPostgresStorageSchema(ctx, dsn, desired, o) default: return nil, fmt.Errorf("no storage schema differ for storage dialect %q (supported: %q, %q)", o.dialect, schema.DialectMySQL, schema.DialectPostgres) } @@ -137,6 +152,11 @@ func DiffStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, opt // command that converged storage differently from a boot would be a second // implementation of the one path that must not have two. // +// It takes no schema source, and the absence is the safety property rather than +// an omission: a convergence runs the schema embedded in the binary running it, +// so "apply is what a boot does" holds by construction (AV-9). A caller that +// wants a later release's schema on a database runs that release's binary. +// // Two reports bracket the run, because "what happened" and "what is left" are // different questions and an operator mid-incident needs both: // @@ -148,7 +168,7 @@ func DiffStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, opt // does not declare, which is the expected steady state during a rollback // (AV-9) rather than a failure. func ApplyStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, opts ...EnsureSchemaOption) (planned, remaining *StorageSchemaReport, err error) { - planned, err = DiffStorageSchema(ctx, dsn, logger, opts...) + planned, err = DiffStorageSchema(ctx, dsn, nil, logger, opts...) if err != nil { return nil, nil, fmt.Errorf("diff storage schema before converging it: %w", err) } @@ -181,7 +201,7 @@ func ApplyStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, op return planned, nil, fmt.Errorf("converge storage schema on database %q (%s): %w", planned.Database, planned.Dialect, err) } - remaining, err = DiffStorageSchema(ctx, dsn, logger, opts...) + remaining, err = DiffStorageSchema(ctx, dsn, nil, logger, opts...) if err != nil { // The convergence succeeded; only the confirming read failed. Report // that distinctly — an operator must not read a failed verification as @@ -197,27 +217,33 @@ func ApplyStorageSchema(ctx context.Context, dsn string, logger *slog.Logger, op return planned, remaining, nil } -// diffMySQLStorageSchema diffs the embedded MySQL schema files against the live +// diffMySQLStorageSchema diffs the desired MySQL schema files against the live // storage database with Spirit's differ — the same Plan call ensureMySQLSchema // makes, so the two cannot disagree about what a boot would run. Spirit emits // one combined ALTER per table, and partitionDestructiveChanges splits it the // way the bootstrap would, so a mixed ALTER is reported as the additive clauses // that run plus the destructive clauses that are refused, not as one statement // whose disposition an operator has to guess. -func diffMySQLStorageSchema(ctx context.Context, dsn string, o ensureSchemaOptions) (*StorageSchemaReport, error) { - report := &StorageSchemaReport{Dialect: schema.DialectMySQL, DestructiveAllowed: o.allowDestructive} +func diffMySQLStorageSchema(ctx context.Context, dsn string, desired *StorageSchemaSource, o ensureSchemaOptions) (*StorageSchemaReport, error) { + report := &StorageSchemaReport{ + Dialect: schema.DialectMySQL, + SchemaSource: desired.Describe(), + DestructiveAllowed: o.allowDestructive, + } - // The database name is what makes the report readable as being about one - // database. Unlike the bootstrap's preamble, a failure here is fatal: the - // bootstrap can converge without knowing the name, but a report that cannot - // say which database it read is one an operator cannot act on. + // The database identity is what makes the report readable as being about + // one database on one server. Unlike the bootstrap's preamble, a failure + // here is fatal: the bootstrap can converge without knowing the name, but a + // report that cannot say which database it read is one an operator cannot + // act on. diag, err := diagnoseStorageTarget(ctx, dsn) if err != nil { return nil, fmt.Errorf("read storage target identity: %w", err) } report.Database = diag.database + report.Host = diag.hostname - schemaFiles, err := readEmbeddedSchemaFiles() + schemaFiles, err := desired.mysqlSchemaFiles() if err != nil { return nil, err } @@ -300,16 +326,16 @@ func storageSchemaOperation(t ddl.StatementType) (string, error) { // diff and the bootstrap from drifting apart on a literal. const storageSchemaNamespace = "schemabot" -// diffPostgresStorageSchema diffs the embedded PostgreSQL schema files against +// diffPostgresStorageSchema diffs the desired PostgreSQL schema files against // the live storage database with the additive convergence's own drift scan, so // the report is exactly what ensurePostgresSchema would decide. The convergence // never drops or alters an existing object, so the report has no destructive // set; what it does have is the manual-remediation set, whose entries abort a // whole convergence pass rather than being skipped. -func diffPostgresStorageSchema(ctx context.Context, dsn string, o ensureSchemaOptions) (*StorageSchemaReport, error) { - report := &StorageSchemaReport{Dialect: schema.DialectPostgres} +func diffPostgresStorageSchema(ctx context.Context, dsn string, desired *StorageSchemaSource, o ensureSchemaOptions) (*StorageSchemaReport, error) { + report := &StorageSchemaReport{Dialect: schema.DialectPostgres, SchemaSource: desired.Describe()} - tables, files, err := readEmbeddedPostgresSchemaFiles() + tables, files, err := desired.postgresSchemaFiles() if err != nil { return nil, err } @@ -322,7 +348,13 @@ func diffPostgresStorageSchema(ctx context.Context, dsn string, o ensureSchemaOp if err := db.PingContext(ctx); err != nil { return nil, fmt.Errorf("ping storage database: %w", err) } - if err := db.QueryRowContext(ctx, "SELECT current_database()").Scan(&report.Database); err != nil { + // inet_server_addr() is null over a Unix socket and on some managed + // platforms, so the host is coalesced to empty rather than failing the + // read: a report that names the database but not the server is still + // actionable, and one that failed outright is not. + if err := db.QueryRowContext(ctx, + "SELECT current_database(), COALESCE(host(inet_server_addr()), '')", + ).Scan(&report.Database, &report.Host); err != nil { return nil, fmt.Errorf("read storage target identity: %w", err) } @@ -374,6 +406,8 @@ func (r *StorageSchemaReport) APIType() *apitypes.StorageSchemaReport { return &apitypes.StorageSchemaReport{ Dialect: string(r.Dialect), Database: r.Database, + Host: r.Host, + SchemaSource: r.SchemaSource, Version: r.Version, Converged: r.Converged(), Outstanding: storageSchemaStatementsAPIType(r.Outstanding), @@ -408,6 +442,8 @@ func StorageSchemaReportProto(r *StorageSchemaReport) *ternv1.StorageSchemaRepor return &ternv1.StorageSchemaReport{ Dialect: string(r.Dialect), Database: r.Database, + Host: r.Host, + SchemaSource: r.SchemaSource, Version: r.Version, Outstanding: storageSchemaStatementsProto(r.Outstanding), Destructive: storageSchemaStatementsProto(r.Destructive), @@ -428,6 +464,8 @@ func StorageSchemaReportFromProto(p *ternv1.StorageSchemaReport) *StorageSchemaR return &StorageSchemaReport{ Dialect: schema.Dialect(p.GetDialect()), Database: p.GetDatabase(), + Host: p.GetHost(), + SchemaSource: p.GetSchemaSource(), Version: p.GetVersion(), Outstanding: storageSchemaStatementsFromProto(p.GetOutstanding()), Destructive: storageSchemaStatementsFromProto(p.GetDestructive()), diff --git a/pkg/api/storage_schema_handlers.go b/pkg/api/storage_schema_handlers.go index 031b1f9fe..a508143c2 100644 --- a/pkg/api/storage_schema_handlers.go +++ b/pkg/api/storage_schema_handlers.go @@ -7,8 +7,9 @@ // that release *would* converge to; it does not tell you what the storage // actually converged to, and the two answers differ exactly when a deploy has // failed. Since that is the only time anyone asks, a version is never an input -// here: the diff is computed by the binary that is running, against the live -// catalog, in one call. +// here: the live side is always read from the catalog, and the desired side is +// always files — the answering binary's own, or a release's, sent by the caller +// and named in the report. // // That is also why a data plane's storage is reached through the data plane // rather than dialed from the control plane. Its storage database usually sits @@ -119,37 +120,51 @@ func (s *Service) resolveStorageSchemaTarget(deployment, environment string) (*s } // handleStorageSchemaDiff is the HTTP handler for -// GET /api/storage/schema/diff. +// POST /api/storage/schema/diff. // -// It is a GET because it only reads: it plans nothing, stores nothing, and -// takes no lock, so it is safe to call repeatedly against production while an -// incident is in progress. It is nonetheless admitted at the write tier and -// gated on admin membership — see storageSchemaOperation below — because what -// it returns is the internal shape of SchemaBot's own bookkeeping database, -// and because its sibling route converges that database. +// It reads and nothing else: it plans no change, stores nothing, and takes no +// lock, so it is safe to call repeatedly against production while an incident +// is in progress. It is a POST because the desired schema travels in the body — +// an operator asking what a database needs in order to match a later release +// sends that release's files, since the answering binary does not carry them. +// Being a POST also puts it at the write tier by the default rule, which is +// where it belongs: what it returns is the internal shape of SchemaBot's own +// bookkeeping database, and its sibling route converges that database. func (s *Service) handleStorageSchemaDiff(w http.ResponseWriter, r *http.Request) { - query := r.URL.Query() - deployment := query.Get("deployment") - environment := query.Get("environment") - allowDestructive := query.Get("allow_destructive") == "true" - + req, err := decodeStorageSchemaDiffRequest(r) + if err != nil { + s.writeBodyDecodeError(w, err) + return + } if !s.authorizeStorageSchemaOperation(w, r, storageSchemaDiffOperation) { return } - target, err := s.resolveStorageSchemaTarget(deployment, environment) + if err := validateStorageSchemaDiffRequest(req); err != nil { + s.logger.Warn("rejecting storage schema diff because its desired schema is incomplete", + "deployment", req.Deployment, "environment", req.Environment, "error", err) + s.writeError(w, http.StatusBadRequest, err.Error()) + return + } + target, err := s.resolveStorageSchemaTarget(req.Deployment, req.Environment) if err != nil { s.logger.Warn("rejecting storage schema diff because its target could not be resolved", - "deployment", deployment, "environment", environment, "error", err) + "deployment", req.Deployment, "environment", req.Environment, "error", err) s.writeError(w, http.StatusBadRequest, err.Error()) return } resp, err := target.service.StorageSchemaDiff(r.Context(), &ternv1.StorageSchemaDiffRequest{ - AllowDestructive: allowDestructive, + AllowDestructive: req.AllowDestructive, + SchemaFiles: req.SchemaFiles, + SchemaSource: req.SchemaSource, }) if err != nil { s.logger.Error("storage schema diff failed", - "deployment", target.deployment, "environment", target.environment, "error", err) + "deployment", target.deployment, + "environment", target.environment, + "schema_source", req.SchemaSource, + "schema_file_count", len(req.SchemaFiles), + "error", err) s.writeError(w, http.StatusInternalServerError, fmt.Sprintf("storage schema diff failed: %v", err)) return } @@ -256,34 +271,67 @@ const ( // // This is the second of two gates and it is not the one that usually bites. // The first is the tier the auth middleware admits the route at, and both -// routes are classified write there (auth.TierForRequest), including the -// read-only diff. That classification is what makes the admin requirement real -// on a deployment whose whole authorization model is read groups and write -// groups: the handler-level scoped-write decision is a pass-through until some -// database configures operator_groups, so a route left on the read tier would -// be readable by every reader no matter what this function said. +// routes are classified write there by auth.TierForRequest's default rule, +// since both are non-GET. That classification is what makes the admin +// requirement real on a deployment whose whole authorization model is read +// groups and write groups: the handler-level scoped-write decision is a +// pass-through until some database configures operator_groups, so a route left +// on the read tier would be readable by every reader no matter what this +// function said. func (s *Service) authorizeStorageSchemaOperation(w http.ResponseWriter, r *http.Request, operation string) bool { return s.authorizeDirectAdminWrite(w, r, operation) } // decodeStorageSchemaApplyRequest decodes the apply body, tolerating an empty -// one. Every field is optional — the defaults name this server's own storage -// and refuse destructive statements — so a caller sending no body at all gets -// the safe defaults rather than a decode error. Unknown fields are still -// rejected, so a misspelled "deployment" cannot quietly become a convergence -// of the wrong storage. +// one. func decodeStorageSchemaApplyRequest(r *http.Request) (apitypes.StorageSchemaApplyRequest, error) { - var req apitypes.StorageSchemaApplyRequest + return decodeOptionalStorageSchemaBody[apitypes.StorageSchemaApplyRequest](r) +} + +// decodeStorageSchemaDiffRequest decodes the diff body, tolerating an empty +// one. +func decodeStorageSchemaDiffRequest(r *http.Request) (apitypes.StorageSchemaDiffRequest, error) { + return decodeOptionalStorageSchemaBody[apitypes.StorageSchemaDiffRequest](r) +} + +// decodeOptionalStorageSchemaBody decodes a storage schema request body, +// tolerating an empty one. Every field of both requests is optional — the +// defaults name this server's own storage, its own embedded schema, and refuse +// destructive statements — so a caller sending no body at all gets the safe +// defaults rather than a decode error. Unknown fields are still rejected, so a +// misspelled "deployment" cannot quietly become a report about, or a +// convergence of, the wrong storage. +func decodeOptionalStorageSchemaBody[T any](r *http.Request) (T, error) { + var req T if r.Body == nil { return req, nil } decoder := json.NewDecoder(r.Body) decoder.DisallowUnknownFields() if err := decoder.Decode(&req); err != nil { + var zero T if errors.Is(err, io.EOF) { - return apitypes.StorageSchemaApplyRequest{}, nil + return zero, nil } - return apitypes.StorageSchemaApplyRequest{}, err + return zero, err } return req, nil } + +// validateStorageSchemaDiffRequest refuses a desired schema that is only half +// supplied. The two fields travel together or not at all: files without a +// source would produce a report that cannot say what it was diffed against, +// and a source without files would label this server's own embedded schema +// with somebody else's name — which is the one way a report of this kind can +// be actively misleading rather than merely wrong. +func validateStorageSchemaDiffRequest(req apitypes.StorageSchemaDiffRequest) error { + source := strings.TrimSpace(req.SchemaSource) + switch { + case len(req.SchemaFiles) > 0 && source == "": + return fmt.Errorf("schema_files was sent without schema_source: a report has to say which schema it was diffed against, so name the source (a release, a directory) alongside the files") + case len(req.SchemaFiles) == 0 && source != "": + return fmt.Errorf("schema_source %q was sent without schema_files: with no files the diff would run against this server's own embedded schema and report it under that name; send the files, or drop schema_source to ask about the embedded schema", source) + default: + return nil + } +} diff --git a/pkg/api/storage_schema_handlers_test.go b/pkg/api/storage_schema_handlers_test.go index 2926a4efc..1adf5a897 100644 --- a/pkg/api/storage_schema_handlers_test.go +++ b/pkg/api/storage_schema_handlers_test.go @@ -86,15 +86,11 @@ func newStorageSchemaService(t *testing.T, cfg *ServerConfig) *Service { return New(nil, cfg, nil, slog.New(slog.DiscardHandler)) } -func storageSchemaDiffRequest(t *testing.T, svc *Service, query string) *httptest.ResponseRecorder { +func storageSchemaDiffRequest(t *testing.T, svc *Service, body string) *httptest.ResponseRecorder { t.Helper() mux := http.NewServeMux() svc.ConfigureRoutes(mux) - path := "/api/storage/schema/diff" - if query != "" { - path += "?" + query - } - req := httptest.NewRequestWithContext(t.Context(), http.MethodGet, path, nil) + req := httptest.NewRequestWithContext(t.Context(), http.MethodPost, "/api/storage/schema/diff", strings.NewReader(body)) rec := httptest.NewRecorder() mux.ServeHTTP(rec, req) return rec @@ -166,7 +162,7 @@ func TestHandleStorageSchemaDiff_ForwardsAllowDestructive(t *testing.T) { } svc.SetStorageSchemaService(local) - rec := storageSchemaDiffRequest(t, svc, "allow_destructive=true") + rec := storageSchemaDiffRequest(t, svc, `{"allow_destructive":true}`) require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) assert.True(t, local.diffReq.GetAllowDestructive()) } @@ -199,7 +195,7 @@ func TestHandleStorageSchemaDiff_ReadsDataPlaneStorage(t *testing.T) { fakeStorageSchemaService: remote, }) - rec := storageSchemaDiffRequest(t, svc, "deployment=west&environment=production") + rec := storageSchemaDiffRequest(t, svc, `{"deployment":"west","environment":"production"}`) require.Equal(t, http.StatusOK, rec.Code, rec.Body.String()) report := decodeDiffResponse(t, rec).Report @@ -222,7 +218,7 @@ func TestHandleStorageSchemaDiff_UnconfiguredDeploymentDoesNotFallBack(t *testin } svc.SetStorageSchemaService(local) - rec := storageSchemaDiffRequest(t, svc, "deployment=east&environment=production") + rec := storageSchemaDiffRequest(t, svc, `{"deployment":"east","environment":"production"}`) require.Equal(t, http.StatusBadRequest, rec.Code) body := rec.Body.String() assert.Contains(t, body, "no data plane configured for deployment") @@ -242,7 +238,7 @@ func TestHandleStorageSchemaDiff_RefusesNonRoutableDeployment(t *testing.T) { }) svc.RegisterTernClient("west", "production", &mockTernClient{}) - rec := storageSchemaDiffRequest(t, svc, "deployment=west&environment=production") + rec := storageSchemaDiffRequest(t, svc, `{"deployment":"west","environment":"production"}`) require.Equal(t, http.StatusBadRequest, rec.Code) assert.Contains(t, rec.Body.String(), "in-process client") assert.Contains(t, rec.Body.String(), "west") @@ -257,11 +253,11 @@ func TestHandleStorageSchemaDiff_RefusesHalfNamedTarget(t *testing.T) { diffResp: &ternv1.StorageSchemaDiffResponse{Report: storageSchemaReportMessage("control_plane_storage")}, }) - deploymentOnly := storageSchemaDiffRequest(t, svc, "deployment=west") + deploymentOnly := storageSchemaDiffRequest(t, svc, `{"deployment":"west"}`) require.Equal(t, http.StatusBadRequest, deploymentOnly.Code) assert.Contains(t, deploymentOnly.Body.String(), "needs an environment") - environmentOnly := storageSchemaDiffRequest(t, svc, "environment=production") + environmentOnly := storageSchemaDiffRequest(t, svc, `{"environment":"production"}`) require.Equal(t, http.StatusBadRequest, environmentOnly.Code) assert.Contains(t, environmentOnly.Body.String(), "without a deployment") } @@ -363,7 +359,7 @@ func TestStorageSchemaRoutes_DenyScopedOperator(t *testing.T) { path string body string }{ - {http.MethodGet, "/api/storage/schema/diff", ""}, + {http.MethodPost, "/api/storage/schema/diff", `{}`}, {http.MethodPost, "/api/storage/schema/apply", `{}`}, } { t.Run(route.method+" "+route.path, func(t *testing.T) { @@ -390,7 +386,7 @@ func TestStorageSchemaRoutes_AllowAdmin(t *testing.T) { mux := http.NewServeMux() svc.ConfigureRoutes(mux) admin := auth.WithUser(t.Context(), &auth.User{Subject: "alice", Groups: []string{"schema-admins"}}) - req := httptest.NewRequestWithContext(admin, http.MethodGet, "/api/storage/schema/diff", nil) + req := httptest.NewRequestWithContext(admin, http.MethodPost, "/api/storage/schema/diff", strings.NewReader(`{}`)) rec := httptest.NewRecorder() mux.ServeHTTP(rec, req) diff --git a/pkg/api/storage_schema_integration_test.go b/pkg/api/storage_schema_integration_test.go index 6c26f1d24..299724d42 100644 --- a/pkg/api/storage_schema_integration_test.go +++ b/pkg/api/storage_schema_integration_test.go @@ -4,7 +4,9 @@ package api import ( "database/sql" + "fmt" "log/slog" + "maps" "os" "strings" "testing" @@ -51,7 +53,7 @@ func statementTables(statements []StorageSchemaStatement) []string { func TestDiffStorageSchemaMySQL_EmptyDatabaseNeedsEveryTable(t *testing.T) { sdb, db := openEnsureSchemaDatabase(t) - report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + report, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger()) require.NoError(t, err) assert.Equal(t, schema.DialectMySQL, report.Dialect) @@ -97,7 +99,7 @@ func TestApplyStorageSchemaMySQL_ConvergesEmptyDatabase(t *testing.T) { } // The confirming read is the same read an operator would run next. - after, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + after, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger()) require.NoError(t, err) assert.True(t, after.Converged()) assert.Empty(t, after.Outstanding) @@ -117,7 +119,7 @@ func TestDiffStorageSchemaMySQL_ReportsMissingColumn(t *testing.T) { require.NoError(t, err, "drop a column the embedded schema declares") require.False(t, testutil.ColumnExists(t, db, sdb.Name, "applies", "deployment")) - report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + report, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger()) require.NoError(t, err) assert.False(t, report.Converged()) @@ -153,7 +155,7 @@ func TestDiffStorageSchemaMySQL_RefusesSurplusTable(t *testing.T) { "CREATE TABLE `newer_release_state` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci") require.NoError(t, err, "create a table a newer release would own") - report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + report, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger()) require.NoError(t, err) assert.False(t, report.Converged(), "a refused statement is not convergence") @@ -175,7 +177,7 @@ func TestDiffStorageSchemaMySQL_RefusesSurplusTable(t *testing.T) { // The same report with destructive changes allowed says the statement would // run, which is what the operator opting in is asking to be told. - allowed, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger(), WithAllowDestructiveSchemaChanges(true)) + allowed, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger(), WithAllowDestructiveSchemaChanges(true)) require.NoError(t, err) assert.True(t, allowed.DestructiveAllowed) require.Len(t, allowed.Destructive, 1) @@ -201,7 +203,7 @@ func TestDiffStorageSchemaMySQL_SurplusIndexIsNotProtected(t *testing.T) { require.NoError(t, err, "pre-create an index the embedded schema does not declare") require.True(t, testutil.IndexExists(t, db, sdb.Name, "applies", "idx_applies_caller")) - report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + report, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger()) require.NoError(t, err) assert.False(t, report.Converged()) @@ -243,7 +245,7 @@ func TestDiffStorageSchemaMySQL_ReadsWhileBootstrapHoldsLock(t *testing.T) { _ = conn.QueryRowContext(t.Context(), "SELECT RELEASE_LOCK(?)", ensureSchemaLockName).Scan(&released) }() - report, err := DiffStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + report, err := DiffStorageSchema(t.Context(), sdb.DSN, nil, storageSchemaTestLogger()) require.NoError(t, err, "a diff must not wait on the bootstrap lock") assert.True(t, report.Converged()) } @@ -257,7 +259,7 @@ func TestDiffStorageSchemaPostgres_ConvergesEmptyDatabase(t *testing.T) { logger := storageSchemaTestLogger() postgres := WithDialect(schema.DialectPostgres) - report, err := DiffStorageSchema(t.Context(), dsn, logger, postgres) + report, err := DiffStorageSchema(t.Context(), dsn, nil, logger, postgres) require.NoError(t, err) assert.Equal(t, schema.DialectPostgres, report.Dialect) assert.Equal(t, "schemabot", report.Database) @@ -294,7 +296,7 @@ func TestDiffStorageSchemaPostgres_ReportsMissingColumn(t *testing.T) { _, err := db.ExecContext(t.Context(), `ALTER TABLE "applies" DROP COLUMN "deployment"`) require.NoError(t, err, "drop a column the embedded schema declares") - report, err := DiffStorageSchema(t.Context(), dsn, logger, postgres) + report, err := DiffStorageSchema(t.Context(), dsn, nil, logger, postgres) require.NoError(t, err) assert.False(t, report.Converged()) assert.Empty(t, report.Manual) @@ -319,11 +321,80 @@ func TestDiffStorageSchemaPostgres_ReportsMissingColumn(t *testing.T) { assert.True(t, testutil.PostgresColumnExists(t, db, "public", "applies", "deployment")) } +// The question a deploy actually asks is whether the storage is ready for the +// release about to roll, not whether it matches the release that is running. A +// database converged against the running schema still reports the next +// release's column as outstanding when the diff is given that release's schema +// files, and the report attributes the answer to those files rather than to the +// binary that read them. +// +// The live side is unaffected by any of it: the column is reported because the +// catalog does not have it, which is why the answer stays correct on a database +// a failed deploy left half converged. +func TestDiffStorageSchemaMySQL_DiffsAgainstASuppliedSchema(t *testing.T) { + sdb, _ := openEnsureSchemaDatabase(t) + require.NoError(t, EnsureSchema(sdb.DSN, storageSchemaTestLogger())) + + converged, err := DiffStorageSchema(t.Context(), sdb.DSN, EmbeddedStorageSchema("v1.2.3"), storageSchemaTestLogger()) + require.NoError(t, err) + require.True(t, converged.Converged(), "outstanding against the running schema: %v", statementTables(converged.Outstanding)) + assert.Equal(t, "the schema embedded in v1.2.3", converged.SchemaSource) + + // The next release's schema: the running one, plus a column on one table. + files, err := storageSchemaFilesForTest() + require.NoError(t, err) + files["applies.sql"] = strings.Replace(files["applies.sql"], + "PRIMARY KEY (`id`)", "`release_note` varchar(255) NOT NULL DEFAULT '',\n PRIMARY KEY (`id`)", 1) + require.Contains(t, files["applies.sql"], "release_note", "the fixture must actually declare the new column") + desired, err := StorageSchemaFromFiles("the schema files of release v1.4.0", files) + require.NoError(t, err) + + report, err := DiffStorageSchema(t.Context(), sdb.DSN, desired, storageSchemaTestLogger()) + require.NoError(t, err) + assert.False(t, report.Converged(), "the next release's column is not on this database yet") + assert.Equal(t, "the schema files of release v1.4.0", report.SchemaSource) + assert.Empty(t, report.Destructive) + assert.Empty(t, report.Manual) + require.Len(t, report.Outstanding, 1, "only the one table diverges: %v", statementTables(report.Outstanding)) + assert.Equal(t, "applies", report.Outstanding[0].Table) + assert.Contains(t, report.Outstanding[0].DDL, "ADD COLUMN") + assert.Contains(t, report.Outstanding[0].DDL, "`release_note`") + + // Attribution survives the stamp the responder applies: the answer came + // from the supplied files, whoever read them. + report.AttributeTo("v1.2.3") + assert.Equal(t, "the schema files of release v1.4.0", report.SchemaSource) + assert.Equal(t, "v1.2.3", report.Version) + + // A convergence has no way to run the supplied schema, so the column stays + // off the database until the release that declares it boots. + _, remaining, err := ApplyStorageSchema(t.Context(), sdb.DSN, storageSchemaTestLogger()) + require.NoError(t, err) + assert.True(t, remaining.Converged(), "an apply converges the running binary's schema, not the supplied one") + assert.Equal(t, "the schema embedded in this binary", remaining.SchemaSource) +} + +// storageSchemaFilesForTest is the embedded MySQL schema as a file-name map, the +// shape a caller-supplied schema takes. +func storageSchemaFilesForTest() (map[string]string, error) { + schemaFiles, err := readEmbeddedSchemaFiles() + if err != nil { + return nil, err + } + namespace := schemaFiles[storageSchemaNamespace] + if namespace == nil { + return nil, fmt.Errorf("embedded schema has no %q namespace", storageSchemaNamespace) + } + files := make(map[string]string, len(namespace.Files)) + maps.Copy(files, namespace.Files) + return files, nil +} + // A storage dialect with no differ fails closed rather than running another // family's catalog queries against it, and the refusal names both the dialect // asked for and the ones that exist. func TestDiffStorageSchema_UnsupportedDialectFailsClosed(t *testing.T) { - _, err := DiffStorageSchema(t.Context(), "unused", storageSchemaTestLogger(), WithDialect(schema.Dialect("sqlite"))) + _, err := DiffStorageSchema(t.Context(), "unused", nil, storageSchemaTestLogger(), WithDialect(schema.Dialect("sqlite"))) require.Error(t, err) assert.Contains(t, err.Error(), "sqlite") assert.Contains(t, err.Error(), string(schema.DialectMySQL)) diff --git a/pkg/api/storage_schema_source.go b/pkg/api/storage_schema_source.go new file mode 100644 index 000000000..1ad2022ee --- /dev/null +++ b/pkg/api/storage_schema_source.go @@ -0,0 +1,174 @@ +package api + +import ( + "fmt" + "maps" + "path" + "sort" + "strings" + + "github.com/block/schemabot/pkg/schema" +) + +// The desired side of a storage schema diff is an input, and naming it is part +// of the answer. +// +// The live side is never in question: it is read from the catalog of the one +// database the request addresses. The desired side is, because the question an +// operator asks during a deploy is not always "does this storage match the +// binary that is running". Deploying a later release asks "does this storage +// match the release I am about to roll", and the running binary cannot answer +// that from files it does not carry. +// +// So the schema may come from elsewhere — a checkout on disk, a published +// release — and wherever it came from, the report says so. Two diffs of one +// database against two releases give different answers, both correct, and a +// report that did not name its desired side would leave an operator holding +// the wrong one with no way to tell. +// +// A supplied schema is a diff-only input. ApplyStorageSchema takes no source +// and has no way to accept one: a convergence runs the schema of the binary +// running it, or "apply is what a boot does" — the property that makes this +// usable as a pre-deploy step — stops being true (AV-9). + +// StorageSchemaSource is the desired schema of a diff, with the words the +// report uses to attribute it. +type StorageSchemaSource struct { + // Description says where the schema came from, as a report renders it + // ("the schema embedded in v1.2.3", "the schema files in ./schema/mysql"). + Description string + // Files is the schema, as file name → file contents. Nil means the embedded + // files of this binary, read the same way a boot reads them. + Files map[string]string +} + +// embeddedStorageSchemaDescription attributes a diff to the running binary's +// own files when nothing said otherwise. +const embeddedStorageSchemaDescription = "the schema embedded in this binary" + +// EmbeddedStorageSchema is the desired schema a boot would converge to: the +// files embedded in this binary. The version is attribution only — it names +// whose files these are and is never used to find them, because a version that +// disagrees with the files it labels is exactly the failure this whole surface +// exists to avoid. +func EmbeddedStorageSchema(version string) *StorageSchemaSource { + description := embeddedStorageSchemaDescription + if v := strings.TrimSpace(version); v != "" { + description = fmt.Sprintf("the schema embedded in %s", v) + } + return &StorageSchemaSource{Description: description} +} + +// StorageSchemaFromFiles is a desired schema supplied by the caller, validated +// before anything reads a database with it. +// +// Validation is strict and happens up front because the failure it prevents is +// a quiet one. A file set that is missing half its tables still diffs cleanly +// and reports the missing ones as surplus — a report full of statements that +// destroy the storage schema an operator was about to extend. Refusing an +// unreadable set at the door turns that into an error naming the file. +func StorageSchemaFromFiles(description string, files map[string]string) (*StorageSchemaSource, error) { + description = strings.TrimSpace(description) + if description == "" { + return nil, fmt.Errorf("a supplied storage schema needs a description saying where it came from, so the report can attribute the answer to it") + } + if len(files) == 0 { + return nil, fmt.Errorf("no schema files supplied for %s: a diff against an empty schema would report every existing storage table as surplus", description) + } + names := make([]string, 0, len(files)) + for name := range files { + names = append(names, name) + } + sort.Strings(names) + for _, name := range names { + if err := validateStorageSchemaFileName(name); err != nil { + return nil, fmt.Errorf("schema file %q from %s: %w", name, description, err) + } + if strings.TrimSpace(files[name]) == "" { + return nil, fmt.Errorf("schema file %q from %s is empty: an empty file declares no table, so the table it is named for would be reported as surplus", name, description) + } + } + return &StorageSchemaSource{Description: description, Files: files}, nil +} + +// validateStorageSchemaFileName holds the file set to the shape both +// convergences already assume: a flat directory of .sql files, one per storage +// table. The PostgreSQL convergence derives a table name from the file name, so +// a nested path or a stray extension there is not a cosmetic problem — it +// becomes a table nothing declares. +func validateStorageSchemaFileName(name string) error { + if name == "" { + return fmt.Errorf("has no name") + } + if name != path.Base(name) || strings.ContainsAny(name, `/\`) { + return fmt.Errorf("is a path, not a file name; a storage schema is a flat directory of .sql files") + } + if !strings.HasSuffix(name, ".sql") { + return fmt.Errorf("is not a .sql file; a storage schema is a flat directory of .sql files, one per storage table") + } + if strings.TrimSuffix(name, ".sql") == "" { + return fmt.Errorf("names no table") + } + return nil +} + +// AttributeTo stamps the report with the SchemaBot version of the binary that +// produced it, and names that version as the desired schema's origin when the +// desired schema was that binary's own embedded files. +// +// It is a stamp applied after the fact because only the caller knows its own +// version, and a version is deliberately not an input to the diff itself. A +// schema the caller supplied keeps its own attribution: the answer came from +// those files, and relabelling them with the responder's version is exactly the +// misattribution the report's wording exists to prevent. +func (r *StorageSchemaReport) AttributeTo(version string) { + if r == nil { + return + } + r.Version = version + if r.SchemaSource == "" || r.SchemaSource == embeddedStorageSchemaDescription { + r.SchemaSource = EmbeddedStorageSchema(version).Description + } +} + +// Describe is the report's attribution for this source. +func (s *StorageSchemaSource) Describe() string { + if s == nil || strings.TrimSpace(s.Description) == "" { + return embeddedStorageSchemaDescription + } + return s.Description +} + +// mysqlSchemaFiles renders the source the way the MySQL differ consumes it — +// the same SchemaFiles shape ensureMySQLSchema builds, under the same +// namespace, so a supplied set and the embedded set travel one code path. +func (s *StorageSchemaSource) mysqlSchemaFiles() (schema.SchemaFiles, error) { + if s == nil || len(s.Files) == 0 { + return readEmbeddedSchemaFiles() + } + files := make(map[string]string, len(s.Files)) + maps.Copy(files, s.Files) + return schema.SchemaFiles{storageSchemaNamespace: &schema.Namespace{Files: files}}, nil +} + +// postgresSchemaFiles renders the source the way the PostgreSQL convergence +// consumes it: sorted table names, and contents keyed by table rather than by +// file. The table name is the file's base name, which is the same invariant the +// embedded reader relies on and the schema parity tests pin. +func (s *StorageSchemaSource) postgresSchemaFiles() ([]string, map[string]string, error) { + if s == nil || len(s.Files) == 0 { + return readEmbeddedPostgresSchemaFiles() + } + tables := make([]string, 0, len(s.Files)) + files := make(map[string]string, len(s.Files)) + for name, content := range s.Files { + table := strings.TrimSuffix(name, ".sql") + if existing, ok := files[table]; ok && existing != content { + return nil, nil, fmt.Errorf("two schema files from %s declare table %q with different contents; one file per storage table", s.Describe(), table) + } + files[table] = content + tables = append(tables, table) + } + sort.Strings(tables) + return tables, files, nil +} diff --git a/pkg/api/storage_schema_source_test.go b/pkg/api/storage_schema_source_test.go new file mode 100644 index 000000000..a479ea2f2 --- /dev/null +++ b/pkg/api/storage_schema_source_test.go @@ -0,0 +1,156 @@ +package api + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/apitypes" +) + +func validStorageSchemaFiles() map[string]string { + return map[string]string{ + "applies.sql": "CREATE TABLE `applies` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)", + "checks.sql": "CREATE TABLE `checks` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)", + } +} + +// A supplied schema is validated before it reads a database, because the +// failure it prevents is silent: a half-read file set diffs cleanly and reports +// the tables it is missing as surplus, which is a report full of statements +// that destroy the storage schema the operator was about to extend. +func TestStorageSchemaFromFiles_Validation(t *testing.T) { + tests := []struct { + name string + description string + files map[string]string + wantErr string + }{ + { + name: "a directory of .sql files, one per table", + description: "the schema files of release v1.4.0", + files: validStorageSchemaFiles(), + }, + { + name: "no description to attribute the answer to", + files: validStorageSchemaFiles(), + wantErr: "needs a description saying where it came from", + }, + { + name: "no files at all", + description: "the schema files in ./empty", + files: map[string]string{}, + wantErr: "would report every existing storage table as surplus", + }, + { + name: "a path instead of a file name", + description: "the schema files in ./checkout", + files: map[string]string{"mysql/applies.sql": "CREATE TABLE `applies` (`id` BIGINT UNSIGNED PRIMARY KEY)"}, + wantErr: "is a path, not a file name", + }, + { + name: "a file that is not .sql", + description: "the schema files in ./checkout", + files: map[string]string{"applies.txt": "CREATE TABLE `applies` (`id` BIGINT UNSIGNED PRIMARY KEY)"}, + wantErr: "is not a .sql file", + }, + { + name: "a file with no table in it", + description: "the schema files in ./checkout", + files: map[string]string{"applies.sql": " \n"}, + wantErr: "is empty", + }, + } + for _, tc := range tests { + t.Run(tc.name, func(t *testing.T) { + desired, err := StorageSchemaFromFiles(tc.description, tc.files) + if tc.wantErr != "" { + require.Error(t, err) + assert.Contains(t, err.Error(), tc.wantErr) + return + } + require.NoError(t, err) + assert.Equal(t, tc.description, desired.Describe()) + assert.Equal(t, tc.files, desired.Files) + }) + } +} + +// The embedded source is attributed to the version that embedded it, and says +// so without one rather than leaving the report's origin blank. +func TestEmbeddedStorageSchema(t *testing.T) { + assert.Equal(t, "the schema embedded in v1.2.3", EmbeddedStorageSchema("v1.2.3").Describe()) + assert.Equal(t, "the schema embedded in this binary", EmbeddedStorageSchema("").Describe()) + assert.Empty(t, EmbeddedStorageSchema("v1.2.3").Files, "the embedded source carries no files of its own") + assert.Equal(t, "the schema embedded in this binary", (*StorageSchemaSource)(nil).Describe(), + "no source at all is the embedded schema") +} + +// The responder's version stamp names the binary that answered, and never +// relabels a schema the caller supplied: a report claiming a release's answer +// came from the answering binary's files is the one way this report can mislead +// rather than merely be wrong. +func TestStorageSchemaReport_AttributeTo(t *testing.T) { + embedded := &StorageSchemaReport{SchemaSource: embeddedStorageSchemaDescription} + embedded.AttributeTo("v1.2.3") + assert.Equal(t, "the schema embedded in v1.2.3", embedded.SchemaSource) + assert.Equal(t, "v1.2.3", embedded.Version) + + unstamped := &StorageSchemaReport{} + unstamped.AttributeTo("v1.2.3") + assert.Equal(t, "the schema embedded in v1.2.3", unstamped.SchemaSource) + + supplied := &StorageSchemaReport{SchemaSource: "the schema files of release v1.4.0"} + supplied.AttributeTo("v1.2.3") + assert.Equal(t, "the schema files of release v1.4.0", supplied.SchemaSource, + "the schema the diff used outranks the version of the binary that read it") + assert.Equal(t, "v1.2.3", supplied.Version) +} + +// Each differ consumes the source in the shape its convergence already uses: a +// namespaced file map on MySQL, sorted table names and contents keyed by table +// on PostgreSQL. A supplied set and the embedded set travel the same path, so +// neither differ has a second way to read a schema. +func TestStorageSchemaSource_PerDialectShapes(t *testing.T) { + desired, err := StorageSchemaFromFiles("the schema files of release v1.4.0", validStorageSchemaFiles()) + require.NoError(t, err) + + mysqlFiles, err := desired.mysqlSchemaFiles() + require.NoError(t, err) + require.Contains(t, mysqlFiles, storageSchemaNamespace) + assert.Len(t, mysqlFiles[storageSchemaNamespace].Files, 2) + assert.Contains(t, mysqlFiles[storageSchemaNamespace].Files["applies.sql"], "CREATE TABLE `applies`") + + tables, contents, err := desired.postgresSchemaFiles() + require.NoError(t, err) + assert.Equal(t, []string{"applies", "checks"}, tables, "tables come out in the order a convergence runs them") + assert.Contains(t, contents["checks"], "CREATE TABLE `checks`") + + // No source falls through to the embedded files, which is what a boot reads. + embeddedMySQL, err := (*StorageSchemaSource)(nil).mysqlSchemaFiles() + require.NoError(t, err) + assert.NotEmpty(t, embeddedMySQL[storageSchemaNamespace].Files) + embeddedTables, _, err := (*StorageSchemaSource)(nil).postgresSchemaFiles() + require.NoError(t, err) + assert.NotEmpty(t, embeddedTables) +} + +// A half-supplied desired schema is refused. Files with no source produce a +// report that cannot say what it was compared against; a source with no files +// would label this server's own embedded schema with another release's name. +func TestValidateStorageSchemaDiffRequest(t *testing.T) { + require.NoError(t, validateStorageSchemaDiffRequest(apitypes.StorageSchemaDiffRequest{})) + require.NoError(t, validateStorageSchemaDiffRequest(apitypes.StorageSchemaDiffRequest{ + SchemaFiles: validStorageSchemaFiles(), + SchemaSource: "the schema files of release v1.4.0", + })) + + err := validateStorageSchemaDiffRequest(apitypes.StorageSchemaDiffRequest{SchemaFiles: validStorageSchemaFiles()}) + require.Error(t, err) + assert.Contains(t, err.Error(), "without schema_source") + + err = validateStorageSchemaDiffRequest(apitypes.StorageSchemaDiffRequest{SchemaSource: "the schema files of release v1.4.0"}) + require.Error(t, err) + assert.Contains(t, err.Error(), "without schema_files") +} diff --git a/pkg/apitypes/storage_schema.go b/pkg/apitypes/storage_schema.go index 76931f971..3900509e2 100644 --- a/pkg/apitypes/storage_schema.go +++ b/pkg/apitypes/storage_schema.go @@ -30,9 +30,17 @@ type StorageSchemaReport struct { // Database is the live database that was read, as its server reports it. // It is what makes a report unmistakably about one database. Database string `json:"database"` - // Version is the SchemaBot version of the binary whose embedded schema - // produced the diff. It is attribution, not an input: the diff came from - // the files themselves. + // Host is the database server as it names itself. Empty when the server + // does not report one. + Host string `json:"host,omitempty"` + // SchemaSource is where the desired side of the diff came from: the + // answering binary's embedded files, a directory, or a release. Always set, + // because one live database yields different answers against different + // releases. + SchemaSource string `json:"schema_source,omitempty"` + // Version is the SchemaBot version of the binary that answered. It is + // attribution, not an input: the diff came from schema files, and + // SchemaSource says which ones. Version string `json:"version,omitempty"` // Converged reports that the storage schema needs nothing at all. A report // carrying only refused destructive statements is not converged. @@ -43,8 +51,34 @@ type StorageSchemaReport struct { Manual []StorageSchemaStatement `json:"manual,omitempty"` } +// StorageSchemaDiffRequest is the HTTP request for +// POST /api/storage/schema/diff. +// +// The request reads and changes nothing, and is a POST because the desired +// schema travels in the body: a caller asking what a database needs in order to +// match a later release sends that release's files, which the answering binary +// does not have. Every field is optional — the defaults ask the addressed +// server what its own storage needs to match its own embedded schema. +type StorageSchemaDiffRequest struct { + // Deployment names the data plane whose storage to read. Empty reads the + // storage of the server the request is made to. + Deployment string `json:"deployment,omitempty"` + // Environment is required alongside Deployment, since a deployment serves + // one gRPC endpoint per environment. + Environment string `json:"environment,omitempty"` + // AllowDestructive reports the destructive statements as ones that would + // run. It executes nothing either way: a diff never does. + AllowDestructive bool `json:"allow_destructive,omitempty"` + // SchemaFiles is the desired schema as file name → file contents. Empty + // diffs against the answering binary's own embedded schema. + SchemaFiles map[string]string `json:"schema_files,omitempty"` + // SchemaSource says where SchemaFiles came from, in words, for the report + // to attribute the answer to. Required with SchemaFiles and never inferred. + SchemaSource string `json:"schema_source,omitempty"` +} + // StorageSchemaDiffResponse is the HTTP response for -// GET /api/storage/schema/diff. +// POST /api/storage/schema/diff. type StorageSchemaDiffResponse struct { Report *StorageSchemaReport `json:"report"` } diff --git a/pkg/auth/tiers.go b/pkg/auth/tiers.go index 918ba9a51..091e8a3eb 100644 --- a/pkg/auth/tiers.go +++ b/pkg/auth/tiers.go @@ -45,33 +45,13 @@ var readPaths = map[string]bool{ "/api/pull": true, } -// writePaths are GET endpoints that nonetheless require write access. They read -// SchemaBot's own storage schema — the internal shape of its bookkeeping -// database — rather than anything about a user's database, and they are the -// read half of a pair whose other half converges that database. An operator -// who may inspect the surplus and missing objects in SchemaBot's own storage is -// the operator who may converge it, so both halves are admitted at one tier. -// -// This is the tier that actually decides the question on a deployment -// configured with nothing but read groups and write groups, which is the common -// case: the handler-level scoped-write gate is a pass-through until some -// database configures operator groups, so a route left on the read tier here -// would be open to every reader regardless of what its handler checked. -var writePaths = map[string]bool{ - "/api/storage/schema/diff": true, -} - // TierForRequest classifies an API request into the access tier it requires. -// GET/HEAD requests are read unless listed in writePaths, the explicit -// read-only endpoints are read, and everything else is write — so a newly -// added mutating-looking endpoint fails closed (requires authorization) until -// it is classified here. Exported so the route authorization sweep test -// enforces per-database scoping against the same rule the middleware admits -// with. +// GET/HEAD requests are read, the explicit read-only endpoints are read, and +// everything else is write — so a newly added mutating-looking endpoint fails +// closed (requires authorization) until it is classified here. Exported so the +// route authorization sweep test enforces per-database scoping against the same +// rule the middleware admits with. func TierForRequest(method, path string) Tier { - if writePaths[path] { - return TierWrite - } if readPaths[path] { return TierRead } diff --git a/pkg/auth/tiers_test.go b/pkg/auth/tiers_test.go index bacc9f44b..ac5c40240 100644 --- a/pkg/auth/tiers_test.go +++ b/pkg/auth/tiers_test.go @@ -29,9 +29,8 @@ func TestTierForRequest(t *testing.T) { // SchemaBot's own storage schema is admin territory on both halves: // the diff exposes the internal shape of its bookkeeping database, and // its sibling route converges it. On a deployment configured with only - // read and write groups this tier is the whole admin decision, so the - // GET must not sit at the read tier. - {http.MethodGet, "/api/storage/schema/diff", TierWrite}, + // read and write groups this tier is the whole admin decision. + {http.MethodPost, "/api/storage/schema/diff", TierWrite}, {http.MethodPost, "/api/storage/schema/apply", TierWrite}, } for _, c := range cases { @@ -39,17 +38,16 @@ func TestTierForRequest(t *testing.T) { } } -// Every write-tier GET is listed in writePaths, and a GET outside that list -// stays a read. The list is the only thing standing between a read-tier -// classification and an endpoint everyone with read access can call, so it has -// to be exact rather than approximate. -func TestWritePathsCoverEveryWriteTierGet(t *testing.T) { - for path := range writePaths { - assert.Equalf(t, TierWrite, TierForRequest(http.MethodGet, path), "GET %s", path) - assert.Equalf(t, TierWrite, TierForRequest(http.MethodHead, path), "HEAD %s", path) +// Only the listed read-only endpoints escape the write tier by name. Every +// other non-GET path is a write, including one nobody has classified, so a new +// mutating endpoint is admitted at the write tier before anyone remembers to +// think about it. +func TestUnclassifiedNonGetPathsAreWrites(t *testing.T) { + for _, path := range []string{"/api/storage/schema/diff", "/api/newly/added/endpoint", "/api/pull/subresource"} { + assert.Equalf(t, TierWrite, TierForRequest(http.MethodPost, path), "POST %s", path) } - assert.Equal(t, TierRead, TierForRequest(http.MethodGet, "/api/storage/schema"), - "a prefix of a write path is not itself a write path") + assert.Equal(t, TierRead, TierForRequest(http.MethodPost, "/api/pull"), + "the read-only endpoints are listed by exact path") } func TestMatchesAnyGroup(t *testing.T) { diff --git a/pkg/cmd/client/client.go b/pkg/cmd/client/client.go index 1456891e3..a02849f92 100644 --- a/pkg/cmd/client/client.go +++ b/pkg/cmd/client/client.go @@ -135,28 +135,15 @@ func ChecksRepos(ctx context.Context, endpoint string) (*apitypes.ChecksReposRes // instance's own storage database — the server addressed by endpoint, or a data // plane reached through it when deployment is set. // -// The server computes the answer from the embedded schema files of the binary -// that is running, against the live catalog. No version is sent, because a -// version is the wrong input: it says what a release would converge to, not +// The server reads the live catalog and compares it against schema files: +// its own embedded ones, or the ones in req.SchemaFiles when the caller is +// asking about a release the server is not running. No version is sent, because +// a version is the wrong input: it says what a release would converge to, not // what the storage actually converged to, and the two differ exactly when a // deploy has failed to converge. -func StorageSchemaDiff(ctx context.Context, endpoint, deployment, environment string, allowDestructive bool) (*apitypes.StorageSchemaDiffResponse, error) { - values := url.Values{} - if deployment != "" { - values.Set("deployment", deployment) - } - if environment != "" { - values.Set("environment", environment) - } - if allowDestructive { - values.Set("allow_destructive", "true") - } - requestPath := "/api/storage/schema/diff" - if encoded := values.Encode(); encoded != "" { - requestPath += "?" + encoded - } +func StorageSchemaDiff(ctx context.Context, endpoint string, req apitypes.StorageSchemaDiffRequest) (*apitypes.StorageSchemaDiffResponse, error) { var result apitypes.StorageSchemaDiffResponse - if err := doSlowGetIntoCtx(ctx, endpoint, requestPath, &result); err != nil { + if err := doSlowPostIntoCtx(ctx, endpoint, "/api/storage/schema/diff", req, &result); err != nil { return nil, err } return &result, nil diff --git a/pkg/cmd/commands/storage_schema.go b/pkg/cmd/commands/storage_schema.go index fcd245dc1..2a0d063c4 100644 --- a/pkg/cmd/commands/storage_schema.go +++ b/pkg/cmd/commands/storage_schema.go @@ -13,16 +13,19 @@ import ( "github.com/block/schemabot/pkg/apitypes" cmdclient "github.com/block/schemabot/pkg/cmd/client" "github.com/block/schemabot/pkg/cmd/cliname" + "github.com/block/schemabot/pkg/schema" ) // Storage schema commands answer, and then close, the one question a deploy // that did not converge leaves open: which storage DDL is still outstanding on // this instance's own storage database. // -// Both read the live database. Neither takes a version: a release tag says what -// that release would converge to, not what the storage converged to, and the -// two answers differ exactly when a deploy has failed — which is the only time -// anyone runs these. +// Both read the live database, always. A release tag can name the schema to +// compare against (--release, --schema-dir), and it never stands in for the +// live side: what a release would converge to and what the storage actually +// converged to differ exactly when a deploy has failed, which is the only time +// anyone runs these. So the desired side is files — the answering binary's, a +// checkout's, or a tag's — and the report names which. // // There are two ways to reach a storage database, and which one applies is the // operator's to state rather than the CLI's to discover: @@ -93,8 +96,13 @@ func (f *storageSchemaTargetFlags) validate() error { // storage needs nothing, 2 when statements are outstanding, and 1 when the // read itself failed. A pre-deploy gate needs those three apart, because // "converged" and "unreachable" call for opposite decisions. +// The schema it compares against defaults to the answering binary's own, and +// --schema-dir or --release point it at another release's instead, for the +// question a deploy actually asks: is this storage ready for the release about +// to roll. type StorageDiffCmd struct { storageSchemaTargetFlags `embed:""` + storageSchemaSourceFlags `embed:""` AllowDestructive bool `help:"Report destructive statements as ones that would run, matching what an apply with the same flag would do" name:"allow-destructive"` JSON bool `help:"Output as JSON"` } @@ -103,6 +111,9 @@ func (cmd *StorageDiffCmd) Run(ctx context.Context, g *Globals) error { if err := cmd.validate(); err != nil { return err } + if err := cmd.validateSource(); err != nil { + return err + } report, err := cmd.read(ctx, g) if err != nil { return err @@ -113,7 +124,7 @@ func (cmd *StorageDiffCmd) Run(ctx context.Context, g *Globals) error { if err := encoder.Encode(apitypes.StorageSchemaDiffResponse{Report: report}); err != nil { return fmt.Errorf("encode storage schema report: %w", err) } - } else if err := renderStorageSchemaReport(os.Stdout, report, storageSchemaDiffHints(cmd)); err != nil { + } else if err := renderStorageSchemaReport(os.Stdout, report, storageSchemaDiffHints(cmd, report)); err != nil { return err } if report.Converged { @@ -125,30 +136,65 @@ func (cmd *StorageDiffCmd) Run(ctx context.Context, g *Globals) error { // read fetches the report over whichever path the flags selected. func (cmd *StorageDiffCmd) read(ctx context.Context, g *Globals) (*apitypes.StorageSchemaReport, error) { if cmd.direct() { - target, err := resolveStorageTarget(cmd.DSN, cmd.Config, cmd.Dialect) - if err != nil { - return nil, err - } - logger := storageSchemaLogger(g) - logger.Info("reading storage schema directly", - "source", target.source, "dialect", target.dialect) - report, err := api.DiffStorageSchema(ctx, target.dsn, logger, - api.WithDialect(target.dialect), - api.WithAllowDestructiveSchemaChanges(cmd.AllowDestructive)) - if err != nil { - return nil, fmt.Errorf("diff storage schema on the database from %s: %w", target.source, err) - } - // The version is this CLI's, because on this path the embedded schema - // files that produced the diff are this binary's own. - report.Version = g.Version - return report.APIType(), nil + return cmd.readDirect(ctx, g) + } + return cmd.readThroughAPI(ctx, g) +} + +// readDirect opens the storage database from this workstation and diffs it +// here, for when the server is down — including when it is down because its own +// schema bootstrap is failing. +func (cmd *StorageDiffCmd) readDirect(ctx context.Context, g *Globals) (*apitypes.StorageSchemaReport, error) { + target, err := resolveStorageTarget(cmd.DSN, cmd.Config, cmd.Dialect) + if err != nil { + return nil, err } + // The dialect is already resolved on this path, so a release fetch costs no + // extra round trip. + desired, err := cmd.resolve(ctx, func() (schema.Dialect, error) { return target.dialect, nil }) + if err != nil { + return nil, err + } + logger := storageSchemaLogger(g) + logger.Info("reading storage schema directly", + "source", target.source, "dialect", target.dialect, "schema_source", desired.Describe()) + report, err := api.DiffStorageSchema(ctx, target.dsn, desired, logger, + api.WithDialect(target.dialect), + api.WithAllowDestructiveSchemaChanges(cmd.AllowDestructive)) + if err != nil { + return nil, fmt.Errorf("diff storage schema on the database from %s: %w", target.source, err) + } + // This CLI answered, so the version is this binary's — and so is the + // schema, unless the operator named another release's. + report.AttributeTo(g.Version) + return report.APIType(), nil +} +// readThroughAPI asks the server, which reads its own storage or has a data +// plane read its own. A desired schema the operator named travels with the +// request, because the answering binary does not carry another release's files. +func (cmd *StorageDiffCmd) readThroughAPI(ctx context.Context, g *Globals) (*apitypes.StorageSchemaReport, error) { endpoint, err := g.Resolve() if err != nil { return nil, err } - response, err := cmdclient.StorageSchemaDiff(ctx, endpoint, cmd.Deployment, cmd.Environment, cmd.AllowDestructive) + request := apitypes.StorageSchemaDiffRequest{ + Deployment: cmd.Deployment, + Environment: cmd.Environment, + AllowDestructive: cmd.AllowDestructive, + } + desired, err := cmd.resolve(ctx, func() (schema.Dialect, error) { + return cmd.dialectThroughAPI(ctx, endpoint) + }) + if err != nil { + return nil, err + } + if desired != nil { + request.SchemaFiles = desired.Files + request.SchemaSource = desired.Description + } + + response, err := cmdclient.StorageSchemaDiff(ctx, endpoint, request) if err != nil { return nil, fmt.Errorf("diff storage schema%s: %w", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment), err) } @@ -158,6 +204,28 @@ func (cmd *StorageDiffCmd) read(ctx context.Context, g *Globals) (*apitypes.Stor return response.Report, nil } +// dialectThroughAPI asks the target which storage family it runs, so a release +// fetch reads the right one of its schema directories. +// +// It is the diff itself, asked with no desired schema: a read-only call that +// the target answers about its own storage, which is the only authority on the +// question. Asking costs a round trip and is why only --release pays for it — +// but asking beats making the operator state a dialect their control plane +// already knows, and beats guessing one and fetching DDL of the wrong family. +func (cmd *StorageDiffCmd) dialectThroughAPI(ctx context.Context, endpoint string) (schema.Dialect, error) { + response, err := cmdclient.StorageSchemaDiff(ctx, endpoint, apitypes.StorageSchemaDiffRequest{ + Deployment: cmd.Deployment, + Environment: cmd.Environment, + }) + if err != nil { + return "", fmt.Errorf("ask which storage family%s runs: %w", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment), err) + } + if response.Report == nil || response.Report.Dialect == "" { + return "", fmt.Errorf("the report for the storage%s named no dialect, so there is no way to tell which of a release's schema files apply to it", storageSchemaTargetSuffix(cmd.Deployment, cmd.Environment)) + } + return schema.Dialect(response.Report.Dialect), nil +} + // StorageApplyCmd converges a SchemaBot instance's storage database by running // the schema bootstrap that instance would run on its next boot. // @@ -165,17 +233,29 @@ func (cmd *StorageDiffCmd) read(ctx context.Context, g *Globals) (*apitypes.Stor // same refusal of destructive statements, and the same advisory lock — so two // operators running this at once serialize exactly the way two booting pods // do, and a pre-deploy convergence step is this command with nothing added. +// It converges to the schema of the binary that runs it, and there is no flag +// to point it at another release's — see storageSchemaSourceRefusal. type StorageApplyCmd struct { storageSchemaTargetFlags `embed:""` AllowDestructive bool `help:"Permit the destructive statements the convergence would otherwise refuse; it widens the target's standing storage policy and never narrows it" name:"allow-destructive"` AutoApprove bool `short:"y" help:"Skip confirmation prompt" name:"auto-approve"` JSON bool `help:"Output as JSON"` + // The diff's schema selectors are accepted here only to be refused with + // the reason and the alternative. An operator who has just run the diff + // against a release reaches for the same flags on the apply, and Kong's + // bare "unknown flag" would leave them guessing at whether the convergence + // silently used a different schema. + SchemaDir string `hidden:"" name:"schema-dir"` + Release string `hidden:""` } func (cmd *StorageApplyCmd) Run(ctx context.Context, g *Globals) error { if err := cmd.validate(); err != nil { return err } + if err := storageSchemaSourceRefusal(cmd.SchemaDir, cmd.Release); err != nil { + return err + } // Confirm against a fresh read rather than against a description of the // command: an operator approving DDL on SchemaBot's own storage should see @@ -291,18 +371,27 @@ func storageSchemaTargetSuffix(deployment, environment string) string { } // storageSchemaDatabaseLabel names the database a report is about, as an -// operator would say it: the database, its dialect, and the deployment it -// belongs to when the report came from one. +// operator would say it: the database, the server it is on, its dialect, and +// the deployment it belongs to when the report came from one. +// +// The server is in the label because "which database does this point at" is the +// question an operator has before they act on any of it, and the answer has to +// be legible without re-deriving it from a DSN, a config file, or a deployment +// name. A server that does not report a name of its own is left out rather than +// guessed at. func storageSchemaDatabaseLabel(report *apitypes.StorageSchemaReport) string { label := report.Database if label == "" { label = "the storage database" } + if report.Host != "" { + label += " on " + report.Host + } if report.Dialect != "" { label += fmt.Sprintf(" (%s)", report.Dialect) } if report.Deployment != "" { - label += fmt.Sprintf(" on deployment %s in %s", report.Deployment, report.Environment) + label += fmt.Sprintf(", deployment %s in %s", report.Deployment, report.Environment) } return label } @@ -357,14 +446,20 @@ func renderStorageSchemaReport(w io.Writer, report *apitypes.StorageSchemaReport } // storageSchemaHeadline is the one line an operator reads first: which database -// was read, and whether it needs anything. +// was read, which schema it was compared against, and whether it needs +// anything. +// +// The schema it was compared against is in the line and not in a footer, +// because it changes what the rest of the output means. The same database is +// converged against the release that is running and three statements short of +// the release about to roll, and both reports are correct. func storageSchemaHeadline(report *apitypes.StorageSchemaReport) string { - version := "" - if report.Version != "" { - version = fmt.Sprintf(", against the schema embedded in %s", report.Version) + against := "" + if report.SchemaSource != "" { + against = fmt.Sprintf(", against %s", report.SchemaSource) } if report.Converged { - return fmt.Sprintf("%s is converged%s.", storageSchemaDatabaseLabel(report), version) + return fmt.Sprintf("%s is converged%s.", storageSchemaDatabaseLabel(report), against) } counts := make([]string, 0, 3) if n := len(report.Outstanding); n > 0 { @@ -378,7 +473,7 @@ func storageSchemaHeadline(report *apitypes.StorageSchemaReport) string { } total := len(report.Outstanding) + len(report.Destructive) + len(report.Manual) return fmt.Sprintf("%s needs %d %s: %s%s.", - storageSchemaDatabaseLabel(report), total, pluralStatements(total), strings.Join(counts, ", "), version) + storageSchemaDatabaseLabel(report), total, pluralStatements(total), strings.Join(counts, ", "), against) } // storageSchemaOutstandingTitle says what the statements in the section are: @@ -399,7 +494,14 @@ func storageSchemaDestructiveTitle(report *apitypes.StorageSchemaReport) string // storageSchemaDiffHints names the next step for the report a diff just // printed, in the command form the operator invoked the CLI as. -func storageSchemaDiffHints(cmd *StorageDiffCmd) []string { +func storageSchemaDiffHints(cmd *StorageDiffCmd, report *apitypes.StorageSchemaReport) []string { + if cmd.selected() { + // Naming `storage apply` here would be wrong: it converges the schema + // of the binary that answers, which is not the schema this report is + // about. The two ways to converge the release's schema are the release + // itself, and this is where an operator is about to look for them. + return []string{fmt.Sprintf("These are what %s needs in order to match %s, not what its own next boot would run. To converge them, run that release's binary against this database — its container image is that release — or let the release's first boot converge them.", storageSchemaDatabaseLabel(report), report.SchemaSource)} + } if cmd.direct() { return []string{fmt.Sprintf("Converge it with: %s storage apply %s", cliname.Name(), storageSchemaDirectFlagHint(cmd.DSN, cmd.Config))} } diff --git a/pkg/cmd/commands/storage_schema_source.go b/pkg/cmd/commands/storage_schema_source.go new file mode 100644 index 000000000..4a853d024 --- /dev/null +++ b/pkg/cmd/commands/storage_schema_source.go @@ -0,0 +1,345 @@ +package commands + +import ( + "context" + "encoding/json" + "fmt" + "io" + "net/http" + "net/url" + "os" + "path/filepath" + "strings" + "time" + + "github.com/block/schemabot/pkg/api" + "github.com/block/schemabot/pkg/schema" +) + +// A diff has two sides, and only one of them is fixed. +// +// The live side is always the catalog of the database the command addresses. +// The desired side is whatever schema the operator is asking about, and during +// a deploy that is usually not the schema of the binary that answers: the +// question is whether the storage is ready for the release about to roll, and +// the release about to roll is by definition not the one running. +// +// So the desired side is selectable, two ways, and the report always says which +// one was used: +// +// --schema-dir the .sql files in a directory — a checkout of the +// release, or an unreleased commit, and offline +// --release the .sql files of a published tag, fetched from the +// repository +// +// Neither is available on `storage apply`, and that is the safety property +// rather than an omission: a convergence runs the schema of the binary running +// it, so "apply is what a boot does" holds by construction (AV-9). To converge +// a release's schema, run that release's binary. + +// storageSchemaSourceFlags selects the desired side of the diff. The two +// selectors are mutually exclusive: each names a complete schema, and silently +// preferring one would answer a question the operator did not ask. +type storageSchemaSourceFlags struct { + SchemaDir string `help:"Diff against the .sql files in this directory instead of the schema built into the binary that answers — a checkout of the release you are about to deploy (e.g. ./pkg/schema/mysql)" name:"schema-dir" type:"path"` + Release string `help:"Diff against the schema files of this published tag, fetched from the SchemaBot repository (e.g. v1.4.0)"` + Repo string `help:"Repository to fetch --release schema files from" name:"release-repo" default:"block/schemabot"` +} + +// selected reports whether the operator named a desired schema other than the +// answering binary's own. +func (f *storageSchemaSourceFlags) selected() bool { + return strings.TrimSpace(f.SchemaDir) != "" || strings.TrimSpace(f.Release) != "" +} + +// validateSource refuses selector combinations rather than resolving them by +// precedence. +func (f *storageSchemaSourceFlags) validateSource() error { + if strings.TrimSpace(f.SchemaDir) != "" && strings.TrimSpace(f.Release) != "" { + return fmt.Errorf("--schema-dir and --release both name a schema to diff against: pass one; --schema-dir reads files you already have, --release fetches a published tag") + } + repo := strings.TrimSpace(f.Repo) + if strings.TrimSpace(f.Release) == "" && repo != "" && repo != defaultStorageSchemaRepo { + return fmt.Errorf("--release-repo only applies with --release: it says which repository to fetch a published tag's schema files from") + } + return nil +} + +// storageSchemaSourceRefusal refuses the diff's schema selectors on a +// convergence, and names the two ways to converge a release's schema instead. +// +// The refusal is the invariant, stated where an operator meets it. A +// convergence runs the schema embedded in the binary running it, which is what +// makes an operator apply identical to the next boot's — so it can be used to +// converge storage ahead of a deploy without the fleet's own boots then +// disagreeing with it (AV-9). A convergence to files named on the command line +// would be a second implementation of the one path that must not have two. +func storageSchemaSourceRefusal(schemaDir, release string) error { + selector := "" + switch { + case strings.TrimSpace(release) != "": + selector = "--release" + case strings.TrimSpace(schemaDir) != "": + selector = "--schema-dir" + default: + return nil + } + return fmt.Errorf("%s cannot be used with a convergence: an apply runs the schema embedded in the binary running it, so that it converges exactly what that binary's next boot would. To converge a release's schema, run that release's binary — its container image is that release — or let the release's own first boot converge it. To see what it would do, use the same flag on `storage diff`", selector) +} + +// resolve reads the desired schema the flags selected, or returns nil for the +// answering binary's own embedded schema. +// +// dialect is resolved lazily, by calling it, because only one selector needs +// it: a release's schema files live in a per-dialect directory of the +// repository, while a directory on disk is read as given. On the API path +// resolving the dialect costs a round trip, so the release path pays for it and +// the directory path does not. +func (f *storageSchemaSourceFlags) resolve(ctx context.Context, dialect func() (schema.Dialect, error)) (*api.StorageSchemaSource, error) { + if dir := strings.TrimSpace(f.SchemaDir); dir != "" { + return storageSchemaFromDirectory(dir) + } + release := strings.TrimSpace(f.Release) + if release == "" { + return nil, nil + } + d, err := dialect() + if err != nil { + return nil, fmt.Errorf("resolve the storage dialect, which says which of release %s's schema files to fetch: %w", release, err) + } + repo := strings.TrimSpace(f.Repo) + if repo == "" { + repo = defaultStorageSchemaRepo + } + return storageSchemaFromRelease(ctx, repo, release, d) +} + +// storageSchemaFromDirectory reads a release's schema files from a checkout. +// +// The directory is the dialect's own schema directory, read as given: a +// checkout of the release is the exact schema that release embeds, which is +// what makes this the offline answer and the one that also works for a commit +// that was never tagged. +func storageSchemaFromDirectory(dir string) (*api.StorageSchemaSource, error) { + absolute, err := filepath.Abs(dir) + if err != nil { + return nil, fmt.Errorf("resolve --schema-dir %s: %w", dir, err) + } + entries, err := os.ReadDir(absolute) + if err != nil { + return nil, fmt.Errorf("read the schema files in --schema-dir %s: %w", absolute, err) + } + files := make(map[string]string) + for _, entry := range entries { + if entry.IsDir() || !strings.HasSuffix(entry.Name(), ".sql") { + continue + } + content, err := os.ReadFile(filepath.Join(absolute, entry.Name())) + if err != nil { + return nil, fmt.Errorf("read schema file %s in --schema-dir %s: %w", entry.Name(), absolute, err) + } + files[entry.Name()] = string(content) + } + if len(files) == 0 { + return nil, fmt.Errorf("no .sql files in --schema-dir %s%s", absolute, storageSchemaDirectoryHint(absolute, entries)) + } + return api.StorageSchemaFromFiles(fmt.Sprintf("the schema files in %s", absolute), files) +} + +// storageSchemaDirectoryHint names the per-dialect subdirectories when the +// operator pointed one level too high, which is the likeliest way to land on a +// directory with no .sql files in it. +func storageSchemaDirectoryHint(dir string, entries []os.DirEntry) string { + subdirectories := make([]string, 0, 2) + for _, entry := range entries { + if !entry.IsDir() { + continue + } + if name := entry.Name(); name == string(schema.DialectMySQL) || name == string(schema.DialectPostgres) { + subdirectories = append(subdirectories, filepath.Join(dir, name)) + } + } + if len(subdirectories) == 0 { + return "; point it at a release's schema directory, which holds one .sql file per storage table" + } + return fmt.Sprintf("; it holds a directory per dialect, so point --schema-dir at the one your storage runs: %s", strings.Join(subdirectories, " or ")) +} + +// defaultStorageSchemaRepo is where a released tag's schema files are +// published. A private mirror or a fork is reachable with --release-repo. +const defaultStorageSchemaRepo = "block/schemabot" + +// storageSchemaReleaseAPIBase is the GitHub API a release's files are fetched +// from. The environment variable is the one GitHub tooling already sets, so a +// GitHub Enterprise host needs no flag of its own. +func storageSchemaReleaseAPIBase() string { + if base := strings.TrimSpace(os.Getenv("GITHUB_API_URL")); base != "" { + return strings.TrimRight(base, "/") + } + return "https://api.github.com" +} + +// storageSchemaReleaseToken authorizes the fetch when one is available. +// Unauthenticated requests work against a public repository and are rate +// limited; a private mirror needs a token. +func storageSchemaReleaseToken() string { + for _, key := range []string{"GITHUB_TOKEN", "GH_TOKEN"} { + if token := strings.TrimSpace(os.Getenv(key)); token != "" { + return token + } + } + return "" +} + +const ( + // storageSchemaFetchTimeout bounds the whole fetch. A diff is run at a + // terminal during a deploy, so a repository that does not answer has to + // fail rather than hang: the operator still has --schema-dir, which needs + // no network at all. + storageSchemaFetchTimeout = 30 * time.Second + // storageSchemaMaxFileBytes and storageSchemaMaxFiles bound what a fetch + // will read. The storage schema is a few dozen small files; anything past + // these bounds means the path being fetched is not a storage schema + // directory, and saying so beats reading a repository into memory. + storageSchemaMaxFileBytes = 1 << 20 + storageSchemaMaxFiles = 256 +) + +// storageSchemaFromRelease fetches one release's storage schema files for the +// dialect the live storage runs. +// +// It reads the files at the tag rather than a release artifact, because the +// files are what the release embeds: the same directory the binary's //go:embed +// captured, at the same commit. A tag that does not exist, or a repository the +// caller cannot read, is an error naming the tag — never an empty schema, which +// would diff as "every storage table is surplus". +func storageSchemaFromRelease(ctx context.Context, repo, tag string, dialect schema.Dialect) (*api.StorageSchemaSource, error) { + directory, err := storageSchemaReleaseDirectory(dialect) + if err != nil { + return nil, err + } + client := &http.Client{Timeout: storageSchemaFetchTimeout} + listing, err := listReleaseSchemaFiles(ctx, client, repo, tag, directory) + if err != nil { + return nil, err + } + files := make(map[string]string, len(listing)) + for _, entry := range listing { + content, err := fetchReleaseSchemaFile(ctx, client, repo, tag, entry.Path) + if err != nil { + return nil, err + } + files[entry.Name] = content + } + description := fmt.Sprintf("the schema files of release %s", tag) + if repo != defaultStorageSchemaRepo { + description = fmt.Sprintf("the schema files of release %s in %s", tag, repo) + } + return api.StorageSchemaFromFiles(description, files) +} + +// storageSchemaReleaseDirectory is the path in the repository holding one +// dialect's schema files. It fails closed for a dialect with no directory +// rather than fetching another family's DDL, which would parse as a schema and +// diff as nonsense. +func storageSchemaReleaseDirectory(dialect schema.Dialect) (string, error) { + switch dialect { + case schema.DialectMySQL, schema.DialectPostgres: + return "pkg/schema/" + string(dialect), nil + default: + return "", fmt.Errorf("no published schema directory for storage dialect %q (%q and %q have one)", dialect, schema.DialectMySQL, schema.DialectPostgres) + } +} + +// releaseSchemaEntry is the one part of a repository listing this needs. +type releaseSchemaEntry struct { + Name string `json:"name"` + Path string `json:"path"` + Type string `json:"type"` +} + +// listReleaseSchemaFiles lists the .sql files in one directory of the +// repository at a tag. +func listReleaseSchemaFiles(ctx context.Context, client *http.Client, repo, tag, directory string) ([]releaseSchemaEntry, error) { + body, err := getReleaseContents(ctx, client, repo, tag, directory, "application/vnd.github+json") + if err != nil { + return nil, err + } + var entries []releaseSchemaEntry + if err := json.Unmarshal(body, &entries); err != nil { + return nil, fmt.Errorf("read the listing of %s at %s in %s: %w", directory, tag, repo, err) + } + files := make([]releaseSchemaEntry, 0, len(entries)) + for _, entry := range entries { + if entry.Type == "file" && strings.HasSuffix(entry.Name, ".sql") { + files = append(files, entry) + } + } + if len(files) == 0 { + return nil, fmt.Errorf("release %s in %s has no .sql files in %s: check that the tag exists and names a SchemaBot release", tag, repo, directory) + } + if len(files) > storageSchemaMaxFiles { + return nil, fmt.Errorf("release %s in %s has %d .sql files in %s, past the %d a storage schema can have; check that the tag names a SchemaBot release", tag, repo, len(files), directory, storageSchemaMaxFiles) + } + return files, nil +} + +// fetchReleaseSchemaFile reads one schema file's contents at a tag. +func fetchReleaseSchemaFile(ctx context.Context, client *http.Client, repo, tag, path string) (string, error) { + body, err := getReleaseContents(ctx, client, repo, tag, path, "application/vnd.github.raw") + if err != nil { + return "", err + } + return string(body), nil +} + +// getReleaseContents reads one path of the repository at a tag. +// +// The contents API is used for both the listing and the files, rather than the +// raw download URLs a listing also carries, so one request shape and one +// authorization header cover a public repository and a private mirror alike. +func getReleaseContents(ctx context.Context, client *http.Client, repo, tag, path, accept string) ([]byte, error) { + endpoint := fmt.Sprintf("%s/repos/%s/contents/%s?ref=%s", + storageSchemaReleaseAPIBase(), repo, path, url.QueryEscape(tag)) + request, err := http.NewRequestWithContext(ctx, http.MethodGet, endpoint, nil) + if err != nil { + return nil, fmt.Errorf("build the request for %s at %s in %s: %w", path, tag, repo, err) + } + request.Header.Set("Accept", accept) + request.Header.Set("X-GitHub-Api-Version", "2022-11-28") + if token := storageSchemaReleaseToken(); token != "" { + request.Header.Set("Authorization", "Bearer "+token) + } + + response, err := client.Do(request) + if err != nil { + return nil, fmt.Errorf("fetch %s at %s from %s: %w; --schema-dir reads the same files from a checkout without the network", path, tag, repo, err) + } + defer response.Body.Close() + if response.StatusCode != http.StatusOK { + return nil, releaseContentsError(response, repo, tag, path) + } + body, err := io.ReadAll(io.LimitReader(response.Body, storageSchemaMaxFileBytes+1)) + if err != nil { + return nil, fmt.Errorf("read %s at %s from %s: %w", path, tag, repo, err) + } + if len(body) > storageSchemaMaxFileBytes { + return nil, fmt.Errorf("%s at %s in %s is larger than the %d MiB a storage schema file can be; check that the tag names a SchemaBot release", path, tag, repo, storageSchemaMaxFileBytes>>20) + } + return body, nil +} + +// releaseContentsError turns a fetch failure into the remediation for it. The +// three that happen are a tag that does not exist, a repository the caller +// cannot read, and a rate limit — each with a different fix, and all three +// solved for good by --schema-dir. +func releaseContentsError(response *http.Response, repo, tag, path string) error { + switch response.StatusCode { + case http.StatusNotFound: + return fmt.Errorf("release %s in %s has no %s: check the tag spelling, or pass --release-repo if the release is published elsewhere", tag, repo, path) + case http.StatusUnauthorized, http.StatusForbidden: + return fmt.Errorf("not allowed to read %s at %s in %s (HTTP %d): set GITHUB_TOKEN to a token that can read the repository, or use --schema-dir to read the files from a checkout instead", path, tag, repo, response.StatusCode) + default: + return fmt.Errorf("fetching %s at %s from %s failed with HTTP %d; --schema-dir reads the same files from a checkout without the network", path, tag, repo, response.StatusCode) + } +} diff --git a/pkg/cmd/commands/storage_schema_source_test.go b/pkg/cmd/commands/storage_schema_source_test.go new file mode 100644 index 000000000..fc6d0d663 --- /dev/null +++ b/pkg/cmd/commands/storage_schema_source_test.go @@ -0,0 +1,239 @@ +package commands + +import ( + "fmt" + "net/http" + "net/http/httptest" + "os" + "path/filepath" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/schema" +) + +// mysqlDialect answers the dialect resolver without a database or a server, +// for the cases where resolving it is not what is under test. +func mysqlDialect() (schema.Dialect, error) { return schema.DialectMySQL, nil } + +// The two schema selectors each name a complete schema, so naming both is +// refused rather than resolved by precedence: silently preferring one would +// answer a question the operator did not ask. +func TestStorageSchemaSourceFlags_ValidateSource(t *testing.T) { + require.NoError(t, (&storageSchemaSourceFlags{}).validateSource()) + require.NoError(t, (&storageSchemaSourceFlags{SchemaDir: "./schema/mysql"}).validateSource()) + require.NoError(t, (&storageSchemaSourceFlags{Release: "v1.4.0", Repo: defaultStorageSchemaRepo}).validateSource()) + require.NoError(t, (&storageSchemaSourceFlags{Release: "v1.4.0", Repo: "example/mirror"}).validateSource()) + + err := (&storageSchemaSourceFlags{SchemaDir: "./schema/mysql", Release: "v1.4.0"}).validateSource() + require.Error(t, err) + assert.Contains(t, err.Error(), "both name a schema to diff against") + + err = (&storageSchemaSourceFlags{Repo: "example/mirror"}).validateSource() + require.Error(t, err) + assert.Contains(t, err.Error(), "--release-repo only applies with --release") +} + +// With no selector the diff is against the schema of the binary that answers, +// which is what a boot would converge to. Resolving that needs no files and no +// dialect, so nothing is read and nothing is fetched. +func TestStorageSchemaSourceFlags_ResolveDefaultsToTheAnsweringBinary(t *testing.T) { + desired, err := (&storageSchemaSourceFlags{}).resolve(t.Context(), func() (schema.Dialect, error) { + t.Fatal("the dialect must not be resolved when no schema was named") + return "", nil + }) + require.NoError(t, err) + assert.Nil(t, desired, "a nil source is the answering binary's own embedded schema") +} + +// A directory of .sql files is read as given, and the report attributes the +// answer to the directory by path — an operator running two diffs from two +// checkouts has to be able to tell the answers apart. +func TestStorageSchemaFromDirectory(t *testing.T) { + dir := t.TempDir() + require.NoError(t, os.WriteFile(filepath.Join(dir, "applies.sql"), + []byte("CREATE TABLE `applies` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)"), 0o600)) + require.NoError(t, os.WriteFile(filepath.Join(dir, "checks.sql"), + []byte("CREATE TABLE `checks` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)"), 0o600)) + require.NoError(t, os.WriteFile(filepath.Join(dir, "README.md"), []byte("not schema"), 0o600)) + + desired, err := (&storageSchemaSourceFlags{SchemaDir: dir}).resolve(t.Context(), mysqlDialect) + require.NoError(t, err) + require.NotNil(t, desired) + assert.Equal(t, fmt.Sprintf("the schema files in %s", dir), desired.Description) + assert.Len(t, desired.Files, 2, "only .sql files are schema") + assert.Contains(t, desired.Files["applies.sql"], "CREATE TABLE `applies`") + assert.NotContains(t, desired.Files, "README.md") +} + +// A directory with no .sql files is refused. A diff against an empty schema +// would report every existing storage table as surplus, which reads as a +// storage database that needs destroying rather than as a mistyped path. +func TestStorageSchemaFromDirectory_RefusesEmptyDirectory(t *testing.T) { + _, err := (&storageSchemaSourceFlags{SchemaDir: t.TempDir()}).resolve(t.Context(), mysqlDialect) + require.Error(t, err) + assert.Contains(t, err.Error(), "no .sql files in --schema-dir") + assert.Contains(t, err.Error(), "one .sql file per storage table") +} + +// Pointing one level too high is the likeliest mistake, so the refusal names +// the per-dialect directories that are actually there rather than restating +// that the path was wrong. +func TestStorageSchemaFromDirectory_NamesDialectSubdirectories(t *testing.T) { + dir := t.TempDir() + require.NoError(t, os.Mkdir(filepath.Join(dir, string(schema.DialectMySQL)), 0o750)) + require.NoError(t, os.Mkdir(filepath.Join(dir, string(schema.DialectPostgres)), 0o750)) + + _, err := (&storageSchemaSourceFlags{SchemaDir: dir}).resolve(t.Context(), mysqlDialect) + require.Error(t, err) + assert.Contains(t, err.Error(), filepath.Join(dir, "mysql")) + assert.Contains(t, err.Error(), filepath.Join(dir, "postgres")) +} + +// A missing directory fails naming the path, rather than resolving to an empty +// schema. +func TestStorageSchemaFromDirectory_RefusesMissingDirectory(t *testing.T) { + missing := filepath.Join(t.TempDir(), "not-a-checkout") + _, err := (&storageSchemaSourceFlags{SchemaDir: missing}).resolve(t.Context(), mysqlDialect) + require.Error(t, err) + assert.Contains(t, err.Error(), missing) +} + +// releaseSchemaServer stands in for the repository API: it serves a listing for +// the dialect's schema directory and the raw contents of each file, and records +// the refs it was asked for so a test can assert the tag was honored. +func releaseSchemaServer(t *testing.T, directory string, files map[string]string) (*httptest.Server, *[]string) { + t.Helper() + refs := make([]string, 0, len(files)+1) + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + refs = append(refs, r.URL.Query().Get("ref")) + prefix := "/repos/example/schemabot/contents/" + path := strings.TrimPrefix(r.URL.Path, prefix) + if path == directory { + entries := make([]string, 0, len(files)) + for name := range files { + entries = append(entries, fmt.Sprintf(`{"name":%q,"path":%q,"type":"file"}`, name, directory+"/"+name)) + } + entries = append(entries, fmt.Sprintf(`{"name":"nested","path":%q,"type":"dir"}`, directory+"/nested")) + _, err := fmt.Fprintf(w, "[%s]", strings.Join(entries, ",")) + assert.NoError(t, err) + return + } + content, ok := files[strings.TrimPrefix(path, directory+"/")] + if !ok { + w.WriteHeader(http.StatusNotFound) + return + } + _, err := w.Write([]byte(content)) + assert.NoError(t, err) + })) + t.Cleanup(server.Close) + t.Setenv("GITHUB_API_URL", server.URL) + return server, &refs +} + +// A release's schema comes from the files at that tag, for the dialect the live +// storage runs — the same directory that release's binary embedded. The report +// attributes the answer to the release, never to the binary that answered. +func TestStorageSchemaFromRelease(t *testing.T) { + _, refs := releaseSchemaServer(t, "pkg/schema/mysql", map[string]string{ + "applies.sql": "CREATE TABLE `applies` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)", + "checks.sql": "CREATE TABLE `checks` (`id` BIGINT UNSIGNED AUTO_INCREMENT PRIMARY KEY)", + }) + + flags := &storageSchemaSourceFlags{Release: "v1.4.0", Repo: "example/schemabot"} + desired, err := flags.resolve(t.Context(), mysqlDialect) + require.NoError(t, err) + require.NotNil(t, desired) + assert.Equal(t, "the schema files of release v1.4.0 in example/schemabot", desired.Description) + assert.Len(t, desired.Files, 2, "the listing's directory entry is not a schema file") + assert.Contains(t, desired.Files["checks.sql"], "CREATE TABLE `checks`") + for _, ref := range *refs { + assert.Equal(t, "v1.4.0", ref, "every read is pinned to the tag that was asked for") + } +} + +// The dialect decides which of a release's schema directories to read, so a +// PostgreSQL storage is never diffed against MySQL DDL that happened to be +// published alongside it. +func TestStorageSchemaFromRelease_FetchesTheStorageDialectsFiles(t *testing.T) { + releaseSchemaServer(t, "pkg/schema/postgres", map[string]string{ + "applies.sql": `CREATE TABLE "applies" (id bigserial PRIMARY KEY)`, + }) + + flags := &storageSchemaSourceFlags{Release: "v1.4.0", Repo: "example/schemabot"} + desired, err := flags.resolve(t.Context(), func() (schema.Dialect, error) { return schema.DialectPostgres, nil }) + require.NoError(t, err) + require.NotNil(t, desired) + assert.Contains(t, desired.Files["applies.sql"], `CREATE TABLE "applies"`) + + _, err = flags.resolve(t.Context(), mysqlDialect) + require.Error(t, err, "the MySQL directory is not published by this fixture") + assert.Contains(t, err.Error(), "pkg/schema/mysql") +} + +// A tag that does not exist is an error naming the tag. The one answer it must +// never produce is an empty schema, which diffs as "every storage table is +// surplus". +func TestStorageSchemaFromRelease_RefusesUnknownTag(t *testing.T) { + releaseSchemaServer(t, "pkg/schema/mysql", map[string]string{}) + + _, err := (&storageSchemaSourceFlags{Release: "v9.9.9", Repo: "example/schemabot"}).resolve(t.Context(), mysqlDialect) + require.Error(t, err) + assert.Contains(t, err.Error(), "v9.9.9") + assert.Contains(t, err.Error(), "no .sql files") +} + +// A repository the caller cannot read names the token to set and the offline +// alternative, because both are decisions the operator makes at the terminal +// mid-deploy. +func TestStorageSchemaFromRelease_RefusalNamesTheRemedy(t *testing.T) { + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, _ *http.Request) { + w.WriteHeader(http.StatusForbidden) + })) + t.Cleanup(server.Close) + t.Setenv("GITHUB_API_URL", server.URL) + + _, err := (&storageSchemaSourceFlags{Release: "v1.4.0", Repo: "example/private"}).resolve(t.Context(), mysqlDialect) + require.Error(t, err) + assert.Contains(t, err.Error(), "GITHUB_TOKEN") + assert.Contains(t, err.Error(), "--schema-dir") +} + +// The fetch is authorized when a token is available, so a private mirror works +// without a flag of its own. +func TestStorageSchemaFromRelease_SendsTheToken(t *testing.T) { + var authorization string + server := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) { + authorization = r.Header.Get("Authorization") + _, err := w.Write([]byte(`[]`)) + assert.NoError(t, err) + })) + t.Cleanup(server.Close) + t.Setenv("GITHUB_API_URL", server.URL) + t.Setenv("GITHUB_TOKEN", "fetch-token") + + _, err := (&storageSchemaSourceFlags{Release: "v1.4.0", Repo: "example/private"}).resolve(t.Context(), mysqlDialect) + require.Error(t, err, "an empty listing is still refused") + assert.Equal(t, "Bearer fetch-token", authorization) +} + +// A convergence runs the schema embedded in the binary running it, so the +// diff's selectors are refused on `storage apply` — with the two ways to +// converge a release named, since that is what the operator is reaching for. +func TestStorageSchemaSourceRefusal(t *testing.T) { + require.NoError(t, storageSchemaSourceRefusal("", "")) + + release := storageSchemaSourceRefusal("", "v1.4.0") + require.Error(t, release) + assert.Contains(t, release.Error(), "--release cannot be used with a convergence") + assert.Contains(t, release.Error(), "run that release's binary") + assert.Contains(t, release.Error(), "storage diff") + + dir := storageSchemaSourceRefusal("./schema/mysql", "") + require.Error(t, dir) + assert.Contains(t, dir.Error(), "--schema-dir cannot be used with a convergence") +} diff --git a/pkg/cmd/commands/storage_schema_test.go b/pkg/cmd/commands/storage_schema_test.go index 0de84f5ff..15cf4912b 100644 --- a/pkg/cmd/commands/storage_schema_test.go +++ b/pkg/cmd/commands/storage_schema_test.go @@ -192,32 +192,37 @@ func TestStorageSchemaRunnableDDL(t *testing.T) { assert.Equal(t, "", storageSchemaRunnableDDL(""), "nothing to terminate stays untouched") } -// The headline is the line an operator reads first, so it names the database -// and the binary whose embedded schema produced the diff. The version matters: -// the answer is only meaningful against the schema files it was compared with. +// The headline is the line an operator reads first, so it names the database, +// the server it is on, and the schema it was compared against. The schema +// matters as much as the counts: the same database is converged against one +// release and short of another, and both answers are correct. func TestStorageSchemaHeadline(t *testing.T) { converged := storageSchemaHeadline(&apitypes.StorageSchemaReport{ - Database: "schemabot", - Dialect: "mysql", - Version: "v1.2.3", - Converged: true, + Database: "schemabot", + Host: "db-1.example", + Dialect: "mysql", + SchemaSource: "the schema embedded in v1.2.3", + Version: "v1.2.3", + Converged: true, }) - assert.Equal(t, "schemabot (mysql) is converged, against the schema embedded in v1.2.3.", converged) + assert.Equal(t, "schemabot on db-1.example (mysql) is converged, against the schema embedded in v1.2.3.", converged) outstanding := storageSchemaHeadline(&apitypes.StorageSchemaReport{ - Database: "schemabot", - Dialect: "postgres", - Deployment: "west", - Environment: "production", - Version: "v1.2.3", - Outstanding: []apitypes.StorageSchemaStatement{{Table: "applies"}, {Table: "checks"}}, - Destructive: []apitypes.StorageSchemaStatement{{Table: "stale_state"}}, - Manual: []apitypes.StorageSchemaStatement{{Table: "plans"}}, + Database: "schemabot", + Dialect: "postgres", + Deployment: "west", + Environment: "production", + SchemaSource: "the schema files of release v1.4.0", + Version: "v1.2.3", + Outstanding: []apitypes.StorageSchemaStatement{{Table: "applies"}, {Table: "checks"}}, + Destructive: []apitypes.StorageSchemaStatement{{Table: "stale_state"}}, + Manual: []apitypes.StorageSchemaStatement{{Table: "plans"}}, }) assert.Equal(t, - "schemabot (postgres) on deployment west in production needs 4 statements: "+ - "2 outstanding, 1 destructive, 1 needing manual remediation, against the schema embedded in v1.2.3.", - outstanding) + "schemabot (postgres), deployment west in production needs 4 statements: "+ + "2 outstanding, 1 destructive, 1 needing manual remediation, against the schema files of release v1.4.0.", + outstanding, + "the answering binary's version never displaces the schema the diff actually used") assert.Equal(t, "the storage database needs 1 statement: 1 outstanding.", storageSchemaHeadline(&apitypes.StorageSchemaReport{ @@ -304,30 +309,50 @@ func TestRenderStorageSchemaReport_ManualSection(t *testing.T) { // direct form never echoes the DSN back — it may carry a password, and the // operator already has it. func TestStorageSchemaDiffHints(t *testing.T) { - local := storageSchemaDiffHints(&StorageDiffCmd{}) + report := &apitypes.StorageSchemaReport{Database: "schemabot", Dialect: "mysql"} + + local := storageSchemaDiffHints(&StorageDiffCmd{}, report) require.Len(t, local, 1) assert.Contains(t, local[0], "storage apply") assert.NotContains(t, local[0], "--deployment") deployment := storageSchemaDiffHints(&StorageDiffCmd{ storageSchemaTargetFlags: storageSchemaTargetFlags{Deployment: "west", Environment: "production"}, - }) + }, report) require.Len(t, deployment, 1) assert.Contains(t, deployment[0], "storage apply --deployment west -e production") const secret = "root:hunter2@tcp(db.example:3306)/schemabot" - dsn := storageSchemaDiffHints(&StorageDiffCmd{storageSchemaTargetFlags: storageSchemaTargetFlags{DSN: secret}}) + dsn := storageSchemaDiffHints(&StorageDiffCmd{storageSchemaTargetFlags: storageSchemaTargetFlags{DSN: secret}}, report) require.Len(t, dsn, 1) assert.Contains(t, dsn[0], "--dsn ") assert.NotContains(t, dsn[0], "hunter2") config := storageSchemaDiffHints(&StorageDiffCmd{ storageSchemaTargetFlags: storageSchemaTargetFlags{Config: "/etc/schemabot/config.yaml"}, - }) + }, report) require.Len(t, config, 1) assert.Contains(t, config[0], "--config /etc/schemabot/config.yaml") } +// A diff against a release the target is not running must not offer +// `storage apply` as the next step: an apply converges the schema of the binary +// that answers, which is not the schema the report describes. The hint names +// the two ways to converge the release instead. +func TestStorageSchemaDiffHints_NamedRelease(t *testing.T) { + hints := storageSchemaDiffHints( + &StorageDiffCmd{storageSchemaSourceFlags: storageSchemaSourceFlags{Release: "v1.4.0"}}, + &apitypes.StorageSchemaReport{ + Database: "schemabot", + Dialect: "mysql", + SchemaSource: "the schema files of release v1.4.0", + }) + require.Len(t, hints, 1) + assert.Contains(t, hints[0], "the schema files of release v1.4.0") + assert.Contains(t, hints[0], "run that release's binary") + assert.NotContains(t, hints[0], "storage apply") +} + // A clean convergence states that nothing is left, rather than leaving an // operator to read the absence of a section as good news. func TestRenderStorageSchemaConvergence_Clean(t *testing.T) { @@ -376,15 +401,20 @@ func TestRenderStorageSchemaConvergence_LeftBehind(t *testing.T) { assert.NotContains(t, out.String(), "is converged") } -// A deployment's report is labelled with the deployment it came from, so an -// operator reading a control plane's answer can tell whose storage it describes. +// A report is labelled with the database, the server it is on, and the +// deployment it came from, so an operator reading a control plane's answer can +// tell which database it describes without re-deriving it from a DSN or a +// deployment name. func TestStorageSchemaDatabaseLabel(t *testing.T) { assert.Equal(t, "schemabot (mysql)", storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{ Database: "schemabot", Dialect: "mysql", + }), "a server that reports no name of its own is left out rather than guessed at") + assert.Equal(t, "schemabot on db-1.example (mysql)", storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{ + Database: "schemabot", Host: "db-1.example", Dialect: "mysql", })) - assert.Equal(t, "schemabot (postgres) on deployment west in production", + assert.Equal(t, "schemabot on 10.0.0.7 (postgres), deployment west in production", storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{ - Database: "schemabot", Dialect: "postgres", Deployment: "west", Environment: "production", + Database: "schemabot", Host: "10.0.0.7", Dialect: "postgres", Deployment: "west", Environment: "production", })) assert.Equal(t, "the storage database", storageSchemaDatabaseLabel(&apitypes.StorageSchemaReport{}), "a report with no database name still reads as a sentence") diff --git a/pkg/proto/tern.proto b/pkg/proto/tern.proto index 0a5805d05..d5b73cea5 100644 --- a/pkg/proto/tern.proto +++ b/pkg/proto/tern.proto @@ -966,6 +966,13 @@ message StartResponse { // on its own storage database. It deliberately carries no target: the instance // answers for the storage it uses, and nothing a caller sends can point it at // a different database. +// +// The live side of the diff is therefore fixed. The desired side is not: a +// caller deploying a later release asks what this storage needs in order to +// match that release, which the serving binary cannot answer from files it does +// not have. Sending the files is the only way to ask the question, and it is +// safe to ask because a diff executes nothing. The apply RPC carries no such +// field, so a convergence can only ever run the serving binary's own schema. message StorageSchemaDiffRequest { // AllowDestructive reports the destructive statements as allowed rather than // refused, matching what an apply carrying the same flag would run. It does @@ -973,6 +980,16 @@ message StorageSchemaDiffRequest { // statements are classified destructive, only whether the report says they // would run. bool allow_destructive = 1; + // SchemaFiles is the desired schema, as file name → file contents (for + // example "applies.sql" → "CREATE TABLE ..."). Empty diffs against the + // serving binary's own embedded files. The serving instance's dialect selects + // how they are read, exactly as it does for the embedded ones. + map schema_files = 2; + // SchemaSource says where schema_files came from, in words, for the report to + // carry back. It is required with schema_files and never inferred: a report + // that does not say whose schema it was diffed against is one a caller can + // misattribute to the serving binary. + string schema_source = 3; } // StorageSchemaStatement is one outstanding storage-schema statement. @@ -998,8 +1015,8 @@ message StorageSchemaReport { string dialect = 1; // The live database the diff read, as the server reports it. string database = 2; - // SchemaBot version of the binary whose embedded schema produced the diff. - // Reported for attribution only; the diff itself is computed from the files. + // SchemaBot version of the binary that answered. Reported for attribution + // only; the diff itself is computed from schema files, never from a version. string version = 3; // Statements that converge the schema and run automatically, in the order // the convergence would run them. @@ -1013,6 +1030,15 @@ message StorageSchemaReport { // remediation. Any entry aborts the whole convergence before a single // statement executes. repeated StorageSchemaStatement manual = 7; + // The database server as it names itself, so a report names the machine it + // read and not only the database on it. Empty when the server does not + // report one. + string host = 8; + // Where the desired side of the diff came from, in words — the serving + // binary's embedded files, a directory, or a release. A report always says + // this, because the same live database yields different answers against + // different releases. + string schema_source = 9; } // StorageSchemaDiffResponse carries the outstanding storage DDL. diff --git a/pkg/proto/ternv1/tern.pb.go b/pkg/proto/ternv1/tern.pb.go index c929a7b8c..af2a12acc 100644 --- a/pkg/proto/ternv1/tern.pb.go +++ b/pkg/proto/ternv1/tern.pb.go @@ -3924,6 +3924,13 @@ func (x *StartResponse) GetSkippedCount() int64 { // on its own storage database. It deliberately carries no target: the instance // answers for the storage it uses, and nothing a caller sends can point it at // a different database. +// +// The live side of the diff is therefore fixed. The desired side is not: a +// caller deploying a later release asks what this storage needs in order to +// match that release, which the serving binary cannot answer from files it does +// not have. Sending the files is the only way to ask the question, and it is +// safe to ask because a diff executes nothing. The apply RPC carries no such +// field, so a convergence can only ever run the serving binary's own schema. type StorageSchemaDiffRequest struct { state protoimpl.MessageState `protogen:"open.v1"` // AllowDestructive reports the destructive statements as allowed rather than @@ -3932,8 +3939,18 @@ type StorageSchemaDiffRequest struct { // statements are classified destructive, only whether the report says they // would run. AllowDestructive bool `protobuf:"varint,1,opt,name=allow_destructive,json=allowDestructive,proto3" json:"allow_destructive,omitempty"` - unknownFields protoimpl.UnknownFields - sizeCache protoimpl.SizeCache + // SchemaFiles is the desired schema, as file name → file contents (for + // example "applies.sql" → "CREATE TABLE ..."). Empty diffs against the + // serving binary's own embedded files. The serving instance's dialect selects + // how they are read, exactly as it does for the embedded ones. + SchemaFiles map[string]string `protobuf:"bytes,2,rep,name=schema_files,json=schemaFiles,proto3" json:"schema_files,omitempty" protobuf_key:"bytes,1,opt,name=key" protobuf_val:"bytes,2,opt,name=value"` + // SchemaSource says where schema_files came from, in words, for the report to + // carry back. It is required with schema_files and never inferred: a report + // that does not say whose schema it was diffed against is one a caller can + // misattribute to the serving binary. + SchemaSource string `protobuf:"bytes,3,opt,name=schema_source,json=schemaSource,proto3" json:"schema_source,omitempty"` + unknownFields protoimpl.UnknownFields + sizeCache protoimpl.SizeCache } func (x *StorageSchemaDiffRequest) Reset() { @@ -3973,6 +3990,20 @@ func (x *StorageSchemaDiffRequest) GetAllowDestructive() bool { return false } +func (x *StorageSchemaDiffRequest) GetSchemaFiles() map[string]string { + if x != nil { + return x.SchemaFiles + } + return nil +} + +func (x *StorageSchemaDiffRequest) GetSchemaSource() string { + if x != nil { + return x.SchemaSource + } + return "" +} + // StorageSchemaStatement is one outstanding storage-schema statement. type StorageSchemaStatement struct { state protoimpl.MessageState `protogen:"open.v1"` @@ -4058,8 +4089,8 @@ type StorageSchemaReport struct { Dialect string `protobuf:"bytes,1,opt,name=dialect,proto3" json:"dialect,omitempty"` // The live database the diff read, as the server reports it. Database string `protobuf:"bytes,2,opt,name=database,proto3" json:"database,omitempty"` - // SchemaBot version of the binary whose embedded schema produced the diff. - // Reported for attribution only; the diff itself is computed from the files. + // SchemaBot version of the binary that answered. Reported for attribution + // only; the diff itself is computed from schema files, never from a version. Version string `protobuf:"bytes,3,opt,name=version,proto3" json:"version,omitempty"` // Statements that converge the schema and run automatically, in the order // the convergence would run them. @@ -4072,7 +4103,16 @@ type StorageSchemaReport struct { // Changes that cannot run automatically, each naming the situation and its // remediation. Any entry aborts the whole convergence before a single // statement executes. - Manual []*StorageSchemaStatement `protobuf:"bytes,7,rep,name=manual,proto3" json:"manual,omitempty"` + Manual []*StorageSchemaStatement `protobuf:"bytes,7,rep,name=manual,proto3" json:"manual,omitempty"` + // The database server as it names itself, so a report names the machine it + // read and not only the database on it. Empty when the server does not + // report one. + Host string `protobuf:"bytes,8,opt,name=host,proto3" json:"host,omitempty"` + // Where the desired side of the diff came from, in words — the serving + // binary's embedded files, a directory, or a release. A report always says + // this, because the same live database yields different answers against + // different releases. + SchemaSource string `protobuf:"bytes,9,opt,name=schema_source,json=schemaSource,proto3" json:"schema_source,omitempty"` unknownFields protoimpl.UnknownFields sizeCache protoimpl.SizeCache } @@ -4156,6 +4196,20 @@ func (x *StorageSchemaReport) GetManual() []*StorageSchemaStatement { return nil } +func (x *StorageSchemaReport) GetHost() string { + if x != nil { + return x.Host + } + return "" +} + +func (x *StorageSchemaReport) GetSchemaSource() string { + if x != nil { + return x.SchemaSource + } + return "" +} + // StorageSchemaDiffResponse carries the outstanding storage DDL. type StorageSchemaDiffResponse struct { state protoimpl.MessageState `protogen:"open.v1"` @@ -4665,14 +4719,19 @@ const file_tern_proto_rawDesc = "" + "\baccepted\x18\x01 \x01(\bR\baccepted\x12#\n" + "\rerror_message\x18\x02 \x01(\tR\ferrorMessage\x12#\n" + "\rstarted_count\x18\x03 \x01(\x03R\fstartedCount\x12#\n" + - "\rskipped_count\x18\x04 \x01(\x03R\fskippedCount\"G\n" + + "\rskipped_count\x18\x04 \x01(\x03R\fskippedCount\"\x83\x02\n" + "\x18StorageSchemaDiffRequest\x12+\n" + - "\x11allow_destructive\x18\x01 \x01(\bR\x10allowDestructive\"v\n" + + "\x11allow_destructive\x18\x01 \x01(\bR\x10allowDestructive\x12U\n" + + "\fschema_files\x18\x02 \x03(\v22.tern.v1.StorageSchemaDiffRequest.SchemaFilesEntryR\vschemaFiles\x12#\n" + + "\rschema_source\x18\x03 \x01(\tR\fschemaSource\x1a>\n" + + "\x10SchemaFilesEntry\x12\x10\n" + + "\x03key\x18\x01 \x01(\tR\x03key\x12\x14\n" + + "\x05value\x18\x02 \x01(\tR\x05value:\x028\x01\"v\n" + "\x16StorageSchemaStatement\x12\x14\n" + "\x05table\x18\x01 \x01(\tR\x05table\x12\x1c\n" + "\toperation\x18\x02 \x01(\tR\toperation\x12\x10\n" + "\x03ddl\x18\x03 \x01(\tR\x03ddl\x12\x16\n" + - "\x06reason\x18\x04 \x01(\tR\x06reason\"\xd5\x02\n" + + "\x06reason\x18\x04 \x01(\tR\x06reason\"\x8e\x03\n" + "\x13StorageSchemaReport\x12\x18\n" + "\adialect\x18\x01 \x01(\tR\adialect\x12\x1a\n" + "\bdatabase\x18\x02 \x01(\tR\bdatabase\x12\x18\n" + @@ -4680,7 +4739,9 @@ const file_tern_proto_rawDesc = "" + "\voutstanding\x18\x04 \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\voutstanding\x12A\n" + "\vdestructive\x18\x05 \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\vdestructive\x12/\n" + "\x13destructive_allowed\x18\x06 \x01(\bR\x12destructiveAllowed\x127\n" + - "\x06manual\x18\a \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\x06manual\"Q\n" + + "\x06manual\x18\a \x03(\v2\x1f.tern.v1.StorageSchemaStatementR\x06manual\x12\x12\n" + + "\x04host\x18\b \x01(\tR\x04host\x12#\n" + + "\rschema_source\x18\t \x01(\tR\fschemaSource\"Q\n" + "\x19StorageSchemaDiffResponse\x124\n" + "\x06report\x18\x01 \x01(\v2\x1c.tern.v1.StorageSchemaReportR\x06report\"`\n" + "\x19StorageSchemaApplyRequest\x12+\n" + @@ -4770,7 +4831,7 @@ func file_tern_proto_rawDescGZIP() []byte { } var file_tern_proto_enumTypes = make([]protoimpl.EnumInfo, 4) -var file_tern_proto_msgTypes = make([]protoimpl.MessageInfo, 61) +var file_tern_proto_msgTypes = make([]protoimpl.MessageInfo, 62) var file_tern_proto_goTypes = []any{ (Engine)(0), // 0: tern.v1.Engine (State)(0), // 1: tern.v1.State @@ -4837,6 +4898,7 @@ var file_tern_proto_goTypes = []any{ nil, // 62: tern.v1.ApplyRequest.OptionsEntry nil, // 63: tern.v1.ApplyRequest.SchemaFilesEntry nil, // 64: tern.v1.ProgressResponse.MetadataEntry + nil, // 65: tern.v1.StorageSchemaDiffRequest.SchemaFilesEntry } var file_tern_proto_depIdxs = []int32{ 53, // 0: tern.v1.SchemaFiles.files:type_name -> tern.v1.SchemaFiles.FilesEntry @@ -4878,51 +4940,52 @@ var file_tern_proto_depIdxs = []int32{ 30, // 36: tern.v1.ProgressResponse.tables:type_name -> tern.v1.TableProgress 64, // 37: tern.v1.ProgressResponse.metadata:type_name -> tern.v1.ProgressResponse.MetadataEntry 31, // 38: tern.v1.ProgressResponse.settled_control_requests:type_name -> tern.v1.SettledControlRequest - 48, // 39: tern.v1.StorageSchemaReport.outstanding:type_name -> tern.v1.StorageSchemaStatement - 48, // 40: tern.v1.StorageSchemaReport.destructive:type_name -> tern.v1.StorageSchemaStatement - 48, // 41: tern.v1.StorageSchemaReport.manual:type_name -> tern.v1.StorageSchemaStatement - 49, // 42: tern.v1.StorageSchemaDiffResponse.report:type_name -> tern.v1.StorageSchemaReport - 49, // 43: tern.v1.StorageSchemaApplyResponse.planned:type_name -> tern.v1.StorageSchemaReport - 49, // 44: tern.v1.StorageSchemaApplyResponse.remaining:type_name -> tern.v1.StorageSchemaReport - 8, // 45: tern.v1.PulledNamespace.TableCatalogEntry.value:type_name -> tern.v1.TableCatalog - 6, // 46: tern.v1.PullSchemaResponse.NamespacesEntry.value:type_name -> tern.v1.PulledNamespace - 4, // 47: tern.v1.PlanRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles - 4, // 48: tern.v1.ApplyRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles - 5, // 49: tern.v1.Tern.PullSchema:input_type -> tern.v1.PullSchemaRequest - 13, // 50: tern.v1.Tern.Plan:input_type -> tern.v1.PlanRequest - 13, // 51: tern.v1.Tern.PlanDiff:input_type -> tern.v1.PlanRequest - 22, // 52: tern.v1.Tern.Apply:input_type -> tern.v1.ApplyRequest - 25, // 53: tern.v1.Tern.Progress:input_type -> tern.v1.ProgressRequest - 26, // 54: tern.v1.Tern.Logs:input_type -> tern.v1.LogsRequest - 33, // 55: tern.v1.Tern.Cutover:input_type -> tern.v1.CutoverRequest - 35, // 56: tern.v1.Tern.Revert:input_type -> tern.v1.RevertRequest - 37, // 57: tern.v1.Tern.SkipRevert:input_type -> tern.v1.SkipRevertRequest - 39, // 58: tern.v1.Tern.Health:input_type -> tern.v1.HealthRequest - 41, // 59: tern.v1.Tern.Stop:input_type -> tern.v1.StopRequest - 43, // 60: tern.v1.Tern.Cancel:input_type -> tern.v1.CancelRequest - 45, // 61: tern.v1.Tern.Start:input_type -> tern.v1.StartRequest - 47, // 62: tern.v1.Tern.StorageSchemaDiff:input_type -> tern.v1.StorageSchemaDiffRequest - 51, // 63: tern.v1.Tern.StorageSchemaApply:input_type -> tern.v1.StorageSchemaApplyRequest - 12, // 64: tern.v1.Tern.PullSchema:output_type -> tern.v1.PullSchemaResponse - 20, // 65: tern.v1.Tern.Plan:output_type -> tern.v1.PlanResponse - 21, // 66: tern.v1.Tern.PlanDiff:output_type -> tern.v1.PlanDiffResponse - 24, // 67: tern.v1.Tern.Apply:output_type -> tern.v1.ApplyResponse - 32, // 68: tern.v1.Tern.Progress:output_type -> tern.v1.ProgressResponse - 28, // 69: tern.v1.Tern.Logs:output_type -> tern.v1.LogsResponse - 34, // 70: tern.v1.Tern.Cutover:output_type -> tern.v1.CutoverResponse - 36, // 71: tern.v1.Tern.Revert:output_type -> tern.v1.RevertResponse - 38, // 72: tern.v1.Tern.SkipRevert:output_type -> tern.v1.SkipRevertResponse - 40, // 73: tern.v1.Tern.Health:output_type -> tern.v1.HealthResponse - 42, // 74: tern.v1.Tern.Stop:output_type -> tern.v1.StopResponse - 44, // 75: tern.v1.Tern.Cancel:output_type -> tern.v1.CancelResponse - 46, // 76: tern.v1.Tern.Start:output_type -> tern.v1.StartResponse - 50, // 77: tern.v1.Tern.StorageSchemaDiff:output_type -> tern.v1.StorageSchemaDiffResponse - 52, // 78: tern.v1.Tern.StorageSchemaApply:output_type -> tern.v1.StorageSchemaApplyResponse - 64, // [64:79] is the sub-list for method output_type - 49, // [49:64] is the sub-list for method input_type - 49, // [49:49] is the sub-list for extension type_name - 49, // [49:49] is the sub-list for extension extendee - 0, // [0:49] is the sub-list for field type_name + 65, // 39: tern.v1.StorageSchemaDiffRequest.schema_files:type_name -> tern.v1.StorageSchemaDiffRequest.SchemaFilesEntry + 48, // 40: tern.v1.StorageSchemaReport.outstanding:type_name -> tern.v1.StorageSchemaStatement + 48, // 41: tern.v1.StorageSchemaReport.destructive:type_name -> tern.v1.StorageSchemaStatement + 48, // 42: tern.v1.StorageSchemaReport.manual:type_name -> tern.v1.StorageSchemaStatement + 49, // 43: tern.v1.StorageSchemaDiffResponse.report:type_name -> tern.v1.StorageSchemaReport + 49, // 44: tern.v1.StorageSchemaApplyResponse.planned:type_name -> tern.v1.StorageSchemaReport + 49, // 45: tern.v1.StorageSchemaApplyResponse.remaining:type_name -> tern.v1.StorageSchemaReport + 8, // 46: tern.v1.PulledNamespace.TableCatalogEntry.value:type_name -> tern.v1.TableCatalog + 6, // 47: tern.v1.PullSchemaResponse.NamespacesEntry.value:type_name -> tern.v1.PulledNamespace + 4, // 48: tern.v1.PlanRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles + 4, // 49: tern.v1.ApplyRequest.SchemaFilesEntry.value:type_name -> tern.v1.SchemaFiles + 5, // 50: tern.v1.Tern.PullSchema:input_type -> tern.v1.PullSchemaRequest + 13, // 51: tern.v1.Tern.Plan:input_type -> tern.v1.PlanRequest + 13, // 52: tern.v1.Tern.PlanDiff:input_type -> tern.v1.PlanRequest + 22, // 53: tern.v1.Tern.Apply:input_type -> tern.v1.ApplyRequest + 25, // 54: tern.v1.Tern.Progress:input_type -> tern.v1.ProgressRequest + 26, // 55: tern.v1.Tern.Logs:input_type -> tern.v1.LogsRequest + 33, // 56: tern.v1.Tern.Cutover:input_type -> tern.v1.CutoverRequest + 35, // 57: tern.v1.Tern.Revert:input_type -> tern.v1.RevertRequest + 37, // 58: tern.v1.Tern.SkipRevert:input_type -> tern.v1.SkipRevertRequest + 39, // 59: tern.v1.Tern.Health:input_type -> tern.v1.HealthRequest + 41, // 60: tern.v1.Tern.Stop:input_type -> tern.v1.StopRequest + 43, // 61: tern.v1.Tern.Cancel:input_type -> tern.v1.CancelRequest + 45, // 62: tern.v1.Tern.Start:input_type -> tern.v1.StartRequest + 47, // 63: tern.v1.Tern.StorageSchemaDiff:input_type -> tern.v1.StorageSchemaDiffRequest + 51, // 64: tern.v1.Tern.StorageSchemaApply:input_type -> tern.v1.StorageSchemaApplyRequest + 12, // 65: tern.v1.Tern.PullSchema:output_type -> tern.v1.PullSchemaResponse + 20, // 66: tern.v1.Tern.Plan:output_type -> tern.v1.PlanResponse + 21, // 67: tern.v1.Tern.PlanDiff:output_type -> tern.v1.PlanDiffResponse + 24, // 68: tern.v1.Tern.Apply:output_type -> tern.v1.ApplyResponse + 32, // 69: tern.v1.Tern.Progress:output_type -> tern.v1.ProgressResponse + 28, // 70: tern.v1.Tern.Logs:output_type -> tern.v1.LogsResponse + 34, // 71: tern.v1.Tern.Cutover:output_type -> tern.v1.CutoverResponse + 36, // 72: tern.v1.Tern.Revert:output_type -> tern.v1.RevertResponse + 38, // 73: tern.v1.Tern.SkipRevert:output_type -> tern.v1.SkipRevertResponse + 40, // 74: tern.v1.Tern.Health:output_type -> tern.v1.HealthResponse + 42, // 75: tern.v1.Tern.Stop:output_type -> tern.v1.StopResponse + 44, // 76: tern.v1.Tern.Cancel:output_type -> tern.v1.CancelResponse + 46, // 77: tern.v1.Tern.Start:output_type -> tern.v1.StartResponse + 50, // 78: tern.v1.Tern.StorageSchemaDiff:output_type -> tern.v1.StorageSchemaDiffResponse + 52, // 79: tern.v1.Tern.StorageSchemaApply:output_type -> tern.v1.StorageSchemaApplyResponse + 65, // [65:80] is the sub-list for method output_type + 50, // [50:65] is the sub-list for method input_type + 50, // [50:50] is the sub-list for extension type_name + 50, // [50:50] is the sub-list for extension extendee + 0, // [0:50] is the sub-list for field type_name } func init() { file_tern_proto_init() } @@ -4938,7 +5001,7 @@ func file_tern_proto_init() { GoPackagePath: reflect.TypeOf(x{}).PkgPath(), RawDescriptor: unsafe.Slice(unsafe.StringData(file_tern_proto_rawDesc), len(file_tern_proto_rawDesc)), NumEnums: 4, - NumMessages: 61, + NumMessages: 62, NumExtensions: 0, NumServices: 1, }, diff --git a/pkg/serve/storage_schema.go b/pkg/serve/storage_schema.go index 518a9ba52..5964b8fe8 100644 --- a/pkg/serve/storage_schema.go +++ b/pkg/serve/storage_schema.go @@ -19,9 +19,14 @@ import ( // The bindings are what make the answer trustworthy, so they are fixed at // construction and nothing on the wire can move them. There is no target in // the request, so no caller — not even the control plane — can point this at a -// different database. The one thing a caller may influence is whether -// destructive statements run, and that only ever widens what the local config -// already allows (see effectiveAllowDestructive). +// different database. +// +// Two things a caller may influence, and neither reaches the database a +// convergence writes to. Whether destructive statements run only ever widens +// what the local config already allows (see effectiveAllowDestructive). A +// desired schema sent with a diff replaces the files the comparison reads, and +// is accepted only there: the convergence RPC carries no schema, so an apply +// always runs this binary's own (see desiredSchema). type storageSchemaAdapter struct { // resolveDSN re-resolves the storage DSN per call rather than capturing a // string, so a credential rotated since startup is picked up the same way @@ -62,17 +67,46 @@ func (a *storageSchemaAdapter) StorageSchemaDiff(ctx context.Context, req *ternv if err != nil { return nil, err } + desired, err := a.desiredSchema(req) + if err != nil { + return nil, err + } diffCtx, cancel := context.WithTimeout(ctx, api.StorageSchemaDiffTimeout) defer cancel() - report, err := api.DiffStorageSchema(diffCtx, dsn, a.logger, opts...) + report, err := api.DiffStorageSchema(diffCtx, dsn, desired, a.logger, opts...) if err != nil { - return nil, fmt.Errorf("diff storage schema (dialect %s): %w", a.dialect, err) + return nil, fmt.Errorf("diff storage schema (dialect %s) against %s: %w", a.dialect, desired.Description, err) } - report.Version = a.version + report.AttributeTo(a.version) return &ternv1.StorageSchemaDiffResponse{Report: api.StorageSchemaReportProto(report)}, nil } +// desiredSchema resolves the schema the diff compares the live database +// against: the files the caller sent, or this binary's own when it sent none. +// +// A caller-supplied schema is accepted here and nowhere else. A diff executes +// nothing, so answering "what would this database need in order to match that +// release" is a read however the release's files arrived; the convergence RPC +// has no such field to send, so the schema a convergence runs is always this +// binary's (AV-9). +func (a *storageSchemaAdapter) desiredSchema(req *ternv1.StorageSchemaDiffRequest) (*api.StorageSchemaSource, error) { + files := req.GetSchemaFiles() + if len(files) == 0 { + return api.EmbeddedStorageSchema(a.version), nil + } + desired, err := api.StorageSchemaFromFiles(req.GetSchemaSource(), files) + if err != nil { + return nil, fmt.Errorf("read the supplied storage schema: %w", err) + } + a.logger.Info("diffing storage schema against a supplied schema", + "dialect", a.dialect, + "schema_source", desired.Description, + "schema_file_count", len(files), + ) + return desired, nil +} + func (a *storageSchemaAdapter) StorageSchemaApply(ctx context.Context, req *ternv1.StorageSchemaApplyRequest) (*ternv1.StorageSchemaApplyResponse, error) { dsn, opts, err := a.target(req.GetAllowDestructive()) if err != nil { @@ -86,8 +120,8 @@ func (a *storageSchemaAdapter) StorageSchemaApply(ctx context.Context, req *tern if err != nil { return nil, fmt.Errorf("converge storage schema (dialect %s): %w", a.dialect, err) } - planned.Version = a.version - remaining.Version = a.version + planned.AttributeTo(a.version) + remaining.AttributeTo(a.version) a.logger.InfoContext(ctx, "storage schema convergence answered", "dialect", a.dialect, "database", remaining.Database, diff --git a/pkg/serve/storage_schema_test.go b/pkg/serve/storage_schema_test.go index 6287d957b..e46420e98 100644 --- a/pkg/serve/storage_schema_test.go +++ b/pkg/serve/storage_schema_test.go @@ -102,6 +102,47 @@ func TestStorageSchemaAdapter_RefusesWithoutStorageDSN(t *testing.T) { require.Error(t, err, "a convergence must not proceed without a database to converge") } +// A diff with no schema on it is answered against this binary's own embedded +// schema, attributed to this binary's version — which is what a boot would +// converge to, and the answer an operator gets when they ask nothing else. +func TestStorageSchemaAdapter_DesiredSchemaDefaultsToThisBinary(t *testing.T) { + adapter := &storageSchemaAdapter{version: "v1.2.3", logger: slog.New(slog.DiscardHandler)} + + desired, err := adapter.desiredSchema(&ternv1.StorageSchemaDiffRequest{}) + require.NoError(t, err) + assert.Equal(t, "the schema embedded in v1.2.3", desired.Description) + assert.Empty(t, desired.Files, "the answering binary's own files are read here, not sent to it") +} + +// A schema on the request replaces the files the comparison reads, so an +// operator can ask what this storage needs in order to match a release this +// binary is not running. The attribution is the caller's, because the answer +// came from the caller's files. +func TestStorageSchemaAdapter_DesiredSchemaAcceptsASuppliedSchema(t *testing.T) { + adapter := &storageSchemaAdapter{version: "v1.2.3", logger: slog.New(slog.DiscardHandler)} + + desired, err := adapter.desiredSchema(&ternv1.StorageSchemaDiffRequest{ + SchemaSource: "the schema files of release v1.4.0", + SchemaFiles: map[string]string{"applies.sql": "CREATE TABLE `applies` (`id` BIGINT UNSIGNED PRIMARY KEY)"}, + }) + require.NoError(t, err) + assert.Equal(t, "the schema files of release v1.4.0", desired.Description) + assert.Len(t, desired.Files, 1) +} + +// An unusable supplied schema is refused before anything reads a database. A +// file set that cannot be read as one .sql file per table would otherwise diff +// as a storage database full of surplus tables. +func TestStorageSchemaAdapter_DesiredSchemaRefusesAnUnusableSchema(t *testing.T) { + adapter := &storageSchemaAdapter{version: "v1.2.3", logger: slog.New(slog.DiscardHandler)} + + _, err := adapter.desiredSchema(&ternv1.StorageSchemaDiffRequest{ + SchemaFiles: map[string]string{"applies.sql": "CREATE TABLE `applies` (`id` BIGINT UNSIGNED PRIMARY KEY)"}, + }) + require.Error(t, err, "files with no source leave the report unable to attribute its answer") + assert.Contains(t, err.Error(), "needs a description") +} + // A DSN the server cannot resolve — an unreadable credential file, say — // surfaces as an error naming what was being resolved, not as a report about // an empty database. From 077582f7b6c8f3e56a5ca569ab6f151aaf4d830b Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Fri, 11 Sep 2026 14:43:43 -0400 Subject: [PATCH 4/5] feat(cli): require the diff to name the schema it compares against MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit `storage diff` had no flag for the common case: with none given it compared against the schema embedded in whichever binary answered. That default is the one thing about this command an operator has to know, and it was the one thing the command line did not say. Name it — `--embedded` for the answering binary's own schema, `--release ` or `--schema-dir ` for another release's — or the command refuses before it reads anything. `storage apply` accepts `--embedded` and still refuses the other two: a convergence runs the schema of the binary running it (AV-9), so there is nothing else for it to converge. Adds docs/storage-schema.md, the operator guide: how a storage schema change reaches the database through either path — every boot's EnsureSchema and an operator's `storage apply`, which are the same code — what each dialect converges by itself, what it refuses and why, the deploy and crashloop playbooks, and the indexes worth pre-creating on a long-lived database. The operator narrative moves out of the configuration reference, which keeps the settings and points at the guide. Co-Authored-By: Claude Opus 5 --- README.md | 1 + docs/.toc-manifest | 1 + docs/cli.md | 1 + docs/configuration.md | 397 +------------ docs/release.md | 4 +- docs/storage-schema.md | 541 ++++++++++++++++++ pkg/cmd/commands/storage_schema.go | 25 +- pkg/cmd/commands/storage_schema_source.go | 62 +- .../commands/storage_schema_source_test.go | 41 +- pkg/cmd/commands/storage_schema_test.go | 17 +- 10 files changed, 676 insertions(+), 414 deletions(-) create mode 100644 docs/storage-schema.md diff --git a/README.md b/README.md index d1726cff8..7902eedf2 100644 --- a/README.md +++ b/README.md @@ -116,6 +116,7 @@ Guides and reference: - [Engines](./docs/engines.md): See how changes run on your database engine - [PostgreSQL](./docs/postgresql.md): Find out what’s supported today - [Configuration](./docs/configuration.md): Set things up for your environment +- [Storage schema](./docs/storage-schema.md): Keep SchemaBot’s own bookkeeping database converged across deploys - [Authentication](./docs/auth.md): Choose who can read and change your databases - [AI agents](./docs/ai-agents.md): Set clear boundaries for your assistants - [Safety invariants](./docs/invariants.md): Understand the guardrails behind each change diff --git a/docs/.toc-manifest b/docs/.toc-manifest index 8b87081ff..28c7c7d68 100644 --- a/docs/.toc-manifest +++ b/docs/.toc-manifest @@ -23,6 +23,7 @@ docs/pre-merge-workflow.md docs/schema-intelligence.md docs/spirit_progress.md docs/storage-outage-behavior.md +docs/storage-schema.md docs/strata-engine.md docs/target-credential-self-heal.md docs/throttle.md diff --git a/docs/cli.md b/docs/cli.md index 6c8e3bb2f..f3486b7bf 100644 --- a/docs/cli.md +++ b/docs/cli.md @@ -13,6 +13,7 @@ to see what is changing across your fleet. | Manage a PlanetScale deploy request | [Deploy, follow shards, and control cutover](#manage-planetscale-deploy-requests) | | Understand a merge check that will not clear | [Explain a blocked check](#explain-a-blocked-check) | | Build an integration | [Use structured output](#use-the-cli-from-scripts-and-agents) | +| Converge SchemaBot's own storage schema | [Storage schema guide](storage-schema.md) | The examples use a MySQL database named `shop` in `staging`. Substitute a database and environment from your server's inventory. The local diff --git a/docs/configuration.md b/docs/configuration.md index 6cb0653b4..91644382f 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -660,7 +660,7 @@ a MySQL-format DSN and is only supported with the `mysql` storage dialect; combining it with `postgres` fails config validation. (This restriction is specific to the storage database — `dsn_from` on a `target_resolver` target supports both `mysql` and `postgres`.) See -[Storage Schema Changes](#storage-schema-changes) for how schema +[docs/storage-schema.md](storage-schema.md) for how schema bootstrapping differs between the two dialects. ### PostgreSQL storage needs a session-per-connection endpoint @@ -900,138 +900,27 @@ Helm chart, mount the certificate secret with `extraVolumes` / ## Storage Schema Changes SchemaBot's internal storage schema is self-bootstrapping: on every startup, -`EnsureSchema` converges the live storage database against the embedded schema -files before the server accepts traffic. How far that convergence goes depends -on the storage dialect: - -- **MySQL** diffs the embedded schema files against the live database and - applies whatever DDL is needed (via Spirit) — new tables, new columns, and - index changes all converge automatically. That convergence is bounded by a - hard five-minute startup budget, and an index added to an existing table - runs as Spirit online DDL — a table copy, not an in-place build — so its - cost grows with the table's row count. On a deployment whose storage - tables carry a long history, create a newly declared index by hand before - rolling out: the startup diff then finds nothing to do, instead of copying - the table inside the budget on every pod. -- **PostgreSQL** automatically creates missing tables, columns, and standalone - indexes. It discovers drift before taking the bootstrap advisory lock, then - re-checks and applies each table's changes transactionally under that lock. - A missing column converges automatically only when the `ADD COLUMN` is - metadata-only. A missing `NOT NULL` column without a `DEFAULT`, a generated - or identity column, a `UNIQUE` column, a `REFERENCES` column with a - `DEFAULT`, or a column with a constraint shape not explicitly classified as - safe fails startup with instructions for manual remediation: generated and - identity columns rewrite the populated table, `UNIQUE` builds a unique - index over it, and a foreign key with a `DEFAULT` validates every existing - row against the referenced table — all under an exclusive lock whose hold - time the startup lock timeout does not bound. Startup also fails when - additive DDL cannot be parsed or executed, or when re-verification finds - unresolved drift. - - A live index only counts as present when PostgreSQL reports it valid. - PostgreSQL marks an index invalid both while a `CREATE INDEX CONCURRENTLY` - is still building it and after one fails part-way — a unique build that - hits duplicate keys, a cancelled session — and in either case the planner - never uses it. Startup fails closed naming that index rather than reading - it as converged or colliding with it on a fresh `CREATE INDEX`, and reads - `pg_stat_progress_create_index` to say which situation it is. When a build - is in progress — the expected state while an operator pre-creates an index - ahead of a release — the error says so and asks for nothing; the pod - restarts on its backoff and starts cleanly once the build completes. When - no build is visible, the error treats the index as a failed build: remove - the cause first — a unique build keeps failing while duplicate keys - remain — then drop the index so the next startup recreates it, or - `REINDEX INDEX CONCURRENTLY` it by hand. That view only shows other roles' - sessions to a caller with `pg_read_all_stats`, so if the storage role - lacks it and the build runs under a different role, confirm from a - privileged session that no build is running before recovering. A - non-unique index under a name the embedded schema requires to be unique - fails startup the same way. Every such problem across every table is named - in the one startup error, and no DDL runs until all of them are resolved. - - Convergence is additive-only: extra columns and indexes remain in place for - binary rollback, and `allow_destructive_schema_changes` has no effect because - this flow never produces destructive DDL. Column verification remains - presence-only, so type, length, and nullability drift is outside its scope and - is not detected. - - Indexes added to an embedded schema file after a database was bootstrapped - converge on the next startup as plain `CREATE INDEX` statements, each in - its own transaction under the bootstrap advisory lock. A plain - `CREATE INDEX` holds a `SHARE` lock on the table for the full build and - blocks writes to it, and the startup budget is the build's only duration - ceiling, so on a deployment whose storage tables carry a long history, - pre-create the index by hand before rolling out — the startup diff then - finds it present and skips the build. The indexes below are the ones a - long-lived database is most likely to be missing. - - A database bootstrapped before `idx_plans_created_at` was added to `plans` - needs: - - ```sql - CREATE INDEX idx_plans_created_at ON plans (created_at); - ``` - - Without it, listing recent plans is a sequential scan plus a top-N sort, - which gets slower as plan history grows. Likewise, one bootstrapped before - the driver claim ordering on `apply_operations` was indexed needs: - - ```sql - CREATE INDEX idx_apply_operations_created_id ON apply_operations (created_at, id); - ``` - - Without it, every driver claim sorts the full claimable set before taking - one row, which slows claiming as apply history grows. One bootstrapped - before refused applies started naming the schema change holding the - database needs: - - ```sql - CREATE INDEX idx_apply_operations_external_id ON apply_operations (external_id); - ``` - - Without it, resolving the holding change behind a refused apply scans the - full operation history for one remote identifier. On PostgreSQL the lookup - is an optimization, never load-bearing: the refusal still reads correctly, - it just gets slower to record as apply history grows. On MySQL the same - index is not optional — `EnsureSchema` applies it as a startup `ALTER` - under the budget described in the MySQL bullet above, and `apply_operations` - grows with total apply history, so large deployments should pre-create it - there too. And one bootstrapped before the webhook inbox claim ordering on - `webhook_events` was indexed needs: - - ```sql - CREATE INDEX idx_webhook_events_created_id ON webhook_events (created_at, id); - ``` - - Without it, every webhook claim sorts the full claimable inbox before - taking one row, which slows claiming as delivery history grows. On MySQL - the same index arrives as a startup `ALTER` under the budget described in - the MySQL bullet above, and `webhook_events` grows with total delivery - history and has no retention sweep, so pre-create it there before rolling - out: - - ```sql - ALTER TABLE `webhook_events` ADD INDEX `idx_created_id` (`created_at`, `id`); - ``` - -The rest of this section describes the MySQL flow. - -By default, destructive statements in that diff — `DROP TABLE`, or an -`ALTER TABLE` containing `DROP COLUMN` — are refused and skipped. A mixed -`ALTER TABLE` is split: its additive clauses still execute and only the -destructive clauses are refused, except that a clause which cannot run -without a refused clause (the `ADD PRIMARY KEY` half of a primary-key change) -is refused with it. The remaining non-destructive statements still apply and -startup proceeds. This protects -against rolling deploys and rollbacks: a pod running an older binary sees a -newer binary's tables and columns as surplus, and without the gate would drop -them (destroying data the newer pods depend on). Each refused statement is -logged at warn level with the exact DDL, and counted in the +`EnsureSchema` converges the live storage database against the schema files +embedded in the binary, before the server accepts traffic. Operators never apply +storage DDL by hand, and there is no schema directory to point the server at. + +**[docs/storage-schema.md](storage-schema.md) is the operator guide** — how the +convergence works on each dialect, what it refuses, the `storage diff` and +`storage apply` commands, and what to do when a deploy or a pod start does not +converge. Every SchemaBot operator should read at least its first three +sections. This section covers only the settings. + +### `allow_destructive_schema_changes` + +On MySQL, destructive statements in the startup diff — `DROP TABLE`, or an +`ALTER TABLE` containing `DROP COLUMN` — are refused and skipped by default, +because a pod running an older binary sees a newer binary's tables and columns +as surplus and would otherwise drop data the newer pods depend on. Each refusal +is logged at warn level with the exact DDL and counted in the `schemabot.storage_schema.destructive_refusals_total` metric. -To intentionally remove a storage table or column, first make sure every -running pod is on a binary whose embedded schema no longer declares it, then -opt in: +To intentionally remove a storage table or column, first make sure every running +pod is on a binary whose embedded schema no longer declares it, then opt in: ```yaml storage: @@ -1040,248 +929,12 @@ storage: ``` Leave the flag false during normal operation and revert it after the removal -converges. +converges. `--allow-destructive` on `schemabot storage apply` opts in for one +invocation instead; it widens this policy and never narrows it. -### Ask what storage DDL is outstanding - -A deploy that did not converge leaves one question open: which storage DDL is -still outstanding. Two commands answer it, and both read the live storage -database. A release tag never stands in for that read: what a release would -converge to and what the storage actually converged to differ exactly when a -deploy has failed. - -`storage diff` is read-only. It takes no lock and holds no transaction, so it -is safe at any time, including against production during an incident. - -```console -$ schemabot storage diff -schemabot on db-1.example (mysql) needs 3 statements: 3 outstanding, against the schema embedded in v1.2.3. - -Outstanding, and run automatically on the next boot or apply (3): - -ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; -ALTER TABLE `checks` ADD COLUMN `blocked_reason` varchar(64) NOT NULL DEFAULT '' AFTER `state`; -CREATE TABLE `check_gate_audit` ( - `id` BIGINT UNSIGNED AUTO_INCREMENT, - `check_id` BIGINT UNSIGNED NOT NULL, - PRIMARY KEY (`id`) -) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci; - -Converge it with: schemabot storage apply -``` - -The statements are printed bare and one per line so a whole section can be -pasted into a client as it stands. The exit status is the machine-readable half -of the answer: `0` when the storage needs nothing, `2` when statements are -outstanding, and `1` when the read itself failed. A pre-deploy gate needs those -three apart, since "converged" and "unreachable" call for opposite decisions. - -The headline answers the two questions that decide what the rest of the output -means. `schemabot on db-1.example (mysql)` is the database that was read — -reported by whoever read it, so it is not re-derived from a DSN, a config file, -or a deployment name. `against the schema embedded in v1.2.3` is the schema it -was compared against; [Which schema you are asking -about](#which-schema-you-are-asking-about) is how to change that. - -`storage apply` converges the database by running the same bootstrap the next -boot would run: the same differ, the same refusal of destructive statements, -and the same advisory lock, so two operators running it at once serialize the -way two booting pods do. It previews the statements and prompts before running -them; `--auto-approve` (`-y`) skips the prompt for scripted maintenance. - -```console -$ schemabot storage apply -schemabot on db-1.example (mysql) needs 1 statement: 1 outstanding, against the schema embedded in v1.2.3. - -Outstanding, and run automatically on the next boot or apply (1): - -ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; - -Run these statements against schemabot on db-1.example (mysql)? Only 'yes' will be accepted: yes -Ran 1 statement against schemabot on db-1.example (mysql). -schemabot on db-1.example (mysql) is converged. -``` - -Destructive statements are refused here exactly as they are at startup, and for -the same reason: a binary older than the storage sees the newer schema's tables -as surplus. A refusal is reported rather than silently dropped, and -`--allow-destructive` opts in per invocation, widening the deployment's standing -`allow_destructive_schema_changes` policy without ever narrowing it. - -```console -$ schemabot storage diff -schemabot on db-1.example (mysql) needs 1 statement: 1 destructive, against the schema embedded in v1.2.3. - -Destructive, and refused; surplus state stays in place (1): - --- check_gate_audit: DROP TABLE destroys data -DROP TABLE `check_gate_audit`; - -Converge it with: schemabot storage apply -``` - -### Reach the right storage database - -Both commands take a target, and which path applies is stated rather than -discovered. Nothing falls back from one to the other: a deployment that cannot -be reached through the API is an error naming the deployment, never a report -about a different database that happened to be reachable. - -| Target | Reads | -|---|---| -| no flags | the storage of the server the CLI is pointed at | -| `--deployment -e ` | that data plane's own storage, over the gRPC connection that already exists between the two | -| `--dsn ` or `--config ` | the storage database this workstation opens itself | - -A data plane owns its storage database and generally sits where a workstation -cannot dial it, so `--deployment` routes through the control plane: the control -plane asks the data plane, and the data plane reads its own storage with its own -embedded schema files. That is also what makes the answer trustworthy, since the -binary that reports the diff is the binary whose next boot would run it. - -The direct path exists for when the server is down, including when it is down -because its own schema bootstrap is failing. `--dialect` states the storage -family when a DSN's form does not say; it applies only to a direct connection. - -Both routes are admin-only and both sit at the write tier, the read-only diff -included, because the diff exposes the internal shape of SchemaBot's bookkeeping -database. Both are `POST` requests, so they take the write tier by the default -rule rather than by an exception. See [Authentication and -authorization](auth.md#what-read-and-write-access-include). - -### Which schema you are asking about - -The live side of the diff is always a read of the database. The desired side is -schema *files*, and three things can supply them: - -| Desired schema | Where the files come from | -|---|---| -| no flag | the embedded files of the binary that answers the request | -| `--schema-dir ` | that directory's `.sql` files, read by the CLI | -| `--release ` | that tag's `pkg/schema//` files, fetched by the CLI | - -With no flag, the answer describes the release that is **currently running**: -the files are compiled in (`go:embed` over `pkg/schema/mysql/` and -`pkg/schema/postgres/`), so through the API it is the server or data plane that -answered, and on the direct path it is the CLI binary you are running. That is -the right default — it is what the next boot would converge — and it is the -wrong question before a roll, when the running release reports convergence while -the release about to deploy still has work to do. - -`--release` asks that question without a binary of that release: - -```console -$ schemabot storage diff --deployment west -e production --release v1.4.0 -schemabot on db-1.example (mysql), deployment west in production needs 1 statement: 1 outstanding, against the schema files of release v1.4.0 in block/schemabot. - -Outstanding, and run automatically on the next boot or apply (1): - -ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; - -These are what schemabot on db-1.example (mysql), deployment west in production needs in order to match the schema files of release v1.4.0 in block/schemabot, not what its own next boot would run. To converge them, run that release's binary against this database — its container image is that release — or let the release's first boot converge them. -``` - -Because the storage schema is declarative, one diff against the release you are -rolling to covers however many releases lie between; there is nothing to step -through. - -The report names the schema it used, always, and never relabels a schema you -supplied as the answering binary's own. That line is the difference between two -correct reports about the same database, so read it before acting on the -statements. - -Details of the two selectors: - -- **`--schema-dir `** reads `*.sql` directly from a checkout or an - extracted image layer, one file per storage table. Point it at the dialect - directory (`pkg/schema/mysql`), not at its parent. A directory with no `.sql` - files is an error naming the path — a diff against an empty schema would - report every existing table as surplus. -- **`--release `** fetches the files over the repository's contents API at - that tag. It reads the schema directory for the dialect the *live storage* - runs, which it learns by first asking the target — one extra read-only diff, - paid only by this flag. `--release-repo` points at a fork or mirror - (`block/schemabot` by default), `GITHUB_API_URL` at a different API host, and - `GITHUB_TOKEN` or `GH_TOKEN` authorizes the fetch. A repository the CLI cannot - read is an error naming the token to set and `--schema-dir` as the offline - alternative. Naming both selectors is refused rather than resolved by - precedence. - -`storage apply` has neither flag. A convergence runs the schema embedded in the -binary running it, so that it does exactly what that binary's next boot would -do — the property that makes it usable as a pre-deploy step at all, and the one -that keeps an older binary from being handed newer schema to destroy. Passing -either flag to `apply` is refused with the two real ways to converge a release: -run that release's binary, or let its first boot do it. - -### Deploying a release that changes the storage schema - -Every startup converges the storage schema on its own, so the routine case -needs none of this. Reach for the commands when the release notes name a -storage schema change, when the tables involved carry a long history, or when a -pod is not starting. - -1. **Before the roll, ask the new release what it will run.** Name the release - being deployed, from whatever CLI you have to hand: - - ```bash - schemabot storage diff --deployment west -e production --release v1.4.0 - ``` - - Exit status 0 means that release's boot has nothing to do and the rest of - this does not apply. A binary of the new release answers the same question - with no flag, which is what to use where the tag cannot be fetched: - - ```bash - schemabot storage diff --dsn "$STORAGE_DSN" # run from the new release's binary - ``` - -2. **Decide whether the boot should do it.** Additive DDL inside the - five-minute startup budget is fine when the tables are small. It is not fine - when they are not: on MySQL an index added to an existing storage table runs - as Spirit online DDL, a table copy whose cost grows with row count, and every - pod in the roll pays it. Converge once, ahead of the roll, instead: - - ```bash - schemabot storage apply --dsn "$STORAGE_DSN" # from the new release's binary - ``` - - The convergence has to come from a binary of the new release: `apply` runs the - schema embedded in whatever binary runs it, and there is no flag that points - it at a release's files. Use that release's container image as a one-shot job - if there is no binary to hand. - -3. **If you converged ahead of the roll, re-check right before it.** A table or - a column you created early survives a boot of the current release, because - dropping one is destructive and is refused. **An index does not.** Dropping - an index destroys no data, so it falls outside that refusal, and any boot of - the still-running older release converges the new index away without - comment — a pod restart, a scale-up, a health-check replacement. Re-run the - step 1 diff immediately before rolling, and treat a long gap between - pre-creating an index and deploying as a gap the index probably did not - survive. - -4. **After the roll, confirm through the API, per deployment.** - - ```bash - schemabot storage diff # this server's storage - schemabot storage diff --deployment west -e production # a data plane's storage - ``` - - Now the running binary is the new release, so exit status 0 is the - confirmation that its storage converged. - -5. **If a pod is crashlooping, ask directly.** A failed storage bootstrap keeps - the server from accepting traffic at all, so the API cannot answer for it. - The direct path can, with the same binary the pod runs, and the statements it - prints are the ones the pod is failing on. `storage apply` from there clears - it under the same advisory lock the pods are contending for. - -During a rollback window the diff reports the newer release's tables and columns -as refused destructive statements and exits 2. That is the expected steady state -rather than drift: the surplus state is deliberate, and it is what lets the -release be rolled forward again. A pre-deploy gate keyed on exit status 0 will -flag it, which is the correct signal to pause on. +On PostgreSQL the setting has no effect: that convergence is additive-only and +never produces destructive DDL. See +[What is never automatic](storage-schema.md#what-is-never-automatic). ## Support Channel diff --git a/docs/release.md b/docs/release.md index 5907f9549..0905b154c 100644 --- a/docs/release.md +++ b/docs/release.md @@ -267,7 +267,7 @@ build is still running fails closed until it completes. A new column whose shape needs manual remediation — `NOT NULL` without a `DEFAULT`, generated or identity, `UNIQUE`, `REFERENCES` with a `DEFAULT` — fails startup until an operator creates it by hand, so it always belongs in the release notes with its -statement (see [configuration.md](./configuration.md)). A destructive +statement (see [storage-schema.md](./storage-schema.md)). A destructive change is a coordinated operation and belongs in the release notes with instructions, not in a routine patch release. @@ -281,7 +281,7 @@ schemabot storage diff --deployment west -e production --release v1.4.0 Name the release, or run the command from a binary of it — with neither, the answer describes whatever release is currently serving, which before a roll is the old one. Operators pre-creating an index ahead of the roll should also read -[Deploying a release that changes the storage schema](./configuration.md#deploying-a-release-that-changes-the-storage-schema) +[Deploying a release that changes the storage schema](./storage-schema.md#deploying-a-release-that-changes-the-storage-schema) — a pre-created index is removed again by any boot of the still-running earlier release, because dropping an index is not destructive and so is not refused. diff --git a/docs/storage-schema.md b/docs/storage-schema.md new file mode 100644 index 000000000..8e7dd4b32 --- /dev/null +++ b/docs/storage-schema.md @@ -0,0 +1,541 @@ +# SchemaBot's Own Storage Schema + + + +## Table of Contents + +- [How a storage schema change reaches the database](#how-a-storage-schema-change-reaches-the-database) +- [What converges by itself](#what-converges-by-itself) +- [What is never automatic](#what-is-never-automatic) +- [Ask what storage DDL is outstanding](#ask-what-storage-ddl-is-outstanding) +- [Which schema you are asking about](#which-schema-you-are-asking-about) +- [Reach the right storage database](#reach-the-right-storage-database) +- [Converge it](#converge-it) +- [Deploying a release that changes the storage schema](#deploying-a-release-that-changes-the-storage-schema) +- [When a pod will not start](#when-a-pod-will-not-start) +- [Pre-creating indexes on a long-lived database](#pre-creating-indexes-on-a-long-lived-database) +- [Adding a storage schema change](#adding-a-storage-schema-change) + + + +SchemaBot runs schema changes against your databases. It also *has* a database: +its own bookkeeping storage, holding plans, applies, checks, leases, locks, and +the webhook inbox. This guide is about that one — how its schema changes get +applied, what converges by itself, what never will, and the two commands that +answer "what is outstanding" and "converge it". + +The two are easy to confuse and are unrelated. A database SchemaBot *manages* +changes through a PR, a plan, a lint gate, and an apply with progress and +rollback. SchemaBot's *own* storage changes through the bootstrap described +here, which has none of that machinery and is deliberately far more +conservative: it is additive, it decides before it writes, and it never +destroys state a peer on another release might still be reading. + +Every operator running SchemaBot needs the first three sections. The rest is +the deploy and incident playbook. + +## How a storage schema change reaches the database + +SchemaBot's storage schema is *self-bootstrapping*. The schema files are +compiled into the binary — `go:embed` over `pkg/schema/mysql/` and +`pkg/schema/postgres/`, one `.sql` file per storage table — and the binary +converges the live database against them. Nobody applies storage DDL by hand, +and there is no schema directory on disk to keep in sync. + +There are exactly two ways that convergence runs, and they are the same code: + +``` +the startup path the CLI path +(every boot, automatic) (an operator, deliberate) + +┌────────────────────────────────┐ ┌────────────────────────────────┐ +│ pod starts │ │ schemabot storage apply │ +│ before it serves traffic │ │ at a terminal, or a job │ +└───────────────┬────────────────┘ └───────────────┬────────────────┘ + │ │ + └───────────────────┬───────────────────┘ + ▼ + ┌──────────────────────────────────┐ + │ EnsureSchema │ + │ │ + │ 1. diff the live catalog │ + │ against the embedded files │ + │ 2. nothing outstanding? done, │ + │ without taking a lock │ + │ 3. take the advisory lock │ + │ 4. re-diff, holding it │ + │ 5. refuse what it must not run │ + │ 6. run what is left │ + └────────────────┬─────────────────┘ + ▼ + the storage database, converged to the + schema of the binary that ran it, and to + nothing else +``` + +Three consequences worth holding onto: + +- **`storage apply` is what a boot does.** Not an equivalent of it — the same + differ, the same refusal, the same lock. So converging ahead of a roll cannot + disagree with what the roll then does, and two operators, or an operator and + a booting pod, serialize on the lock instead of racing. +- **A binary converges its own schema, never another's.** `storage apply` has no + flag pointing it at a release's files, and the RPC behind it has no field to + carry them. That is what keeps an older binary from being handed a newer + release's schema, where the newer tables it does not know about look like + surplus to prune. +- **Startup convergence is bounded.** The startup path runs inside a hard + five-minute budget before the pod serves traffic. A change that cannot finish + in it keeps the pod out of service, which is why large index builds belong + ahead of the roll rather than inside it. +- **A converged storage costs one diff.** Steps 1 and 2 run without the lock, so + the overwhelmingly common case — a pod booting against storage that already + matches it — never contends with anything. The re-diff at step 4 is what makes + that safe: whoever wins the lock decides again, holding it, so two pods that + both saw work do not both run it. + +The storage database is the one database SchemaBot cannot be down for, so the +bootstrap fails closed: uncertainty keeps the pod out of service rather than +becoming a half-converged schema. That is invariant AV-9 in +[invariants.md](invariants.md) — worth reading once if you operate this. + +## What converges by itself + +How far the automatic convergence goes depends on the storage dialect. + +**MySQL** diffs the embedded schema files against the live database and applies +whatever DDL is needed (via Spirit) — new tables, new columns, and index changes +all converge automatically. An index added to an existing table runs as Spirit +online DDL — a table copy, not an in-place build — so its cost grows with the +table's row count. On a deployment whose storage tables carry a long history, +create a newly declared index by hand before rolling out: the startup diff then +finds nothing to do, instead of copying the table inside the budget on every +pod. See [Pre-creating indexes on a long-lived +database](#pre-creating-indexes-on-a-long-lived-database). + +**PostgreSQL** automatically creates missing tables, columns, and standalone +indexes. It discovers drift before taking the bootstrap advisory lock, then +re-checks and applies each table's changes transactionally under that lock. +Convergence is additive-only: extra columns and indexes remain in place for +binary rollback, and `allow_destructive_schema_changes` has no effect because +this flow never produces destructive DDL. Column verification is +presence-only, so type, length, and nullability drift is outside its scope and +is not detected. + +A missing column converges automatically only when the `ADD COLUMN` is +metadata-only. A missing `NOT NULL` column without a `DEFAULT`, a generated or +identity column, a `UNIQUE` column, a `REFERENCES` column with a `DEFAULT`, or a +column with a constraint shape not explicitly classified as safe fails startup +with instructions for manual remediation: generated and identity columns rewrite +the populated table, `UNIQUE` builds a unique index over it, and a foreign key +with a `DEFAULT` validates every existing row against the referenced table — all +under an exclusive lock whose hold time the startup lock timeout does not bound. +Startup also fails when additive DDL cannot be parsed or executed, or when +re-verification finds unresolved drift. Every such problem across every table is +named in the one startup error, and no DDL runs until all of them are resolved. + +A live index only counts as present when PostgreSQL reports it valid. +PostgreSQL marks an index invalid both while a `CREATE INDEX CONCURRENTLY` is +still building it and after one fails part-way — a unique build that hits +duplicate keys, a cancelled session — and in either case the planner never uses +it. Startup fails closed naming that index rather than reading it as converged +or colliding with it on a fresh `CREATE INDEX`, and reads +`pg_stat_progress_create_index` to say which situation it is: + +- **A build is in progress** — the expected state while an operator pre-creates + an index ahead of a release. The error says so and asks for nothing; the pod + restarts on its backoff and starts cleanly once the build completes. +- **No build is visible** — the error treats the index as a failed build. Remove + the cause first (a unique build keeps failing while duplicate keys remain), + then drop the index so the next startup recreates it, or `REINDEX INDEX + CONCURRENTLY` it by hand. + +That view only shows other roles' sessions to a caller with +`pg_read_all_stats`, so if the storage role lacks it and the build runs under a +different role, confirm from a privileged session that no build is running +before recovering. A non-unique index under a name the embedded schema requires +to be unique fails startup the same way. + +## What is never automatic + +On MySQL, destructive statements in the diff — `DROP TABLE`, or an `ALTER TABLE` +containing `DROP COLUMN` — are refused and skipped by default. A mixed `ALTER +TABLE` is split: its additive clauses still execute and only the destructive +clauses are refused, except that a clause which cannot run without a refused +clause (the `ADD PRIMARY KEY` half of a primary-key change) is refused with it. +The remaining non-destructive statements still apply and startup proceeds. + +This protects against rolling deploys and rollbacks: a pod running an older +binary sees a newer binary's tables and columns as surplus, and without the gate +would drop them, destroying data the newer pods depend on. Each refused +statement is logged at warn level with the exact DDL, and counted in the +`schemabot.storage_schema.destructive_refusals_total` metric. + +**One asymmetry bites operators, and it is worth memorizing:** dropping an index +destroys no data, so it is *not* destructive and *not* refused. A table or +column you create ahead of a roll survives a boot of the still-running older +release. **An index does not.** Any boot of the earlier release converges a +newly created index away without comment — a pod restart, a scale-up, a +health-check replacement. Treat a long gap between pre-creating an index and +rolling the release that declares it as a gap the index probably did not +survive, and re-check before you roll. + +To intentionally remove a storage table or column, first make sure every running +pod is on a binary whose embedded schema no longer declares it, then opt in with +[`allow_destructive_schema_changes`](configuration.md#allow_destructive_schema_changes). +Leave the flag false during normal operation and revert it after the removal +converges. `--allow-destructive` on the CLI opts in for one invocation; it +widens the deployment's standing policy and never narrows it. + +## Ask what storage DDL is outstanding + +A deploy that did not converge leaves one question open: which storage DDL is +still outstanding. Two commands answer it, and both read the live storage +database. A release tag never stands in for that read: what a release would +converge to and what the storage actually converged to differ exactly when a +deploy has failed. + +`storage diff` is read-only. It takes no lock and holds no transaction, so it is +safe at any time, including against production during an incident. It names the +schema to compare the database against — `--embedded` for the schema of the +binary that answers, which is what its own next boot would converge to: + +```console +$ schemabot storage diff --embedded +schemabot on db-1.example (mysql) needs 3 statements: 3 outstanding, against the schema embedded in v1.2.3. + +Outstanding, and run automatically on the next boot or apply (3): + +ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; +ALTER TABLE `checks` ADD COLUMN `blocked_reason` varchar(64) NOT NULL DEFAULT '' AFTER `state`; +CREATE TABLE `check_gate_audit` ( + `id` BIGINT UNSIGNED AUTO_INCREMENT, + `check_id` BIGINT UNSIGNED NOT NULL, + PRIMARY KEY (`id`) +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci; + +Converge it with: schemabot storage apply +``` + +The statements are printed bare and one per line so a whole section can be +pasted into a client as it stands. + +The headline answers the two questions that decide what the rest of the output +means. `schemabot on db-1.example (mysql)` is the database that was read — +reported by whoever read it, so it is not re-derived from a DSN, a config file, +or a deployment name. `against the schema embedded in v1.2.3` is the schema it +was compared against, repeating the selector's answer so a report read later, or +out of a pipeline, still says what it means. + +The exit status is the machine-readable half of the answer: `0` when the storage +needs nothing, `2` when statements are outstanding, and `1` when the read itself +failed. A pre-deploy gate needs those three apart, since "converged" and +"unreachable" call for opposite decisions. + +## Which schema you are asking about + +The live side of the diff is always a read of the database. The desired side is +schema *files*, and `storage diff` requires you to say which: + +| Desired schema | Where the files come from | +|---|---| +| `--embedded` | the files compiled into the binary that answers the request | +| `--schema-dir ` | that directory's `.sql` files, read by the CLI | +| `--release ` | that tag's `pkg/schema//` files, fetched by the CLI | + +There is no default. A diff whose desired side you did not choose is not a +weaker answer, it is an unusable one: the same storage is converged against the +release that is running and short of the release about to roll, and an operator +who assumed the wrong one of those either rolls into a failing bootstrap or +converges storage they did not mean to touch. Naming more than one selector is +refused too, rather than resolved by precedence. + +`--embedded` answers for the release that is **currently running**, so through +the API it is the server or data plane that answered, and on the direct path it +is the CLI binary you are running. It is what that binary's next boot would +converge — and it is the wrong question before a roll, when the running release +reports convergence while the release about to deploy still has work to do. + +`--release` asks that question without a binary of that release: + +```console +$ schemabot storage diff --deployment west -e production --release v1.4.0 +schemabot on db-1.example (mysql), deployment west in production needs 1 statement: 1 outstanding, against the schema files of release v1.4.0 in block/schemabot. + +Outstanding, and run automatically on the next boot or apply (1): + +ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; + +These are what schemabot on db-1.example (mysql), deployment west in production needs in order to match the schema files of release v1.4.0 in block/schemabot, not what its own next boot would run. To converge them, run that release's binary against this database — its container image is that release — or let the release's first boot converge them. +``` + +Because the storage schema is declarative, one diff against the release you are +rolling to covers however many releases lie between; there is nothing to step +through. + +Details of the two file selectors: + +- **`--schema-dir `** reads `*.sql` directly from a checkout or an + extracted image layer, one file per storage table. Point it at the dialect + directory (`pkg/schema/mysql`), not at its parent. A directory with no `.sql` + files is an error naming the path — a diff against an empty schema would + report every existing table as surplus. +- **`--release `** fetches the files over the repository's contents API at + that tag. It reads the schema directory for the dialect the *live storage* + runs, which it learns by first asking the target — one extra read-only diff, + paid only by this flag. `--release-repo` points at a fork or mirror + (`block/schemabot` by default), `GITHUB_API_URL` at a different API host, and + `GITHUB_TOKEN` or `GH_TOKEN` authorizes the fetch. A repository the CLI cannot + read is an error naming the token to set and `--schema-dir` as the offline + alternative. + +`storage apply` takes `--embedded` and neither of the other two, for the reason +in [How a storage schema change reaches the +database](#how-a-storage-schema-change-reaches-the-database): a convergence runs +the schema of the binary running it. `--embedded` is therefore optional there, +since there is nothing else it could converge; passing `--schema-dir` or +`--release` is refused with the two real ways to converge a release — run that +release's binary, or let its first boot do it. + +## Reach the right storage database + +Both commands take a target, and which path applies is stated rather than +discovered. Nothing falls back from one to the other: a deployment that cannot +be reached through the API is an error naming the deployment, never a report +about a different database that happened to be reachable. + +| Target | Reads | +|---|---| +| no flags | the storage of the server the CLI is pointed at | +| `--deployment -e ` | that data plane's own storage, over the gRPC connection that already exists between the two | +| `--dsn ` or `--config ` | the storage database this workstation opens itself | + +A data plane owns its storage database and generally sits where a workstation +cannot dial it, so `--deployment` routes through the control plane: the control +plane asks the data plane, and the data plane reads its own storage. That is +also what makes the answer trustworthy, since the binary that reports the diff +is the binary whose next boot would run it. + +The direct path exists for when the server is down, including when it is down +because its own schema bootstrap is failing. `--dialect` states the storage +family when a DSN's form does not say; it applies only to a direct connection. + +Both routes are admin-only and both sit at the write tier, the read-only diff +included, because the diff exposes the internal shape of SchemaBot's bookkeeping +database. See [Authentication and +authorization](auth.md#what-read-and-write-access-include). + +## Converge it + +`storage apply` runs the bootstrap described above: the same differ, the same +refusal of destructive statements, and the same advisory lock, so two operators +running it at once serialize the way two booting pods do. It previews the +statements and prompts before running them; `--auto-approve` (`-y`) skips the +prompt for scripted maintenance. + +```console +$ schemabot storage apply +schemabot on db-1.example (mysql) needs 1 statement: 1 outstanding, against the schema embedded in v1.2.3. + +Outstanding, and run automatically on the next boot or apply (1): + +ALTER TABLE `applies` ADD COLUMN `driver_note` varchar(255) NOT NULL DEFAULT '' AFTER `lease_owner`; + +Run these statements against schemabot on db-1.example (mysql)? Only 'yes' will be accepted: yes +Ran 1 statement against schemabot on db-1.example (mysql). +schemabot on db-1.example (mysql) is converged. +``` + +Destructive statements are refused here exactly as they are at startup, and for +the same reason. A refusal is reported rather than silently dropped: + +```console +$ schemabot storage diff --embedded +schemabot on db-1.example (mysql) needs 1 statement: 1 destructive, against the schema embedded in v1.2.3. + +Destructive, and refused; surplus state stays in place (1): + +-- check_gate_audit: DROP TABLE destroys data +DROP TABLE `check_gate_audit`; + +Converge it with: schemabot storage apply +``` + +On PostgreSQL a change that needs manual remediation stops the convergence +before it runs anything, the same way it stops a startup: + +```console +$ schemabot storage apply +schemabot on db-2.example (postgres) needs 1 statement: 1 needing manual remediation, against the schema embedded in v1.2.3. + +Needs manual remediation before anything converges (1): + +-- checks: column is NOT NULL without a DEFAULT +ALTER TABLE "checks" ADD COLUMN "head_sha" varchar(64) NOT NULL; + +Error: refusing to converge storage schema on schemabot on db-2.example (postgres): 1 change(s) need manual remediation first (listed above) +``` + +## Deploying a release that changes the storage schema + +Every startup converges the storage schema on its own, so the routine case needs +none of this. Reach for the commands when the release notes name a storage +schema change, when the tables involved carry a long history, or when a pod is +not starting. + +1. **Before the roll, ask the new release what it will run.** Name the release + being deployed, from whatever CLI you have to hand: + + ```bash + schemabot storage diff --deployment west -e production --release v1.4.0 + ``` + + Exit status 0 means that release's boot has nothing to do and the rest of + this does not apply. Where the tag cannot be fetched, a binary of the new + release answers the same question about itself: + + ```bash + schemabot storage diff --embedded --dsn "$STORAGE_DSN" # from the new release's binary + ``` + +2. **Decide whether the boot should do it.** Additive DDL inside the + five-minute startup budget is fine when the tables are small. It is not fine + when they are not: on MySQL an index added to an existing storage table runs + as Spirit online DDL, a table copy whose cost grows with row count, and every + pod in the roll pays it. Converge once, ahead of the roll, instead: + + ```bash + schemabot storage apply --embedded --dsn "$STORAGE_DSN" # from the new release's binary + ``` + + The convergence has to come from a binary of the new release: `apply` runs + the schema embedded in whatever binary runs it. Use that release's container + image as a one-shot job if there is no binary to hand. + +3. **If you converged ahead of the roll, re-check right before it.** A table or + a column you created early survives a boot of the current release; **an index + does not**, for the reason in [What is never + automatic](#what-is-never-automatic). Re-run the step 1 diff immediately + before rolling. + +4. **After the roll, confirm through the API, per deployment.** + + ```bash + schemabot storage diff --embedded # this server's storage + schemabot storage diff --embedded --deployment west -e production # a data plane's storage + ``` + + Now the running binary is the new release, so exit status 0 is the + confirmation that its storage converged. + +During a rollback window the diff reports the newer release's tables and columns +as refused destructive statements and exits 2. That is the expected steady state +rather than drift: the surplus state is deliberate, and it is what lets the +release be rolled forward again. A pre-deploy gate keyed on exit status 0 will +flag it, which is the correct signal to pause on. + +## When a pod will not start + +A failed storage bootstrap keeps the server from accepting traffic at all, so +the API cannot answer for it — which is exactly when an operator needs the +answer. Ask the database directly, with the same binary the pod runs: + +```bash +schemabot storage diff --embedded --dsn "$STORAGE_DSN" +``` + +The statements it prints are the ones the pod is failing on. `storage apply` +from there clears them under the same advisory lock the pods are contending +for, and the pod starts on its next backoff. + +Two failures look similar in the logs and are not: + +- **The convergence is refusing something.** The diff names it — a destructive + statement, or a PostgreSQL column shape needing manual remediation. Nothing + will clear on its own; resolve the named item. +- **The convergence is not finishing in the budget.** The diff names outstanding + DDL that is legal but slow, typically an index over a table with a long + history. Converging it once from the CLI, outside the startup budget, is the + fix; the pod then finds nothing to do. + +## Pre-creating indexes on a long-lived database + +Indexes added to an embedded schema file after a database was bootstrapped +converge on the next startup. On PostgreSQL they run as plain `CREATE INDEX` +statements, each in its own transaction under the bootstrap advisory lock; a +plain `CREATE INDEX` holds a `SHARE` lock on the table for the full build and +blocks writes to it, and the startup budget is the build's only duration +ceiling. On MySQL they arrive as startup `ALTER`s under the same budget. Either +way, on a deployment whose storage tables carry a long history, pre-create the +index by hand before rolling out — the startup diff then finds it present and +skips the build. Use `CREATE INDEX CONCURRENTLY` on PostgreSQL, and finish the +build before rolling the release, since a pod that starts while the build is +still running fails closed until it completes. + +The indexes below are the ones a long-lived database is most likely to be +missing. `storage diff` names whichever of them this database actually needs; +this list is here for the ones worth pre-creating rather than leaving to a boot. + +A database bootstrapped before `idx_plans_created_at` was added to `plans` +needs: + +```sql +CREATE INDEX idx_plans_created_at ON plans (created_at); +``` + +Without it, listing recent plans is a sequential scan plus a top-N sort, which +gets slower as plan history grows. Likewise, one bootstrapped before the driver +claim ordering on `apply_operations` was indexed needs: + +```sql +CREATE INDEX idx_apply_operations_created_id ON apply_operations (created_at, id); +``` + +Without it, every driver claim sorts the full claimable set before taking one +row, which slows claiming as apply history grows. One bootstrapped before +refused applies started naming the schema change holding the database needs: + +```sql +CREATE INDEX idx_apply_operations_external_id ON apply_operations (external_id); +``` + +Without it, resolving the holding change behind a refused apply scans the full +operation history for one remote identifier. On PostgreSQL the lookup is an +optimization, never load-bearing: the refusal still reads correctly, it just +gets slower to record as apply history grows. On MySQL the same index is not +optional, and `apply_operations` grows with total apply history, so large +deployments should pre-create it there too. And one bootstrapped before the +webhook inbox claim ordering on `webhook_events` was indexed needs: + +```sql +CREATE INDEX idx_webhook_events_created_id ON webhook_events (created_at, id); +``` + +Without it, every webhook claim sorts the full claimable inbox before taking one +row, which slows claiming as delivery history grows. On MySQL `webhook_events` +grows with total delivery history and has no retention sweep, so pre-create it +there before rolling out: + +```sql +ALTER TABLE `webhook_events` ADD INDEX `idx_created_id` (`created_at`, `id`); +``` + +## Adding a storage schema change + +For contributors: adding a table or a column to SchemaBot's storage means adding +or editing a file under `pkg/schema/mysql/` and its counterpart under +`pkg/schema/postgres/`. The next deploy picks it up — there is no schema +directory to ship and no DDL to write into a runbook. Schema parity tests pin +the two dialect directories against each other, so a table file present for one +dialect and not the other fails CI, as does an index that differs in table, +ordered columns, or uniqueness. + +Design the change so that a binary *without* it can keep running against a +database that has it. That is what makes a rolling deploy and a rollback safe, +and it is why the convergence is additive: a new column is nullable or carries a +default, a removal is a coordinated operation rather than a patch release, and +nothing assumes the whole fleet is on one release at once. + +[release.md](release.md) covers what a storage schema change obliges a release +to say: which statements operators may want to pre-create, and which column +shapes will stop a PostgreSQL startup until an operator runs them by hand. diff --git a/pkg/cmd/commands/storage_schema.go b/pkg/cmd/commands/storage_schema.go index 2a0d063c4..8148d988b 100644 --- a/pkg/cmd/commands/storage_schema.go +++ b/pkg/cmd/commands/storage_schema.go @@ -96,10 +96,11 @@ func (f *storageSchemaTargetFlags) validate() error { // storage needs nothing, 2 when statements are outstanding, and 1 when the // read itself failed. A pre-deploy gate needs those three apart, because // "converged" and "unreachable" call for opposite decisions. -// The schema it compares against defaults to the answering binary's own, and -// --schema-dir or --release point it at another release's instead, for the -// question a deploy actually asks: is this storage ready for the release about -// to roll. +// +// The schema it compares against is always named — --embedded, --schema-dir, or +// --release — because the question a deploy asks is whether the storage is +// ready for the release about to roll, and that is a different question from +// whether it matches the release now running. type StorageDiffCmd struct { storageSchemaTargetFlags `embed:""` storageSchemaSourceFlags `embed:""` @@ -240,10 +241,15 @@ type StorageApplyCmd struct { AllowDestructive bool `help:"Permit the destructive statements the convergence would otherwise refuse; it widens the target's standing storage policy and never narrows it" name:"allow-destructive"` AutoApprove bool `short:"y" help:"Skip confirmation prompt" name:"auto-approve"` JSON bool `help:"Output as JSON"` - // The diff's schema selectors are accepted here only to be refused with - // the reason and the alternative. An operator who has just run the diff - // against a release reaches for the same flags on the apply, and Kong's - // bare "unknown flag" would leave them guessing at whether the convergence + // Embedded names the one schema a convergence can run, so an operator + // moving from a diff to an apply can carry the flag over and have it mean + // what it said. It changes nothing, because there is nothing else to + // converge. + Embedded bool `help:"Converge the schema built into the binary that runs the convergence — the only schema a convergence can run; see storage diff for the others"` + // The diff's file selectors are accepted here only to be refused with the + // reason and the alternative. An operator who has just run the diff against + // a release reaches for the same flags on the apply, and Kong's bare + // "unknown flag" would leave them guessing at whether the convergence // silently used a different schema. SchemaDir string `hidden:"" name:"schema-dir"` Release string `hidden:""` @@ -265,6 +271,7 @@ func (cmd *StorageApplyCmd) Run(ctx context.Context, g *Globals) error { if !cmd.AutoApprove { preview := &StorageDiffCmd{ storageSchemaTargetFlags: cmd.storageSchemaTargetFlags, + storageSchemaSourceFlags: storageSchemaSourceFlags{Embedded: true}, AllowDestructive: cmd.AllowDestructive, } report, err := preview.read(ctx, g) @@ -495,7 +502,7 @@ func storageSchemaDestructiveTitle(report *apitypes.StorageSchemaReport) string // storageSchemaDiffHints names the next step for the report a diff just // printed, in the command form the operator invoked the CLI as. func storageSchemaDiffHints(cmd *StorageDiffCmd, report *apitypes.StorageSchemaReport) []string { - if cmd.selected() { + if cmd.suppliesFiles() { // Naming `storage apply` here would be wrong: it converges the schema // of the binary that answers, which is not the schema this report is // about. The two ways to converge the release's schema are the release diff --git a/pkg/cmd/commands/storage_schema_source.go b/pkg/cmd/commands/storage_schema_source.go index 4a853d024..32df175e4 100644 --- a/pkg/cmd/commands/storage_schema_source.go +++ b/pkg/cmd/commands/storage_schema_source.go @@ -24,39 +24,62 @@ import ( // question is whether the storage is ready for the release about to roll, and // the release about to roll is by definition not the one running. // -// So the desired side is selectable, two ways, and the report always says which -// one was used: +// So the desired side is named, always, one of three ways, and the report +// repeats which one was used: // +// --embedded the schema built into the binary that answers — what +// its own next boot would converge to // --schema-dir the .sql files in a directory — a checkout of the // release, or an unreleased commit, and offline // --release the .sql files of a published tag, fetched from the // repository // -// Neither is available on `storage apply`, and that is the safety property -// rather than an omission: a convergence runs the schema of the binary running -// it, so "apply is what a boot does" holds by construction (AV-9). To converge -// a release's schema, run that release's binary. +// There is deliberately no default. A diff read without knowing which schema it +// compared against is not a weaker answer, it is an unusable one: the same +// database is converged against the release that is running and three +// statements short of the release about to roll, and an operator who assumed +// the wrong side of that either rolls into a failing bootstrap or converges +// storage they did not mean to touch. +// +// Only --embedded is available on `storage apply`, and that is the safety +// property rather than an omission: a convergence runs the schema of the binary +// running it, so "apply is what a boot does" holds by construction (AV-9). To +// converge a release's schema, run that release's binary. -// storageSchemaSourceFlags selects the desired side of the diff. The two +// storageSchemaSourceFlags names the desired side of the diff. The three // selectors are mutually exclusive: each names a complete schema, and silently // preferring one would answer a question the operator did not ask. type storageSchemaSourceFlags struct { - SchemaDir string `help:"Diff against the .sql files in this directory instead of the schema built into the binary that answers — a checkout of the release you are about to deploy (e.g. ./pkg/schema/mysql)" name:"schema-dir" type:"path"` + Embedded bool `help:"Diff against the schema built into the binary that answers — through the API that is the release currently running, which is what its next boot would converge to"` + SchemaDir string `help:"Diff against the .sql files in this directory instead — a checkout of the release you are about to deploy (e.g. ./pkg/schema/mysql)" name:"schema-dir" type:"path"` Release string `help:"Diff against the schema files of this published tag, fetched from the SchemaBot repository (e.g. v1.4.0)"` Repo string `help:"Repository to fetch --release schema files from" name:"release-repo" default:"block/schemabot"` } -// selected reports whether the operator named a desired schema other than the -// answering binary's own. -func (f *storageSchemaSourceFlags) selected() bool { +// suppliesFiles reports whether the desired schema is files the CLI carries to +// the target, rather than the schema the answering binary already has. +func (f *storageSchemaSourceFlags) suppliesFiles() bool { return strings.TrimSpace(f.SchemaDir) != "" || strings.TrimSpace(f.Release) != "" } -// validateSource refuses selector combinations rather than resolving them by -// precedence. +// validateSource requires exactly one desired schema: one named, rather than +// several resolved by precedence, and never none resolved by default. func (f *storageSchemaSourceFlags) validateSource() error { - if strings.TrimSpace(f.SchemaDir) != "" && strings.TrimSpace(f.Release) != "" { - return fmt.Errorf("--schema-dir and --release both name a schema to diff against: pass one; --schema-dir reads files you already have, --release fetches a published tag") + named := make([]string, 0, 3) + if f.Embedded { + named = append(named, "--embedded") + } + if strings.TrimSpace(f.SchemaDir) != "" { + named = append(named, "--schema-dir") + } + if strings.TrimSpace(f.Release) != "" { + named = append(named, "--release") + } + switch { + case len(named) == 0: + return fmt.Errorf("name the schema to diff the live database against: --embedded for the schema of the binary that answers, which is what its own next boot would converge to; --release for a published release's schema files; --schema-dir for a checkout's. There is no default because the answer means different things: the same storage is converged against the release that is running and short of the release about to roll") + case len(named) > 1: + return fmt.Errorf("%s each name a whole schema to diff against: pass one. --embedded is the answering binary's own, --release fetches a published tag, --schema-dir reads files you already have", strings.Join(named, " and ")) } repo := strings.TrimSpace(f.Repo) if strings.TrimSpace(f.Release) == "" && repo != "" && repo != defaultStorageSchemaRepo { @@ -65,8 +88,10 @@ func (f *storageSchemaSourceFlags) validateSource() error { return nil } -// storageSchemaSourceRefusal refuses the diff's schema selectors on a +// storageSchemaSourceRefusal refuses the diff's file selectors on a // convergence, and names the two ways to converge a release's schema instead. +// --embedded is not refused: it is the schema a convergence runs, so naming it +// is the operator stating what they are about to do. // // The refusal is the invariant, stated where an operator meets it. A // convergence runs the schema embedded in the binary running it, which is what @@ -87,8 +112,9 @@ func storageSchemaSourceRefusal(schemaDir, release string) error { return fmt.Errorf("%s cannot be used with a convergence: an apply runs the schema embedded in the binary running it, so that it converges exactly what that binary's next boot would. To converge a release's schema, run that release's binary — its container image is that release — or let the release's own first boot converge it. To see what it would do, use the same flag on `storage diff`", selector) } -// resolve reads the desired schema the flags selected, or returns nil for the -// answering binary's own embedded schema. +// resolve reads the desired schema the flags named, or returns nil for +// --embedded: the answering binary already has those files, so carrying a copy +// of them to it would only create a way for the two to disagree. // // dialect is resolved lazily, by calling it, because only one selector needs // it: a release's schema files live in a per-dialect directory of the diff --git a/pkg/cmd/commands/storage_schema_source_test.go b/pkg/cmd/commands/storage_schema_source_test.go index fc6d0d663..38374ca9b 100644 --- a/pkg/cmd/commands/storage_schema_source_test.go +++ b/pkg/cmd/commands/storage_schema_source_test.go @@ -19,30 +19,46 @@ import ( // for the cases where resolving it is not what is under test. func mysqlDialect() (schema.Dialect, error) { return schema.DialectMySQL, nil } -// The two schema selectors each name a complete schema, so naming both is -// refused rather than resolved by precedence: silently preferring one would -// answer a question the operator did not ask. +// Exactly one schema is named. Naming several is refused rather than resolved +// by precedence, and naming none is refused rather than resolved by default: a +// report whose desired side the operator did not choose is the one report that +// can be read as the opposite of what it says. func TestStorageSchemaSourceFlags_ValidateSource(t *testing.T) { - require.NoError(t, (&storageSchemaSourceFlags{}).validateSource()) + require.NoError(t, (&storageSchemaSourceFlags{Embedded: true}).validateSource()) require.NoError(t, (&storageSchemaSourceFlags{SchemaDir: "./schema/mysql"}).validateSource()) require.NoError(t, (&storageSchemaSourceFlags{Release: "v1.4.0", Repo: defaultStorageSchemaRepo}).validateSource()) require.NoError(t, (&storageSchemaSourceFlags{Release: "v1.4.0", Repo: "example/mirror"}).validateSource()) + unnamed := (&storageSchemaSourceFlags{}).validateSource() + require.Error(t, unnamed, "the desired schema has no default") + assert.Contains(t, unnamed.Error(), "name the schema to diff the live database against") + assert.Contains(t, unnamed.Error(), "--embedded") + assert.Contains(t, unnamed.Error(), "--release") + assert.Contains(t, unnamed.Error(), "--schema-dir") + err := (&storageSchemaSourceFlags{SchemaDir: "./schema/mysql", Release: "v1.4.0"}).validateSource() require.Error(t, err) - assert.Contains(t, err.Error(), "both name a schema to diff against") + assert.Contains(t, err.Error(), "--schema-dir and --release each name a whole schema") + + err = (&storageSchemaSourceFlags{Embedded: true, Release: "v1.4.0"}).validateSource() + require.Error(t, err) + assert.Contains(t, err.Error(), "--embedded and --release each name a whole schema") err = (&storageSchemaSourceFlags{Repo: "example/mirror"}).validateSource() require.Error(t, err) + assert.Contains(t, err.Error(), "name the schema to diff the live database against") + + err = (&storageSchemaSourceFlags{Embedded: true, Repo: "example/mirror"}).validateSource() + require.Error(t, err) assert.Contains(t, err.Error(), "--release-repo only applies with --release") } -// With no selector the diff is against the schema of the binary that answers, -// which is what a boot would converge to. Resolving that needs no files and no -// dialect, so nothing is read and nothing is fetched. -func TestStorageSchemaSourceFlags_ResolveDefaultsToTheAnsweringBinary(t *testing.T) { - desired, err := (&storageSchemaSourceFlags{}).resolve(t.Context(), func() (schema.Dialect, error) { - t.Fatal("the dialect must not be resolved when no schema was named") +// --embedded is the schema of the binary that answers, which already has those +// files: resolving it reads nothing, fetches nothing, and needs no dialect, so +// the target is asked about its own schema rather than handed a copy of it. +func TestStorageSchemaSourceFlags_ResolveEmbedded(t *testing.T) { + desired, err := (&storageSchemaSourceFlags{Embedded: true}).resolve(t.Context(), func() (schema.Dialect, error) { + t.Fatal("the dialect must not be resolved for the answering binary's own schema") return "", nil }) require.NoError(t, err) @@ -222,8 +238,9 @@ func TestStorageSchemaFromRelease_SendsTheToken(t *testing.T) { } // A convergence runs the schema embedded in the binary running it, so the -// diff's selectors are refused on `storage apply` — with the two ways to +// diff's file selectors are refused on `storage apply` — with the two ways to // converge a release named, since that is what the operator is reaching for. +// --embedded is not refused: it names what the convergence already does. func TestStorageSchemaSourceRefusal(t *testing.T) { require.NoError(t, storageSchemaSourceRefusal("", "")) diff --git a/pkg/cmd/commands/storage_schema_test.go b/pkg/cmd/commands/storage_schema_test.go index 15cf4912b..ba152c94d 100644 --- a/pkg/cmd/commands/storage_schema_test.go +++ b/pkg/cmd/commands/storage_schema_test.go @@ -78,6 +78,19 @@ func TestStorageSchemaTargetFlags_Validate(t *testing.T) { } } +// A diff names the schema it compares the database against, and refuses before +// it reads anything when the operator did not. The report would say which +// schema it used either way; the point of refusing is that the operator has +// already decided by the time they read it. +func TestStorageDiffCmd_RequiresADesiredSchema(t *testing.T) { + cmd := &StorageDiffCmd{ + storageSchemaTargetFlags: storageSchemaTargetFlags{DSN: "root@tcp(127.0.0.1:3306)/schemabot"}, + } + err := cmd.Run(t.Context(), &Globals{}) + require.Error(t, err) + assert.Contains(t, err.Error(), "name the schema to diff the live database against") +} + // Naming a DSN source is what selects the direct path, so a command can tell // which path it is on without inspecting what happened to resolve. func TestStorageSchemaTargetFlags_Direct(t *testing.T) { @@ -311,7 +324,9 @@ func TestRenderStorageSchemaReport_ManualSection(t *testing.T) { func TestStorageSchemaDiffHints(t *testing.T) { report := &apitypes.StorageSchemaReport{Database: "schemabot", Dialect: "mysql"} - local := storageSchemaDiffHints(&StorageDiffCmd{}, report) + local := storageSchemaDiffHints(&StorageDiffCmd{ + storageSchemaSourceFlags: storageSchemaSourceFlags{Embedded: true}, + }, report) require.Len(t, local, 1) assert.Contains(t, local[0], "storage apply") assert.NotContains(t, local[0], "--deployment") From 191e79896fad8718acc970127872920b53f7f50d Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Fri, 11 Sep 2026 15:17:57 -0400 Subject: [PATCH 5/5] feat(cli): require the schema selector as a flag, not a paragraph The refusal when no desired schema is named is now one line naming the three flags, and the parser refuses two selectors at once. The reasoning for having no default lives in the flag help and in the storage schema guide, where an operator reads it before they are blocked by it. Co-Authored-By: Claude Opus 5 --- docs/storage-schema.md | 12 ++++---- pkg/cmd/commands/storage_schema_source.go | 21 +++++++------- .../commands/storage_schema_source_test.go | 8 +++--- pkg/cmd/commands/storage_schema_test.go | 28 +++++++++++++++---- 4 files changed, 44 insertions(+), 25 deletions(-) diff --git a/docs/storage-schema.md b/docs/storage-schema.md index 8e7dd4b32..fd963b00e 100644 --- a/docs/storage-schema.md +++ b/docs/storage-schema.md @@ -243,12 +243,12 @@ schema *files*, and `storage diff` requires you to say which: | `--schema-dir ` | that directory's `.sql` files, read by the CLI | | `--release ` | that tag's `pkg/schema//` files, fetched by the CLI | -There is no default. A diff whose desired side you did not choose is not a -weaker answer, it is an unusable one: the same storage is converged against the -release that is running and short of the release about to roll, and an operator -who assumed the wrong one of those either rolls into a failing bootstrap or -converges storage they did not mean to touch. Naming more than one selector is -refused too, rather than resolved by precedence. +One of the three is required, and naming two is refused rather than resolved by +precedence. There is no default because a diff whose desired side you did not +choose is not a weaker answer, it is an unusable one: the same storage is +converged against the release that is running and short of the release about to +roll, and an operator who assumed the wrong one of those either rolls into a +failing bootstrap or converges storage they did not mean to touch. `--embedded` answers for the release that is **currently running**, so through the API it is the server or data plane that answered, and on the direct path it diff --git a/pkg/cmd/commands/storage_schema_source.go b/pkg/cmd/commands/storage_schema_source.go index 32df175e4..673fcd4f0 100644 --- a/pkg/cmd/commands/storage_schema_source.go +++ b/pkg/cmd/commands/storage_schema_source.go @@ -46,13 +46,13 @@ import ( // running it, so "apply is what a boot does" holds by construction (AV-9). To // converge a release's schema, run that release's binary. -// storageSchemaSourceFlags names the desired side of the diff. The three -// selectors are mutually exclusive: each names a complete schema, and silently -// preferring one would answer a question the operator did not ask. +// storageSchemaSourceFlags names the desired side of the diff. Exactly one +// selector is required: each names a complete schema, so the parser refuses +// two of them and validateSource refuses none. type storageSchemaSourceFlags struct { - Embedded bool `help:"Diff against the schema built into the binary that answers — through the API that is the release currently running, which is what its next boot would converge to"` - SchemaDir string `help:"Diff against the .sql files in this directory instead — a checkout of the release you are about to deploy (e.g. ./pkg/schema/mysql)" name:"schema-dir" type:"path"` - Release string `help:"Diff against the schema files of this published tag, fetched from the SchemaBot repository (e.g. v1.4.0)"` + Embedded bool `help:"Diff against the schema built into the binary that answers — through the API that is the release currently running, which is what its next boot would converge to" xor:"desired-schema"` + SchemaDir string `help:"Diff against the .sql files in this directory instead — a checkout of the release you are about to deploy (e.g. ./pkg/schema/mysql)" name:"schema-dir" type:"path" xor:"desired-schema"` + Release string `help:"Diff against the schema files of this published tag, fetched from the SchemaBot repository (e.g. v1.4.0)" xor:"desired-schema"` Repo string `help:"Repository to fetch --release schema files from" name:"release-repo" default:"block/schemabot"` } @@ -62,8 +62,9 @@ func (f *storageSchemaSourceFlags) suppliesFiles() bool { return strings.TrimSpace(f.SchemaDir) != "" || strings.TrimSpace(f.Release) != "" } -// validateSource requires exactly one desired schema: one named, rather than -// several resolved by precedence, and never none resolved by default. +// validateSource requires a desired schema and refuses --release-repo without +// the flag it modifies. Two selectors at once is the parser's own refusal, and +// restated here for callers that build the command in Go. func (f *storageSchemaSourceFlags) validateSource() error { named := make([]string, 0, 3) if f.Embedded { @@ -77,9 +78,9 @@ func (f *storageSchemaSourceFlags) validateSource() error { } switch { case len(named) == 0: - return fmt.Errorf("name the schema to diff the live database against: --embedded for the schema of the binary that answers, which is what its own next boot would converge to; --release for a published release's schema files; --schema-dir for a checkout's. There is no default because the answer means different things: the same storage is converged against the release that is running and short of the release about to roll") + return fmt.Errorf("missing flags: --embedded or --schema-dir=STRING or --release=STRING") case len(named) > 1: - return fmt.Errorf("%s each name a whole schema to diff against: pass one. --embedded is the answering binary's own, --release fetches a published tag, --schema-dir reads files you already have", strings.Join(named, " and ")) + return fmt.Errorf("%s can't be used together", strings.Join(named, " and ")) } repo := strings.TrimSpace(f.Repo) if strings.TrimSpace(f.Release) == "" && repo != "" && repo != defaultStorageSchemaRepo { diff --git a/pkg/cmd/commands/storage_schema_source_test.go b/pkg/cmd/commands/storage_schema_source_test.go index 38374ca9b..8e9fb026a 100644 --- a/pkg/cmd/commands/storage_schema_source_test.go +++ b/pkg/cmd/commands/storage_schema_source_test.go @@ -31,22 +31,22 @@ func TestStorageSchemaSourceFlags_ValidateSource(t *testing.T) { unnamed := (&storageSchemaSourceFlags{}).validateSource() require.Error(t, unnamed, "the desired schema has no default") - assert.Contains(t, unnamed.Error(), "name the schema to diff the live database against") + assert.Contains(t, unnamed.Error(), "missing flags") assert.Contains(t, unnamed.Error(), "--embedded") assert.Contains(t, unnamed.Error(), "--release") assert.Contains(t, unnamed.Error(), "--schema-dir") err := (&storageSchemaSourceFlags{SchemaDir: "./schema/mysql", Release: "v1.4.0"}).validateSource() require.Error(t, err) - assert.Contains(t, err.Error(), "--schema-dir and --release each name a whole schema") + assert.Contains(t, err.Error(), "--schema-dir and --release can't be used together") err = (&storageSchemaSourceFlags{Embedded: true, Release: "v1.4.0"}).validateSource() require.Error(t, err) - assert.Contains(t, err.Error(), "--embedded and --release each name a whole schema") + assert.Contains(t, err.Error(), "--embedded and --release can't be used together") err = (&storageSchemaSourceFlags{Repo: "example/mirror"}).validateSource() require.Error(t, err) - assert.Contains(t, err.Error(), "name the schema to diff the live database against") + assert.Contains(t, err.Error(), "missing flags") err = (&storageSchemaSourceFlags{Embedded: true, Repo: "example/mirror"}).validateSource() require.Error(t, err) diff --git a/pkg/cmd/commands/storage_schema_test.go b/pkg/cmd/commands/storage_schema_test.go index ba152c94d..1d170ccd1 100644 --- a/pkg/cmd/commands/storage_schema_test.go +++ b/pkg/cmd/commands/storage_schema_test.go @@ -6,6 +6,7 @@ import ( "fmt" "testing" + "github.com/alecthomas/kong" "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" @@ -78,17 +79,34 @@ func TestStorageSchemaTargetFlags_Validate(t *testing.T) { } } -// A diff names the schema it compares the database against, and refuses before -// it reads anything when the operator did not. The report would say which -// schema it used either way; the point of refusing is that the operator has -// already decided by the time they read it. +// A diff names the schema it compares the database against: one selector is +// required, and naming two is refused by the parser. The report would say which +// schema it used either way; requiring the flag is what makes the operator +// decide before they read the report rather than after. func TestStorageDiffCmd_RequiresADesiredSchema(t *testing.T) { cmd := &StorageDiffCmd{ storageSchemaTargetFlags: storageSchemaTargetFlags{DSN: "root@tcp(127.0.0.1:3306)/schemabot"}, } err := cmd.Run(t.Context(), &Globals{}) require.Error(t, err) - assert.Contains(t, err.Error(), "name the schema to diff the live database against") + assert.Contains(t, err.Error(), "missing flags: --embedded or --schema-dir=STRING or --release=STRING") + + parse := func(args ...string) error { + var cli struct { + Diff StorageDiffCmd `cmd:"" name:"diff"` + } + parser, err := kong.New(&cli, kong.Name("schemabot")) + require.NoError(t, err) + _, err = parser.Parse(append([]string{"diff"}, args...)) + return err + } + require.NoError(t, parse("--embedded")) + require.NoError(t, parse("--release", "v1.4.0")) + require.NoError(t, parse("--schema-dir", ".")) + + err = parse("--embedded", "--release", "v1.4.0") + require.Error(t, err) + assert.Contains(t, err.Error(), "--embedded and --release can't be used together") } // Naming a DSN source is what selects the direct path, so a command can tell