From 9a6ffa7a0333547185680a5a82cad8ab798ada59 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 13:41:09 -0400 Subject: [PATCH 01/34] refactor(api): identify a rollout member by deployment and target The review-time drift rollup identified each member of a multi-deployment rollout by its deployment name alone. That holds only while a deployment addresses exactly one target, so the positional producer/rollup contract could not distinguish two members that share a deployment. Key the contract on the (deployment, target) pair via a MemberID helper on routing.ExecutionTarget, thread the pair through the producer's primary invariant, and carry the plan's origin target alongside its origin deployment so the baseline is verified against the member it was created for. Behavior is unchanged for every configured shape today, where each deployment addresses one target. Co-Authored-By: Claude Fable 5 --- pkg/api/plan_deployment_diffs.go | 29 ++++----- pkg/api/plan_deployment_diffs_test.go | 21 ++++--- pkg/api/plan_handlers.go | 6 +- pkg/api/plan_review_drift.go | 30 ++++----- pkg/api/plan_review_drift_test.go | 8 +-- pkg/api/plan_rollup.go | 41 ++++++------ pkg/api/plan_rollup_test.go | 90 ++++++++++++++++++++------- pkg/apitypes/apitypes.go | 13 ++-- pkg/routing/resolver.go | 9 +++ pkg/webhook/plan.go | 5 +- pkg/webhook/plan_drift.go | 8 ++- 11 files changed, 165 insertions(+), 95 deletions(-) diff --git a/pkg/api/plan_deployment_diffs.go b/pkg/api/plan_deployment_diffs.go index b1dda9746..d0c7b9e36 100644 --- a/pkg/api/plan_deployment_diffs.go +++ b/pkg/api/plan_deployment_diffs.go @@ -53,28 +53,29 @@ type DeploymentPlanDiff struct { // supplied by the caller so a database/environment is resolved once per rollup // rather than re-resolved here. It must be non-empty. // -// primaryDeployment is the deployment the reviewed primaryPlan was created -// against (rollout index 0 at plan time). When a primaryPlan is reused, it is -// checked against targets[0] here so a deployment-order change between plan and -// rollup — which would map the reviewed baseline onto a different deployment — -// fails closed rather than being compared against the wrong live schema. -func (s *Service) PlanDeploymentDiffs(ctx context.Context, req PlanRequest, primaryPlan *ternv1.PlanResponse, primaryDeployment string, targets []routing.ExecutionTarget) ([]DeploymentPlanDiff, error) { +// primaryMember is the rollout member the reviewed primaryPlan was created +// against (rollout index 0 at plan time), identified by deployment and target +// together because one deployment can address several targets. When a +// primaryPlan is reused, it is checked against targets[0] here so a rollout +// order change between plan and rollup — which would map the reviewed baseline +// onto a different member — fails closed rather than being compared against the +// wrong live schema. +func (s *Service) PlanDeploymentDiffs(ctx context.Context, req PlanRequest, primaryPlan *ternv1.PlanResponse, primaryMember routing.ExecutionTarget, targets []routing.ExecutionTarget) ([]DeploymentPlanDiff, error) { if len(targets) == 0 { return nil, fmt.Errorf("no deployment targets for %s/%s", req.Database, req.Environment) } // Reusing primaryPlan for the primary member assumes it was created against - // the deployment now at rollout index 0. Verify that against the plan's - // recorded origin deployment (captured at plan time) and fail closed on a - // mismatch — e.g. deployment_order changed between plan and rollup — rather - // than comparing deployments against a baseline built for a different - // deployment. + // the member now at rollout index 0. Verify that against the plan's recorded + // origin member (captured at plan time) and fail closed on a mismatch — e.g. + // deployment_order changed between plan and rollup — rather than comparing + // members against a baseline built for a different one. if primaryPlan != nil { - if primaryDeployment == "" { + if primaryMember.Deployment == "" { return nil, fmt.Errorf("plan diff for %s/%s: reviewed plan has no origin deployment to verify the primary against", req.Database, req.Environment) } - if targets[0].Deployment != primaryDeployment { - return nil, fmt.Errorf("primary invariant violated for %s/%s: rollout index 0 is %q but the reviewed plan was created against %q", req.Database, req.Environment, targets[0].Deployment, primaryDeployment) + if targets[0].Deployment != primaryMember.Deployment || targets[0].Target != primaryMember.Target { + return nil, fmt.Errorf("primary invariant violated for %s/%s: rollout index 0 is %q but the reviewed plan was created against %q", req.Database, req.Environment, targets[0].MemberID(), primaryMember.MemberID()) } } diff --git a/pkg/api/plan_deployment_diffs_test.go b/pkg/api/plan_deployment_diffs_test.go index 5f4f6a26d..9c5292a54 100644 --- a/pkg/api/plan_deployment_diffs_test.go +++ b/pkg/api/plan_deployment_diffs_test.go @@ -43,6 +43,13 @@ func twoDeploymentService(t *testing.T, eu, us *mockTernClient) *Service { }, logger) } +// productionMember names one rollout member of the testapp/production set. Both +// deployments address the same target, so the deployment half is what +// distinguishes them here. +func productionMember(deployment string) routing.ExecutionTarget { + return routing.ExecutionTarget{Deployment: deployment, Target: "testapp"} +} + // productionTargets resolves the testapp/production deployment set the producer // tests exercise, so each PlanDeploymentDiffs call is given the same resolved // targets the rollup would pass in. @@ -102,7 +109,7 @@ func TestPlanDeploymentDiffs_PrimaryReusesReviewedPlan(t *testing.T) { }}, } - results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, "eu", productionTargets(t, svc)) + results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, productionMember("eu"), productionTargets(t, svc)) require.NoError(t, err) require.Len(t, results, 2) @@ -126,7 +133,7 @@ func TestPlanDeploymentDiffs_DiffsAllDeploymentsWhenNoPrimary(t *testing.T) { us := &mockTernClient{planDiffResp: alterUsersDiff("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")} svc := twoDeploymentService(t, eu, us) - results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), nil, "", productionTargets(t, svc)) + results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), nil, routing.ExecutionTarget{}, productionTargets(t, svc)) require.NoError(t, err) require.Len(t, results, 2) @@ -144,7 +151,7 @@ func TestPlanDeploymentDiffs_PerDeploymentErrorIsCaptured(t *testing.T) { us := &mockTernClient{planDiffErr: errors.New("deployment unreachable")} svc := twoDeploymentService(t, eu, us) - results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), nil, "", productionTargets(t, svc)) + results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), nil, routing.ExecutionTarget{}, productionTargets(t, svc)) require.NoError(t, err, "a single deployment failure must not abort the rollup") require.Len(t, results, 2) @@ -206,7 +213,7 @@ func TestPlanDeploymentDiffs_PreservesShards(t *testing.T) { }}, } - results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, "eu", productionTargets(t, svc)) + results, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, productionMember("eu"), productionTargets(t, svc)) require.NoError(t, err) require.Len(t, results, 2) @@ -235,7 +242,7 @@ func TestPlanDeploymentDiffs_PrimaryDeploymentMismatchFailsClosed(t *testing.T) primaryPlan := &ternv1.PlanResponse{PlanId: "plan_eu", Engine: ternv1.Engine_ENGINE_SPIRIT} // targets[0] is "eu", but the reviewed plan was created against "us". - _, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, "us", productionTargets(t, svc)) + _, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, productionMember("us"), productionTargets(t, svc)) require.Error(t, err) assert.Contains(t, err.Error(), "primary invariant violated") assert.Nil(t, eu.planDiffReq, "must fail before diffing any deployment") @@ -252,7 +259,7 @@ func TestPlanDeploymentDiffs_PrimaryPlanWithoutOriginFailsClosed(t *testing.T) { primaryPlan := &ternv1.PlanResponse{PlanId: "plan_eu", Engine: ternv1.Engine_ENGINE_SPIRIT} - _, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, "", productionTargets(t, svc)) + _, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), primaryPlan, routing.ExecutionTarget{}, productionTargets(t, svc)) require.Error(t, err) assert.Contains(t, err.Error(), "no origin deployment") } @@ -262,7 +269,7 @@ func TestPlanDeploymentDiffs_PrimaryPlanWithoutOriginFailsClosed(t *testing.T) { func TestPlanDeploymentDiffs_NoTargetsError(t *testing.T) { svc := twoDeploymentService(t, &mockTernClient{}, &mockTernClient{}) - _, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), nil, "", nil) + _, err := svc.PlanDeploymentDiffs(t.Context(), planDiffReq(t), nil, routing.ExecutionTarget{}, nil) require.Error(t, err) assert.Contains(t, err.Error(), "no deployment targets") } diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index e319258d2..7a3025c8c 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -819,10 +819,12 @@ func (s *Service) ExecutePlanProto(ctx context.Context, req PlanRequest) (*ternv } planResp := planResponseFromProto(resp) - // Record the primary deployment this plan was created against so the + // Record the primary rollout member this plan was created against so the // review-time drift rollup can verify the baseline still maps to the primary - // at rollup time. + // at rollup time. Both halves are needed: one deployment can address several + // targets, so the deployment alone does not identify the member. planResp.Deployment = deployment + planResp.Target = resolvedTarget.Target return resp, planResp, nil } diff --git a/pkg/api/plan_review_drift.go b/pkg/api/plan_review_drift.go index be99eec58..2bfc7f519 100644 --- a/pkg/api/plan_review_drift.go +++ b/pkg/api/plan_review_drift.go @@ -7,6 +7,7 @@ import ( ternv1 "github.com/block/schemabot/pkg/proto/ternv1" "github.com/block/schemabot/pkg/metrics" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/tern" ) @@ -17,33 +18,28 @@ import ( // primaryPlan is the just-reviewed primary plan proto, reused as the rollup's // baseline so the comparison is against exactly what the user reviewed rather // than a fresh read of the primary's live schema (which could have drifted and -// tripped a spurious primary-vs-primary mismatch). primaryDeployment is the -// deployment that plan was created against; the producer fails closed if it no -// longer maps to rollout index 0 at rollup time. +// tripped a spurious primary-vs-primary mismatch). primaryMember is the +// deployment and target that plan was created against; the producer fails +// closed if that pair no longer maps to rollout index 0 at rollup time. // -// The database/environment is resolved once here to the configured deployment -// set in rollout order, then shared with the producer. The resolved order is -// also passed to RollupDeploymentDiffs as the expected set so the rollup can -// enforce that the producer returned one diff per deployment in that order — a -// missing, extra, or reordered result fails closed rather than being mistaken -// for agreement. The returned rollup is Clean only when every deployment -// matches. -func (s *Service) RollupReviewTimeDrift(ctx context.Context, req PlanRequest, primaryPlan *ternv1.PlanResponse, primaryDeployment string) (PlanRollup, error) { +// The database/environment is resolved once here to the configured member set +// in rollout order, then shared with the producer. The resolved order is also +// passed to RollupDeploymentDiffs as the expected set so the rollup can enforce +// that the producer returned one diff per member in that order — a missing, +// extra, or reordered result fails closed rather than being mistaken for +// agreement. The returned rollup is Clean only when every member matches. +func (s *Service) RollupReviewTimeDrift(ctx context.Context, req PlanRequest, primaryPlan *ternv1.PlanResponse, primaryMember routing.ExecutionTarget) (PlanRollup, error) { targets, err := s.config.ResolveDatabaseTargets(req.Database, req.Environment) if err != nil { return PlanRollup{}, fmt.Errorf("resolve deployment targets for %s/%s: %w", req.Database, req.Environment, err) } - expectedDeployments := make([]string, len(targets)) - for i, t := range targets { - expectedDeployments[i] = t.Deployment - } - diffs, err := s.PlanDeploymentDiffs(ctx, req, primaryPlan, primaryDeployment, targets) + diffs, err := s.PlanDeploymentDiffs(ctx, req, primaryPlan, primaryMember, targets) if err != nil { return PlanRollup{}, fmt.Errorf("plan deployment diffs for %s/%s: %w", req.Database, req.Environment, err) } - rollup, err := RollupDeploymentDiffs(diffs, expectedDeployments) + rollup, err := RollupDeploymentDiffs(diffs, targets) if err != nil { return PlanRollup{}, fmt.Errorf("roll up deployment diffs for %s/%s: %w", req.Database, req.Environment, err) } diff --git a/pkg/api/plan_review_drift_test.go b/pkg/api/plan_review_drift_test.go index 3e2257cfe..124b0f9b5 100644 --- a/pkg/api/plan_review_drift_test.go +++ b/pkg/api/plan_review_drift_test.go @@ -37,7 +37,7 @@ func TestRollupReviewTimeDrift_CleanWhenAllMatch(t *testing.T) { us := &mockTernClient{planDiffResp: alterUsersDiff(ddl)} svc := twoDeploymentService(t, eu, us) - rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewedUsersPlan(ddl), "eu") + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewedUsersPlan(ddl), productionMember("eu")) require.NoError(t, err) assert.True(t, rollup.Clean, "matching deployments must roll up clean") require.Len(t, rollup.Entries, 2) @@ -55,7 +55,7 @@ func TestRollupReviewTimeDrift_DivergingDeploymentBlocks(t *testing.T) { svc := twoDeploymentService(t, eu, us) reviewed := reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)") - rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewed, "eu") + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewed, productionMember("eu")) require.NoError(t, err) assert.False(t, rollup.Clean, "a diverging deployment must block the rollup") require.Len(t, rollup.Entries, 2) @@ -73,7 +73,7 @@ func TestRollupReviewTimeDrift_UnreachableDeploymentBlocks(t *testing.T) { us := &mockTernClient{planDiffErr: errors.New("us unreachable")} svc := twoDeploymentService(t, eu, us) - rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewedUsersPlan(ddl), "eu") + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewedUsersPlan(ddl), productionMember("eu")) require.NoError(t, err) assert.False(t, rollup.Clean, "an unreachable deployment must block the rollup") require.Len(t, rollup.Entries, 2) @@ -91,7 +91,7 @@ func TestRollupReviewTimeDrift_UnresolvedTargetsError(t *testing.T) { req := planDiffReq(t) req.Database = "unknown" - _, err := svc.RollupReviewTimeDrift(t.Context(), req, reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), "eu") + _, err := svc.RollupReviewTimeDrift(t.Context(), req, reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), productionMember("eu")) require.Error(t, err) } diff --git a/pkg/api/plan_rollup.go b/pkg/api/plan_rollup.go index bbf7b6d02..0b12c822f 100644 --- a/pkg/api/plan_rollup.go +++ b/pkg/api/plan_rollup.go @@ -3,6 +3,7 @@ package api import ( "fmt" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/schema" "github.com/block/schemabot/pkg/tern" ) @@ -58,31 +59,35 @@ type PlanRollup struct { Clean bool } -// RollupDeploymentDiffs classifies each deployment's review-time diff against the -// reviewed primary plan and reports whether the rollup is clean. +// RollupDeploymentDiffs classifies each rollout member's review-time diff +// against the reviewed primary plan and reports whether the rollup is clean. // -// expectedDeployments is the configured deployment set in rollout order, primary -// first — the same order PlanDeploymentDiffs produces. The diffs must match it -// positionally: this turns the producer's structural convention (primary first, -// one entry per configured deployment) into an enforced contract, so a -// reordered, short, or otherwise mismatched result is rejected rather than -// letting a missing or misidentified deployment silently pass the gate. +// expectedMembers is the configured member set in rollout order, primary first +// — the same order PlanDeploymentDiffs produces. The diffs must match it +// positionally on both deployment and target: this turns the producer's +// structural convention (primary first, one entry per configured member) into +// an enforced contract, so a reordered, short, or otherwise mismatched result +// is rejected rather than letting a missing or misidentified member silently +// pass the gate. Matching on the deployment alone would not be enough, because +// one deployment can address several targets and they would be +// indistinguishable. // // The primary (index 0) is the reviewed baseline and classifies Match against -// itself; every other deployment is compared to it with tern.CompareChangeSets. +// itself; every other member is compared to it with tern.CompareChangeSets. // The result fails closed: a contract mismatch, a primary baseline that errored -// or is otherwise unusable, or any deployment that errored or diverged makes the +// or is otherwise unusable, or any member that errored or diverged makes the // rollup not Clean. -func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedDeployments []string) (PlanRollup, error) { - if len(expectedDeployments) == 0 { - return PlanRollup{}, fmt.Errorf("no expected deployments to roll up") +func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing.ExecutionTarget) (PlanRollup, error) { + if len(expectedMembers) == 0 { + return PlanRollup{}, fmt.Errorf("no expected rollout members to roll up") } - if len(diffs) != len(expectedDeployments) { - return PlanRollup{}, fmt.Errorf("expected %d deployment diffs in rollout order, got %d", len(expectedDeployments), len(diffs)) + if len(diffs) != len(expectedMembers) { + return PlanRollup{}, fmt.Errorf("expected %d member diffs in rollout order, got %d", len(expectedMembers), len(diffs)) } - for i, name := range expectedDeployments { - if diffs[i].Deployment != name { - return PlanRollup{}, fmt.Errorf("deployment diff %d is %q, expected %q; diffs must be in rollout order with the primary first and every configured deployment present", i, diffs[i].Deployment, name) + for i, member := range expectedMembers { + got := routing.ExecutionTarget{Deployment: diffs[i].Deployment, Target: diffs[i].Target} + if got.Deployment != member.Deployment || got.Target != member.Target { + return PlanRollup{}, fmt.Errorf("member diff %d is %q, expected %q; diffs must be in rollout order with the primary first and every configured member present", i, got.MemberID(), member.MemberID()) } } diff --git a/pkg/api/plan_rollup_test.go b/pkg/api/plan_rollup_test.go index 199b65f49..19ebc9dd9 100644 --- a/pkg/api/plan_rollup_test.go +++ b/pkg/api/plan_rollup_test.go @@ -8,6 +8,7 @@ import ( "github.com/stretchr/testify/require" ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/routing" ) func rollupAlterUsers(ddl string) *ternv1.SchemaChange { @@ -23,22 +24,38 @@ func rollupAlterUsers(ddl string) *ternv1.SchemaChange { } func rollupDeployment(name string, changes ...*ternv1.SchemaChange) DeploymentPlanDiff { + return rollupMember(name, name, changes...) +} + +// rollupMember builds a diff for one rollout member, for cases where the +// deployment and target differ. +func rollupMember(deployment, target string, changes ...*ternv1.SchemaChange) DeploymentPlanDiff { return DeploymentPlanDiff{ DatabaseType: "vitess", - Deployment: name, - Target: name, + Deployment: deployment, + Target: target, Changes: changes, } } -// rollupNames returns the deployment names of diffs in order, the expected -// deployment contract PlanDeploymentDiffs would produce for them. -func rollupNames(diffs []DeploymentPlanDiff) []string { - names := make([]string, len(diffs)) +// rollupMembers returns the rollout members of diffs in order, the expected +// member contract PlanDeploymentDiffs would produce for them. +func rollupMembers(diffs []DeploymentPlanDiff) []routing.ExecutionTarget { + members := make([]routing.ExecutionTarget, len(diffs)) for i, d := range diffs { - names[i] = d.Deployment + members[i] = routing.ExecutionTarget{Deployment: d.Deployment, Target: d.Target} + } + return members +} + +// rollupMemberList builds an expected member set from "deployment/target" +// pairs, for contract cases that deliberately disagree with the diffs. +func rollupMemberList(pairs ...[2]string) []routing.ExecutionTarget { + members := make([]routing.ExecutionTarget, len(pairs)) + for i, p := range pairs { + members[i] = routing.ExecutionTarget{Deployment: p[0], Target: p[1]} } - return names + return members } // When every deployment would plan exactly the reviewed changes, the rollup is @@ -50,7 +67,7 @@ func TestRollupDeploymentDiffs_AllMatchIsClean(t *testing.T) { rollupDeployment("au", rollupAlterUsers(change)), rollupDeployment("us", rollupAlterUsers(change)), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 3) @@ -66,7 +83,7 @@ func TestRollupDeploymentDiffs_DivergenceBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), rollupDeployment("au", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `phone` varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentMatch, rollup.Entries[0].Class) @@ -84,7 +101,7 @@ func TestRollupDeploymentDiffs_ProducerErrorBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers(change)), errored, } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentMatch, rollup.Entries[0].Class) @@ -99,7 +116,7 @@ func TestRollupDeploymentDiffs_ComparisonErrorBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), rollupDeployment("au", rollupAlterUsers("not valid sql")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) @@ -115,7 +132,7 @@ func TestRollupDeploymentDiffs_UnusablePrimaryBlocksAll(t *testing.T) { primary, rollupDeployment("au", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentErrored, rollup.Entries[0].Class) @@ -135,7 +152,7 @@ func TestRollupDeploymentDiffs_SingleDeploymentClean(t *testing.T) { diffs := []DeploymentPlanDiff{ rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 1) @@ -149,7 +166,7 @@ func TestRollupDeploymentDiffs_MalformedSingleDeploymentBaselineBlocks(t *testin diffs := []DeploymentPlanDiff{ rollupDeployment("eu", rollupAlterUsers("not valid sql")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) require.Len(t, rollup.Entries, 1) @@ -183,7 +200,7 @@ func TestRollupDeploymentDiffs_PostgresDialectClean(t *testing.T) { rollupPostgresDeployment("eu", change()), rollupPostgresDeployment("us", change()), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 2) @@ -199,7 +216,7 @@ func TestRollupDeploymentDiffs_UnregisteredPrimaryDialectBlocks(t *testing.T) { primary := rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")) primary.DatabaseType = "oracle" diffs := []DeploymentPlanDiff{primary} - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) require.Len(t, rollup.Entries, 1) @@ -219,7 +236,7 @@ func TestRollupDeploymentDiffs_MySQLFamilyTypesShareDialect(t *testing.T) { rollupDeployment("eu", rollupAlterUsers(change)), mysqlDeployment, } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 2) @@ -236,7 +253,7 @@ func TestRollupDeploymentDiffs_MixedDialectBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), rollupPostgresDeployment("us", rollupAlterUsers("ALTER TABLE users ADD COLUMN email varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupNames(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentMatch, rollup.Entries[0].Class) @@ -256,15 +273,42 @@ func TestRollupDeploymentDiffs_ContractMismatchErrors(t *testing.T) { } t.Run("wrong primary", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, []string{"au", "eu"}) + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"au", "au"}, [2]string{"eu", "eu"})) require.Error(t, err) }) - t.Run("missing deployment", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, []string{"eu", "au", "us"}) + t.Run("missing member", func(t *testing.T) { + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"eu", "eu"}, [2]string{"au", "au"}, [2]string{"us", "us"})) require.Error(t, err) }) t.Run("extra diff", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, []string{"eu"}) + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"eu", "eu"})) + require.Error(t, err) + }) +} + +// Members of one deployment are distinguished by their target, so a result +// carrying the right deployments against the wrong targets is a contract +// mismatch rather than a silent pass. Without the target half of the member +// identity these two entries would be indistinguishable. +func TestRollupDeploymentDiffs_SameDeploymentDifferentTargets(t *testing.T) { + change := "ALTER TABLE `users` ADD COLUMN `email` varchar(255)" + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001", rollupAlterUsers(change)), + rollupMember("cake", "orders-002", rollupAlterUsers(change)), + } + + t.Run("matching members roll up clean", func(t *testing.T) { + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + require.NoError(t, err) + assert.True(t, rollup.Clean) + require.Len(t, rollup.Entries, 2) + assert.Equal(t, "orders-001", rollup.Entries[0].Target) + assert.Equal(t, "orders-002", rollup.Entries[1].Target) + }) + + t.Run("swapped targets are a contract mismatch", func(t *testing.T) { + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"cake", "orders-002"}, [2]string{"cake", "orders-001"})) require.Error(t, err) + assert.ErrorContains(t, err, "cake/orders-001") }) } diff --git a/pkg/apitypes/apitypes.go b/pkg/apitypes/apitypes.go index c3de7eb12..284b99e64 100644 --- a/pkg/apitypes/apitypes.go +++ b/pkg/apitypes/apitypes.go @@ -536,12 +536,15 @@ type PlanResponse struct { Database string `json:"database,omitempty"` DatabaseType string `json:"database_type,omitempty"` Environment string `json:"environment,omitempty"` - // Deployment is the primary deployment this plan was created against - // (rollout index 0 at plan time). The review-time drift rollup carries it - // forward so it can verify the plan's baseline still maps to the primary at - // rollup time, rather than trusting that current config re-resolves the same - // primary. + // Deployment and Target together identify the primary rollout member this + // plan was created against (rollout index 0 at plan time). The review-time + // drift rollup carries both forward so it can verify the plan's baseline + // still maps to the primary at rollup time, rather than trusting that + // current config re-resolves the same primary. The deployment alone is not + // sufficient: one deployment can address several targets, so a member is + // identified by the pair. Deployment string `json:"deployment,omitempty"` + Target string `json:"target,omitempty"` Engine string `json:"engine"` Changes []*SchemaChangeResponse `json:"changes"` LintResults []*LintViolationResponse `json:"lint_violations"` diff --git a/pkg/routing/resolver.go b/pkg/routing/resolver.go index 8c273d287..34777c014 100644 --- a/pkg/routing/resolver.go +++ b/pkg/routing/resolver.go @@ -23,6 +23,15 @@ type ExecutionTarget struct { Target string } +// MemberID is the rollout-member identity of this execution target: the +// deployment it routes through and the target it addresses, together. A +// deployment name alone does not identify a member, because one deployment can +// address several targets; callers that compare, order, or report members must +// use this pair rather than the deployment. +func (t ExecutionTarget) MemberID() string { + return t.Deployment + "/" + t.Target +} + // Resolver resolves logical SchemaBot targets to concrete execution targets. type Resolver interface { ResolveTargets(ctx context.Context, req Request) ([]ExecutionTarget, error) diff --git a/pkg/webhook/plan.go b/pkg/webhook/plan.go index 33cd8771a..840a66d98 100644 --- a/pkg/webhook/plan.go +++ b/pkg/webhook/plan.go @@ -12,6 +12,7 @@ import ( "github.com/block/schemabot/pkg/apitypes" ghclient "github.com/block/schemabot/pkg/github" "github.com/block/schemabot/pkg/metrics" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/storage" "github.com/block/schemabot/pkg/ui" "github.com/block/schemabot/pkg/webhook/action" @@ -141,7 +142,7 @@ func (h *Handler) handlePlanCommand(w http.ResponseWriter, repo string, pr int, // Roll up every deployment's diff against the reviewed plan so drift on a // non-primary deployment fails the check closed at review time. - drift, driftPreview := h.reviewTimeDrift(ctx, planReq, planProto, planResp.Deployment, repo, pr) + drift, driftPreview := h.reviewTimeDrift(ctx, planReq, planProto, routing.ExecutionTarget{Deployment: planResp.Deployment, Target: planResp.Target}, repo, pr) // Build plan comment data commentData := buildPlanCommentData(schemaResult, planResp, environment, tenant, requestedBy, h.agentHint()) @@ -438,7 +439,7 @@ func (h *Handler) handleMultiEnvPlan(repo string, pr int, databaseName, tenant s // Roll up every deployment's diff against the reviewed plan so drift on a // non-primary deployment fails the check closed at review time. - drift, driftPreview := h.reviewTimeDrift(ctx, planReq, planProto, planResp.Deployment, repo, pr) + drift, driftPreview := h.reviewTimeDrift(ctx, planReq, planProto, routing.ExecutionTarget{Deployment: planResp.Deployment, Target: planResp.Target}, repo, pr) // Store per-database check record per environment var recoveredApplyOwnedCheckState bool diff --git a/pkg/webhook/plan_drift.go b/pkg/webhook/plan_drift.go index d4e1fc28f..cc353080a 100644 --- a/pkg/webhook/plan_drift.go +++ b/pkg/webhook/plan_drift.go @@ -7,6 +7,7 @@ import ( "github.com/block/schemabot/pkg/api" ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/tern" "github.com/block/schemabot/pkg/webhook/templates" ) @@ -35,15 +36,16 @@ import ( // // primaryPlan is the reviewed primary plan proto returned by // executePlanProtoWithTransientRetry, reused as the rollup baseline so the -// comparison is against exactly what was reviewed. -func (h *Handler) reviewTimeDrift(ctx context.Context, planReq api.PlanRequest, primaryPlan *ternv1.PlanResponse, primaryDeployment string, repo string, pr int) (reviewDriftOutcome, *templates.DeploymentDriftData) { +// comparison is against exactly what was reviewed. primaryMember is the +// deployment and target that plan was created against. +func (h *Handler) reviewTimeDrift(ctx context.Context, planReq api.PlanRequest, primaryPlan *ternv1.PlanResponse, primaryMember routing.ExecutionTarget, repo string, pr int) (reviewDriftOutcome, *templates.DeploymentDriftData) { if len(primaryPlan.GetErrors()) > 0 { h.logger.Debug("skipping review-time drift rollup: primary plan reported errors", "repo", repo, "pr", pr, "database", planReq.Database, "environment", planReq.Environment) return reviewDriftOutcome{state: driftNotEvaluated}, nil } - rollup, err := h.service.RollupReviewTimeDrift(ctx, planReq, primaryPlan, primaryDeployment) + rollup, err := h.service.RollupReviewTimeDrift(ctx, planReq, primaryPlan, primaryMember) if err != nil { h.logger.Error("review-time drift rollup failed; the plan check will block the PR closed", "repo", repo, From 0327376741e0a33d5aa6cfff47ab2bf9afed12ba Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 14:56:34 -0400 Subject: [PATCH 02/34] feat(storage): let one apply's members carry their own plans An apply whose members are planned together shares one plan, and that is still the common case. When each member is planned against its own live schema, though, there is no single plan for the apply to point at. Add a nullable apply_operations.plan_id so a member can name the plan it executes, with PlanIDForOperation resolving to the parent apply's plan when the member has none. An operation with no plan on either row is not executable and now errors rather than resolving to a row ID no plan has. TargetOperationKey names the operation key for one target's work when a single apply addresses several targets, composing with the shard-scoped key so a sharded target still gets one key per shard. Co-Authored-By: Claude Fable 5 --- pkg/schema/mysql/apply_operations.sql | 1 + pkg/schema/postgres/apply_operations.sql | 1 + .../internal/sqlstore/apply_operations.go | 14 ++-- pkg/storage/internal/sqlstore/sql_helpers.go | 10 +++ pkg/storage/storagetest/apply_operations.go | 46 +++++++++++++ pkg/storage/types.go | 53 +++++++++++++++ pkg/storage/types_test.go | 65 +++++++++++++++++++ 7 files changed, 185 insertions(+), 5 deletions(-) diff --git a/pkg/schema/mysql/apply_operations.sql b/pkg/schema/mysql/apply_operations.sql index d08629c51..f91e14404 100644 --- a/pkg/schema/mysql/apply_operations.sql +++ b/pkg/schema/mysql/apply_operations.sql @@ -1,6 +1,7 @@ CREATE TABLE `apply_operations` ( `id` bigint unsigned NOT NULL AUTO_INCREMENT, `apply_id` bigint unsigned NOT NULL, + `plan_id` bigint unsigned DEFAULT NULL, `deployment` varchar(255) NOT NULL, `operation_key` varchar(255) NOT NULL DEFAULT '', `operation_kind` varchar(32) NOT NULL DEFAULT 'work', diff --git a/pkg/schema/postgres/apply_operations.sql b/pkg/schema/postgres/apply_operations.sql index 7818470a7..d67229487 100644 --- a/pkg/schema/postgres/apply_operations.sql +++ b/pkg/schema/postgres/apply_operations.sql @@ -1,6 +1,7 @@ CREATE TABLE apply_operations ( id bigint GENERATED BY DEFAULT AS IDENTITY, apply_id bigint NOT NULL, + plan_id bigint DEFAULT NULL, deployment varchar(255) NOT NULL, operation_key varchar(255) NOT NULL DEFAULT '', operation_kind varchar(32) NOT NULL DEFAULT 'work', diff --git a/pkg/storage/internal/sqlstore/apply_operations.go b/pkg/storage/internal/sqlstore/apply_operations.go index 57d511bcb..46ce86acd 100644 --- a/pkg/storage/internal/sqlstore/apply_operations.go +++ b/pkg/storage/internal/sqlstore/apply_operations.go @@ -22,7 +22,7 @@ import ( ) // applyOperationColumns lists all columns for SELECT queries. -const applyOperationColumns = `id, apply_id, deployment, operation_key, operation_kind, target, external_id, external_operation_id, state, error_message, +const applyOperationColumns = `id, apply_id, plan_id, deployment, operation_key, operation_kind, target, external_id, external_operation_id, state, error_message, cutover_policy, on_failure, attempt, started_at, completed_at, lease_owner, lease_token, lease_acquired_at, engine_resume_context, engine_resume_metadata, progress_metadata, created_at, updated_at` @@ -82,11 +82,11 @@ func insertApplyOperation(ctx context.Context, exec queryExecer, identity identi id, err := identity.InsertID(ctx, exec, ` INSERT INTO apply_operations ( - apply_id, deployment, operation_key, operation_kind, target, external_id, external_operation_id, state, error_message, cutover_policy, on_failure, + apply_id, plan_id, deployment, operation_key, operation_kind, target, external_id, external_operation_id, state, error_message, cutover_policy, on_failure, started_at, completed_at, engine_resume_context, engine_resume_metadata - ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) `, - ad.ApplyID, ad.Deployment, ad.OperationKey, operationKind, ad.Target, nullString(ad.ExternalID), nullString(ad.ExternalOperationID), stateVal, nullString(ad.ErrorMessage), cutoverPolicy, onFailure, + ad.ApplyID, nullInt64(ad.PlanID), ad.Deployment, ad.OperationKey, operationKind, ad.Target, nullString(ad.ExternalID), nullString(ad.ExternalOperationID), stateVal, nullString(ad.ErrorMessage), cutoverPolicy, onFailure, ad.StartedAt, ad.CompletedAt, nullString(ad.EngineResumeContext), nullString(ad.EngineResumeMetadata), ) if err != nil { @@ -2281,15 +2281,19 @@ func scanApplyOperationInto(s scanner) (*storage.ApplyOperation, error) { var externalOperationID sql.NullString var engineResumeContext, engineResumeMetadata, progressMetadata sql.NullString var startedAt, completedAt, leaseAcquiredAt sql.NullTime + var planID sql.NullInt64 if err := s.Scan( - &ad.ID, &ad.ApplyID, &ad.Deployment, &ad.OperationKey, &ad.OperationKind, &ad.Target, &externalID, &externalOperationID, &ad.State, &errMsg, + &ad.ID, &ad.ApplyID, &planID, &ad.Deployment, &ad.OperationKey, &ad.OperationKind, &ad.Target, &externalID, &externalOperationID, &ad.State, &errMsg, &ad.CutoverPolicy, &ad.OnFailure, &ad.Attempt, &startedAt, &completedAt, &ad.LeaseOwner, &ad.LeaseToken, &leaseAcquiredAt, &engineResumeContext, &engineResumeMetadata, &progressMetadata, &ad.CreatedAt, &ad.UpdatedAt, ); err != nil { return nil, err } + if planID.Valid { + ad.PlanID = planID.Int64 + } if errMsg.Valid { ad.ErrorMessage = errMsg.String } diff --git a/pkg/storage/internal/sqlstore/sql_helpers.go b/pkg/storage/internal/sqlstore/sql_helpers.go index eb855efd4..b43b2d7d7 100644 --- a/pkg/storage/internal/sqlstore/sql_helpers.go +++ b/pkg/storage/internal/sqlstore/sql_helpers.go @@ -33,6 +33,16 @@ func nullString(s string) sql.NullString { return sql.NullString{String: s, Valid: true} } +// nullInt64 returns a sql.NullInt64 for a row reference held as an int64, where +// zero means "no reference" and must be stored as NULL rather than as a row ID +// of 0 that no row can have. +func nullInt64(v int64) sql.NullInt64 { + if v == 0 { + return sql.NullInt64{} + } + return sql.NullInt64{Int64: v, Valid: true} +} + // nullInt64Ptr returns a sql.NullInt64 for a *int64 value. func nullInt64Ptr(v *int64) sql.NullInt64 { if v == nil { diff --git a/pkg/storage/storagetest/apply_operations.go b/pkg/storage/storagetest/apply_operations.go index 65eeecf07..701c12680 100644 --- a/pkg/storage/storagetest/apply_operations.go +++ b/pkg/storage/storagetest/apply_operations.go @@ -455,6 +455,52 @@ func TestApplyOperations(t *testing.T, h Harness) { assert.NotEmpty(t, claims[0].LeaseToken) }) + // PlanIDRoundTrip verifies which plan each member of an apply executes. A + // member planned together with its siblings stores no plan of its own and + // resolves to the apply's plan, while a member planned against its own live + // schema carries that plan on its row and keeps it across a claim. + t.Run("PlanIDRoundTrip", func(t *testing.T) { + ctx := t.Context() + store := h.NewStorage(t) + lock := CreateLock(t, store, "operation_plan_db", storage.DatabaseTypeMySQL) + apply := CreateApply(t, store, lock, "apply_operation_plan", 910) + + sharedID := createOperation(t, store, apply.ID, "region-a", "schema") + ownID, err := store.ApplyOperations().Insert(ctx, &storage.ApplyOperation{ + ApplyID: apply.ID, Deployment: "region-b", OperationKey: "schema", PlanID: 911, + }) + require.NoError(t, err) + + shared, err := store.ApplyOperations().Get(ctx, sharedID) + require.NoError(t, err) + require.NotNil(t, shared) + assert.Zero(t, shared.PlanID, "a member planned with its siblings stores no plan of its own") + sharedPlan, err := storage.PlanIDForOperation(apply, shared) + require.NoError(t, err) + assert.Equal(t, int64(910), sharedPlan) + + own, err := store.ApplyOperations().Get(ctx, ownID) + require.NoError(t, err) + require.NotNil(t, own) + assert.Equal(t, int64(911), own.PlanID) + ownPlan, err := storage.PlanIDForOperation(apply, own) + require.NoError(t, err) + assert.Equal(t, int64(911), ownPlan) + + claimed, err := store.ApplyOperations().FindNextApplyOperation(ctx, "driver-a") + require.NoError(t, err) + require.NotNil(t, claimed) + assert.Equal(t, sharedID, claimed.ID) + assert.Zero(t, claimed.PlanID) + require.NoError(t, store.ApplyOperations().MarkCompleted(ctx, sharedID)) + + claimed, err = store.ApplyOperations().FindNextApplyOperation(ctx, "driver-b") + require.NoError(t, err) + require.NotNil(t, claimed) + assert.Equal(t, ownID, claimed.ID) + assert.Equal(t, int64(911), claimed.PlanID, "a member's own plan survives the claim") + }) + t.Run("FindNextApplyOperation_DBError", func(t *testing.T) { store := h.NewUnreachableStorage(t) _, err := store.ApplyOperations().FindNextApplyOperation(t.Context(), "driver") diff --git a/pkg/storage/types.go b/pkg/storage/types.go index d9f921905..169fd0d1b 100644 --- a/pkg/storage/types.go +++ b/pkg/storage/types.go @@ -330,6 +330,48 @@ func ShardOperationKey(namespace, shard, table string) string { return namespace + "/" + shard + "/" + table } +// TargetOperationKey builds the operation key for one target's work when a +// single apply addresses several targets. The target leads the key because it +// is the coarsest scope: an apply's members are its targets, and any narrower +// division of one target's work hangs off it. +// +// scopedKey is the narrower key within that target, or empty for work that +// covers the whole target. Passing a ShardOperationKey composes the two, so a +// sharded target yields one key per (target, namespace, shard, table): +// +// TargetOperationKey("orders-002", "") -> "orders-002" +// TargetOperationKey("orders-002", ShardOperationKey("main", "-80", "t")) -> "orders-002/main/-80/t" +func TargetOperationKey(target, scopedKey string) string { + if scopedKey == "" { + return target + } + return target + "/" + scopedKey +} + +// PlanIDForOperation resolves which plan an operation executes: its own when it +// has one, and its parent apply's otherwise. Members of one apply share the +// apply's plan when they are planned together, and carry their own plan when +// each was planned against its own live schema. +// +// An operation with no plan on either row is not executable — a dispatch would +// have no DDL to run — so that case is an error rather than a zero return the +// caller might mistake for a valid plan. +func PlanIDForOperation(apply *Apply, op *ApplyOperation) (int64, error) { + if op == nil { + return 0, fmt.Errorf("resolve plan for operation: no operation") + } + if op.PlanID != 0 { + return op.PlanID, nil + } + if apply == nil { + return 0, fmt.Errorf("resolve plan for operation %d: operation has no plan and its apply was not loaded", op.ID) + } + if apply.PlanID == 0 { + return 0, fmt.Errorf("resolve plan for operation %d: neither the operation nor apply %s names a plan", op.ID, apply.ApplyIdentifier) + } + return apply.PlanID, nil +} + // EngineForType returns the engine name for a database type. func EngineForType(dbType string) string { switch dbType { @@ -859,6 +901,17 @@ type ApplyOperation struct { // OperationKey. ApplyID int64 + // PlanID points to the plans.id this operation executes, when the operation + // has a plan of its own. Zero means the operation executes its parent + // apply's plan; resolve it with PlanIDForOperation rather than reading this + // field directly, so the fallback is applied consistently. + // + // An operation carries its own plan when the members of one apply do not + // share a single desired-vs-live diff — each member is planned against its + // own live schema, so each gets its own persisted plan row. Members that do + // share the parent's plan leave this zero. + PlanID int64 + // Deployment is the Tern deployment name this child row targets // (e.g. "region-a", "payments-eu"). Drawn from the resolved // environment-level deployments map in server config. diff --git a/pkg/storage/types_test.go b/pkg/storage/types_test.go index efdf06528..4c4f0a465 100644 --- a/pkg/storage/types_test.go +++ b/pkg/storage/types_test.go @@ -390,3 +390,68 @@ func TestApplyOperationParseProgressMetadata(t *testing.T) { assert.Nil(t, got) }) } + +// TestTargetOperationKey covers the operation keys an apply that addresses +// several targets stamps on its members: one key per target for whole-target +// work, and a target-led composition when a target's work is further divided by +// shard so no two members of the same apply can collide on a key. +func TestTargetOperationKey(t *testing.T) { + cases := []struct { + name string + target string + scopedKey string + want string + }{ + {"whole-target work keys on the target alone", "orders-002", "", "orders-002"}, + {"shard work hangs off its target", "orders-002", ShardOperationKey("main", "-80", "customers"), "orders-002/main/-80/customers"}, + {"the same shard on another target is a distinct key", "orders-003", ShardOperationKey("main", "-80", "customers"), "orders-003/main/-80/customers"}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.Equal(t, tc.want, TargetOperationKey(tc.target, tc.scopedKey)) + }) + } +} + +// TestPlanIDForOperation covers which plan a member executes: members planned +// together share their apply's plan, a member planned against its own live +// schema carries its own, and a member with neither is not executable and must +// error rather than resolve to a plan ID no row can have. +func TestPlanIDForOperation(t *testing.T) { + apply := &Apply{ApplyIdentifier: "apply-1", PlanID: 7} + + t.Run("member without its own plan executes the apply's", func(t *testing.T) { + planID, err := PlanIDForOperation(apply, &ApplyOperation{ID: 1}) + require.NoError(t, err) + assert.Equal(t, int64(7), planID) + }) + + t.Run("member with its own plan executes that one", func(t *testing.T) { + planID, err := PlanIDForOperation(apply, &ApplyOperation{ID: 2, PlanID: 42}) + require.NoError(t, err) + assert.Equal(t, int64(42), planID) + }) + + t.Run("an unloaded apply cannot supply the fallback", func(t *testing.T) { + _, err := PlanIDForOperation(nil, &ApplyOperation{ID: 3}) + require.Error(t, err) + assert.Contains(t, err.Error(), "was not loaded") + }) + + t.Run("an unloaded apply is irrelevant once the member has its own plan", func(t *testing.T) { + planID, err := PlanIDForOperation(nil, &ApplyOperation{ID: 4, PlanID: 42}) + require.NoError(t, err) + assert.Equal(t, int64(42), planID) + }) + + t.Run("no plan on either row is an error", func(t *testing.T) { + _, err := PlanIDForOperation(&Apply{ApplyIdentifier: "apply-2"}, &ApplyOperation{ID: 5}) + require.Error(t, err) + assert.Contains(t, err.Error(), "apply-2") + }) + + t.Run("no operation is an error", func(t *testing.T) { + _, err := PlanIDForOperation(apply, nil) + require.Error(t, err) + }) +} From 3a54e8e5b20bf841a4a50dae7d15af0e7976bac7 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 17:03:50 -0400 Subject: [PATCH 03/34] fix(api): name an unresolvable operation's plan error by deployment The error identified the operation by its internal numeric row ID, which is not a triage handle an operator can look up. Name the operation by the identifiers that route it instead. Co-Authored-By: Claude Fable 5 --- pkg/storage/types.go | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/pkg/storage/types.go b/pkg/storage/types.go index 169fd0d1b..dc919694b 100644 --- a/pkg/storage/types.go +++ b/pkg/storage/types.go @@ -364,10 +364,10 @@ func PlanIDForOperation(apply *Apply, op *ApplyOperation) (int64, error) { return op.PlanID, nil } if apply == nil { - return 0, fmt.Errorf("resolve plan for operation %d: operation has no plan and its apply was not loaded", op.ID) + return 0, fmt.Errorf("resolve plan for operation on deployment %q (operation key %q): operation has no plan and its apply was not loaded", op.Deployment, op.OperationKey) } if apply.PlanID == 0 { - return 0, fmt.Errorf("resolve plan for operation %d: neither the operation nor apply %s names a plan", op.ID, apply.ApplyIdentifier) + return 0, fmt.Errorf("resolve plan for operation on deployment %q (operation key %q): neither the operation nor apply %s names a plan", op.Deployment, op.OperationKey, apply.ApplyIdentifier) } return apply.PlanID, nil } From 8cad46baba1d0295b015433e5b04266322cb4589 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:03:03 -0400 Subject: [PATCH 04/34] feat(api): let one environment address several targets MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A database that lives on more than one target has no single target to name. Add a targets list, on an environment and on a deployments-map entry alike, and resolve it to one rollout member per target — the deployment and target pair a member is identified by. Validation and routing share one resolver for the list, so a config that validates resolves to exactly the member list validation checked. An entry names either a target or a targets list, never both; a list may not be empty, hold an empty entry, or repeat a target, since one deployment cannot address the same target twice. Members resolve deployments outermost, so a rollout finishes one deployment's targets before moving to the next. Co-Authored-By: Claude Fable 5 --- docs/configuration.md | 41 +++++++++ pkg/api/config.go | 120 +++++++++++++++++++++++---- pkg/api/config_test.go | 183 +++++++++++++++++++++++++++++++++++++++++ 3 files changed, 327 insertions(+), 17 deletions(-) diff --git a/docs/configuration.md b/docs/configuration.md index 2b65f4728..1e5d9d0f4 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -7,6 +7,7 @@ - [Local Mode](#local-mode) - [gRPC Mode](#grpc-mode) - [Multi-Deployment Environments](#multi-deployment-environments) +- [Multi-Target Environment (preview)](#multi-target-environment-preview) - [Environment Order](#environment-order) - [Hybrid Mode](#hybrid-mode) - [Drivers](#drivers) @@ -266,6 +267,46 @@ A multi-deployment environment is gated on every deployment agreeing with one re Every deployment name in `deployment_order` must be lowercase; the server refuses to start otherwise. +## Multi-Target Environment (preview) + +A database that lives on more than one target replaces the scalar `target` with a `targets` list. Each entry is one target the environment addresses, and rollout follows the listed order. + +> **Config surface only in this release.** `targets` resolves to one rollout member per target, but the plan and apply paths still treat every member of an environment the way they treat deployments. + +```yaml +databases: + payments: + type: mysql + environments: + production: + deployment: payments-a + targets: + - payments-001 + - payments-002 +``` + +A `targets` list can also sit inside a `deployments` map entry, for a database whose targets are spread across several deployments: + +```yaml + production: + deployments: + payments-a: + targets: + - payments-001 + - payments-002 + payments-b: + target: payments-003 +``` + +Rules: + +- `targets` is mutually exclusive with `target` at the same level, and with a local `dsn` / `dsn_from`. +- The list MUST contain at least one entry, and no entry may be empty. +- One deployment may not list the same target twice. A rollout member is identified by its deployment and target together, so the same target under two different deployments is two distinct members and is allowed. +- Members resolve deployments outermost: every target of the first deployment, then every target of the next. + +`targets` and `deployments` both fan an environment out across several members, but they mean different things. The deployments of one environment are expected to hold the same schema, so a difference between them is drift to surface. The targets of one environment are each planned on their own, so a difference between them is ordinary. + ## Environment Order Environment availability and promotion order are server-owned. Repository `schemabot.yaml` files identify the database and type; they do not list or opt into environments. SchemaBot resolves the environments for a database from server config: diff --git a/pkg/api/config.go b/pkg/api/config.go index df863d5a6..bd37b2ce0 100644 --- a/pkg/api/config.go +++ b/pkg/api/config.go @@ -1109,9 +1109,26 @@ type EnvironmentConfig struct { DSNFrom *DSNFromConfig `yaml:"dsn_from,omitempty"` // Target is the opaque Tern-facing target identifier for gRPC mode. - // Mutually exclusive with Deployments. + // Mutually exclusive with Targets and with Deployments. Target string `yaml:"target,omitempty"` + // Targets lists the Tern-facing target identifiers this environment + // addresses through a single deployment, for a database that lives on more + // than one target. Rollout follows the listed order. Mutually exclusive with + // Target and with Deployments. + // + // Targets and Deployments both fan an environment out across several + // members, and a member is identified by its deployment and target together + // either way. They differ in what the fan-out means: the deployments of one + // environment are meant to hold the same schema, so a difference between + // them is drift to surface; the targets of one environment are each planned + // on their own, so a difference between them is ordinary and is converged. + // + // Example: + // deployment: region-a + // targets: [orders-001, orders-002] + Targets []string `yaml:"targets,omitempty"` + // Deployment is the lowercase Tern deployment key for gRPC mode. Deployment // names are storage identity keys compared byte-wise across storage dialects. // Mutually exclusive with Deployments. @@ -1577,7 +1594,53 @@ func (c *ServerConfig) SchemaDirHintsForRepo(repo string) (dirs []string, exhaus // the top-level tern_deployments map). type DeploymentTarget struct { // Target is the opaque Tern-facing target identifier for this deployment. + // Mutually exclusive with Targets. Target string `yaml:"target"` + + // Targets lists the Tern-facing target identifiers this deployment + // addresses, for a deployment that reaches more than one. Rollout follows + // the listed order. Mutually exclusive with Target. + // + // Example: + // deployments: + // region-a: + // targets: [orders-001, orders-002] + Targets []string `yaml:"targets,omitempty"` +} + +// resolveTargetList returns the ordered target list for one routing entry — +// either an environment's scalar routing or one entry of its deployments map. +// Validation and routing both call it, so a config that validates resolves to +// exactly the member list validation checked. +// +// An entry names either a single target or a targets list, never both: the two +// spellings would otherwise disagree about how many members the entry has. An +// entry that names neither returns an empty list without an error, leaving the +// caller to report the missing target in its own terms. +func resolveTargetList(what, target string, targets []string) ([]string, error) { + if target != "" && len(targets) > 0 { + return nil, fmt.Errorf("%s cannot configure both target and targets", what) + } + if targets == nil { + if target == "" { + return nil, nil + } + return []string{target}, nil + } + if len(targets) == 0 { + return nil, fmt.Errorf("%s targets list is empty", what) + } + seen := make(map[string]bool, len(targets)) + for i, t := range targets { + if t == "" { + return nil, fmt.Errorf("%s targets entry %d is empty", what, i) + } + if seen[t] { + return nil, fmt.Errorf("%s lists target %q more than once; a rollout member is identified by its deployment and target together, so one deployment cannot address the same target twice", what, t) + } + seen[t] = true + } + return targets, nil } var defaultEnvironmentOrder = []string{"staging", "production"} @@ -1808,7 +1871,7 @@ func (c *ServerConfig) Validate() error { return err } hasDSN := envConfig.HasLocalDSN() - hasScalarRouting := envConfig.Target != "" || envConfig.Deployment != "" + hasScalarRouting := envConfig.Target != "" || envConfig.Targets != nil || envConfig.Deployment != "" hasMapRouting := envConfig.Deployments != nil if len(envConfig.DeploymentOrder) > 0 && !hasMapRouting { return fmt.Errorf("database %q environment %q sets deployment_order without a deployments map", name, env) @@ -1856,7 +1919,11 @@ func (c *ServerConfig) Validate() error { if err := validateIdentifier(fmt.Sprintf("database %q environment %q deployment name", name, env), deployment); err != nil { return err } - if dt.Target == "" { + targets, err := resolveTargetList(fmt.Sprintf("database %q environment %q deployment %q", name, env, deployment), dt.Target, dt.Targets) + if err != nil { + return err + } + if len(targets) == 0 { return fmt.Errorf("database %q environment %q deployment %q missing target", name, env, deployment) } endpoints, ok := c.TernDeployments[deployment] @@ -1870,11 +1937,16 @@ func (c *ServerConfig) Validate() error { continue case !hasScalarRouting: return fmt.Errorf("database %q environment %q missing local DSN or target/deployment(s)", name, env) - case envConfig.Target == "": - return fmt.Errorf("database %q environment %q missing target", name, env) case envConfig.Deployment == "": return fmt.Errorf("database %q environment %q missing deployment", name, env) } + scalarTargets, err := resolveTargetList(fmt.Sprintf("database %q environment %q", name, env), envConfig.Target, envConfig.Targets) + if err != nil { + return err + } + if len(scalarTargets) == 0 { + return fmt.Errorf("database %q environment %q missing target", name, env) + } endpoints, ok := c.TernDeployments[envConfig.Deployment] if !ok { return fmt.Errorf("database %q environment %q references unknown deployment %q", name, env, envConfig.Deployment) @@ -2683,29 +2755,43 @@ func (c *ServerConfig) ResolveDatabaseTargets(database, environment string) ([]r out := make([]routing.ExecutionTarget, 0, len(deployments)) for _, deployment := range deployments { dt := envConfig.Deployments[deployment] - if dt.Target == "" { + targets, err := resolveTargetList(fmt.Sprintf("database %q environment %q deployment %q", database, environment, deployment), dt.Target, dt.Targets) + if err != nil { + return nil, err + } + if len(targets) == 0 { return nil, fmt.Errorf("database %q environment %q deployment %q missing target", database, environment, deployment) } - out = append(out, routing.ExecutionTarget{ - DatabaseType: dbConfig.Type, - Deployment: deployment, - Target: dt.Target, - }) + for _, target := range targets { + out = append(out, routing.ExecutionTarget{ + DatabaseType: dbConfig.Type, + Deployment: deployment, + Target: target, + }) + } } return out, nil } - if envConfig.Target == "" { + targets, err := resolveTargetList(fmt.Sprintf("database %q environment %q", database, environment), envConfig.Target, envConfig.Targets) + if err != nil { + return nil, err + } + if len(targets) == 0 { return nil, fmt.Errorf("database %q environment %q missing server-side target", database, environment) } if envConfig.Deployment == "" { return nil, fmt.Errorf("database %q environment %q missing server-side deployment", database, environment) } - return []routing.ExecutionTarget{{ - DatabaseType: dbConfig.Type, - Deployment: envConfig.Deployment, - Target: envConfig.Target, - }}, nil + out := make([]routing.ExecutionTarget, 0, len(targets)) + for _, target := range targets { + out = append(out, routing.ExecutionTarget{ + DatabaseType: dbConfig.Type, + Deployment: envConfig.Deployment, + Target: target, + }) + } + return out, nil } // IsRepoAllowed returns whether the given repository is permitted to use SchemaBot. diff --git a/pkg/api/config_test.go b/pkg/api/config_test.go index 5aed1eb6f..af3a0caa5 100644 --- a/pkg/api/config_test.go +++ b/pkg/api/config_test.go @@ -2249,6 +2249,109 @@ func TestServerConfig_DeploymentsMapValidation(t *testing.T) { }, tern: baseTern, }, + { + name: "environment targets list is accepted", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Targets: []string{"payments-001", "payments-002"}, + }, + tern: baseTern, + }, + { + name: "environment target and targets together are rejected", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Target: "payments-001", + Targets: []string{"payments-001", "payments-002"}, + }, + tern: baseTern, + wantErrSub: "cannot configure both target and targets", + }, + { + name: "empty environment targets list is rejected", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Targets: []string{}, + }, + tern: baseTern, + wantErrSub: "targets list is empty", + }, + { + name: "empty entry in environment targets is rejected", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Targets: []string{"payments-001", ""}, + }, + tern: baseTern, + wantErrSub: "targets entry 1 is empty", + }, + { + name: "repeated environment target is rejected", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Targets: []string{"payments-001", "payments-001"}, + }, + tern: baseTern, + wantErrSub: `lists target "payments-001" more than once`, + }, + { + name: "environment targets without a deployment is rejected", + envConfig: EnvironmentConfig{ + Targets: []string{"payments-001", "payments-002"}, + }, + tern: baseTern, + wantErrSub: "missing deployment", + }, + { + name: "local DSN together with targets is rejected", + envConfig: EnvironmentConfig{ + DSN: "root@tcp(localhost)/payments", + Deployment: "payments-a", + Targets: []string{"payments-001"}, + }, + tern: baseTern, + wantErrSub: "cannot configure both local DSN and target/deployment(s)", + }, + { + name: "deployment targets list is accepted", + envConfig: EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{ + "payments-a": {Targets: []string{"payments-001", "payments-002"}}, + "payments-b": {Target: "payments-003"}, + }, + }, + tern: baseTern, + }, + { + name: "deployment target and targets together are rejected", + envConfig: EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{ + "payments-a": {Target: "payments-001", Targets: []string{"payments-002"}}, + }, + }, + tern: baseTern, + wantErrSub: `deployment "payments-a" cannot configure both target and targets`, + }, + { + name: "repeated target within one deployment is rejected", + envConfig: EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{ + "payments-a": {Targets: []string{"payments-001", "payments-001"}}, + }, + }, + tern: baseTern, + wantErrSub: `deployment "payments-a" lists target "payments-001" more than once`, + }, + { + name: "the same target under different deployments is accepted", + envConfig: EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{ + "payments-a": {Targets: []string{"payments-001", "payments-002"}}, + "payments-b": {Targets: []string{"payments-001", "payments-002"}}, + }, + }, + tern: baseTern, + }, } for _, tc := range cases { @@ -4778,3 +4881,83 @@ postgres: assert.Equal(t, 45*time.Second, cfg.Postgres.StatementTimeoutOrDefault()) }) } + +// TestServerConfig_ResolveDatabaseTargets_MultiTarget covers an environment +// that addresses several targets. A rollout member is a deployment and a target +// together, so resolution returns one member per target — under a single +// deployment, under each deployment of a deployments map, and in an order that +// walks the deployments outermost so a rollout finishes one deployment before +// moving to the next. +func TestServerConfig_ResolveDatabaseTargets_MultiTarget(t *testing.T) { + cfg := ServerConfig{ + Databases: map[string]DatabaseConfig{ + "scalar": { + Type: "mysql", + Environments: map[string]EnvironmentConfig{ + "production": { + Deployment: "payments-a", + Targets: []string{"payments-002", "payments-001"}, + }, + }, + }, + "mapped": { + Type: "mysql", + Environments: map[string]EnvironmentConfig{ + "production": { + Deployments: map[string]DeploymentTarget{ + "payments-b": {Targets: []string{"payments-003", "payments-004"}}, + "payments-a": {Target: "payments-001"}, + }, + DeploymentOrder: []string{"payments-a", "payments-b"}, + }, + }, + }, + "broken": { + Type: "mysql", + Environments: map[string]EnvironmentConfig{ + "production": { + Deployment: "payments-a", + Target: "payments-001", + Targets: []string{"payments-002"}, + }, + }, + }, + }, + TernDeployments: TernConfig{ + "payments-a": {"production": "tern-a:9090"}, + "payments-b": {"production": "tern-b:9090"}, + }, + } + + t.Run("one deployment fans out to one member per target in listed order", func(t *testing.T) { + got, err := cfg.ResolveDatabaseTargets("scalar", "production") + require.NoError(t, err) + assert.Equal(t, []routing.ExecutionTarget{ + {DatabaseType: "mysql", Deployment: "payments-a", Target: "payments-002"}, + {DatabaseType: "mysql", Deployment: "payments-a", Target: "payments-001"}, + }, got, "targets resolve in the order they are listed, not sorted") + }) + + t.Run("a deployments map fans out deployments outermost", func(t *testing.T) { + got, err := cfg.ResolveDatabaseTargets("mapped", "production") + require.NoError(t, err) + assert.Equal(t, []routing.ExecutionTarget{ + {DatabaseType: "mysql", Deployment: "payments-a", Target: "payments-001"}, + {DatabaseType: "mysql", Deployment: "payments-b", Target: "payments-003"}, + {DatabaseType: "mysql", Deployment: "payments-b", Target: "payments-004"}, + }, got) + }) + + t.Run("the primary is the first member, not the first deployment", func(t *testing.T) { + primary, err := cfg.ResolvePrimaryDatabaseTarget("scalar", "production") + require.NoError(t, err) + assert.Equal(t, "payments-a", primary.Deployment) + assert.Equal(t, "payments-002", primary.Target) + }) + + t.Run("target and targets together fail closed at resolution too", func(t *testing.T) { + _, err := cfg.ResolveDatabaseTargets("broken", "production") + require.Error(t, err) + assert.Contains(t, err.Error(), "cannot configure both target and targets") + }) +} From ee4a5dea16de57eb37e0513564e854ca55aa9d6a Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:26:59 -0400 Subject: [PATCH 05/34] feat(api): gate multi-target on the engine, MySQL only MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Addressing several targets from one database entry has to be represented on every operator-facing surface — the plan comment, progress, the terminal summary — and that presentation is built per engine. Enable the targets spelling per engine as that work lands rather than assuming it everywhere: a targets list on a vitess, strata, or postgres database now fails config validation at startup. The gate is on the spelling, not the member count. A deployments map whose entries each name a single target is not multi-target however many entries it has, and stays available on every engine. Co-Authored-By: Claude Fable 5 --- docs/configuration.md | 3 ++ pkg/api/config.go | 38 +++++++++++++++++ pkg/api/config_test.go | 85 ++++++++++++++++++++++++++++++++++++++ pkg/schema/dialect.go | 9 ++++ pkg/schema/dialect_test.go | 5 +++ 5 files changed, 140 insertions(+) diff --git a/docs/configuration.md b/docs/configuration.md index 1e5d9d0f4..3cf9dc9d9 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -271,6 +271,8 @@ refuses to start otherwise. A database that lives on more than one target replaces the scalar `target` with a `targets` list. Each entry is one target the environment addresses, and rollout follows the listed order. +> **MySQL only.** `targets` is rejected at startup on any other database type. Addressing N targets has to be represented on every operator-facing surface — the plan comment, progress, the terminal summary — and that presentation is built per engine, so the feature is enabled per engine as the work lands. + > **Config surface only in this release.** `targets` resolves to one rollout member per target, but the plan and apply paths still treat every member of an environment the way they treat deployments. ```yaml @@ -300,6 +302,7 @@ A `targets` list can also sit inside a `deployments` map entry, for a database w Rules: +- `targets` requires `type: mysql`. Configuring it on a `vitess`, `strata`, or `postgres` database fails validation at startup. - `targets` is mutually exclusive with `target` at the same level, and with a local `dsn` / `dsn_from`. - The list MUST contain at least one entry, and no entry may be empty. - One deployment may not list the same target twice. A rollout member is identified by its deployment and target together, so the same target under two different deployments is two distinct members and is allowed. diff --git a/pkg/api/config.go b/pkg/api/config.go index bd37b2ce0..a458a8af3 100644 --- a/pkg/api/config.go +++ b/pkg/api/config.go @@ -1608,6 +1608,41 @@ type DeploymentTarget struct { Targets []string `yaml:"targets,omitempty"` } +// UsesTargetsList reports whether an environment spells any of its routing as a +// targets list — at the environment level, or inside one entry of its +// deployments map. That spelling is what makes an environment multi-target: +// several distinct targets under one deployment, each holding its own schema. +// +// A deployments map whose entries each name a single target is not multi-target +// however many entries it has: those members are expected to hold the same +// schema as each other. +func (c EnvironmentConfig) UsesTargetsList() bool { + if c.Targets != nil { + return true + } + for _, dt := range c.Deployments { + if dt.Targets != nil { + return true + } + } + return false +} + +// validateMultiTargetSupport rejects a targets list on a database whose engine +// cannot drive several targets from one database entry. Multi-target work is +// planned and applied per target, which not every engine supports yet, and a +// config that looks like it addresses several targets must never be silently +// collapsed to one. +func (c EnvironmentConfig) validateMultiTargetSupport(context, databaseType string) error { + if !c.UsesTargetsList() { + return nil + } + if !schema.SupportsFeature(databaseType, schema.FeatureMultiTarget) { + return fmt.Errorf("%s configures targets, which is only supported for %s databases (type is %q)", context, storage.DatabaseTypeMySQL, databaseType) + } + return nil +} + // resolveTargetList returns the ordered target list for one routing entry — // either an environment's scalar routing or one entry of its deployments map. // Validation and routing both call it, so a config that validates resolves to @@ -1870,6 +1905,9 @@ func (c *ServerConfig) Validate() error { if err := envConfig.validateDirectExecution(fmt.Sprintf("database %q environment %q", name, env), dbConfig.Type); err != nil { return err } + if err := envConfig.validateMultiTargetSupport(fmt.Sprintf("database %q environment %q", name, env), dbConfig.Type); err != nil { + return err + } hasDSN := envConfig.HasLocalDSN() hasScalarRouting := envConfig.Target != "" || envConfig.Targets != nil || envConfig.Deployment != "" hasMapRouting := envConfig.Deployments != nil diff --git a/pkg/api/config_test.go b/pkg/api/config_test.go index af3a0caa5..b2ef80107 100644 --- a/pkg/api/config_test.go +++ b/pkg/api/config_test.go @@ -4961,3 +4961,88 @@ func TestServerConfig_ResolveDatabaseTargets_MultiTarget(t *testing.T) { assert.Contains(t, err.Error(), "cannot configure both target and targets") }) } + +// TestServerConfig_MultiTargetIsMySQLOnly covers the engine gate on the targets +// spelling. A targets list makes an environment address several distinct +// targets, which only MySQL drives today, so configuring it on another engine +// fails validation rather than being silently collapsed to one target. A +// deployments map with several single-target entries is not multi-target and +// stays available on every engine. +func TestServerConfig_MultiTargetIsMySQLOnly(t *testing.T) { + configFor := func(dbType string, env EnvironmentConfig) ServerConfig { + return ServerConfig{ + // Strata is opt-in, and its gate runs ahead of this one. Enabling it + // keeps the Strata case exercising the engine gate under test rather + // than stopping at the experimental refusal. + ExperimentalStrataEnabled: true, + TernDeployments: TernConfig{ + "payments-a": {"production": "localhost:9090"}, + "payments-b": {"production": "localhost:9091"}, + }, + Databases: map[string]DatabaseConfig{ + "payments": { + Type: dbType, + Environments: map[string]EnvironmentConfig{"production": env}, + }, + }, + } + } + envTargets := EnvironmentConfig{Deployment: "payments-a", Targets: []string{"payments-001", "payments-002"}} + mapTargets := EnvironmentConfig{Deployments: map[string]DeploymentTarget{ + "payments-a": {Targets: []string{"payments-001", "payments-002"}}, + }} + mirrored := EnvironmentConfig{Deployments: map[string]DeploymentTarget{ + "payments-a": {Target: "payments"}, + "payments-b": {Target: "payments"}, + }} + + cases := []struct { + name string + dbType string + env EnvironmentConfig + wantErr string + }{ + {name: "mysql env targets", dbType: storage.DatabaseTypeMySQL, env: envTargets}, + {name: "mysql deployments targets", dbType: storage.DatabaseTypeMySQL, env: mapTargets}, + {name: "vitess env targets", dbType: storage.DatabaseTypeVitess, env: envTargets, wantErr: `configures targets, which is only supported for mysql databases (type is "vitess")`}, + {name: "postgres env targets", dbType: storage.DatabaseTypePostgres, env: envTargets, wantErr: `configures targets, which is only supported for mysql databases (type is "postgres")`}, + {name: "strata deployments targets", dbType: storage.DatabaseTypeStrata, env: mapTargets, wantErr: `configures targets, which is only supported for mysql databases (type is "strata")`}, + {name: "vitess mirrored deployments", dbType: storage.DatabaseTypeVitess, env: mirrored}, + {name: "postgres mirrored deployments", dbType: storage.DatabaseTypePostgres, env: mirrored}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + cfg := configFor(tc.dbType, tc.env) + err := cfg.Validate() + if tc.wantErr == "" { + require.NoError(t, err) + return + } + require.Error(t, err) + assert.Contains(t, err.Error(), tc.wantErr) + }) + } +} + +func TestEnvironmentConfig_UsesTargetsList(t *testing.T) { + cases := []struct { + name string + env EnvironmentConfig + want bool + }{ + {name: "single target", env: EnvironmentConfig{Deployment: "a", Target: "payments"}}, + {name: "local dsn", env: EnvironmentConfig{DSN: "root@tcp(localhost)/payments"}}, + {name: "deployments map of single targets", env: EnvironmentConfig{Deployments: map[string]DeploymentTarget{ + "a": {Target: "payments"}, "b": {Target: "payments"}, + }}}, + {name: "environment targets", env: EnvironmentConfig{Deployment: "a", Targets: []string{"payments-001"}}, want: true}, + {name: "deployments entry targets", env: EnvironmentConfig{Deployments: map[string]DeploymentTarget{ + "a": {Target: "payments-001"}, "b": {Targets: []string{"payments-002"}}, + }}, want: true}, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.Equal(t, tc.want, tc.env.UsesTargetsList()) + }) + } +} diff --git a/pkg/schema/dialect.go b/pkg/schema/dialect.go index 6dbe985a7..35e71eaf7 100644 --- a/pkg/schema/dialect.go +++ b/pkg/schema/dialect.go @@ -18,6 +18,13 @@ const ( DialectPostgres Dialect = "postgres" FeatureDeferredCutover Feature = "deferred cutover" + // FeatureMultiTarget covers addressing several distinct targets from one + // database entry, where each target holds its own schema and is planned and + // applied on its own. Every operator-facing surface — the plan comment, + // progress, the terminal summary — has to represent N targets rather than + // one, and that presentation is built per engine, so the feature is enabled + // per engine as the work lands rather than assumed available everywhere. + FeatureMultiTarget Feature = "multi-target" ) // DialectForDatabaseType maps a database_type to its database family for @@ -45,6 +52,8 @@ func SupportsFeature(databaseType string, feature Feature) bool { switch feature { case FeatureDeferredCutover: return databaseType == "mysql" || databaseType == "vitess" + case FeatureMultiTarget: + return databaseType == "mysql" default: return false } diff --git a/pkg/schema/dialect_test.go b/pkg/schema/dialect_test.go index eb4741eef..d779908c0 100644 --- a/pkg/schema/dialect_test.go +++ b/pkg/schema/dialect_test.go @@ -116,6 +116,11 @@ func TestSupportsFeature(t *testing.T) { {name: "postgres deferred cutover", databaseType: "postgres", feature: FeatureDeferredCutover}, {name: "unknown database type", databaseType: "unknown", feature: FeatureDeferredCutover}, {name: "unknown feature", databaseType: "mysql", feature: Feature("unknown")}, + {name: "mysql multi-target", databaseType: "mysql", feature: FeatureMultiTarget, want: true}, + {name: "vitess multi-target", databaseType: "vitess", feature: FeatureMultiTarget}, + {name: "strata multi-target", databaseType: "strata", feature: FeatureMultiTarget}, + {name: "postgres multi-target", databaseType: "postgres", feature: FeatureMultiTarget}, + {name: "unknown database type multi-target", databaseType: "unknown", feature: FeatureMultiTarget}, } for _, tt := range tests { From 5a02275d57ff050f1e0849bc026c27b310560a7c Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 17:04:16 -0400 Subject: [PATCH 06/34] fix(api): report a target beside an empty targets list as a conflict An entry that spells out both `target` and `targets: []` was reported as an empty list rather than as the conflict it is. Both spellings being present is what conflicts, so gate on the list being configured at all. Document that an environment-level targets list excludes a deployments map the same way an environment-level target does. Co-Authored-By: Claude Fable 5 --- docs/configuration.md | 1 + pkg/api/config.go | 10 ++++++---- pkg/api/config_test.go | 10 ++++++++++ 3 files changed, 17 insertions(+), 4 deletions(-) diff --git a/docs/configuration.md b/docs/configuration.md index 3cf9dc9d9..288d792c8 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -304,6 +304,7 @@ Rules: - `targets` requires `type: mysql`. Configuring it on a `vitess`, `strata`, or `postgres` database fails validation at startup. - `targets` is mutually exclusive with `target` at the same level, and with a local `dsn` / `dsn_from`. +- An environment-level `targets` list is mutually exclusive with an environment-level `deployments` map, the same way an environment-level `target` is. A `targets` list inside a `deployments` entry is how the two combine. - The list MUST contain at least one entry, and no entry may be empty. - One deployment may not list the same target twice. A rollout member is identified by its deployment and target together, so the same target under two different deployments is two distinct members and is allowed. - Members resolve deployments outermost: every target of the first deployment, then every target of the next. diff --git a/pkg/api/config.go b/pkg/api/config.go index a458a8af3..8006583eb 100644 --- a/pkg/api/config.go +++ b/pkg/api/config.go @@ -1649,11 +1649,13 @@ func (c EnvironmentConfig) validateMultiTargetSupport(context, databaseType stri // exactly the member list validation checked. // // An entry names either a single target or a targets list, never both: the two -// spellings would otherwise disagree about how many members the entry has. An -// entry that names neither returns an empty list without an error, leaving the -// caller to report the missing target in its own terms. +// spellings would otherwise disagree about how many members the entry has. Both +// spellings being present is what conflicts, so an explicitly empty targets list +// alongside a target is reported as the conflict it is rather than as an empty +// list. An entry that names neither returns an empty list without an error, +// leaving the caller to report the missing target in its own terms. func resolveTargetList(what, target string, targets []string) ([]string, error) { - if target != "" && len(targets) > 0 { + if target != "" && targets != nil { return nil, fmt.Errorf("%s cannot configure both target and targets", what) } if targets == nil { diff --git a/pkg/api/config_test.go b/pkg/api/config_test.go index b2ef80107..66805651e 100644 --- a/pkg/api/config_test.go +++ b/pkg/api/config_test.go @@ -2267,6 +2267,16 @@ func TestServerConfig_DeploymentsMapValidation(t *testing.T) { tern: baseTern, wantErrSub: "cannot configure both target and targets", }, + { + name: "environment target and an empty targets list are rejected as a conflict", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Target: "payments-001", + Targets: []string{}, + }, + tern: baseTern, + wantErrSub: "cannot configure both target and targets", + }, { name: "empty environment targets list is rejected", envConfig: EnvironmentConfig{ From 2785ad8fdb2745194188a1c5db0ea447d5150202 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:57:46 -0400 Subject: [PATCH 07/34] feat(api): report how an environment's targets diverge on pull An environment whose targets each hold their own schema has no single live schema, so pulling only the primary presented one target's schema as the environment's. A pull now fans out across every target and reports how each one differs from the primary. The response body is still the primary's schema, which is what a caller materializes. Alongside it, each other target reports its table count and the tables the two do not agree on: held by both with different DDL, held only by the primary, or held only by that target. Tables are compared by the dialect parser's canonical form, so formatting never reads as a schema difference. Deployments that are expected to hold the same schema are neither pulled nor compared. A difference between them is drift for the review-time rollup to block on, and pulling them would cost a round trip each to learn what the configuration already asserts. A target that cannot be pulled, or a table whose DDL cannot be compared, fails the pull. Reporting no divergence for a target that was never compared would describe the environment as converged on the strength of a comparison that did not happen. Co-Authored-By: Claude Fable 5 --- pkg/api/plan_handlers.go | 84 ++------ pkg/api/pull_members.go | 245 ++++++++++++++++++++++++ pkg/api/pull_members_test.go | 208 ++++++++++++++++++++ pkg/apitypes/apitypes.go | 34 ++++ pkg/cmd/internal/templates/pull.go | 47 +++++ pkg/cmd/internal/templates/pull_test.go | 57 ++++++ 6 files changed, 605 insertions(+), 70 deletions(-) create mode 100644 pkg/api/pull_members.go create mode 100644 pkg/api/pull_members_test.go diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index bf0ab033b..42c996f02 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -28,7 +28,6 @@ import ( "github.com/block/schemabot/pkg/schema" "github.com/block/schemabot/pkg/state" "github.com/block/schemabot/pkg/storage" - "github.com/block/schemabot/pkg/tern" ) const applyOperationKeyMaxLen = 255 @@ -304,79 +303,22 @@ func (s *Service) ExecutePullSchema(ctx context.Context, req apitypes.PullSchema return nil, err } - client, err := s.TernClient(resolvedTarget.Deployment, req.Environment) + merged, err := s.pullTargetSchema(ctx, req, resolvedTarget, namespaces, catalogDetail) if err != nil { span.RecordError(err) - span.SetStatus(otelcodes.Error, "tern client") - return nil, fmt.Errorf("database %q (%s): %w", req.Database, req.Environment, err) + span.SetStatus(otelcodes.Error, "pull schema failed") + return nil, err } - isRemoteTarget := client.IsRemote() - s.logger.Info("ExecutePullSchema: calling PullSchema", - "database", req.Database, - "type", resolvedTarget.DatabaseType, - "deployment", resolvedTarget.Deployment, - "target", resolvedTarget.Target, - "environment", req.Environment, - "pull_call_count", len(namespaces), - "explicit_namespace_count", len(req.Namespaces), - "is_remote", isRemoteTarget, - ) - - merged := &ternv1.PullSchemaResponse{ - Database: req.Database, - Type: resolvedTarget.DatabaseType, - Environment: req.Environment, - Namespaces: make(map[string]*ternv1.PulledNamespace), - } - for _, namespace := range namespaces { - resp, pullErr := client.PullSchema(ctx, &ternv1.PullSchemaRequest{ - Database: req.Database, - Type: resolvedTarget.DatabaseType, - Target: resolvedTarget.Target, - Environment: req.Environment, - Namespace: namespace, - CatalogDetail: catalogDetail, - }) - if pullErr != nil { - span.RecordError(pullErr) - span.SetStatus(otelcodes.Error, "pull schema failed") - s.logger.Error("ExecutePullSchema: routing client PullSchema failed", - "database", req.Database, - "type", resolvedTarget.DatabaseType, - "deployment", resolvedTarget.Deployment, - "target", resolvedTarget.Target, - "environment", req.Environment, - "namespace", namespace, - "endpoint", client.Endpoint(), - "is_remote", isRemoteTarget, - "error", pullErr, - ) - if isRemoteTarget && grpcstatus.Code(pullErr) == grpccodes.Unavailable { - return nil, &RemoteDeploymentUnavailableError{ - Deployment: resolvedTarget.Deployment, - Target: resolvedTarget.Target, - Err: pullErr, - } - } - // Whether pull is supported is the data plane's answer — it - // depends on which engine backs the deployment — so the 501 is - // derived from the pull attempt instead of gating on database - // type. One sentinel check covers both routes: the local client - // returns ErrPullSchemaUnsupportedType directly, and the gRPC - // client re-derives the same sentinel from the remote data - // plane's own unsupported verdict (infrastructure Unimplemented - // errors deliberately fall through as ordinary failures). - if errors.Is(pullErr, tern.ErrPullSchemaUnsupportedType) { - return nil, &unsupportedPullSchemaError{DatabaseType: resolvedTarget.DatabaseType} - } - return nil, pullErr - } - if err := mergePullSchemaResponse(merged, resp, namespace); err != nil { - span.RecordError(err) - span.SetStatus(otelcodes.Error, "merge pull schema response") - return nil, err - } + // An environment whose targets each hold their own schema has no single live + // schema, so the primary's is reported alongside how the others differ from + // it. Comparing here, rather than leaving it to the caller, keeps a pull from + // presenting one target's schema as the environment's. + divergences, err := s.pullMemberDivergence(ctx, req, resolvedTarget, merged, namespaces, catalogDetail) + if err != nil { + span.RecordError(err) + span.SetStatus(otelcodes.Error, "compare rollout members") + return nil, err } span.SetAttributes(attribute.Int("table_count", int(merged.TableCount))) @@ -386,6 +328,7 @@ func (s *Service) ExecutePullSchema(ctx context.Context, req apitypes.PullSchema "environment", merged.Environment, "table_count", merged.TableCount, "namespace_count", len(merged.Namespaces), + "compared_target_count", len(divergences), ) httpResp := pullSchemaResponseFromProto(merged) @@ -395,6 +338,7 @@ func (s *Service) ExecutePullSchema(ctx context.Context, req apitypes.PullSchema if dbConfig, ok := s.config.DatabaseConfigs()[req.Database]; ok { httpResp.App = dbConfig.App } + httpResp.Targets = divergences if req.Lint { if err := lintPulledNamespaces(httpResp); err != nil { span.RecordError(err) diff --git a/pkg/api/pull_members.go b/pkg/api/pull_members.go new file mode 100644 index 000000000..3be6a1372 --- /dev/null +++ b/pkg/api/pull_members.go @@ -0,0 +1,245 @@ +package api + +import ( + "context" + "errors" + "fmt" + "sort" + + grpccodes "google.golang.org/grpc/codes" + grpcstatus "google.golang.org/grpc/status" + + "github.com/block/schemabot/pkg/apitypes" + "github.com/block/schemabot/pkg/ddl" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/routing" + "github.com/block/schemabot/pkg/schema" + "github.com/block/schemabot/pkg/tern" +) + +// pullTargetSchema fetches one execution target's live schema, one call per +// namespace, and merges the results into a single response. +// +// It is the single place a target is pulled from, so the primary target whose +// schema the caller materializes and a non-primary rollout member pulled only to +// compare against it are fetched identically and fail the same way. +func (s *Service) pullTargetSchema( + ctx context.Context, + req apitypes.PullSchemaRequest, + target routing.ExecutionTarget, + namespaces []string, + catalogDetail ternv1.PullCatalogDetail, +) (*ternv1.PullSchemaResponse, error) { + client, err := s.TernClient(target.Deployment, req.Environment) + if err != nil { + return nil, fmt.Errorf("database %q (%s): %w", req.Database, req.Environment, err) + } + + isRemoteTarget := client.IsRemote() + s.logger.Info("ExecutePullSchema: calling PullSchema", + "database", req.Database, + "type", target.DatabaseType, + "deployment", target.Deployment, + "target", target.Target, + "environment", req.Environment, + "pull_call_count", len(namespaces), + "explicit_namespace_count", len(req.Namespaces), + "is_remote", isRemoteTarget, + ) + + merged := &ternv1.PullSchemaResponse{ + Database: req.Database, + Type: target.DatabaseType, + Environment: req.Environment, + Namespaces: make(map[string]*ternv1.PulledNamespace), + } + for _, namespace := range namespaces { + resp, pullErr := client.PullSchema(ctx, &ternv1.PullSchemaRequest{ + Database: req.Database, + Type: target.DatabaseType, + Target: target.Target, + Environment: req.Environment, + Namespace: namespace, + CatalogDetail: catalogDetail, + }) + if pullErr != nil { + s.logger.Error("ExecutePullSchema: routing client PullSchema failed", + "database", req.Database, + "type", target.DatabaseType, + "deployment", target.Deployment, + "target", target.Target, + "environment", req.Environment, + "namespace", namespace, + "endpoint", client.Endpoint(), + "is_remote", isRemoteTarget, + "error", pullErr, + ) + if isRemoteTarget && grpcstatus.Code(pullErr) == grpccodes.Unavailable { + return nil, &RemoteDeploymentUnavailableError{ + Deployment: target.Deployment, + Target: target.Target, + Err: pullErr, + } + } + // Whether pull is supported is the data plane's answer — it depends + // on which engine backs the deployment — so the 501 is derived from + // the pull attempt instead of gating on database type. One sentinel + // check covers both routes: the local client returns + // ErrPullSchemaUnsupportedType directly, and the gRPC client + // re-derives the same sentinel from the remote data plane's own + // unsupported verdict (infrastructure Unimplemented errors + // deliberately fall through as ordinary failures). + if errors.Is(pullErr, tern.ErrPullSchemaUnsupportedType) { + return nil, &unsupportedPullSchemaError{DatabaseType: target.DatabaseType} + } + return nil, pullErr + } + if err := mergePullSchemaResponse(merged, resp, namespace); err != nil { + return nil, err + } + } + return merged, nil +} + +// pullMemberDivergence pulls every non-primary rollout member of an environment +// whose members hold their own schemas and reports how each one's live schema +// differs from the primary's. +// +// It returns nil for an environment whose members are expected to hold the same +// schema: pulling them would cost one round trip per member to learn what the +// deployments contract already asserts, and any difference there is drift for +// the review-time rollup to block on, not divergence for a pull to describe. +// +// Divergence is what a multi-target environment is for, so it is reported rather +// than treated as an error — but a member that cannot be pulled, or whose DDL +// cannot be compared, fails the request. Reporting "no divergence" for a member +// that was never successfully compared would describe the environment as +// converged on the strength of a comparison that did not happen. +func (s *Service) pullMemberDivergence( + ctx context.Context, + req apitypes.PullSchemaRequest, + primary routing.ExecutionTarget, + primarySchema *ternv1.PullSchemaResponse, + namespaces []string, + catalogDetail ternv1.PullCatalogDetail, +) ([]*apitypes.TargetDivergence, error) { + planning, err := s.config.MemberPlanningFor(req.Database, req.Environment) + if err != nil { + return nil, fmt.Errorf("resolve member planning for %s/%s: %w", req.Database, req.Environment, err) + } + if planning != PlanIndependent { + s.logger.Debug("pull reports no per-target divergence; this environment's members are expected to hold the same schema", + "database", req.Database, + "environment", req.Environment) + return nil, nil + } + + targets, err := s.config.ResolveDatabaseTargets(req.Database, req.Environment) + if err != nil { + return nil, fmt.Errorf("resolve targets for %s/%s: %w", req.Database, req.Environment, err) + } + + dialect := schema.DialectForDatabaseType(primary.DatabaseType) + parser, err := ddl.ParserForDialect(dialect) + if err != nil { + return nil, fmt.Errorf("compare targets of %s/%s: %w", req.Database, req.Environment, err) + } + primaryTables, err := canonicalTablesByNamespace(parser, primarySchema) + if err != nil { + return nil, fmt.Errorf("canonicalize schema of rollout member %s: %w", primary.MemberID(), err) + } + + divergences := make([]*apitypes.TargetDivergence, 0, len(targets)-1) + for _, target := range targets { + if target.Deployment == primary.Deployment && target.Target == primary.Target { + continue + } + memberSchema, err := s.pullTargetSchema(ctx, req, target, namespaces, catalogDetail) + if err != nil { + return nil, fmt.Errorf("pull rollout member %s: %w", target.MemberID(), err) + } + memberTables, err := canonicalTablesByNamespace(parser, memberSchema) + if err != nil { + return nil, fmt.Errorf("canonicalize schema of rollout member %s: %w", target.MemberID(), err) + } + divergences = append(divergences, &apitypes.TargetDivergence{ + Deployment: target.Deployment, + Target: target.Target, + TableCount: memberSchema.TableCount, + DivergedTables: divergedTables(primaryTables, memberTables), + }) + } + return divergences, nil +} + +// namespaceTable identifies one pulled table within its namespace. +type namespaceTable struct { + namespace string + table string +} + +// canonicalTablesByNamespace reduces a pulled schema to the canonical form of +// each table's DDL, which is what two targets are compared by: a difference in +// whitespace, keyword case, or clause order is the same schema, and must not be +// reported as divergence. +// +// A table whose DDL the dialect's parser cannot canonicalize fails the request. +// Comparing raw text for it would report every formatting difference as a +// schema difference, and skipping it would drop a table out of the comparison +// without saying so. +func canonicalTablesByNamespace(parser ddl.StatementParser, pulled *ternv1.PullSchemaResponse) (map[namespaceTable]string, error) { + canonical := make(map[namespaceTable]string) + for namespace, ns := range pulled.Namespaces { + if ns == nil { + return nil, fmt.Errorf("namespace %q has no pulled content", namespace) + } + for table, tableDDL := range ns.Tables { + stmtType, _, err := parser.Classify(tableDDL) + if err != nil { + return nil, fmt.Errorf("namespace %q table %q: DDL rejected by the statement parser: %w", namespace, table, err) + } + if stmtType != ddl.StatementCreateTable { + return nil, fmt.Errorf("namespace %q table %q: expected a CREATE TABLE statement, got %s", namespace, table, stmtType) + } + canonical[namespaceTable{namespace: namespace, table: table}] = parser.Canonicalize(tableDDL) + } + } + return canonical, nil +} + +// divergedTables reports every table the two targets do not agree on, sorted for +// a stable response body. A table only one target holds is as much a divergence +// as a table both hold with different DDL, so all three cases are reported +// together and distinguished by Difference. +func divergedTables(primary, member map[namespaceTable]string) []apitypes.DivergedTable { + diverged := make([]apitypes.DivergedTable, 0) + for key, primaryDDL := range primary { + memberDDL, ok := member[key] + if !ok { + diverged = append(diverged, divergedTable(key, apitypes.DivergenceOnlyOnPrimary)) + continue + } + if memberDDL != primaryDDL { + diverged = append(diverged, divergedTable(key, apitypes.DivergenceDiffers)) + } + } + for key := range member { + if _, ok := primary[key]; !ok { + diverged = append(diverged, divergedTable(key, apitypes.DivergenceOnlyOnTarget)) + } + } + sort.Slice(diverged, func(i, j int) bool { + if diverged[i].Namespace != diverged[j].Namespace { + return diverged[i].Namespace < diverged[j].Namespace + } + return diverged[i].Table < diverged[j].Table + }) + if len(diverged) == 0 { + return nil + } + return diverged +} + +func divergedTable(key namespaceTable, difference string) apitypes.DivergedTable { + return apitypes.DivergedTable{Namespace: key.namespace, Table: key.table, Difference: difference} +} diff --git a/pkg/api/pull_members_test.go b/pkg/api/pull_members_test.go new file mode 100644 index 000000000..aa9492427 --- /dev/null +++ b/pkg/api/pull_members_test.go @@ -0,0 +1,208 @@ +package api + +import ( + "errors" + "log/slog" + "os" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/apitypes" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/storage" + "github.com/block/schemabot/pkg/tern" +) + +const ( + pullUsersDDL = "CREATE TABLE `users` (`id` bigint NOT NULL, PRIMARY KEY (`id`));\n" + pullUsersEmailDDL = "CREATE TABLE `users` (`id` bigint NOT NULL, `email` varchar(255) DEFAULT NULL, PRIMARY KEY (`id`));\n" + pullAuditsDDL = "CREATE TABLE `audits` (`id` bigint NOT NULL, PRIMARY KEY (`id`));\n" +) + +// pulledTables builds a one-namespace pull response holding the given tables. +func pulledTables(tables map[string]string) *ternv1.PullSchemaResponse { + return &ternv1.PullSchemaResponse{ + Database: "testapp", + Type: storage.DatabaseTypeMySQL, + Environment: "production", + Namespaces: map[string]*ternv1.PulledNamespace{"testapp": {Tables: tables}}, + TableCount: int32(len(tables)), + } +} + +// pullTargetService wires one Tern client per deployment for a database whose +// production environment routes to env. +func pullTargetService(t *testing.T, env EnvironmentConfig, clients map[string]tern.Client) *Service { + t.Helper() + cfg := &ServerConfig{ + Databases: map[string]DatabaseConfig{ + "testapp": { + Type: storage.DatabaseTypeMySQL, + Environments: map[string]EnvironmentConfig{"production": env}, + }, + }, + TernDeployments: TernConfig{ + "eu": {"production": "eu.example.com:80"}, + "us": {"production": "us.example.com:80"}, + }, + } + logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelError})) + return New(&mockStorageWithApplyStores{plans: &staticPlanStore{}, applies: &staticApplyStore{}}, cfg, clients, logger) +} + +func pullRequest() apitypes.PullSchemaRequest { + return apitypes.PullSchemaRequest{Database: "testapp", Environment: "production", Type: storage.DatabaseTypeMySQL} +} + +// perTargetPullClient answers each target with its own schema, so a fan-out that +// pulled only one target cannot look like it pulled all of them. +type perTargetPullClient struct { + mockTernClient + byTarget map[string]*ternv1.PullSchemaResponse + errs map[string]error +} + +func newPerTargetPullClient(byTarget map[string]*ternv1.PullSchemaResponse, errs map[string]error) *perTargetPullClient { + c := &perTargetPullClient{byTarget: byTarget, errs: errs} + c.pullSchemaHook = func(req *ternv1.PullSchemaRequest) (*ternv1.PullSchemaResponse, error) { + if err, ok := c.errs[req.Target]; ok { + return nil, err + } + resp, ok := c.byTarget[req.Target] + if !ok { + return nil, errors.New("no schema seeded for target " + req.Target) + } + return resp, nil + } + return c +} + +// pulledTargetsOf returns the targets a fan-out actually pulled, in call order. +func (c *perTargetPullClient) pulledTargets() []string { + out := make([]string, 0, len(c.pullSchemaReqs)) + for _, req := range c.pullSchemaReqs { + out = append(out, req.Target) + } + return out +} + +func multiTargetPullEnv() EnvironmentConfig { + return EnvironmentConfig{Deployment: "eu", Targets: []string{"testapp-001", "testapp-002"}} +} + +// An environment whose targets each hold their own schema has no single live +// schema. A pull returns the primary's schema, which is what a caller +// materializes, plus how each other target differs from it. +func TestExecutePullSchema_ReportsPerTargetDivergence(t *testing.T) { + client := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{ + "testapp-001": pulledTables(map[string]string{"users": pullUsersDDL}), + "testapp-002": pulledTables(map[string]string{"users": pullUsersEmailDDL, "audits": pullAuditsDDL}), + }, nil) + svc := pullTargetService(t, multiTargetPullEnv(), map[string]tern.Client{"eu/production": client}) + + resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.NoError(t, err) + + assert.Equal(t, pullUsersDDL, resp.Namespaces["testapp"].Tables["users"], "the response body is the primary's schema") + assert.Equal(t, []string{"testapp-001", "testapp-002"}, client.pulledTargets(), "every target is pulled") + + require.Len(t, resp.Targets, 1, "only non-primary targets are compared against the primary") + diverged := resp.Targets[0] + assert.Equal(t, "eu", diverged.Deployment) + assert.Equal(t, "testapp-002", diverged.Target) + assert.Equal(t, int32(2), diverged.TableCount) + assert.Equal(t, []apitypes.DivergedTable{ + {Namespace: "testapp", Table: "audits", Difference: apitypes.DivergenceOnlyOnTarget}, + {Namespace: "testapp", Table: "users", Difference: apitypes.DivergenceDiffers}, + }, diverged.DivergedTables) +} + +// Targets that hold the same schema report no diverged tables, which is a +// positive statement that they agree rather than an absence of information. +func TestExecutePullSchema_ConvergedTargetsReportNoDivergedTables(t *testing.T) { + client := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{ + "testapp-001": pulledTables(map[string]string{"users": pullUsersDDL}), + "testapp-002": pulledTables(map[string]string{"users": pullUsersDDL}), + }, nil) + svc := pullTargetService(t, multiTargetPullEnv(), map[string]tern.Client{"eu/production": client}) + + resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.NoError(t, err) + + require.Len(t, resp.Targets, 1) + assert.Equal(t, "testapp-002", resp.Targets[0].Target) + assert.Empty(t, resp.Targets[0].DivergedTables) +} + +// Two targets holding the same schema written differently are not diverged: +// tables are compared by the parser's canonical form, so formatting never +// reads as a schema difference. +func TestExecutePullSchema_FormattingIsNotDivergence(t *testing.T) { + client := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{ + "testapp-001": pulledTables(map[string]string{"users": pullUsersDDL}), + "testapp-002": pulledTables(map[string]string{ + "users": "create table `users` (\n `id` BIGINT not null,\n primary key (`id`)\n)", + }), + }, nil) + svc := pullTargetService(t, multiTargetPullEnv(), map[string]tern.Client{"eu/production": client}) + + resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.NoError(t, err) + + require.Len(t, resp.Targets, 1) + assert.Empty(t, resp.Targets[0].DivergedTables, "the same schema written differently is the same schema") +} + +// A target that cannot be pulled fails the request. Returning the primary's +// schema with that target simply absent would report the environment as +// converged on the strength of a comparison that never happened. +func TestExecutePullSchema_UnreachableTargetFailsThePull(t *testing.T) { + client := newPerTargetPullClient( + map[string]*ternv1.PullSchemaResponse{"testapp-001": pulledTables(map[string]string{"users": pullUsersDDL})}, + map[string]error{"testapp-002": errors.New("target unreachable")}, + ) + svc := pullTargetService(t, multiTargetPullEnv(), map[string]tern.Client{"eu/production": client}) + + _, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.Error(t, err) + assert.Contains(t, err.Error(), "pull rollout member eu/testapp-002") + assert.Contains(t, err.Error(), "target unreachable") +} + +// A pulled table whose DDL is not a CREATE TABLE cannot be compared, so the +// pull fails rather than dropping that table out of the comparison silently. +func TestExecutePullSchema_UncomparableTableFailsThePull(t *testing.T) { + client := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{ + "testapp-001": pulledTables(map[string]string{"users": pullUsersDDL}), + "testapp-002": pulledTables(map[string]string{"users": "SELECT 1"}), + }, nil) + svc := pullTargetService(t, multiTargetPullEnv(), map[string]tern.Client{"eu/production": client}) + + _, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.Error(t, err) + assert.Contains(t, err.Error(), "eu/testapp-002") + assert.Contains(t, err.Error(), "users") +} + +// Deployments that are expected to hold the same schema are not compared: a +// difference between them is drift for the review-time rollup to block on, not +// divergence for a pull to describe. Pulling them would cost a round trip per +// deployment to learn what the configuration already asserts. +func TestExecutePullSchema_MirroredDeploymentsAreNotPulledOrCompared(t *testing.T) { + eu := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{"testapp": pulledTables(map[string]string{"users": pullUsersDDL})}, nil) + us := newPerTargetPullClient(nil, map[string]error{"testapp": errors.New("a mirrored deployment must not be pulled")}) + env := EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{"eu": {Target: "testapp"}, "us": {Target: "testapp"}}, + DeploymentOrder: []string{"eu", "us"}, + } + svc := pullTargetService(t, env, map[string]tern.Client{"eu/production": eu, "us/production": us}) + + resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.NoError(t, err) + + assert.Empty(t, resp.Targets, "an environment whose deployments should match reports no per-target divergence") + assert.Equal(t, []string{"testapp"}, eu.pulledTargets(), "only the primary deployment is pulled") + assert.Empty(t, us.pulledTargets()) +} diff --git a/pkg/apitypes/apitypes.go b/pkg/apitypes/apitypes.go index 284b99e64..7845fbce4 100644 --- a/pkg/apitypes/apitypes.go +++ b/pkg/apitypes/apitypes.go @@ -448,6 +448,11 @@ type PullSchemaRequest struct { } // PullSchemaResponse is the HTTP response body for POST /api/pull. +// +// Namespaces holds the primary target's live schema, which is the schema a +// caller materializes. Targets is populated only for an environment whose +// targets each hold their own schema, and describes how every other target +// differs from the primary. type PullSchemaResponse struct { Database string `json:"database"` Type string `json:"type"` @@ -458,6 +463,35 @@ type PullSchemaResponse struct { App string `json:"app,omitempty"` Namespaces map[string]*PulledNamespace `json:"namespaces"` TableCount int32 `json:"table_count"` + Targets []*TargetDivergence `json:"targets,omitempty"` +} + +// Difference values for DivergedTable. +const ( + // DivergenceDiffers means both targets hold the table with different DDL. + DivergenceDiffers = "differs" + // DivergenceOnlyOnPrimary means only the primary target holds the table. + DivergenceOnlyOnPrimary = "only_on_primary" + // DivergenceOnlyOnTarget means only this target holds the table. + DivergenceOnlyOnTarget = "only_on_target" +) + +// TargetDivergence reports how one non-primary target's live schema differs from +// the primary's. An empty DivergedTables means the two targets hold the same +// schema; it never means the comparison was skipped, since a target that could +// not be pulled or compared fails the pull instead. +type TargetDivergence struct { + Deployment string `json:"deployment"` + Target string `json:"target"` + TableCount int32 `json:"table_count"` + DivergedTables []DivergedTable `json:"diverged_tables,omitempty"` +} + +// DivergedTable names one table two targets do not agree on, and how. +type DivergedTable struct { + Namespace string `json:"namespace"` + Table string `json:"table"` + Difference string `json:"difference"` } // DatabaseListResponse is the HTTP response body for GET /api/databases. diff --git a/pkg/cmd/internal/templates/pull.go b/pkg/cmd/internal/templates/pull.go index 2552e27fa..c63de97ac 100644 --- a/pkg/cmd/internal/templates/pull.go +++ b/pkg/cmd/internal/templates/pull.go @@ -30,6 +30,7 @@ func WritePullSchema(resp *apitypes.PullSchemaResponse) { } rows = append(rows, BoxRow{Label: "Tables", Value: strconv.Itoa(int(resp.TableCount))}) WriteBox(rows, "", nil) + writeTargetDivergence(resp.Targets) for _, name := range sortedKeys(resp.Namespaces) { ns := resp.Namespaces[name] @@ -59,6 +60,52 @@ func WritePullSchema(resp *apitypes.PullSchemaResponse) { } } +// divergenceLabels phrases each difference for a reader looking at the primary +// target's schema: the DDL printed below is the primary's, so the wording says +// what the other target has instead. +var divergenceLabels = map[string]string{ + apitypes.DivergenceDiffers: "differs", + apitypes.DivergenceOnlyOnPrimary: "missing", + apitypes.DivergenceOnlyOnTarget: "extra", +} + +// writeTargetDivergence reports how each of an environment's other targets +// differs from the primary, whose schema is the DDL printed below. It renders as +// "--" comments like the rest of the pull output, so redirecting a multi-target +// pull into a .sql file still produces valid SQL. +// +// A target with no diverged tables is still listed: "these two hold the same +// schema" is the answer an operator is usually looking for, and omitting the +// converged targets would leave it indistinguishable from not having checked. +// An environment whose targets are expected to hold the same schema carries no +// divergence at all and prints nothing. +func writeTargetDivergence(targets []*apitypes.TargetDivergence) { + if len(targets) == 0 { + return + } + for _, target := range targets { + fmt.Println() + if len(target.DivergedTables) == 0 { + fmt.Println(annotation(fmt.Sprintf("-- Target %s — same schema as the primary target", + emphasis("`"+target.Target+"`")))) + continue + } + fmt.Println(annotation(fmt.Sprintf("-- Target %s — %d %s differ from the primary target", + emphasis("`"+target.Target+"`"), + len(target.DivergedTables), + ui.Pluralize("table", len(target.DivergedTables))))) + for _, table := range target.DivergedTables { + label, ok := divergenceLabels[table.Difference] + if !ok { + // A difference this client does not know how to phrase is still + // reported, since dropping it would understate the divergence. + label = table.Difference + } + fmt.Println(annotation(fmt.Sprintf("-- %s.%s: %s", table.Namespace, table.Table, label))) + } + } +} + // tableKindView is the catalog kind an engine reports for a view rather than // a stored table. const tableKindView = "view" diff --git a/pkg/cmd/internal/templates/pull_test.go b/pkg/cmd/internal/templates/pull_test.go index 2290de48c..c6a1d731e 100644 --- a/pkg/cmd/internal/templates/pull_test.go +++ b/pkg/cmd/internal/templates/pull_test.go @@ -230,3 +230,60 @@ func TestWritePullSchema_StylesOutputOnInteractiveTerminals(t *testing.T) { assert.NotContains(t, plain, "\033[", "piped output carries no ANSI escapes") assert.Contains(t, plain, "{\n \"sharded\": true\n}\n") } + +// A pull of an environment whose targets each hold their own schema reports how +// every other target differs from the primary, whose schema is the DDL printed +// below. A converged target says so rather than being omitted, so "they agree" +// is distinguishable from "not checked". The whole section renders as "--" +// comments, so a redirected pull stays valid SQL. +func TestWritePullSchema_RendersPerTargetDivergence(t *testing.T) { + setColors(t, false) + out := captureStdout(t, func() { + WritePullSchema(&apitypes.PullSchemaResponse{ + Database: "orders-db", + Type: "mysql", + Environment: "production", + TableCount: 1, + Namespaces: map[string]*apitypes.PulledNamespace{ + "orders": {Tables: map[string]string{"users": "CREATE TABLE `users` (`id` bigint NOT NULL);\n"}}, + }, + Targets: []*apitypes.TargetDivergence{ + {Deployment: "eu", Target: "orders-002", TableCount: 2, DivergedTables: []apitypes.DivergedTable{ + {Namespace: "orders", Table: "audits", Difference: apitypes.DivergenceOnlyOnTarget}, + {Namespace: "orders", Table: "users", Difference: apitypes.DivergenceDiffers}, + }}, + {Deployment: "eu", Target: "orders-003", TableCount: 1}, + }, + }) + }) + + assert.Contains(t, out, "-- Target `orders-002` — 2 tables differ from the primary target") + assert.Contains(t, out, "-- orders.audits: extra") + assert.Contains(t, out, "-- orders.users: differs") + assert.Contains(t, out, "-- Target `orders-003` — same schema as the primary target") + for line := range strings.SplitSeq(out, "\n") { + if strings.Contains(line, "orders-002") || strings.Contains(line, "orders-003") || strings.Contains(line, "orders.audits") { + assert.True(t, strings.HasPrefix(strings.TrimSpace(line), "--"), + "divergence line %q must be a SQL comment", line) + } + } +} + +// An environment whose targets are expected to hold the same schema carries no +// divergence, and a pull of it renders exactly as it always did. +func TestWritePullSchema_NoDivergenceSectionWithoutTargets(t *testing.T) { + setColors(t, false) + out := captureStdout(t, func() { + WritePullSchema(&apitypes.PullSchemaResponse{ + Database: "orders-db", + Type: "mysql", + Environment: "production", + TableCount: 1, + Namespaces: map[string]*apitypes.PulledNamespace{ + "orders": {Tables: map[string]string{"users": "CREATE TABLE `users` (`id` bigint NOT NULL);\n"}}, + }, + }) + }) + + assert.NotContains(t, out, "-- Target ") +} From e5afea2ef2ba075d57e4ae8beb2feb90afeeb389 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:20:47 -0400 Subject: [PATCH 08/34] feat(api): persist one plan per rollout member when members are planned independently An environment whose members are planned against their own live schemas has no single plan that covers them, so each non-primary member's plan is now stored as a plan row of its own and its identifier recorded on the member's rollup entry. The primary is deliberately left without one: its plan is the reviewed plan the apply is created from. Plan-row construction moves into one helper shared by the reviewed primary plan and a member plan, so the two are stored identically and are indistinguishable downstream. An identifier that already exists still means "the same plan, re-stored", and the existing row is now reloaded so its ID is returned rather than lost. A member whose plan cannot be stored is reclassified as errored and blocks the review: an apply dispatches each member against its stored plan, so a member with no stored plan has nothing to run. Co-Authored-By: Claude Fable 5 --- pkg/api/plan_handlers.go | 51 +++++-- pkg/api/plan_member_plans.go | 83 ++++++++++++ pkg/api/plan_member_plans_test.go | 216 ++++++++++++++++++++++++++++++ pkg/api/plan_review_drift.go | 8 ++ pkg/api/plan_rollup.go | 7 + 5 files changed, 356 insertions(+), 9 deletions(-) create mode 100644 pkg/api/plan_member_plans.go create mode 100644 pkg/api/plan_member_plans_test.go diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index 7a3025c8c..e9483599c 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -887,6 +887,28 @@ type storedPlanRoute struct { } func (s *Service) storePlanResponse(ctx context.Context, req PlanRequest, resp *ternv1.PlanResponse, route storedPlanRoute) error { + _, err := s.storePlan(ctx, req, resp.PlanId, resp.Changes, resp.Shards, route) + return err +} + +// storePlan writes one plan row for a single rollout member: the changes and +// shards that member would run, stamped with the member's own route and the +// request's PR context. It is the one place a plan row is built, so the primary +// member's reviewed plan and a non-primary member's independently produced plan +// are stored identically and are indistinguishable to everything downstream. +// +// planIdentifier is the plan's external identifier — minted by the planner for +// the primary, minted here for a member whose plan came from the non-persisting +// diff RPC. The stored row's ID is returned so a caller can point an apply +// operation at exactly this plan. +// +// An identifier that already exists is not an error: a re-plan of unchanged +// content re-stores the same plan, and the existing row is the same plan. The +// existing row is reloaded so its ID is still returned. +func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier string, changes []*ternv1.SchemaChange, shards []*ternv1.ShardPlan, route storedPlanRoute) (int64, error) { + if planIdentifier == "" { + return 0, fmt.Errorf("store plan for database %s deployment %q target %q: plan has no identifier", req.Database, route.Deployment, route.Target) + } prInt := 0 if req.PullRequest != nil { prInt = int(*req.PullRequest) @@ -899,16 +921,16 @@ func (s *Service) storePlanResponse(ctx context.Context, req PlanRequest, resp * if req.HeadSHA != nil { headSHA = *req.HeadSHA } - namespaces, err := protoChangesToNamespaces(resp.Changes, req.SchemaFiles) + namespaces, err := protoChangesToNamespaces(changes, req.SchemaFiles) if err != nil { - return fmt.Errorf("convert plan namespaces: %w", err) + return 0, fmt.Errorf("convert plan namespaces: %w", err) } - shards, err := protoShardPlansToStorage(resp.Shards) + storedShards, err := protoShardPlansToStorage(shards) if err != nil { - return fmt.Errorf("convert plan shards: %w", err) + return 0, fmt.Errorf("convert plan shards: %w", err) } storedPlan := &storage.Plan{ - PlanIdentifier: resp.PlanId, + PlanIdentifier: planIdentifier, Database: req.Database, DatabaseType: route.DatabaseType, Deployment: route.Deployment, @@ -919,14 +941,25 @@ func (s *Service) storePlanResponse(ctx context.Context, req PlanRequest, resp * Environment: req.Environment, SchemaFiles: protoToSchemaFiles(req.SchemaFiles), Namespaces: namespaces, - Shards: shards, + Shards: storedShards, HeadSHA: headSHA, CreatedAt: time.Now(), } - if _, err := s.storage.Plans().Create(ctx, storedPlan); err != nil && !errors.Is(err, storage.ErrPlanIDExists) { - return fmt.Errorf("store plan: %w", err) + id, err := s.storage.Plans().Create(ctx, storedPlan) + if err == nil { + return id, nil } - return nil + if !errors.Is(err, storage.ErrPlanIDExists) { + return 0, fmt.Errorf("store plan %s: %w", planIdentifier, err) + } + existing, getErr := s.storage.Plans().Get(ctx, planIdentifier) + if getErr != nil { + return 0, fmt.Errorf("reload already-stored plan %s: %w", planIdentifier, getErr) + } + if existing == nil { + return 0, fmt.Errorf("plan %s was reported as already stored but could not be read back", planIdentifier) + } + return existing.ID, nil } // handleApply handles POST /api/apply requests. diff --git a/pkg/api/plan_member_plans.go b/pkg/api/plan_member_plans.go new file mode 100644 index 000000000..3a6cf873e --- /dev/null +++ b/pkg/api/plan_member_plans.go @@ -0,0 +1,83 @@ +package api + +import ( + "context" + "fmt" + + "github.com/block/schemabot/pkg/engine" + "github.com/block/schemabot/pkg/routing" +) + +// persistMemberPlans stores one plan row per non-primary rollout member that was +// planned against its own live schema, and records each stored plan's identifier +// on its rollup entry. +// +// It does work only under PlanIndependent. Under PlanMirrored every member is +// expected to run exactly the reviewed changes, which the primary's plan row +// already holds, so there is no second plan to store. +// +// The primary member is deliberately left without a plan identifier of its own. +// Its plan is the reviewed plan, already stored, and is the plan the apply is +// created from — so the primary's work runs its apply's plan, which is what an +// operation with no plan of its own already means. +// +// A member whose plan cannot be stored is reclassified as errored and blocks the +// review. An apply dispatches each member against its stored plan, so a member +// with no stored plan has nothing to run; letting the rollup stay clean would +// gate the PR on a member that could not have been applied. +func (s *Service) persistMemberPlans(ctx context.Context, req PlanRequest, planning MemberPlanning, diffs []DeploymentPlanDiff, rollup *PlanRollup) error { + if planning != PlanIndependent { + return nil + } + if len(diffs) != len(rollup.Entries) { + return fmt.Errorf("persist member plans for %s/%s: %d member diffs for %d rollup entries", req.Database, req.Environment, len(diffs), len(rollup.Entries)) + } + + // Index 0 is the primary, whose reviewed plan is already stored. + for i := 1; i < len(rollup.Entries); i++ { + entry := &rollup.Entries[i] + member := routing.ExecutionTarget{Deployment: entry.Deployment, Target: entry.Target} + if entry.Class != DeploymentPlanned { + s.logger.Debug("rollout member produced no usable plan to store; it already blocks the review", + "database", req.Database, + "environment", req.Environment, + "member", member.MemberID(), + "class", entry.Class.String()) + continue + } + + // The plan came from the non-persisting diff RPC, so it arrived without an + // identifier and gets one minted here. + planIdentifier := engine.NewPlanID() + route := storedPlanRoute{ + DatabaseType: entry.DatabaseType, + Deployment: entry.Deployment, + Target: entry.Target, + } + if _, err := s.storePlan(ctx, req, planIdentifier, diffs[i].Changes, diffs[i].Shards, route); err != nil { + s.logger.Error("failed to store a rollout member's plan; the member will block the review because an apply would have no plan to run for it", + "repository", req.Repository, + "database", req.Database, + "database_type", entry.DatabaseType, + "environment", req.Environment, + "deployment", entry.Deployment, + "target", entry.Target, + "plan_id", planIdentifier, + "error", err) + entry.Class = DeploymentErrored + entry.Err = fmt.Errorf("store plan for rollout member %s: %w", member.MemberID(), err) + rollup.Clean = false + continue + } + + entry.PlanIdentifier = planIdentifier + s.logger.Info("stored plan for rollout member", + "repository", req.Repository, + "database", req.Database, + "environment", req.Environment, + "deployment", entry.Deployment, + "target", entry.Target, + "plan_id", planIdentifier) + } + return nil +} diff --git a/pkg/api/plan_member_plans_test.go b/pkg/api/plan_member_plans_test.go new file mode 100644 index 000000000..f3e44d49e --- /dev/null +++ b/pkg/api/plan_member_plans_test.go @@ -0,0 +1,216 @@ +package api + +import ( + "context" + "errors" + "log/slog" + "os" + "strings" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/routing" + "github.com/block/schemabot/pkg/storage" + "github.com/block/schemabot/pkg/tern" +) + +// recordingPlanStore captures every plan row a rollup stores, so a test can +// assert both how many member plans were persisted and what each one holds. +type recordingPlanStore struct { + mockPlanLookupStore + created []*storage.Plan + createErr error +} + +func (s *recordingPlanStore) Create(_ context.Context, plan *storage.Plan) (int64, error) { + if s.createErr != nil { + return 0, s.createErr + } + s.created = append(s.created, plan) + return int64(len(s.created)), nil +} + +// multiTargetService builds a database whose production environment addresses +// two distinct targets through one deployment, which is the shape that makes its +// members independently planned. +func multiTargetService(t *testing.T, client *mockTernClient, plans storage.PlanStore) *Service { + t.Helper() + cfg := &ServerConfig{ + Databases: map[string]DatabaseConfig{ + "testapp": { + Type: storage.DatabaseTypeMySQL, + Environments: map[string]EnvironmentConfig{ + "production": { + Deployment: "eu", + Targets: []string{"testapp-001", "testapp-002"}, + }, + }, + }, + }, + } + logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelError})) + return New(&mockStorageWithPlanLookup{plans: plans}, cfg, map[string]tern.Client{ + "eu/production": client, + }, logger) +} + +// mirroredService builds the two-deployment, one-target shape whose members are +// expected to hold the same schema, wired to a plan store so a test can assert +// no per-member plan is written for it. +func mirroredService(t *testing.T, eu, us *mockTernClient, plans storage.PlanStore) *Service { + t.Helper() + cfg := &ServerConfig{ + Databases: map[string]DatabaseConfig{ + "testapp": { + Type: storage.DatabaseTypeMySQL, + Environments: map[string]EnvironmentConfig{ + "production": { + Deployments: map[string]DeploymentTarget{ + "eu": {Target: "testapp"}, + "us": {Target: "testapp"}, + }, + DeploymentOrder: []string{"eu", "us"}, + }, + }, + }, + }, + } + logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelError})) + return New(&mockStorageWithPlanLookup{plans: plans}, cfg, map[string]tern.Client{ + "eu/production": eu, + "us/production": us, + }, logger) +} + +func multiTargetMember(target string) routing.ExecutionTarget { + return routing.ExecutionTarget{Deployment: "eu", Target: target} +} + +// Each target of a multi-target environment holds its own schema, so the second +// target legitimately plans different DDL than the reviewed primary. The review +// stays clean and the second target's plan is persisted as a row of its own, so +// an apply has something to run for it. +func TestRollupReviewTimeDrift_IndependentMemberPlanIsPersisted(t *testing.T) { + reviewed := reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)") + secondDDL := "ALTER TABLE `users` ADD COLUMN `phone` varchar(32)" + plans := &recordingPlanStore{} + svc := multiTargetService(t, &mockTernClient{planDiffResp: alterUsersDiff(secondDDL)}, plans) + + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewed, multiTargetMember("testapp-001")) + require.NoError(t, err) + assert.True(t, rollup.Clean, "targets that hold their own schemas must not block each other") + require.Len(t, rollup.Entries, 2) + + assert.Equal(t, DeploymentPlanned, rollup.Entries[0].Class) + assert.Empty(t, rollup.Entries[0].PlanIdentifier, "the primary runs the plan its apply is created from") + + assert.Equal(t, "testapp-002", rollup.Entries[1].Target) + assert.Equal(t, DeploymentPlanned, rollup.Entries[1].Class) + require.NotEmpty(t, rollup.Entries[1].PlanIdentifier) + assert.True(t, strings.HasPrefix(rollup.Entries[1].PlanIdentifier, "plan-"), + "member plan identifier %q should be a minted plan id", rollup.Entries[1].PlanIdentifier) + + require.Len(t, plans.created, 1, "only the non-primary member needs a plan row of its own") + stored := plans.created[0] + assert.Equal(t, rollup.Entries[1].PlanIdentifier, stored.PlanIdentifier) + assert.Equal(t, "testapp", stored.Database) + assert.Equal(t, "eu", stored.Deployment) + assert.Equal(t, "testapp-002", stored.Target) + assert.Equal(t, "production", stored.Environment) + require.Contains(t, stored.Namespaces, "testapp") + require.Len(t, stored.Namespaces["testapp"].Tables, 1) + assert.Equal(t, "users", stored.Namespaces["testapp"].Tables[0].Table) + assert.Contains(t, stored.Namespaces["testapp"].Tables[0].DDL, "ADD COLUMN `phone`") +} + +// A member whose plan cannot be stored has nothing an apply could run, so it is +// reclassified as errored and blocks the review rather than passing on a plan +// that was never persisted. +func TestRollupReviewTimeDrift_MemberPlanStoreFailureBlocks(t *testing.T) { + reviewed := reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)") + plans := &recordingPlanStore{createErr: errors.New("storage unavailable")} + svc := multiTargetService(t, &mockTernClient{planDiffResp: alterUsersDiff("ALTER TABLE `users` ADD COLUMN `phone` varchar(32)")}, plans) + + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewed, multiTargetMember("testapp-001")) + require.NoError(t, err) + assert.False(t, rollup.Clean, "a member whose plan was not stored must block the review") + require.Len(t, rollup.Entries, 2) + assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) + assert.Empty(t, rollup.Entries[1].PlanIdentifier) + require.Error(t, rollup.Entries[1].Err) + assert.Contains(t, rollup.Entries[1].Err.Error(), "eu/testapp-002") + assert.Contains(t, rollup.Entries[1].Err.Error(), "storage unavailable") +} + +// A member that could not be planned at all never reaches plan storage: it +// already blocks, and there is no plan to write. +func TestRollupReviewTimeDrift_UnplannableMemberStoresNothing(t *testing.T) { + reviewed := reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)") + plans := &recordingPlanStore{} + svc := multiTargetService(t, &mockTernClient{planDiffErr: errors.New("target unreachable")}, plans) + + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewed, multiTargetMember("testapp-001")) + require.NoError(t, err) + assert.False(t, rollup.Clean) + require.Len(t, rollup.Entries, 2) + assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) + assert.Empty(t, plans.created, "an unplannable member has no plan to store") +} + +// Deployments that are expected to hold the same schema all run the reviewed +// plan, so no second plan row is written for them. +func TestRollupReviewTimeDrift_MirroredMembersStoreNoMemberPlan(t *testing.T) { + ddl := "ALTER TABLE `users` ADD COLUMN `email` varchar(255)" + plans := &recordingPlanStore{} + svc := mirroredService(t, &mockTernClient{}, &mockTernClient{planDiffResp: alterUsersDiff(ddl)}, plans) + + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewedUsersPlan(ddl), productionMember("eu")) + require.NoError(t, err) + assert.True(t, rollup.Clean) + require.Len(t, rollup.Entries, 2) + assert.Equal(t, DeploymentMatch, rollup.Entries[1].Class) + assert.Empty(t, rollup.Entries[1].PlanIdentifier) + assert.Empty(t, plans.created, "mirrored members run the reviewed plan") +} + +// A plan identifier that is already stored is the same plan re-stored by a +// re-plan of unchanged content, so the existing row is reused rather than +// treated as a failure. +func TestStorePlan_ExistingIdentifierReusesStoredRow(t *testing.T) { + plans := &existingPlanStore{existing: &storage.Plan{ID: 77, PlanIdentifier: "plan-existing"}} + svc := multiTargetService(t, &mockTernClient{}, plans) + + id, err := svc.storePlan(t.Context(), planDiffReq(t), "plan-existing", nil, nil, + storedPlanRoute{DatabaseType: storage.DatabaseTypeMySQL, Deployment: "eu", Target: "testapp-002"}) + require.NoError(t, err) + assert.Equal(t, int64(77), id) +} + +// A plan reported as already stored but not readable back leaves the caller with +// no plan row to point at, which must be an error rather than a zero ID. +func TestStorePlan_ExistingIdentifierNotReadableBackErrors(t *testing.T) { + plans := &existingPlanStore{} + svc := multiTargetService(t, &mockTernClient{}, plans) + + _, err := svc.storePlan(t.Context(), planDiffReq(t), "plan-vanished", nil, nil, + storedPlanRoute{DatabaseType: storage.DatabaseTypeMySQL, Deployment: "eu", Target: "testapp-002"}) + require.Error(t, err) + assert.Contains(t, err.Error(), "could not be read back") +} + +// existingPlanStore rejects every create as an already-used identifier and +// returns whatever row a read-back finds. +type existingPlanStore struct { + mockPlanLookupStore + existing *storage.Plan +} + +func (s *existingPlanStore) Create(context.Context, *storage.Plan) (int64, error) { + return 0, storage.ErrPlanIDExists +} + +func (s *existingPlanStore) Get(context.Context, string) (*storage.Plan, error) { + return s.existing, nil +} diff --git a/pkg/api/plan_review_drift.go b/pkg/api/plan_review_drift.go index b3a67fcb1..7976b9a3a 100644 --- a/pkg/api/plan_review_drift.go +++ b/pkg/api/plan_review_drift.go @@ -49,6 +49,14 @@ func (s *Service) RollupReviewTimeDrift(ctx context.Context, req PlanRequest, pr return PlanRollup{}, fmt.Errorf("roll up deployment diffs for %s/%s: %w", req.Database, req.Environment, err) } + // Members planned on their own each need a plan row of their own, since an + // apply has no single plan that covers them. Storing them here, before the + // rollup is reported, keeps "the review says this member is fine" and "this + // member has a plan to run" from being separately true. + if err := s.persistMemberPlans(ctx, req, planning, diffs, &rollup); err != nil { + return PlanRollup{}, fmt.Errorf("persist member plans for %s/%s: %w", req.Database, req.Environment, err) + } + // Include repo/pr/head SHA so an operator can tell which PR is blocked from // the drift warn log alone. var pr int32 diff --git a/pkg/api/plan_rollup.go b/pkg/api/plan_rollup.go index 6d9f9d380..12d803dbc 100644 --- a/pkg/api/plan_rollup.go +++ b/pkg/api/plan_rollup.go @@ -83,6 +83,13 @@ type DeploymentRollupEntry struct { Class DeploymentClassification Diff tern.ChangeSetDiff Err error + + // PlanIdentifier names the stored plan this member will run, set when the + // member was planned on its own and its plan was persisted as a row of its + // own. Empty means the member runs the plan the apply itself was created + // from — which is every member under mirrored planning, and the primary + // under either. + PlanIdentifier string } // PlanRollup aggregates every deployment's review-time classification for a From 98320c8d76020a7abfe172549a654c6ccb3af85e Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:47:55 -0400 Subject: [PATCH 09/34] feat(api): fan an apply out to one operation per rollout member An apply resolves each rollout member to the plan its work is built from before building operations. Members of an environment whose members hold the same schema all carry the apply's own plan. A member that was planned against its own live schema carries its own plan, looked up by member id and bound to the head SHA the apply's plan was created for, so plans from an earlier push of the same pull request are never picked up. A member with no plan for that review round fails apply creation. The apply's plan describes a different target's schema, so substituting it would run DDL that target was never planned for. Operations are grouped by member, and a member is its deployment and target together, so two targets of one deployment each get their own operation for a given (namespace, shard, table) rather than sharing one. An operation names a plan of its own only when it runs a different plan than its apply. Co-Authored-By: Claude Fable 5 --- pkg/api/apply_members.go | 123 ++++++++++++ pkg/api/apply_members_test.go | 248 ++++++++++++++++++++++++ pkg/api/plan_handlers.go | 93 +++++---- pkg/api/sharded_pershard_fanout_test.go | 22 ++- 4 files changed, 445 insertions(+), 41 deletions(-) create mode 100644 pkg/api/apply_members.go create mode 100644 pkg/api/apply_members_test.go diff --git a/pkg/api/apply_members.go b/pkg/api/apply_members.go new file mode 100644 index 000000000..5264cf234 --- /dev/null +++ b/pkg/api/apply_members.go @@ -0,0 +1,123 @@ +package api + +import ( + "context" + "fmt" + + "github.com/block/schemabot/pkg/routing" + "github.com/block/schemabot/pkg/storage" +) + +// memberPlanLookupLimit bounds the plan listing that resolves member plans. A +// review round stores one plan per member, and a PR is re-planned on every push, +// so the listing has to reach back far enough to cover several rounds of a +// wide environment while staying a bounded read. +const memberPlanLookupLimit = 200 + +// applyMember is one rollout member of an apply together with the plan its work +// is built from. Members of an environment whose members hold the same schema +// all carry the apply's own plan; a member that was planned against its own live +// schema carries its own. +type applyMember struct { + Target routing.ExecutionTarget + Plan *storage.Plan +} + +// MemberID names the member for logs and errors. +func (m applyMember) MemberID() string { + return m.Target.MemberID() +} + +// resolveApplyMembers pairs each rollout member with the plan its operations +// must be built from. +// +// When the environment's members are expected to hold the same schema, every +// member runs the reviewed plan and the apply's own plan is used throughout. +// +// When the members were planned independently, each one has a plan of its own +// stored at review time, and running one member's DDL against another's target +// would apply a schema that target was never planned for. So each non-primary +// member is paired with its own stored plan, matched on the head SHA the apply's +// plan was created for, which binds the member plans to the same review round +// the operator approved. +// +// A member with no plan for that review round fails apply creation. There is no +// safe fallback: the apply's plan describes a different target's schema, so +// substituting it would run DDL that was never planned for this member. +func (s *Service) resolveApplyMembers(ctx context.Context, plan *storage.Plan, environment string, targets []routing.ExecutionTarget) ([]applyMember, error) { + planning, err := s.config.MemberPlanningFor(plan.Database, environment) + if err != nil { + // A database/environment the config no longer resolves cannot be shown to + // have independently planned members, and mirrored is the shape that reuses + // one plan for every member. Fail rather than assume it. + return nil, fmt.Errorf("resolve member planning for %s/%s: %w", plan.Database, environment, err) + } + + members := make([]applyMember, 0, len(targets)) + if planning == PlanMirrored { + for _, target := range targets { + members = append(members, applyMember{Target: target, Plan: plan}) + } + return members, nil + } + + memberPlans, err := s.memberPlansForReviewRound(ctx, plan, environment) + if err != nil { + return nil, err + } + primary := routing.ExecutionTarget{Deployment: plan.Deployment, Target: plan.Target} + for _, target := range targets { + // The apply is created from the primary's plan, so the primary needs no + // lookup — and must not take one, since a re-plan could have stored a newer + // row for the same member than the plan the operator approved. + if target.Deployment == primary.Deployment && target.Target == primary.Target { + members = append(members, applyMember{Target: target, Plan: plan}) + continue + } + memberPlan, ok := memberPlans[target.MemberID()] + if !ok { + return nil, fmt.Errorf("apply for %s/%s has no stored plan for rollout member %s at head %q; plan the environment again so every target is planned before applying", + plan.Database, environment, target.MemberID(), plan.HeadSHA) + } + members = append(members, applyMember{Target: target, Plan: memberPlan}) + } + return members, nil +} + +// memberPlansForReviewRound loads the member plans stored alongside plan, keyed +// by member id. Plans are listed newest first, so the first row seen for a +// member is that member's latest plan within the round. +// +// The round is identified by the apply plan's head SHA: every member of one +// review round is planned against the same commit, so a plan from an earlier +// push is a different round and must not be picked up. An apply plan with no +// head SHA was not produced by a PR review, which is the only place member plans +// are written, so there is nothing to match it against. +func (s *Service) memberPlansForReviewRound(ctx context.Context, plan *storage.Plan, environment string) (map[string]*storage.Plan, error) { + if plan.HeadSHA == "" { + return nil, fmt.Errorf("apply for %s/%s addresses several targets but its plan has no head SHA to match member plans against; plan from a pull request so every target is planned", + plan.Database, environment) + } + stored, err := s.storage.Plans().List(ctx, storage.ListPlansOptions{ + Database: plan.Database, + Environment: environment, + Repository: plan.Repository, + PullRequest: plan.PullRequest, + Limit: memberPlanLookupLimit, + }) + if err != nil { + return nil, fmt.Errorf("list member plans for %s/%s pr %d: %w", plan.Database, environment, plan.PullRequest, err) + } + byMember := make(map[string]*storage.Plan, len(stored)) + for _, candidate := range stored { + if candidate.HeadSHA != plan.HeadSHA { + continue + } + memberID := routing.ExecutionTarget{Deployment: candidate.Deployment, Target: candidate.Target}.MemberID() + if _, seen := byMember[memberID]; seen { + continue + } + byMember[memberID] = candidate + } + return byMember, nil +} diff --git a/pkg/api/apply_members_test.go b/pkg/api/apply_members_test.go new file mode 100644 index 000000000..7d5b66287 --- /dev/null +++ b/pkg/api/apply_members_test.go @@ -0,0 +1,248 @@ +package api + +import ( + "context" + "errors" + "log/slog" + "os" + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + "github.com/block/schemabot/pkg/routing" + "github.com/block/schemabot/pkg/storage" + "github.com/block/schemabot/pkg/tern" +) + +// listingPlanStore serves a fixed set of stored plans to List, newest first, so +// a test can control exactly which member plans apply creation can find. +type listingPlanStore struct { + mockPlanLookupStore + plans []*storage.Plan + listErr error +} + +func (s *listingPlanStore) List(context.Context, storage.ListPlansOptions) ([]*storage.Plan, error) { + if s.listErr != nil { + return nil, s.listErr + } + return s.plans, nil +} + +func memberResolutionService(t *testing.T, env EnvironmentConfig, plans storage.PlanStore) *Service { + t.Helper() + cfg := &ServerConfig{ + Databases: map[string]DatabaseConfig{ + "testapp": { + Type: storage.DatabaseTypeMySQL, + Environments: map[string]EnvironmentConfig{"production": env}, + }, + }, + } + logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelError})) + return New(&mockStorageWithPlanLookup{plans: plans}, cfg, map[string]tern.Client{}, logger) +} + +func multiTargetEnv() EnvironmentConfig { + return EnvironmentConfig{Deployment: "eu", Targets: []string{"testapp-001", "testapp-002"}} +} + +func mirroredEnv() EnvironmentConfig { + return EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{"eu": {Target: "testapp"}, "us": {Target: "testapp"}}, + DeploymentOrder: []string{"eu", "us"}, + } +} + +func primaryPlanRow(target string) *storage.Plan { + return &storage.Plan{ + ID: 10, + PlanIdentifier: "plan-primary", + Database: "testapp", + DatabaseType: storage.DatabaseTypeMySQL, + Deployment: "eu", + Target: target, + Repository: "org/repo", + PullRequest: 7, + HeadSHA: "abc123", + } +} + +func targetsFor(t *testing.T, svc *Service) []routing.ExecutionTarget { + t.Helper() + targets, err := svc.config.ResolveDatabaseTargets("testapp", "production") + require.NoError(t, err) + return targets +} + +// Members that are expected to hold the same schema all run the plan the +// operator reviewed, so every member is paired with the apply's own plan and no +// member-plan lookup is needed. +func TestResolveApplyMembers_MirroredMembersShareTheApplyPlan(t *testing.T) { + plans := &listingPlanStore{listErr: errors.New("List must not be called for mirrored members")} + svc := memberResolutionService(t, mirroredEnv(), plans) + plan := primaryPlanRow("testapp") + + members, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.NoError(t, err) + require.Len(t, members, 2) + for _, member := range members { + assert.Same(t, plan, member.Plan, "member %s must run the reviewed plan", member.MemberID()) + } +} + +// Each target of a multi-target environment runs the plan stored for that +// target in the same review round, matched on the head SHA the apply's plan was +// created for. +func TestResolveApplyMembers_IndependentMembersRunTheirOwnPlans(t *testing.T) { + plan := primaryPlanRow("testapp-001") + secondPlan := &storage.Plan{ID: 11, PlanIdentifier: "plan-second", Deployment: "eu", Target: "testapp-002", HeadSHA: "abc123"} + plans := &listingPlanStore{plans: []*storage.Plan{secondPlan}} + svc := memberResolutionService(t, multiTargetEnv(), plans) + + members, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.NoError(t, err) + require.Len(t, members, 2) + assert.Equal(t, "eu/testapp-001", members[0].MemberID()) + assert.Same(t, plan, members[0].Plan, "the primary runs the plan the apply was created from") + assert.Equal(t, "eu/testapp-002", members[1].MemberID()) + assert.Same(t, secondPlan, members[1].Plan) +} + +// A plan stored for an earlier push is a different review round. Matching it to +// this apply would run DDL the operator never reviewed on this commit, so the +// member counts as unplanned and apply creation fails. +func TestResolveApplyMembers_MemberPlanFromAnotherRoundIsNotUsed(t *testing.T) { + plan := primaryPlanRow("testapp-001") + stale := &storage.Plan{ID: 9, PlanIdentifier: "plan-stale", Deployment: "eu", Target: "testapp-002", HeadSHA: "older"} + plans := &listingPlanStore{plans: []*storage.Plan{stale}} + svc := memberResolutionService(t, multiTargetEnv(), plans) + + _, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.Error(t, err) + assert.Contains(t, err.Error(), "no stored plan for rollout member eu/testapp-002") +} + +// A member with no plan at all cannot be applied: the apply's plan describes a +// different target's schema, so there is no safe substitute. +func TestResolveApplyMembers_MissingMemberPlanFailsClosed(t *testing.T) { + plan := primaryPlanRow("testapp-001") + plans := &listingPlanStore{} + svc := memberResolutionService(t, multiTargetEnv(), plans) + + _, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.Error(t, err) + assert.Contains(t, err.Error(), "no stored plan for rollout member eu/testapp-002") +} + +// Member plans are only written by a pull request review, so a plan with no head +// SHA has no round to match against and cannot drive a multi-target apply. +func TestResolveApplyMembers_PlanWithoutHeadSHAFailsClosed(t *testing.T) { + plan := primaryPlanRow("testapp-001") + plan.HeadSHA = "" + plans := &listingPlanStore{} + svc := memberResolutionService(t, multiTargetEnv(), plans) + + _, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.Error(t, err) + assert.Contains(t, err.Error(), "no head SHA to match member plans against") +} + +// A storage failure while loading member plans is not an absence of members: it +// blocks apply creation rather than falling back to the apply's plan. +func TestResolveApplyMembers_MemberPlanLookupFailureBlocks(t *testing.T) { + plan := primaryPlanRow("testapp-001") + plans := &listingPlanStore{listErr: errors.New("storage unavailable")} + svc := memberResolutionService(t, multiTargetEnv(), plans) + + _, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.Error(t, err) + assert.Contains(t, err.Error(), "storage unavailable") +} + +// An operation names a plan of its own exactly when it runs a different plan +// than its apply. Members that run the apply's plan leave it unset, which is +// what "runs the apply's plan" already means downstream. +func TestNewPendingApplyOperation_StampsPlanIDOnlyForOwnPlan(t *testing.T) { + applyPlan := primaryPlanRow("testapp-001") + ownPlan := &storage.Plan{ID: 11, Deployment: "eu", Target: "testapp-002"} + now := pershardTestTime() + + shared := newPendingApplyOperation( + applyMember{Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-001"}, Plan: applyPlan}, + applyPlan, "", "", "", now) + assert.Zero(t, shared.PlanID, "a member running the apply's plan names no plan of its own") + assert.Equal(t, "testapp-001", shared.Target) + + own := newPendingApplyOperation( + applyMember{Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}, Plan: ownPlan}, + applyPlan, "", "", "", now) + assert.Equal(t, int64(11), own.PlanID) + assert.Equal(t, "testapp-002", own.Target) + assert.Equal(t, "eu", own.Deployment) +} + +// Two targets of one deployment are two distinct members, so each gets its own +// operation for the same table rather than one member's work being folded into +// the other's. +func TestBuildApplyOperationGroups_TargetsOfOneDeploymentGetOwnOperations(t *testing.T) { + applyPlan := primaryPlanRow("testapp-001") + secondPlan := &storage.Plan{ID: 11, Deployment: "eu", Target: "testapp-002"} + members := []applyMember{ + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-001"}, Plan: applyPlan}, + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}, Plan: secondPlan}, + } + taskChanges := []storage.TableChange{{Namespace: "testapp", Table: "users", DDL: "ALTER TABLE `users` ADD COLUMN `email` varchar(255)", Operation: "alter"}} + + groups, sharded, err := buildApplyOperationGroups(applyPlan, taskChanges, members, "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + require.NoError(t, err) + assert.False(t, sharded) + require.Len(t, groups, 2) + assert.Equal(t, "testapp-001", groups[0].Operation.Target) + assert.Zero(t, groups[0].Operation.PlanID) + assert.Equal(t, "testapp-002", groups[1].Operation.Target) + assert.Equal(t, int64(11), groups[1].Operation.PlanID) +} + +// A sharded plan produces the same operation key for the same (namespace, shard, +// table) on every member, so operations are grouped by member and key together. +// Two targets of one deployment each get their own operation for that key rather +// than one target's shard work being folded into the other's. +func TestBuildShardedApplyOperationGroups_TargetsOfOneDeploymentDoNotShareOperations(t *testing.T) { + mutesDDL := "ALTER TABLE `mutes` ADD INDEX (`created_at`)" + applyPlan := &storage.Plan{ + ID: 10, + Database: "testapp", + Shards: []storage.ShardPlan{ + {Namespace: pershardNamespace, Shard: "-80", Changes: []storage.TableChange{ + {Namespace: pershardNamespace, Table: "mutes", DDL: mutesDDL, Operation: "alter"}, + }}, + }, + } + secondPlan := &storage.Plan{ID: 11, Database: "testapp", Shards: applyPlan.Shards} + members := []applyMember{ + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-001"}, Plan: applyPlan}, + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}, Plan: secondPlan}, + } + + groups, err := buildShardedApplyOperationGroups(applyPlan, members, "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + require.NoError(t, err) + require.Len(t, groups, 2, "each target needs its own operation for the shard's table") + + byTarget := map[string]*storage.ApplyOperationWithTasks{} + for _, g := range groups { + byTarget[g.Operation.Target] = g + } + require.Contains(t, byTarget, "testapp-001") + require.Contains(t, byTarget, "testapp-002") + for target, group := range byTarget { + assert.Equal(t, pershardNamespace+"/-80/mutes", group.Operation.OperationKey) + assert.Equal(t, "eu", group.Operation.Deployment) + require.Len(t, group.Tasks, 1, "target %s must carry its shard's work exactly once", target) + assert.Equal(t, mutesDDL, group.Tasks[0].DDL) + assert.Equal(t, "-80", group.Tasks[0].Shard) + } + assert.Zero(t, byTarget["testapp-001"].Operation.PlanID) + assert.Equal(t, int64(11), byTarget["testapp-002"].Operation.PlanID) +} diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index ac92907f3..bf0ab033b 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -1349,7 +1349,11 @@ func (s *Service) createStoredApply( taskChanges := applyTaskChanges(plan) cutoverPolicy := s.config.CutoverPolicyFor(plan.Database, req.Environment) onFailure := s.config.OnFailure(plan.Database, req.Environment) - groups, shardedFanout, err := buildApplyOperationGroups(plan, taskChanges, targets, req.Environment, applyOpts, cutoverPolicy, onFailure, now) + members, err := s.resolveApplyMembers(ctx, plan, req.Environment, targets) + if err != nil { + return nil, 0, err + } + groups, shardedFanout, err := buildApplyOperationGroups(plan, taskChanges, members, req.Environment, applyOpts, cutoverPolicy, onFailure, now) if err != nil { return nil, 0, err } @@ -1431,7 +1435,7 @@ func applyTaskChanges(plan *storage.Plan) []storage.TableChange { func buildApplyOperationGroups( plan *storage.Plan, taskChanges []storage.TableChange, - targets []routing.ExecutionTarget, + members []applyMember, environment string, applyOpts storage.ApplyOptions, cutoverPolicy string, @@ -1443,7 +1447,7 @@ func buildApplyOperationGroups( // externally-authoritative engine (e.g. PlanetScale) — whose plans never // carry per-shard changes — is never fanned out, regardless of transport. if canBuildShardedOperationGroups(plan, taskChanges) { - groups, err := buildShardedApplyOperationGroups(plan, targets, environment, applyOpts, cutoverPolicy, onFailure, now) + groups, err := buildShardedApplyOperationGroups(plan, members, environment, applyOpts, cutoverPolicy, onFailure, now) if err != nil { return nil, false, err } @@ -1460,20 +1464,23 @@ func buildApplyOperationGroups( // every keyspace in it, so splitting the namespaces across operations would // have each drive validating keyspaces whose VSchema it never applied. if len(taskChanges) == 0 && len(plan.VSchemaNamespaces()) > 0 { - groups := make([]*storage.ApplyOperationWithTasks, 0, len(targets)) - for _, target := range targets { - operation := newPendingApplyOperation(target, finalizerOperationKeySegment, cutoverPolicy, onFailure, now) + groups := make([]*storage.ApplyOperationWithTasks, 0, len(members)) + for _, member := range members { + operation := newPendingApplyOperation(member, plan, finalizerOperationKeySegment, cutoverPolicy, onFailure, now) operation.OperationKind = storage.ApplyOperationKindGroupFinalizer groups = append(groups, &storage.ApplyOperationWithTasks{Operation: operation}) } return groups, false, nil } - groups := make([]*storage.ApplyOperationWithTasks, 0, len(targets)) - for _, target := range targets { - tasks := buildApplyTasks(plan, taskChanges, environment, applyOpts, "", now) + groups := make([]*storage.ApplyOperationWithTasks, 0, len(members)) + for _, member := range members { + // A member planned on its own runs its own changes, not the apply plan's: + // its target holds a schema the apply plan never described. + memberChanges := applyTaskChanges(member.Plan) + tasks := buildApplyTasks(member.Plan, memberChanges, environment, applyOpts, "", now) groups = append(groups, &storage.ApplyOperationWithTasks{ - Operation: newPendingApplyOperation(target, "", cutoverPolicy, onFailure, now), + Operation: newPendingApplyOperation(member, plan, "", cutoverPolicy, onFailure, now), Tasks: tasks, }) } @@ -1487,13 +1494,13 @@ func buildApplyOperationGroups( // synthetic task. A namespace with no shard work still gets a finalizer so its // VSchema change is never dropped. func buildNamespaceFinalizerOperations( - plan *storage.Plan, - target routing.ExecutionTarget, + applyPlan *storage.Plan, + member applyMember, cutoverPolicy string, onFailure string, now time.Time, ) ([]*storage.ApplyOperationWithTasks, error) { - namespaces := plan.VSchemaNamespaces() + namespaces := member.Plan.VSchemaNamespaces() groups := make([]*storage.ApplyOperationWithTasks, 0, len(namespaces)) for _, namespace := range namespaces { if err := validateOperationKeyPart("namespace", namespace); err != nil { @@ -1503,7 +1510,7 @@ func buildNamespaceFinalizerOperations( if len(operationKey) > applyOperationKeyMaxLen { return nil, fmt.Errorf("operation key for namespace %q finalizer exceeds %d characters", namespace, applyOperationKeyMaxLen) } - operation := newPendingApplyOperation(target, operationKey, cutoverPolicy, onFailure, now) + operation := newPendingApplyOperation(member, applyPlan, operationKey, cutoverPolicy, onFailure, now) operation.OperationKind = storage.ApplyOperationKindGroupFinalizer groups = append(groups, &storage.ApplyOperationWithTasks{ Operation: operation, @@ -1534,24 +1541,30 @@ func canBuildShardedOperationGroups(plan *storage.Plan, taskChanges []storage.Ta } func buildShardedApplyOperationGroups( - plan *storage.Plan, - targets []routing.ExecutionTarget, + applyPlan *storage.Plan, + members []applyMember, environment string, applyOpts storage.ApplyOptions, cutoverPolicy string, onFailure string, now time.Time, ) ([]*storage.ApplyOperationWithTasks, error) { - shardsByNamespace := changingShardsByNamespace(plan.Shards) - namespaces := make([]string, 0, len(shardsByNamespace)) - for namespace := range shardsByNamespace { - namespaces = append(namespaces, namespace) - } - sort.Strings(namespaces) + groups := make([]*storage.ApplyOperationWithTasks, 0, len(members)*(len(applyPlan.Shards)+1)) + // One group per (member, operation key). A member is identified by its + // deployment and target together, so two targets of one deployment get their + // own groups instead of one member's shard work being folded into the other's. + groupsByMemberAndKey := make(map[string]*storage.ApplyOperationWithTasks) + for _, member := range members { + // A member planned on its own carries its own shards and changes; a member + // of a mirrored environment carries the apply's plan, so this is the same + // shard set for every member there. + shardsByNamespace := changingShardsByNamespace(member.Plan.Shards) + namespaces := make([]string, 0, len(shardsByNamespace)) + for namespace := range shardsByNamespace { + namespaces = append(namespaces, namespace) + } + sort.Strings(namespaces) - groups := make([]*storage.ApplyOperationWithTasks, 0, len(targets)*(len(plan.Shards)+1)) - groupsByTargetAndKey := make(map[string]*storage.ApplyOperationWithTasks) - for _, target := range targets { for _, namespace := range namespaces { for _, shard := range shardsByNamespace[namespace] { // Each shard is driven from its own changes; it is in @@ -1572,20 +1585,20 @@ func buildShardedApplyOperationGroups( if len(operationKey) > applyOperationKeyMaxLen { return nil, fmt.Errorf("operation key for namespace %q shard %q table %q exceeds %d characters", namespace, shard.Shard, ddlChange.Table, applyOperationKeyMaxLen) } - groupKey := target.Deployment + "\x00" + operationKey - group := groupsByTargetAndKey[groupKey] + groupKey := member.MemberID() + "\x00" + operationKey + group := groupsByMemberAndKey[groupKey] if group == nil { group = &storage.ApplyOperationWithTasks{ - Operation: newPendingApplyOperation(target, operationKey, cutoverPolicy, onFailure, now), + Operation: newPendingApplyOperation(member, applyPlan, operationKey, cutoverPolicy, onFailure, now), } - groupsByTargetAndKey[groupKey] = group + groupsByMemberAndKey[groupKey] = group groups = append(groups, group) } - group.Tasks = append(group.Tasks, buildApplyTask(plan, ddlChange, environment, applyOpts, shard.Shard, now)) + group.Tasks = append(group.Tasks, buildApplyTask(member.Plan, ddlChange, environment, applyOpts, shard.Shard, now)) } } } - finalizers, err := buildNamespaceFinalizerOperations(plan, target, cutoverPolicy, onFailure, now) + finalizers, err := buildNamespaceFinalizerOperations(applyPlan, member, cutoverPolicy, onFailure, now) if err != nil { return nil, err } @@ -1638,18 +1651,28 @@ func finalizerOperationKey(namespace string) string { return namespace + "/" + finalizerOperationKeySegment } -func newPendingApplyOperation(target routing.ExecutionTarget, operationKey, cutoverPolicy, onFailure string, now time.Time) *storage.ApplyOperation { - return &storage.ApplyOperation{ - Deployment: target.Deployment, +// newPendingApplyOperation builds one member's pending operation. +// +// applyPlan is the plan the apply itself was created from. PlanID is stamped +// only when the member runs a different plan, so an operation names a plan of +// its own exactly when it was planned on its own — every other operation runs +// its apply's plan, which is what an unset PlanID already means. +func newPendingApplyOperation(member applyMember, applyPlan *storage.Plan, operationKey, cutoverPolicy, onFailure string, now time.Time) *storage.ApplyOperation { + op := &storage.ApplyOperation{ + Deployment: member.Target.Deployment, OperationKey: operationKey, OperationKind: storage.ApplyOperationKindWork, - Target: target.Target, + Target: member.Target.Target, State: state.ApplyOperation.Pending, CutoverPolicy: cutoverPolicy, OnFailure: onFailure, CreatedAt: now, UpdatedAt: now, } + if member.Plan != nil && applyPlan != nil && member.Plan.ID != applyPlan.ID { + op.PlanID = member.Plan.ID + } + return op } func buildApplyTasks( diff --git a/pkg/api/sharded_pershard_fanout_test.go b/pkg/api/sharded_pershard_fanout_test.go index 1e37d264d..5c921fa55 100644 --- a/pkg/api/sharded_pershard_fanout_test.go +++ b/pkg/api/sharded_pershard_fanout_test.go @@ -18,6 +18,16 @@ func pershardTargets() []routing.ExecutionTarget { return []routing.ExecutionTarget{{DatabaseType: storage.DatabaseTypeStrata, Deployment: "cdb-resolute", Target: "cdb-resolute"}} } +// pershardMembers is the single rollout member these tests build operations +// for, paired with the plan its work comes from. +func pershardMembers(plan *storage.Plan) []applyMember { + members := make([]applyMember, 0, 1) + for _, target := range pershardTargets() { + members = append(members, applyMember{Target: target, Plan: plan}) + } + return members +} + func pershardTestTime() time.Time { return time.Unix(1700000000, 0).UTC() } // operationKeys returns each group's operation key paired with its tasks' DDLs, @@ -56,7 +66,7 @@ func TestBuildShardedApplyOperationGroupsUsesPerShardDDL(t *testing.T) { }, } - groups, err := buildShardedApplyOperationGroups(plan, pershardTargets(), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) got := operationDDLByKey(groups) @@ -86,7 +96,7 @@ func TestBuildShardedApplyOperationGroupsSkipsShardsWithoutChanges(t *testing.T) }, } - groups, err := buildShardedApplyOperationGroups(plan, pershardTargets(), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) assert.Equal(t, map[string][]string{ @@ -130,7 +140,7 @@ func TestBuildShardedApplyOperationGroupsFailsClosedOnMalformedChange(t *testing Changes: []storage.TableChange{{Namespace: pershardNamespace, Table: "", DDL: "ALTER TABLE `mutes` ADD INDEX (`x`)", Operation: "alter"}}, }}, } - _, err := buildShardedApplyOperationGroups(plan, pershardTargets(), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + _, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.Error(t, err) assert.Contains(t, err.Error(), "empty table") } @@ -169,7 +179,7 @@ func TestBuildApplyOperationGroupsVSchemaOnlyPlanBuildsFinalizer(t *testing.T) { }, } - groups, shardedFanout, err := buildApplyOperationGroups(plan, nil, pershardTargets(), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, shardedFanout, err := buildApplyOperationGroups(plan, nil, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) assert.False(t, shardedFanout) require.Len(t, groups, 1) @@ -192,7 +202,7 @@ func TestBuildApplyOperationGroupsVSchemaOnlyPlanMultiNamespaceSingleFinalizer(t }, } - groups, shardedFanout, err := buildApplyOperationGroups(plan, nil, pershardTargets(), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, shardedFanout, err := buildApplyOperationGroups(plan, nil, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) assert.False(t, shardedFanout) require.Len(t, groups, 1) @@ -217,7 +227,7 @@ func TestBuildApplyOperationGroupsTableDDLKeepsWorkShape(t *testing.T) { taskChanges := plan.FlatDDLChanges() require.Len(t, taskChanges, 1) - groups, shardedFanout, err := buildApplyOperationGroups(plan, taskChanges, pershardTargets(), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, shardedFanout, err := buildApplyOperationGroups(plan, taskChanges, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) assert.False(t, shardedFanout) require.Len(t, groups, 1) From 595b8555aad3e497604d19b5a17c190c0303836a Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 10 Sep 2026 18:08:41 -0400 Subject: [PATCH 10/34] feat(api): name every rollout member on a multi-target pull A pull of an environment whose targets each hold their own schema listed only the targets compared against the primary, leaving the primary itself implicit in the response body. A caller reconciling the environment against its own shard inventory then reads a four-target environment as three, and can only recover the missing member by knowing which target the schema in `namespaces` came from. Every target the environment addresses is now named on `targets`, in configuration order, with the primary carrying `"primary": true` and no comparison of its own. The primary's schema is already in hand from the pull that produced the response body, so naming it costs no extra round trip. Co-Authored-By: Claude Fable 5 --- pkg/api/plan_handlers.go | 6 +-- pkg/api/pull_members.go | 30 +++++++++++---- pkg/api/pull_members_test.go | 51 +++++++++++++++++++++---- pkg/apitypes/apitypes.go | 25 ++++++++---- pkg/cmd/internal/templates/pull.go | 19 ++++++--- pkg/cmd/internal/templates/pull_test.go | 12 ++++-- 6 files changed, 109 insertions(+), 34 deletions(-) diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index 42c996f02..54bc49196 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -314,7 +314,7 @@ func (s *Service) ExecutePullSchema(ctx context.Context, req apitypes.PullSchema // schema, so the primary's is reported alongside how the others differ from // it. Comparing here, rather than leaving it to the caller, keeps a pull from // presenting one target's schema as the environment's. - divergences, err := s.pullMemberDivergence(ctx, req, resolvedTarget, merged, namespaces, catalogDetail) + members, err := s.pullMemberDivergence(ctx, req, resolvedTarget, merged, namespaces, catalogDetail) if err != nil { span.RecordError(err) span.SetStatus(otelcodes.Error, "compare rollout members") @@ -328,7 +328,7 @@ func (s *Service) ExecutePullSchema(ctx context.Context, req apitypes.PullSchema "environment", merged.Environment, "table_count", merged.TableCount, "namespace_count", len(merged.Namespaces), - "compared_target_count", len(divergences), + "member_target_count", len(members), ) httpResp := pullSchemaResponseFromProto(merged) @@ -338,7 +338,7 @@ func (s *Service) ExecutePullSchema(ctx context.Context, req apitypes.PullSchema if dbConfig, ok := s.config.DatabaseConfigs()[req.Database]; ok { httpResp.App = dbConfig.App } - httpResp.Targets = divergences + httpResp.Targets = members if req.Lint { if err := lintPulledNamespaces(httpResp); err != nil { span.RecordError(err) diff --git a/pkg/api/pull_members.go b/pkg/api/pull_members.go index 3be6a1372..748f4822d 100644 --- a/pkg/api/pull_members.go +++ b/pkg/api/pull_members.go @@ -101,9 +101,16 @@ func (s *Service) pullTargetSchema( return merged, nil } -// pullMemberDivergence pulls every non-primary rollout member of an environment -// whose members hold their own schemas and reports how each one's live schema -// differs from the primary's. +// pullMemberDivergence reports every rollout member of an environment whose +// members hold their own schemas: the primary, marked as such and carrying no +// comparison of its own, and each other member with how its live schema differs +// from the primary's. +// +// The primary is listed rather than left implicit so the response names the +// whole member set. A caller reconciling an environment against its own shard +// inventory can then read the members straight off the payload, instead of +// having to know that the schema in Namespaces belongs to a member the list +// omits — which would leave a four-target environment describing three. // // It returns nil for an environment whose members are expected to hold the same // schema: pulling them would cost one round trip per member to learn what the @@ -149,9 +156,18 @@ func (s *Service) pullMemberDivergence( return nil, fmt.Errorf("canonicalize schema of rollout member %s: %w", primary.MemberID(), err) } - divergences := make([]*apitypes.TargetDivergence, 0, len(targets)-1) + members := make([]*apitypes.TargetDivergence, 0, len(targets)) for _, target := range targets { - if target.Deployment == primary.Deployment && target.Target == primary.Target { + // The primary is already pulled — its schema is what every other member + // is compared against — so it is recorded from the schema in hand rather + // than fetched a second time, and carries no comparison against itself. + if target.MemberID() == primary.MemberID() { + members = append(members, &apitypes.TargetDivergence{ + Deployment: target.Deployment, + Target: target.Target, + TableCount: primarySchema.TableCount, + Primary: true, + }) continue } memberSchema, err := s.pullTargetSchema(ctx, req, target, namespaces, catalogDetail) @@ -162,14 +178,14 @@ func (s *Service) pullMemberDivergence( if err != nil { return nil, fmt.Errorf("canonicalize schema of rollout member %s: %w", target.MemberID(), err) } - divergences = append(divergences, &apitypes.TargetDivergence{ + members = append(members, &apitypes.TargetDivergence{ Deployment: target.Deployment, Target: target.Target, TableCount: memberSchema.TableCount, DivergedTables: divergedTables(primaryTables, memberTables), }) } - return divergences, nil + return members, nil } // namespaceTable identifies one pulled table within its namespace. diff --git a/pkg/api/pull_members_test.go b/pkg/api/pull_members_test.go index aa9492427..7f5099db8 100644 --- a/pkg/api/pull_members_test.go +++ b/pkg/api/pull_members_test.go @@ -108,8 +108,17 @@ func TestExecutePullSchema_ReportsPerTargetDivergence(t *testing.T) { assert.Equal(t, pullUsersDDL, resp.Namespaces["testapp"].Tables["users"], "the response body is the primary's schema") assert.Equal(t, []string{"testapp-001", "testapp-002"}, client.pulledTargets(), "every target is pulled") - require.Len(t, resp.Targets, 1, "only non-primary targets are compared against the primary") - diverged := resp.Targets[0] + require.Len(t, resp.Targets, 2, "every target the environment addresses is named, primary included") + + primary := resp.Targets[0] + assert.True(t, primary.Primary) + assert.Equal(t, "eu", primary.Deployment) + assert.Equal(t, "testapp-001", primary.Target) + assert.Equal(t, int32(1), primary.TableCount) + assert.Empty(t, primary.DivergedTables, "the primary is the baseline and never diverges from itself") + + diverged := resp.Targets[1] + assert.False(t, diverged.Primary) assert.Equal(t, "eu", diverged.Deployment) assert.Equal(t, "testapp-002", diverged.Target) assert.Equal(t, int32(2), diverged.TableCount) @@ -119,6 +128,34 @@ func TestExecutePullSchema_ReportsPerTargetDivergence(t *testing.T) { }, diverged.DivergedTables) } +// A caller reconciling an environment against its own shard inventory reads the +// member set off the pull payload, so exactly one target is marked primary and +// the whole configured list is present in configuration order. +func TestExecutePullSchema_NamesEveryTargetExactlyOncePrimaryFirst(t *testing.T) { + client := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{ + "testapp-001": pulledTables(map[string]string{"users": pullUsersDDL}), + "testapp-002": pulledTables(map[string]string{"users": pullUsersDDL}), + "testapp-003": pulledTables(map[string]string{"users": pullUsersDDL}), + }, nil) + env := EnvironmentConfig{Deployment: "eu", Targets: []string{"testapp-001", "testapp-002", "testapp-003"}} + svc := pullTargetService(t, env, map[string]tern.Client{"eu/production": client}) + + resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.NoError(t, err) + + named := make([]string, 0, len(resp.Targets)) + primaries := 0 + for _, target := range resp.Targets { + named = append(named, target.Target) + if target.Primary { + primaries++ + } + } + assert.Equal(t, []string{"testapp-001", "testapp-002", "testapp-003"}, named) + assert.Equal(t, 1, primaries, "exactly one target is the baseline the others are compared against") + assert.True(t, resp.Targets[0].Primary) +} + // Targets that hold the same schema report no diverged tables, which is a // positive statement that they agree rather than an absence of information. func TestExecutePullSchema_ConvergedTargetsReportNoDivergedTables(t *testing.T) { @@ -131,9 +168,9 @@ func TestExecutePullSchema_ConvergedTargetsReportNoDivergedTables(t *testing.T) resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) require.NoError(t, err) - require.Len(t, resp.Targets, 1) - assert.Equal(t, "testapp-002", resp.Targets[0].Target) - assert.Empty(t, resp.Targets[0].DivergedTables) + require.Len(t, resp.Targets, 2) + assert.Equal(t, "testapp-002", resp.Targets[1].Target) + assert.Empty(t, resp.Targets[1].DivergedTables) } // Two targets holding the same schema written differently are not diverged: @@ -151,8 +188,8 @@ func TestExecutePullSchema_FormattingIsNotDivergence(t *testing.T) { resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) require.NoError(t, err) - require.Len(t, resp.Targets, 1) - assert.Empty(t, resp.Targets[0].DivergedTables, "the same schema written differently is the same schema") + require.Len(t, resp.Targets, 2) + assert.Empty(t, resp.Targets[1].DivergedTables, "the same schema written differently is the same schema") } // A target that cannot be pulled fails the request. Returning the primary's diff --git a/pkg/apitypes/apitypes.go b/pkg/apitypes/apitypes.go index 7845fbce4..1b0f10f39 100644 --- a/pkg/apitypes/apitypes.go +++ b/pkg/apitypes/apitypes.go @@ -451,8 +451,9 @@ type PullSchemaRequest struct { // // Namespaces holds the primary target's live schema, which is the schema a // caller materializes. Targets is populated only for an environment whose -// targets each hold their own schema, and describes how every other target -// differs from the primary. +// targets each hold their own schema. It names every target the environment +// addresses, including the primary, and describes how each of the others +// differs from it. type PullSchemaResponse struct { Database string `json:"database"` Type string `json:"type"` @@ -476,14 +477,24 @@ const ( DivergenceOnlyOnTarget = "only_on_target" ) -// TargetDivergence reports how one non-primary target's live schema differs from -// the primary's. An empty DivergedTables means the two targets hold the same +// TargetDivergence reports how one target's live schema differs from the +// primary's. An empty DivergedTables means the two targets hold the same // schema; it never means the comparison was skipped, since a target that could // not be pulled or compared fails the pull instead. +// +// Exactly one entry carries Primary, and it is the target whose schema +// Namespaces holds. It is listed alongside the others so the response names the +// environment's whole member set: a caller reconciling shards against its own +// inventory can read the members off the payload instead of having to know +// which target was left out for being the baseline. type TargetDivergence struct { - Deployment string `json:"deployment"` - Target string `json:"target"` - TableCount int32 `json:"table_count"` + Deployment string `json:"deployment"` + Target string `json:"target"` + TableCount int32 `json:"table_count"` + // Primary marks the target the other targets are compared against, whose + // schema is the one in PullSchemaResponse.Namespaces. It never carries + // diverged tables, since it is the baseline of the comparison. + Primary bool `json:"primary,omitempty"` DivergedTables []DivergedTable `json:"diverged_tables,omitempty"` } diff --git a/pkg/cmd/internal/templates/pull.go b/pkg/cmd/internal/templates/pull.go index c63de97ac..3e99a1e92 100644 --- a/pkg/cmd/internal/templates/pull.go +++ b/pkg/cmd/internal/templates/pull.go @@ -69,22 +69,29 @@ var divergenceLabels = map[string]string{ apitypes.DivergenceOnlyOnTarget: "extra", } -// writeTargetDivergence reports how each of an environment's other targets -// differs from the primary, whose schema is the DDL printed below. It renders as -// "--" comments like the rest of the pull output, so redirecting a multi-target -// pull into a .sql file still produces valid SQL. +// writeTargetDivergence lists every target the environment addresses and how +// each differs from the primary, whose schema is the DDL printed below. It +// renders as "--" comments like the rest of the pull output, so redirecting a +// multi-target pull into a .sql file still produces valid SQL. // // A target with no diverged tables is still listed: "these two hold the same // schema" is the answer an operator is usually looking for, and omitting the // converged targets would leave it indistinguishable from not having checked. -// An environment whose targets are expected to hold the same schema carries no -// divergence at all and prints nothing. +// The primary is named on the same list for the same reason — an operator +// counting targets against what they expect the environment to hold should not +// have to add one back. An environment whose targets are expected to hold the +// same schema carries no member list at all and prints nothing. func writeTargetDivergence(targets []*apitypes.TargetDivergence) { if len(targets) == 0 { return } for _, target := range targets { fmt.Println() + if target.Primary { + fmt.Println(annotation(fmt.Sprintf("-- Target %s — primary target, whose schema is below", + emphasis("`"+target.Target+"`")))) + continue + } if len(target.DivergedTables) == 0 { fmt.Println(annotation(fmt.Sprintf("-- Target %s — same schema as the primary target", emphasis("`"+target.Target+"`")))) diff --git a/pkg/cmd/internal/templates/pull_test.go b/pkg/cmd/internal/templates/pull_test.go index c6a1d731e..8edebb5d9 100644 --- a/pkg/cmd/internal/templates/pull_test.go +++ b/pkg/cmd/internal/templates/pull_test.go @@ -233,9 +233,11 @@ func TestWritePullSchema_StylesOutputOnInteractiveTerminals(t *testing.T) { // A pull of an environment whose targets each hold their own schema reports how // every other target differs from the primary, whose schema is the DDL printed -// below. A converged target says so rather than being omitted, so "they agree" -// is distinguishable from "not checked". The whole section renders as "--" -// comments, so a redirected pull stays valid SQL. +// below. Every target is listed, the primary included: a converged target says +// so rather than being omitted, so "they agree" is distinguishable from "not +// checked", and an operator counting members against what they expect the +// environment to hold does not have to add the primary back. The whole section +// renders as "--" comments, so a redirected pull stays valid SQL. func TestWritePullSchema_RendersPerTargetDivergence(t *testing.T) { setColors(t, false) out := captureStdout(t, func() { @@ -248,6 +250,7 @@ func TestWritePullSchema_RendersPerTargetDivergence(t *testing.T) { "orders": {Tables: map[string]string{"users": "CREATE TABLE `users` (`id` bigint NOT NULL);\n"}}, }, Targets: []*apitypes.TargetDivergence{ + {Deployment: "eu", Target: "orders-001", TableCount: 1, Primary: true}, {Deployment: "eu", Target: "orders-002", TableCount: 2, DivergedTables: []apitypes.DivergedTable{ {Namespace: "orders", Table: "audits", Difference: apitypes.DivergenceOnlyOnTarget}, {Namespace: "orders", Table: "users", Difference: apitypes.DivergenceDiffers}, @@ -257,12 +260,13 @@ func TestWritePullSchema_RendersPerTargetDivergence(t *testing.T) { }) }) + assert.Contains(t, out, "-- Target `orders-001` — primary target, whose schema is below") assert.Contains(t, out, "-- Target `orders-002` — 2 tables differ from the primary target") assert.Contains(t, out, "-- orders.audits: extra") assert.Contains(t, out, "-- orders.users: differs") assert.Contains(t, out, "-- Target `orders-003` — same schema as the primary target") for line := range strings.SplitSeq(out, "\n") { - if strings.Contains(line, "orders-002") || strings.Contains(line, "orders-003") || strings.Contains(line, "orders.audits") { + if strings.Contains(line, "orders-001") || strings.Contains(line, "orders-002") || strings.Contains(line, "orders-003") || strings.Contains(line, "orders.audits") { assert.True(t, strings.HasPrefix(strings.TrimSpace(line), "--"), "divergence line %q must be a SQL comment", line) } From b82c0b2c81beb31a0409b7a44814db759f2e6940 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:13:47 -0400 Subject: [PATCH 11/34] feat(api): plan multi-target environments independently instead of blocking on drift An environment that spells its routing as a targets list holds several distinct databases under one deployment, so its members are not expected to hold the same schema and a difference between them is not drift. Plan such an environment's members independently: classify each member that produced a usable plan as Planned rather than comparing it to the reviewed primary. The choice is fail-closed. An environment plans independently only when a targets list appears, at the environment level or inside a deployments entry; every other shape, including every config that predates the targets spelling, stays mirrored and keeps blocking on drift. Independent planning still blocks on a member that could not be planned: a producer error, or change content that will not parse under the member's own grammar. Co-Authored-By: Claude Fable 5 --- pkg/api/config.go | 36 ++++++++++++ pkg/api/config_test.go | 54 +++++++++++++++++ pkg/api/plan_review_drift.go | 8 ++- pkg/api/plan_rollup.go | 93 +++++++++++++++++++++++++++-- pkg/api/plan_rollup_test.go | 111 +++++++++++++++++++++++++++++------ 5 files changed, 278 insertions(+), 24 deletions(-) diff --git a/pkg/api/config.go b/pkg/api/config.go index 8006583eb..338a91779 100644 --- a/pkg/api/config.go +++ b/pkg/api/config.go @@ -2748,6 +2748,42 @@ func validateDeploymentOrder(deployments map[string]DeploymentTarget, order []st return nil } +// MemberPlanningFor reports how the rollout members of a database/environment +// relate to each other, which decides whether a difference between them is +// drift that blocks a review. +// +// An environment plans its members independently when it spells any of its +// routing as a targets list — at the environment level or inside a deployments +// map entry. Every other shape is mirrored, so an environment that predates the +// targets spelling, or does not use it, keeps blocking on drift. +// +// The choice is per environment rather than per member: an environment whose +// members are distinct targets has no pair of members that should be expected +// to match, so extending independent planning to the whole environment does not +// silence a comparison that would otherwise have been meaningful. +func (c *ServerConfig) MemberPlanningFor(database, environment string) (MemberPlanning, error) { + if c == nil { + return PlanMirrored, fmt.Errorf("server config is nil") + } + dbConfig := c.Database(database) + if dbConfig == nil { + return PlanMirrored, &DatabaseNotConfiguredError{Database: database} + } + envConfig, ok := dbConfig.Environments[environment] + if !ok { + return PlanMirrored, &EnvironmentNotConfiguredError{Database: database, Environment: environment} + } + if envConfig.Targets != nil { + return PlanIndependent, nil + } + for _, dt := range envConfig.Deployments { + if dt.Targets != nil { + return PlanIndependent, nil + } + } + return PlanMirrored, nil +} + // ResolveTargets implements routing.Resolver using this server's static // configuration. func (c *ServerConfig) ResolveTargets(_ context.Context, req routing.Request) ([]routing.ExecutionTarget, error) { diff --git a/pkg/api/config_test.go b/pkg/api/config_test.go index 66805651e..4ea7c11a2 100644 --- a/pkg/api/config_test.go +++ b/pkg/api/config_test.go @@ -5056,3 +5056,57 @@ func TestEnvironmentConfig_UsesTargetsList(t *testing.T) { }) } } + +// TestServerConfig_MemberPlanningFor covers which environments plan their +// members independently. Spelling any routing as a targets list means the +// members are distinct targets with no expectation of matching; every other +// shape, including every shape that predates the targets spelling, stays +// mirrored so a difference between members keeps blocking the review. +func TestServerConfig_MemberPlanningFor(t *testing.T) { + cfg := ServerConfig{ + Databases: map[string]DatabaseConfig{ + "payments": { + Type: "mysql", + Environments: map[string]EnvironmentConfig{ + "scalar": {Deployment: "payments-a", Target: "payments-001"}, + "local": {DSN: "root@tcp(localhost)/payments"}, + "mirrored": {Deployments: map[string]DeploymentTarget{"payments-a": {Target: "payments"}, "payments-b": {Target: "payments"}}}, + "targets": {Deployment: "payments-a", Targets: []string{"payments-001", "payments-002"}}, + "map-targets": {Deployments: map[string]DeploymentTarget{"payments-a": {Targets: []string{"payments-001", "payments-002"}}}}, + "mixed-shapes": {Deployments: map[string]DeploymentTarget{"payments-a": {Target: "payments-001"}, "payments-b": {Targets: []string{"payments-002"}}}}, + }, + }, + }, + } + + cases := []struct { + environment string + want MemberPlanning + }{ + {"scalar", PlanMirrored}, + {"local", PlanMirrored}, + {"mirrored", PlanMirrored}, + {"targets", PlanIndependent}, + {"map-targets", PlanIndependent}, + {"mixed-shapes", PlanIndependent}, + } + for _, tc := range cases { + t.Run(tc.environment, func(t *testing.T) { + got, err := cfg.MemberPlanningFor("payments", tc.environment) + require.NoError(t, err) + assert.Equal(t, tc.want, got) + }) + } + + t.Run("unknown database errors", func(t *testing.T) { + _, err := cfg.MemberPlanningFor("missing", "scalar") + require.Error(t, err) + assert.Contains(t, err.Error(), "not configured") + }) + + t.Run("unknown environment errors", func(t *testing.T) { + _, err := cfg.MemberPlanningFor("payments", "missing") + require.Error(t, err) + assert.Contains(t, err.Error(), "missing") + }) +} diff --git a/pkg/api/plan_review_drift.go b/pkg/api/plan_review_drift.go index 2bfc7f519..b3a67fcb1 100644 --- a/pkg/api/plan_review_drift.go +++ b/pkg/api/plan_review_drift.go @@ -34,12 +34,17 @@ func (s *Service) RollupReviewTimeDrift(ctx context.Context, req PlanRequest, pr return PlanRollup{}, fmt.Errorf("resolve deployment targets for %s/%s: %w", req.Database, req.Environment, err) } + planning, err := s.config.MemberPlanningFor(req.Database, req.Environment) + if err != nil { + return PlanRollup{}, fmt.Errorf("resolve member planning for %s/%s: %w", req.Database, req.Environment, err) + } + diffs, err := s.PlanDeploymentDiffs(ctx, req, primaryPlan, primaryMember, targets) if err != nil { return PlanRollup{}, fmt.Errorf("plan deployment diffs for %s/%s: %w", req.Database, req.Environment, err) } - rollup, err := RollupDeploymentDiffs(diffs, targets) + rollup, err := RollupDeploymentDiffs(diffs, targets, planning) if err != nil { return PlanRollup{}, fmt.Errorf("roll up deployment diffs for %s/%s: %w", req.Database, req.Environment, err) } @@ -77,6 +82,7 @@ func (s *Service) RollupReviewTimeDrift(ctx context.Context, req PlanRequest, pr "environment", req.Environment, "deployment", entry.Deployment, "target", entry.Target, + "member_planning", planning.String(), "error", entry.Err) } } diff --git a/pkg/api/plan_rollup.go b/pkg/api/plan_rollup.go index 0b12c822f..6d9f9d380 100644 --- a/pkg/api/plan_rollup.go +++ b/pkg/api/plan_rollup.go @@ -22,6 +22,11 @@ const ( // DeploymentErrored means the deployment's diff could not be computed or // compared. It must be treated as blocking, never as agreement. DeploymentErrored + // DeploymentPlanned means the member was planned against its own live + // schema and was never compared to the reviewed plan, because its + // environment's members are not expected to hold the same schema. Its + // changes are its own and do not block. + DeploymentPlanned ) func (c DeploymentClassification) String() string { @@ -32,11 +37,41 @@ func (c DeploymentClassification) String() string { return "diverged" case DeploymentErrored: return "errored" + case DeploymentPlanned: + return "planned" default: return fmt.Sprintf("unknown(%d)", int(c)) } } +// MemberPlanning is how the rollout members of one database/environment relate +// to each other, which decides whether a difference between them is drift. +type MemberPlanning int + +const ( + // PlanMirrored means the members are expected to hold the same schema, so + // each is compared to the reviewed plan and any difference is drift that + // blocks the review. This is the default: an environment opts out of it, and + // never into it, so a config that does not say otherwise keeps blocking. + PlanMirrored MemberPlanning = iota + // PlanIndependent means each member is planned against its own live schema, + // so a difference between members is ordinary rather than drift. Members are + // still individually required to be plannable — an error on any of them + // blocks the review. + PlanIndependent +) + +func (p MemberPlanning) String() string { + switch p { + case PlanMirrored: + return "mirrored" + case PlanIndependent: + return "independent" + default: + return fmt.Sprintf("unknown(%d)", int(p)) + } +} + // DeploymentRollupEntry is one deployment's place in the review-time rollup: how // it classified against the reviewed plan, the diff when it diverged, and the // error when it could not be computed or compared. @@ -72,12 +107,18 @@ type PlanRollup struct { // one deployment can address several targets and they would be // indistinguishable. // -// The primary (index 0) is the reviewed baseline and classifies Match against -// itself; every other member is compared to it with tern.CompareChangeSets. -// The result fails closed: a contract mismatch, a primary baseline that errored -// or is otherwise unusable, or any member that errored or diverged makes the -// rollup not Clean. -func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing.ExecutionTarget) (PlanRollup, error) { +// planning decides what a difference between members means. Under +// PlanMirrored the primary (index 0) is the reviewed baseline and classifies +// Match against itself, every other member is compared to it with +// tern.CompareChangeSets, and a difference is drift that blocks. Under +// PlanIndependent no member is compared to another: each was planned against +// its own live schema, so every member that produced a usable diff classifies +// Planned. +// +// The result fails closed under either planning: a contract mismatch, or any +// member that errored, makes the rollup not Clean. Under PlanMirrored a +// diverged member, or a primary baseline that is unusable, also blocks. +func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing.ExecutionTarget, planning MemberPlanning) (PlanRollup, error) { if len(expectedMembers) == 0 { return PlanRollup{}, fmt.Errorf("no expected rollout members to roll up") } @@ -91,6 +132,10 @@ func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing } } + if planning == PlanIndependent { + return rollupIndependentMembers(diffs), nil + } + baseline := tern.ChangeSet{Changes: diffs[0].Changes, Shards: diffs[0].Shards} // The primary's database type selects the grammar every comparison in this @@ -177,3 +222,39 @@ func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing return PlanRollup{Entries: entries, Clean: clean}, nil } + +// rollupIndependentMembers classifies members that were each planned against +// their own live schema. No member is compared to another, so a difference +// between them is never drift. What still blocks is a member that could not be +// planned at all: a producer error, or change content that will not parse under +// the member's own grammar. Content is checked by comparing a member's change +// set to itself, which is provably empty when the content is well-formed, so the +// check surfaces malformed content without ever false-diverging a real plan. +func rollupIndependentMembers(diffs []DeploymentPlanDiff) PlanRollup { + entries := make([]DeploymentRollupEntry, len(diffs)) + clean := true + for i, d := range diffs { + entry := DeploymentRollupEntry{ + DatabaseType: d.DatabaseType, + Deployment: d.Deployment, + Target: d.Target, + } + switch { + case d.Err != nil: + entry.Class = DeploymentErrored + entry.Err = d.Err + clean = false + default: + own := tern.ChangeSet{Changes: d.Changes, Shards: d.Shards} + if _, err := tern.CompareChangeSets(schema.DialectForDatabaseType(d.DatabaseType), own, own); err != nil { + entry.Class = DeploymentErrored + entry.Err = fmt.Errorf("member plan is not usable: %w", err) + clean = false + } else { + entry.Class = DeploymentPlanned + } + } + entries[i] = entry + } + return PlanRollup{Entries: entries, Clean: clean} +} diff --git a/pkg/api/plan_rollup_test.go b/pkg/api/plan_rollup_test.go index 19ebc9dd9..f0a17db0d 100644 --- a/pkg/api/plan_rollup_test.go +++ b/pkg/api/plan_rollup_test.go @@ -67,7 +67,7 @@ func TestRollupDeploymentDiffs_AllMatchIsClean(t *testing.T) { rollupDeployment("au", rollupAlterUsers(change)), rollupDeployment("us", rollupAlterUsers(change)), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 3) @@ -83,7 +83,7 @@ func TestRollupDeploymentDiffs_DivergenceBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), rollupDeployment("au", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `phone` varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentMatch, rollup.Entries[0].Class) @@ -101,7 +101,7 @@ func TestRollupDeploymentDiffs_ProducerErrorBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers(change)), errored, } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentMatch, rollup.Entries[0].Class) @@ -116,7 +116,7 @@ func TestRollupDeploymentDiffs_ComparisonErrorBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), rollupDeployment("au", rollupAlterUsers("not valid sql")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) @@ -132,7 +132,7 @@ func TestRollupDeploymentDiffs_UnusablePrimaryBlocksAll(t *testing.T) { primary, rollupDeployment("au", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentErrored, rollup.Entries[0].Class) @@ -143,7 +143,7 @@ func TestRollupDeploymentDiffs_UnusablePrimaryBlocksAll(t *testing.T) { // An empty result set is a fail-closed error: there is nothing to prove the // deployments agree. func TestRollupDeploymentDiffs_EmptyErrors(t *testing.T) { - _, err := RollupDeploymentDiffs(nil, nil) + _, err := RollupDeploymentDiffs(nil, nil, PlanMirrored) require.Error(t, err) } @@ -152,7 +152,7 @@ func TestRollupDeploymentDiffs_SingleDeploymentClean(t *testing.T) { diffs := []DeploymentPlanDiff{ rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 1) @@ -166,7 +166,7 @@ func TestRollupDeploymentDiffs_MalformedSingleDeploymentBaselineBlocks(t *testin diffs := []DeploymentPlanDiff{ rollupDeployment("eu", rollupAlterUsers("not valid sql")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) require.Len(t, rollup.Entries, 1) @@ -200,7 +200,7 @@ func TestRollupDeploymentDiffs_PostgresDialectClean(t *testing.T) { rollupPostgresDeployment("eu", change()), rollupPostgresDeployment("us", change()), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 2) @@ -216,7 +216,7 @@ func TestRollupDeploymentDiffs_UnregisteredPrimaryDialectBlocks(t *testing.T) { primary := rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")) primary.DatabaseType = "oracle" diffs := []DeploymentPlanDiff{primary} - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) require.Len(t, rollup.Entries, 1) @@ -236,7 +236,7 @@ func TestRollupDeploymentDiffs_MySQLFamilyTypesShareDialect(t *testing.T) { rollupDeployment("eu", rollupAlterUsers(change)), mysqlDeployment, } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 2) @@ -253,7 +253,7 @@ func TestRollupDeploymentDiffs_MixedDialectBlocks(t *testing.T) { rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), rollupPostgresDeployment("us", rollupAlterUsers("ALTER TABLE users ADD COLUMN email varchar(255)")), } - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.False(t, rollup.Clean) assert.Equal(t, DeploymentMatch, rollup.Entries[0].Class) @@ -273,15 +273,15 @@ func TestRollupDeploymentDiffs_ContractMismatchErrors(t *testing.T) { } t.Run("wrong primary", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"au", "au"}, [2]string{"eu", "eu"})) + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"au", "au"}, [2]string{"eu", "eu"}), PlanMirrored) require.Error(t, err) }) t.Run("missing member", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"eu", "eu"}, [2]string{"au", "au"}, [2]string{"us", "us"})) + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"eu", "eu"}, [2]string{"au", "au"}, [2]string{"us", "us"}), PlanMirrored) require.Error(t, err) }) t.Run("extra diff", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"eu", "eu"})) + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"eu", "eu"}), PlanMirrored) require.Error(t, err) }) } @@ -298,7 +298,7 @@ func TestRollupDeploymentDiffs_SameDeploymentDifferentTargets(t *testing.T) { } t.Run("matching members roll up clean", func(t *testing.T) { - rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs)) + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) require.NoError(t, err) assert.True(t, rollup.Clean) require.Len(t, rollup.Entries, 2) @@ -307,8 +307,85 @@ func TestRollupDeploymentDiffs_SameDeploymentDifferentTargets(t *testing.T) { }) t.Run("swapped targets are a contract mismatch", func(t *testing.T) { - _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"cake", "orders-002"}, [2]string{"cake", "orders-001"})) + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"cake", "orders-002"}, [2]string{"cake", "orders-001"}), PlanMirrored) require.Error(t, err) assert.ErrorContains(t, err, "cake/orders-001") }) } + +// An environment whose members are distinct targets plans each one against its +// own live schema, so members that would run different changes are ordinary +// rather than drift. The rollup stays clean and every member classifies as +// planned — the same change sets under mirrored planning would block. +func TestRollupDeploymentDiffs_IndependentMembersDoNotBlockOnDifference(t *testing.T) { + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001", rollupAlterUsers("ALTER TABLE users ADD COLUMN email VARCHAR(255)")), + rollupMember("cake", "orders-002", rollupAlterUsers("ALTER TABLE users ADD COLUMN phone VARCHAR(32)")), + rollupMember("cake", "orders-003"), + } + members := rollupMembers(diffs) + + independent, err := RollupDeploymentDiffs(diffs, members, PlanIndependent) + require.NoError(t, err) + assert.True(t, independent.Clean, "targets planned on their own do not drift against each other") + require.Len(t, independent.Entries, 3) + for i, entry := range independent.Entries { + assert.Equal(t, DeploymentPlanned, entry.Class, "entry %d", i) + assert.NoError(t, entry.Err, "entry %d", i) + } + + mirrored, err := RollupDeploymentDiffs(diffs, members, PlanMirrored) + require.NoError(t, err) + assert.False(t, mirrored.Clean, "the same change sets are drift when members are expected to match") + assert.Equal(t, DeploymentDiverged, mirrored.Entries[1].Class) +} + +// Independent planning removes the comparison between members, not the +// requirement that each member be plannable: a member the producer could not +// diff still blocks the review closed. +func TestRollupDeploymentDiffs_IndependentMemberErrorBlocks(t *testing.T) { + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001", rollupAlterUsers("ALTER TABLE users ADD COLUMN email VARCHAR(255)")), + rollupMember("cake", "orders-002"), + } + diffs[1].Err = fmt.Errorf("target unreachable") + + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanIndependent) + require.NoError(t, err) + assert.False(t, rollup.Clean) + assert.Equal(t, DeploymentPlanned, rollup.Entries[0].Class) + assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) + require.Error(t, rollup.Entries[1].Err) + assert.Contains(t, rollup.Entries[1].Err.Error(), "target unreachable") +} + +// A member whose change content will not parse under its own grammar has no +// usable plan, so it blocks even though nothing is compared against it. +func TestRollupDeploymentDiffs_IndependentUnparseableMemberBlocks(t *testing.T) { + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001", rollupAlterUsers("ALTER TABLE users ADD COLUMN email VARCHAR(255)")), + rollupMember("cake", "orders-002", rollupAlterUsers("this is not valid DDL at all")), + } + + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanIndependent) + require.NoError(t, err) + assert.False(t, rollup.Clean) + assert.Equal(t, DeploymentPlanned, rollup.Entries[0].Class) + assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) + require.Error(t, rollup.Entries[1].Err) + assert.Contains(t, rollup.Entries[1].Err.Error(), "not usable") +} + +// The member contract is enforced whatever the planning: independent planning +// stops members being compared to each other, it does not stop a missing or +// misidentified member from failing the rollup closed. +func TestRollupDeploymentDiffs_IndependentEnforcesMemberContract(t *testing.T) { + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001"), + rollupMember("cake", "orders-002"), + } + + _, err := RollupDeploymentDiffs(diffs, rollupMemberList([2]string{"cake", "orders-002"}, [2]string{"cake", "orders-001"}), PlanIndependent) + require.Error(t, err) + assert.Contains(t, err.Error(), "cake/orders-002") +} From 8256fdea6039d389cfc2c9d2db67035719332e2e Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 15:35:36 -0400 Subject: [PATCH 12/34] feat(api): keep plan storage tolerant of an already-stored identifier MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A plan whose identifier is already stored is that same plan re-stored, and the store's job is done. Storing a member plan needs no row ID today — the member's rollup entry carries the plan identifier — so plan storage stays a write that either succeeds or reports why, with no read-back that could fail a plan the store already holds. Co-Authored-By: Claude Fable 5 --- pkg/api/plan_handlers.go | 36 +++++++++------------------- pkg/api/plan_member_plans.go | 2 +- pkg/api/plan_member_plans_test.go | 40 ------------------------------- 3 files changed, 12 insertions(+), 66 deletions(-) diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index e9483599c..ac92907f3 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -887,8 +887,7 @@ type storedPlanRoute struct { } func (s *Service) storePlanResponse(ctx context.Context, req PlanRequest, resp *ternv1.PlanResponse, route storedPlanRoute) error { - _, err := s.storePlan(ctx, req, resp.PlanId, resp.Changes, resp.Shards, route) - return err + return s.storePlan(ctx, req, resp.PlanId, resp.Changes, resp.Shards, route) } // storePlan writes one plan row for a single rollout member: the changes and @@ -899,15 +898,13 @@ func (s *Service) storePlanResponse(ctx context.Context, req PlanRequest, resp * // // planIdentifier is the plan's external identifier — minted by the planner for // the primary, minted here for a member whose plan came from the non-persisting -// diff RPC. The stored row's ID is returned so a caller can point an apply -// operation at exactly this plan. +// diff RPC. // -// An identifier that already exists is not an error: a re-plan of unchanged -// content re-stores the same plan, and the existing row is the same plan. The -// existing row is reloaded so its ID is still returned. -func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier string, changes []*ternv1.SchemaChange, shards []*ternv1.ShardPlan, route storedPlanRoute) (int64, error) { +// An identifier that is already stored is not an error: a re-plan of unchanged +// content re-stores the same plan, and the row already there is that plan. +func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier string, changes []*ternv1.SchemaChange, shards []*ternv1.ShardPlan, route storedPlanRoute) error { if planIdentifier == "" { - return 0, fmt.Errorf("store plan for database %s deployment %q target %q: plan has no identifier", req.Database, route.Deployment, route.Target) + return fmt.Errorf("store plan for database %s deployment %q target %q: plan has no identifier", req.Database, route.Deployment, route.Target) } prInt := 0 if req.PullRequest != nil { @@ -923,11 +920,11 @@ func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier } namespaces, err := protoChangesToNamespaces(changes, req.SchemaFiles) if err != nil { - return 0, fmt.Errorf("convert plan namespaces: %w", err) + return fmt.Errorf("convert plan namespaces: %w", err) } storedShards, err := protoShardPlansToStorage(shards) if err != nil { - return 0, fmt.Errorf("convert plan shards: %w", err) + return fmt.Errorf("convert plan shards: %w", err) } storedPlan := &storage.Plan{ PlanIdentifier: planIdentifier, @@ -945,21 +942,10 @@ func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier HeadSHA: headSHA, CreatedAt: time.Now(), } - id, err := s.storage.Plans().Create(ctx, storedPlan) - if err == nil { - return id, nil + if _, err := s.storage.Plans().Create(ctx, storedPlan); err != nil && !errors.Is(err, storage.ErrPlanIDExists) { + return fmt.Errorf("store plan %s: %w", planIdentifier, err) } - if !errors.Is(err, storage.ErrPlanIDExists) { - return 0, fmt.Errorf("store plan %s: %w", planIdentifier, err) - } - existing, getErr := s.storage.Plans().Get(ctx, planIdentifier) - if getErr != nil { - return 0, fmt.Errorf("reload already-stored plan %s: %w", planIdentifier, getErr) - } - if existing == nil { - return 0, fmt.Errorf("plan %s was reported as already stored but could not be read back", planIdentifier) - } - return existing.ID, nil + return nil } // handleApply handles POST /api/apply requests. diff --git a/pkg/api/plan_member_plans.go b/pkg/api/plan_member_plans.go index 3a6cf873e..150bf5d46 100644 --- a/pkg/api/plan_member_plans.go +++ b/pkg/api/plan_member_plans.go @@ -54,7 +54,7 @@ func (s *Service) persistMemberPlans(ctx context.Context, req PlanRequest, plann Deployment: entry.Deployment, Target: entry.Target, } - if _, err := s.storePlan(ctx, req, planIdentifier, diffs[i].Changes, diffs[i].Shards, route); err != nil { + if err := s.storePlan(ctx, req, planIdentifier, diffs[i].Changes, diffs[i].Shards, route); err != nil { s.logger.Error("failed to store a rollout member's plan; the member will block the review because an apply would have no plan to run for it", "repository", req.Repository, "database", req.Database, diff --git a/pkg/api/plan_member_plans_test.go b/pkg/api/plan_member_plans_test.go index f3e44d49e..497853c2f 100644 --- a/pkg/api/plan_member_plans_test.go +++ b/pkg/api/plan_member_plans_test.go @@ -174,43 +174,3 @@ func TestRollupReviewTimeDrift_MirroredMembersStoreNoMemberPlan(t *testing.T) { assert.Empty(t, rollup.Entries[1].PlanIdentifier) assert.Empty(t, plans.created, "mirrored members run the reviewed plan") } - -// A plan identifier that is already stored is the same plan re-stored by a -// re-plan of unchanged content, so the existing row is reused rather than -// treated as a failure. -func TestStorePlan_ExistingIdentifierReusesStoredRow(t *testing.T) { - plans := &existingPlanStore{existing: &storage.Plan{ID: 77, PlanIdentifier: "plan-existing"}} - svc := multiTargetService(t, &mockTernClient{}, plans) - - id, err := svc.storePlan(t.Context(), planDiffReq(t), "plan-existing", nil, nil, - storedPlanRoute{DatabaseType: storage.DatabaseTypeMySQL, Deployment: "eu", Target: "testapp-002"}) - require.NoError(t, err) - assert.Equal(t, int64(77), id) -} - -// A plan reported as already stored but not readable back leaves the caller with -// no plan row to point at, which must be an error rather than a zero ID. -func TestStorePlan_ExistingIdentifierNotReadableBackErrors(t *testing.T) { - plans := &existingPlanStore{} - svc := multiTargetService(t, &mockTernClient{}, plans) - - _, err := svc.storePlan(t.Context(), planDiffReq(t), "plan-vanished", nil, nil, - storedPlanRoute{DatabaseType: storage.DatabaseTypeMySQL, Deployment: "eu", Target: "testapp-002"}) - require.Error(t, err) - assert.Contains(t, err.Error(), "could not be read back") -} - -// existingPlanStore rejects every create as an already-used identifier and -// returns whatever row a read-back finds. -type existingPlanStore struct { - mockPlanLookupStore - existing *storage.Plan -} - -func (s *existingPlanStore) Create(context.Context, *storage.Plan) (int64, error) { - return 0, storage.ErrPlanIDExists -} - -func (s *existingPlanStore) Get(context.Context, string) (*storage.Plan, error) { - return s.existing, nil -} From 20de329b2ea36df22f2a8f4ec0cd240ad056ddde Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 16:58:11 -0400 Subject: [PATCH 13/34] fix(api): create a single-member apply without database config The trusted control-plane enqueue path holds no `Databases` config, so resolving the member planning contract there fails and apply creation with it. A single member is the plan's own primary: it runs the apply's plan under either contract, and no sibling's plan could be substituted for it, so the contract lookup is unnecessary work at that point. Co-Authored-By: Claude Fable 5 --- pkg/api/apply_members.go | 10 ++++++++++ pkg/api/apply_members_test.go | 20 ++++++++++++++++++++ 2 files changed, 30 insertions(+) diff --git a/pkg/api/apply_members.go b/pkg/api/apply_members.go index 5264cf234..140fe4a7d 100644 --- a/pkg/api/apply_members.go +++ b/pkg/api/apply_members.go @@ -45,6 +45,16 @@ func (m applyMember) MemberID() string { // safe fallback: the apply's plan describes a different target's schema, so // substituting it would run DDL that was never planned for this member. func (s *Service) resolveApplyMembers(ctx context.Context, plan *storage.Plan, environment string, targets []routing.ExecutionTarget) ([]applyMember, error) { + // A single member is the plan's own primary, so it runs the apply's plan + // under either contract and there is no sibling whose plan could be + // substituted for it. Deciding the contract first would make apply creation + // depend on database config that the trusted control-plane enqueue path is + // not required to have — the same reason the caller falls back to the plan's + // stored target when config does not resolve the environment. + if len(targets) == 1 { + return []applyMember{{Target: targets[0], Plan: plan}}, nil + } + planning, err := s.config.MemberPlanningFor(plan.Database, environment) if err != nil { // A database/environment the config no longer resolves cannot be shown to diff --git a/pkg/api/apply_members_test.go b/pkg/api/apply_members_test.go index 7d5b66287..77b4b7f8b 100644 --- a/pkg/api/apply_members_test.go +++ b/pkg/api/apply_members_test.go @@ -92,6 +92,26 @@ func TestResolveApplyMembers_MirroredMembersShareTheApplyPlan(t *testing.T) { } } +// The trusted control-plane enqueue path creates applies on servers that hold +// no database config, and apply creation falls back to the plan's own stored +// target there. That single member is the plan's own primary, so it runs the +// apply's plan without consulting config for a contract that could not change +// the outcome. +func TestResolveApplyMembers_SingleMemberNeedsNoDatabaseConfig(t *testing.T) { + plans := &listingPlanStore{listErr: errors.New("List must not be called for a single member")} + logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelError})) + svc := New(&mockStorageWithPlanLookup{plans: plans}, &ServerConfig{}, map[string]tern.Client{}, logger) + plan := primaryPlanRow("testapp-001") + + members, err := svc.resolveApplyMembers(t.Context(), plan, "production", []routing.ExecutionTarget{ + {DatabaseType: plan.DatabaseType, Deployment: plan.Deployment, Target: plan.Target}, + }) + require.NoError(t, err) + require.Len(t, members, 1) + assert.Equal(t, "eu/testapp-001", members[0].MemberID()) + assert.Same(t, plan, members[0].Plan) +} + // Each target of a multi-target environment runs the plan stored for that // target in the same review round, matched on the head SHA the apply's plan was // created for. From 99970fdc6ab91ad8ca82c85049bcdbad5a16fd5f Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 10 Sep 2026 18:48:44 -0400 Subject: [PATCH 14/34] docs(api): describe the pull payload's target member list The schema intelligence guide documents the pull endpoint but not what a pull of a multi-target environment returns. Describe the `targets` array, what the primary entry means, the three `difference` values, and that an empty `diverged_tables` is a positive statement that two targets agree rather than a comparison that was skipped. Co-Authored-By: Claude Fable 5 --- docs/schema-intelligence.md | 74 +++++++++++++++++++++++++++++++++++++ 1 file changed, 74 insertions(+) diff --git a/docs/schema-intelligence.md b/docs/schema-intelligence.md index 3399bcfd5..e4a88f405 100644 --- a/docs/schema-intelligence.md +++ b/docs/schema-intelligence.md @@ -310,6 +310,80 @@ A clean audit returns `lint: []`; an omitted field means lint was not requested. Pull runs schema-shape rules only. Rules about proposed changes, such as unsafe drops, require a plan. See [lint and safety levels](lint-and-safety-levels.md#auditing-a-live-schema-pull---lint). +### Databases that span several targets + +A database whose environment lists `targets` addresses several targets at once, +and each holds its own schema. There is no single live schema to return, so a +pull returns the primary target's — the one a caller materializes — plus a +`targets` array naming every target the environment addresses and how each of +the others differs from the primary. + +```sh +schemabot pull -d shop -e production +``` + +```sql +-- Target `shop-001` — primary target, whose schema is below +-- Target `shop-002` — same schema as the primary target +-- Target `shop-003` — 2 tables differ from the primary target +-- shop.audit_log: differs +-- shop.order_events: missing +``` + +
+API equivalent + +```http +POST /api/pull +Content-Type: application/json + +{"database": "shop", "environment": "production"} +``` + +```json +{ + "database": "shop", + "type": "mysql", + "environment": "production", + "table_count": 4, + "namespaces": { + "shop": {"tables": {"…": "CREATE TABLE …"}} + }, + "targets": [ + {"deployment": "commerce-a", "target": "shop-001", "table_count": 4, "primary": true}, + {"deployment": "commerce-a", "target": "shop-002", "table_count": 4}, + {"deployment": "commerce-a", "target": "shop-003", "table_count": 3, "diverged_tables": [ + {"namespace": "shop", "table": "audit_log", "difference": "differs"}, + {"namespace": "shop", "table": "order_events", "difference": "only_on_primary"} + ]} + ] +} +``` + +
+ +Read the array as the environment's whole member set. Exactly one entry carries +`"primary": true`, and it is the target whose schema is in `namespaces`; it +never carries `diverged_tables`, because it is the baseline the others are +compared against. A reconciling caller can take the member set straight from +this array rather than deriving it from target names. + +`difference` is one of: + +| Value | Meaning | +|---|---| +| `differs` | both targets hold the table, with different DDL | +| `only_on_primary` | only the primary target holds the table | +| `only_on_target` | only this target holds the table | + +An empty `diverged_tables` means the two targets genuinely agree. It never means +the comparison was skipped: a target that cannot be pulled, or whose DDL cannot +be parsed, fails the whole pull rather than being reported as converged. Tables +are compared by their canonical parsed form, so formatting differences are not +divergence. + +An environment that does not list `targets` carries no `targets` array at all. + ### Engine support The envelope differs by dialect: From a4023dc550bb2184e6bb385c92b250ee582571af Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 16:34:54 -0400 Subject: [PATCH 15/34] feat(github): name every rollout member unambiguously MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A rollout member is identified by its deployment and target together, but the deployment alone is what the plan comment, check summary, apply comment and CLI progress have been showing. Once one deployment can address several targets, two members render under the same name, and a label that points at another member ("waiting for X", "halted by X") no longer identifies one. routing.DisplayNames is the single naming rule every surface now uses: a member is named by its deployment alone unless that deployment addresses more than one distinct target in the rollout, in which case every member of that deployment is named deployment/target. Keying on distinct targets is what keeps a keyed or sharded apply — several operations of one deployment against the same target — named by the deployment, where the extra half would be noise that still did not tell the operations apart. presentation.Derive resolves each member's name once and hands it to every consumer, including the labels that reference a sibling and the suggested next action. The apply comment's per-member detail bodies move from a name-keyed map to a slice paired positionally with the model, since a map collapses two members of one deployment onto one body. The independent-planning contract also reaches the wording: an errored member reads "could not plan" where targets hold their own schemas and "could not verify" where they are expected to mirror each other. --- pkg/api/plan_rollup.go | 11 +- pkg/cmd/commands/watch_tui_view_multi.go | 64 +++++----- pkg/cmd/internal/templates/progress_multi.go | 17 ++- .../internal/templates/progress_multi_test.go | 28 ++++ pkg/presentation/presentation.go | 98 ++++++++++---- pkg/presentation/presentation_test.go | 75 ++++++++++- pkg/routing/resolver.go | 36 ++++++ pkg/routing/resolver_test.go | 75 +++++++++++ pkg/webhook/multi_apply.go | 10 +- pkg/webhook/multi_apply_test.go | 20 +-- pkg/webhook/plan_drift.go | 42 +++++- pkg/webhook/plan_drift_test.go | 19 +++ pkg/webhook/templates/multi_apply.go | 44 +++++-- .../templates/multi_apply_tenant_test.go | 4 +- pkg/webhook/templates/multi_apply_test.go | 120 ++++++++++++++---- pkg/webhook/templates/plan.go | 83 ++++++++---- pkg/webhook/templates/plan_drift_test.go | 61 +++++++++ pkg/webhook/templates/preview.go | 38 +++--- 18 files changed, 674 insertions(+), 171 deletions(-) create mode 100644 pkg/routing/resolver_test.go diff --git a/pkg/api/plan_rollup.go b/pkg/api/plan_rollup.go index 12d803dbc..01e94cd3c 100644 --- a/pkg/api/plan_rollup.go +++ b/pkg/api/plan_rollup.go @@ -99,6 +99,13 @@ type DeploymentRollupEntry struct { type PlanRollup struct { Entries []DeploymentRollupEntry Clean bool + // Planning is the contract the members were classified under, and decides + // what Clean means. Under PlanMirrored a clean rollup says every member + // would run the same plan. Under PlanIndependent it says every member + // produced a plan of its own, which are not expected to match — so a reader + // of the rollup cannot describe it without knowing which contract produced + // it. + Planning MemberPlanning } // RollupDeploymentDiffs classifies each rollout member's review-time diff @@ -227,7 +234,7 @@ func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing entries[i] = entry } - return PlanRollup{Entries: entries, Clean: clean}, nil + return PlanRollup{Entries: entries, Clean: clean, Planning: planning}, nil } // rollupIndependentMembers classifies members that were each planned against @@ -263,5 +270,5 @@ func rollupIndependentMembers(diffs []DeploymentPlanDiff) PlanRollup { } entries[i] = entry } - return PlanRollup{Entries: entries, Clean: clean} + return PlanRollup{Entries: entries, Clean: clean, Planning: PlanIndependent} } diff --git a/pkg/cmd/commands/watch_tui_view_multi.go b/pkg/cmd/commands/watch_tui_view_multi.go index bffa71c82..d629e6d65 100644 --- a/pkg/cmd/commands/watch_tui_view_multi.go +++ b/pkg/cmd/commands/watch_tui_view_multi.go @@ -19,8 +19,12 @@ func (m WatchModel) multiDeploymentProgressView() string { var b strings.Builder m.writeMultiDeploymentHeader(&b, model) - for _, deployment := range model.Deployments { - m.writeDeploymentSection(&b, deployment) + // Derive returns one Deployment per input operation, in input order, so + // model.Deployments[i] projects m.operations[i]. Pairing by index lets each + // section render its own operation's identifiers; a deployment can own + // several operations, so a name-based lookup cannot tell them apart. + for i, deployment := range model.Deployments { + m.writeDeploymentSection(&b, deployment, m.operations[i]) } b.WriteString(templates.FormatThrottleReference(m.tables)) @@ -37,6 +41,7 @@ func tuiOperationsForPresentation(ops []templates.ProgressOperation, released bo for _, op := range ops { presentationOps = append(presentationOps, presentation.Operation{ Deployment: op.Deployment, + Target: op.Target, State: op.State, Barrier: op.CutoverPolicy == storage.CutoverPolicyBarrier, Parallel: op.CutoverPolicy == storage.CutoverPolicyParallel, @@ -61,9 +66,9 @@ func (m WatchModel) writeMultiDeploymentHeader(b *strings.Builder, model present if model.FirstFailure != nil { errStyle := lipgloss.NewStyle().Foreground(lipgloss.Color("9")) if model.FirstFailure.Error != "" { - fmt.Fprintf(b, "%s\n", errStyle.Render(fmt.Sprintf(glyph.Failed+" First failure: %s — %s", model.FirstFailure.Deployment, model.FirstFailure.Error))) + fmt.Fprintf(b, "%s\n", errStyle.Render(fmt.Sprintf(glyph.Failed+" First failure: %s — %s", model.FirstFailure.Name, model.FirstFailure.Error))) } else { - fmt.Fprintf(b, "%s\n", errStyle.Render(fmt.Sprintf(glyph.Failed+" First failure: %s", model.FirstFailure.Deployment))) + fmt.Fprintf(b, "%s\n", errStyle.Render(fmt.Sprintf(glyph.Failed+" First failure: %s", model.FirstFailure.Name))) } } if m.applyID != "" { @@ -83,16 +88,20 @@ func formatTUIDeploymentCounts(counts []presentation.StateCount) string { return strings.Join(parts, " · ") } -func (m WatchModel) writeDeploymentSection(b *strings.Builder, deployment presentation.Deployment) { - fmt.Fprintf(b, "%s %s — %s", deployment.Emoji, deployment.Deployment, deployment.Label) - if target := targetForTUIDeployment(m.operations, deployment.Deployment); target != "" { - fmt.Fprintf(b, " (%s)", target) +func (m WatchModel) writeDeploymentSection(b *strings.Builder, deployment presentation.Deployment, op templates.ProgressOperation) { + fmt.Fprintf(b, "%s %s — %s", deployment.Emoji, deployment.Name, deployment.Label) + // A member whose name already carries its target does not repeat it in the + // trailing parenthetical. + if op.Target != "" && deployment.Name == deployment.Deployment { + fmt.Fprintf(b, " (%s)", op.Target) } b.WriteString("\n") - if externalOperationID := externalOperationIDForTUIDeployment(m.operations, deployment.Deployment); externalOperationID != "" { - fmt.Fprintf(b, " External operation ID: %s\n", externalOperationID) + // The external operation ID identifies this operation's own data-plane row, + // so it never falls back to a sibling's value. + if op.ExternalOperationID != "" { + fmt.Fprintf(b, " External operation ID: %s\n", op.ExternalOperationID) } - if externalID := externalIDForTUIDeployment(m.operations, deployment.Deployment); externalID != "" { + if externalID := externalIDForTUIMember(m.operations, op); externalID != "" { fmt.Fprintf(b, " External apply ID: %s\n", externalID) } @@ -109,28 +118,17 @@ func (m WatchModel) writeDeploymentSection(b *strings.Builder, deployment presen b.WriteString("\n") } -func targetForTUIDeployment(ops []templates.ProgressOperation, deployment string) string { - for _, op := range ops { - if op.Deployment == deployment { - return op.Target - } - } - return "" -} - -func externalOperationIDForTUIDeployment(ops []templates.ProgressOperation, deployment string) string { - for _, op := range ops { - if op.Deployment == deployment && op.ExternalOperationID != "" { - return op.ExternalOperationID - } - } - return "" -} - -func externalIDForTUIDeployment(ops []templates.ProgressOperation, deployment string) string { - for _, op := range ops { - if op.Deployment == deployment && op.ExternalID != "" { - return op.ExternalID +// externalIDForTUIMember resolves the external apply ID shown in a section: the +// section's own operation when set, falling back to a sibling's — a keyed +// apply's operations share one data-plane apply, so an operation that has not +// dispatched yet still shows it. +func externalIDForTUIMember(ops []templates.ProgressOperation, op templates.ProgressOperation) string { + if op.ExternalID != "" { + return op.ExternalID + } + for _, sibling := range ops { + if sibling.Deployment == op.Deployment && sibling.ExternalID != "" { + return sibling.ExternalID } } return "" diff --git a/pkg/cmd/internal/templates/progress_multi.go b/pkg/cmd/internal/templates/progress_multi.go index 590003c37..1a61e054f 100644 --- a/pkg/cmd/internal/templates/progress_multi.go +++ b/pkg/cmd/internal/templates/progress_multi.go @@ -41,6 +41,7 @@ func progressOperationsForPresentation(ops []ProgressOperation, released bool) [ for _, op := range ops { presentationOps = append(presentationOps, presentation.Operation{ Deployment: op.Deployment, + Target: op.Target, State: op.State, Barrier: op.CutoverPolicy == storage.CutoverPolicyBarrier, Parallel: op.CutoverPolicy == storage.CutoverPolicyParallel, @@ -90,35 +91,37 @@ func writeMultiDeploymentFirstFailure(failure *presentation.Deployment) { return } if failure.Error == "" { - fmt.Printf("\n %s"+glyph.Failed+" First failure: %s%s\n", ANSIRed, failure.Deployment, ANSIReset) + fmt.Printf("\n %s"+glyph.Failed+" First failure: %s%s\n", ANSIRed, failure.Name, ANSIReset) return } - fmt.Printf("\n %s"+glyph.Failed+" First failure: %s — %s%s\n", ANSIRed, failure.Deployment, failure.Error, ANSIReset) + fmt.Printf("\n %s"+glyph.Failed+" First failure: %s — %s%s\n", ANSIRed, failure.Name, failure.Error, ANSIReset) } func writeMultiDeploymentNextAction(next presentation.NextAction) { switch next.Kind { case presentation.NextActionCutover: - fmt.Printf("\n Next: cut over %s\n", next.Deployment) + fmt.Printf("\n Next: cut over %s\n", next.Name) case presentation.NextActionResume: fmt.Println("\n Next: resume apply") case presentation.NextActionReviewFailure: - if next.Deployment == "" { + if next.Name == "" { fmt.Println("\n Next: review failure") return } - fmt.Printf("\n Next: review failure in %s\n", next.Deployment) + fmt.Printf("\n Next: review failure in %s\n", next.Name) case presentation.NextActionNone: } } func writeDeploymentProgressSection(deployment presentation.Deployment, op ProgressOperation, data ProgressData) { - fmt.Printf("%s %s", deployment.Emoji, deployment.Deployment) + fmt.Printf("%s %s", deployment.Emoji, deployment.Name) if op.OperationKey != "" { fmt.Printf(" · %s", op.OperationKey) } fmt.Printf(" — %s", deployment.Label) - if target := sectionTarget(op, data.Operations); target != "" { + // A member whose name already carries its target does not repeat it in the + // trailing parenthetical. + if target := sectionTarget(op, data.Operations); target != "" && deployment.Name == deployment.Deployment { fmt.Printf(" (%s)", target) } fmt.Println() diff --git a/pkg/cmd/internal/templates/progress_multi_test.go b/pkg/cmd/internal/templates/progress_multi_test.go index 1538d1771..96f4b9e3c 100644 --- a/pkg/cmd/internal/templates/progress_multi_test.go +++ b/pkg/cmd/internal/templates/progress_multi_test.go @@ -163,3 +163,31 @@ func assertLess(t *testing.T, output, left, right string) { assert.NotEqual(t, -1, rightIndex, "expected output to contain %q", right) assert.Less(t, leftIndex, rightIndex, "expected %q before %q", left, right) } + +// One deployment can address several targets, each running its own copy of the +// change. Every member is named by its routing pair so no two sections carry the +// same heading, while a sibling deployment that addresses a single target keeps +// its plain name. +func TestWriteProgressMultiTargetSectionsNameEachMember(t *testing.T) { + output := captureStdout(t, func() { + WriteProgress(ProgressData{ + ApplyID: "apply-multi-target", + Environment: "staging", + State: state.Apply.Running, + Operations: []ProgressOperation{ + {Deployment: "primary", Target: "testapp-001", State: state.ApplyOperation.Completed, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + {Deployment: "primary", Target: "testapp-002", State: state.ApplyOperation.Running, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + {Deployment: "eu-west", Target: "orders-eu", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + }, + }) + }) + + assert.Contains(t, output, "✅ primary/testapp-001 — completed") + assert.Contains(t, output, "🔄 primary/testapp-002 — running table copy") + assert.Contains(t, output, "⏳ eu-west — waiting for primary/testapp-002 (orders-eu)") + + // A name that already carries the target does not repeat it in the + // trailing parenthetical. + assert.NotContains(t, output, "primary/testapp-001 — completed (testapp-001)") + assert.NotContains(t, output, "primary/testapp-002 — running table copy (testapp-002)") +} diff --git a/pkg/presentation/presentation.go b/pkg/presentation/presentation.go index f165396cf..b6cbf3f22 100644 --- a/pkg/presentation/presentation.go +++ b/pkg/presentation/presentation.go @@ -20,6 +20,7 @@ import ( "fmt" "github.com/block/schemabot/pkg/glyph" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/state" ) @@ -32,6 +33,13 @@ type Operation struct { // Deployment is the deployment name this operation targets. Deployment string + // Target is the address this operation runs against within its deployment. + // One deployment can address several targets, in which case the deployment + // name alone labels two different rollout members identically — the + // derivation resolves an unambiguous name for each member from the two + // together (see routing.DisplayNames). + Target string + // State is the canonical operation state (state.ApplyOperation, == state.Apply). State string @@ -121,6 +129,17 @@ type Deployment struct { // Deployment is the deployment name. Deployment string + // Target is the address this member runs against within its deployment. + Target string + + // Name is the operator-facing identity of this member: the deployment name + // on its own, or deployment/target when the deployment addresses several + // targets in this apply and the deployment name would not tell its members + // apart. Surfaces label a member with this rather than with Deployment, and + // labels that reference another member (waiting for, halted by) name it the + // same way. + Name string + // State is the raw operation state it was derived from. State string @@ -169,6 +188,11 @@ type NextAction struct { Kind NextActionKind // Deployment is the action's target, when the action is deployment-scoped. Deployment string + // Target is that deployment's address, when the action is member-scoped. + Target string + // Name is the member's operator-facing identity (see Deployment.Name), for + // surfaces that render the action as prose. + Name string } // StateCount is one entry of the aggregate's per-status histogram. @@ -235,9 +259,10 @@ func Derive(ops []Operation) Apply { } } + names := memberNames(ops) deployments := make([]Deployment, len(ops)) for i := range ops { - deployments[i] = deriveDeployment(ops, i) + deployments[i] = deriveDeployment(ops, names, i) } aggState := state.DeriveRolloutApplyState(children) @@ -267,11 +292,25 @@ func firstFailure(deps []Deployment) *Deployment { return nil } +// memberNames resolves the operator-facing name of every operation, so a label +// that names a member — its own or an earlier sibling it is waiting on — always +// identifies exactly one of them. The naming rule is shared with the plan +// comment and the check summary. +func memberNames(ops []Operation) []string { + members := make([]routing.ExecutionTarget, len(ops)) + for i, op := range ops { + members[i] = routing.ExecutionTarget{Deployment: op.Deployment, Target: op.Target} + } + return routing.DisplayNames(members) +} + // deriveDeployment projects operation i, using its earlier siblings for the -// ordering context that disambiguates pending and waiting_for_cutover. -func deriveDeployment(ops []Operation, i int) Deployment { +// ordering context that disambiguates pending and waiting_for_cutover. names is +// index-parallel to ops, so a label that references sibling j names it exactly +// as sibling j's own section is labelled. +func deriveDeployment(ops []Operation, names []string, i int) Deployment { op := ops[i] - d := Deployment{Deployment: op.Deployment, State: op.State, Error: op.Error} + d := Deployment{Deployment: op.Deployment, Target: op.Target, Name: names[i], State: op.State, Error: op.Error} switch op.State { case state.ApplyOperation.Completed: @@ -293,9 +332,9 @@ func deriveDeployment(ops []Operation, i int) Deployment { case state.ApplyOperation.Reverted: d.set(StateReverted, "reverted", "↩️", false) case state.ApplyOperation.Pending: - derivePending(&d, ops, i) + derivePending(&d, ops, names, i) case state.ApplyOperation.WaitingForCutover: - deriveWaitingForCutover(&d, ops, i) + deriveWaitingForCutover(&d, ops, names, i) default: // Unknown / engine-specific transient state: show it verbatim and keep it // open so an operator is never left without a status for a deployment. @@ -307,7 +346,7 @@ func deriveDeployment(ops []Operation, i int) Deployment { // derivePending splits a pending operation into next-in-order, waiting-on, or // halted, mirroring the pending sibling gate in FindNextApplyOperation so the // label agrees with what the operator will claim next. -func derivePending(d *Deployment, ops []Operation, i int) { +func derivePending(d *Deployment, ops []Operation, names []string, i int) { op := ops[i] // Under parallel the copy phase has no earlier-sibling gate, so a pending // operation is immediately claimable regardless of any earlier sibling — it @@ -317,17 +356,17 @@ func derivePending(d *Deployment, ops []Operation, i int) { d.set(StateQueuedNext, "queued — next in order", "⏳", false) return } - if h, paused := blockingSibling(ops, i); h != nil { + if h, paused := blockingSibling(ops, i); h >= 0 { if paused { - d.set(StatePaused, fmt.Sprintf("paused — %s failed; release or stop", h.Deployment), "⏸️", true) + d.set(StatePaused, fmt.Sprintf("paused — %s failed; release or stop", names[h]), "⏸️", true) return } - d.set(StateHalted, fmt.Sprintf("halted — %s %s", h.Deployment, haltedReason(h.State)), "⏸️", true) + d.set(StateHalted, fmt.Sprintf("halted — %s %s", names[h], haltedReason(ops[h].State)), "⏸️", true) return } for j := range i { if blocksPending(ops[j].State, op.Barrier, op.continuesPastFailure()) { - d.set(StateWaiting, fmt.Sprintf("waiting for %s", ops[j].Deployment), "⏳", false) + d.set(StateWaiting, fmt.Sprintf("waiting for %s", names[j]), "⏳", false) return } } @@ -337,24 +376,27 @@ func derivePending(d *Deployment, ops []Operation, i int) { // deriveWaitingForCutover splits a copied-and-parked operation into ready-now or // ready-but-waiting-on-an-earlier-cutover. Cutover ordering is strict-complete // and independent of cutover_policy, which only relaxes copy start. -func deriveWaitingForCutover(d *Deployment, ops []Operation, i int) { +func deriveWaitingForCutover(d *Deployment, ops []Operation, names []string, i int) { op := ops[i] for j := range i { if blocksCutover(ops[j].State, op.continuesPastFailure()) { - d.set(StateReadyForCutoverWaiting, fmt.Sprintf("ready for cutover — waiting for %s", ops[j].Deployment), "🟡", true) + d.set(StateReadyForCutoverWaiting, fmt.Sprintf("ready for cutover — waiting for %s", names[j]), "🟡", true) return } } d.set(StateReadyForCutoverNext, "ready for cutover — next in order", "🟢", true) } -// blockingSibling returns the earliest earlier sibling that holds the rollout -// for a pending operation, and whether the hold is a human-gated pause rather -// than a halt. A terminal-failed sibling holds unless the policy continues past -// it (continue, or a released pause); when it does hold under on_failure=pause -// the hold is a pause (paused=true). A cancelled/reverted sibling always halts, -// matching the predicate's lack of an exemption for them. -func blockingSibling(ops []Operation, i int) (blocker *Operation, paused bool) { +// blockingSibling returns the index of the earliest earlier sibling that holds +// the rollout for a pending operation, or -1 when none does, and whether the +// hold is a human-gated pause rather than a halt. A terminal-failed sibling +// holds unless the policy continues past it (continue, or a released pause); +// when it does hold under on_failure=pause the hold is a pause (paused=true). A +// cancelled/reverted sibling always halts, matching the predicate's lack of an +// exemption for them. The index, rather than the operation, is what the caller +// needs: the label names the blocker by its resolved member name, which is only +// addressable by position. +func blockingSibling(ops []Operation, i int) (blocker int, paused bool) { op := ops[i] for j := range i { switch ops[j].State { @@ -362,12 +404,12 @@ func blockingSibling(ops []Operation, i int) (blocker *Operation, paused bool) { if op.continuesPastFailure() { continue } - return &ops[j], op.pausesOnFailure() + return j, op.pausesOnFailure() case state.ApplyOperation.Cancelled, state.ApplyOperation.Reverted: - return &ops[j], false + return j, false } } - return nil, false + return -1, false } // blocksPending reports whether an earlier sibling in earlierState blocks a @@ -523,7 +565,7 @@ func nextAction(aggState string, deps []Deployment, failClosed bool) NextAction // than the aggregate is what says there is nothing left to wait for. if aggState == state.Apply.Failed || failClosed { if d, ok := firstWithState(deps, state.ApplyOperation.Failed); ok { - return NextAction{Kind: NextActionReviewFailure, Deployment: d.Deployment} + return memberAction(NextActionReviewFailure, d) } return NextAction{Kind: NextActionReviewFailure} } @@ -537,11 +579,17 @@ func nextAction(aggState string, deps []Deployment, failClosed bool) NextAction // up until every deployment has reached it). This covers both the // waiting_for_cutover aggregate and the running-with-one-ready case. if d, ok := firstWithPresentation(deps, StateReadyForCutoverNext); ok { - return NextAction{Kind: NextActionCutover, Deployment: d.Deployment} + return memberAction(NextActionCutover, d) } return NextAction{Kind: NextActionNone} } +// memberAction scopes a next action to one rollout member, carrying both the +// routing pair a surface needs to address it and the name it renders it under. +func memberAction(kind NextActionKind, d Deployment) NextAction { + return NextAction{Kind: kind, Deployment: d.Deployment, Target: d.Target, Name: d.Name} +} + // summaryCategoryOrder is the stable display order for the aggregate histogram. var summaryCategoryOrder = []struct { label string diff --git a/pkg/presentation/presentation_test.go b/pkg/presentation/presentation_test.go index d82b3e341..f837c3ea3 100644 --- a/pkg/presentation/presentation_test.go +++ b/pkg/presentation/presentation_test.go @@ -328,7 +328,7 @@ func TestDerive_AggregateBarrierWorkedExample(t *testing.T) { assert.Equal(t, StateWaiting, got.Deployments[3].Presentation) assert.Equal(t, "waiting for us", got.Deployments[3].Label) - assert.Equal(t, NextAction{Kind: NextActionCutover, Deployment: "eu"}, got.NextAction) + assert.Equal(t, NextAction{Kind: NextActionCutover, Deployment: "eu", Name: "eu"}, got.NextAction) assert.Equal(t, []StateCount{ {Label: "ready for cutover", Count: 1}, {Label: "running", Count: 1}, @@ -352,7 +352,7 @@ func TestDerive_AggregateFailedHaltExample(t *testing.T) { assert.Equal(t, state.Apply.RunningDegraded, got.State) assert.Equal(t, "running (degraded)", got.Label) - assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "us"}, got.NextAction) + assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "us", Name: "us"}, got.NextAction) assert.Equal(t, StateHalted, got.Deployments[2].Presentation) assert.Equal(t, "halted — us failed", got.Deployments[2].Label) assert.Equal(t, StateHalted, got.Deployments[3].Presentation) @@ -372,7 +372,7 @@ func TestDerive_AggregateFailedHaltSettlesOnceStartedWorkEnds(t *testing.T) { assert.Equal(t, state.Apply.Failed, got.State) assert.Equal(t, "failed", got.Label) - assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "us"}, got.NextAction) + assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "us", Name: "us"}, got.NextAction) assert.Equal(t, StateHalted, got.Deployments[2].Presentation) assert.Equal(t, StateHalted, got.Deployments[3].Presentation) } @@ -473,7 +473,7 @@ func TestDerive_HaltFailureWithRunningSiblingRunsDegraded(t *testing.T) { }) assert.Equal(t, state.Apply.RunningDegraded, got.State) assert.Equal(t, "running (degraded)", got.Label) - assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "eu"}, got.NextAction) + assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "eu", Name: "eu"}, got.NextAction) } // TestDerive_BothFailurePolicyFlagsPointsAtTheFailure: the two on_failure flags @@ -487,7 +487,7 @@ func TestDerive_BothFailurePolicyFlagsPointsAtTheFailure(t *testing.T) { {Deployment: "us", State: so.Running, ContinueOnFailure: true, PauseOnFailure: true}, }) assert.Equal(t, state.Apply.RunningDegraded, got.State) - assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "eu"}, got.NextAction) + assert.Equal(t, NextAction{Kind: NextActionReviewFailure, Deployment: "eu", Name: "eu"}, got.NextAction) } // TestDerive_PauseFailureWithPendingSiblingHoldsPaused: under on_failure pause an @@ -531,3 +531,68 @@ func TestDerive_FirstFailureExcludesRetrying(t *testing.T) { }) assert.Nil(t, got.FirstFailure) } + +// TestDerive_MultiTargetMemberNames: when one deployment addresses several +// targets, every member of that deployment is named by the routing pair, so a +// surface never labels two members identically. A sibling deployment that +// addresses a single target keeps its plain name. +func TestDerive_MultiTargetMemberNames(t *testing.T) { + got := Derive([]Operation{ + {Deployment: "primary", Target: "testapp-001", State: so.Completed}, + {Deployment: "primary", Target: "testapp-002", State: so.Running}, + {Deployment: "eu-west", Target: "orders-eu", State: so.Pending}, + }) + require.Len(t, got.Deployments, 3) + assert.Equal(t, "primary/testapp-001", got.Deployments[0].Name) + assert.Equal(t, "primary/testapp-002", got.Deployments[1].Name) + assert.Equal(t, "eu-west", got.Deployments[2].Name) + // The routing pair travels alongside the name so a surface can still + // address the member it just labelled. + assert.Equal(t, "primary", got.Deployments[1].Deployment) + assert.Equal(t, "testapp-002", got.Deployments[1].Target) +} + +// TestDerive_MultiTargetLabelsNameTheBlockingMember: a member held by an earlier +// sibling names that sibling the same way the sibling's own section is named, so +// an operator reading "halted — X failed" can find X. +func TestDerive_MultiTargetLabelsNameTheBlockingMember(t *testing.T) { + got := Derive([]Operation{ + {Deployment: "primary", Target: "testapp-001", State: so.Failed, Error: "lock wait timeout"}, + {Deployment: "primary", Target: "testapp-002", State: so.Pending}, + }) + require.Len(t, got.Deployments, 2) + assert.Equal(t, StateHalted, got.Deployments[1].Presentation) + assert.Equal(t, "halted — primary/testapp-001 failed", got.Deployments[1].Label) + require.NotNil(t, got.FirstFailure) + assert.Equal(t, "primary/testapp-001", got.FirstFailure.Name) +} + +// TestDerive_MultiTargetNextActionCarriesMemberIdentity: a cutover suggestion +// names the member it applies to and carries the routing pair that addresses it, +// so the two targets of one deployment produce distinguishable next actions. +func TestDerive_MultiTargetNextActionCarriesMemberIdentity(t *testing.T) { + got := Derive([]Operation{ + {Deployment: "primary", Target: "testapp-001", State: so.Completed}, + {Deployment: "primary", Target: "testapp-002", State: so.WaitingForCutover}, + }) + assert.Equal(t, NextAction{ + Kind: NextActionCutover, + Deployment: "primary", + Target: "testapp-002", + Name: "primary/testapp-002", + }, got.NextAction) +} + +// TestDerive_KeyedApplyStaysNamedByDeployment: several operations of one +// deployment against the same target are a keyed apply, not separate members. +// The target half would not tell them apart, so the plain deployment name is +// kept and the operation key does the disambiguating on the surface. +func TestDerive_KeyedApplyStaysNamedByDeployment(t *testing.T) { + got := Derive([]Operation{ + {Deployment: "us-east", Target: "orders-us", State: so.Completed}, + {Deployment: "us-east", Target: "orders-us", State: so.Running}, + }) + require.Len(t, got.Deployments, 2) + assert.Equal(t, "us-east", got.Deployments[0].Name) + assert.Equal(t, "us-east", got.Deployments[1].Name) +} diff --git a/pkg/routing/resolver.go b/pkg/routing/resolver.go index 34777c014..ffd11f7bb 100644 --- a/pkg/routing/resolver.go +++ b/pkg/routing/resolver.go @@ -32,6 +32,42 @@ func (t ExecutionTarget) MemberID() string { return t.Deployment + "/" + t.Target } +// DisplayNames returns the operator-facing name of each member of a rollout, in +// the order given. MemberID is the identity every comparison must use, but it is +// not always the right thing to show: a deployment that addresses exactly one +// target is already unambiguous, and naming it "deployment/target" everywhere +// would add a second half that never distinguishes anything. So a member is +// named by its deployment alone unless its deployment addresses more than one +// distinct target in this rollout, in which case every member of that deployment +// is named by its full MemberID. +// +// Keying on distinct targets, rather than on how many members a deployment has, +// is what keeps a keyed or sharded apply — several operations of one deployment +// against the same target — named by the deployment, where the extra half would +// be noise that still did not tell the operations apart. +func DisplayNames(members []ExecutionTarget) []string { + targetsByDeployment := make(map[string]map[string]struct{}, len(members)) + for _, m := range members { + if m.Target == "" { + continue + } + if targetsByDeployment[m.Deployment] == nil { + targetsByDeployment[m.Deployment] = make(map[string]struct{}, 1) + } + targetsByDeployment[m.Deployment][m.Target] = struct{}{} + } + + names := make([]string, len(members)) + for i, m := range members { + if len(targetsByDeployment[m.Deployment]) > 1 { + names[i] = m.MemberID() + continue + } + names[i] = m.Deployment + } + return names +} + // Resolver resolves logical SchemaBot targets to concrete execution targets. type Resolver interface { ResolveTargets(ctx context.Context, req Request) ([]ExecutionTarget, error) diff --git a/pkg/routing/resolver_test.go b/pkg/routing/resolver_test.go new file mode 100644 index 000000000..f7e6068bf --- /dev/null +++ b/pkg/routing/resolver_test.go @@ -0,0 +1,75 @@ +package routing + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestDisplayNames(t *testing.T) { + tests := []struct { + name string + members []ExecutionTarget + want []string + }{ + { + name: "one deployment per target is named by the deployment", + members: []ExecutionTarget{ + {Deployment: "us-east", Target: "orders-us"}, + {Deployment: "eu-west", Target: "orders-eu"}, + }, + want: []string{"us-east", "eu-west"}, + }, + { + name: "a deployment addressing several targets names every one of its members in full", + members: []ExecutionTarget{ + {Deployment: "primary", Target: "testapp-001"}, + {Deployment: "primary", Target: "testapp-002"}, + {Deployment: "eu-west", Target: "orders-eu"}, + }, + want: []string{"primary/testapp-001", "primary/testapp-002", "eu-west"}, + }, + { + // A keyed or sharded apply runs several operations of one deployment + // against the same target. The target half would not tell them apart, + // so it is not added. + name: "several members on one target stay named by the deployment", + members: []ExecutionTarget{ + {Deployment: "us-east", Target: "orders-us"}, + {Deployment: "us-east", Target: "orders-us"}, + }, + want: []string{"us-east", "us-east"}, + }, + { + name: "members that have not recorded a target are named by the deployment", + members: []ExecutionTarget{ + {Deployment: "us-east"}, + {Deployment: "eu-west"}, + }, + want: []string{"us-east", "eu-west"}, + }, + { + // A member whose own target is still unrecorded is named in full when + // its deployment addresses several: the deployment name is ambiguous + // for it too, even though its own half is empty. + name: "an unrecorded target under a multi-target deployment is still named in full", + members: []ExecutionTarget{ + {Deployment: "primary", Target: "testapp-001"}, + {Deployment: "primary", Target: "testapp-002"}, + {Deployment: "primary"}, + }, + want: []string{"primary/testapp-001", "primary/testapp-002", "primary/"}, + }, + { + name: "no members", + members: nil, + want: []string{}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, DisplayNames(tt.members)) + }) + } +} diff --git a/pkg/webhook/multi_apply.go b/pkg/webhook/multi_apply.go index da3aa60f3..abd257102 100644 --- a/pkg/webhook/multi_apply.go +++ b/pkg/webhook/multi_apply.go @@ -94,9 +94,14 @@ func buildMultiApplyData(apply *storage.Apply, ops []*storage.ApplyOperation, re tasksByOp := groupTasksByOperation(tasks) model := deriveApplyPresentation(ops, released) - details := make(map[string]templates.ApplyStatusCommentData, len(ops)) + // Derive returns one presentation per operation in input order, so the + // details are built in that same order and consumed positionally. A + // deployment can own several operations, which a name-keyed map could not + // tell apart. + details := make([]*templates.ApplyStatusCommentData, 0, len(ops)) for _, op := range ops { - details[op.Deployment] = buildDeploymentDetail(apply, op, tasksByOp[op.ID], displayByOp[op.ID], shardsByTable, tenant) + detail := buildDeploymentDetail(apply, op, tasksByOp[op.ID], displayByOp[op.ID], shardsByTable, tenant) + details = append(details, &detail) } data := templates.MultiDeploymentApplyData{ @@ -140,6 +145,7 @@ func deriveApplyPresentation(ops []*storage.ApplyOperation, released bool) prese func applyOperationToPresentation(op *storage.ApplyOperation, released bool) presentation.Operation { return presentation.Operation{ Deployment: op.Deployment, + Target: op.Target, State: op.State, Barrier: op.CutoverPolicy == storage.CutoverPolicyBarrier, Parallel: op.CutoverPolicy == storage.CutoverPolicyParallel, diff --git a/pkg/webhook/multi_apply_test.go b/pkg/webhook/multi_apply_test.go index 88e0978b7..df07ad06b 100644 --- a/pkg/webhook/multi_apply_test.go +++ b/pkg/webhook/multi_apply_test.go @@ -122,10 +122,10 @@ func TestBuildMultiApplyData_RoutesTasksByOperation(t *testing.T) { } data := buildMultiApplyData(runningApply(), ops, false, tasks, nil, nil, "") - require.Len(t, data.Details["eu"].Tables, 1) - assert.Equal(t, "customers", data.Details["eu"].Tables[0].TableName) - require.Len(t, data.Details["us"].Tables, 1) - assert.Equal(t, "orders", data.Details["us"].Tables[0].TableName) + require.Len(t, data.Details[0].Tables, 1) + assert.Equal(t, "customers", data.Details[0].Tables[0].TableName) + require.Len(t, data.Details[1].Tables, 1) + assert.Equal(t, "orders", data.Details[1].Tables[0].TableName) } // Per-shard rows are scoped to their owning deployment: when two deployments @@ -152,12 +152,12 @@ func TestBuildMultiApplyData_ScopesShardsByOperation(t *testing.T) { data := buildMultiApplyData(runningApply(), ops, false, tasks, nil, shardsByTable, "") - require.Len(t, data.Details["eu"].Tables, 1) - require.Len(t, data.Details["eu"].Tables[0].Shards, 1) - assert.Equal(t, "-80", data.Details["eu"].Tables[0].Shards[0].Shard) - require.Len(t, data.Details["us"].Tables, 1) - require.Len(t, data.Details["us"].Tables[0].Shards, 1) - assert.Equal(t, "80-", data.Details["us"].Tables[0].Shards[0].Shard) + require.Len(t, data.Details[0].Tables, 1) + require.Len(t, data.Details[0].Tables[0].Shards, 1) + assert.Equal(t, "-80", data.Details[0].Tables[0].Shards[0].Shard) + require.Len(t, data.Details[1].Tables, 1) + require.Len(t, data.Details[1].Tables[0].Shards, 1) + assert.Equal(t, "80-", data.Details[1].Tables[0].Shards[0].Shard) } // The per-deployment section reflects the operation's own state and error, not diff --git a/pkg/webhook/plan_drift.go b/pkg/webhook/plan_drift.go index cc353080a..a89d9ca89 100644 --- a/pkg/webhook/plan_drift.go +++ b/pkg/webhook/plan_drift.go @@ -89,6 +89,7 @@ func deploymentDriftPreview(rollup api.PlanRollup) *templates.DeploymentDriftDat for i, e := range rollup.Entries { entry := templates.DeploymentDriftEntry{ Deployment: e.Deployment, + Target: e.Target, Primary: i == 0, Class: e.Class.String(), } @@ -108,6 +109,7 @@ func deploymentDriftPreview(rollup api.PlanRollup) *templates.DeploymentDriftDat Deployments: entries, Clean: rollup.Clean, Computed: true, + Independent: rollup.Planning == api.PlanIndependent, } } @@ -135,6 +137,16 @@ func describeDriftDiff(diff tern.ChangeSetDiff) string { return strings.Join(parts, ", ") + " change(s) vs the reviewed plan" } +// rollupMemberNames renders each rollup entry the way an operator addresses it, +// index-parallel to rollup.Entries. +func rollupMemberNames(rollup api.PlanRollup) []string { + members := make([]routing.ExecutionTarget, len(rollup.Entries)) + for i, e := range rollup.Entries { + members[i] = routing.ExecutionTarget{Deployment: e.Deployment, Target: e.Target} + } + return routing.DisplayNames(members) +} + // maxDriftSummaryLen bounds the stored drift summary to the checks table's // change_summary column width. The summary is truncated on a rune boundary so it // never exceeds the column or splits a multibyte character. @@ -145,13 +157,18 @@ const maxDriftSummaryLen = 255 // from the reviewed plan and those that could not be diffed or compared, so the // check's Change column tells an operator exactly which deployment to reconcile. func summarizeReviewDrift(rollup api.PlanRollup) string { + independent := rollup.Planning == api.PlanIndependent + // One deployment can address several targets, so the deployment name alone + // does not always say which member failed. The shared naming rule adds the + // target only where it disambiguates. + names := rollupMemberNames(rollup) var diverged, errored []string - for _, entry := range rollup.Entries { + for i, entry := range rollup.Entries { switch entry.Class { case api.DeploymentDiverged: - diverged = append(diverged, entry.Deployment) + diverged = append(diverged, names[i]) case api.DeploymentErrored: - errored = append(errored, entry.Deployment) + errored = append(errored, names[i]) } } @@ -160,13 +177,28 @@ func summarizeReviewDrift(rollup api.PlanRollup) string { parts = append(parts, fmt.Sprintf("diverged: %s", strings.Join(diverged, ", "))) } if len(errored) > 0 { - parts = append(parts, fmt.Sprintf("could not verify: %s", strings.Join(errored, ", "))) + // An errored member means different things under the two contracts: a + // mirrored member's diff could not be confirmed against the reviewed plan, + // while an independent member has no plan of its own at all. + reason := "could not verify" + if independent { + reason = "could not plan" + } + parts = append(parts, fmt.Sprintf("%s: %s", reason, strings.Join(errored, ", "))) } if len(parts) == 0 { - // A not-clean rollup always has at least one non-matching entry; guard + // A not-clean rollup always has at least one non-passing entry; guard // anyway so the check never records an empty, uninformative reason. + if independent { + return "blocks apply: not every target could be planned" + } return "drift blocks apply: deployments differ from the reviewed plan" } + // Targets that hold their own schemas are never expected to agree, so their + // failure is an unplanned target, not drift between them. + if independent { + return clampDriftSummary("blocks apply — " + strings.Join(parts, "; ")) + } return clampDriftSummary("drift blocks apply — " + strings.Join(parts, "; ")) } diff --git a/pkg/webhook/plan_drift_test.go b/pkg/webhook/plan_drift_test.go index 1b658ab71..d01c5caa6 100644 --- a/pkg/webhook/plan_drift_test.go +++ b/pkg/webhook/plan_drift_test.go @@ -173,3 +173,22 @@ func TestDeploymentDriftPreview_DivergedAndErroredDetails(t *testing.T) { assert.Equal(t, erroredDriftDetail, preview.Deployments[2].Detail) assert.NotContains(t, preview.Deployments[2].Detail, assert.AnError.Error()) } + +// A rollup where one deployment addresses several targets names each failing +// member by its routing pair, so the check's Change column identifies exactly +// which member to reconcile rather than a deployment with two of them. +func TestSummarizeReviewDrift_NamesMultiTargetMembers(t *testing.T) { + rollup := api.PlanRollup{ + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + {Deployment: "primary", Target: "testapp-001", Class: api.DeploymentPlanned}, + {Deployment: "primary", Target: "testapp-002", Class: api.DeploymentErrored}, + {Deployment: "eu-west", Target: "orders-eu", Class: api.DeploymentErrored}, + }, + } + summary := summarizeReviewDrift(rollup) + // Independent targets are never expected to agree, so an unplannable one is + // not drift between them. + assert.Contains(t, summary, "could not plan: primary/testapp-002, eu-west") + assert.NotContains(t, summary, "drift blocks apply") +} diff --git a/pkg/webhook/templates/multi_apply.go b/pkg/webhook/templates/multi_apply.go index 8dac37b91..41c6a90c8 100644 --- a/pkg/webhook/templates/multi_apply.go +++ b/pkg/webhook/templates/multi_apply.go @@ -39,10 +39,16 @@ type MultiDeploymentApplyData struct { StartedAt string CompletedAt string - // Details maps a deployment name to that deployment's single-deployment - // comment data (its tables, error, timing, database). Each deployment's - //
body is rendered from its entry via RenderApplyStatusComment. - Details map[string]ApplyStatusCommentData + // Details is each member's single-deployment comment data (its tables, + // error, timing, database), index-parallel to Model.Deployments. Each + // member's
body is rendered from its entry via + // RenderApplyStatusComment; a nil entry, or an index past the end, renders + // the no-detail placeholder instead. It is positional rather than keyed by + // deployment name because a deployment can own several operations — + // different targets of one deployment, or the several operations of a keyed + // apply — and a name-keyed lookup would render one member's tables under + // every one of them. + Details []*ApplyStatusCommentData // Tenant is the deployment's tenant identity, appended as --tenant to every // pasteable command hint so copied commands address this deployment in @@ -161,7 +167,7 @@ func writeAggregateFirstFailure(sb *strings.Builder, failure *presentation.Deplo if failure == nil { return } - name := html.EscapeString(failure.Deployment) + name := html.EscapeString(failure.Name) msg := SanitizeInlineError(failure.Error) if msg == "" { fmt.Fprintf(sb, "\n> "+glyph.Failed+" **First failure:** %s\n", name) @@ -181,7 +187,7 @@ func writeAggregateNextAction(sb *strings.Builder, data MultiDeploymentApplyData switch na.Kind { case presentation.NextActionCutover: writeFooterAction(sb, - fmt.Sprintf("To cut over `%s`:", na.Deployment), + fmt.Sprintf("To cut over `%s`:", na.Name), appendTenantFlag(fmt.Sprintf("schemabot cutover %s -e %s", data.ApplyID, data.Environment), data.Tenant)) case presentation.NextActionResume: writeFooterAction(sb, "Paused — to resume from where it stopped:", appendTenantFlag(fmt.Sprintf("schemabot start %s -e %s", data.ApplyID, data.Environment), data.Tenant)) @@ -233,15 +239,16 @@ func writeDeploymentSummarySections(sb *strings.Builder, data MultiDeploymentApp // already carries the title and the line names the deployment, so // repeating the headline inside every section is noise. func writeDeploymentDetailSections(sb *strings.Builder, data MultiDeploymentApplyData, renderDetail func(ApplyStatusCommentData) string) { - for _, d := range data.Model.Deployments { + for i, d := range data.Model.Deployments { openAttr := "" if d.Open { openAttr = " open" } fmt.Fprintf(sb, "\n\n%s — %s\n\n", openAttr, deploymentTag(d), html.EscapeString(d.Label)) - if detail, ok := data.Details[d.Deployment]; ok { - detail.DerivedStatus = siblingDerivedStatus(d) - sb.WriteString(stripLeadingHeading(renderDetail(detail))) + if detail := memberDetail(data.Details, i); detail != nil { + body := *detail + body.DerivedStatus = siblingDerivedStatus(d) + sb.WriteString(stripLeadingHeading(renderDetail(body))) } else { sb.WriteString("_No details available yet._\n") } @@ -249,6 +256,16 @@ func writeDeploymentDetailSections(sb *strings.Builder, data MultiDeploymentAppl } } +// memberDetail returns member i's comment data, or nil when the caller supplied +// no detail for it — either a nil entry or a details slice that stops short of +// the member set, both of which mean the member has nothing to show yet. +func memberDetail(details []*ApplyStatusCommentData, i int) *ApplyStatusCommentData { + if i >= len(details) { + return nil + } + return details[i] +} + // siblingDerivedStatus returns the
body status for a deployment whose // presentation is derived from its earlier siblings: queued, waiting, halted, // or paused. The raw operation state for all four is pending, which the @@ -286,10 +303,11 @@ func stripLeadingHeading(body string) string { return strings.TrimLeft(rest, "\n") } -// deploymentTag renders the " " prefix, omitting the leading -// space when a state has no glyph. +// deploymentTag renders the " " prefix, omitting the leading +// space when a state has no glyph. The member is named by the derivation's +// resolved name, so two targets of one deployment are labelled distinctly. func deploymentTag(d presentation.Deployment) string { - name := html.EscapeString(d.Deployment) + name := html.EscapeString(d.Name) if d.Emoji == "" { return name } diff --git a/pkg/webhook/templates/multi_apply_tenant_test.go b/pkg/webhook/templates/multi_apply_tenant_test.go index 9dca7b239..2932e181b 100644 --- a/pkg/webhook/templates/multi_apply_tenant_test.go +++ b/pkg/webhook/templates/multi_apply_tenant_test.go @@ -62,8 +62,8 @@ func TestMultiDeploymentApplyDetailHintsCarryTenant(t *testing.T) { ApplyID: "apply-123", Environment: "production", Tenant: "acme", - Details: map[string]ApplyStatusCommentData{ - "eu": {ApplyID: "apply-123", Environment: "production", State: state.Apply.Stopped, Tenant: "acme"}, + Details: []*ApplyStatusCommentData{ + {ApplyID: "apply-123", Environment: "production", State: state.Apply.Stopped, Tenant: "acme"}, }, } diff --git a/pkg/webhook/templates/multi_apply_test.go b/pkg/webhook/templates/multi_apply_test.go index 57186450d..2e9b3abb5 100644 --- a/pkg/webhook/templates/multi_apply_test.go +++ b/pkg/webhook/templates/multi_apply_test.go @@ -118,8 +118,8 @@ func TestRenderMultiDeploymentApplyComment_UsesOneRenderTimestamp(t *testing.T) ApplyID: "apply-123", Environment: "production", RequestedBy: "aparajon", - Details: map[string]ApplyStatusCommentData{ - "us": { + Details: []*ApplyStatusCommentData{ + { Database: "payments_us", Environment: "production", State: state.Apply.Running, @@ -247,8 +247,10 @@ func TestRenderMultiDeploymentApplyComment_DetailsReuseSingleRenderer(t *testing Model: model, ApplyID: "apply-123", Environment: "production", - Details: map[string]ApplyStatusCommentData{ - "us": { + Details: []*ApplyStatusCommentData{ + // eu has no detail yet, so its section renders the placeholder. + nil, + { Database: "payments_us", State: state.Apply.Running, Tables: []TableProgressData{ @@ -282,8 +284,8 @@ func TestRenderMultiDeploymentApplySummaryComment_CompletedReusesSummaryRenderer Model: model, ApplyID: "apply-123", Environment: "production", - Details: map[string]ApplyStatusCommentData{ - "eu": { + Details: []*ApplyStatusCommentData{ + { Database: "payments_eu", Environment: "production", State: state.Apply.Completed, @@ -291,7 +293,7 @@ func TestRenderMultiDeploymentApplySummaryComment_CompletedReusesSummaryRenderer {TableName: "orders", Status: state.Task.Completed}, }, }, - "us": { + { Database: "payments_us", Environment: "production", State: state.Apply.Completed, @@ -328,14 +330,14 @@ func TestRenderMultiDeploymentApplyComment_DetailsOmitDuplicateHeadline(t *testi Model: model, ApplyID: "apply-123", Environment: "production", - Details: map[string]ApplyStatusCommentData{ - "eu": { + Details: []*ApplyStatusCommentData{ + { Database: "payments_eu", Environment: "production", State: state.Apply.Completed, Tables: []TableProgressData{{TableName: "orders", Status: state.Task.Completed}}, }, - "us": { + { Database: "payments_us", Environment: "production", State: state.Apply.Running, @@ -364,14 +366,14 @@ func TestRenderMultiDeploymentApplySummaryComment_DetailsOmitDuplicateHeadline(t Model: model, ApplyID: "apply-123", Environment: "production", - Details: map[string]ApplyStatusCommentData{ - "eu": { + Details: []*ApplyStatusCommentData{ + { Database: "payments_eu", Environment: "production", State: state.Apply.Completed, Tables: []TableProgressData{{TableName: "orders", Status: state.Task.Completed}}, }, - "us": { + { Database: "payments_us", Environment: "production", State: state.Apply.Completed, @@ -397,8 +399,10 @@ func TestRenderMultiDeploymentApplySummaryComment_FailedDeploymentSummary(t *tes Model: model, ApplyID: "apply-123", Environment: "production", - Details: map[string]ApplyStatusCommentData{ - "us": { + Details: []*ApplyStatusCommentData{ + // eu carries no summary detail; only the failed member's is asserted. + nil, + { Database: "payments_us", Environment: "production", State: state.Apply.Failed, @@ -431,8 +435,8 @@ func TestRenderMultiDeploymentApplyComment_PreflightFailureHasNoProgressBar(t *t Model: model, ApplyID: "apply-123", Environment: "qa", - Details: map[string]ApplyStatusCommentData{ - "apse2": { + Details: []*ApplyStatusCommentData{ + { Database: "profiles_db", Environment: "qa", State: state.Apply.Failed, @@ -461,9 +465,9 @@ func TestRenderMultiDeploymentApplyComment_HaltedDetailUsesDerivedStatus(t *test Model: model, ApplyID: "apply-123", Environment: "qa", - Details: map[string]ApplyStatusCommentData{ - "apse2": {Database: "app_db", Environment: "qa", State: state.Apply.Failed}, - "euwe1": {Database: "app_db", Environment: "qa", State: state.Apply.Pending}, + Details: []*ApplyStatusCommentData{ + {Database: "app_db", Environment: "qa", State: state.Apply.Failed}, + {Database: "app_db", Environment: "qa", State: state.Apply.Pending}, }, }) @@ -486,9 +490,9 @@ func TestRenderMultiDeploymentApplyComment_PendingDetailUsesDerivedStatus(t *tes Model: model, ApplyID: "apply-123", Environment: "staging", - Details: map[string]ApplyStatusCommentData{ - "eu": {Database: "app_db", Environment: "staging", State: state.Apply.Running}, - "us": {Database: "app_db", Environment: "staging", State: state.Apply.Pending}, + Details: []*ApplyStatusCommentData{ + {Database: "app_db", Environment: "staging", State: state.Apply.Running}, + {Database: "app_db", Environment: "staging", State: state.Apply.Pending}, }, }) @@ -510,9 +514,9 @@ func TestRenderMultiDeploymentApplyComment_DerivedStatusEscapesLabel(t *testing. Model: model, ApplyID: "apply-123", Environment: "staging", - Details: map[string]ApplyStatusCommentData{ - "eu": {Database: "app_db", Environment: "staging", State: state.Apply.Running}, - "us": {Database: "app_db", Environment: "staging", State: state.Apply.Pending}, + Details: []*ApplyStatusCommentData{ + {Database: "app_db", Environment: "staging", State: state.Apply.Running}, + {Database: "app_db", Environment: "staging", State: state.Apply.Pending}, }, }) @@ -605,3 +609,67 @@ func TestRenderMultiDeploymentApplyComment_FirstFailureErrorSanitized(t *testing assert.Contains(t, out, "> ❌ **First failure:** us — dial tcp [endpoint redacted]: refused second line\n", "the first-failure line stays on one line") } + +// memberOp builds a rolling operation for one rollout member, addressing a +// target within its deployment. +func memberOp(dep, target, st string) presentation.Operation { + return presentation.Operation{Deployment: dep, Target: target, State: st} +} + +// When one deployment addresses several targets, each member gets its own +// summary line, its own
section, and its own body — the deployment +// name alone would label two sections identically, and pairing details by name +// would give both members the same body. +func TestRenderMultiDeploymentApplyComment_MultiTargetMembersRenderSeparately(t *testing.T) { + model := presentation.Derive([]presentation.Operation{ + memberOp("primary", "testapp-001", so.Completed), + memberOp("primary", "testapp-002", so.Running), + memberOp("eu-west", "orders-eu", so.Pending), + }) + out := RenderMultiDeploymentApplyComment(MultiDeploymentApplyData{ + Model: model, + ApplyID: "apply-123", + Environment: "production", + Details: []*ApplyStatusCommentData{ + {Database: "testapp_001", State: state.Apply.Completed}, + { + Database: "testapp_002", + State: state.Apply.Running, + Tables: []TableProgressData{ + {TableName: "orders", Status: state.Task.Running, PercentComplete: 42, RowsCopied: 420, RowsTotal: 1000}, + }, + }, + nil, + }, + }) + + // Both members of "primary" are named in full; the single-target sibling is not. + assert.Contains(t, out, "- ✅ primary/testapp-001 — completed") + assert.Contains(t, out, "- 🔄 primary/testapp-002 — running table copy") + assert.Contains(t, out, "- ⏳ eu-west — waiting for primary/testapp-002") + assert.Contains(t, out, "
\n✅ primary/testapp-001 — completed") + assert.Contains(t, out, "
\n🔄 primary/testapp-002 — running table copy") + + // Each member's body is its own: the details are paired positionally, so the + // two members of one deployment do not collapse onto a single body. + assert.Contains(t, out, "**Database**: `testapp_001`") + assert.Contains(t, out, "**Database**: `testapp_002`") + assert.Equal(t, 1, strings.Count(out, "**Database**: `testapp_002`")) + assert.Contains(t, out, "_No details available yet._") +} + +// A cutover suggestion names the member it applies to, so an operator reading it +// on a deployment with several targets knows which one is parked at the barrier. +func TestRenderMultiDeploymentApplyComment_NextActionNamesMultiTargetMember(t *testing.T) { + model := presentation.Derive([]presentation.Operation{ + {Deployment: "primary", Target: "testapp-001", State: so.Completed, Barrier: true}, + {Deployment: "primary", Target: "testapp-002", State: so.WaitingForCutover, Barrier: true}, + }) + out := RenderMultiDeploymentApplyComment(MultiDeploymentApplyData{ + Model: model, + ApplyID: "apply-123", + Environment: "production", + }) + + assert.Contains(t, out, "To cut over `primary/testapp-002`:") +} diff --git a/pkg/webhook/templates/plan.go b/pkg/webhook/templates/plan.go index d315fff5d..92228268f 100644 --- a/pkg/webhook/templates/plan.go +++ b/pkg/webhook/templates/plan.go @@ -10,6 +10,7 @@ import ( "github.com/block/schemabot/pkg/caller" "github.com/block/schemabot/pkg/ddl" "github.com/block/schemabot/pkg/glyph" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/schema" "github.com/block/schemabot/pkg/storage" "github.com/block/schemabot/pkg/ui" @@ -187,27 +188,51 @@ type ExemptTablesData struct { // per-deployment breakdown when some deployment diverged from — or could not be // confirmed against — the reviewed plan. type DeploymentDriftData struct { - // Deployments is every configured deployment in rollout order, primary first. + // Deployments is every configured rollout member in rollout order, primary + // first. Deployments []DeploymentDriftEntry - // Clean is true only when every deployment matches the reviewed plan. + // Clean means every member passed its contract: under mirrored members, that + // they all match the reviewed plan; under independent members, that they all + // produced a plan of their own. Clean bool // Computed is false when the rollup itself could not be evaluated; the check // still fails closed, and the preview says the deployments are unverified. Computed bool + // Independent reports that each member holds its own schema and was planned + // against it. Members are then not expected to match each other, so a clean + // rollup means every target was planned rather than that they agree — which + // is the opposite of what the mirrored wording says. + Independent bool } -// DeploymentDriftEntry is one deployment's classification against the reviewed -// primary plan. +// DeploymentDriftEntry is one rollout member's classification against the +// reviewed primary plan. type DeploymentDriftEntry struct { Deployment string - Primary bool - // Class is "match", "diverged", or "errored". + // Target is the member's target within its deployment. One deployment can + // address several targets, so the deployment name alone does not always name + // the member. + Target string + Primary bool + // Class is "match", "planned", "diverged", or "errored". Class string - // Detail is a short human explanation for a diverged or errored deployment; - // empty for a match. + // Detail is a short human explanation for a diverged or errored member; + // empty for a member that passed. Detail string } +// driftMemberNames renders each rollup entry the way an operator addresses it, +// index-parallel to the entries. The naming rule is shared with every other +// member-facing surface, so a deployment that addresses several targets is named +// the same way in the plan comment, the check summary, and the progress comment. +func driftMemberNames(entries []DeploymentDriftEntry) []string { + members := make([]routing.ExecutionTarget, len(entries)) + for i, e := range entries { + members[i] = routing.ExecutionTarget{Deployment: e.Deployment, Target: e.Target} + } + return routing.DisplayNames(members) +} + // applyingWithoutConfirmation reports whether this comment announces an apply // that is already running rather than one waiting on the operator: a locked // comment with nothing pausing it. Nothing on such a comment is a question, so @@ -1106,25 +1131,48 @@ func writeDeploymentDrift(sb *strings.Builder, drift *DeploymentDriftData) { return } + names := driftMemberNames(drift.Deployments) if drift.Clean { + if drift.Independent { + fmt.Fprintf(sb, "✅ **Planned separately for all %d targets** (%s) — each target holds its own schema, so their plans are not expected to match.\n\n", + len(drift.Deployments), strings.Join(names, ", ")) + return + } fmt.Fprintf(sb, "✅ **Same plan on all %d deployments** (%s).\n\n", - len(drift.Deployments), joinDeploymentNames(drift.Deployments)) + len(drift.Deployments), strings.Join(names, ", ")) return } - sb.WriteString(glyph.Attention + " **Deployment drift detected** — some deployments no longer match the reviewed plan, so the plan check is failing closed:\n\n") - for _, d := range drift.Deployments { - name := "`" + d.Deployment + "`" + // A member that could not be planned blocks under either contract, but only + // mirrored members can be *out of agreement* with each other. Calling an + // independent environment's failure "drift" would tell an operator to go + // reconcile targets that are supposed to differ. + if drift.Independent { + sb.WriteString(glyph.Attention + " **Some targets could not be planned** — every target must have a plan before an apply can run, so the plan check is failing closed:\n\n") + } else { + sb.WriteString(glyph.Attention + " **Deployment drift detected** — some deployments no longer match the reviewed plan, so the plan check is failing closed:\n\n") + } + for i, d := range drift.Deployments { + name := "`" + names[i] + "`" if d.Primary { name += " (primary)" } switch d.Class { case "match": fmt.Fprintf(sb, "- %s ✅ matches the reviewed plan\n", name) + case "planned": + fmt.Fprintf(sb, "- %s ✅ planned against its own schema\n", name) case "diverged": fmt.Fprintf(sb, "- %s "+glyph.Attention+" diverged%s\n", name, driftDetailSuffix(d.Detail)) default: - fmt.Fprintf(sb, "- %s "+glyph.Failed+" could not verify%s\n", name, driftDetailSuffix(d.Detail)) + // An errored member means different things under the two contracts: + // a mirrored member's diff could not be confirmed against the + // reviewed plan, while an independent member has no plan at all. + reason := "could not verify" + if drift.Independent { + reason = "could not plan" + } + fmt.Fprintf(sb, "- %s "+glyph.Failed+" %s%s\n", name, reason, driftDetailSuffix(d.Detail)) } } sb.WriteString("\n") @@ -1139,15 +1187,6 @@ func driftDetailSuffix(detail string) string { return " — " + detail } -// joinDeploymentNames lists deployment names for the uniform drift line. -func joinDeploymentNames(deployments []DeploymentDriftEntry) string { - names := make([]string, len(deployments)) - for i, d := range deployments { - names[i] = d.Deployment - } - return strings.Join(names, ", ") -} - // writeBlockedChanges writes the section for statements the engine refuses, // naming each table and a sanitized, Markdown-safe engine reason. There is no // opt-in flag that lets these through — the remedy is whatever each reason diff --git a/pkg/webhook/templates/plan_drift_test.go b/pkg/webhook/templates/plan_drift_test.go index 50e028413..ed7f45ead 100644 --- a/pkg/webhook/templates/plan_drift_test.go +++ b/pkg/webhook/templates/plan_drift_test.go @@ -237,3 +237,64 @@ func TestAnyEnvHasDriftToShow(t *testing.T) { }) } } + +// When one deployment addresses several targets, the deployment name alone +// labels two different members identically. The plan comment names every member +// of that deployment by its routing pair, while a sibling deployment that +// addresses a single target keeps its plain name. +func TestRenderPlanComment_DriftNamesMultiTargetMembers(t *testing.T) { + data := PlanCommentData{ + Database: "testapp", Environment: "production", IsMySQL: true, + Changes: []KeyspaceChangeData{{ + Keyspace: "testapp", + Statements: []string{"ALTER TABLE `users` ADD COLUMN `email` varchar(255)"}, + }}, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, + Clean: false, + Independent: true, + Deployments: []DeploymentDriftEntry{ + {Deployment: "primary", Target: "testapp-001", Primary: true, Class: "planned"}, + {Deployment: "primary", Target: "testapp-002", Class: "errored", Detail: "diff failed; see server logs"}, + {Deployment: "eu-west", Target: "orders-eu", Class: "planned"}, + }, + }, + } + + out := RenderPlanComment(data) + assert.Contains(t, out, "Some targets could not be planned") + assert.Contains(t, out, "`primary/testapp-001` (primary)") + assert.Contains(t, out, "`primary/testapp-002`") + assert.Contains(t, out, "`eu-west`") + // An independent target has no plan of its own to compare, so its failure + // is reported as unplanned rather than unverified. + assert.Contains(t, out, "could not plan") + assert.NotContains(t, out, "could not verify") +} + +// The uniform clean line names members the same way the per-member breakdown +// does, so a reviewer sees one vocabulary for the rollout across both renderings. +func TestRenderPlanComment_DriftCleanNamesMultiTargetMembers(t *testing.T) { + data := PlanCommentData{ + Database: "testapp", Environment: "production", IsMySQL: true, + Changes: []KeyspaceChangeData{{ + Keyspace: "testapp", + Statements: []string{"ALTER TABLE `users` ADD COLUMN `email` varchar(255)"}, + }}, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, + Clean: true, + Independent: true, + Deployments: []DeploymentDriftEntry{ + {Deployment: "primary", Target: "testapp-001", Primary: true, Class: "planned"}, + {Deployment: "primary", Target: "testapp-002", Class: "planned"}, + {Deployment: "eu-west", Target: "orders-eu", Class: "planned"}, + }, + }, + } + + out := RenderPlanComment(data) + assert.Contains(t, out, "Planned separately for all 3 targets") + assert.Contains(t, out, "primary/testapp-001, primary/testapp-002, eu-west") + assert.True(t, strings.Contains(out, "each target holds its own schema")) +} diff --git a/pkg/webhook/templates/preview.go b/pkg/webhook/templates/preview.go index 55520fbac..bb039bcee 100644 --- a/pkg/webhook/templates/preview.go +++ b/pkg/webhook/templates/preview.go @@ -2073,8 +2073,8 @@ func PreviewCommentApplyReverting() string { // sampleDeploymentDetail builds one deployment's single-deployment comment data // for the per-deployment
body, with its own database name. -func sampleDeploymentDetail(database, applyState string, tables []TableProgressData) ApplyStatusCommentData { - return ApplyStatusCommentData{ +func sampleDeploymentDetail(database, applyState string, tables []TableProgressData) *ApplyStatusCommentData { + return &ApplyStatusCommentData{ Database: database, Environment: "production", RequestedBy: "aparajon", @@ -2116,9 +2116,9 @@ func PreviewCommentMultiDeploymentApplyInProgress() string { Environment: "production", RequestedBy: "aparajon", StartedAt: sampleTime().Add(-12 * time.Minute).UTC().Format(time.RFC3339), - Details: map[string]ApplyStatusCommentData{ - "eu": sampleDeploymentDetail("payments_eu", state.Apply.WaitingForCutover, euTables), - "us": sampleDeploymentDetail("payments_us", state.Apply.Running, usTables), + Details: []*ApplyStatusCommentData{ + sampleDeploymentDetail("payments_eu", state.Apply.WaitingForCutover, euTables), + sampleDeploymentDetail("payments_us", state.Apply.Running, usTables), }, }) } @@ -2155,9 +2155,9 @@ func PreviewCommentMultiDeploymentApplyFailed() string { Environment: "production", RequestedBy: "aparajon", StartedAt: sampleTime().Add(-20 * time.Minute).UTC().Format(time.RFC3339), - Details: map[string]ApplyStatusCommentData{ - "eu": sampleDeploymentDetail("payments_eu", state.Apply.Completed, euTables), - "us": usDetail, + Details: []*ApplyStatusCommentData{ + sampleDeploymentDetail("payments_eu", state.Apply.Completed, euTables), + usDetail, }, }) } @@ -2186,10 +2186,10 @@ func PreviewCommentMultiDeploymentApplyCompleted() string { RequestedBy: "aparajon", StartedAt: sampleTime().Add(-30 * time.Minute).UTC().Format(time.RFC3339), CompletedAt: sampleTime().Add(-2 * time.Minute).UTC().Format(time.RFC3339), - Details: map[string]ApplyStatusCommentData{ - "eu": sampleDeploymentDetail("payments_eu", state.Apply.Completed, completedTables()), - "us": sampleDeploymentDetail("payments_us", state.Apply.Completed, completedTables()), - "au": sampleDeploymentDetail("payments_au", state.Apply.Completed, completedTables()), + Details: []*ApplyStatusCommentData{ + sampleDeploymentDetail("payments_eu", state.Apply.Completed, completedTables()), + sampleDeploymentDetail("payments_us", state.Apply.Completed, completedTables()), + sampleDeploymentDetail("payments_au", state.Apply.Completed, completedTables()), }, }) } @@ -2219,10 +2219,10 @@ func PreviewCommentMultiDeploymentApplySummaryCompleted() string { RequestedBy: "aparajon", StartedAt: sampleTime().Add(-30 * time.Minute).UTC().Format(time.RFC3339), CompletedAt: sampleTime().Add(-2 * time.Minute).UTC().Format(time.RFC3339), - Details: map[string]ApplyStatusCommentData{ - "eu": sampleDeploymentDetail("payments_eu", state.Apply.Completed, completedTables()), - "us": sampleDeploymentDetail("payments_us", state.Apply.Completed, completedTables()), - "au": sampleDeploymentDetail("payments_au", state.Apply.Completed, completedTables()), + Details: []*ApplyStatusCommentData{ + sampleDeploymentDetail("payments_eu", state.Apply.Completed, completedTables()), + sampleDeploymentDetail("payments_us", state.Apply.Completed, completedTables()), + sampleDeploymentDetail("payments_au", state.Apply.Completed, completedTables()), }, }) } @@ -2258,9 +2258,9 @@ func PreviewCommentMultiDeploymentApplySummaryFailed() string { RequestedBy: "aparajon", StartedAt: sampleTime().Add(-20 * time.Minute).UTC().Format(time.RFC3339), CompletedAt: sampleTime().Add(-1 * time.Minute).UTC().Format(time.RFC3339), - Details: map[string]ApplyStatusCommentData{ - "eu": sampleDeploymentDetail("payments_eu", state.Apply.Completed, euTables), - "us": usDetail, + Details: []*ApplyStatusCommentData{ + sampleDeploymentDetail("payments_eu", state.Apply.Completed, euTables), + usDetail, }, }) } From d0bdeefd8c73cce8785999c1f7df9b77c332ad40 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 3 Sep 2026 16:39:44 -0400 Subject: [PATCH 16/34] feat(observability): attribute per-table progress to its rollout member Table progress rows are attributed to a deployment, which was enough while a deployment addressed one target. When it addresses several, each member runs its own copy of the same tables, and a section scoped to the deployment lists both members' copies under each of them. Progress now carries the target alongside the deployment, from the stored apply_operation through the API response to the CLI, and every consumer selects on both halves of the routing pair. The external apply ID a member inherits from a sibling that has already dispatched is gated the same way: a keyed apply's operations share one data-plane apply and may inherit it, while a deployment addressing several targets runs a separate apply per target and inherits nothing. --- TEMPLATES.md | 63 ------------------- pkg/api/progress_handlers.go | 40 ++++++------ pkg/api/progress_handlers_test.go | 37 +++++++++++ pkg/apitypes/apitypes.go | 7 ++- pkg/cmd/commands/watch_tui_test.go | 35 ++++++++++- pkg/cmd/commands/watch_tui_view_multi.go | 29 ++++++--- pkg/cmd/internal/templates/progress_multi.go | 52 ++++++++++----- .../internal/templates/progress_multi_test.go | 45 +++++++++++-- pkg/cmd/internal/templates/progress_parse.go | 9 ++- 9 files changed, 203 insertions(+), 114 deletions(-) diff --git a/TEMPLATES.md b/TEMPLATES.md index 067b5d290..28987068b 100644 --- a/TEMPLATES.md +++ b/TEMPLATES.md @@ -8000,23 +8000,10 @@ This schema change was cancelled and cannot be resumed. Open a new schema change 🟢 us-east — ready for cutover — next in order (orders-us-east) - ~ orders: 🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨 Waiting for cutover - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - 🔄 eu-west — running table copy (orders-eu-west) - ~ orders: 🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ 35.00% - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - • Rows: 42,000 / 120,000 · ETA: 4m 0s - - ⏳ ap-south — waiting for eu-west (orders-ap-south) - ~ orders: ⏳ Queued - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ```
@@ -8043,23 +8030,11 @@ This schema change was cancelled and cannot be resumed. Open a new schema change ✅ us-east — completed (orders-us-east) - ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ❌ eu-west — failed (orders-eu-west) duplicate key name 'idx_orders_source' - ~ orders: ⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ ❌ Failed - ALTER TABLE `orders` ADD INDEX `idx_orders_source`(`source`); - - ⏸️ ap-south — halted — eu-west failed (orders-ap-south) - ~ orders: 🚫 Cancelled (not started) - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ```
@@ -8087,23 +8062,10 @@ This schema change was cancelled and cannot be resumed. Open a new schema change ❌ us-east — failed (orders-us-east) duplicate key name 'idx_orders_source' - ~ orders: ⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ ❌ Failed - ALTER TABLE `orders` ADD INDEX `idx_orders_source`(`source`); - - 🔄 eu-west — running table copy (orders-eu-west) - ~ orders: 🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ 35.00% - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - • Rows: 42,000 / 120,000 · ETA: 4m 0s - - ⏸️ ap-south — halted — us-east failed (orders-ap-south) - ~ orders: ⏳ Queued - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ```
@@ -8126,22 +8088,10 @@ This schema change was cancelled and cannot be resumed. Open a new schema change ✅ us-east — completed (orders-us-east) - ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ✅ eu-west — completed (orders-eu-west) - ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ✅ ap-south — completed (orders-ap-south) - ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ```
@@ -8738,25 +8688,12 @@ Environment: production External operation ID: remote-op-us-east-001 External apply ID: remote-apply-us-east-001 - ~ orders: 🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨 Waiting for cutover - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - 🔄 eu-west — running table copy (orders-eu-west) External operation ID: remote-op-eu-west-001 External apply ID: remote-apply-eu-west-001 - ~ orders: 🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ 35.42% - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - • Rows: 42,500 / 120,000 · ETA: 4m 0s - - ⏳ ap-south — waiting for eu-west (orders-ap-south) - ~ orders: ⏳ Queued - ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; - - ESC to detach ``` diff --git a/pkg/api/progress_handlers.go b/pkg/api/progress_handlers.go index a5c8ae9a1..79e8a63fe 100644 --- a/pkg/api/progress_handlers.go +++ b/pkg/api/progress_handlers.go @@ -16,6 +16,7 @@ import ( "github.com/block/schemabot/pkg/caller" "github.com/block/schemabot/pkg/ddl" ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/state" "github.com/block/schemabot/pkg/storage" "github.com/block/schemabot/pkg/tern" @@ -245,7 +246,7 @@ func progressOperationResponseFromStorage(op *storage.ApplyOperation) *apitypes. return resp } -func (s *Service) progressOperationsForApply(ctx context.Context, apply *storage.Apply) ([]*apitypes.ProgressOperationResponse, map[int64]string, []*storage.ApplyOperation, error) { +func (s *Service) progressOperationsForApply(ctx context.Context, apply *storage.Apply) ([]*apitypes.ProgressOperationResponse, map[int64]routing.ExecutionTarget, []*storage.ApplyOperation, error) { if apply == nil { return nil, nil, nil, fmt.Errorf("apply is required") } @@ -253,8 +254,8 @@ func (s *Service) progressOperationsForApply(ctx context.Context, apply *storage if err != nil { return nil, nil, nil, fmt.Errorf("list apply operations for apply %d (%s): %w", apply.ID, apply.ApplyIdentifier, err) } - responses, deploymentByOperationID := progressOperationsFromRows(ops) - return responses, deploymentByOperationID, ops, nil + responses, memberByOperationID := progressOperationsFromRows(ops) + return responses, memberByOperationID, ops, nil } // resolveReleaseLatch best-effort reports whether a paused rollout has been @@ -273,30 +274,32 @@ func (s *Service) resolveReleaseLatch(ctx context.Context, apply *storage.Apply, } // progressOperationsFromRows projects already-fetched operation rows into the -// API response shape and the operation-id→deployment map. Keeping the +// API response shape and the operation-id→member map. Keeping the // transformation separate from the storage read lets a single ListByApply -// result feed both multi-operation detection and per-deployment enrichment on -// the polled progress path. -func progressOperationsFromRows(ops []*storage.ApplyOperation) ([]*apitypes.ProgressOperationResponse, map[int64]string) { +// result feed both multi-operation detection and per-member enrichment on the +// polled progress path. The map carries the whole routing pair rather than the +// deployment alone, because a deployment can address several targets and a task +// attributed to the deployment would not say which of them ran it. +func progressOperationsFromRows(ops []*storage.ApplyOperation) ([]*apitypes.ProgressOperationResponse, map[int64]routing.ExecutionTarget) { responses := make([]*apitypes.ProgressOperationResponse, 0, len(ops)) - deploymentByOperationID := make(map[int64]string, len(ops)) + memberByOperationID := make(map[int64]routing.ExecutionTarget, len(ops)) for _, op := range ops { responses = append(responses, progressOperationResponseFromStorage(op)) - deploymentByOperationID[op.ID] = op.Deployment + memberByOperationID[op.ID] = routing.ExecutionTarget{Deployment: op.Deployment, Target: op.Target} } - return responses, deploymentByOperationID + return responses, memberByOperationID } // bestEffortProgressOperations loads the apply's operation rows once and // returns every projection the storage-served progress response needs from -// them: the API operation entries, the operation-id→deployment map, the raw -// rows (for the stored engine metadata overlay), and the release latch. -func (s *Service) bestEffortProgressOperations(ctx context.Context, apply *storage.Apply) ([]*apitypes.ProgressOperationResponse, map[int64]string, []*storage.ApplyOperation, bool) { +// them: the API operation entries, the operation-id→member map, the raw rows +// (for the stored engine metadata overlay), and the release latch. +func (s *Service) bestEffortProgressOperations(ctx context.Context, apply *storage.Apply) ([]*apitypes.ProgressOperationResponse, map[int64]routing.ExecutionTarget, []*storage.ApplyOperation, bool) { if apply == nil { s.logger.Warn("progress response will omit per-deployment operations: apply is nil") return nil, nil, nil, false } - operations, deploymentByOperationID, ops, err := s.progressOperationsForApply(ctx, apply) + operations, memberByOperationID, ops, err := s.progressOperationsForApply(ctx, apply) if err != nil { // Operation rows are observability enrichment, not an apply safety gate. // Serve progress without the enrichment and log the storage uncertainty. @@ -305,7 +308,7 @@ func (s *Service) bestEffortProgressOperations(ctx context.Context, apply *stora "error", err)...) return nil, nil, nil, false } - return operations, deploymentByOperationID, ops, s.resolveReleaseLatch(ctx, apply, ops) + return operations, memberByOperationID, ops, s.resolveReleaseLatch(ctx, apply, ops) } // handleProgressByApplyID handles GET /api/progress/apply/{apply_id} requests. @@ -1166,7 +1169,7 @@ func (s *Service) progressFromLocalStorage(ctx context.Context, apply *storage.A } overlayApplyOptions(httpResp, apply) setRevertSkippedMetadata(httpResp, apply) - operations, deploymentByOperationID, ops, released := s.bestEffortProgressOperations(ctx, apply) + operations, memberByOperationID, ops, released := s.bestEffortProgressOperations(ctx, apply) httpResp.Operations = operations httpResp.Released = released overlayStoredDisplayMetadata(httpResp, apply, ops) @@ -1189,8 +1192,9 @@ func (s *Service) progressFromLocalStorage(ctx context.Context, apply *storage.A TaskID: task.TaskIdentifier, } if task.ApplyOperationID != nil { - if deployment, ok := deploymentByOperationID[*task.ApplyOperationID]; ok { - tpr.Deployment = deployment + if member, ok := memberByOperationID[*task.ApplyOperationID]; ok { + tpr.Deployment = member.Deployment + tpr.Target = member.Target } } if task.StartedAt != nil { diff --git a/pkg/api/progress_handlers_test.go b/pkg/api/progress_handlers_test.go index b44c8ec01..b192e69b1 100644 --- a/pkg/api/progress_handlers_test.go +++ b/pkg/api/progress_handlers_test.go @@ -10,6 +10,7 @@ import ( "github.com/block/schemabot/pkg/apitypes" ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/state" "github.com/block/schemabot/pkg/storage" ) @@ -186,3 +187,39 @@ func TestProgressOperationsCarryOperationKey(t *testing.T) { assert.Equal(t, "commerce/-80/users", decoded.Operations[0].OperationKey) assert.Equal(t, "commerce/80-/users", decoded.Operations[1].OperationKey) } + +// The operation-id map that attributes tasks to rollout members carries the +// whole routing pair. One deployment can address several targets, each copying +// the same tables against its own schema, so a task attributed to the deployment +// alone would not say which member's progress it reports. +func TestProgressOperationsMapCarriesTheRoutingPair(t *testing.T) { + ops := []*storage.ApplyOperation{ + {ID: 1, Deployment: "primary", Target: "testapp-001", State: state.Apply.Running}, + {ID: 2, Deployment: "primary", Target: "testapp-002", State: state.Apply.Pending}, + {ID: 3, Deployment: "eu-west", Target: "orders-eu", State: state.Apply.Pending}, + } + + _, memberByOperationID := progressOperationsFromRows(ops) + require.Len(t, memberByOperationID, 3) + assert.Equal(t, routing.ExecutionTarget{Deployment: "primary", Target: "testapp-001"}, memberByOperationID[1]) + assert.Equal(t, routing.ExecutionTarget{Deployment: "primary", Target: "testapp-002"}, memberByOperationID[2]) + assert.Equal(t, routing.ExecutionTarget{Deployment: "eu-west", Target: "orders-eu"}, memberByOperationID[3]) +} + +// A table's target survives the API's JSON encoding, so the CLI can scope each +// member's section to the copies that member actually ran. +func TestTableProgressResponseEncodesTarget(t *testing.T) { + encoded, err := json.Marshal(&apitypes.ProgressResponse{ + State: state.Apply.Running, + Tables: []*apitypes.TableProgressResponse{ + {TableName: "users", Deployment: "primary", Target: "testapp-002", Status: state.Task.Running}, + }, + }) + require.NoError(t, err) + assert.Contains(t, string(encoded), `"target":"testapp-002"`) + + var decoded apitypes.ProgressResponse + require.NoError(t, json.Unmarshal(encoded, &decoded)) + require.Len(t, decoded.Tables, 1) + assert.Equal(t, "testapp-002", decoded.Tables[0].Target) +} diff --git a/pkg/apitypes/apitypes.go b/pkg/apitypes/apitypes.go index 1b0f10f39..11b8ecdb7 100644 --- a/pkg/apitypes/apitypes.go +++ b/pkg/apitypes/apitypes.go @@ -1132,7 +1132,12 @@ type TableProgressResponse struct { DDL string `json:"ddl"` // Deployment attributes this table/task to a deployment in a multi-deployment apply. // Empty for single-deployment applies. - Deployment string `json:"deployment,omitempty"` + Deployment string `json:"deployment,omitempty"` + // Target attributes this table/task to a target within that deployment. One + // deployment can address several targets, each running its own copy of the + // change, so the deployment alone does not say which member's progress this + // row reports. + Target string `json:"target,omitempty"` Keyspace string `json:"keyspace,omitempty"` ChangeType string `json:"change_type,omitempty"` // create, alter, drop Status string `json:"status"` diff --git a/pkg/cmd/commands/watch_tui_test.go b/pkg/cmd/commands/watch_tui_test.go index de6a96282..41af90f84 100644 --- a/pkg/cmd/commands/watch_tui_test.go +++ b/pkg/cmd/commands/watch_tui_test.go @@ -388,9 +388,9 @@ func multiDeploymentTUITestProgress() apitypes.ProgressResponse { {Deployment: "ap-south", Target: "orders-ap-south", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, }, Tables: []*apitypes.TableProgressResponse{ - {Deployment: "us-east", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, - {Deployment: "eu-west", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD INDEX `idx_orders_source` (`source`)", Status: state.Task.Failed}, - {Deployment: "ap-south", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, + {Deployment: "us-east", Target: "orders-us-east", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, + {Deployment: "eu-west", Target: "orders-eu-west", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD INDEX `idx_orders_source` (`source`)", Status: state.Task.Failed}, + {Deployment: "ap-south", Target: "orders-ap-south", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, }, } } @@ -724,3 +724,32 @@ func TestGetProgress_ServerReturns500_CLIReturnsError(t *testing.T) { require.Error(t, err) assert.Contains(t, err.Error(), "500") } + +// Two targets of one deployment each copy the same tables against their own +// schema. The watch view lists each member's own copies under that member, so a +// deployment addressing several targets does not show every copy twice. +func TestWatchModel_MultiTargetSectionsScopeTablesToTheirMember(t *testing.T) { + m := NewWatchModel("http://localhost:8080", "testapp", "production", false) + m.applyID = "apply-multi-target" + m.state = state.Apply.Running + m.initialized = true + m.operations = []templates.ProgressOperation{ + {Deployment: "primary", Target: "testapp-001", State: state.ApplyOperation.Completed, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + {Deployment: "primary", Target: "testapp-002", State: state.ApplyOperation.Running, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + } + m.tables = []templates.TableProgress{ + {Deployment: "primary", Target: "testapp-001", TableName: "users_001", ChangeType: "alter", Status: state.Task.Completed}, + {Deployment: "primary", Target: "testapp-002", TableName: "users_002", ChangeType: "alter", Status: state.Task.Running}, + } + + view := m.View() + + assertContainsInOrder(t, view, + "primary/testapp-001", + "users_001", + "primary/testapp-002", + "users_002", + ) + assert.Equal(t, 1, strings.Count(view, "users_001")) + assert.Equal(t, 1, strings.Count(view, "users_002")) +} diff --git a/pkg/cmd/commands/watch_tui_view_multi.go b/pkg/cmd/commands/watch_tui_view_multi.go index d629e6d65..3a93a92be 100644 --- a/pkg/cmd/commands/watch_tui_view_multi.go +++ b/pkg/cmd/commands/watch_tui_view_multi.go @@ -110,7 +110,7 @@ func (m WatchModel) writeDeploymentSection(b *strings.Builder, deployment presen fmt.Fprintf(b, " %s\n", errStyle.Render(deployment.Error)) } - tables := tablesForDeployment(m.tables, deployment.Deployment) + tables := tablesForMember(m.tables, deployment.Deployment, deployment.Target) if len(tables) > 0 && !state.IsSetupPhase(m.state) { sortTablesByProgress(tables) m.renderTables(b, tables) @@ -121,11 +121,22 @@ func (m WatchModel) writeDeploymentSection(b *strings.Builder, deployment presen // externalIDForTUIMember resolves the external apply ID shown in a section: the // section's own operation when set, falling back to a sibling's — a keyed // apply's operations share one data-plane apply, so an operation that has not -// dispatched yet still shows it. +// dispatched yet still shows it. The fallback applies only while the deployment +// addresses a single target, so it never shows one member the ID of another. func externalIDForTUIMember(ops []templates.ProgressOperation, op templates.ProgressOperation) string { if op.ExternalID != "" { return op.ExternalID } + target := "" + for _, sibling := range ops { + if sibling.Deployment != op.Deployment || sibling.Target == "" { + continue + } + if target != "" && target != sibling.Target { + return "" + } + target = sibling.Target + } for _, sibling := range ops { if sibling.Deployment == op.Deployment && sibling.ExternalID != "" { return sibling.ExternalID @@ -134,14 +145,18 @@ func externalIDForTUIMember(ops []templates.ProgressOperation, op templates.Prog return "" } -func tablesForDeployment(tables []templates.TableProgress, deployment string) []templates.TableProgress { - deploymentTables := make([]templates.TableProgress, 0, len(tables)) +// tablesForMember selects the tables copied by one rollout member. Both halves +// of the routing pair are matched: two targets of one deployment each copy the +// same tables, and matching the deployment alone would list both members' +// copies under each of them. +func tablesForMember(tables []templates.TableProgress, deployment, target string) []templates.TableProgress { + memberTables := make([]templates.TableProgress, 0, len(tables)) for _, table := range tables { - if table.Deployment == deployment && table.TableName != "" { - deploymentTables = append(deploymentTables, table) + if table.Deployment == deployment && table.Target == target && table.TableName != "" { + memberTables = append(memberTables, table) } } - return deploymentTables + return memberTables } func (m WatchModel) writeMultiDeploymentFooter(b *strings.Builder, model presentation.Apply) { diff --git a/pkg/cmd/internal/templates/progress_multi.go b/pkg/cmd/internal/templates/progress_multi.go index 1a61e054f..936d23f0c 100644 --- a/pkg/cmd/internal/templates/progress_multi.go +++ b/pkg/cmd/internal/templates/progress_multi.go @@ -138,7 +138,7 @@ func writeDeploymentProgressSection(deployment presentation.Deployment, op Progr fmt.Printf(" %s%s%s\n", ANSIRed, deployment.Error, ANSIReset) } - tables := activeTablesForDeployment(data.Tables, deployment.Deployment) + tables := activeTablesForMember(data.Tables, deployment.Deployment, deployment.Target) if len(tables) > 0 && !state.IsSetupPhase(data.State) { sortActiveTables(tables) if hasTableNamespaces(tables) { @@ -153,30 +153,48 @@ func writeDeploymentProgressSection(deployment presentation.Deployment, op Progr fmt.Println() } +// deploymentTarget reports the one target every operation of op's deployment +// addresses, and whether there is exactly one. It is what decides when an +// operation that has not recorded its own routing may inherit a sibling's: a +// deployment addressing several targets runs a separate data-plane apply per +// target, so nothing about one member can be read off another. +func deploymentTarget(op ProgressOperation, ops []ProgressOperation) (target string, single bool) { + for _, sibling := range ops { + if sibling.Deployment != op.Deployment || sibling.Target == "" { + continue + } + if target != "" && target != sibling.Target { + return "", false + } + target = sibling.Target + } + return target, true +} + // sectionTarget resolves the target shown in a section header: the section's -// own operation when set, falling back to any same-deployment sibling — the -// target is deployment-level routing, identical across a keyed apply's -// operations. +// own operation when set, otherwise its deployment's when the deployment +// addresses a single one. A deployment addressing several shows no target +// rather than another member's. func sectionTarget(op ProgressOperation, ops []ProgressOperation) string { if op.Target != "" { return op.Target } - for _, sibling := range ops { - if sibling.Deployment == op.Deployment && sibling.Target != "" { - return sibling.Target - } - } - return "" + target, _ := deploymentTarget(op, ops) + return target } // sectionExternalID resolves the external apply ID shown in a section: the -// section's own operation when set, falling back to any same-deployment -// sibling — a keyed apply's operations share one data-plane apply, so a -// not-yet-dispatched operation still shows the deployment's shared ID. +// section's own operation when set, falling back to a sibling's — a keyed +// apply's operations share one data-plane apply, so an operation that has not +// dispatched yet still shows it. The fallback applies only while the deployment +// addresses a single target, so it never shows one member the ID of another. func sectionExternalID(op ProgressOperation, ops []ProgressOperation) string { if op.ExternalID != "" { return op.ExternalID } + if _, single := deploymentTarget(op, ops); !single { + return "" + } for _, sibling := range ops { if sibling.Deployment == op.Deployment && sibling.ExternalID != "" { return sibling.ExternalID @@ -185,10 +203,14 @@ func sectionExternalID(op ProgressOperation, ops []ProgressOperation) string { return "" } -func activeTablesForDeployment(tables []TableProgress, deployment string) []TableProgress { +// activeTablesForMember selects the tables copied by one rollout member. Both +// halves of the routing pair are matched: two targets of one deployment each +// copy the same tables, and matching the deployment alone would list both +// members' copies under each of them. +func activeTablesForMember(tables []TableProgress, deployment, target string) []TableProgress { activeTables := make([]TableProgress, 0, len(tables)) for _, table := range tables { - if table.Deployment == deployment && table.TableName != "" { + if table.Deployment == deployment && table.Target == target && table.TableName != "" { activeTables = append(activeTables, table) } } diff --git a/pkg/cmd/internal/templates/progress_multi_test.go b/pkg/cmd/internal/templates/progress_multi_test.go index 96f4b9e3c..c3f33be6c 100644 --- a/pkg/cmd/internal/templates/progress_multi_test.go +++ b/pkg/cmd/internal/templates/progress_multi_test.go @@ -23,9 +23,9 @@ func TestWriteProgressMultiDeploymentRendersAggregateAndSections(t *testing.T) { {Deployment: "region-c", Target: "orders-c", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, }, Tables: []TableProgress{ - {Deployment: "region-a", TableName: "users_a", ChangeType: "alter", DDL: "ALTER TABLE `users_a` ADD COLUMN `region` varchar(20)", Status: state.Task.Completed}, - {Deployment: "region-b", TableName: "users_b", ChangeType: "alter", DDL: "ALTER TABLE `users_b` ADD COLUMN `region` varchar(20)", Status: state.Task.Failed}, - {Deployment: "region-c", TableName: "users_c", ChangeType: "alter", DDL: "ALTER TABLE `users_c` ADD COLUMN `region` varchar(20)", Status: state.Task.Running}, + {Deployment: "region-a", Target: "orders-a", TableName: "users_a", ChangeType: "alter", DDL: "ALTER TABLE `users_a` ADD COLUMN `region` varchar(20)", Status: state.Task.Completed}, + {Deployment: "region-b", Target: "orders-b", TableName: "users_b", ChangeType: "alter", DDL: "ALTER TABLE `users_b` ADD COLUMN `region` varchar(20)", Status: state.Task.Failed}, + {Deployment: "region-c", Target: "orders-c", TableName: "users_c", ChangeType: "alter", DDL: "ALTER TABLE `users_c` ADD COLUMN `region` varchar(20)", Status: state.Task.Running}, }, }) }) @@ -91,8 +91,8 @@ func TestWriteProgressMultiDeploymentContinueFailureShowsRunningDegraded(t *test {Deployment: "region-b", Target: "orders-b", State: state.ApplyOperation.Running, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureContinue}, }, Tables: []TableProgress{ - {Deployment: "region-a", TableName: "users_a", ChangeType: "alter", DDL: "ALTER TABLE `users_a` ADD COLUMN `region` varchar(20)", Status: state.Task.Failed}, - {Deployment: "region-b", TableName: "users_b", ChangeType: "alter", DDL: "ALTER TABLE `users_b` ADD COLUMN `region` varchar(20)", Status: state.Task.Running}, + {Deployment: "region-a", Target: "orders-a", TableName: "users_a", ChangeType: "alter", DDL: "ALTER TABLE `users_a` ADD COLUMN `region` varchar(20)", Status: state.Task.Failed}, + {Deployment: "region-b", Target: "orders-b", TableName: "users_b", ChangeType: "alter", DDL: "ALTER TABLE `users_b` ADD COLUMN `region` varchar(20)", Status: state.Task.Running}, }, }) }) @@ -191,3 +191,38 @@ func TestWriteProgressMultiTargetSectionsNameEachMember(t *testing.T) { assert.NotContains(t, output, "primary/testapp-001 — completed (testapp-001)") assert.NotContains(t, output, "primary/testapp-002 — running table copy (testapp-002)") } + +// Two targets of one deployment each run their own copy of the change against +// their own schema. Each member's section shows only the tables its own target +// copied and only its own data-plane identifiers — nothing is read off the +// sibling member it shares a deployment with. +func TestWriteProgressMultiTargetSectionsAreMemberScoped(t *testing.T) { + output := captureStdout(t, func() { + WriteProgress(ProgressData{ + ApplyID: "apply-multi-target", + Environment: "staging", + State: state.Apply.Running, + Operations: []ProgressOperation{ + {Deployment: "primary", Target: "testapp-001", ExternalID: "remote-apply-001", ExternalOperationID: "remote-op-001", State: state.ApplyOperation.Completed, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + {Deployment: "primary", Target: "testapp-002", State: state.ApplyOperation.Running, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, + }, + Tables: []TableProgress{ + {Deployment: "primary", Target: "testapp-001", TableName: "users_001", ChangeType: "alter", DDL: "ALTER TABLE `users` ADD COLUMN `region` varchar(20)", Status: state.Task.Completed}, + {Deployment: "primary", Target: "testapp-002", TableName: "users_002", ChangeType: "alter", DDL: "ALTER TABLE `users` ADD COLUMN `region` varchar(20)", Status: state.Task.Running}, + }, + }) + }) + + // The dispatched member's identifiers stay with it: the other member runs a + // separate data-plane apply, so it inherits neither. + assert.Equal(t, 1, strings.Count(output, "External apply ID: remote-apply-001"), + "a sibling target must not inherit another member's external apply ID") + assert.Equal(t, 1, strings.Count(output, "External operation ID: remote-op-001")) + + // Each member lists only the tables its own target copied. + assertLess(t, output, "primary/testapp-001", "users_001") + assertLess(t, output, "users_001", "primary/testapp-002") + assertLess(t, output, "primary/testapp-002", "users_002") + assert.Equal(t, 1, strings.Count(output, "users_001")) + assert.Equal(t, 1, strings.Count(output, "users_002")) +} diff --git a/pkg/cmd/internal/templates/progress_parse.go b/pkg/cmd/internal/templates/progress_parse.go index f834f4fd7..b984a58e3 100644 --- a/pkg/cmd/internal/templates/progress_parse.go +++ b/pkg/cmd/internal/templates/progress_parse.go @@ -62,8 +62,12 @@ type ProgressOperation struct { // TableProgress represents progress for a single table schema change. type TableProgress struct { - TableName string - Deployment string + TableName string + Deployment string + // Target is the address within Deployment this table's copy ran against. One + // deployment can address several targets, each copying the same table + // separately, so a section scoped to a deployment alone would merge them. + Target string Namespace string // Keyspace (Vitess) or schema name (MySQL) Dialect schema.Dialect ChangeType string // create, alter, drop @@ -199,6 +203,7 @@ func ParseProgressResponse(result *apitypes.ProgressResponse) ProgressData { tp := TableProgress{ TableName: tbl.TableName, Deployment: tbl.Deployment, + Target: tbl.Target, Namespace: tbl.Keyspace, Dialect: dialect, ChangeType: tbl.ChangeType, From 550f1cad0cbeea3507880333c59dfb72e660b854 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 15:30:02 -0400 Subject: [PATCH 17/34] feat(storage): name the operation key delimiter and cut a target prefix An operation key names its target only when that target's deployment addresses more than one, so a reader meets both shapes. CutTargetPrefix matches the prefix against the operation's own target and reports whether it cut one, leaving the caller to parse unqualified first so a key that already reads whole is never mistaken for a qualified one. Co-Authored-By: Claude Opus 5 --- pkg/storage/types.go | 27 ++++++++++++++++-- pkg/storage/types_test.go | 58 +++++++++++++++++++++++++++++++++++++++ 2 files changed, 83 insertions(+), 2 deletions(-) diff --git a/pkg/storage/types.go b/pkg/storage/types.go index dc919694b..5f3976d27 100644 --- a/pkg/storage/types.go +++ b/pkg/storage/types.go @@ -320,6 +320,11 @@ const ( ApplyOperationKindGroupFinalizer = "group_finalizer" ) +// OperationKeyDelimiter separates the components of an operation key. A +// component containing it would make the key ambiguous to split, so producers +// refuse the delimiter inside a component rather than escaping it. +const OperationKeyDelimiter = "/" + // ShardOperationKey builds the operation key for one shard's work on one table // ("//"). It is the canonical key for shard-scoped // work operations: the control plane's sharded fan-out stamps it on each @@ -327,7 +332,7 @@ const ( // plane's operation row, and the task loaders match shard-tagged task rows // against it to distinguish drive tasks from reflected per-shard progress rows. func ShardOperationKey(namespace, shard, table string) string { - return namespace + "/" + shard + "/" + table + return namespace + OperationKeyDelimiter + shard + OperationKeyDelimiter + table } // TargetOperationKey builds the operation key for one target's work when a @@ -345,7 +350,25 @@ func TargetOperationKey(target, scopedKey string) string { if scopedKey == "" { return target } - return target + "/" + scopedKey + return target + OperationKeyDelimiter + scopedKey +} + +// CutTargetPrefix removes an operation key's leading target component, reporting +// whether the key carried one. +// +// A key names its target only when that target's deployment addresses more than +// one, so a reader meets both shapes and cannot tell them apart by inspection: a +// target named after the namespace it holds produces the same leading component +// either way. The operation row carries its target, so the prefix is matched +// rather than guessed — but matching it is not proof that the key is qualified, +// which is why this reports what it did instead of deciding. Parse the key +// unqualified first and only fall back to the cut form, so a shape that already +// reads as a whole key is never mistaken for a qualified one. +func CutTargetPrefix(target, operationKey string) (scopedKey string, qualified bool) { + if target == "" { + return operationKey, false + } + return strings.CutPrefix(operationKey, target+OperationKeyDelimiter) } // PlanIDForOperation resolves which plan an operation executes: its own when it diff --git a/pkg/storage/types_test.go b/pkg/storage/types_test.go index 4c4f0a465..08d5e0bf0 100644 --- a/pkg/storage/types_test.go +++ b/pkg/storage/types_test.go @@ -413,6 +413,64 @@ func TestTargetOperationKey(t *testing.T) { } } +// TestCutTargetPrefix covers how a reader recovers the scoped part of an +// operation key. Both key shapes are live at once, so the cut reports what it +// did rather than deciding, and a target named after a component of an +// unqualified key is the case that proves the caller must parse unqualified +// first. +func TestCutTargetPrefix(t *testing.T) { + cases := []struct { + name string + target string + operationKey string + wantScoped string + wantQualified bool + }{ + { + name: "a qualified key cuts down to its scoped part", + target: "orders-002", + operationKey: TargetOperationKey("orders-002", ShardOperationKey("main", "-80", "customers")), + wantScoped: "main/-80/customers", + wantQualified: true, + }, + { + name: "an unqualified key is returned whole", + target: "orders-002", + operationKey: ShardOperationKey("main", "-80", "customers"), + wantScoped: "main/-80/customers", + wantQualified: false, + }, + { + name: "another member's key is not this member's prefix", + target: "orders-003", + operationKey: TargetOperationKey("orders-002", ShardOperationKey("main", "-80", "customers")), + wantScoped: "orders-002/main/-80/customers", + wantQualified: false, + }, + { + name: "an operation with no target cuts nothing", + target: "", + operationKey: ShardOperationKey("main", "-80", "customers"), + wantScoped: "main/-80/customers", + wantQualified: false, + }, + { + name: "a target named for its namespace cuts a real component", + target: "main", + operationKey: ShardOperationKey("main", "-80", "customers"), + wantScoped: "-80/customers", + wantQualified: true, + }, + } + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + scoped, qualified := CutTargetPrefix(tc.target, tc.operationKey) + assert.Equal(t, tc.wantScoped, scoped) + assert.Equal(t, tc.wantQualified, qualified) + }) + } +} + // TestPlanIDForOperation covers which plan a member executes: members planned // together share their apply's plan, a member planned against its own live // schema carries its own, and a member with neither is not executable and must From db192799283f15a08c814b54af1ab164fd7094db Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 15:32:20 -0400 Subject: [PATCH 18/34] fix(api): refuse a listed target that contains the operation key delimiter A deployment addressing several targets names each one in its members' operation keys. A target carrying the delimiter would write a key no reader can split back into the target it came from, and once such a key is written the ambiguity is in the data. Config validation is the last point at which it is still recoverable, so the name is refused there. Co-Authored-By: Claude Opus 5 --- docs/configuration.md | 1 + pkg/api/config.go | 10 ++++++++++ pkg/api/config_test.go | 19 +++++++++++++++++++ 3 files changed, 30 insertions(+) diff --git a/docs/configuration.md b/docs/configuration.md index 288d792c8..f7288bebb 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -306,6 +306,7 @@ Rules: - `targets` is mutually exclusive with `target` at the same level, and with a local `dsn` / `dsn_from`. - An environment-level `targets` list is mutually exclusive with an environment-level `deployments` map, the same way an environment-level `target` is. A `targets` list inside a `deployments` entry is how the two combine. - The list MUST contain at least one entry, and no entry may be empty. +- No entry may contain `/`. A deployment addressing several targets names each one in its members' operation keys, and `/` separates a key's components. - One deployment may not list the same target twice. A rollout member is identified by its deployment and target together, so the same target under two different deployments is two distinct members and is allowed. - Members resolve deployments outermost: every target of the first deployment, then every target of the next. diff --git a/pkg/api/config.go b/pkg/api/config.go index 8006583eb..b65f795ef 100644 --- a/pkg/api/config.go +++ b/pkg/api/config.go @@ -1654,6 +1654,13 @@ func (c EnvironmentConfig) validateMultiTargetSupport(context, databaseType stri // alongside a target is reported as the conflict it is rather than as an empty // list. An entry that names neither returns an empty list without an error, // leaving the caller to report the missing target in its own terms. +// +// A listed target may not contain the operation key delimiter. A deployment +// addressing several targets names each one in its members' operation keys, and +// a target carrying the delimiter would write a key no reader can split back +// into the target it came from. Refusing the name is the only point at which +// that is still recoverable: once such a key is written, the ambiguity is in the +// data. func resolveTargetList(what, target string, targets []string) ([]string, error) { if target != "" && targets != nil { return nil, fmt.Errorf("%s cannot configure both target and targets", what) @@ -1672,6 +1679,9 @@ func resolveTargetList(what, target string, targets []string) ([]string, error) if t == "" { return nil, fmt.Errorf("%s targets entry %d is empty", what, i) } + if strings.Contains(t, storage.OperationKeyDelimiter) { + return nil, fmt.Errorf("%s targets entry %d %q contains reserved delimiter %q; a target's name leads the operation keys of the members that address it, so it cannot contain the character that separates their components", what, i, t, storage.OperationKeyDelimiter) + } if seen[t] { return nil, fmt.Errorf("%s lists target %q more than once; a rollout member is identified by its deployment and target together, so one deployment cannot address the same target twice", what, t) } diff --git a/pkg/api/config_test.go b/pkg/api/config_test.go index 66805651e..7af3ef464 100644 --- a/pkg/api/config_test.go +++ b/pkg/api/config_test.go @@ -2304,6 +2304,15 @@ func TestServerConfig_DeploymentsMapValidation(t *testing.T) { tern: baseTern, wantErrSub: `lists target "payments-001" more than once`, }, + { + name: "environment target containing the operation key delimiter is rejected", + envConfig: EnvironmentConfig{ + Deployment: "payments-a", + Targets: []string{"payments-001", "payments/002"}, + }, + tern: baseTern, + wantErrSub: `targets entry 1 "payments/002" contains reserved delimiter "/"`, + }, { name: "environment targets without a deployment is rejected", envConfig: EnvironmentConfig{ @@ -2352,6 +2361,16 @@ func TestServerConfig_DeploymentsMapValidation(t *testing.T) { tern: baseTern, wantErrSub: `deployment "payments-a" lists target "payments-001" more than once`, }, + { + name: "deployment target containing the operation key delimiter is rejected", + envConfig: EnvironmentConfig{ + Deployments: map[string]DeploymentTarget{ + "payments-a": {Targets: []string{"payments/001", "payments-002"}}, + }, + }, + tern: baseTern, + wantErrSub: `deployment "payments-a" targets entry 0 "payments/001" contains reserved delimiter "/"`, + }, { name: "the same target under different deployments is accepted", envConfig: EnvironmentConfig{ From 7412afa03ebe83e71e86e91fea8fc0d8ce076476 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 15:33:33 -0400 Subject: [PATCH 19/34] refactor(api): resolve the planning contract through UsesTargetsList MemberPlanningFor re-derived the targets-list predicate instead of asking the helper that is meant to own it. A routing spelling added to the helper would then select independent planning in validation and routing but mirrored planning here, which is the one disagreement this contract cannot survive. Co-Authored-By: Claude Opus 5 --- pkg/api/config.go | 7 +------ 1 file changed, 1 insertion(+), 6 deletions(-) diff --git a/pkg/api/config.go b/pkg/api/config.go index 9079c4932..f2cbec125 100644 --- a/pkg/api/config.go +++ b/pkg/api/config.go @@ -2783,14 +2783,9 @@ func (c *ServerConfig) MemberPlanningFor(database, environment string) (MemberPl if !ok { return PlanMirrored, &EnvironmentNotConfiguredError{Database: database, Environment: environment} } - if envConfig.Targets != nil { + if envConfig.UsesTargetsList() { return PlanIndependent, nil } - for _, dt := range envConfig.Deployments { - if dt.Targets != nil { - return PlanIndependent, nil - } - } return PlanMirrored, nil } From 6f9ba19b54f5d3a98b9068318759b47b7e9a40cc Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 15:41:14 -0400 Subject: [PATCH 20/34] fix(api): bind each member plan to the review round it was produced in MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A commit can be planned more than once — a re-plan, or two deliveries racing — and each round stores its own plan per member with the same route and the same head SHA. Nothing durable said which round a member plan belonged to, so an apply created from one round's reviewed plan could pair its members with another round's plans and dispatch DDL the operator never saw. Member plans now carry the reviewed plan's identifier, and a member plan that cannot be attributed to a round is not stored at all. The member diff also carries the request contract the primary's does: a namespace the caller withheld is excluded from the desired state, not absent from it, and the grouping choice is always stated. The execution-mode vocabulary moves to the shared plan writer, so it is enforced once for every plan row whichever RPC produced it. Co-Authored-By: Claude Opus 5 --- pkg/api/plan_deployment_diffs.go | 8 ++++ pkg/api/plan_handlers.go | 65 +++++++++++++++++-------- pkg/api/plan_member_plans.go | 21 ++++++-- pkg/api/plan_member_plans_test.go | 66 ++++++++++++++++++++++++++ pkg/api/plan_review_drift.go | 2 +- pkg/schema/mysql/plans.sql | 1 + pkg/schema/postgres/plans.sql | 1 + pkg/storage/internal/sqlstore/plans.go | 15 ++++-- pkg/storage/storage.go | 4 ++ pkg/storage/storagetest/plans.go | 61 ++++++++++++++++++++++++ pkg/storage/types.go | 11 +++++ 11 files changed, 224 insertions(+), 31 deletions(-) diff --git a/pkg/api/plan_deployment_diffs.go b/pkg/api/plan_deployment_diffs.go index d0c7b9e36..9e592e965 100644 --- a/pkg/api/plan_deployment_diffs.go +++ b/pkg/api/plan_deployment_diffs.go @@ -174,6 +174,14 @@ func (s *Service) planDeploymentDiff(ctx context.Context, req PlanRequest, targe Environment: req.Environment, Target: target.Target, SchemaPath: trustedSchemaPath, + // The desired state is partial in exactly the way the primary's is: a + // namespace the caller withheld is excluded, not absent. Without this + // the data plane reads the omission as intent to remove and plans DROPs + // for namespaces the configuration excluded on purpose. + IgnoredNamespaces: req.IgnoredNamespaces, + // Always stated, never left absent: absence tells the data plane the + // caller predates the grouping choice, and this caller has made one. + GroupedExecution: new(req.GroupedExecution), } if req.PullRequest != nil { ternReq.PullRequest = *req.PullRequest diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index ac92907f3..edf64f5f9 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -828,18 +828,31 @@ func (s *Service) ExecutePlanProto(ctx context.Context, req PlanRequest) (*ternv return resp, planResp, nil } -// normalizeExecutionVerdicts validates every table change's execution-mode +// normalizeExecutionVerdicts normalizes a whole plan response, for the paths +// that hold one. A nil response is a no-op so callers that may not have reached +// a planner do not have to guard the call themselves. +func (s *Service) normalizeExecutionVerdicts(resp *ternv1.PlanResponse, database, deployment string) { + if resp == nil { + return + } + s.normalizePlanExecutionVerdicts(resp.Changes, resp.Shards, database, deployment) +} + +// normalizePlanExecutionVerdicts validates every table change's execution-mode // verdict against its closed vocabulary — empty (executable), "blocked", or // "direct" — before the plan is stored or returned. The verdict crosses the // wire as a free-form string and everything except "blocked" executes as the // engine's default path downstream, so a value this build does not recognize // (a skewed remote planner, a newer plan contract) must fail closed rather // than run. -func (s *Service) normalizeExecutionVerdicts(resp *ternv1.PlanResponse, database, deployment string) { - if resp == nil { - return - } - for _, change := range resp.Changes { +// +// It takes the plan's parts rather than a response because not every path that +// stores a plan holds one: a member planned against its own live schema arrives +// from the non-persisting diff RPC, and its changes reach storage through the +// same shared writer. Normalizing there is what keeps the vocabulary enforced +// once for every plan row, whichever RPC produced it. +func (s *Service) normalizePlanExecutionVerdicts(changes []*ternv1.SchemaChange, shards []*ternv1.ShardPlan, database, deployment string) { + for _, change := range changes { if change == nil { continue } @@ -847,7 +860,7 @@ func (s *Service) normalizeExecutionVerdicts(resp *ternv1.PlanResponse, database s.normalizeExecutionVerdict(tc, database, deployment) } } - for _, shard := range resp.Shards { + for _, shard := range shards { if shard == nil { continue } @@ -880,10 +893,18 @@ func recognizedExecutionMode(mode string) bool { strings.EqualFold(mode, engine.ExecutionModeDirect) } +// storedPlanRoute is what a stored plan row is stamped with beyond the request: +// the member the plan was produced for, and — for a member planned against its +// own live schema — the reviewed plan it was produced alongside. type storedPlanRoute struct { DatabaseType string Deployment string Target string + + // PrimaryPlanIdentifier names the reviewed plan of this member's review + // round. Empty for the reviewed plan itself and for every plan of an + // environment whose members all run it. + PrimaryPlanIdentifier string } func (s *Service) storePlanResponse(ctx context.Context, req PlanRequest, resp *ternv1.PlanResponse, route storedPlanRoute) error { @@ -918,6 +939,7 @@ func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier if req.HeadSHA != nil { headSHA = *req.HeadSHA } + s.normalizePlanExecutionVerdicts(changes, shards, req.Database, route.Deployment) namespaces, err := protoChangesToNamespaces(changes, req.SchemaFiles) if err != nil { return fmt.Errorf("convert plan namespaces: %w", err) @@ -927,20 +949,21 @@ func (s *Service) storePlan(ctx context.Context, req PlanRequest, planIdentifier return fmt.Errorf("convert plan shards: %w", err) } storedPlan := &storage.Plan{ - PlanIdentifier: planIdentifier, - Database: req.Database, - DatabaseType: route.DatabaseType, - Deployment: route.Deployment, - Target: route.Target, - Repository: req.Repository, - PullRequest: prInt, - SchemaPath: trustedSchemaPath, - Environment: req.Environment, - SchemaFiles: protoToSchemaFiles(req.SchemaFiles), - Namespaces: namespaces, - Shards: storedShards, - HeadSHA: headSHA, - CreatedAt: time.Now(), + PlanIdentifier: planIdentifier, + Database: req.Database, + DatabaseType: route.DatabaseType, + Deployment: route.Deployment, + Target: route.Target, + Repository: req.Repository, + PullRequest: prInt, + SchemaPath: trustedSchemaPath, + Environment: req.Environment, + SchemaFiles: protoToSchemaFiles(req.SchemaFiles), + Namespaces: namespaces, + Shards: storedShards, + HeadSHA: headSHA, + PrimaryPlanIdentifier: route.PrimaryPlanIdentifier, + CreatedAt: time.Now(), } if _, err := s.storage.Plans().Create(ctx, storedPlan); err != nil && !errors.Is(err, storage.ErrPlanIDExists) { return fmt.Errorf("store plan %s: %w", planIdentifier, err) diff --git a/pkg/api/plan_member_plans.go b/pkg/api/plan_member_plans.go index 150bf5d46..c96f35dfa 100644 --- a/pkg/api/plan_member_plans.go +++ b/pkg/api/plan_member_plans.go @@ -25,13 +25,25 @@ import ( // review. An apply dispatches each member against its stored plan, so a member // with no stored plan has nothing to run; letting the rollup stay clean would // gate the PR on a member that could not have been applied. -func (s *Service) persistMemberPlans(ctx context.Context, req PlanRequest, planning MemberPlanning, diffs []DeploymentPlanDiff, rollup *PlanRollup) error { +// +// Every member plan is stamped with the reviewed plan's identifier, which is +// what durably binds it to this review round. A commit can be planned more than +// once — a re-plan, or two deliveries racing — and each round stores its own row +// per member with the same route and the same head SHA. Without the stamp an +// apply created from one round's reviewed plan could pair its members with +// another round's plans, dispatching DDL the operator never saw. +func (s *Service) persistMemberPlans(ctx context.Context, req PlanRequest, planning MemberPlanning, primaryPlanIdentifier string, diffs []DeploymentPlanDiff, rollup *PlanRollup) error { if planning != PlanIndependent { return nil } if len(diffs) != len(rollup.Entries) { return fmt.Errorf("persist member plans for %s/%s: %d member diffs for %d rollup entries", req.Database, req.Environment, len(diffs), len(rollup.Entries)) } + if primaryPlanIdentifier == "" { + // A member plan that cannot be attributed to a review round is exactly + // what the stamp exists to prevent, so it is not stored at all. + return fmt.Errorf("persist member plans for %s/%s: the reviewed plan has no identifier to bind member plans to", req.Database, req.Environment) + } // Index 0 is the primary, whose reviewed plan is already stored. for i := 1; i < len(rollup.Entries); i++ { @@ -50,9 +62,10 @@ func (s *Service) persistMemberPlans(ctx context.Context, req PlanRequest, plann // identifier and gets one minted here. planIdentifier := engine.NewPlanID() route := storedPlanRoute{ - DatabaseType: entry.DatabaseType, - Deployment: entry.Deployment, - Target: entry.Target, + DatabaseType: entry.DatabaseType, + Deployment: entry.Deployment, + Target: entry.Target, + PrimaryPlanIdentifier: primaryPlanIdentifier, } if err := s.storePlan(ctx, req, planIdentifier, diffs[i].Changes, diffs[i].Shards, route); err != nil { s.logger.Error("failed to store a rollout member's plan; the member will block the review because an apply would have no plan to run for it", diff --git a/pkg/api/plan_member_plans_test.go b/pkg/api/plan_member_plans_test.go index 497853c2f..530e77c6f 100644 --- a/pkg/api/plan_member_plans_test.go +++ b/pkg/api/plan_member_plans_test.go @@ -11,6 +11,7 @@ import ( "github.com/stretchr/testify/assert" "github.com/stretchr/testify/require" + "github.com/block/schemabot/pkg/engine" "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/storage" "github.com/block/schemabot/pkg/tern" @@ -123,6 +124,71 @@ func TestRollupReviewTimeDrift_IndependentMemberPlanIsPersisted(t *testing.T) { require.Len(t, stored.Namespaces["testapp"].Tables, 1) assert.Equal(t, "users", stored.Namespaces["testapp"].Tables[0].Table) assert.Contains(t, stored.Namespaces["testapp"].Tables[0].DDL, "ADD COLUMN `phone`") + assert.Equal(t, reviewed.PlanId, stored.PrimaryPlanIdentifier, + "a member's plan is bound to the reviewed plan it was produced alongside") +} + +// A commit can be planned more than once, and each round stores its own plan per +// member with the same route and the same head SHA. The reviewed plan's +// identifier is what tells the rounds apart, so a member plan that cannot be +// attributed to one is not stored at all: the member then has no plan for the +// round and blocks the review, rather than being paired with a plan from a round +// the operator never saw. +func TestRollupReviewTimeDrift_MemberPlanNeedsAReviewRoundToBindTo(t *testing.T) { + reviewed := reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)") + reviewed.PlanId = "" + plans := &recordingPlanStore{} + svc := multiTargetService(t, &mockTernClient{planDiffResp: alterUsersDiff("ALTER TABLE `users` ADD COLUMN `phone` varchar(32)")}, plans) + + _, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewed, multiTargetMember("testapp-001")) + require.Error(t, err) + assert.Contains(t, err.Error(), "no identifier to bind member plans to") + assert.Empty(t, plans.created) +} + +// A namespace the caller withheld per the config's ignore_namespaces is +// excluded from the desired state, not absent from it. A member planned against +// its own live schema is planned from the same partial desired state as the +// primary, or the data plane reads the omission as intent to remove and the +// member's stored plan carries DROPs for namespaces the configuration excludes. +func TestRollupReviewTimeDrift_MemberDiffCarriesTheRequestContract(t *testing.T) { + client := &mockTernClient{planDiffResp: alterUsersDiff("ALTER TABLE `users` ADD COLUMN `phone` varchar(32)")} + svc := multiTargetService(t, client, &recordingPlanStore{}) + + req := planDiffReq(t) + req.IgnoredNamespaces = []string{"legacy_reports"} + req.GroupedExecution = true + + _, err := svc.RollupReviewTimeDrift(t.Context(), req, reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), multiTargetMember("testapp-001")) + require.NoError(t, err) + + require.NotNil(t, client.planDiffReq) + assert.Equal(t, []string{"legacy_reports"}, client.planDiffReq.IgnoredNamespaces) + require.NotNil(t, client.planDiffReq.GroupedExecution, "the grouping choice is always stated, never left absent") + assert.True(t, *client.planDiffReq.GroupedExecution) +} + +// The execution-mode verdict crosses the wire as free-form text and everything +// except "blocked" runs on the engine's default path. A member's plan reaches +// storage from the diff RPC rather than the plan RPC, so the vocabulary is +// enforced where plan rows are written and an unrecognized verdict is persisted +// blocked rather than applyable. +func TestRollupReviewTimeDrift_MemberPlanBlocksUnrecognizedVerdict(t *testing.T) { + diff := alterUsersDiff("ALTER TABLE `users` ADD COLUMN `phone` varchar(32)") + diff.Changes[0].TableChanges[0].ExecutionMode = "future-mode" + diff.Changes[0].TableChanges[0].ModeReason = "untrusted planner reason" + plans := &recordingPlanStore{} + svc := multiTargetService(t, &mockTernClient{planDiffResp: diff}, plans) + + rollup, err := svc.RollupReviewTimeDrift(t.Context(), planDiffReq(t), reviewedUsersPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), multiTargetMember("testapp-001")) + require.NoError(t, err) + require.Len(t, rollup.Entries, 2) + + require.Len(t, plans.created, 1) + table := plans.created[0].Namespaces["testapp"].Tables[0] + assert.Equal(t, engine.ExecutionModeBlocked, table.ExecutionMode) + assert.Contains(t, table.ModeReason, `"future-mode"`) + assert.NotContains(t, table.ModeReason, "untrusted planner reason") } // A member whose plan cannot be stored has nothing an apply could run, so it is diff --git a/pkg/api/plan_review_drift.go b/pkg/api/plan_review_drift.go index 7976b9a3a..91140c8f1 100644 --- a/pkg/api/plan_review_drift.go +++ b/pkg/api/plan_review_drift.go @@ -53,7 +53,7 @@ func (s *Service) RollupReviewTimeDrift(ctx context.Context, req PlanRequest, pr // apply has no single plan that covers them. Storing them here, before the // rollup is reported, keeps "the review says this member is fine" and "this // member has a plan to run" from being separately true. - if err := s.persistMemberPlans(ctx, req, planning, diffs, &rollup); err != nil { + if err := s.persistMemberPlans(ctx, req, planning, primaryPlan.GetPlanId(), diffs, &rollup); err != nil { return PlanRollup{}, fmt.Errorf("persist member plans for %s/%s: %w", req.Database, req.Environment, err) } diff --git a/pkg/schema/mysql/plans.sql b/pkg/schema/mysql/plans.sql index 918822fb9..2dbaa3672 100644 --- a/pkg/schema/mysql/plans.sql +++ b/pkg/schema/mysql/plans.sql @@ -12,6 +12,7 @@ CREATE TABLE `plans` ( `schema_files` json NOT NULL, `plan_data` json NOT NULL, `head_sha` varchar(64) NOT NULL DEFAULT '', + `primary_plan_identifier` varchar(255) NOT NULL DEFAULT '', `created_at` datetime NOT NULL DEFAULT CURRENT_TIMESTAMP, PRIMARY KEY (`id`), UNIQUE KEY `idx_plan_identifier` (`plan_identifier`), diff --git a/pkg/schema/postgres/plans.sql b/pkg/schema/postgres/plans.sql index 82b799c55..3aad7a87c 100644 --- a/pkg/schema/postgres/plans.sql +++ b/pkg/schema/postgres/plans.sql @@ -12,6 +12,7 @@ CREATE TABLE plans ( schema_files jsonb NOT NULL, plan_data jsonb NOT NULL, head_sha varchar(64) NOT NULL DEFAULT '', + primary_plan_identifier varchar(255) NOT NULL DEFAULT '', created_at timestamp NOT NULL DEFAULT CURRENT_TIMESTAMP, PRIMARY KEY (id) ); diff --git a/pkg/storage/internal/sqlstore/plans.go b/pkg/storage/internal/sqlstore/plans.go index f4412de47..bff76846d 100644 --- a/pkg/storage/internal/sqlstore/plans.go +++ b/pkg/storage/internal/sqlstore/plans.go @@ -17,7 +17,7 @@ import ( // planColumns lists all columns for SELECT queries. const planColumns = `id, plan_identifier, database_name, database_type, - deployment, target, repository, pull_request, schema_path, environment, schema_files, plan_data, head_sha, created_at` + deployment, target, repository, pull_request, schema_path, environment, schema_files, plan_data, head_sha, primary_plan_identifier, created_at` // planListColumns matches planColumns except schema_files, which is replaced // by a NULL placeholder so the scan shape stays identical. schema_files holds @@ -25,7 +25,7 @@ const planColumns = `id, plan_identifier, database_name, database_type, // listings never need it, so List leaves SchemaFiles unhydrated rather than // transferring megabytes per page. const planListColumns = `id, plan_identifier, database_name, database_type, - deployment, target, repository, pull_request, schema_path, environment, NULL AS schema_files, plan_data, head_sha, created_at` + deployment, target, repository, pull_request, schema_path, environment, NULL AS schema_files, plan_data, head_sha, primary_plan_identifier, created_at` // planStore implements storage.PlanStore using MySQL. type planStore struct { @@ -49,9 +49,9 @@ func (s *planStore) Create(ctx context.Context, plan *storage.Plan) (int64, erro } id, err := s.identity.InsertID(ctx, s.db, ` - INSERT INTO plans (plan_identifier, database_name, database_type, deployment, target, repository, pull_request, schema_path, environment, schema_files, plan_data, head_sha, created_at) - VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) - `, plan.PlanIdentifier, plan.Database, plan.DatabaseType, plan.Deployment, plan.Target, plan.Repository, plan.PullRequest, plan.SchemaPath, plan.Environment, string(schemaFilesJSON), string(planDataJSON), plan.HeadSHA, plan.CreatedAt) + INSERT INTO plans (plan_identifier, database_name, database_type, deployment, target, repository, pull_request, schema_path, environment, schema_files, plan_data, head_sha, primary_plan_identifier, created_at) + VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?) + `, plan.PlanIdentifier, plan.Database, plan.DatabaseType, plan.Deployment, plan.Target, plan.Repository, plan.PullRequest, plan.SchemaPath, plan.Environment, string(schemaFilesJSON), string(planDataJSON), plan.HeadSHA, plan.PrimaryPlanIdentifier, plan.CreatedAt) if err != nil { if s.classifier.IsDuplicateKey(err) { return 0, storage.ErrPlanIDExists @@ -151,6 +151,10 @@ func (s *planStore) List(ctx context.Context, opts storage.ListPlansOptions) ([] where = append(where, "pull_request = ?") args = append(args, opts.PullRequest) } + if opts.PrimaryPlanIdentifier != "" { + where = append(where, "primary_plan_identifier = ?") + args = append(args, opts.PrimaryPlanIdentifier) + } if !opts.Since.IsZero() { where = append(where, "created_at >= ?") args = append(args, opts.Since) @@ -237,6 +241,7 @@ func scanPlanInto(s scanner) (*storage.Plan, error) { &schemaFilesJSON, &planDataJSON, &plan.HeadSHA, + &plan.PrimaryPlanIdentifier, &plan.CreatedAt, ) if err != nil { diff --git a/pkg/storage/storage.go b/pkg/storage/storage.go index 95f228607..1d7cf5ca1 100644 --- a/pkg/storage/storage.go +++ b/pkg/storage/storage.go @@ -501,6 +501,10 @@ type ListPlansOptions struct { // that PR number. Requires Repository — a PR number is only meaningful // within one repository, so List errors when it is set alone. PullRequest int + // PrimaryPlanIdentifier, when set, restricts results to the member plans + // produced alongside that reviewed plan — the one review round's members, + // rather than every plan stored for the pull request. + PrimaryPlanIdentifier string // Since, when set, restricts results to plans created at or after this // instant. Since time.Time diff --git a/pkg/storage/storagetest/plans.go b/pkg/storage/storagetest/plans.go index 0023a1267..0e460d3e1 100644 --- a/pkg/storage/storagetest/plans.go +++ b/pkg/storage/storagetest/plans.go @@ -83,6 +83,9 @@ func TestPlans(t *testing.T, h Harness) { SchemaPath: "schema/commerce", Environment: "staging", HeadSHA: "sha_head", + // A member planned against its own live schema is bound to the + // reviewed plan of its round; every store must carry that link. + PrimaryPlanIdentifier: "plan_reviewed", SchemaFiles: schema.SchemaFiles{ "commerce": {Files: map[string]string{"users.sql": "CREATE TABLE `users` (`id` bigint unsigned NOT NULL)"}}, }, @@ -105,6 +108,7 @@ func TestPlans(t *testing.T, h Harness) { assert.Equal(t, "schema/commerce", got.SchemaPath) assert.Equal(t, "staging", got.Environment) assert.Equal(t, "sha_head", got.HeadSHA) + assert.Equal(t, "plan_reviewed", got.PrimaryPlanIdentifier) require.Contains(t, got.SchemaFiles, "commerce") assert.Equal(t, "CREATE TABLE `users` (`id` bigint unsigned NOT NULL)", got.SchemaFiles["commerce"].Files["users.sql"]) @@ -376,6 +380,63 @@ func TestPlans(t *testing.T, h Harness) { assert.Empty(t, unmatched, "a filter matching nothing lists as empty, not an error") }) + t.Run("List_FiltersByReviewRound", func(t *testing.T) { + ctx := t.Context() + store := h.NewStorage(t) + + // Two review rounds of the same commit: same database, environment, + // repository, pull request, and head SHA, differing only in which + // reviewed plan each member plan was produced alongside. The round is + // the only thing that tells them apart, so the filter is what keeps an + // apply from pairing a member with a round the operator never saw. + base := time.Now().UTC().Truncate(time.Second).Add(-time.Hour) + create := func(identifier, target, primary string, createdAt time.Time) { + t.Helper() + _, err := store.Plans().Create(ctx, &storage.Plan{ + PlanIdentifier: identifier, + Database: "commerce", + DatabaseType: storage.DatabaseTypeMySQL, + Environment: "production", + Deployment: "eu", + Target: target, + Repository: "org/repo", + PullRequest: 7, + HeadSHA: "sha_head", + PrimaryPlanIdentifier: primary, + CreatedAt: createdAt, + }) + require.NoError(t, err) + } + + create("plan_reviewed_first", "commerce-001", "", base) + create("plan_member_first", "commerce-002", "plan_reviewed_first", base.Add(time.Second)) + create("plan_reviewed_second", "commerce-001", "", base.Add(2*time.Second)) + create("plan_member_second", "commerce-002", "plan_reviewed_second", base.Add(3*time.Second)) + + firstRound, err := store.Plans().List(ctx, storage.ListPlansOptions{ + Repository: "org/repo", + PullRequest: 7, + PrimaryPlanIdentifier: "plan_reviewed_first", + Limit: 10, + }) + require.NoError(t, err) + assert.Equal(t, []string{"plan_member_first"}, planIdentifiers(firstRound), + "the later round's member plan must not answer a lookup for the earlier round") + + secondRound, err := store.Plans().List(ctx, storage.ListPlansOptions{ + Repository: "org/repo", + PullRequest: 7, + PrimaryPlanIdentifier: "plan_reviewed_second", + Limit: 10, + }) + require.NoError(t, err) + assert.Equal(t, []string{"plan_member_second"}, planIdentifiers(secondRound)) + + unfiltered, err := store.Plans().List(ctx, storage.ListPlansOptions{Repository: "org/repo", PullRequest: 7, Limit: 10}) + require.NoError(t, err) + assert.Len(t, unfiltered, 4, "an unfiltered listing still sees every round") + }) + t.Run("List_RejectsNonPositiveLimit", func(t *testing.T) { ctx := t.Context() store := h.NewStorage(t) diff --git a/pkg/storage/types.go b/pkg/storage/types.go index 5f3976d27..409cebac9 100644 --- a/pkg/storage/types.go +++ b/pkg/storage/types.go @@ -573,6 +573,17 @@ type Plan struct { // invariant cannot be evaluated) rather than fail closed. HeadSHA string + // PrimaryPlanIdentifier names the reviewed plan this one was produced + // alongside, for a rollout member planned against its own live schema. It is + // the durable link between a member's plan and the review round the operator + // approved: an apply created from the reviewed plan selects its members' + // plans by this identifier, so a plan from a later re-plan of the same commit + // is a different round and is never substituted for the reviewed one. + // + // Empty on the reviewed plan itself, and on every plan of an environment + // whose members all run the reviewed plan. + PrimaryPlanIdentifier string + // CreatedAt is when the plan was generated. CreatedAt time.Time } From 888c7a847677dfe904fc3808ba4c5c7bda4b996b Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 15:55:00 -0400 Subject: [PATCH 21/34] fix(api): name the target in an operation key only where it has to be named An operation key is unique per deployment, so it identifies one operation only when the deployment addresses one target. Where a deployment addresses several, two members of it would otherwise store the same key for the same table and collide on the target-blind unique index. Keys are now qualified with the member's target in exactly that case, by the same predicate that decides when a member is worth naming after its target. A deployment with one target keeps the key shape every reader already parses. A member whose own plan found nothing to change is recorded as completed on creation, so the apply covers every member it addressed rather than leaving a pending operation no driver can finish. Member plans are now resolved by the review round they were stamped with rather than by head SHA and first-row-wins. The round takes precedence over current config: a round that planned its members independently is applied that way even if the environment has since been respelled as mirrored. Config decides only the case the round is silent about, and fails closed there. Co-Authored-By: Claude Opus 5 --- pkg/api/apply_members.go | 112 ++++++++++------- pkg/api/apply_members_test.go | 152 ++++++++++++++++++++---- pkg/api/handlers_test.go | 21 +++- pkg/api/plan_handlers.go | 97 +++++++++++++-- pkg/api/sharded_pershard_fanout_test.go | 6 +- pkg/routing/resolver.go | 31 +++++ pkg/routing/resolver_test.go | 75 ++++++++++++ 7 files changed, 412 insertions(+), 82 deletions(-) create mode 100644 pkg/routing/resolver_test.go diff --git a/pkg/api/apply_members.go b/pkg/api/apply_members.go index 140fe4a7d..f27bd9caa 100644 --- a/pkg/api/apply_members.go +++ b/pkg/api/apply_members.go @@ -8,10 +8,10 @@ import ( "github.com/block/schemabot/pkg/storage" ) -// memberPlanLookupLimit bounds the plan listing that resolves member plans. A -// review round stores one plan per member, and a PR is re-planned on every push, -// so the listing has to reach back far enough to cover several rounds of a -// wide environment while staying a bounded read. +// memberPlanLookupLimit bounds the plan listing that resolves member plans. The +// listing is already narrowed to one review round, which stores one plan per +// member, so the limit only has to cover the widest environment a single round +// can address. const memberPlanLookupLimit = 200 // applyMember is one rollout member of an apply together with the plan its work @@ -37,13 +37,20 @@ func (m applyMember) MemberID() string { // When the members were planned independently, each one has a plan of its own // stored at review time, and running one member's DDL against another's target // would apply a schema that target was never planned for. So each non-primary -// member is paired with its own stored plan, matched on the head SHA the apply's -// plan was created for, which binds the member plans to the same review round -// the operator approved. +// member is paired with its own stored plan from the same review round. // -// A member with no plan for that review round fails apply creation. There is no -// safe fallback: the apply's plan describes a different target's schema, so -// substituting it would run DDL that was never planned for this member. +// The round takes precedence over current config. A round that planned its +// members independently stored a plan for each, stamped with the reviewed plan's +// identifier, and those plans are what the operator approved — so they are used +// even if the environment's routing has since been respelled as mirrored. +// Config decides only the case the round is silent about: no member plans stored +// means either that every member was verified against the reviewed plan, or that +// nothing planned the members at all, and only config can tell those apart. +// +// A member with no plan for an independently planned round fails apply creation. +// There is no safe fallback: the apply's plan describes a different target's +// schema, so substituting it would run DDL that was never planned for this +// member. func (s *Service) resolveApplyMembers(ctx context.Context, plan *storage.Plan, environment string, targets []routing.ExecutionTarget) ([]applyMember, error) { // A single member is the plan's own primary, so it runs the apply's plan // under either contract and there is no sibling whose plan could be @@ -55,76 +62,89 @@ func (s *Service) resolveApplyMembers(ctx context.Context, plan *storage.Plan, e return []applyMember{{Target: targets[0], Plan: plan}}, nil } - planning, err := s.config.MemberPlanningFor(plan.Database, environment) + memberPlans, err := s.memberPlansForReviewRound(ctx, plan, environment) if err != nil { - // A database/environment the config no longer resolves cannot be shown to - // have independently planned members, and mirrored is the shape that reuses - // one plan for every member. Fail rather than assume it. - return nil, fmt.Errorf("resolve member planning for %s/%s: %w", plan.Database, environment, err) + return nil, err } members := make([]applyMember, 0, len(targets)) - if planning == PlanMirrored { - for _, target := range targets { - members = append(members, applyMember{Target: target, Plan: plan}) + if len(memberPlans) == 0 { + planning, err := s.config.MemberPlanningFor(plan.Database, environment) + if err != nil { + // A database/environment the config no longer resolves cannot be shown + // to have had its members verified against the reviewed plan, and that + // is the only shape in which one plan runs everywhere. Fail rather than + // assume it. + return nil, fmt.Errorf("resolve member planning for %s/%s: %w", plan.Database, environment, err) + } + if planning == PlanMirrored { + for _, target := range targets { + members = append(members, applyMember{Target: target, Plan: plan}) + } + return members, nil } - return members, nil } - memberPlans, err := s.memberPlansForReviewRound(ctx, plan, environment) - if err != nil { - return nil, err - } primary := routing.ExecutionTarget{Deployment: plan.Deployment, Target: plan.Target} for _, target := range targets { // The apply is created from the primary's plan, so the primary needs no - // lookup — and must not take one, since a re-plan could have stored a newer - // row for the same member than the plan the operator approved. + // lookup — and must not take one, since it stores no plan of its own. if target.Deployment == primary.Deployment && target.Target == primary.Target { members = append(members, applyMember{Target: target, Plan: plan}) continue } memberPlan, ok := memberPlans[target.MemberID()] if !ok { - return nil, fmt.Errorf("apply for %s/%s has no stored plan for rollout member %s at head %q; plan the environment again so every target is planned before applying", - plan.Database, environment, target.MemberID(), plan.HeadSHA) + if plan.HeadSHA == "" { + // Member plans are only ever written by a pull request review, so + // an apply from a plan that had none is unplannable by + // construction rather than by a missed round. + return nil, fmt.Errorf("apply for %s/%s has no stored plan for rollout member %s, and its plan %s was not produced by a pull request review; plan from a pull request so every target is planned", + plan.Database, environment, target.MemberID(), plan.PlanIdentifier) + } + return nil, fmt.Errorf("apply for %s/%s has no stored plan for rollout member %s in the reviewed round (plan %s); plan the environment again so every target is planned before applying", + plan.Database, environment, target.MemberID(), plan.PlanIdentifier) } members = append(members, applyMember{Target: target, Plan: memberPlan}) } return members, nil } -// memberPlansForReviewRound loads the member plans stored alongside plan, keyed -// by member id. Plans are listed newest first, so the first row seen for a -// member is that member's latest plan within the round. +// memberPlansForReviewRound loads the plans stored for the members of this +// apply's own review round, keyed by member id. // -// The round is identified by the apply plan's head SHA: every member of one -// review round is planned against the same commit, so a plan from an earlier -// push is a different round and must not be picked up. An apply plan with no -// head SHA was not produced by a PR review, which is the only place member plans -// are written, so there is nothing to match it against. +// The round is identified by the reviewed plan's identifier, which every member +// plan is stamped with at review time. A commit can be planned more than once — +// a re-plan, or two deliveries racing — and each round writes a plan per member +// carrying the same database, environment, repository, pull request, route, and +// head SHA. The stamp is the only thing that separates them, so selecting on it +// is what keeps an apply from dispatching a member plan the operator never saw. +// +// An empty result means the round planned no member on its own, which is how a +// mirrored round reads: it is the answer, not a lookup that came up short. func (s *Service) memberPlansForReviewRound(ctx context.Context, plan *storage.Plan, environment string) (map[string]*storage.Plan, error) { - if plan.HeadSHA == "" { - return nil, fmt.Errorf("apply for %s/%s addresses several targets but its plan has no head SHA to match member plans against; plan from a pull request so every target is planned", + if plan.PlanIdentifier == "" { + return nil, fmt.Errorf("apply for %s/%s addresses several targets but its plan has no identifier to match member plans against", plan.Database, environment) } stored, err := s.storage.Plans().List(ctx, storage.ListPlansOptions{ - Database: plan.Database, - Environment: environment, - Repository: plan.Repository, - PullRequest: plan.PullRequest, - Limit: memberPlanLookupLimit, + Database: plan.Database, + Environment: environment, + Repository: plan.Repository, + PullRequest: plan.PullRequest, + PrimaryPlanIdentifier: plan.PlanIdentifier, + Limit: memberPlanLookupLimit, }) if err != nil { - return nil, fmt.Errorf("list member plans for %s/%s pr %d: %w", plan.Database, environment, plan.PullRequest, err) + return nil, fmt.Errorf("list member plans for %s/%s round %s: %w", plan.Database, environment, plan.PlanIdentifier, err) } byMember := make(map[string]*storage.Plan, len(stored)) for _, candidate := range stored { - if candidate.HeadSHA != plan.HeadSHA { - continue - } memberID := routing.ExecutionTarget{Deployment: candidate.Deployment, Target: candidate.Target}.MemberID() if _, seen := byMember[memberID]; seen { + // One round stores one plan per member. A second row for the same + // member means the round was written twice; the listing is newest + // first, so the later write is the one the round ended with. continue } byMember[memberID] = candidate diff --git a/pkg/api/apply_members_test.go b/pkg/api/apply_members_test.go index 77b4b7f8b..e8dd99708 100644 --- a/pkg/api/apply_members_test.go +++ b/pkg/api/apply_members_test.go @@ -11,6 +11,7 @@ import ( "github.com/stretchr/testify/require" "github.com/block/schemabot/pkg/routing" + "github.com/block/schemabot/pkg/state" "github.com/block/schemabot/pkg/storage" "github.com/block/schemabot/pkg/tern" ) @@ -23,11 +24,17 @@ type listingPlanStore struct { listErr error } -func (s *listingPlanStore) List(context.Context, storage.ListPlansOptions) ([]*storage.Plan, error) { +func (s *listingPlanStore) List(_ context.Context, opts storage.ListPlansOptions) ([]*storage.Plan, error) { if s.listErr != nil { return nil, s.listErr } - return s.plans, nil + var matched []*storage.Plan + for _, plan := range s.plans { + if plan.PrimaryPlanIdentifier == opts.PrimaryPlanIdentifier { + matched = append(matched, plan) + } + } + return matched, nil } func memberResolutionService(t *testing.T, env EnvironmentConfig, plans storage.PlanStore) *Service { @@ -69,6 +76,16 @@ func primaryPlanRow(target string) *storage.Plan { } } +// memberPlanRow is a plan stored for a non-primary member, stamped with the +// reviewed plan it was produced alongside. +func memberPlanRow(identifier, target, primaryPlanIdentifier string) *storage.Plan { + plan := primaryPlanRow(target) + plan.ID = 0 + plan.PlanIdentifier = identifier + plan.PrimaryPlanIdentifier = primaryPlanIdentifier + return plan +} + func targetsFor(t *testing.T, svc *Service) []routing.ExecutionTarget { t.Helper() targets, err := svc.config.ResolveDatabaseTargets("testapp", "production") @@ -76,11 +93,10 @@ func targetsFor(t *testing.T, svc *Service) []routing.ExecutionTarget { return targets } -// Members that are expected to hold the same schema all run the plan the -// operator reviewed, so every member is paired with the apply's own plan and no -// member-plan lookup is needed. +// Members that are expected to hold the same schema store no plans of their +// own, because every one of them runs the plan the operator reviewed. func TestResolveApplyMembers_MirroredMembersShareTheApplyPlan(t *testing.T) { - plans := &listingPlanStore{listErr: errors.New("List must not be called for mirrored members")} + plans := &listingPlanStore{} svc := memberResolutionService(t, mirroredEnv(), plans) plan := primaryPlanRow("testapp") @@ -98,7 +114,7 @@ func TestResolveApplyMembers_MirroredMembersShareTheApplyPlan(t *testing.T) { // apply's plan without consulting config for a contract that could not change // the outcome. func TestResolveApplyMembers_SingleMemberNeedsNoDatabaseConfig(t *testing.T) { - plans := &listingPlanStore{listErr: errors.New("List must not be called for a single member")} + plans := &listingPlanStore{listErr: errors.New("a single member needs no member-plan lookup")} logger := slog.New(slog.NewTextHandler(os.Stdout, &slog.HandlerOptions{Level: slog.LevelError})) svc := New(&mockStorageWithPlanLookup{plans: plans}, &ServerConfig{}, map[string]tern.Client{}, logger) plan := primaryPlanRow("testapp-001") @@ -113,11 +129,10 @@ func TestResolveApplyMembers_SingleMemberNeedsNoDatabaseConfig(t *testing.T) { } // Each target of a multi-target environment runs the plan stored for that -// target in the same review round, matched on the head SHA the apply's plan was -// created for. +// target in the review round the apply's own plan is the reviewed plan of. func TestResolveApplyMembers_IndependentMembersRunTheirOwnPlans(t *testing.T) { plan := primaryPlanRow("testapp-001") - secondPlan := &storage.Plan{ID: 11, PlanIdentifier: "plan-second", Deployment: "eu", Target: "testapp-002", HeadSHA: "abc123"} + secondPlan := memberPlanRow("plan-second", "testapp-002", "plan-primary") plans := &listingPlanStore{plans: []*storage.Plan{secondPlan}} svc := memberResolutionService(t, multiTargetEnv(), plans) @@ -130,12 +145,12 @@ func TestResolveApplyMembers_IndependentMembersRunTheirOwnPlans(t *testing.T) { assert.Same(t, secondPlan, members[1].Plan) } -// A plan stored for an earlier push is a different review round. Matching it to -// this apply would run DDL the operator never reviewed on this commit, so the -// member counts as unplanned and apply creation fails. +// A plan stored in another review round of the same pull request — an earlier +// push, or a re-plan of this very commit — would run DDL the operator never +// approved, so the member counts as unplanned and apply creation fails. func TestResolveApplyMembers_MemberPlanFromAnotherRoundIsNotUsed(t *testing.T) { plan := primaryPlanRow("testapp-001") - stale := &storage.Plan{ID: 9, PlanIdentifier: "plan-stale", Deployment: "eu", Target: "testapp-002", HeadSHA: "older"} + stale := memberPlanRow("plan-stale", "testapp-002", "plan-earlier-round") plans := &listingPlanStore{plans: []*storage.Plan{stale}} svc := memberResolutionService(t, multiTargetEnv(), plans) @@ -156,9 +171,10 @@ func TestResolveApplyMembers_MissingMemberPlanFailsClosed(t *testing.T) { assert.Contains(t, err.Error(), "no stored plan for rollout member eu/testapp-002") } -// Member plans are only written by a pull request review, so a plan with no head -// SHA has no round to match against and cannot drive a multi-target apply. -func TestResolveApplyMembers_PlanWithoutHeadSHAFailsClosed(t *testing.T) { +// Member plans are only written by a pull request review, so a CLI apply against +// a multi-target environment has none and fails closed rather than running the +// primary's DDL against every target. +func TestResolveApplyMembers_PlanWithoutPullRequestReviewFailsClosed(t *testing.T) { plan := primaryPlanRow("testapp-001") plan.HeadSHA = "" plans := &listingPlanStore{} @@ -166,7 +182,24 @@ func TestResolveApplyMembers_PlanWithoutHeadSHAFailsClosed(t *testing.T) { _, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) require.Error(t, err) - assert.Contains(t, err.Error(), "no head SHA to match member plans against") + assert.Contains(t, err.Error(), "was not produced by a pull request review") +} + +// An environment respelled as mirrored after its review still applies what was +// reviewed: the round stored a plan per member, and those plans are what the +// operator approved. Current config cannot reinterpret a finished review. +func TestResolveApplyMembers_ReviewedRoundOutranksCurrentConfig(t *testing.T) { + plan := primaryPlanRow("testapp") + secondPlan := memberPlanRow("plan-second", "testapp", "plan-primary") + secondPlan.Deployment = "us" + plans := &listingPlanStore{plans: []*storage.Plan{secondPlan}} + svc := memberResolutionService(t, mirroredEnv(), plans) + + members, err := svc.resolveApplyMembers(t.Context(), plan, "production", targetsFor(t, svc)) + require.NoError(t, err) + require.Len(t, members, 2) + assert.Same(t, plan, members[0].Plan, "the primary runs the plan the apply was created from") + assert.Same(t, secondPlan, members[1].Plan, "a member planned on its own keeps its reviewed plan") } // A storage failure while loading member plans is not an absence of members: it @@ -225,10 +258,81 @@ func TestBuildApplyOperationGroups_TargetsOfOneDeploymentGetOwnOperations(t *tes assert.Equal(t, int64(11), groups[1].Operation.PlanID) } -// A sharded plan produces the same operation key for the same (namespace, shard, -// table) on every member, so operations are grouped by member and key together. -// Two targets of one deployment each get their own operation for that key rather -// than one target's shard work being folded into the other's. +// An operation key is unique per deployment, so it only has to name a target +// where the deployment addresses more than one. A deployment with a single +// target leaves the key alone, keeping it the shape every reader already parses. +func TestMemberOperationKeys_QualifiesOnlyMultiTargetDeployments(t *testing.T) { + members := []applyMember{ + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-001"}}, + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}}, + {Target: routing.ExecutionTarget{Deployment: "us", Target: "testapp"}}, + } + keys := newMemberOperationKeys(members) + + qualified, err := keys.qualify(members[0], "testapp/-80/mutes") + require.NoError(t, err) + assert.Equal(t, "testapp-001/testapp/-80/mutes", qualified) + + whole, err := keys.qualify(members[0], "") + require.NoError(t, err) + assert.Equal(t, "testapp-001", whole, "a member with no scoped work is named by its target alone") + + sole, err := keys.qualify(members[2], "testapp/-80/mutes") + require.NoError(t, err) + assert.Equal(t, "testapp/-80/mutes", sole, "a deployment with one target needs no target component") +} + +// Config refuses the delimiter in a target's name, but an apply can be created +// from a plan stored under a config that no longer applies. A target that would +// make its members' operation keys ambiguous to split fails apply creation +// rather than producing keys no reader can take apart. +func TestMemberOperationKeys_RefusesTargetCarryingTheDelimiter(t *testing.T) { + members := []applyMember{ + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp/001"}}, + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}}, + } + keys := newMemberOperationKeys(members) + + _, err := keys.qualify(members[0], "testapp/users") + require.Error(t, err) + assert.Contains(t, err.Error(), "target") +} + +// A member whose own plan found nothing to change is already converged. Its +// operation is recorded as completed so the apply covers every member it +// addressed, rather than leaving a pending operation no driver can ever finish. +func TestBuildApplyOperationGroups_ConvergedMemberIsCompletedOnCreation(t *testing.T) { + usersDDL := "ALTER TABLE `users` ADD COLUMN `email` varchar(255)" + applyPlan := primaryPlanRow("testapp-001") + applyPlan.Namespaces = map[string]*storage.NamespacePlanData{ + "testapp": {Tables: []storage.TableChange{{Namespace: "testapp", Table: "users", DDL: usersDDL, Operation: "alter"}}}, + } + convergedPlan := &storage.Plan{ID: 11, Deployment: "eu", Target: "testapp-002"} + members := []applyMember{ + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-001"}, Plan: applyPlan}, + {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}, Plan: convergedPlan}, + } + taskChanges := applyTaskChanges(applyPlan) + + groups, _, err := buildApplyOperationGroups(applyPlan, taskChanges, members, "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + require.NoError(t, err) + require.Len(t, groups, 2) + + assert.Equal(t, state.ApplyOperation.Pending, groups[0].Operation.State) + require.Len(t, groups[0].Tasks, 1) + + converged := groups[1].Operation + assert.Empty(t, groups[1].Tasks, "a converged member has no work to drive") + assert.Equal(t, state.ApplyOperation.Completed, converged.State) + require.NotNil(t, converged.StartedAt) + require.NotNil(t, converged.CompletedAt) + assert.Equal(t, pershardTestTime(), *converged.CompletedAt) +} + +// A sharded plan describes the same (namespace, shard, table) work on every +// member, and the operation key is unique per deployment. Two targets of one +// deployment therefore lead their keys with the target's name, so each gets its +// own operation rather than one target's shard work being folded into the other's. func TestBuildShardedApplyOperationGroups_TargetsOfOneDeploymentDoNotShareOperations(t *testing.T) { mutesDDL := "ALTER TABLE `mutes` ADD INDEX (`created_at`)" applyPlan := &storage.Plan{ @@ -246,7 +350,7 @@ func TestBuildShardedApplyOperationGroups_TargetsOfOneDeploymentDoNotShareOperat {Target: routing.ExecutionTarget{Deployment: "eu", Target: "testapp-002"}, Plan: secondPlan}, } - groups, err := buildShardedApplyOperationGroups(applyPlan, members, "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, err := buildShardedApplyOperationGroups(applyPlan, members, newMemberOperationKeys(members), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) require.Len(t, groups, 2, "each target needs its own operation for the shard's table") @@ -257,7 +361,7 @@ func TestBuildShardedApplyOperationGroups_TargetsOfOneDeploymentDoNotShareOperat require.Contains(t, byTarget, "testapp-001") require.Contains(t, byTarget, "testapp-002") for target, group := range byTarget { - assert.Equal(t, pershardNamespace+"/-80/mutes", group.Operation.OperationKey) + assert.Equal(t, target+"/"+pershardNamespace+"/-80/mutes", group.Operation.OperationKey) assert.Equal(t, "eu", group.Operation.Deployment) require.Len(t, group.Tasks, 1, "target %s must carry its shard's work exactly once", target) assert.Equal(t, mutesDDL, group.Tasks[0].DDL) diff --git a/pkg/api/handlers_test.go b/pkg/api/handlers_test.go index 2bc387b51..d2e8254a9 100644 --- a/pkg/api/handlers_test.go +++ b/pkg/api/handlers_test.go @@ -199,7 +199,26 @@ type staticPlanStore struct { storage.PlanStore plan *storage.Plan plansByID map[int64]*storage.Plan - err error + // memberPlans are the plans stored for the members of a review round, + // listed by the round the apply was created from. + memberPlans []*storage.Plan + err error +} + +func (s *staticPlanStore) List(_ context.Context, opts storage.ListPlansOptions) ([]*storage.Plan, error) { + if s.err != nil { + return nil, s.err + } + if opts.PrimaryPlanIdentifier == "" { + return s.memberPlans, nil + } + var matched []*storage.Plan + for _, plan := range s.memberPlans { + if plan.PrimaryPlanIdentifier == opts.PrimaryPlanIdentifier { + matched = append(matched, plan) + } + } + return matched, nil } func (s *staticPlanStore) Get(context.Context, string) (*storage.Plan, error) { diff --git a/pkg/api/plan_handlers.go b/pkg/api/plan_handlers.go index 4238a2443..211856f5f 100644 --- a/pkg/api/plan_handlers.go +++ b/pkg/api/plan_handlers.go @@ -1469,8 +1469,9 @@ func buildApplyOperationGroups( // instance-local sharded engine (Strata) produces those, so an // externally-authoritative engine (e.g. PlanetScale) — whose plans never // carry per-shard changes — is never fanned out, regardless of transport. + keys := newMemberOperationKeys(members) if canBuildShardedOperationGroups(plan, taskChanges) { - groups, err := buildShardedApplyOperationGroups(plan, members, environment, applyOpts, cutoverPolicy, onFailure, now) + groups, err := buildShardedApplyOperationGroups(plan, members, keys, environment, applyOpts, cutoverPolicy, onFailure, now) if err != nil { return nil, false, err } @@ -1489,7 +1490,11 @@ func buildApplyOperationGroups( if len(taskChanges) == 0 && len(plan.VSchemaNamespaces()) > 0 { groups := make([]*storage.ApplyOperationWithTasks, 0, len(members)) for _, member := range members { - operation := newPendingApplyOperation(member, plan, finalizerOperationKeySegment, cutoverPolicy, onFailure, now) + operationKey, err := keys.qualify(member, finalizerOperationKeySegment) + if err != nil { + return nil, false, err + } + operation := newPendingApplyOperation(member, plan, operationKey, cutoverPolicy, onFailure, now) operation.OperationKind = storage.ApplyOperationKindGroupFinalizer groups = append(groups, &storage.ApplyOperationWithTasks{Operation: operation}) } @@ -1502,14 +1507,40 @@ func buildApplyOperationGroups( // its target holds a schema the apply plan never described. memberChanges := applyTaskChanges(member.Plan) tasks := buildApplyTasks(member.Plan, memberChanges, environment, applyOpts, "", now) + operationKey, err := keys.qualify(member, "") + if err != nil { + return nil, false, err + } + operation := newPendingApplyOperation(member, plan, operationKey, cutoverPolicy, onFailure, now) + if len(tasks) == 0 { + // A member planned on its own can already hold the reviewed change, + // so its plan has nothing left to run. The member still belongs to + // the rollout: dropping it would make the apply silently address + // fewer targets than the operator asked for, and a work operation + // with no tasks can never be driven, so it is recorded as the + // already-settled work it is. + settleConvergedMemberOperation(operation, now) + } groups = append(groups, &storage.ApplyOperationWithTasks{ - Operation: newPendingApplyOperation(member, plan, "", cutoverPolicy, onFailure, now), + Operation: operation, Tasks: tasks, }) } return groups, false, nil } +// settleConvergedMemberOperation records a member that had nothing left to run +// as completed at creation. It is the one operation shape that is terminal +// before a driver ever claims it: there is no work to drive, and the alternative +// shapes are both wrong — a pending row that no drive can satisfy halts the +// rollout, and omitting the member entirely removes it from every per-member +// surface the operator reads. +func settleConvergedMemberOperation(operation *storage.ApplyOperation, now time.Time) { + operation.State = state.ApplyOperation.Completed + operation.StartedAt = &now + operation.CompletedAt = &now +} + // buildNamespaceFinalizerOperations builds one task-less group_finalizer per // VSchema-changed namespace in the plan, for one target. The VSchema is applied // once the namespace's shard work (if any) completes; the finalizer drives it @@ -1519,6 +1550,7 @@ func buildApplyOperationGroups( func buildNamespaceFinalizerOperations( applyPlan *storage.Plan, member applyMember, + keys memberOperationKeys, cutoverPolicy string, onFailure string, now time.Time, @@ -1529,7 +1561,10 @@ func buildNamespaceFinalizerOperations( if err := validateOperationKeyPart("namespace", namespace); err != nil { return nil, err } - operationKey := finalizerOperationKey(namespace) + operationKey, err := keys.qualify(member, finalizerOperationKey(namespace)) + if err != nil { + return nil, err + } if len(operationKey) > applyOperationKeyMaxLen { return nil, fmt.Errorf("operation key for namespace %q finalizer exceeds %d characters", namespace, applyOperationKeyMaxLen) } @@ -1566,6 +1601,7 @@ func canBuildShardedOperationGroups(plan *storage.Plan, taskChanges []storage.Ta func buildShardedApplyOperationGroups( applyPlan *storage.Plan, members []applyMember, + keys memberOperationKeys, environment string, applyOpts storage.ApplyOptions, cutoverPolicy string, @@ -1604,7 +1640,10 @@ func buildShardedApplyOperationGroups( if err := validateShardOperationKeyParts(namespace, shard.Shard, ddlChange.Table); err != nil { return nil, err } - operationKey := storage.ShardOperationKey(namespace, shard.Shard, ddlChange.Table) + operationKey, err := keys.qualify(member, storage.ShardOperationKey(namespace, shard.Shard, ddlChange.Table)) + if err != nil { + return nil, err + } if len(operationKey) > applyOperationKeyMaxLen { return nil, fmt.Errorf("operation key for namespace %q shard %q table %q exceeds %d characters", namespace, shard.Shard, ddlChange.Table, applyOperationKeyMaxLen) } @@ -1621,7 +1660,7 @@ func buildShardedApplyOperationGroups( } } } - finalizers, err := buildNamespaceFinalizerOperations(applyPlan, member, cutoverPolicy, onFailure, now) + finalizers, err := buildNamespaceFinalizerOperations(applyPlan, member, keys, cutoverPolicy, onFailure, now) if err != nil { return nil, err } @@ -1630,6 +1669,48 @@ func buildShardedApplyOperationGroups( return groups, nil } +// memberOperationKeys decides how one apply's operation keys are qualified. +// +// An operation is unique on (apply, deployment, operation key), so a deployment +// addressing several targets needs the target in the key or its members collide +// on one row. A deployment addressing one target does not: its key is already +// unique, and naming the target would change the shape of every key every +// existing reader parses, for no gain. So the target leads the key exactly where +// the deployment stops identifying the member — the same rule that decides +// whether an operator sees a member named "eu" or "eu/shop-002". +// +// Keeping both shapes live is what makes widening the rule later a change to +// this one predicate: readers already recover the scoped key by matching the +// operation's own target rather than by counting components. +type memberOperationKeys struct { + multiTargetDeployments map[string]bool +} + +func newMemberOperationKeys(members []applyMember) memberOperationKeys { + targets := make([]routing.ExecutionTarget, len(members)) + for i, member := range members { + targets[i] = member.Target + } + return memberOperationKeys{multiTargetDeployments: routing.MultiTargetDeployments(targets)} +} + +// qualify returns the operation key for one member's scoped work. scopedKey is +// the key within the member — a shard key, a finalizer key, or empty for work +// covering the whole target. +func (k memberOperationKeys) qualify(member applyMember, scopedKey string) (string, error) { + if !k.multiTargetDeployments[member.Target.Deployment] { + return scopedKey, nil + } + // The config that admits a target refuses the delimiter in its name, but a + // stored plan can carry a target from a config that no longer applies. A key + // that cannot be split back into the target it came from is not recoverable + // once written, so it is refused here too. + if err := validateOperationKeyPart("target", member.Target.Target); err != nil { + return "", err + } + return storage.TargetOperationKey(member.Target.Target, scopedKey), nil +} + func validateShardOperationKeyParts(namespace, shard, table string) error { for _, part := range []struct { label string @@ -1647,8 +1728,8 @@ func validateShardOperationKeyParts(namespace, shard, table string) error { } func validateOperationKeyPart(label, value string) error { - if strings.Contains(value, "/") { - return fmt.Errorf("operation key %s component %q contains reserved delimiter %q", label, value, "/") + if strings.Contains(value, storage.OperationKeyDelimiter) { + return fmt.Errorf("operation key %s component %q contains reserved delimiter %q", label, value, storage.OperationKeyDelimiter) } return nil } diff --git a/pkg/api/sharded_pershard_fanout_test.go b/pkg/api/sharded_pershard_fanout_test.go index 5c921fa55..39d6d2a10 100644 --- a/pkg/api/sharded_pershard_fanout_test.go +++ b/pkg/api/sharded_pershard_fanout_test.go @@ -66,7 +66,7 @@ func TestBuildShardedApplyOperationGroupsUsesPerShardDDL(t *testing.T) { }, } - groups, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), newMemberOperationKeys(pershardMembers(plan)), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) got := operationDDLByKey(groups) @@ -96,7 +96,7 @@ func TestBuildShardedApplyOperationGroupsSkipsShardsWithoutChanges(t *testing.T) }, } - groups, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + groups, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), newMemberOperationKeys(pershardMembers(plan)), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.NoError(t, err) assert.Equal(t, map[string][]string{ @@ -140,7 +140,7 @@ func TestBuildShardedApplyOperationGroupsFailsClosedOnMalformedChange(t *testing Changes: []storage.TableChange{{Namespace: pershardNamespace, Table: "", DDL: "ALTER TABLE `mutes` ADD INDEX (`x`)", Operation: "alter"}}, }}, } - _, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) + _, err := buildShardedApplyOperationGroups(plan, pershardMembers(plan), newMemberOperationKeys(pershardMembers(plan)), "production", storage.ApplyOptions{}, "", "", pershardTestTime()) require.Error(t, err) assert.Contains(t, err.Error(), "empty table") } diff --git a/pkg/routing/resolver.go b/pkg/routing/resolver.go index 34777c014..ef33c982f 100644 --- a/pkg/routing/resolver.go +++ b/pkg/routing/resolver.go @@ -32,6 +32,37 @@ func (t ExecutionTarget) MemberID() string { return t.Deployment + "/" + t.Target } +// MultiTargetDeployments reports which deployments of a member set address more +// than one distinct target. +// +// It is the point at which a deployment name stops identifying one member, and +// so the one place anything qualifies itself with the target: the name an +// operator reads, and the key an operation is stored under. Deriving both from +// this keeps them from disagreeing about when a target is worth naming. +// +// Distinct targets rather than member count is the test on purpose. A keyed or +// sharded apply has several members on one target, and naming the target there +// would add a component that still does not tell them apart. +func MultiTargetDeployments(members []ExecutionTarget) map[string]bool { + targetsByDeployment := make(map[string]map[string]struct{}, len(members)) + for _, m := range members { + if m.Target == "" { + continue + } + if targetsByDeployment[m.Deployment] == nil { + targetsByDeployment[m.Deployment] = make(map[string]struct{}, 1) + } + targetsByDeployment[m.Deployment][m.Target] = struct{}{} + } + multi := make(map[string]bool, len(targetsByDeployment)) + for deployment, targets := range targetsByDeployment { + if len(targets) > 1 { + multi[deployment] = true + } + } + return multi +} + // Resolver resolves logical SchemaBot targets to concrete execution targets. type Resolver interface { ResolveTargets(ctx context.Context, req Request) ([]ExecutionTarget, error) diff --git a/pkg/routing/resolver_test.go b/pkg/routing/resolver_test.go new file mode 100644 index 000000000..d1fa9505c --- /dev/null +++ b/pkg/routing/resolver_test.go @@ -0,0 +1,75 @@ +package routing + +import ( + "testing" + + "github.com/stretchr/testify/assert" +) + +func TestMultiTargetDeployments(t *testing.T) { + tests := []struct { + name string + members []ExecutionTarget + want map[string]bool + }{ + { + name: "a deployment addressing two targets is multi-target", + members: []ExecutionTarget{ + {Deployment: "eu", Target: "testapp-001"}, + {Deployment: "eu", Target: "testapp-002"}, + }, + want: map[string]bool{"eu": true}, + }, + { + // A keyed or sharded apply puts several members on one target. The + // target is the same for all of them, so naming it would add a + // component that still does not tell them apart. + name: "repeated members on one target are not multi-target", + members: []ExecutionTarget{ + {Deployment: "eu", Target: "testapp"}, + {Deployment: "eu", Target: "testapp"}, + }, + want: map[string]bool{}, + }, + { + // Mirrored deployments each address one target, so no deployment of + // the set has to name it. + name: "one target per deployment is not multi-target", + members: []ExecutionTarget{ + {Deployment: "eu", Target: "testapp"}, + {Deployment: "us", Target: "testapp"}, + }, + want: map[string]bool{}, + }, + { + name: "only the deployment with several targets is reported", + members: []ExecutionTarget{ + {Deployment: "eu", Target: "testapp-001"}, + {Deployment: "eu", Target: "testapp-002"}, + {Deployment: "us", Target: "testapp"}, + }, + want: map[string]bool{"eu": true}, + }, + { + // A target the resolver left blank names nothing, so it cannot make + // a deployment's members distinguishable by target. + name: "blank targets do not count toward the set", + members: []ExecutionTarget{ + {Deployment: "eu", Target: "testapp"}, + {Deployment: "eu", Target: ""}, + }, + want: map[string]bool{}, + }, + { + name: "no members", + members: nil, + want: map[string]bool{}, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + assert.Equal(t, tt.want, MultiTargetDeployments(tt.members)) + }) + } +} From 95f4bd073efdb7d2561db701b5857062b783ee13 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 15:59:16 -0400 Subject: [PATCH 22/34] fix(tern): dispatch the plan the operation runs, not the apply's MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A rollout member planned against its own live schema names its plan on its operation row. Dispatch loaded the apply's plan regardless, so a member's work was sent with the primary's schema files and recorded against the primary's plan identifier — one target's DDL described by another target's plan. The scope now resolves the plan it runs: an operation's own when it has one, its apply's otherwise, and an error when neither row names a plan rather than a zero a caller could mistake for a valid plan. Co-Authored-By: Claude Opus 5 --- pkg/tern/grpc_client.go | 30 +++++++++++++++++++++++---- pkg/tern/grpc_client_test.go | 40 ++++++++++++++++++++++++++++++++++++ 2 files changed, 66 insertions(+), 4 deletions(-) diff --git a/pkg/tern/grpc_client.go b/pkg/tern/grpc_client.go index 2b1f38866..71e9a1d8d 100644 --- a/pkg/tern/grpc_client.go +++ b/pkg/tern/grpc_client.go @@ -1584,6 +1584,21 @@ func (s applyTaskScope) tasklessOperationScope() bool { return s.isOperationScoped() && s.tasklessOperation } +// planID resolves the plan this drive runs. A rollout member planned against +// its own live schema names its plan on its operation row, and dispatching the +// apply's plan there would send another target's DDL and record another +// target's plan identifier against this member's work. A whole-apply drive has +// no operation to name one, so it runs the apply's plan. +func (s applyTaskScope) planID(apply *storage.Apply) (int64, error) { + if s.operation == nil { + if apply == nil || apply.PlanID == 0 { + return 0, fmt.Errorf("resolve plan for whole-apply drive: apply names no plan") + } + return apply.PlanID, nil + } + return storage.PlanIDForOperation(apply, s.operation) +} + // remoteApplyID resolves the remote Tern apply id sent on this drive's // Progress/Stop/Start/Cutover calls. Operation-owning drives read the claimed // operation's recorded remote apply id (which may be empty before dispatch); @@ -3048,15 +3063,22 @@ func hasAmbiguousRemoteDispatchState(apply *storage.Apply, scope applyTaskScope) } func (c *GRPCClient) dispatchPendingApply(ctx context.Context, apply *storage.Apply, scope applyTaskScope) error { - plan, err := c.storage.Plans().GetByID(ctx, apply.PlanID) + planID, err := scope.planID(apply) + if err != nil { + if markErr := c.markRemoteApplyFailed(ctx, apply, nil, fmt.Sprintf("queued gRPC apply failed: %v", err), false, scope); markErr != nil { + return fmt.Errorf("mark queued gRPC apply %s failed after plan resolution error: %w", apply.ApplyIdentifier, markErr) + } + return fmt.Errorf("queued gRPC apply %s: %w", apply.ApplyIdentifier, err) + } + plan, err := c.storage.Plans().GetByID(ctx, planID) if err != nil { - if markErr := c.markRemoteApplyFailed(ctx, apply, nil, fmt.Sprintf("queued gRPC apply failed: load plan %d: %v", apply.PlanID, err), false, scope); markErr != nil { + if markErr := c.markRemoteApplyFailed(ctx, apply, nil, fmt.Sprintf("queued gRPC apply failed: load plan %d: %v", planID, err), false, scope); markErr != nil { return fmt.Errorf("mark queued gRPC apply %s failed after plan load error: %w", apply.ApplyIdentifier, markErr) } - return fmt.Errorf("load plan %d for queued gRPC apply %s: %w", apply.PlanID, apply.ApplyIdentifier, err) + return fmt.Errorf("load plan %d for queued gRPC apply %s: %w", planID, apply.ApplyIdentifier, err) } if plan == nil { - errMsg := fmt.Sprintf("queued gRPC apply failed: plan %d not found", apply.PlanID) + errMsg := fmt.Sprintf("queued gRPC apply failed: plan %d not found", planID) if markErr := c.markRemoteApplyFailed(ctx, apply, nil, errMsg, false, scope); markErr != nil { return fmt.Errorf("mark queued gRPC apply %s failed after missing plan: %w", apply.ApplyIdentifier, markErr) } diff --git a/pkg/tern/grpc_client_test.go b/pkg/tern/grpc_client_test.go index fff96f321..5f6696a26 100644 --- a/pkg/tern/grpc_client_test.go +++ b/pkg/tern/grpc_client_test.go @@ -9296,3 +9296,43 @@ func TestGRPCClient_StoppedTasklessOperationStaysStoppedWithoutAStartRequest(t * assert.True(t, state.IsState(operations.ops[operationID].State, state.ApplyOperation.Stopped), "the operation stays stopped, but is %q", operations.ops[operationID].State) } + +// A rollout member planned against its own live schema names its plan on its +// operation row, and the dispatch must run that plan rather than the apply's — +// the apply's plan describes the primary's target, so dispatching it would send +// one target's DDL to another and record the wrong plan identifier against the +// member's work. +func TestApplyTaskScopePlanID(t *testing.T) { + apply := &storage.Apply{ApplyIdentifier: "apply-1", PlanID: 10} + + t.Run("a whole-apply drive runs the apply's plan", func(t *testing.T) { + planID, err := applyTaskScope{}.planID(apply) + require.NoError(t, err) + assert.Equal(t, int64(10), planID) + }) + + t.Run("a member that shares the reviewed plan runs the apply's plan", func(t *testing.T) { + scope := applyTaskScope{operation: &storage.ApplyOperation{ID: 1, Target: "testapp-001"}} + planID, err := scope.planID(apply) + require.NoError(t, err) + assert.Equal(t, int64(10), planID) + }) + + t.Run("a member planned on its own runs its own plan", func(t *testing.T) { + scope := applyTaskScope{operation: &storage.ApplyOperation{ID: 2, Target: "testapp-002", PlanID: 11}} + planID, err := scope.planID(apply) + require.NoError(t, err) + assert.Equal(t, int64(11), planID) + }) + + t.Run("an operation with no plan on either row is not dispatchable", func(t *testing.T) { + scope := applyTaskScope{operation: &storage.ApplyOperation{ID: 3, Deployment: "eu"}} + _, err := scope.planID(&storage.Apply{ApplyIdentifier: "apply-2"}) + require.Error(t, err) + }) + + t.Run("a whole-apply drive with no plan is not dispatchable", func(t *testing.T) { + _, err := applyTaskScope{}.planID(&storage.Apply{ApplyIdentifier: "apply-3"}) + require.Error(t, err) + }) +} From 9370b76b15351ecee97d9dd9230779e93cafc4a2 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 16:16:44 -0400 Subject: [PATCH 23/34] fix(storage): accept a converged rollout member's task-less operation MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A rollout member whose own plan found nothing to change has no task to carry. Apply creation records it already completed, so the apply covers every member it addressed instead of silently omitting one — but the grouped-insert guard refused any task-less work operation and rejected the whole apply. The guard's reason is that an operation-scoped drive must never claim work it cannot find, which is a statement about operations still to be driven. It now refuses a task-less work operation only while the operation is non-terminal; a group_finalizer rebuilds its work from the plan and a terminal operation is never claimed, so neither can strand a driver by arriving empty. Co-Authored-By: Claude Opus 5 --- pkg/storage/internal/sqlstore/applies.go | 22 +++++- pkg/storage/internal/sqlstore/applies_test.go | 73 +++++++++++++++++++ 2 files changed, 91 insertions(+), 4 deletions(-) diff --git a/pkg/storage/internal/sqlstore/applies.go b/pkg/storage/internal/sqlstore/applies.go index b419e4788..39bf10166 100644 --- a/pkg/storage/internal/sqlstore/applies.go +++ b/pkg/storage/internal/sqlstore/applies.go @@ -877,6 +877,17 @@ func insertApplyTasksAndOperations(ctx context.Context, tx *rebindTx, identity i return nil } +// needsTasksToBeDrivable reports whether an operation has to carry tasks for a +// drive to have work to claim. A group_finalizer rebuilds its work from the +// plan, and an operation stored already terminal is never claimed at all, so +// neither can strand a driver by arriving empty. +func needsTasksToBeDrivable(op *storage.ApplyOperation) bool { + if op.OperationKind == storage.ApplyOperationKindGroupFinalizer { + return false + } + return !state.IsApplyOperationTerminal(op.State) +} + func insertApplyGroupedOperations(ctx context.Context, tx *rebindTx, identity identityInserter, classifier ErrorClassifier, apply *storage.Apply, applyID int64, groups []*storage.ApplyOperationWithTasks) error { if len(groups) == 0 { return fmt.Errorf("create apply %s: grouped operations are empty", apply.ApplyIdentifier) @@ -893,10 +904,13 @@ func insertApplyGroupedOperations(ctx context.Context, tx *rebindTx, identity id return fmt.Errorf("create apply %s deployment %s: grouped operation is missing its operation row", apply.ApplyIdentifier, deployment) } // A group_finalizer carries no tasks — it applies namespace-level work - // reconstructed from the plan at drive time. Every work operation must - // have at least one task so operation-scoped drives fail closed on bad - // scoping. - if len(group.Tasks) == 0 && group.Operation.OperationKind != storage.ApplyOperationKindGroupFinalizer { + // reconstructed from the plan at drive time. A rollout member whose own + // plan found nothing to change carries none either, and is recorded + // already terminal so the apply covers every member it addressed. + // Every other work operation must have at least one task, so an + // operation-scoped drive fails closed on bad scoping rather than + // claiming work it cannot find. + if len(group.Tasks) == 0 && needsTasksToBeDrivable(group.Operation) { return fmt.Errorf("create apply %s deployment %s: grouped work operation has no tasks", apply.ApplyIdentifier, deployment) } diff --git a/pkg/storage/internal/sqlstore/applies_test.go b/pkg/storage/internal/sqlstore/applies_test.go index b3f5164bf..661e86a84 100644 --- a/pkg/storage/internal/sqlstore/applies_test.go +++ b/pkg/storage/internal/sqlstore/applies_test.go @@ -535,6 +535,79 @@ func TestApplyStore_CreateWithGroupedOperationsAllowsTaskLessFinalizer(t *testin assert.Equal(t, "commerce/group_finalizer", finalizer.OperationKey) } +// A rollout member whose own plan found nothing to change has no task to carry, +// and is recorded already completed so the apply covers every member it +// addressed. CreateWithGroupedOperations must accept that converged member +// alongside its working siblings: no driver will ever claim a terminal +// operation, so it cannot be stranded by arriving empty. +func TestApplyStore_CreateWithGroupedOperationsAllowsConvergedMember(t *testing.T) { + clearTables(t) + ctx := t.Context() + store := NewMySQL(testDB) + now := time.Now() + + apply := newGroupedCreateApply(now, "apply_grouped_converged_member") + groups := []*storage.ApplyOperationWithTasks{ + newGroupedCreateGroup(now, "payments-a", "payments-001", "users"), + {Operation: &storage.ApplyOperation{ + Deployment: "payments-a", + OperationKey: "payments-002", + OperationKind: storage.ApplyOperationKindWork, + Target: "payments-002", + State: state.ApplyOperation.Completed, + CutoverPolicy: storage.CutoverPolicyRolling, + OnFailure: storage.OnFailureHalt, + StartedAt: &now, + CompletedAt: &now, + CreatedAt: now, + UpdatedAt: now, + }}, + } + + applyID, err := store.Applies().CreateWithGroupedOperations(ctx, apply, groups) + require.NoError(t, err) + + ops, err := store.ApplyOperations().ListByApply(ctx, applyID) + require.NoError(t, err) + require.Len(t, ops, 2) + var converged *storage.ApplyOperation + for _, op := range ops { + if op.Target == "payments-002" { + converged = op + } + } + require.NotNil(t, converged, "the converged member must be recorded so the apply covers it") + assert.Equal(t, state.ApplyOperation.Completed, converged.State) + assert.Equal(t, storage.ApplyOperationKindWork, converged.OperationKind) +} + +// A work operation that is still to be driven must carry its tasks. Without +// them an operation-scoped drive would claim the operation and find nothing to +// run, so it is refused at creation rather than stranding a driver. +func TestApplyStore_CreateWithGroupedOperationsRejectsPendingWorkWithoutTasks(t *testing.T) { + clearTables(t) + ctx := t.Context() + store := NewMySQL(testDB) + now := time.Now() + + apply := newGroupedCreateApply(now, "apply_grouped_pending_no_tasks") + groups := []*storage.ApplyOperationWithTasks{ + {Operation: &storage.ApplyOperation{ + Deployment: "payments-a", + OperationKey: "payments-002", + OperationKind: storage.ApplyOperationKindWork, + Target: "payments-002", + State: state.ApplyOperation.Pending, + CreatedAt: now, + UpdatedAt: now, + }}, + } + + _, err := store.Applies().CreateWithGroupedOperations(ctx, apply, groups) + require.Error(t, err) + assert.Contains(t, err.Error(), "grouped work operation has no tasks") +} + // TestApplyStore_CreateWithGroupedOperationsBlocksOverlapOnSecondaryDeployment // proves the active-apply invariant covers every deployment a fan-out apply // owns, not just the parent's primary deployment. A non-terminal apply spanning From 187943de6ea3c4ff00aa04905570868e8fb44615 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 16:23:23 -0400 Subject: [PATCH 24/34] fix(cli): filter and name a multi-target pull's divergence report A --table filter is a projection over the whole pulled response, so the divergence report has to be narrowed with the schema it describes. Unfiltered it named divergence in tables the operator did not ask about, beside a schema that no longer contained them, and each target's count described a different table set than the header. Two targets of one pull can also share a name, since a target name is only unique within the deployment that addresses it. Name those by member so no two rows of the report read alike, and leave the bare name where it is already unambiguous. Co-Authored-By: Claude Opus 5 --- pkg/api/pull_members_test.go | 23 ++++++++++ pkg/cmd/commands/pull.go | 41 +++++++++++++++++ pkg/cmd/commands/pull_test.go | 51 +++++++++++++++++++++ pkg/cmd/internal/templates/pull.go | 43 ++++++++++++++---- pkg/cmd/internal/templates/pull_test.go | 60 +++++++++++++++++++++++++ 5 files changed, 210 insertions(+), 8 deletions(-) diff --git a/pkg/api/pull_members_test.go b/pkg/api/pull_members_test.go index 7f5099db8..d01d2ca0b 100644 --- a/pkg/api/pull_members_test.go +++ b/pkg/api/pull_members_test.go @@ -128,6 +128,29 @@ func TestExecutePullSchema_ReportsPerTargetDivergence(t *testing.T) { }, diverged.DivergedTables) } +// A table the primary holds and another target does not is the opposite +// one-sided case from an extra table, and the two must never be confused: one +// says the target is behind the schema being materialized, the other says it +// carries something the primary does not. +func TestExecutePullSchema_ReportsTablesMissingFromATarget(t *testing.T) { + client := newPerTargetPullClient(map[string]*ternv1.PullSchemaResponse{ + "testapp-001": pulledTables(map[string]string{"users": pullUsersDDL, "audits": pullAuditsDDL}), + "testapp-002": pulledTables(map[string]string{"users": pullUsersDDL}), + }, nil) + svc := pullTargetService(t, multiTargetPullEnv(), map[string]tern.Client{"eu/production": client}) + + resp, err := svc.ExecutePullSchema(t.Context(), pullRequest()) + require.NoError(t, err) + + require.Len(t, resp.Targets, 2) + diverged := resp.Targets[1] + assert.Equal(t, "testapp-002", diverged.Target) + assert.Equal(t, int32(1), diverged.TableCount) + assert.Equal(t, []apitypes.DivergedTable{ + {Namespace: "testapp", Table: "audits", Difference: apitypes.DivergenceOnlyOnPrimary}, + }, diverged.DivergedTables) +} + // A caller reconciling an environment against its own shard inventory reads the // member set off the pull payload, so exactly one target is marked primary and // the whole configured list is present in configuration order. diff --git a/pkg/cmd/commands/pull.go b/pkg/cmd/commands/pull.go index f43497a79..dfaeeb2da 100644 --- a/pkg/cmd/commands/pull.go +++ b/pkg/cmd/commands/pull.go @@ -130,9 +130,50 @@ func filterPullSchemaTables(resp *apitypes.PullSchemaResponse, filter string) er } } resp.TableCount = kept + filterPullTargetDivergence(resp.Targets, needle, kept) return nil } +// filterPullTargetDivergence narrows a multi-target pull's divergence report to +// the same tables the filter kept, so the report describes the schema printed +// beside it rather than the whole database. Left unfiltered it would name +// divergence in tables the operator did not ask about, while the header and the +// pulled DDL described a different set. +// +// Each target's table count is recomputed over what remains, which the filtered +// response already determines: a target holds the tables the primary holds, +// less the ones only the primary has and plus the ones only it has. +func filterPullTargetDivergence(targets []*apitypes.TargetDivergence, needle string, primaryKept int32) { + for _, target := range targets { + if target.Primary { + target.TableCount = primaryKept + continue + } + kept := make([]apitypes.DivergedTable, 0, len(target.DivergedTables)) + count := primaryKept + for _, table := range target.DivergedTables { + if !strings.Contains(strings.ToLower(table.Table), needle) { + continue + } + kept = append(kept, table) + switch table.Difference { + case apitypes.DivergenceOnlyOnPrimary: + count-- + case apitypes.DivergenceOnlyOnTarget: + count++ + } + } + if len(kept) == 0 { + // Absent rather than empty: a target with nothing left to report + // holds the same filtered schema as the primary, which is what an + // unset list already says. + kept = nil + } + target.DivergedTables = kept + target.TableCount = count + } +} + // errNoTableMatches names the filter, the database, and the environment, and // lists the pulled table names when the list is short enough to read, so a // typo is a one-round-trip fix. diff --git a/pkg/cmd/commands/pull_test.go b/pkg/cmd/commands/pull_test.go index 1bb7d0e97..d14cc56d3 100644 --- a/pkg/cmd/commands/pull_test.go +++ b/pkg/cmd/commands/pull_test.go @@ -135,6 +135,57 @@ func TestFilterPullSchemaTablesKeepsSubstringMatches(t *testing.T) { assert.Equal(t, int32(2), resp.TableCount) } +// A table filter narrows what the pull describes, so the divergence report has +// to be narrowed with it. Left alone it would name divergence in tables the +// operator did not ask about, beside a schema that no longer contains them, and +// each target's count would describe a different table set than the header. +func TestFilterPullSchemaTablesNarrowsTargetDivergence(t *testing.T) { + resp := &apitypes.PullSchemaResponse{ + Database: "orders", + Environment: "production", + Namespaces: map[string]*apitypes.PulledNamespace{ + "orders": { + Tables: map[string]string{ + "users": "CREATE TABLE `users` (`id` bigint NOT NULL);", + "user_settings": "CREATE TABLE `user_settings` (`id` bigint NOT NULL);", + "payments": "CREATE TABLE `payments` (`id` bigint NOT NULL);", + }, + }, + }, + TableCount: 3, + Targets: []*apitypes.TargetDivergence{ + {Deployment: "eu", Target: "orders-001", TableCount: 3, Primary: true}, + {Deployment: "eu", Target: "orders-002", TableCount: 3, DivergedTables: []apitypes.DivergedTable{ + {Namespace: "orders", Table: "users", Difference: apitypes.DivergenceDiffers}, + {Namespace: "orders", Table: "user_settings", Difference: apitypes.DivergenceOnlyOnPrimary}, + {Namespace: "orders", Table: "receipts", Difference: apitypes.DivergenceOnlyOnTarget}, + {Namespace: "orders", Table: "payments", Difference: apitypes.DivergenceDiffers}, + }}, + {Deployment: "eu", Target: "orders-003", TableCount: 3, DivergedTables: []apitypes.DivergedTable{ + {Namespace: "orders", Table: "payments", Difference: apitypes.DivergenceDiffers}, + }}, + }, + } + + require.NoError(t, filterPullSchemaTables(resp, "user")) + + require.Len(t, resp.Targets, 3) + assert.Equal(t, int32(2), resp.Targets[0].TableCount, "the primary's count is the filtered schema's") + + diverged := resp.Targets[1] + assert.Equal(t, []apitypes.DivergedTable{ + {Namespace: "orders", Table: "users", Difference: apitypes.DivergenceDiffers}, + {Namespace: "orders", Table: "user_settings", Difference: apitypes.DivergenceOnlyOnPrimary}, + }, diverged.DivergedTables, "divergence in tables the filter excluded must not be reported") + assert.Equal(t, int32(1), diverged.TableCount, + "the target holds the filtered tables the primary holds, less the one only the primary has") + + converged := resp.Targets[2] + assert.Nil(t, converged.DivergedTables, + "a target whose only divergence was filtered out holds the same filtered schema as the primary") + assert.Equal(t, int32(2), converged.TableCount) +} + // A filtered pull that requested lint keeps the explicit empty audit when the // selected tables are clean, so "no violations" stays distinguishable from // lint not being requested. diff --git a/pkg/cmd/internal/templates/pull.go b/pkg/cmd/internal/templates/pull.go index 3e99a1e92..bd71a7041 100644 --- a/pkg/cmd/internal/templates/pull.go +++ b/pkg/cmd/internal/templates/pull.go @@ -69,6 +69,32 @@ var divergenceLabels = map[string]string{ apitypes.DivergenceOnlyOnTarget: "extra", } +// targetDivergenceNames names each target of a pull so no two members of the +// report read alike. A target name is opaque to SchemaBot and only unique +// within the deployment that addresses it, so an environment whose deployments +// address targets of the same name would otherwise print the same header twice +// and leave the operator unable to tell which member a divergence belongs to. +// Such a target is named by its full member identity; every other target keeps +// the bare name, which is already the thing that distinguishes it. +func targetDivergenceNames(targets []*apitypes.TargetDivergence) []string { + deploymentsByTarget := make(map[string]map[string]struct{}, len(targets)) + for _, target := range targets { + if deploymentsByTarget[target.Target] == nil { + deploymentsByTarget[target.Target] = make(map[string]struct{}, 1) + } + deploymentsByTarget[target.Target][target.Deployment] = struct{}{} + } + names := make([]string, len(targets)) + for i, target := range targets { + if len(deploymentsByTarget[target.Target]) > 1 { + names[i] = target.Deployment + "/" + target.Target + continue + } + names[i] = target.Target + } + return names +} + // writeTargetDivergence lists every target the environment addresses and how // each differs from the primary, whose schema is the DDL printed below. It // renders as "--" comments like the rest of the pull output, so redirecting a @@ -85,22 +111,23 @@ func writeTargetDivergence(targets []*apitypes.TargetDivergence) { if len(targets) == 0 { return } - for _, target := range targets { + names := targetDivergenceNames(targets) + for i, target := range targets { + name := emphasis("`" + names[i] + "`") fmt.Println() if target.Primary { - fmt.Println(annotation(fmt.Sprintf("-- Target %s — primary target, whose schema is below", - emphasis("`"+target.Target+"`")))) + fmt.Println(annotation(fmt.Sprintf("-- Target %s — primary target, whose schema is below", name))) continue } if len(target.DivergedTables) == 0 { - fmt.Println(annotation(fmt.Sprintf("-- Target %s — same schema as the primary target", - emphasis("`"+target.Target+"`")))) + fmt.Println(annotation(fmt.Sprintf("-- Target %s — same schema as the primary target", name))) continue } - fmt.Println(annotation(fmt.Sprintf("-- Target %s — %d %s differ from the primary target", - emphasis("`"+target.Target+"`"), + fmt.Println(annotation(fmt.Sprintf("-- Target %s — %d %s %s from the primary target", + name, len(target.DivergedTables), - ui.Pluralize("table", len(target.DivergedTables))))) + ui.Pluralize("table", len(target.DivergedTables)), + ui.PluralizeLabel("differs", "differ", len(target.DivergedTables))))) for _, table := range target.DivergedTables { label, ok := divergenceLabels[table.Difference] if !ok { diff --git a/pkg/cmd/internal/templates/pull_test.go b/pkg/cmd/internal/templates/pull_test.go index 8edebb5d9..3f8339fa5 100644 --- a/pkg/cmd/internal/templates/pull_test.go +++ b/pkg/cmd/internal/templates/pull_test.go @@ -273,6 +273,66 @@ func TestWritePullSchema_RendersPerTargetDivergence(t *testing.T) { } } +// The two one-sided differences read as opposites from the primary's point of +// view: a table only the primary holds is missing from the other target, and a +// table only the other target holds is extra. Rendering either label for the +// other would tell the operator to change the wrong schema. +func TestWritePullSchema_RendersBothOneSidedDifferences(t *testing.T) { + setColors(t, false) + out := captureStdout(t, func() { + WritePullSchema(&apitypes.PullSchemaResponse{ + Database: "orders-db", + Type: "mysql", + Environment: "production", + TableCount: 2, + Namespaces: map[string]*apitypes.PulledNamespace{ + "orders": {Tables: map[string]string{"users": "CREATE TABLE `users` (`id` bigint NOT NULL);\n"}}, + }, + Targets: []*apitypes.TargetDivergence{ + {Deployment: "eu", Target: "orders-001", TableCount: 2, Primary: true}, + {Deployment: "eu", Target: "orders-002", TableCount: 2, DivergedTables: []apitypes.DivergedTable{ + {Namespace: "orders", Table: "audits", Difference: apitypes.DivergenceOnlyOnPrimary}, + {Namespace: "orders", Table: "receipts", Difference: apitypes.DivergenceOnlyOnTarget}, + }}, + }, + }) + }) + + assert.Contains(t, out, "-- orders.audits: missing") + assert.Contains(t, out, "-- orders.receipts: extra") +} + +// A target name is opaque and only unique within its deployment, so two +// deployments addressing the same name would render the same header twice. Such +// a target is named by its full member identity so the operator can tell which +// member a divergence belongs to. +func TestWritePullSchema_NamesTargetsThatShareANameByMember(t *testing.T) { + setColors(t, false) + out := captureStdout(t, func() { + WritePullSchema(&apitypes.PullSchemaResponse{ + Database: "orders-db", + Type: "mysql", + Environment: "production", + TableCount: 1, + Namespaces: map[string]*apitypes.PulledNamespace{ + "orders": {Tables: map[string]string{"users": "CREATE TABLE `users` (`id` bigint NOT NULL);\n"}}, + }, + Targets: []*apitypes.TargetDivergence{ + {Deployment: "eu", Target: "orders", TableCount: 1, Primary: true}, + {Deployment: "us", Target: "orders", TableCount: 1, DivergedTables: []apitypes.DivergedTable{ + {Namespace: "orders", Table: "users", Difference: apitypes.DivergenceDiffers}, + }}, + {Deployment: "ap", Target: "orders-ap", TableCount: 1}, + }, + }) + }) + + assert.Contains(t, out, "-- Target `eu/orders` — primary target, whose schema is below") + assert.Contains(t, out, "-- Target `us/orders` — 1 table differs from the primary target") + assert.Contains(t, out, "-- Target `orders-ap` — same schema as the primary target", + "a target whose name is already unique keeps the bare name") +} + // An environment whose targets are expected to hold the same schema carries no // divergence, and a pull of it renders exactly as it always did. func TestWritePullSchema_NoDivergenceSectionWithoutTargets(t *testing.T) { From bc0ffd460e00b905bb8a563f735087fb44006e5d Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 16:26:58 -0400 Subject: [PATCH 25/34] fix(github): render every rollout member name as a code span A member name is assembled from server config, so it reaches the plan comment as text SchemaBot did not choose. A name carrying a backtick closed the span it sat in, and one carrying a line break started markdown of its own in a comment operators act on. Every name in the drift rollup now goes through the same span renderer the rest of the comment's identifiers use, which also gives the clean line the same vocabulary as the per-member breakdown beside it. Co-Authored-By: Claude Opus 5 --- TEMPLATES.md | 2 +- pkg/webhook/templates/plan.go | 8 +++++-- pkg/webhook/templates/plan_drift_test.go | 30 ++++++++++++++++++++++-- 3 files changed, 35 insertions(+), 5 deletions(-) diff --git a/TEMPLATES.md b/TEMPLATES.md index 067b5d290..54f2bb039 100644 --- a/TEMPLATES.md +++ b/TEMPLATES.md @@ -1289,7 +1289,7 @@ schemabot apply -e production *Requested by @jackjackbits at 2026-01-01 00:00:00 UTC · planned from [`abcdef1`](https://github.com/block/schemabot/commit/abcdef1234567890abcdef1234567890abcdef12)* -✅ **Same plan on all 3 deployments** (eu, au, us). +✅ **Same plan on all 3 deployments** (`eu`, `au`, `us`). ```sql CREATE TABLE `users` ( diff --git a/pkg/webhook/templates/plan.go b/pkg/webhook/templates/plan.go index 92228268f..d7b270d32 100644 --- a/pkg/webhook/templates/plan.go +++ b/pkg/webhook/templates/plan.go @@ -1131,7 +1131,11 @@ func writeDeploymentDrift(sb *strings.Builder, drift *DeploymentDriftData) { return } - names := driftMemberNames(drift.Deployments) + // A member name is assembled from server config, so it reaches this comment + // as text SchemaBot did not choose. Rendering every one as a code span keeps + // a name carrying a backtick or a line break from closing the span it sits + // in and writing markdown of its own into a comment operators act on. + names := inlineCodeList(driftMemberNames(drift.Deployments)) if drift.Clean { if drift.Independent { fmt.Fprintf(sb, "✅ **Planned separately for all %d targets** (%s) — each target holds its own schema, so their plans are not expected to match.\n\n", @@ -1153,7 +1157,7 @@ func writeDeploymentDrift(sb *strings.Builder, drift *DeploymentDriftData) { sb.WriteString(glyph.Attention + " **Deployment drift detected** — some deployments no longer match the reviewed plan, so the plan check is failing closed:\n\n") } for i, d := range drift.Deployments { - name := "`" + names[i] + "`" + name := names[i] if d.Primary { name += " (primary)" } diff --git a/pkg/webhook/templates/plan_drift_test.go b/pkg/webhook/templates/plan_drift_test.go index ed7f45ead..b3112d773 100644 --- a/pkg/webhook/templates/plan_drift_test.go +++ b/pkg/webhook/templates/plan_drift_test.go @@ -30,7 +30,7 @@ func TestRenderPlanComment_DriftCleanShowsUniformLine(t *testing.T) { out := RenderPlanComment(data) assert.Contains(t, out, "Same plan on all 3 deployments") - assert.Contains(t, out, "eu, au, us") + assert.Contains(t, out, "`eu`, `au`, `us`") } // A diverged deployment is named with a compact change summary, and an errored @@ -295,6 +295,32 @@ func TestRenderPlanComment_DriftCleanNamesMultiTargetMembers(t *testing.T) { out := RenderPlanComment(data) assert.Contains(t, out, "Planned separately for all 3 targets") - assert.Contains(t, out, "primary/testapp-001, primary/testapp-002, eu-west") + assert.Contains(t, out, "`primary/testapp-001`, `primary/testapp-002`, `eu-west`") assert.True(t, strings.Contains(out, "each target holds its own schema")) } + +// A member name reaches the comment from server config, so the rollup renders +// it as a code span it cannot break out of: a name carrying a backtick or a +// line break stays one readable name on one line instead of closing its span +// and writing markdown into a comment operators act on. +func TestRenderPlanComment_DriftContainsHostileMemberNames(t *testing.T) { + data := PlanCommentData{ + Database: "testapp", Environment: "production", IsMySQL: true, + Changes: []KeyspaceChangeData{{ + Keyspace: "testapp", + Statements: []string{"ALTER TABLE `users` ADD COLUMN `email` varchar(255)"}, + }}, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, + Clean: false, + Deployments: []DeploymentDriftEntry{ + {Deployment: "eu", Primary: true, Class: "match"}, + {Deployment: "us`\n## Injected", Class: "diverged", Detail: "1 unexpected change(s) vs the reviewed plan"}, + }, + }, + } + + out := RenderPlanComment(data) + assert.NotContains(t, out, "\n## Injected", "a name must not start a heading of its own") + assert.Contains(t, out, "`` us` ## Injected ``") +} From 9135242dc02385cfe39d37dcad1c050b5d5a48e0 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 16:31:38 -0400 Subject: [PATCH 26/34] fix(cli): attribute preview progress rows to their target A rollout's table rows are matched to the member running them by deployment and target together, so preview fixtures that named only a deployment matched no member and their per-table progress dropped out of the rendered multi-deployment preview entirely. Naming each fixture row's target restores the rows the previews exist to show, in TEMPLATES.md as well as the TUI snapshot. Also documents `target` on the progress endpoint's table entries, and covers it in the local-storage projection test, so a caller reads a table's member from the pair rather than the deployment alone. Co-Authored-By: Claude Opus 5 --- TEMPLATES.md | 63 +++++++++++++++++++ docs/schema-intelligence.md | 55 +++++++++++++++- pkg/api/handlers_test.go | 4 ++ pkg/cmd/commands/preview_tui.go | 6 +- .../templates/preview_progress_multi.go | 24 +++---- 5 files changed, 136 insertions(+), 16 deletions(-) diff --git a/TEMPLATES.md b/TEMPLATES.md index 8ecb70885..54f2bb039 100644 --- a/TEMPLATES.md +++ b/TEMPLATES.md @@ -8000,10 +8000,23 @@ This schema change was cancelled and cannot be resumed. Open a new schema change 🟢 us-east — ready for cutover — next in order (orders-us-east) + ~ orders: 🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨 Waiting for cutover + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + 🔄 eu-west — running table copy (orders-eu-west) + ~ orders: 🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ 35.00% + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + • Rows: 42,000 / 120,000 · ETA: 4m 0s + + ⏳ ap-south — waiting for eu-west (orders-ap-south) + ~ orders: ⏳ Queued + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ``` @@ -8030,11 +8043,23 @@ This schema change was cancelled and cannot be resumed. Open a new schema change ✅ us-east — completed (orders-us-east) + ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ❌ eu-west — failed (orders-eu-west) duplicate key name 'idx_orders_source' + ~ orders: ⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ ❌ Failed + ALTER TABLE `orders` ADD INDEX `idx_orders_source`(`source`); + + ⏸️ ap-south — halted — eu-west failed (orders-ap-south) + ~ orders: 🚫 Cancelled (not started) + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ``` @@ -8062,10 +8087,23 @@ This schema change was cancelled and cannot be resumed. Open a new schema change ❌ us-east — failed (orders-us-east) duplicate key name 'idx_orders_source' + ~ orders: ⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ ❌ Failed + ALTER TABLE `orders` ADD INDEX `idx_orders_source`(`source`); + + 🔄 eu-west — running table copy (orders-eu-west) + ~ orders: 🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ 35.00% + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + • Rows: 42,000 / 120,000 · ETA: 4m 0s + + ⏸️ ap-south — halted — us-east failed (orders-ap-south) + ~ orders: ⏳ Queued + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ``` @@ -8088,10 +8126,22 @@ This schema change was cancelled and cannot be resumed. Open a new schema change ✅ us-east — completed (orders-us-east) + ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ✅ eu-west — completed (orders-eu-west) + ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ✅ ap-south — completed (orders-ap-south) + ~ orders: 🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩🟩 ✓ Complete + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ``` @@ -8688,12 +8738,25 @@ Environment: production External operation ID: remote-op-us-east-001 External apply ID: remote-apply-us-east-001 + ~ orders: 🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨🟨 Waiting for cutover + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + 🔄 eu-west — running table copy (orders-eu-west) External operation ID: remote-op-eu-west-001 External apply ID: remote-apply-eu-west-001 + ~ orders: 🟦🟦🟦🟦🟦🟦🟦⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜⬜ 35.42% + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + • Rows: 42,500 / 120,000 · ETA: 4m 0s + + ⏳ ap-south — waiting for eu-west (orders-ap-south) + ~ orders: ⏳ Queued + ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL; + + ESC to detach ``` diff --git a/docs/schema-intelligence.md b/docs/schema-intelligence.md index e4a88f405..d804d0022 100644 --- a/docs/schema-intelligence.md +++ b/docs/schema-intelligence.md @@ -571,7 +571,12 @@ from the server's build phase and its counters; it stays below 100 until the apply completes and holds its last value between phases (see [postgresql.md](postgresql.md)). Sharded engines can add per-shard progress, and multi-deployment applies list -operations with their deployment, target, state, and cutover policy. +operations with their deployment, target, state, and cutover policy. A rollout's +table entries carry `deployment` and `target`, naming the member whose copy the +row reports. Attribute a table by the pair, never by `deployment` alone: one +deployment can address several targets, each running its own copy of the change, +so several rows for the same table share a deployment and differ only in their +target. Both fields are absent on an apply that runs against a single target. The top-level `metadata` object carries engine-specific display fields when the engine reports them: PostgreSQL applies report their position through `phase`, `step`, `steps_total`, and `statement`; PlanetScale applies report deploy @@ -612,6 +617,54 @@ Response excerpt (illustrative values): +
+Multi-target rollout response example + +```http +GET /api/progress/apply/apply-example-75 +``` + +Response excerpt (illustrative values): + +```json +{ + "apply_id": "apply-example-75", + "database": "shop", + "environment": "production", + "engine": "spirit", + "state": "running", + "operations": [ + {"deployment": "commerce-a", "target": "shop-001", "state": "completed", "cutover_policy": "rolling"}, + {"deployment": "commerce-a", "target": "shop-002", "state": "running", "cutover_policy": "rolling"} + ], + "tables": [ + { + "table_name": "orders", + "deployment": "commerce-a", + "target": "shop-001", + "ddl": "ALTER TABLE `orders` ADD INDEX `idx_status` (`status`)", + "status": "completed", + "percent_complete": 100 + }, + { + "table_name": "orders", + "deployment": "commerce-a", + "target": "shop-002", + "ddl": "ALTER TABLE `orders` ADD INDEX `idx_status` (`status`)", + "status": "running", + "rows_copied": 2000000, + "rows_total": 8000000, + "percent_complete": 25 + } + ] +} +``` + +Both rows report the same table under the same deployment, and only `target` +tells them apart. + +
+
PostgreSQL response example diff --git a/pkg/api/handlers_test.go b/pkg/api/handlers_test.go index d2e8254a9..532221039 100644 --- a/pkg/api/handlers_test.go +++ b/pkg/api/handlers_test.go @@ -4380,6 +4380,10 @@ func TestProgressFromLocalStorageIncludesOperationProgressAndTableDeployment(t * require.Len(t, resp.Tables, 2) assert.Equal(t, "deploy-a", resp.Tables[0].Deployment) assert.Equal(t, "deploy-b", resp.Tables[1].Deployment) + // One deployment can address several targets, so the deployment alone does + // not say which target a table's copy is running against. + assert.Equal(t, "target-a", resp.Tables[0].Target) + assert.Equal(t, "target-b", resp.Tables[1].Target) } func newActiveProgressServiceWithOperations(client tern.Client, apply *storage.Apply, operations storage.ApplyOperationStore) *Service { diff --git a/pkg/cmd/commands/preview_tui.go b/pkg/cmd/commands/preview_tui.go index 76a89eb4c..7c3d540ee 100644 --- a/pkg/cmd/commands/preview_tui.go +++ b/pkg/cmd/commands/preview_tui.go @@ -68,9 +68,9 @@ var tuiPreviewScenarios = map[string]tuiPreviewScenario{ {Deployment: "ap-south", Target: "orders-ap-south", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyBarrier, OnFailure: storage.OnFailureHalt}, }, Tables: []*apitypes.TableProgressResponse{ - {Deployment: "us-east", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.WaitingForCutover, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, - {Deployment: "eu-west", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Running, RowsCopied: 42000 + poll*500, RowsTotal: 120000, PercentComplete: 35, ETASeconds: 240}, - {Deployment: "ap-south", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, + {Deployment: "us-east", Target: "orders-us-east", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.WaitingForCutover, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, + {Deployment: "eu-west", Target: "orders-eu-west", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Running, RowsCopied: 42000 + poll*500, RowsTotal: 120000, PercentComplete: 35, ETASeconds: 240}, + {Deployment: "ap-south", Target: "orders-ap-south", TableName: "orders", ChangeType: "alter", DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, }, } }, diff --git a/pkg/cmd/internal/templates/preview_progress_multi.go b/pkg/cmd/internal/templates/preview_progress_multi.go index 0c7061391..f5f3a824f 100644 --- a/pkg/cmd/internal/templates/preview_progress_multi.go +++ b/pkg/cmd/internal/templates/preview_progress_multi.go @@ -15,9 +15,9 @@ func previewCLIMultiDeploymentApplyInProgress() { {Deployment: "eu-west", Target: "orders-eu-west", State: state.ApplyOperation.Running, CutoverPolicy: storage.CutoverPolicyBarrier, OnFailure: storage.OnFailureHalt}, {Deployment: "ap-south", Target: "orders-ap-south", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyBarrier, OnFailure: storage.OnFailureHalt}, }, []TableProgress{ - {Deployment: "us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.WaitingForCutover, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, - {Deployment: "eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Running, RowsCopied: 42000, RowsTotal: 120000, PercentComplete: 35, ETASeconds: 240}, - {Deployment: "ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, + {Deployment: "us-east", Target: "orders-us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.WaitingForCutover, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, + {Deployment: "eu-west", Target: "orders-eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Running, RowsCopied: 42000, RowsTotal: 120000, PercentComplete: 35, ETASeconds: 240}, + {Deployment: "ap-south", Target: "orders-ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, })) } @@ -27,9 +27,9 @@ func previewCLIMultiDeploymentApplyFailed() { {Deployment: "eu-west", Target: "orders-eu-west", State: state.ApplyOperation.Failed, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt, ErrorMessage: "duplicate key name 'idx_orders_source'"}, {Deployment: "ap-south", Target: "orders-ap-south", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, }, []TableProgress{ - {Deployment: "us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, - {Deployment: "eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD INDEX `idx_orders_source` (`source`)", Status: state.Task.Failed, RowsCopied: 0, RowsTotal: 120000, PercentComplete: 0}, - {Deployment: "ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: TaskCancelled}, + {Deployment: "us-east", Target: "orders-us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, + {Deployment: "eu-west", Target: "orders-eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD INDEX `idx_orders_source` (`source`)", Status: state.Task.Failed, RowsCopied: 0, RowsTotal: 120000, PercentComplete: 0}, + {Deployment: "ap-south", Target: "orders-ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: TaskCancelled}, })) } @@ -39,9 +39,9 @@ func previewCLIMultiDeploymentApplyHaltedWithLiveSibling() { {Deployment: "eu-west", Target: "orders-eu-west", State: state.ApplyOperation.Running, CutoverPolicy: storage.CutoverPolicyBarrier, OnFailure: storage.OnFailureHalt}, {Deployment: "ap-south", Target: "orders-ap-south", State: state.ApplyOperation.Pending, CutoverPolicy: storage.CutoverPolicyBarrier, OnFailure: storage.OnFailureHalt}, }, []TableProgress{ - {Deployment: "us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD INDEX `idx_orders_source` (`source`)", Status: state.Task.Failed, RowsCopied: 0, RowsTotal: 80000, PercentComplete: 0}, - {Deployment: "eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Running, RowsCopied: 42000, RowsTotal: 120000, PercentComplete: 35, ETASeconds: 240}, - {Deployment: "ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, + {Deployment: "us-east", Target: "orders-us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD INDEX `idx_orders_source` (`source`)", Status: state.Task.Failed, RowsCopied: 0, RowsTotal: 80000, PercentComplete: 0}, + {Deployment: "eu-west", Target: "orders-eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Running, RowsCopied: 42000, RowsTotal: 120000, PercentComplete: 35, ETASeconds: 240}, + {Deployment: "ap-south", Target: "orders-ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Pending}, })) } @@ -51,9 +51,9 @@ func previewCLIMultiDeploymentApplyCompleted() { {Deployment: "eu-west", Target: "orders-eu-west", State: state.ApplyOperation.Completed, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, {Deployment: "ap-south", Target: "orders-ap-south", State: state.ApplyOperation.Completed, CutoverPolicy: storage.CutoverPolicyRolling, OnFailure: storage.OnFailureHalt}, }, []TableProgress{ - {Deployment: "us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, - {Deployment: "eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 120000, RowsTotal: 120000, PercentComplete: 100}, - {Deployment: "ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 60000, RowsTotal: 60000, PercentComplete: 100}, + {Deployment: "us-east", Target: "orders-us-east", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 80000, RowsTotal: 80000, PercentComplete: 100}, + {Deployment: "eu-west", Target: "orders-eu-west", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 120000, RowsTotal: 120000, PercentComplete: 100}, + {Deployment: "ap-south", Target: "orders-ap-south", TableName: "orders", ChangeType: "alter", Dialect: schema.DialectMySQL, DDL: "ALTER TABLE `orders` ADD COLUMN `source` varchar(32) DEFAULT NULL", Status: state.Task.Completed, RowsCopied: 60000, RowsTotal: 60000, PercentComplete: 100}, }) data.CompletedAt = previewTime.Add(-1 * time.Minute).Format(time.RFC3339) WriteProgress(data) From d15d638e1811f963d71368ed70f625f434cef9a1 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 16:42:31 -0400 Subject: [PATCH 27/34] fix(tern): drive every operation against the plan it names The local drive and the two remaining remote paths loaded the apply's plan whatever they were driving, so an operation-scoped drive of a member planned against its own live schema would run the reviewed primary's DDL against that member's target and record the primary's plan against its work. Every drive now resolves its plan through storage.PlanIDForOperation: an operation runs the plan its own row names, a whole-apply drive runs the apply's, and a drive that can resolve neither errors rather than falling back to a plan describing another target's schema. Co-Authored-By: Claude Opus 5 --- pkg/tern/grpc_client.go | 20 +++++++--- pkg/tern/local_control_resume.go | 53 ++++++++++++++++++++++----- pkg/tern/local_control_resume_test.go | 46 ++++++++++++++++++++--- 3 files changed, 98 insertions(+), 21 deletions(-) diff --git a/pkg/tern/grpc_client.go b/pkg/tern/grpc_client.go index 71e9a1d8d..4355d61c1 100644 --- a/pkg/tern/grpc_client.go +++ b/pkg/tern/grpc_client.go @@ -2195,14 +2195,18 @@ func (c *GRPCClient) ResumeApplyOperation(ctx context.Context, apply *storage.Ap // modelled as a task row. Dispatch it as a VSchema-only apply, which the // data plane applies via its own task-less VSchema-only path, mirroring // LocalClient.ResumeApplyOperation. - plan, err := c.storage.Plans().GetByID(ctx, apply.PlanID) + planID, err := scope.planID(apply) if err != nil { - return fmt.Errorf("load plan %d for task-less apply_operation %d (apply %s): %w", apply.PlanID, applyOperationID, apply.ApplyIdentifier, err) + return fmt.Errorf("resolve plan for task-less apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, err) + } + plan, err := c.storage.Plans().GetByID(ctx, planID) + if err != nil { + return fmt.Errorf("load plan %d for task-less apply_operation %d (apply %s): %w", planID, applyOperationID, apply.ApplyIdentifier, err) } // A missing plan row is its own cause, separate from a claim that resolved // to the wrong operation, so name it rather than reporting a stale claim. if plan == nil { - return fmt.Errorf("plan %d for task-less apply_operation %d (apply %s): %w", apply.PlanID, applyOperationID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) + return fmt.Errorf("plan %d for task-less apply_operation %d (apply %s): %w", planID, applyOperationID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) } // Fail closed before any dispatch or state mutation on every other // task-less work shape: it is an invalid or stale claim. The shared resume @@ -2239,12 +2243,16 @@ func (c *GRPCClient) dispatchRemoteGroupFinalizer(ctx context.Context, apply *st if namespace == "" && op.OperationKey != finalizerDeploymentScopedKey { return fmt.Errorf("group_finalizer apply_operation %d (apply %s): malformed operation key %q", op.ID, apply.ApplyIdentifier, op.OperationKey) } - plan, err := c.storage.Plans().GetByID(ctx, apply.PlanID) + planID, err := scope.planID(apply) + if err != nil { + return fmt.Errorf("resolve plan for group_finalizer apply_operation %d (apply %s): %w", op.ID, apply.ApplyIdentifier, err) + } + plan, err := c.storage.Plans().GetByID(ctx, planID) if err != nil { - return fmt.Errorf("load plan %d for group_finalizer apply_operation %d (apply %s): %w", apply.PlanID, op.ID, apply.ApplyIdentifier, err) + return fmt.Errorf("load plan %d for group_finalizer apply_operation %d (apply %s): %w", planID, op.ID, apply.ApplyIdentifier, err) } if plan == nil { - return fmt.Errorf("plan %d for group_finalizer apply_operation %d (apply %s): %w", apply.PlanID, op.ID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) + return fmt.Errorf("plan %d for group_finalizer apply_operation %d (apply %s): %w", planID, op.ID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) } // Fail closed if the operation's scope carries no VSchema artifact, // mirroring the local finalizer drive. diff --git a/pkg/tern/local_control_resume.go b/pkg/tern/local_control_resume.go index 979708501..13927974c 100644 --- a/pkg/tern/local_control_resume.go +++ b/pkg/tern/local_control_resume.go @@ -1517,7 +1517,7 @@ func (c *LocalClient) ResumeApply(ctx context.Context, apply *storage.Apply) err // apply is handled inside the shared resume path: VSchema-only plans are // re-driven so the VSchema is applied, and any other task-less shape (e.g. a // sharded dispatch whose shard already matches) completes as a no-op. - return c.resumeApplyWithTasks(ctx, apply, tasks, apply.GetOptions().Map(), false, false) + return c.resumeApplyWithTasks(ctx, apply, nil, tasks, apply.GetOptions().Map(), false, false) } // ResumeApplyOperation starts or resumes a single apply_operation (one @@ -1555,19 +1555,23 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A if op.OperationKind == storage.ApplyOperationKindGroupFinalizer { return c.driveGroupFinalizer(ctx, apply, op) } - plan, planErr := c.storage.Plans().GetByID(ctx, apply.PlanID) + planID, planErr := storage.PlanIDForOperation(apply, op) + if planErr != nil { + return fmt.Errorf("resolve plan for task-less apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, planErr) + } + plan, planErr := c.storage.Plans().GetByID(ctx, planID) if planErr != nil { return fmt.Errorf("get plan for task-less apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, planErr) } // A missing plan row is its own cause, separate from a claim that resolved // to the wrong operation, so name it rather than reporting a stale claim. if plan == nil { - return fmt.Errorf("plan %d for task-less apply_operation %d (apply %s): %w", apply.PlanID, applyOperationID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) + return fmt.Errorf("plan %d for task-less apply_operation %d (apply %s): %w", planID, applyOperationID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) } if !op.IsTasklessVSchemaOnlyWork(plan) { return fmt.Errorf("apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, ErrNoTasksForApplyOperation) } - return c.resumeApplyWithTasks(ctx, apply, tasks, apply.GetOptions().Map(), false, false) + return c.resumeApplyWithTasks(ctx, apply, op, tasks, apply.GetOptions().Map(), false, false) } siblings, err := c.storage.ApplyOperations().ListByApply(ctx, apply.ID) if err != nil { @@ -1580,7 +1584,7 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A // unchanged. releaseAtCutoverBarrier := shouldReleaseAtCutoverBarrier(apply, multiOperation, op) options := effectiveCopyDriveOptions(apply, multiOperation, op).Map() - return c.resumeApplyWithTasks(ctx, apply, tasks, options, releaseAtCutoverBarrier, false) + return c.resumeApplyWithTasks(ctx, apply, op, tasks, options, releaseAtCutoverBarrier, false) } // ResumeApplyOperationCutover drives a single apply_operation parked at the @@ -1636,7 +1640,7 @@ func (c *LocalClient) ResumeApplyOperationCutover(ctx context.Context, apply *st // the parked engine checkpoint before driving. opts := apply.GetOptions() opts.DeferCutover = false - return c.resumeApplyWithTasks(ctx, apply, tasks, opts.Map(), false, true) + return c.resumeApplyWithTasks(ctx, apply, op, tasks, opts.Map(), false, true) } // finalizerOperationKeySuffix is the trailing segment of a namespace-scoped @@ -1680,12 +1684,16 @@ func namespaceFromFinalizerKey(operationKey string) string { // terminal state), so the operator never advances the parent's aggregate as if // the VSchema applied when it did not. func (c *LocalClient) driveGroupFinalizer(ctx context.Context, apply *storage.Apply, op *storage.ApplyOperation) error { - plan, err := c.storage.Plans().GetByID(ctx, apply.PlanID) + planID, err := storage.PlanIDForOperation(apply, op) + if err != nil { + return fmt.Errorf("resolve plan for group_finalizer apply_operation %d (apply %s): %w", op.ID, apply.ApplyIdentifier, err) + } + plan, err := c.storage.Plans().GetByID(ctx, planID) if err != nil { return fmt.Errorf("load plan for group_finalizer apply_operation %d (apply %s): %w", op.ID, apply.ApplyIdentifier, err) } if plan == nil { - return fmt.Errorf("plan %d for group_finalizer apply_operation %d (apply %s): %w", apply.PlanID, op.ID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) + return fmt.Errorf("plan %d for group_finalizer apply_operation %d (apply %s): %w", planID, op.ID, apply.ApplyIdentifier, ErrPlanMissingForApplyOperation) } namespace := namespaceFromFinalizerKey(op.OperationKey) if namespace == "" && op.OperationKey != finalizerDeploymentScopedKey { @@ -1869,10 +1877,31 @@ func (c *LocalClient) driveFinalizerToTerminal(ctx context.Context, eng engine.E } } +// drivePlanID resolves the plan a drive runs. An operation-scoped drive runs the +// plan its operation names, which for a member planned against its own live +// schema is that member's plan rather than the apply's. A whole-apply drive has +// no operation to name one and runs the apply's plan, erroring rather than +// guessing when the apply names none. +func (c *LocalClient) drivePlanID(apply *storage.Apply, op *storage.ApplyOperation) (int64, error) { + if op == nil { + if apply == nil || apply.PlanID == 0 { + return 0, fmt.Errorf("whole-apply drive: apply names no plan") + } + return apply.PlanID, nil + } + return storage.PlanIDForOperation(apply, op) +} + // resumeApplyWithTasks drives an apply (or one of its operations) from the set // of tasks the caller has loaded. Callers choose whether tasks are scoped to the // whole apply or to a single operation. -func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.Apply, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { +// +// op is the operation being driven, or nil for a whole-apply drive. It names the +// plan this drive runs: a rollout member planned against its own live schema +// stores its plan on its operation row, and running the apply's plan there would +// dispatch another target's DDL. A whole-apply drive has no operation to name +// one, so it runs the apply's plan. +func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.Apply, op *storage.ApplyOperation, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { // Bind the apply's identity once so every line of this resume is // filterable by apply_id/repo/pr without hand-listing the attrs per call. // Mutable attrs (state, deployment) stay per-call so the bound logger @@ -1915,7 +1944,11 @@ func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.A // apply state — the engine-side work (a checkpointed copy or a live deploy // request) is untouched. The recovery attempt exits with an error so the // claim is released and a later attempt retries against intact storage. - plan, err := c.storage.Plans().GetByID(ctx, apply.PlanID) + planID, err := c.drivePlanID(apply, op) + if err != nil { + return fmt.Errorf("resolve plan for drive of apply %s (database %s): %w", apply.ApplyIdentifier, apply.Database, err) + } + plan, err := c.storage.Plans().GetByID(ctx, planID) if err != nil { logger.Warn("failed to load plan during recovery; current apply owner will exit for operator retry", append(apply.MutableLogAttrs(), "error", err)...) diff --git a/pkg/tern/local_control_resume_test.go b/pkg/tern/local_control_resume_test.go index 99f6e822f..046a3f234 100644 --- a/pkg/tern/local_control_resume_test.go +++ b/pkg/tern/local_control_resume_test.go @@ -766,7 +766,7 @@ func TestResumeApplyPlanLoadStorageErrorStaysRecoverable(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.ErrorIs(t, err, storageErr) assert.ErrorContains(t, err, "apply-recover-plan") @@ -785,7 +785,7 @@ func TestResumeApplyMissingPlanFailsApply(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.NoError(t, err) assert.True(t, state.IsState(applyStore.apply.State, state.Apply.Failed), @@ -813,7 +813,7 @@ func TestResumeApplyMissingPlanAdoptsConcurrentTerminalState(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.NoError(t, err) assert.True(t, state.IsState(applyStore.apply.State, state.Apply.Stopped), @@ -1175,7 +1175,7 @@ func TestResumeApplyWithTasks_RefusesSequentialResumeOfRevertPhaseTask(t *testin tasks[0].State = tc.taskState taskStore.tasks = tasks - err := c.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := c.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.ErrorIs(t, err, errRevertPhaseTaskInSequentialResume) assert.True(t, state.IsState(tasks[0].State, tc.taskState), @@ -1222,7 +1222,7 @@ func TestResumeApplyWithTasks_RevertPhaseSiblingDoesNotVouchForLandedStatement(t tasks[1].State = siblingState taskStore.tasks = tasks - err := c.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := c.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.Error(t, err) assert.Contains(t, err.Error(), "sibling task task_name ("+siblingState+"), which will not run it") @@ -1411,3 +1411,39 @@ func TestStartDeferredDeployRejectsMixedNamespaces(t *testing.T) { assert.ErrorContains(t, err, "tasks span multiple namespaces") assert.ErrorContains(t, err, apply.ApplyIdentifier) } + +// The local drive resolves the plan from the operation it drives, so a rollout +// member planned against its own live schema runs its own DDL rather than the +// reviewed primary's. +func TestLocalClientDrivePlanID(t *testing.T) { + c := &LocalClient{} + apply := &storage.Apply{ApplyIdentifier: "apply-1", PlanID: 10} + + t.Run("a whole-apply drive runs the apply's plan", func(t *testing.T) { + planID, err := c.drivePlanID(apply, nil) + require.NoError(t, err) + assert.Equal(t, int64(10), planID) + }) + + t.Run("a member that shares the reviewed plan runs the apply's plan", func(t *testing.T) { + planID, err := c.drivePlanID(apply, &storage.ApplyOperation{ID: 1, Target: "testapp-001"}) + require.NoError(t, err) + assert.Equal(t, int64(10), planID) + }) + + t.Run("a member planned on its own runs its own plan", func(t *testing.T) { + planID, err := c.drivePlanID(apply, &storage.ApplyOperation{ID: 2, Target: "testapp-002", PlanID: 11}) + require.NoError(t, err) + assert.Equal(t, int64(11), planID) + }) + + t.Run("an operation with no plan on either row is not drivable", func(t *testing.T) { + _, err := c.drivePlanID(&storage.Apply{ApplyIdentifier: "apply-2"}, &storage.ApplyOperation{ID: 3, Deployment: "eu"}) + require.Error(t, err) + }) + + t.Run("a whole-apply drive with no plan is not drivable", func(t *testing.T) { + _, err := c.drivePlanID(&storage.Apply{ApplyIdentifier: "apply-3"}, nil) + require.Error(t, err) + }) +} From ef652f07bb79328f5ba18e06dcbb34f0438e15c2 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 16:56:21 -0400 Subject: [PATCH 28/34] fix(tern): keep the whole-apply drive entry point intact An operation-scoped drive resolves its plan from the operation, so the operation had been threaded through the shared drive by widening it. The whole-apply callers have their own entry point instead, which passes no operation and reads as what it is, leaving the drive body the only place that knows about both scopes. Co-Authored-By: Claude Opus 5 --- pkg/tern/local_control_resume.go | 25 ++++++++++++++++--------- pkg/tern/local_control_resume_test.go | 10 +++++----- 2 files changed, 21 insertions(+), 14 deletions(-) diff --git a/pkg/tern/local_control_resume.go b/pkg/tern/local_control_resume.go index 13927974c..757abbc04 100644 --- a/pkg/tern/local_control_resume.go +++ b/pkg/tern/local_control_resume.go @@ -1517,7 +1517,7 @@ func (c *LocalClient) ResumeApply(ctx context.Context, apply *storage.Apply) err // apply is handled inside the shared resume path: VSchema-only plans are // re-driven so the VSchema is applied, and any other task-less shape (e.g. a // sharded dispatch whose shard already matches) completes as a no-op. - return c.resumeApplyWithTasks(ctx, apply, nil, tasks, apply.GetOptions().Map(), false, false) + return c.resumeApplyWithTasks(ctx, apply, tasks, apply.GetOptions().Map(), false, false) } // ResumeApplyOperation starts or resumes a single apply_operation (one @@ -1571,7 +1571,7 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A if !op.IsTasklessVSchemaOnlyWork(plan) { return fmt.Errorf("apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, ErrNoTasksForApplyOperation) } - return c.resumeApplyWithTasks(ctx, apply, op, tasks, apply.GetOptions().Map(), false, false) + return c.driveApplyTasks(ctx, apply, op, tasks, apply.GetOptions().Map(), false, false) } siblings, err := c.storage.ApplyOperations().ListByApply(ctx, apply.ID) if err != nil { @@ -1584,7 +1584,7 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A // unchanged. releaseAtCutoverBarrier := shouldReleaseAtCutoverBarrier(apply, multiOperation, op) options := effectiveCopyDriveOptions(apply, multiOperation, op).Map() - return c.resumeApplyWithTasks(ctx, apply, op, tasks, options, releaseAtCutoverBarrier, false) + return c.driveApplyTasks(ctx, apply, op, tasks, options, releaseAtCutoverBarrier, false) } // ResumeApplyOperationCutover drives a single apply_operation parked at the @@ -1640,7 +1640,7 @@ func (c *LocalClient) ResumeApplyOperationCutover(ctx context.Context, apply *st // the parked engine checkpoint before driving. opts := apply.GetOptions() opts.DeferCutover = false - return c.resumeApplyWithTasks(ctx, apply, op, tasks, opts.Map(), false, true) + return c.driveApplyTasks(ctx, apply, op, tasks, opts.Map(), false, true) } // finalizerOperationKeySuffix is the trailing segment of a namespace-scoped @@ -1892,16 +1892,23 @@ func (c *LocalClient) drivePlanID(apply *storage.Apply, op *storage.ApplyOperati return storage.PlanIDForOperation(apply, op) } -// resumeApplyWithTasks drives an apply (or one of its operations) from the set -// of tasks the caller has loaded. Callers choose whether tasks are scoped to the +// resumeApplyWithTasks drives a whole apply from the set of tasks the caller has +// loaded. Whole-apply scope has no operation to name a plan, so the drive runs +// the apply's own plan; a drive scoped to one operation calls driveApplyTasks +// with that operation instead. +func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.Apply, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { + return c.driveApplyTasks(ctx, apply, nil, tasks, options, releaseAtCutoverBarrier, forceCutoverResume) +} + +// driveApplyTasks drives an apply (or one of its operations) from the set of +// tasks the caller has loaded. Callers choose whether tasks are scoped to the // whole apply or to a single operation. // // op is the operation being driven, or nil for a whole-apply drive. It names the // plan this drive runs: a rollout member planned against its own live schema // stores its plan on its operation row, and running the apply's plan there would -// dispatch another target's DDL. A whole-apply drive has no operation to name -// one, so it runs the apply's plan. -func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.Apply, op *storage.ApplyOperation, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { +// dispatch another target's DDL. +func (c *LocalClient) driveApplyTasks(ctx context.Context, apply *storage.Apply, op *storage.ApplyOperation, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { // Bind the apply's identity once so every line of this resume is // filterable by apply_id/repo/pr without hand-listing the attrs per call. // Mutable attrs (state, deployment) stay per-call so the bound logger diff --git a/pkg/tern/local_control_resume_test.go b/pkg/tern/local_control_resume_test.go index 046a3f234..4baa66508 100644 --- a/pkg/tern/local_control_resume_test.go +++ b/pkg/tern/local_control_resume_test.go @@ -766,7 +766,7 @@ func TestResumeApplyPlanLoadStorageErrorStaysRecoverable(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) require.ErrorIs(t, err, storageErr) assert.ErrorContains(t, err, "apply-recover-plan") @@ -785,7 +785,7 @@ func TestResumeApplyMissingPlanFailsApply(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) require.NoError(t, err) assert.True(t, state.IsState(applyStore.apply.State, state.Apply.Failed), @@ -813,7 +813,7 @@ func TestResumeApplyMissingPlanAdoptsConcurrentTerminalState(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) require.NoError(t, err) assert.True(t, state.IsState(applyStore.apply.State, state.Apply.Stopped), @@ -1175,7 +1175,7 @@ func TestResumeApplyWithTasks_RefusesSequentialResumeOfRevertPhaseTask(t *testin tasks[0].State = tc.taskState taskStore.tasks = tasks - err := c.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) + err := c.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) require.ErrorIs(t, err, errRevertPhaseTaskInSequentialResume) assert.True(t, state.IsState(tasks[0].State, tc.taskState), @@ -1222,7 +1222,7 @@ func TestResumeApplyWithTasks_RevertPhaseSiblingDoesNotVouchForLandedStatement(t tasks[1].State = siblingState taskStore.tasks = tasks - err := c.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) + err := c.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) require.Error(t, err) assert.Contains(t, err.Error(), "sibling task task_name ("+siblingState+"), which will not run it") From 5abe6bba6723db4a188166e2a2cc70a362fb449a Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 17:03:33 -0400 Subject: [PATCH 29/34] fix(tern): name the operation on every local drive call The drive resolves its plan from the operation it is driving, so the operation is passed at every call rather than through a second entry point that only ever passed none. Co-Authored-By: Claude Opus 5 --- pkg/tern/local_control_resume.go | 25 +++++++++---------------- pkg/tern/local_control_resume_test.go | 12 ++++++------ 2 files changed, 15 insertions(+), 22 deletions(-) diff --git a/pkg/tern/local_control_resume.go b/pkg/tern/local_control_resume.go index 6781fc74f..311e29469 100644 --- a/pkg/tern/local_control_resume.go +++ b/pkg/tern/local_control_resume.go @@ -1517,7 +1517,7 @@ func (c *LocalClient) ResumeApply(ctx context.Context, apply *storage.Apply) err // apply is handled inside the shared resume path: VSchema-only plans are // re-driven so the VSchema is applied, and any other task-less shape (e.g. a // sharded dispatch whose shard already matches) completes as a no-op. - return c.resumeApplyWithTasks(ctx, apply, tasks, apply.GetOptions().Map(), false, false) + return c.resumeApplyWithTasks(ctx, apply, nil, tasks, apply.GetOptions().Map(), false, false) } // ResumeApplyOperation starts or resumes a single apply_operation (one @@ -1571,7 +1571,7 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A if !op.IsTasklessVSchemaOnlyWork(plan) { return fmt.Errorf("apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, ErrNoTasksForApplyOperation) } - return c.driveApplyTasks(ctx, apply, op, tasks, apply.GetOptions().Map(), false, false) + return c.resumeApplyWithTasks(ctx, apply, op, tasks, apply.GetOptions().Map(), false, false) } siblings, err := c.storage.ApplyOperations().ListByApply(ctx, apply.ID) if err != nil { @@ -1584,7 +1584,7 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A // unchanged. releaseAtCutoverBarrier := shouldReleaseAtCutoverBarrier(apply, multiOperation, op) options := effectiveCopyDriveOptions(apply, multiOperation, op).Map() - return c.driveApplyTasks(ctx, apply, op, tasks, options, releaseAtCutoverBarrier, false) + return c.resumeApplyWithTasks(ctx, apply, op, tasks, options, releaseAtCutoverBarrier, false) } // ResumeApplyOperationCutover drives a single apply_operation parked at the @@ -1640,7 +1640,7 @@ func (c *LocalClient) ResumeApplyOperationCutover(ctx context.Context, apply *st // the parked engine checkpoint before driving. opts := apply.GetOptions() opts.DeferCutover = false - return c.driveApplyTasks(ctx, apply, op, tasks, opts.Map(), false, true) + return c.resumeApplyWithTasks(ctx, apply, op, tasks, opts.Map(), false, true) } // finalizerOperationKeySuffix is the trailing segment of a namespace-scoped @@ -1892,23 +1892,16 @@ func (c *LocalClient) drivePlanID(apply *storage.Apply, op *storage.ApplyOperati return storage.PlanIDForOperation(apply, op) } -// resumeApplyWithTasks drives a whole apply from the set of tasks the caller has -// loaded. Whole-apply scope has no operation to name a plan, so the drive runs -// the apply's own plan; a drive scoped to one operation calls driveApplyTasks -// with that operation instead. -func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.Apply, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { - return c.driveApplyTasks(ctx, apply, nil, tasks, options, releaseAtCutoverBarrier, forceCutoverResume) -} - -// driveApplyTasks drives an apply (or one of its operations) from the set of -// tasks the caller has loaded. Callers choose whether tasks are scoped to the +// resumeApplyWithTasks drives an apply (or one of its operations) from the set +// of tasks the caller has loaded. Callers choose whether tasks are scoped to the // whole apply or to a single operation. // // op is the operation being driven, or nil for a whole-apply drive. It names the // plan this drive runs: a rollout member planned against its own live schema // stores its plan on its operation row, and running the apply's plan there would -// dispatch another target's DDL. -func (c *LocalClient) driveApplyTasks(ctx context.Context, apply *storage.Apply, op *storage.ApplyOperation, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { +// dispatch another target's DDL. A whole-apply drive has no operation to name +// one, so it runs the apply's plan. +func (c *LocalClient) resumeApplyWithTasks(ctx context.Context, apply *storage.Apply, op *storage.ApplyOperation, tasks []*storage.Task, options map[string]string, releaseAtCutoverBarrier bool, forceCutoverResume bool) error { // Bind the apply's identity once so every line of this resume is // filterable by apply_id/repo/pr without hand-listing the attrs per call. // Mutable attrs (state, deployment) stay per-call so the bound logger diff --git a/pkg/tern/local_control_resume_test.go b/pkg/tern/local_control_resume_test.go index 91da485b9..d0a48d8a3 100644 --- a/pkg/tern/local_control_resume_test.go +++ b/pkg/tern/local_control_resume_test.go @@ -766,7 +766,7 @@ func TestResumeApplyPlanLoadStorageErrorStaysRecoverable(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.ErrorIs(t, err, storageErr) assert.ErrorContains(t, err, "apply-recover-plan") @@ -796,7 +796,7 @@ func TestResumeApplyContinuesPastARefusedCancelInTheRevertWindow(t *testing.T) { }}} client.storage.(*exactProgressStorage).controlRequests = requests - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.ErrorIs(t, err, storageErr, "the resume must have continued past the refusal to reach the plan load") assert.Equal(t, state.Task.RevertWindow, tasks[0].State, "the cut-over task keeps its revert window") @@ -814,7 +814,7 @@ func TestResumeApplyMissingPlanFailsApply(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.NoError(t, err) assert.True(t, state.IsState(applyStore.apply.State, state.Apply.Failed), @@ -842,7 +842,7 @@ func TestResumeApplyMissingPlanAdoptsConcurrentTerminalState(t *testing.T) { observer := &terminalRecordingObserver{} client.SetObserver(apply.ID, observer) - err := client.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := client.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.NoError(t, err) assert.True(t, state.IsState(applyStore.apply.State, state.Apply.Stopped), @@ -1204,7 +1204,7 @@ func TestResumeApplyWithTasks_RefusesSequentialResumeOfRevertPhaseTask(t *testin tasks[0].State = tc.taskState taskStore.tasks = tasks - err := c.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := c.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.ErrorIs(t, err, errRevertPhaseTaskInSequentialResume) assert.True(t, state.IsState(tasks[0].State, tc.taskState), @@ -1251,7 +1251,7 @@ func TestResumeApplyWithTasks_RevertPhaseSiblingDoesNotVouchForLandedStatement(t tasks[1].State = siblingState taskStore.tasks = tasks - err := c.resumeApplyWithTasks(t.Context(), apply, tasks, nil, false, false) + err := c.resumeApplyWithTasks(t.Context(), apply, nil, tasks, nil, false, false) require.Error(t, err) assert.Contains(t, err.Error(), "sibling task task_name ("+siblingState+"), which will not run it") From d2dc6053dd3a4d8e2331199551f274b6404caf9f Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Mon, 14 Sep 2026 17:35:00 -0400 Subject: [PATCH 30/34] fix(tern): fail closed when a task-less operation resolves to no plan A work operation with no tasks is valid only when its plan carries VSchema work, so resolving the plan is part of deciding that. An operation naming no plan, under an apply naming none either, has nothing that could make it valid and is the same invalid or stale claim: it must terminalize on the fail-closed signal the operator matches, not on a distinct resolution error that leaves the claim to be re-leased forever. The resolution failure rides along as context so the cause stays triageable. Co-Authored-By: Claude Opus 5 --- pkg/tern/grpc_client.go | 5 ++++- pkg/tern/local_control_resume.go | 7 ++++++- 2 files changed, 10 insertions(+), 2 deletions(-) diff --git a/pkg/tern/grpc_client.go b/pkg/tern/grpc_client.go index 34971745f..ba63f8d0b 100644 --- a/pkg/tern/grpc_client.go +++ b/pkg/tern/grpc_client.go @@ -2224,9 +2224,12 @@ func (c *GRPCClient) ResumeApplyOperation(ctx context.Context, apply *storage.Ap // modelled as a task row. Dispatch it as a VSchema-only apply, which the // data plane applies via its own task-less VSchema-only path, mirroring // LocalClient.ResumeApplyOperation. + // A plan is what makes a task-less work operation valid, so an operation + // that resolves to none fails closed on the same signal as one whose plan + // carries no VSchema work, with the resolution failure as context. planID, err := scope.planID(apply) if err != nil { - return fmt.Errorf("resolve plan for task-less apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, err) + return fmt.Errorf("apply_operation %d (apply %s) resolves to no plan (%w): %w", applyOperationID, apply.ApplyIdentifier, err, ErrNoTasksForApplyOperation) } plan, err := c.storage.Plans().GetByID(ctx, planID) if err != nil { diff --git a/pkg/tern/local_control_resume.go b/pkg/tern/local_control_resume.go index 311e29469..c2321c058 100644 --- a/pkg/tern/local_control_resume.go +++ b/pkg/tern/local_control_resume.go @@ -1555,9 +1555,14 @@ func (c *LocalClient) ResumeApplyOperation(ctx context.Context, apply *storage.A if op.OperationKind == storage.ApplyOperationKindGroupFinalizer { return c.driveGroupFinalizer(ctx, apply, op) } + // A plan is what makes a task-less work operation valid, so an operation + // that resolves to none is the same fail-closed signal as one whose plan + // carries no VSchema work: an invalid or stale claim, terminalized on this + // operation rather than failing the whole parent apply. The resolution + // failure rides along as context so the cause stays triageable. planID, planErr := storage.PlanIDForOperation(apply, op) if planErr != nil { - return fmt.Errorf("resolve plan for task-less apply_operation %d (apply %s): %w", applyOperationID, apply.ApplyIdentifier, planErr) + return fmt.Errorf("apply_operation %d (apply %s) resolves to no plan (%w): %w", applyOperationID, apply.ApplyIdentifier, planErr, ErrNoTasksForApplyOperation) } plan, planErr := c.storage.Plans().GetByID(ctx, planID) if planErr != nil { From 6595a9c26550cdb155920582c174073020d77ac5 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 17 Sep 2026 10:33:56 -0400 Subject: [PATCH 31/34] feat(tern): key a change set by the work it would run A multi-target environment plans every member against its own live schema, so one review round produces N plans. The plan comment has to group the members that would run the same work, and the only honest way to decide "same work" is the comparison the drift rollup already performs -- comparing rendered DDL text would split a group over a backtick. ChangeSetFingerprint returns a stable key over the same canonicalized multiset CompareChangeSets keys on, so two change sets share a key exactly when the comparison reports them identical. That makes grouping by it sound rather than a heuristic: members of one environment are planned by the same differ against the same desired schema, so the only thing that can make two plans differ is the live schema each was diffed against, and any such difference changes the multiset. A table test pins fingerprint equality against the comparison's own verdict across restyled, reordered, duplicated, and diverging sets, so the two cannot drift apart without failing. Three properties the key needs and the tests pin by construction. It sorts the rendered records, because a multiset and a set are both unordered and the engine's return order is not work. It counts changes rather than collecting them, because running a change twice is not running it once. And it separates the fields it joins with a byte no field can hold -- a namespace, table, or DDL body is whatever a schema author wrote, so a concatenated key would let a namespace ending where a shard name begins collide two members into one group. It errors on exactly what the comparison errors on, so a member whose change set cannot be canonicalized has no key at all and a caller has to refuse to group it rather than file it with members it was never compared against. Nothing calls it yet; the plan comment's grouping lands on top of it. Co-Authored-By: Claude Opus 5 --- pkg/tern/change_set_fingerprint.go | 78 ++++++++++++ pkg/tern/change_set_fingerprint_test.go | 152 ++++++++++++++++++++++++ 2 files changed, 230 insertions(+) create mode 100644 pkg/tern/change_set_fingerprint.go create mode 100644 pkg/tern/change_set_fingerprint_test.go diff --git a/pkg/tern/change_set_fingerprint.go b/pkg/tern/change_set_fingerprint.go new file mode 100644 index 000000000..9eabf23c2 --- /dev/null +++ b/pkg/tern/change_set_fingerprint.go @@ -0,0 +1,78 @@ +package tern + +import ( + "crypto/sha256" + "encoding/hex" + "fmt" + "sort" + "strconv" + "strings" + + "github.com/block/schemabot/pkg/ddl" + "github.com/block/schemabot/pkg/schema" +) + +// ChangeSetFingerprint returns a stable key for a change set: two change sets +// share a fingerprint exactly when CompareChangeSets reports them identical, and +// differ otherwise. It exists so callers can group members that would run the +// same work without comparing every pair. +// +// It is built on the same canonicalized multiset the comparison keys on, which +// is what makes grouping by it sound rather than a textual heuristic. Members of +// one environment are planned by the same differ against the same desired +// schema, so the only thing that can make two members' plans differ is the live +// schema each was diffed against — and any such difference changes the multiset. +// Equal fingerprints therefore mean equal work, not merely similar-looking DDL. +// +// The fingerprint is opaque and not stable across releases: it is a grouping +// key for one rollup, never something to persist, compare across versions, or +// show an operator. Nothing about it identifies which plan it belongs to. +// +// Errors on exactly what the comparison errors on — a malformed change set or +// DDL that cannot be canonicalized — so a caller that cannot fingerprint a +// member treats it as unclassifiable rather than grouping it with members it was +// never compared against. +func ChangeSetFingerprint(dialect schema.Dialect, cs ChangeSet) (string, error) { + parser, err := ddl.ParserForDialect(dialect) + if err != nil { + return "", err + } + ms, vschema, err := changeSetMultiset(parser, cs) + if err != nil { + return "", fmt.Errorf("fingerprint change set: %w", err) + } + + // A multiset and a set are both unordered, so the digest has to consume them + // in an order neither one carries. Sorting the rendered lines is what makes + // the fingerprint depend on the change set's content and not on the order the + // engine happened to return it in. + lines := make([]string, 0, len(ms)+len(vschema)) + for key, count := range ms { + lines = append(lines, "c"+fingerprintRecord( + key.namespace, key.shard, key.table, key.operation, key.ddl, strconv.Itoa(count))) + } + for ns := range vschema { + lines = append(lines, "v"+fingerprintRecord(ns)) + } + sort.Strings(lines) + + digest := sha256.New() + for _, line := range lines { + // Record separator, not part of any field: the field separator inside + // fingerprintRecord already keeps fields from running together, and this + // keeps two records from doing the same. + digest.Write([]byte(line)) + digest.Write([]byte{0x1e}) + } + return hex.EncodeToString(digest.Sum(nil)), nil +} + +// fingerprintRecord joins one multiset entry's fields with a separator no field +// can contain, so no pair of distinct entries can render to the same line. A +// namespace, table, or DDL body can contain anything a schema author wrote, +// which is why the separator is a control byte rather than a punctuation +// character: joining on one that a field could contain would let two different +// change sets collide into one fingerprint and be grouped as identical work. +func fingerprintRecord(fields ...string) string { + return strings.Join(fields, "\x1f") +} diff --git a/pkg/tern/change_set_fingerprint_test.go b/pkg/tern/change_set_fingerprint_test.go new file mode 100644 index 000000000..c8beb40b4 --- /dev/null +++ b/pkg/tern/change_set_fingerprint_test.go @@ -0,0 +1,152 @@ +package tern + +import ( + "testing" + + "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" + + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/schema" +) + +func fingerprint(t *testing.T, cs ChangeSet) string { + t.Helper() + fp, err := ChangeSetFingerprint(schema.DialectMySQL, cs) + require.NoError(t, err) + require.NotEmpty(t, fp) + return fp +} + +// Two members whose plans differ only in how the DDL is written run the same +// work, so they group together. This is the property that makes grouping by +// fingerprint sound: it keys on the canonicalized change, not on the text. +func TestChangeSetFingerprint_CanonicallyIdenticalSetsShareAKey(t *testing.T) { + backticked := protoNonShardedSet(protoAlterUsersEmail()) + bare := protoNonShardedSet(&ternv1.TableChange{ + TableName: "users", + Ddl: "ALTER TABLE users ADD COLUMN email varchar(255)", + ChangeType: ternv1.ChangeType_CHANGE_TYPE_ALTER, + Namespace: "testapp", + }) + + assert.Equal(t, fingerprint(t, backticked), fingerprint(t, bare)) +} + +// Members that would run different DDL must not be grouped together, however +// similar the change looks. +func TestChangeSetFingerprint_DifferentWorkGetsDifferentKeys(t *testing.T) { + email := protoNonShardedSet(protoAlterUsersEmail()) + phone := protoNonShardedSet(protoAlterUsersPhone()) + + assert.NotEqual(t, fingerprint(t, email), fingerprint(t, phone)) +} + +// The engine may return one member's changes in a different order than +// another's. Order is not work, so it must not split a group. +func TestChangeSetFingerprint_IgnoresChangeOrder(t *testing.T) { + forward := protoNonShardedSet(protoAlterUsersEmail(), protoAlterUsersPhone()) + reversed := protoNonShardedSet(protoAlterUsersPhone(), protoAlterUsersEmail()) + + assert.Equal(t, fingerprint(t, forward), fingerprint(t, reversed)) +} + +// Running a change once and running it twice are different amounts of work, so +// the key counts changes rather than collecting them into a set. +func TestChangeSetFingerprint_CountsDuplicateChanges(t *testing.T) { + once := protoNonShardedSet(protoAlterUsersEmail()) + twice := protoNonShardedSet(protoAlterUsersEmail(), protoAlterUsersEmail()) + + assert.NotEqual(t, fingerprint(t, once), fingerprint(t, twice)) +} + +// A member already at the desired schema plans nothing. That is a group of its +// own — "nothing to apply here" — not a member that failed to produce a key. +func TestChangeSetFingerprint_EmptySetIsItsOwnGroup(t *testing.T) { + empty := fingerprint(t, ChangeSet{}) + noChanges := fingerprint(t, protoNonShardedSet()) + withWork := fingerprint(t, protoNonShardedSet(protoAlterUsersEmail())) + + assert.Equal(t, empty, noChanges, "a namespace planning nothing is the same work as no namespace at all") + assert.NotEqual(t, empty, withWork) +} + +// A vschema change carries no table DDL, so it would be invisible to a key built +// from table changes alone: two members would group together while only one of +// them rewrites the vschema. +func TestChangeSetFingerprint_VSchemaChangeSplitsAGroup(t *testing.T) { + plain := ChangeSet{Changes: []*ternv1.SchemaChange{{ + Namespace: "testapp", + TableChanges: []*ternv1.TableChange{protoAlterUsersEmail()}, + }}} + withVSchema := ChangeSet{Changes: []*ternv1.SchemaChange{{ + Namespace: "testapp", + TableChanges: []*ternv1.TableChange{protoAlterUsersEmail()}, + Metadata: map[string]string{"vschema_changed": "true"}, + }}} + + assert.NotEqual(t, fingerprint(t, plain), fingerprint(t, withVSchema)) +} + +// The key joins fields that can each hold anything a schema author wrote, so it +// has to separate them. These two change sets run against different shards of +// different namespaces and share every other field, including the DDL — so the +// boundary between namespace and shard is the only thing telling them apart, and +// a key that concatenated its fields would group them as identical work. +func TestChangeSetFingerprint_SeparatesItsFields(t *testing.T) { + sharded := func(namespace, shard string) ChangeSet { + return ChangeSet{Shards: []*ternv1.ShardPlan{{ + Namespace: namespace, + Shard: shard, + Changes: []*ternv1.TableChange{protoAlterUsersEmail()}, + }}} + } + + assert.NotEqual(t, fingerprint(t, sharded("app", "a1")), fingerprint(t, sharded("appa", "1"))) +} + +// A change set the comparison cannot canonicalize has no key. A caller must be +// able to tell that apart from a key, so it can refuse to group the member +// rather than group it with members it was never compared against. +func TestChangeSetFingerprint_MalformedSetHasNoKey(t *testing.T) { + _, err := ChangeSetFingerprint(schema.DialectMySQL, ChangeSet{ + Changes: []*ternv1.SchemaChange{{Namespace: "testapp", TableChanges: []*ternv1.TableChange{nil}}}, + }) + require.Error(t, err) +} + +// The contract the grouping rests on: two members share a key exactly when the +// comparison reports no difference between them. If these two ever disagree, a +// comment would either split one plan across several blocks or render two +// different plans as one. +func TestChangeSetFingerprint_AgreesWithCompareChangeSets(t *testing.T) { + restyled := protoNonShardedSet(&ternv1.TableChange{ + TableName: "users", + Ddl: "ALTER TABLE users ADD COLUMN email varchar(255)", + ChangeType: ternv1.ChangeType_CHANGE_TYPE_ALTER, + Namespace: "testapp", + }) + + cases := []struct { + name string + baseline, candidate ChangeSet + }{ + {"identical", protoNonShardedSet(protoAlterUsersEmail()), protoNonShardedSet(protoAlterUsersEmail())}, + {"restyled DDL", protoNonShardedSet(protoAlterUsersEmail()), restyled}, + {"reordered", protoNonShardedSet(protoAlterUsersEmail(), protoAlterUsersPhone()), protoNonShardedSet(protoAlterUsersPhone(), protoAlterUsersEmail())}, + {"both empty", protoNonShardedSet(), ChangeSet{}}, + {"missing change", protoNonShardedSet(protoAlterUsersEmail()), protoNonShardedSet()}, + {"extra change", protoNonShardedSet(protoAlterUsersEmail()), protoNonShardedSet(protoAlterUsersEmail(), protoAlterUsersPhone())}, + {"different DDL", protoNonShardedSet(protoAlterUsersEmail()), protoNonShardedSet(protoAlterUsersPhone())}, + {"duplicated change", protoNonShardedSet(protoAlterUsersEmail()), protoNonShardedSet(protoAlterUsersEmail(), protoAlterUsersEmail())}, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + diff, err := CompareChangeSets(schema.DialectMySQL, tc.baseline, tc.candidate) + require.NoError(t, err) + assert.Equal(t, diff.Empty(), fingerprint(t, tc.baseline) == fingerprint(t, tc.candidate), + "fingerprint equality must agree with the comparison; diff: %+v", diff) + }) + } +} From 0157f0f0facc13e94acb113da0e7338725015e5b Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 17 Sep 2026 12:37:06 -0400 Subject: [PATCH 32/34] feat(api): carry each rollout member's plan and its grouping key A rollup entry recorded how a member classified but dropped the change set that produced the classification, so a reader could say a member diverged but not show what it would run. Every classified member now carries its own change set and a fingerprint keying that change set by the work it performs: members share the key exactly when the comparison reports their plans identical, which lets a reader group members by plan without comparing every pair. A member that could not be planned carries neither. It has no plan to describe, and keying it would group it with members it was never compared against. Co-Authored-By: Claude Opus 5 --- pkg/api/plan_rollup.go | 53 +++++++++++++++++++++++ pkg/api/plan_rollup_test.go | 85 +++++++++++++++++++++++++++++++++++++ 2 files changed, 138 insertions(+) diff --git a/pkg/api/plan_rollup.go b/pkg/api/plan_rollup.go index 01e94cd3c..4019e69c5 100644 --- a/pkg/api/plan_rollup.go +++ b/pkg/api/plan_rollup.go @@ -84,6 +84,20 @@ type DeploymentRollupEntry struct { Diff tern.ChangeSetDiff Err error + // ChangeSet is what this member would run: the change set its own diff + // produced, or the reviewed plan's for the primary. Empty for a member that + // errored, which has no plan to describe. + ChangeSet tern.ChangeSet + // PlanFingerprint keys ChangeSet by the work it would run, so a reader can + // group members that would run the same plan without comparing every pair. + // Two members share it exactly when tern.CompareChangeSets reports them + // identical. Empty for a member that errored. + // + // A member classified Match or Planned always has one: both classifications + // are reached through a self-comparison that proves the member's content + // canonicalizes, which is the same thing the fingerprint needs. + PlanFingerprint string + // PlanIdentifier names the stored plan this member will run, set when the // member was planned on its own and its plan was persisted as a row of its // own. Empty means the member runs the plan the apply itself was created @@ -231,12 +245,45 @@ func RollupDeploymentDiffs(diffs []DeploymentPlanDiff, expectedMembers []routing entry.Class = DeploymentMatch } } + if !recordMemberPlan(&entry, baselineDialect, tern.ChangeSet{Changes: d.Changes, Shards: d.Shards}) { + clean = false + } entries[i] = entry } return PlanRollup{Entries: entries, Clean: clean, Planning: planning}, nil } +// recordMemberPlan records what a classified member would run — its change set +// and the key that groups it with members running the same work — and reports +// whether the entry still passes. +// +// It is only reached for a member whose content a comparison already +// canonicalized, so a fingerprint failure here contradicts that comparison. It +// still fails the member closed rather than leaving the key empty: a member +// SchemaBot cannot key is one it cannot group, and an ungrouped member renders +// as work nobody reviewed. +// +// An errored member is left alone and reported as not passing. It has no plan to +// describe, and overwriting its cause with a second one would bury the reason it +// blocked. The caller has already failed that member closed, so the repeated +// signal changes nothing; the point is that the result means the same thing for +// every member, whichever branch classified it. +func recordMemberPlan(entry *DeploymentRollupEntry, dialect schema.Dialect, cs tern.ChangeSet) bool { + if entry.Class == DeploymentErrored { + return false + } + fingerprint, err := tern.ChangeSetFingerprint(dialect, cs) + if err != nil { + entry.Class = DeploymentErrored + entry.Err = fmt.Errorf("member plan could not be keyed for grouping: %w", err) + return false + } + entry.ChangeSet = cs + entry.PlanFingerprint = fingerprint + return true +} + // rollupIndependentMembers classifies members that were each planned against // their own live schema. No member is compared to another, so a difference // between them is never drift. What still blocks is a member that could not be @@ -268,6 +315,12 @@ func rollupIndependentMembers(diffs []DeploymentPlanDiff) PlanRollup { entry.Class = DeploymentPlanned } } + // Each member is keyed under its own grammar. Members here are never + // compared to each other, so nothing has established that they share a + // dialect the way the mirrored path's baseline does. + if !recordMemberPlan(&entry, schema.DialectForDatabaseType(d.DatabaseType), tern.ChangeSet{Changes: d.Changes, Shards: d.Shards}) { + clean = false + } entries[i] = entry } return PlanRollup{Entries: entries, Clean: clean, Planning: PlanIndependent} diff --git a/pkg/api/plan_rollup_test.go b/pkg/api/plan_rollup_test.go index f0a17db0d..b8f9843b9 100644 --- a/pkg/api/plan_rollup_test.go +++ b/pkg/api/plan_rollup_test.go @@ -376,6 +376,91 @@ func TestRollupDeploymentDiffs_IndependentUnparseableMemberBlocks(t *testing.T) assert.Contains(t, rollup.Entries[1].Err.Error(), "not usable") } +// Every member that classified carries the plan it would run and a key for it, +// so a reader can render the member's DDL and group members by the work they +// share. Members planning the same changes share the key; a member planning +// different work does not. +func TestRollupDeploymentDiffs_ClassifiedMembersCarryTheirPlan(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001", rollupAlterUsers(email)), + rollupMember("cake", "orders-002", rollupAlterUsers(email)), + rollupMember("cake", "orders-003", rollupAlterUsers("ALTER TABLE users ADD COLUMN phone VARCHAR(32)")), + } + + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanIndependent) + require.NoError(t, err) + require.Len(t, rollup.Entries, 3) + for i, entry := range rollup.Entries { + assert.Equal(t, DeploymentPlanned, entry.Class, "entry %d", i) + require.NotEmpty(t, entry.PlanFingerprint, "entry %d", i) + require.Len(t, entry.ChangeSet.Changes, 1, "entry %d", i) + assert.Equal(t, diffs[i].Changes[0].TableChanges[0].Ddl, entry.ChangeSet.Changes[0].TableChanges[0].Ddl, "entry %d", i) + } + assert.Equal(t, rollup.Entries[0].PlanFingerprint, rollup.Entries[1].PlanFingerprint, + "members planning the same changes group together") + assert.NotEqual(t, rollup.Entries[0].PlanFingerprint, rollup.Entries[2].PlanFingerprint, + "a member planning different changes is its own group") +} + +// A member already at the desired schema plans nothing, which is work of its own +// and keys to a group of its own: the members with nothing to run are rendered +// together, and never folded in with the members that would change a table. +func TestRollupDeploymentDiffs_MembersWithNothingToRunGroupTogether(t *testing.T) { + diffs := []DeploymentPlanDiff{ + rollupMember("cake", "orders-001", rollupAlterUsers("ALTER TABLE users ADD COLUMN email VARCHAR(255)")), + rollupMember("cake", "orders-002"), + rollupMember("cake", "orders-003"), + } + + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanIndependent) + require.NoError(t, err) + require.Len(t, rollup.Entries, 3) + assert.Equal(t, rollup.Entries[1].PlanFingerprint, rollup.Entries[2].PlanFingerprint) + assert.NotEqual(t, rollup.Entries[0].PlanFingerprint, rollup.Entries[1].PlanFingerprint) + assert.Empty(t, rollup.Entries[1].ChangeSet.Changes) +} + +// Mirrored members carry their plans too, and a diverged member carries the plan +// it would actually run rather than the reviewed one — the divergence is the +// difference between them, so rendering the reviewed plan against a diverged +// member would show work that member will not do. +func TestRollupDeploymentDiffs_DivergedMemberCarriesItsOwnPlan(t *testing.T) { + reviewed := "ALTER TABLE `users` ADD COLUMN `email` varchar(255)" + diverged := "ALTER TABLE `users` ADD COLUMN `phone` varchar(255)" + diffs := []DeploymentPlanDiff{ + rollupDeployment("eu", rollupAlterUsers(reviewed)), + rollupDeployment("au", rollupAlterUsers(diverged)), + } + + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) + require.NoError(t, err) + require.Len(t, rollup.Entries, 2) + assert.Equal(t, DeploymentDiverged, rollup.Entries[1].Class) + require.Len(t, rollup.Entries[1].ChangeSet.Changes, 1) + assert.Equal(t, diverged, rollup.Entries[1].ChangeSet.Changes[0].TableChanges[0].Ddl) + assert.NotEqual(t, rollup.Entries[0].PlanFingerprint, rollup.Entries[1].PlanFingerprint) +} + +// A member that could not be planned has no plan to carry and no group to join, +// and keeps the cause it blocked on. Keying it would mean grouping it with +// members it was never compared against. +func TestRollupDeploymentDiffs_ErroredMemberCarriesNoPlan(t *testing.T) { + diffs := []DeploymentPlanDiff{ + rollupDeployment("eu", rollupAlterUsers("ALTER TABLE `users` ADD COLUMN `email` varchar(255)")), + rollupDeployment("au"), + } + diffs[1].Err = fmt.Errorf("deployment unreachable") + + rollup, err := RollupDeploymentDiffs(diffs, rollupMembers(diffs), PlanMirrored) + require.NoError(t, err) + require.Len(t, rollup.Entries, 2) + assert.Equal(t, DeploymentErrored, rollup.Entries[1].Class) + assert.Empty(t, rollup.Entries[1].PlanFingerprint) + assert.Empty(t, rollup.Entries[1].ChangeSet.Changes) + assert.ErrorContains(t, rollup.Entries[1].Err, "deployment unreachable") +} + // The member contract is enforced whatever the planning: independent planning // stops members being compared to each other, it does not stop a missing or // misidentified member from failing the rollup closed. From bf36a875780da7e14e8c5542dd4018c75932cf2d Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 17 Sep 2026 13:33:44 -0400 Subject: [PATCH 33/34] feat(github): say how much a rollout's targets agree this round MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An environment whose members are distinct targets plans each one against its own live schema, so the members are free to run different work. The plan comment said only that: "each target holds its own schema, so their plans are not expected to match". True of the contract, and silent about the round in front of the reviewer. Targets free to differ usually do not, and a fleet converging over several PRs — some targets changed, the rest already there — was invisible. The members are now grouped by the plan each would run, and the comment states the result: every target needs the same change, or how many of them are already at this schema, or how many distinct plans the apply would run. Members group on the plan fingerprint, so they share a group exactly when their plans are the same work. Each group carries the plan its own members would run, in the shape the comment already renders the reviewed plan in, so a later change can show it. Grouping is confined to a clean rollup of independent members. A blocked rollup still lists every member on its own, because the operator's next step is the target that could not be planned. Mirrored members stay ungrouped: a clean mirrored rollup has already proved they are one group, and re-reporting that in the vocabulary of a fleet free to diverge would read as an outcome rather than the requirement that let the check pass. Co-Authored-By: Claude Opus 5 --- pkg/webhook/plan_drift.go | 129 +++++++++- pkg/webhook/plan_drift_test.go | 306 +++++++++++++++++++++++ pkg/webhook/templates/plan.go | 91 ++++++- pkg/webhook/templates/plan_drift_test.go | 147 ++++++++++- 4 files changed, 668 insertions(+), 5 deletions(-) diff --git a/pkg/webhook/plan_drift.go b/pkg/webhook/plan_drift.go index a89d9ca89..66b31bc1f 100644 --- a/pkg/webhook/plan_drift.go +++ b/pkg/webhook/plan_drift.go @@ -3,9 +3,11 @@ package webhook import ( "context" "fmt" + "slices" "strings" "github.com/block/schemabot/pkg/api" + "github.com/block/schemabot/pkg/apitypes" ternv1 "github.com/block/schemabot/pkg/proto/ternv1" "github.com/block/schemabot/pkg/routing" "github.com/block/schemabot/pkg/tern" @@ -105,12 +107,135 @@ func deploymentDriftPreview(rollup api.PlanRollup) *templates.DeploymentDriftDat } entries[i] = entry } - return &templates.DeploymentDriftData{ + independent := rollup.Planning == api.PlanIndependent + data := &templates.DeploymentDriftData{ Deployments: entries, Clean: rollup.Clean, Computed: true, - Independent: rollup.Planning == api.PlanIndependent, + Independent: independent, + } + // Grouping describes the targets that were planned, so it is only meaningful + // once every one of them was. A blocked rollup lists each member on its own + // instead: the operator's next step is the member that could not be planned, + // not the plans of an apply that cannot run. + // + // Mirrored members are left ungrouped because a clean mirrored rollup has + // already proved they are one group. Saying so a second time, in the + // vocabulary of a fleet that may diverge, would suggest the agreement was an + // outcome rather than the requirement that let the check pass. + if rollup.Clean && independent { + data.Plans = deploymentPlanGroups(rollup) + } + return data +} + +// deploymentPlanGroups groups the rollout's members by the plan each would run, +// one entry per distinct plan. +// +// Members are grouped on the plan fingerprint, which two members share exactly +// when their plans are the same work — so a group can be described once and +// attributed to all of its members without comparing every pair. Groups come out +// in the rollout order of their first member, with the primary's group first: +// the reviewed plan is the one an operator has already seen, and a fixed order +// keeps a comment that is re-rendered on a later push from reshuffling under a +// reader who is looking for what changed. +func deploymentPlanGroups(rollup api.PlanRollup) []templates.DeploymentPlanGroup { + names := rollupMemberNames(rollup) + var groups []templates.DeploymentPlanGroup + byPlan := make(map[string]int, len(rollup.Entries)) + for i, e := range rollup.Entries { + at, ok := byPlan[e.PlanFingerprint] + if !ok { + groups = append(groups, templates.DeploymentPlanGroup{ + Primary: i == 0, + Changes: memberPlanChanges(e.ChangeSet), + }) + at = len(groups) - 1 + byPlan[e.PlanFingerprint] = at + } + groups[at].Members = append(groups[at].Members, names[i]) + } + // The primary is the first member, so its group is already first. Ordering is + // stated as a property of the result rather than left to that coincidence, + // which a later change to rollout order would silently break. + slices.SortStableFunc(groups, func(a, b templates.DeploymentPlanGroup) int { + switch { + case a.Primary == b.Primary: + return 0 + case a.Primary: + return -1 + default: + return 1 + } + }) + return groups +} + +// memberPlanChanges renders one member's plan in the shape the comment renders +// the reviewed plan in, so a group's changes are described by the same code that +// describes the plan a reviewer has already read. +// +// A sharded namespace carries its changes twice: once per shard, and once in a +// collapsed namespace view that dedupes tables across shards. Both are kept, the +// same way the reviewed plan keeps them, so the rendering can show what applies +// where rather than a namespace-level view that hides a shard. +// +// A namespace that appears only on shard rows still gets an entry. Dropping it +// would silently remove work from a plan the comment claims to describe in full. +func memberPlanChanges(cs tern.ChangeSet) []templates.KeyspaceChangeData { + shardsByNamespace := make(map[string][]templates.KeyspaceShardChange, len(cs.Shards)) + var shardedNamespaces []string + for _, sp := range cs.Shards { + if sp == nil { + continue + } + shard := templates.KeyspaceShardChange{Shard: sp.GetShard()} + for _, tc := range sp.GetChanges() { + if tc.GetDdl() == "" { + continue + } + shard.Statements = append(shard.Statements, tc.GetDdl()) + } + // A shard with nothing to run already matches the desired schema while + // its siblings change. It is carried as a satisfied group rather than + // dropped, so a partially-applied namespace shows its divergent state. + shard.Satisfied = len(shard.Statements) == 0 + if _, seen := shardsByNamespace[sp.GetNamespace()]; !seen { + shardedNamespaces = append(shardedNamespaces, sp.GetNamespace()) + } + shardsByNamespace[sp.GetNamespace()] = append(shardsByNamespace[sp.GetNamespace()], shard) + } + + changes := make([]templates.KeyspaceChangeData, 0, len(cs.Changes)) + named := make(map[string]bool, len(cs.Changes)) + for _, sc := range cs.Changes { + if sc == nil { + continue + } + named[sc.GetNamespace()] = true + ks := templates.KeyspaceChangeData{ + Keyspace: sc.GetNamespace(), + Shards: shardsByNamespace[sc.GetNamespace()], + } + for _, tc := range sc.GetTableChanges() { + if tc.GetDdl() == "" { + continue + } + ks.Statements = append(ks.Statements, tc.GetDdl()) + } + if sc.GetMetadata()[apitypes.VSchemaChangedMetadataKey] == "true" { + ks.VSchemaChanged = true + ks.VSchemaDiff = sc.GetMetadata()[apitypes.VSchemaDiffMetadataKey] + } + changes = append(changes, ks) + } + for _, ns := range shardedNamespaces { + if named[ns] { + continue + } + changes = append(changes, templates.KeyspaceChangeData{Keyspace: ns, Shards: shardsByNamespace[ns]}) } + return changes } // describeDriftDiff renders a short, count-based summary of how a diverged diff --git a/pkg/webhook/plan_drift_test.go b/pkg/webhook/plan_drift_test.go index d01c5caa6..868dceab1 100644 --- a/pkg/webhook/plan_drift_test.go +++ b/pkg/webhook/plan_drift_test.go @@ -5,11 +5,15 @@ import ( "unicode/utf8" "github.com/stretchr/testify/assert" + "github.com/stretchr/testify/require" "github.com/block/schemabot/pkg/api" "github.com/block/schemabot/pkg/apitypes" + ternv1 "github.com/block/schemabot/pkg/proto/ternv1" + "github.com/block/schemabot/pkg/schema" "github.com/block/schemabot/pkg/storage" "github.com/block/schemabot/pkg/tern" + "github.com/block/schemabot/pkg/webhook/templates" ) // The drift summary names diverged deployments so the check's Change column @@ -192,3 +196,305 @@ func TestSummarizeReviewDrift_NamesMultiTargetMembers(t *testing.T) { assert.Contains(t, summary, "could not plan: primary/testapp-002, eu-west") assert.NotContains(t, summary, "drift blocks apply") } + +// plannedMember builds a clean, independently-planned rollup member running the +// given DDL. A member with no DDL is already at the desired schema. +func plannedMember(deployment, target string, ddl ...string) api.DeploymentRollupEntry { + cs := tern.ChangeSet{} + if len(ddl) > 0 { + change := &ternv1.SchemaChange{Namespace: "testapp"} + for _, stmt := range ddl { + change.TableChanges = append(change.TableChanges, &ternv1.TableChange{ + TableName: "users", + Ddl: stmt, + ChangeType: ternv1.ChangeType_CHANGE_TYPE_ALTER, + Namespace: "testapp", + }) + } + cs.Changes = []*ternv1.SchemaChange{change} + } + fp, err := tern.ChangeSetFingerprint(schema.DialectMySQL, cs) + if err != nil { + panic(err) + } + return api.DeploymentRollupEntry{ + DatabaseType: "vitess", + Deployment: deployment, + Target: target, + Class: api.DeploymentPlanned, + ChangeSet: cs, + PlanFingerprint: fp, + } +} + +// groupMembers flattens the grouped members for assertions. +func groupMembers(groups []templates.DeploymentPlanGroup) [][]string { + out := make([][]string, len(groups)) + for i, g := range groups { + out[i] = g.Members + } + return out +} + +// Targets running the same work are described once and attributed to all of +// them, so a converged fleet does not repeat one plan per target. +func TestDeploymentPlanGroups_SameWorkGroupsTogether(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + rollup := api.PlanRollup{ + Clean: true, + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + plannedMember("primary", "testapp_1", email), + plannedMember("primary", "testapp_2", email), + plannedMember("primary", "testapp_3", email), + }, + } + + groups := deploymentPlanGroups(rollup) + assert.Equal(t, [][]string{{"primary/testapp_1", "primary/testapp_2", "primary/testapp_3"}}, groupMembers(groups)) + assert.True(t, groups[0].Primary) + assert.Equal(t, []string{email}, groups[0].Changes[0].Statements) + assert.False(t, groups[0].Empty()) +} + +// Targets that hold their own schemas can need different work. Each distinct +// plan is its own group, so the comment describes every plan the apply would +// run rather than the reviewed one alone. +func TestDeploymentPlanGroups_DifferentWorkSplits(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + phone := "ALTER TABLE users ADD COLUMN phone VARCHAR(32)" + rollup := api.PlanRollup{ + Clean: true, + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + plannedMember("primary", "testapp_1", email), + plannedMember("primary", "testapp_2", phone, email), + plannedMember("primary", "testapp_3", email), + }, + } + + groups := deploymentPlanGroups(rollup) + assert.Equal(t, [][]string{ + {"primary/testapp_1", "primary/testapp_3"}, + {"primary/testapp_2"}, + }, groupMembers(groups)) + assert.Equal(t, []string{email}, groups[0].Changes[0].Statements) + assert.Equal(t, []string{phone, email}, groups[1].Changes[0].Statements, + "a group carries the plan its own members would run, not the reviewed one") +} + +// Targets already at the desired schema form a group of their own, which the +// comment can name. Folding them into the changing targets would tell an +// operator the apply runs DDL on targets it will not touch. +func TestDeploymentPlanGroups_ConvergedTargetsAreTheirOwnGroup(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + rollup := api.PlanRollup{ + Clean: true, + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + plannedMember("primary", "testapp_1", email), + plannedMember("primary", "testapp_2"), + plannedMember("primary", "testapp_3", email), + plannedMember("primary", "testapp_4"), + }, + } + + groups := deploymentPlanGroups(rollup) + assert.Equal(t, [][]string{ + {"primary/testapp_1", "primary/testapp_3"}, + {"primary/testapp_2", "primary/testapp_4"}, + }, groupMembers(groups)) + assert.False(t, groups[0].Empty()) + assert.True(t, groups[1].Empty(), "targets with nothing to apply are named, not dropped") +} + +// The primary's group comes first whatever the primary's own plan, because the +// reviewed plan is the one the operator has already read. +func TestDeploymentPlanGroups_PrimaryGroupComesFirst(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + rollup := api.PlanRollup{ + Clean: true, + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + plannedMember("primary", "testapp_1"), + plannedMember("primary", "testapp_2", email), + plannedMember("primary", "testapp_3", email), + }, + } + + groups := deploymentPlanGroups(rollup) + assert.True(t, groups[0].Primary) + assert.Equal(t, []string{"primary/testapp_1"}, groups[0].Members) + assert.True(t, groups[0].Empty(), "the primary having nothing to apply does not move its group") +} + +// Grouping describes targets that were planned, so a rollup that blocked +// carries none: the operator's next step is the target that could not be +// planned, not the plans of an apply that cannot run. +func TestDeploymentDriftPreview_BlockedRollupIsNotGrouped(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + blocked := plannedMember("primary", "testapp_2") + blocked.Class = api.DeploymentErrored + blocked.PlanFingerprint = "" + blocked.ChangeSet = tern.ChangeSet{} + rollup := api.PlanRollup{ + Clean: false, + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + plannedMember("primary", "testapp_1", email), + blocked, + }, + } + + preview := deploymentDriftPreview(rollup) + assert.Empty(t, preview.Plans) + assert.Len(t, preview.Deployments, 2) +} + +// Members required to match each other are not grouped: a clean mirrored rollup +// has already proved they are one group, and re-reporting that in the vocabulary +// of a fleet free to diverge would read as an outcome rather than the +// requirement that let the check pass. +func TestDeploymentDriftPreview_MirroredMembersAreNotGrouped(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + eu := plannedMember("eu", "eu", email) + eu.Class = api.DeploymentMatch + au := plannedMember("au", "au", email) + au.Class = api.DeploymentMatch + rollup := api.PlanRollup{ + Clean: true, + Planning: api.PlanMirrored, + Entries: []api.DeploymentRollupEntry{eu, au}, + } + + preview := deploymentDriftPreview(rollup) + assert.Empty(t, preview.Plans) +} + +// A clean independent rollup reaches the comment already grouped, so the +// rendering never has to fall back to describing the contract instead of this +// round's plans. +func TestDeploymentDriftPreview_CleanIndependentRollupCarriesGroups(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + rollup := api.PlanRollup{ + Clean: true, + Planning: api.PlanIndependent, + Entries: []api.DeploymentRollupEntry{ + plannedMember("primary", "testapp_1", email), + plannedMember("primary", "testapp_2"), + }, + } + + preview := deploymentDriftPreview(rollup) + assert.Len(t, preview.Plans, 2) +} + +// A member's plan reaches the comment in the same shape the reviewed plan does, +// so a group's changes render through the code that renders the plan a reviewer +// has already read. +func TestMemberPlanChanges_CarriesNamespaceStatements(t *testing.T) { + cs := tern.ChangeSet{Changes: []*ternv1.SchemaChange{{ + Namespace: "testapp", + TableChanges: []*ternv1.TableChange{ + {TableName: "users", Ddl: "ALTER TABLE users ADD COLUMN email VARCHAR(255)"}, + {TableName: "orders", Ddl: "ALTER TABLE orders ADD COLUMN total BIGINT"}, + }, + }}} + + changes := memberPlanChanges(cs) + assert.Equal(t, []templates.KeyspaceChangeData{{ + Keyspace: "testapp", + Statements: []string{ + "ALTER TABLE users ADD COLUMN email VARCHAR(255)", + "ALTER TABLE orders ADD COLUMN total BIGINT", + }, + }}, changes) +} + +// A sharded namespace keeps both views of its changes, so the comment can show +// what applies to which shard rather than a namespace-level view that hides a +// shard. +func TestMemberPlanChanges_KeepsPerShardChanges(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + cs := tern.ChangeSet{ + Changes: []*ternv1.SchemaChange{{ + Namespace: "testapp", + TableChanges: []*ternv1.TableChange{{TableName: "users", Ddl: email}}, + }}, + Shards: []*ternv1.ShardPlan{ + {Namespace: "testapp", Shard: "-80", Changes: []*ternv1.TableChange{{TableName: "users", Ddl: email}}}, + {Namespace: "testapp", Shard: "80-", Changes: []*ternv1.TableChange{{TableName: "users", Ddl: email}}}, + }, + } + + changes := memberPlanChanges(cs) + require.Len(t, changes, 1) + assert.Equal(t, []string{email}, changes[0].Statements) + assert.Equal(t, []templates.KeyspaceShardChange{ + {Shard: "-80", Statements: []string{email}}, + {Shard: "80-", Statements: []string{email}}, + }, changes[0].Shards) +} + +// A shard already at the desired schema while its siblings change is carried as +// satisfied rather than dropped, so a partially-applied namespace shows its +// divergent state instead of looking uniform. +func TestMemberPlanChanges_MarksSatisfiedShards(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + cs := tern.ChangeSet{ + Changes: []*ternv1.SchemaChange{{ + Namespace: "testapp", + TableChanges: []*ternv1.TableChange{{TableName: "users", Ddl: email}}, + }}, + Shards: []*ternv1.ShardPlan{ + {Namespace: "testapp", Shard: "-80", Changes: []*ternv1.TableChange{{TableName: "users", Ddl: email}}}, + {Namespace: "testapp", Shard: "80-"}, + }, + } + + changes := memberPlanChanges(cs) + require.Len(t, changes[0].Shards, 2) + assert.False(t, changes[0].Shards[0].Satisfied) + assert.True(t, changes[0].Shards[1].Satisfied) +} + +// A namespace carried only by shard rows still reaches the comment. Dropping it +// would remove work from a plan the comment claims to describe in full. +func TestMemberPlanChanges_KeepsShardOnlyNamespace(t *testing.T) { + email := "ALTER TABLE users ADD COLUMN email VARCHAR(255)" + cs := tern.ChangeSet{Shards: []*ternv1.ShardPlan{ + {Namespace: "testapp", Shard: "-80", Changes: []*ternv1.TableChange{{TableName: "users", Ddl: email}}}, + }} + + changes := memberPlanChanges(cs) + require.Len(t, changes, 1) + assert.Equal(t, "testapp", changes[0].Keyspace) + assert.Equal(t, []string{email}, changes[0].Shards[0].Statements) +} + +// A vschema rewrite carries no table DDL. It reaches the comment as a change the +// namespace needs, so a plan that only rewrites the vschema is not mistaken for +// a namespace with nothing to apply. +func TestMemberPlanChanges_CarriesVSchemaChange(t *testing.T) { + cs := tern.ChangeSet{Changes: []*ternv1.SchemaChange{{ + Namespace: "testapp", + Metadata: map[string]string{ + apitypes.VSchemaChangedMetadataKey: "true", + apitypes.VSchemaDiffMetadataKey: "+ table users", + }, + }}} + + changes := memberPlanChanges(cs) + require.Len(t, changes, 1) + assert.True(t, changes[0].VSchemaChanged) + assert.Equal(t, "+ table users", changes[0].VSchemaDiff) + assert.Empty(t, changes[0].Statements) + assert.False(t, templates.DeploymentPlanGroup{Changes: changes}.Empty()) +} + +// A member already at the desired schema produces no changes at all, which is +// the group the comment names as having nothing to apply. +func TestMemberPlanChanges_EmptyPlanHasNoChanges(t *testing.T) { + assert.Empty(t, memberPlanChanges(tern.ChangeSet{})) + assert.True(t, templates.DeploymentPlanGroup{}.Empty()) +} diff --git a/pkg/webhook/templates/plan.go b/pkg/webhook/templates/plan.go index d7b270d32..e74a31e88 100644 --- a/pkg/webhook/templates/plan.go +++ b/pkg/webhook/templates/plan.go @@ -203,6 +203,42 @@ type DeploymentDriftData struct { // rollup means every target was planned rather than that they agree — which // is the opposite of what the mirrored wording says. Independent bool + // Plans is the members grouped by the plan they would run, one entry per + // distinct plan, the primary's first. It says how much the members actually + // agree this round, which the contract alone cannot: members that are free + // to differ usually do not. Set only for a clean rollup of independent + // members — members expected to match each other say nothing by matching, + // and a blocked rollup describes each member on its own instead. + Plans []DeploymentPlanGroup +} + +// DeploymentPlanGroup is the members of a rollout that would run the same plan. +// Members share a group exactly when their plans are identical work, so a group +// is what the comment can describe once and attribute to all of them. +type DeploymentPlanGroup struct { + // Members names the group's members the way an operator addresses them, in + // rollout order. + Members []string + // Primary marks the group the reviewed primary member belongs to. Exactly + // one group carries it, and it is the group operators read first: the + // reviewed plan is the one they have already seen. + Primary bool + // Changes is the plan every member of the group would run, in the same shape + // the comment renders the reviewed plan itself. Empty for a group whose + // members are already at the desired schema. + Changes []KeyspaceChangeData +} + +// Empty reports that the group's members are already at the desired schema and +// would apply nothing. That is a plan in its own right, not a missing one, and +// naming it is the difference between a fleet that is converging and one the +// comment has quietly left out. +// A vschema rewrite carries no DDL and is still work, so a group is counted the +// same way the comment counts the reviewed plan: statements and vschema +// rewrites together. +func (g DeploymentPlanGroup) Empty() bool { + statements, vschema := countChanges(g.Changes) + return statements+vschema == 0 } // DeploymentDriftEntry is one rollout member's classification against the @@ -1138,8 +1174,8 @@ func writeDeploymentDrift(sb *strings.Builder, drift *DeploymentDriftData) { names := inlineCodeList(driftMemberNames(drift.Deployments)) if drift.Clean { if drift.Independent { - fmt.Fprintf(sb, "✅ **Planned separately for all %d targets** (%s) — each target holds its own schema, so their plans are not expected to match.\n\n", - len(drift.Deployments), strings.Join(names, ", ")) + fmt.Fprintf(sb, "✅ **Planned separately for all %d targets** (%s) — %s\n\n", + len(drift.Deployments), strings.Join(names, ", "), describePlanGroups(drift.Plans)) return } fmt.Fprintf(sb, "✅ **Same plan on all %d deployments** (%s).\n\n", @@ -1182,6 +1218,57 @@ func writeDeploymentDrift(sb *strings.Builder, drift *DeploymentDriftData) { sb.WriteString("\n") } +// describePlanGroups states how much the members' plans actually agree this +// round: how many distinct plans there are, and how many members already hold +// the desired schema and would apply nothing. +// +// The contract alone cannot say this. Targets that are free to differ usually do +// not, and a fleet converging over time — some targets changed, the rest already +// there — is otherwise invisible in a comment that only reports what members are +// permitted to do. +// +// With no groups it falls back to the contract, which is all that is known: a +// caller that did not group the members has established nothing about this round +// beyond what the configuration already said. +func describePlanGroups(groups []DeploymentPlanGroup) string { + if len(groups) == 0 { + return "each target holds its own schema, so their plans are not expected to match." + } + var plans, changing, converged int + for _, g := range groups { + if g.Empty() { + converged += len(g.Members) + continue + } + plans++ + changing += len(g.Members) + } + + switch { + case plans == 0: + return "every target is already at this schema." + case plans == 1 && converged == 0: + return "every target needs the same change." + case plans == 1: + return fmt.Sprintf("%s this change, %s already at this schema.", + countedVerb(changing, "needs", "need"), countedVerb(converged, "is", "are")) + case converged == 0: + return fmt.Sprintf("%d distinct plans. Each target applies its own.", plans) + default: + return fmt.Sprintf("%d distinct plans across the %d targets that change; %s already at this schema.", + plans, changing, countedVerb(converged, "is", "are")) + } +} + +// countedVerb renders a count and the verb that agrees with it, e.g. "3 need" or +// "1 needs". +func countedVerb(n int, singular, plural string) string { + if n == 1 { + return "1 " + singular + } + return fmt.Sprintf("%d %s", n, plural) +} + // driftDetailSuffix renders a deployment's drift detail as a trailing clause, or // an empty string when there is no detail. func driftDetailSuffix(detail string) string { diff --git a/pkg/webhook/templates/plan_drift_test.go b/pkg/webhook/templates/plan_drift_test.go index b3112d773..613ac477b 100644 --- a/pkg/webhook/templates/plan_drift_test.go +++ b/pkg/webhook/templates/plan_drift_test.go @@ -1,6 +1,7 @@ package templates import ( + "fmt" "strings" "testing" @@ -290,13 +291,144 @@ func TestRenderPlanComment_DriftCleanNamesMultiTargetMembers(t *testing.T) { {Deployment: "primary", Target: "testapp-002", Class: "planned"}, {Deployment: "eu-west", Target: "orders-eu", Class: "planned"}, }, + Plans: []DeploymentPlanGroup{{ + Members: []string{"primary/testapp-001", "primary/testapp-002", "eu-west"}, + Primary: true, + Changes: planGroupChanges(1), + }}, }, } out := RenderPlanComment(data) assert.Contains(t, out, "Planned separately for all 3 targets") assert.Contains(t, out, "`primary/testapp-001`, `primary/testapp-002`, `eu-west`") - assert.True(t, strings.Contains(out, "each target holds its own schema")) + assert.True(t, strings.Contains(out, "every target needs the same change")) +} + +// Targets are free to hold different schemas, so what an operator needs to know +// is how much they agree this round. The comment says how many distinct plans +// the apply would run and how many targets are already there, which the contract +// alone cannot tell them. +func TestRenderPlanComment_PlanGroupsDescribeThisRound(t *testing.T) { + render := func(plans []DeploymentPlanGroup) string { + members := make([]DeploymentDriftEntry, 0, 5) + for _, g := range plans { + for range g.Members { + members = append(members, DeploymentDriftEntry{Deployment: "primary", Class: "planned"}) + } + } + members[0].Primary = true + return RenderPlanComment(PlanCommentData{ + Database: "testapp", Environment: "production", IsMySQL: true, + Changes: []KeyspaceChangeData{{ + Keyspace: "testapp", + Statements: []string{"ALTER TABLE `users` ADD COLUMN `email` varchar(255)"}, + }}, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, Clean: true, Independent: true, + Deployments: members, + Plans: plans, + }, + }) + } + group := func(statements int, members ...string) DeploymentPlanGroup { + return DeploymentPlanGroup{Members: members, Changes: planGroupChanges(statements)} + } + + cases := []struct { + name string + plans []DeploymentPlanGroup + expect string + }{ + { + name: "every target needs the same change", + plans: []DeploymentPlanGroup{group(1, "a", "b", "c")}, + expect: "every target needs the same change.", + }, + { + name: "some targets are already there", + plans: []DeploymentPlanGroup{group(1, "a", "c", "d"), group(0, "b", "e")}, + expect: "3 need this change, 2 are already at this schema.", + }, + { + name: "a single target still needs it", + plans: []DeploymentPlanGroup{group(1, "a"), group(0, "b")}, + expect: "1 needs this change, 1 is already at this schema.", + }, + { + name: "targets need different changes", + plans: []DeploymentPlanGroup{group(1, "a", "b", "c"), group(2, "d", "e")}, + expect: "2 distinct plans. Each target applies its own.", + }, + { + name: "different changes with some already there", + plans: []DeploymentPlanGroup{group(1, "a", "b"), group(2, "c"), group(0, "d", "e")}, + expect: "2 distinct plans across the 3 targets that change; 2 are already at this schema.", + }, + { + name: "the whole fleet is already there", + plans: []DeploymentPlanGroup{group(0, "a", "b", "c")}, + expect: "every target is already at this schema.", + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.Contains(t, render(tc.plans), tc.expect) + }) + } +} + +// A plan that only rewrites the vschema runs no DDL, and is still work. It is +// described as a change the targets need rather than as a schema they already +// hold, which would tell an operator the apply does nothing. +func TestRenderPlanComment_VSchemaOnlyPlanIsNotAlreadyApplied(t *testing.T) { + data := PlanCommentData{ + Database: "testapp", Environment: "production", + Changes: []KeyspaceChangeData{{Keyspace: "testapp", VSchemaChanged: true}}, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, Clean: true, Independent: true, + Deployments: []DeploymentDriftEntry{ + {Deployment: "primary", Target: "testapp_1", Primary: true, Class: "planned"}, + {Deployment: "primary", Target: "testapp_2", Class: "planned"}, + }, + Plans: []DeploymentPlanGroup{ + { + Members: []string{"primary/testapp_1"}, + Primary: true, + Changes: []KeyspaceChangeData{{Keyspace: "testapp", VSchemaChanged: true}}, + }, + {Members: []string{"primary/testapp_2"}}, + }, + }, + } + + out := RenderPlanComment(data) + assert.Contains(t, out, "1 needs this change, 1 is already at this schema.") +} + +// A rollup that reaches the comment ungrouped states the contract and nothing +// more. Claiming the targets agree — or that they do not — would be a claim +// about plans nobody compared. +func TestRenderPlanComment_UngroupedIndependentRollupStatesTheContract(t *testing.T) { + data := PlanCommentData{ + Database: "testapp", Environment: "production", IsMySQL: true, + Changes: []KeyspaceChangeData{{ + Keyspace: "testapp", + Statements: []string{"ALTER TABLE `users` ADD COLUMN `email` varchar(255)"}, + }}, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, Clean: true, Independent: true, + Deployments: []DeploymentDriftEntry{ + {Deployment: "primary", Target: "testapp_1", Primary: true, Class: "planned"}, + {Deployment: "primary", Target: "testapp_2", Class: "planned"}, + }, + }, + } + + out := RenderPlanComment(data) + assert.Contains(t, out, "each target holds its own schema, so their plans are not expected to match.") + assert.NotContains(t, out, "distinct plans") } // A member name reaches the comment from server config, so the rollup renders @@ -324,3 +456,16 @@ func TestRenderPlanComment_DriftContainsHostileMemberNames(t *testing.T) { assert.NotContains(t, out, "\n## Injected", "a name must not start a heading of its own") assert.Contains(t, out, "`` us` ## Injected ``") } + +// planGroupChanges builds a group plan running the given number of statements. +// A group running none is already at the desired schema. +func planGroupChanges(statements int) []KeyspaceChangeData { + if statements == 0 { + return nil + } + ks := KeyspaceChangeData{Keyspace: "testapp"} + for i := range statements { + ks.Statements = append(ks.Statements, fmt.Sprintf("ALTER TABLE `t%d` ADD COLUMN `c` int", i)) + } + return []KeyspaceChangeData{ks} +} From f2ca09b1d26448d0b707213220ab77c525debb30 Mon Sep 17 00:00:00 2001 From: Armand Parajon Date: Thu, 17 Sep 2026 14:19:56 -0400 Subject: [PATCH 34/34] feat(github): show every target's plan, not just the reviewed one Targets under a `targets:` list are planned each and converge on their own, so they are free to hold different schemas and usually do. The plan comment named that divergence but still rendered only the reviewed plan's DDL, which left an operator authorizing work the comment never showed them. Each distinct plan now renders under the members that would run it. The reviewed plan's block stays open, the rest collapse behind a consent line saying the apply runs them too, and a member group already at the desired schema is named rather than hidden. The summary line counts the rollout instead of the reviewed plan alone. A primary already at the desired schema no longer short-circuits the comment to "no schema changes detected" when its siblings still have work to apply, which upholds UX-3: the comment describes the apply an operator would authorize, not one member of it. Co-Authored-By: Claude Opus 5 --- TEMPLATES.md | 78 ++++++++ pkg/cmd/internal/templates/preview_comment.go | 4 + pkg/webhook/templates/plan.go | 166 +++++++++++++++++- pkg/webhook/templates/plan_drift_test.go | 130 ++++++++++++++ pkg/webhook/templates/preview.go | 83 +++++++++ 5 files changed, 457 insertions(+), 4 deletions(-) diff --git a/TEMPLATES.md b/TEMPLATES.md index 86939d55c..ab1b05034 100644 --- a/TEMPLATES.md +++ b/TEMPLATES.md @@ -1421,6 +1421,84 @@ ALTER TABLE `products` ADD INDEX `idx_category_price`(`category`, `price`); 📋 **Plan**: **2** tables to create, **1** table to alter +--- + +▶️ **To apply** all schema changes from this PR, comment: +``` +schemabot apply -e production +``` + +
+ +
+Targets Converging + + +## Schema Change Plan — Production + +**Database**: `testapp` | **Type**: `MySQL` | **Schema Name**: `testapp` + +*Requested by @jackjackbits at 2026-01-01 00:00:00 UTC · planned from [`abcdef1`](https://github.com/block/schemabot/commit/abcdef1234567890abcdef1234567890abcdef12)* + +✅ **Planned separately for all 3 targets** (`primary/testapp_1`, `primary/testapp_2`, `primary/testapp_3`) — 2 need this change, 1 is already at this schema. + +**`primary/testapp_1` (primary), `primary/testapp_3`** — 1 DDL statement + +```sql +ALTER TABLE `users` ADD COLUMN `email` varchar(255); +``` + +**`primary/testapp_2`** — already at this schema, nothing to apply. + +📋 **Plan**: 1 DDL statement on 2 of 3 targets + + +--- + +▶️ **To apply** all schema changes from this PR, comment: +``` +schemabot apply -e production +``` + +
+ +
+Targets Diverging + + +## Schema Change Plan — Production + +**Database**: `testapp` | **Type**: `MySQL` | **Schema Name**: `testapp` + +*Requested by @jackjackbits at 2026-01-01 00:00:00 UTC · planned from [`abcdef1`](https://github.com/block/schemabot/commit/abcdef1234567890abcdef1234567890abcdef12)* + +✅ **Planned separately for all 3 targets** (`primary/testapp_1`, `primary/testapp_2`, `primary/testapp_3`) — 2 distinct plans. Each target applies its own. + +
+`primary/testapp_1` (primary), `primary/testapp_2` — 1 DDL statement + +```sql +ALTER TABLE `users` ADD COLUMN `email` varchar(255); +``` + +
+ +
+`primary/testapp_3` — 2 DDL statements + +```sql +ALTER TABLE `users` ADD COLUMN `email` varchar(255); + +ALTER TABLE `users` ADD INDEX `idx_email`(`email`); +``` + +
+ +⚠️ Applying runs each target's own plan, including the ones collapsed above. + +📋 **Plan**: 2 distinct plans on 3 targets + + --- ▶️ **To apply** all schema changes from this PR, comment: diff --git a/pkg/cmd/internal/templates/preview_comment.go b/pkg/cmd/internal/templates/preview_comment.go index 0f91b985f..c3e697dab 100644 --- a/pkg/cmd/internal/templates/preview_comment.go +++ b/pkg/cmd/internal/templates/preview_comment.go @@ -85,6 +85,8 @@ func previewCommentAllOutput() { {"DEPLOYMENT DRIFT (CLEAN)", func() { fmt.Print(webhooktemplates.PreviewCommentPlanDriftClean()) }}, {"DEPLOYMENT DRIFT (DETECTED)", func() { fmt.Print(webhooktemplates.PreviewCommentPlanDriftDetected()) }}, {"DEPLOYMENT DRIFT (COULD NOT VERIFY)", func() { fmt.Print(webhooktemplates.PreviewCommentPlanDriftUnverified()) }}, + {"TARGETS CONVERGING", func() { fmt.Print(webhooktemplates.PreviewCommentPlanTargetsConverging()) }}, + {"TARGETS DIVERGING", func() { fmt.Print(webhooktemplates.PreviewCommentPlanTargetsDiverging()) }}, {"HELP COMMENT", func() { fmt.Print(webhooktemplates.PreviewCommentHelp()) }}, {"SUPPORT CHANNEL FOOTER", func() { fmt.Print(webhooktemplates.PreviewCommentSupportChannel()) }}, {"AGENT HINT FOOTER", func() { fmt.Print(webhooktemplates.PreviewCommentAgentHint()) }}, @@ -193,6 +195,8 @@ func previewCommentPlanAllOutput() { {"DEPLOYMENT DRIFT (CLEAN)", func() { fmt.Print(webhooktemplates.PreviewCommentPlanDriftClean()) }}, {"DEPLOYMENT DRIFT (DETECTED)", func() { fmt.Print(webhooktemplates.PreviewCommentPlanDriftDetected()) }}, {"DEPLOYMENT DRIFT (COULD NOT VERIFY)", func() { fmt.Print(webhooktemplates.PreviewCommentPlanDriftUnverified()) }}, + {"TARGETS CONVERGING", func() { fmt.Print(webhooktemplates.PreviewCommentPlanTargetsConverging()) }}, + {"TARGETS DIVERGING", func() { fmt.Print(webhooktemplates.PreviewCommentPlanTargetsDiverging()) }}, {"DROP COLUMN BLOCKED", func() { fmt.Print(webhooktemplates.PreviewCommentDropColumnBlocked()) }}, {"DROP INDEX BLOCKED", func() { fmt.Print(webhooktemplates.PreviewCommentDropIndexBlocked()) }}, {"SCHEMA LINT ERRORS BLOCKED", func() { fmt.Print(webhooktemplates.PreviewCommentLintErrorsBlocked()) }}, diff --git a/pkg/webhook/templates/plan.go b/pkg/webhook/templates/plan.go index e74a31e88..29e218905 100644 --- a/pkg/webhook/templates/plan.go +++ b/pkg/webhook/templates/plan.go @@ -349,11 +349,22 @@ func RenderPlanComment(data PlanCommentData) string { totalStatements, keyspacesWithVSchema := countChanges(data.Changes) totalChanges := totalStatements + keyspacesWithVSchema + // When the rollout's members would run more than one plan, each group's plan + // is rendered under the members it applies to. The reviewed plan is then one + // group among several, so the single block below is suppressed: rendering it + // would show part of the apply as though it were all of it. + groups := data.planGroups() + // No changes — short-circuit with a single clean message. The // ignore_namespaces disclosure still renders: a no-changes result is // exactly where a reviewer needs to tell a withheld namespace apart from a // genuinely unchanged one. - if totalChanges == 0 { + // + // The count is of the reviewed plan, which is the primary's. A primary + // already at the desired schema says nothing about its sibling targets, so a + // rollout whose groups carry work is never short-circuited on it — that would + // report an apply that changes several targets as changing nothing. + if totalChanges == 0 && !groupsCarryWork(groups) { writeNoChangesDetected(&sb, data) if len(data.IgnoredNamespaces) > 0 || hasExemptTables(data.ExemptTables) { sb.WriteString("\n") @@ -364,7 +375,11 @@ func RenderPlanComment(data PlanCommentData) string { } // Detailed changes - writeKeyspaceChanges(&sb, data) + if len(groups) > 0 { + writePlanGroups(&sb, data, groups) + } else { + writeKeyspaceChanges(&sb, data) + } // Blocked changes — statements the engine refuses. Unlike unsafe changes, // these cannot be acknowledged away: the apply will fail on them. Shown on @@ -426,8 +441,15 @@ func RenderPlanComment(data PlanCommentData) string { writeErrors(&sb, data.Errors) } - // Summary and options (after DDL, matching CLI layout) - writePlanSummary(&sb, data, totalStatements, keyspacesWithVSchema) + // Summary and options (after DDL, matching CLI layout). When the members run + // more than one plan, the summary counts the rollout rather than the reviewed + // plan: reporting the primary's tables as the plan would understate an apply + // that runs different work on other targets. + if len(groups) > 0 { + writePlanGroupSummary(&sb, data, groups) + } else { + writePlanSummary(&sb, data, totalStatements, keyspacesWithVSchema) + } writeOptions(&sb, data) // Footer @@ -1218,6 +1240,142 @@ func writeDeploymentDrift(sb *strings.Builder, drift *DeploymentDriftData) { sb.WriteString("\n") } +// planGroups returns the member plan groups when they, rather than the reviewed +// plan on its own, should carry the comment's DDL. +// +// That is when the members would run more than one distinct plan. With a single +// group every member runs the reviewed plan, so the comment already shows what +// the apply does and attributing it to a member list would only repeat the line +// above it. +func (d PlanCommentData) planGroups() []DeploymentPlanGroup { + if d.DeploymentDrift == nil || len(d.DeploymentDrift.Plans) < 2 { + return nil + } + return d.DeploymentDrift.Plans +} + +// groupsCarryWork reports whether any group would apply something. +func groupsCarryWork(groups []DeploymentPlanGroup) bool { + return slices.ContainsFunc(groups, func(g DeploymentPlanGroup) bool { return !g.Empty() }) +} + +// writePlanGroups renders each distinct plan under the members that would run +// it, so a reviewer reads one block per plan rather than one per target, and +// every target's work is on the comment rather than the reviewed one's alone. +// +// The primary's group is open and the rest are collapsed when there is more than +// one plan to run: the reviewed plan is the one an operator has already read, so +// it is the block they should not have to expand. A single plan is never +// collapsed, since there is nothing to collapse it against. +func writePlanGroups(sb *strings.Builder, data PlanCommentData, groups []DeploymentPlanGroup) { + collapse := slices.ContainsFunc(groups, func(g DeploymentPlanGroup) bool { return !g.Empty() && !g.Primary }) + + for _, g := range groups { + heading := planGroupHeading(g) + if g.Empty() { + // Nothing to expand, and the summary is the whole story: these + // members apply nothing. Naming them is what keeps a converging + // fleet visible. + fmt.Fprintf(sb, "**%s** — already at this schema, nothing to apply.\n\n", heading) + continue + } + + label := planGroupWorkLabel(countChanges(g.Changes)) + + // A group's plan renders through the same code that renders the reviewed + // plan, so one target's DDL is never formatted by a second renderer that + // agrees with the first until it does not. + scoped := data + scoped.Changes = g.Changes + + if !collapse { + fmt.Fprintf(sb, "**%s** — %s\n\n", heading, label) + writeKeyspaceChanges(sb, scoped) + continue + } + + open := "" + if g.Primary { + open = " open" + } + fmt.Fprintf(sb, "\n%s — %s\n\n", open, heading, label) + writeKeyspaceChanges(sb, scoped) + sb.WriteString("
\n\n") + } + + if collapse { + // The operator is about to authorize every group's plan, not just the one + // they can see. Saying so is the consent statement, which is why it is the + // only attention line here: the targets differing is the contract. + sb.WriteString(glyph.Attention + " Applying runs each target's own plan, including the ones collapsed above.\n\n") + } +} + +// writePlanGroupSummary states what the apply runs across the whole rollout, +// standing in for the reviewed plan's own summary. Counting the primary's tables +// would understate an apply that runs different work on the other targets, and +// summing every group's would overstate it: one target's statement is not two +// because a sibling runs it too. +// +// So the summary counts targets. A rollout with one plan says how much of the +// fleet still needs it; a rollout with several says how many plans there are, +// and the blocks above say what each one is. +// +// There is no "on all N targets" wording, because the summary is only reached +// once the members hold more than one distinct plan. A single plan among them +// therefore means the rest are already at the desired schema, and the reader +// needs to be told how much of the fleet that leaves. +func writePlanGroupSummary(sb *strings.Builder, data PlanCommentData, groups []DeploymentPlanGroup) { + var plans, changing, targets int + var only DeploymentPlanGroup + for _, g := range groups { + targets += len(g.Members) + if g.Empty() { + continue + } + plans++ + changing += len(g.Members) + only = g + } + + if plans == 1 { + fmt.Fprintf(sb, "📋 **Plan**: %s on %d of %d targets\n\n", planGroupWorkLabel(countChanges(only.Changes)), changing, targets) + } else { + fmt.Fprintf(sb, "📋 **Plan**: %d distinct plans on %d targets\n\n", plans, targets) + } + + // Disclosed directly under the plan summary so the exclusion reads as part of + // the plan result, the same way the single-plan summary discloses it. + writeIgnoredNamespaces(sb, data.IgnoredNamespaces) + writeExemptTables(sb, data.ExemptTables) +} + +// planGroupHeading names a group's members, marking the reviewed member so an +// operator can tell the plan they have already read from the ones they have not. +func planGroupHeading(g DeploymentPlanGroup) string { + names := inlineCodeList(g.Members) + // The primary is first in rollout order, so it is its group's first member. + if g.Primary && len(names) > 0 { + names[0] += " (primary)" + } + return strings.Join(names, ", ") +} + +// planGroupWorkLabel says what a group's plan runs, e.g. "1 DDL statement" or +// "2 DDL statements and a vschema update". +func planGroupWorkLabel(statements, vschemaNamespaces int) string { + switch { + case statements == 0: + return fmt.Sprintf("%d vschema %s", vschemaNamespaces, pluralize("update", vschemaNamespaces)) + case vschemaNamespaces == 0: + return fmt.Sprintf("%d DDL %s", statements, pluralize("statement", statements)) + default: + return fmt.Sprintf("%d DDL %s and %d vschema %s", + statements, pluralize("statement", statements), + vschemaNamespaces, pluralize("update", vschemaNamespaces)) + } +} + // describePlanGroups states how much the members' plans actually agree this // round: how many distinct plans there are, and how many members already hold // the desired schema and would apply nothing. diff --git a/pkg/webhook/templates/plan_drift_test.go b/pkg/webhook/templates/plan_drift_test.go index 613ac477b..7b532a1b5 100644 --- a/pkg/webhook/templates/plan_drift_test.go +++ b/pkg/webhook/templates/plan_drift_test.go @@ -457,6 +457,136 @@ func TestRenderPlanComment_DriftContainsHostileMemberNames(t *testing.T) { assert.Contains(t, out, "`` us` ## Injected ``") } +// renderGroupedPlan renders a plan comment for an independent rollout whose +// members run the given groups. reviewed is the primary's own plan, which is the +// one the comment would render on its own if the members did not disagree. +func renderGroupedPlan(reviewed []KeyspaceChangeData, plans []DeploymentPlanGroup) string { + var members []DeploymentDriftEntry + for _, g := range plans { + for range g.Members { + members = append(members, DeploymentDriftEntry{Deployment: "primary", Class: "planned"}) + } + } + members[0].Primary = true + return RenderPlanComment(PlanCommentData{ + Database: "testapp", Environment: "production", DatabaseType: "mysql", IsMySQL: true, + Changes: reviewed, + DeploymentDrift: &DeploymentDriftData{ + Computed: true, Clean: true, Independent: true, + Deployments: members, + Plans: plans, + }, + }) +} + +// A rollout whose targets all run the reviewed plan is described by the reviewed +// plan itself. Attributing the one block to a member list would only repeat the +// line above it, so the comment renders exactly as it does without a rollout. +func TestRenderPlanComment_OnePlanRendersAsTheReviewedPlan(t *testing.T) { + out := renderGroupedPlan(planGroupChanges(1), []DeploymentPlanGroup{ + {Members: []string{"primary/a", "primary/b"}, Primary: true, Changes: planGroupChanges(1)}, + }) + + assert.Contains(t, out, "ALTER TABLE `t0` ADD COLUMN `c` int") + assert.NotContains(t, out, "
") + assert.NotContains(t, out, "Applying runs each target's own plan") + // The reviewed plan's own summary, not the rollout's. + assert.Contains(t, out, "📋 **Plan**: **1** table to alter") +} + +// Targets that are free to differ usually do, so each distinct plan is rendered +// under the members that would run it. The reviewed plan is the one an operator +// has already read, so its block is the one left open. +func TestRenderPlanComment_DistinctPlansRenderUnderTheirMembers(t *testing.T) { + out := renderGroupedPlan(planGroupChanges(1), []DeploymentPlanGroup{ + {Members: []string{"primary/a", "primary/b"}, Primary: true, Changes: planGroupChanges(1)}, + {Members: []string{"eu/c"}, Changes: planGroupChanges(2)}, + }) + + assert.Contains(t, out, "
\n`primary/a` (primary), `primary/b` — 1 DDL statement") + assert.Contains(t, out, "
\n`eu/c` — 2 DDL statements") + // Every group's DDL is on the comment, not the reviewed one's alone. + assert.Contains(t, out, "ALTER TABLE `t1` ADD COLUMN `c` int") + assert.Contains(t, out, "⚠️ Applying runs each target's own plan, including the ones collapsed above.") + assert.Contains(t, out, "📋 **Plan**: 2 distinct plans on 3 targets") +} + +// A target already holding the desired schema has no plan to collapse, so it is +// named rather than hidden: a converging fleet is what the operator is watching +// for, and the remaining group's plan stays open. +func TestRenderPlanComment_ConvergedGroupIsNamedNotCollapsed(t *testing.T) { + out := renderGroupedPlan(planGroupChanges(1), []DeploymentPlanGroup{ + {Members: []string{"primary/a"}, Primary: true, Changes: planGroupChanges(1)}, + {Members: []string{"primary/b", "eu/c"}, Changes: nil}, + }) + + assert.Contains(t, out, "**`primary/a` (primary)** — 1 DDL statement") + assert.Contains(t, out, "**`primary/b`, `eu/c`** — already at this schema, nothing to apply.") + assert.NotContains(t, out, "
") + assert.NotContains(t, out, "Applying runs each target's own plan") + assert.Contains(t, out, "📋 **Plan**: 1 DDL statement on 1 of 3 targets") +} + +// The reviewed plan is the primary's, so a primary that is already at the +// desired schema says nothing about its siblings. The comment reports the work +// the apply would do on them rather than reporting the round as a no-op. +func TestRenderPlanComment_ConvergedPrimaryStillShowsSiblingWork(t *testing.T) { + out := renderGroupedPlan(nil, []DeploymentPlanGroup{ + {Members: []string{"primary/a"}, Primary: true, Changes: nil}, + {Members: []string{"eu/c"}, Changes: planGroupChanges(2)}, + }) + + assert.NotContains(t, out, "No schema changes detected") + assert.Contains(t, out, "**`primary/a` (primary)** — already at this schema, nothing to apply.") + assert.Contains(t, out, "ALTER TABLE `t1` ADD COLUMN `c` int") + assert.Contains(t, out, "📋 **Plan**: 2 DDL statements on 1 of 2 targets") +} + +// The summary line stands in for the reviewed plan's own, so it counts the +// rollout: how much of the fleet still needs the one plan, or how many plans +// there are when the targets disagree. +func TestRenderPlanComment_PlanSummaryCountsTheRollout(t *testing.T) { + cases := []struct { + name string + plans []DeploymentPlanGroup + expect string + }{ + { + name: "part of the fleet is already there", + plans: []DeploymentPlanGroup{ + {Members: []string{"a", "b"}, Primary: true, Changes: planGroupChanges(1)}, + {Members: []string{"c"}, Changes: nil}, + }, + expect: "📋 **Plan**: 1 DDL statement on 2 of 3 targets", + }, + { + name: "the targets disagree", + plans: []DeploymentPlanGroup{ + {Members: []string{"a"}, Primary: true, Changes: planGroupChanges(1)}, + {Members: []string{"b"}, Changes: planGroupChanges(3)}, + {Members: []string{"c"}, Changes: nil}, + }, + expect: "📋 **Plan**: 2 distinct plans on 3 targets", + }, + } + + for _, tc := range cases { + t.Run(tc.name, func(t *testing.T) { + assert.Contains(t, renderGroupedPlan(planGroupChanges(1), tc.plans), tc.expect) + }) + } +} + +// A vschema rewrite carries no DDL and is still work, so a group's label counts +// it alongside statements rather than describing the plan by its DDL alone. +func TestPlanGroupWorkLabel(t *testing.T) { + assert.Equal(t, "1 DDL statement", planGroupWorkLabel(1, 0)) + assert.Equal(t, "2 DDL statements", planGroupWorkLabel(2, 0)) + assert.Equal(t, "1 vschema update", planGroupWorkLabel(0, 1)) + assert.Equal(t, "2 vschema updates", planGroupWorkLabel(0, 2)) + assert.Equal(t, "2 DDL statements and 1 vschema update", planGroupWorkLabel(2, 1)) +} + // planGroupChanges builds a group plan running the given number of statements. // A group running none is already at the desired schema. func planGroupChanges(statements int) []KeyspaceChangeData { diff --git a/pkg/webhook/templates/preview.go b/pkg/webhook/templates/preview.go index bb039bcee..b4748b2ac 100644 --- a/pkg/webhook/templates/preview.go +++ b/pkg/webhook/templates/preview.go @@ -511,6 +511,89 @@ func PreviewCommentPlanDriftDetected() string { }) } +// previewTargetPlan is one target group's plan in the target-fleet previews. +func previewTargetPlan(statements ...string) []KeyspaceChangeData { + return []KeyspaceChangeData{{Keyspace: "testapp", Statements: statements}} +} + +// PreviewCommentPlanTargetsConverging renders a plan comment for an environment +// whose targets each hold their own schema, where some already have the reviewed +// change and the rest still need it — the shape a fleet rolling out over several +// PRs actually has. +func PreviewCommentPlanTargetsConverging() string { + return RenderPlanComment(PlanCommentData{ + Database: "testapp", + SchemaName: "testapp", + Environment: "production", + HeadSHA: previewHeadSHA, + Repository: previewRepository, + RequestedBy: previewRequestedBy, + IsMySQL: true, + DatabaseType: "mysql", + Changes: previewTargetPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), + DeploymentDrift: &DeploymentDriftData{ + Computed: true, + Clean: true, + Independent: true, + Deployments: []DeploymentDriftEntry{ + {Deployment: "primary", Target: "testapp_1", Primary: true, Class: "planned"}, + {Deployment: "primary", Target: "testapp_2", Class: "planned"}, + {Deployment: "primary", Target: "testapp_3", Class: "planned"}, + }, + Plans: []DeploymentPlanGroup{ + { + Members: []string{"primary/testapp_1", "primary/testapp_3"}, + Primary: true, + Changes: previewTargetPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), + }, + {Members: []string{"primary/testapp_2"}}, + }, + }, + }) +} + +// PreviewCommentPlanTargetsDiverging renders a plan comment for an environment +// whose targets would run different changes from each other. Under a targets +// list that is the contract rather than drift, so it renders on the success +// glyph, and every target's plan is on the comment the apply is authorized from. +func PreviewCommentPlanTargetsDiverging() string { + return RenderPlanComment(PlanCommentData{ + Database: "testapp", + SchemaName: "testapp", + Environment: "production", + HeadSHA: previewHeadSHA, + Repository: previewRepository, + RequestedBy: previewRequestedBy, + IsMySQL: true, + DatabaseType: "mysql", + Changes: previewTargetPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), + DeploymentDrift: &DeploymentDriftData{ + Computed: true, + Clean: true, + Independent: true, + Deployments: []DeploymentDriftEntry{ + {Deployment: "primary", Target: "testapp_1", Primary: true, Class: "planned"}, + {Deployment: "primary", Target: "testapp_2", Class: "planned"}, + {Deployment: "primary", Target: "testapp_3", Class: "planned"}, + }, + Plans: []DeploymentPlanGroup{ + { + Members: []string{"primary/testapp_1", "primary/testapp_2"}, + Primary: true, + Changes: previewTargetPlan("ALTER TABLE `users` ADD COLUMN `email` varchar(255)"), + }, + { + Members: []string{"primary/testapp_3"}, + Changes: previewTargetPlan( + "ALTER TABLE `users` ADD COLUMN `email` varchar(255)", + "ALTER TABLE `users` ADD INDEX `idx_email`(`email`)", + ), + }, + }, + }, + }) +} + // PreviewCommentPlanDriftUnverified renders a plan comment whose review-time // drift rollup could not be computed, so the plan check fails closed. func PreviewCommentPlanDriftUnverified() string {