Fix two defects a production-clone dress rehearsal exposed
Streamed a clone of a production store into the smoke VM - 3.6 GB, 12,361 settings, 6 accounts across 9 domains - and migrated it 0.15.5 -> 0.16.14 with the tool's own phases. The migration succeeded. Two defects surfaced that no smaller instance could have shown, plus one finding worth recording. 1. Account roles broke on production-shaped names. v0.16 stores an account as a local part plus a domain reference: a v0.15 account named "[email protected]" becomes name "john" with a domainId. The generator passed the full address and the server rejected it outright ("Invalid email local part"), failing the apply. The smoke instance used bare usernames - alice, bob - and never exercised this. Fixed to use the local part. And because local parts are unique only within a domain - [email protected] and [email protected] both become "postmaster" - an ambiguous one is now refused with a warning rather than risking an upsert that grants Admin to the wrong account. Verified on the clone: the one admin came out with roles {"@type": "Admin"} and the other five accounts untouched. 2. Cutover's health check conflated liveness with credentials. A config fallback-admin does not survive the migration - v0.16's config is a store pointer, so the old [authentication.fallback-admin] block simply ceases to exist - so the credentials supplied for the pre-migration instance came back 401 on the migrated one, and the check reported the service as never having answered. It had answered; it was up and serving on all ten ports. Liveness and credentials are now separate: any response proves the service is up, and credentials that stopped working are a warning that names this cause. Also recorded: a failed apply leaves the store in bootstrap mode, where only Bootstrap objects are accessible. A half-applied plan is not a partially configured server but an unusable one. Timing, which is the other reason to rehearse: the recovery-mode conversion of that 3.6 GB store took 2 seconds. A migration window is dominated by waiting and verification, not data volume. No production data in this commit; fixtures use example.net and the shapes involved.
This commit is contained in:
@@ -340,9 +340,25 @@ func Run(ctx context.Context, store *checkpoint.Store, rs *checkpoint.RunState,
|
||||
}, nil
|
||||
}
|
||||
client := newClient(opts)
|
||||
if err := client.WaitForPing(ctx, healthTimeout); err != nil {
|
||||
// Liveness first, and on its own. Any response - 401 included -
|
||||
// proves the service is up and routing.
|
||||
if err := client.WaitForResponse(ctx, healthTimeout); err != nil {
|
||||
return checkpoint.StepOutcome{}, fmt.Errorf("the migrated service started but never answered at %s within %s: %w", opts.AdminURL, healthTimeout, err)
|
||||
}
|
||||
// Credentials are a separate question, and failing them is not a
|
||||
// failed cutover. A config fallback-admin does not survive into
|
||||
// v0.16 - its config is just a store pointer, so the old
|
||||
// [authentication.fallback-admin] block is gone - so the
|
||||
// credentials that worked before the migration routinely stop
|
||||
// working after it, on an instance that is otherwise fine.
|
||||
if err := client.Ping(ctx); err != nil {
|
||||
return checkpoint.StepOutcome{
|
||||
Verdict: string(StatusWarn),
|
||||
Detail: fmt.Sprintf("migrated instance is up and answering at %s, but these admin credentials no longer work: %v. "+
|
||||
"If they were a config fallback-admin, that does not survive the migration - v0.16 keeps its config in the store. "+
|
||||
"Authenticate as an account that exists in the directory instead", opts.AdminURL, err),
|
||||
}, nil
|
||||
}
|
||||
return checkpoint.StepOutcome{Detail: fmt.Sprintf("migrated instance answered an authenticated JMAP session request at %s", opts.AdminURL)}, nil
|
||||
}); err != nil {
|
||||
return report, err
|
||||
|
||||
@@ -231,16 +231,13 @@ func TestRunResumesWithoutRedoingCompletedSteps(t *testing.T) {
|
||||
|
||||
func TestRunFailsWhenTheMigratedServiceNeverAnswers(t *testing.T) {
|
||||
store, rs, opts := migratedRun(t)
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusBadGateway)
|
||||
}))
|
||||
defer srv.Close()
|
||||
opts.AdminURL = srv.URL
|
||||
// Nothing listening at all: a closed port, not an error response.
|
||||
opts.AdminURL = "http://127.0.0.1:1"
|
||||
opts.HealthTimeout = 300 * time.Millisecond
|
||||
|
||||
report, err := Run(context.Background(), store, rs, opts)
|
||||
if err == nil {
|
||||
t.Fatal("Run: want failure when the started service never answers, got nil")
|
||||
t.Fatal("Run: want failure when nothing answers at all, got nil")
|
||||
}
|
||||
if !strings.Contains(err.Error(), "never answered") {
|
||||
t.Errorf("error %q should distinguish 'started but not answering' from 'failed to start'", err)
|
||||
@@ -250,6 +247,37 @@ func TestRunFailsWhenTheMigratedServiceNeverAnswers(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// A 401 proves the service is up and routing. Treating it as unhealthy
|
||||
// failed a cutover that had actually succeeded: the credentials supplied
|
||||
// for the pre-migration instance were a config fallback-admin, which does
|
||||
// not survive into v0.16.
|
||||
func TestRunWarnsRatherThanFailsWhenCredentialsStopWorking(t *testing.T) {
|
||||
store, rs, opts := migratedRun(t)
|
||||
srv := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusUnauthorized)
|
||||
}))
|
||||
defer srv.Close()
|
||||
opts.AdminURL = srv.URL
|
||||
opts.HealthTimeout = 2 * time.Second
|
||||
|
||||
report, err := Run(context.Background(), store, rs, opts)
|
||||
if err != nil {
|
||||
t.Fatalf("a service that is up but rejects these credentials is not a failed cutover: %v\n%s", err, report)
|
||||
}
|
||||
var warned bool
|
||||
for _, res := range report.Results {
|
||||
if res.Name == "wait-healthy" {
|
||||
warned = res.Status == StatusWarn
|
||||
if !strings.Contains(res.Detail, "fallback-admin") {
|
||||
t.Errorf("warning %q should name the likely cause", res.Detail)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !warned {
|
||||
t.Errorf("wait-healthy should warn, not fail or pass silently:\n%s", report)
|
||||
}
|
||||
}
|
||||
|
||||
// Rolling back a migration that completed successfully, because a counter
|
||||
// didn't get rebuilt, would be worse than a stale counter.
|
||||
func TestRunWarnsRatherThanFailsWhenQuotaRecalculationFails(t *testing.T) {
|
||||
|
||||
Reference in New Issue
Block a user