diff --git a/README.md b/README.md index e1230fbf9f..08413b4950 100644 --- a/README.md +++ b/README.md @@ -35,7 +35,7 @@ the `launchSettings.json` file of each instance. When started in setup mode, the ## Migrating from RavenDB to SQL Server or PostgreSQL -See [Migrating from RavenDB to SQL Server or PostgreSQL](docs/migration/ravendb-to-sql-migration-instructions.md). +- [Migrate from RavenDB to SQL Server or PostgreSQL](docs/migration/ravendb-to-sql-migration-instructions.md): what an operator can run today, and what is not built yet. ## Secrets diff --git a/docs/migration/migration-system-design-diagram.png b/docs/migration/migration-system-design-diagram.png index 15845976ae..85636355e7 100644 Binary files a/docs/migration/migration-system-design-diagram.png and b/docs/migration/migration-system-design-diagram.png differ diff --git a/docs/migration/ravendb-to-sql-migration-instructions.md b/docs/migration/ravendb-to-sql-migration-instructions.md index d229d03d15..ab4490e079 100644 --- a/docs/migration/ravendb-to-sql-migration-instructions.md +++ b/docs/migration/ravendb-to-sql-migration-instructions.md @@ -1,29 +1,39 @@ -# Migrating from RavenDB to SQL Server or PostgreSQL +# Migrate from RavenDB to SQL Server or PostgreSQL -This page covers reporting on the RavenDB source before you migrate. How the migration works is in the [migration overview](ravendb-to-sql-migration-overview.md) and the [system design diagram](migration-system-design-diagram.png). +This page tells you what you can do today. [The migration overview](ravendb-to-sql-migration-overview.md) tells you how the migration will work. [How the migration is put together](ravendb-to-sql-migration-system-design.md) tells you which class does what. > [!NOTE] -> The source report sends RavenDB only reads, but loading a database lets RavenDB's own expiration, its automatic deletion of documents past their retention date, run against it. If you are keeping the RavenDB database as a fallback, back it up before you run the report, as [Goals](ravendb-to-sql-migration-overview.md#goals) explains. +> You cannot copy data yet. This build has one migration command, the source report. If you set `ServiceControl/Migration/Enabled` to `true`, ServiceControl does not start. [What is not built yet](#what-is-not-built-yet-and-what-happens-if-you-turn-it-on) tells you why. +> +> The report opens RavenDB read-only, and the client refuses every write. But RavenDB deletes documents that are past their retention date, and this deletion starts when RavenDB loads a database. If you keep the RavenDB database as a fallback, make a backup of it before you start the report. ## Before you start -The source is a ServiceControl error instance on RavenDB. Keep its RavenDB settings in its configuration: the migration reads RavenDB through them, including after `PersistenceType` is switched to SQL Server or PostgreSQL. +The source is a ServiceControl error instance on RavenDB. Keep the RavenDB settings in the configuration of that instance. The migration reads RavenDB through these settings. This is also true after you change `ServiceControl/PersistenceType` to SQL Server or PostgreSQL. | Setting | Environment variable | What it is | | --- | --- | --- | -| `ServiceControl/RavenDB/ConnectionString` | `SERVICECONTROL_RAVENDB_CONNECTIONSTRING` | An external RavenDB server. Leave unset for an embedded database | -| `ServiceControl/DbPath` | `SERVICECONTROL_DBPATH` | The embedded database's data directory | -| `ServiceControl/RavenDB/DatabaseName` | `SERVICECONTROL_RAVENDB_DATABASENAME` | The primary database, `primary` by default | -| `LicensingComponent/RavenDB/ThroughputDatabaseName` | `LICENSINGCOMPONENT_RAVENDB_THROUGHPUTDATABASENAME` | The throughput database, `throughput` by default | -| `ServiceControl/RavenDB/ClientCertificatePath` or `ServiceControl/RavenDB/ClientCertificateBase64`, with `ServiceControl/RavenDB/ClientCertificatePassword` | `SERVICECONTROL_RAVENDB_CLIENTCERTIFICATEPATH` and so on | A secured external server's client certificate | -| `ServiceControl/ErrorRetentionPeriod` | `SERVICECONTROL_ERRORRETENTIONPERIOD` | Required. Don't change it during the move | +| `ServiceControl/RavenDB/ConnectionString` | `SERVICECONTROL_RAVENDB_CONNECTIONSTRING` | The address of an external RavenDB server. Leave it unset for an embedded database | +| `ServiceControl/DbPath` | `SERVICECONTROL_DBPATH` | The data directory of the embedded database | +| `ServiceControl/RavenDB/DatabaseName` | `SERVICECONTROL_RAVENDB_DATABASENAME` | The primary database. The default is `primary` | +| `LicensingComponent/RavenDB/ThroughputDatabaseName` | `LICENSINGCOMPONENT_RAVENDB_THROUGHPUTDATABASENAME` | The throughput database. The default is `throughput` | +| `ServiceControl/RavenDB/ClientCertificatePath` or `ServiceControl/RavenDB/ClientCertificateBase64`, with `ServiceControl/RavenDB/ClientCertificatePassword` | `SERVICECONTROL_RAVENDB_CLIENTCERTIFICATEPATH` and so on | The client certificate for a secured external server | +| `ServiceControl/ErrorRetentionPeriod` | `SERVICECONTROL_ERRORRETENTIONPERIOD` | Required. ServiceControl does not start without it | -## Report on the source +In an environment variable you can omit the `SERVICECONTROL_` prefix. `ServiceControl` is one of three namespaces that permit the short form. You cannot omit the `LICENSINGCOMPONENT_` prefix. -Run the instance's executable with `--migration-source-report`: +A SQL Server target must have Full-Text Search installed. Message search needs it, so `--setup` fails without it. A stock SQL Server container image does not have it. PostgreSQL needs nothing extra. + +Start `ServiceControl.exe` with the `--setup` argument on every instance that already uses SQL Server or PostgreSQL. Do this when you upgrade to this build, and do it even if you do not intend to migrate. `--setup` adds the `MigrationCheckpoints` table. Every SQL instance reads this table one time at startup, to make sure that the table is there. An instance that you upgrade without `--setup` does not have the table, and it does not start. + +Not built yet: when the copy is available, you must not change `ServiceControl/ErrorRetentionPeriod` during the move. The copier will use it to decide which rows are past the retention period. + +## Make the source report + +Start the instance executable with the `--migration-source-report` argument. ```powershell -# Installed on Windows, from the instance's installation folder +# Windows installation. Start this from the installation folder of the instance. .\ServiceControl.exe --migration-source-report ``` @@ -32,22 +42,45 @@ Run the instance's executable with `--migration-source-report`: docker run --rm --env-file servicecontrol.env ghcr.io/particular/servicecontrol: --migration-source-report ``` -From source, build `src/ServiceControl` and run the same command from its output folder, as in [How to run/debug locally](../../README.md#how-to-rundebug-locally). +To do this from source, build `src/ServiceControl`. Then start the same command from its output folder. [How to run/debug locally](../../README.md#how-to-rundebug-locally) gives the steps. + +The report prints: + +- The source persistence that it read. +- The RavenDB server version. +- Whether the source is embedded or external, and where it is. +- The name of each of the two databases, and the setting that each name came from. +- A row count for every collection in both databases. -The report prints the RavenDB server version, whether the source is embedded or external and where it is, both database names with the setting each came from, and a row count for every collection in both databases. +The report exits with 0 after it prints a report. It exits with 1 if it cannot print one. A script can thus tell the two apart without a read of the output. -- **External server:** run it while ServiceControl is running. It sends only reads, and the note above about expiration applies to a server you are keeping as a fallback. -- **Embedded database:** stop the ServiceControl service, run the report, then start the service again. The report starts its own RavenDB process against the data directory, which cannot happen while the instance holds it. -- **Container with an embedded database:** not supported, because the container image does not ship the RavenDB server. Point the instance at an external RavenDB server instead. +Where you start the report depends on the source: + +- External server: start the report while ServiceControl runs. The report sends only reads. The expiration warning at the top of this page applies to a fallback server. +- Embedded database: stop the ServiceControl service first. Start the report, then start the service again. The report starts its own RavenDB process against the data directory. This is not possible while the instance holds that directory. +- Container with an embedded database: not supported. The container image does not contain the RavenDB server. Point the instance at an external RavenDB server instead. ## If the report fails -The error says what to fix: +The error message tells you what to correct. + +- `has no database named ...`: the database name in the setting that the message quotes is wrong. Correct that setting. +- `refused its client certificate access ...`: give that certificate Read access to the database. Or supply a certificate that already has this access. +- `is secured but no client certificate is configured`: the connection string starts with `https://`. Set `ServiceControl/RavenDB/ClientCertificatePath` or `ServiceControl/RavenDB/ClientCertificateBase64`. +- `is valid from ... to ..., which does not include now`: the client certificate expired, or it is not yet valid. Supply a current certificate. The report does this check before the first request. Without the check, an expired certificate gives the same error as a refused connection. +- `ServiceControl expects RavenDB Server version ... or higher`: the external server is older than the RavenDB client in this build. Upgrade the server. The report does not do this check on an embedded source, because that server is installed with the client. +- `could not start a server for the embedded database ...`: the remainder of that line gives the reason from the server. The usual reason is a ServiceControl instance that still holds the data directory. Stop that instance, then start the report again. The other two reasons are a port already in use, and a missing RavenDB server. +- `did not finish loading within 5 minutes`: another process holds the embedded data directory, or the directory is corrupt. The wait is fixed at 5 minutes and no setting changes it. + +## What is not built yet, and what happens if you turn it on + +This build does not have the copy, the dry run, the status command or the verify command. [Migration workflow](ravendb-to-sql-migration-overview.md#migration-workflow) gives the planned steps. + +If you set `ServiceControl/Migration/Enabled` to `true`, the migration startup checks run before ServiceControl opens. One check always fails in this build. The host does not start, nothing is copied, and nothing is opened on the target. Leave the setting at its default of `false`. -- **"has no database named ..."**: the database name setting it quotes is wrong. -- **"refused its client certificate access ..."**: grant that certificate Read access to the database, or supply a certificate that has it. -- **"could not start a server for the embedded database ..."**: a ServiceControl instance is still running against that data directory and holds it. Stop the instance, run the report, then start it again. +Which check fails depends on `ServiceControl/PersistenceType`: -## Not available yet +- `RavenDB`: the first check refuses the pair, with the message `Migrating from 'RavenDB' to 'RavenDB' is not supported`. +- SQL Server or PostgreSQL: the check that asks whether this build can copy every required category fails. This build can copy 2 of the 12 required categories, and the message names the 10 that it cannot copy. -Copying the data (`MigrationMode`), the dry run, and the status and verify commands are planned but not built. The planned steps are in [Migration workflow](ravendb-to-sql-migration-overview.md#migration-workflow). +Five more settings sit beside `ServiceControl/Migration/Enabled`: `ServiceControl/Migration/OptionalCategories`, `ServiceControl/Migration/ThrottlePauseMilliseconds`, `ServiceControl/Migration/HaltThresholdPercent`, `ServiceControl/Migration/HaltThresholdMinimum` and `ServiceControl/Migration/AllowIncompleteExit`. The code reads the first four only after the check that always fails, so they do nothing in this build. The fifth does one thing today. An `--error-ingestion-only` worker refuses to start while a copy into its database is unfinished, because ingesting into a part-copied database means the copy can no longer be abandoned without loss. Setting `ServiceControl/Migration/AllowIncompleteExit` to `true` lets that worker start anyway. Its main purpose, the exit gate when `ServiceControl/Migration/Enabled` is turned off, is not built yet. diff --git a/docs/migration/ravendb-to-sql-migration-overview.md b/docs/migration/ravendb-to-sql-migration-overview.md index b1277bd1e8..bb5e0c7f32 100644 --- a/docs/migration/ravendb-to-sql-migration-overview.md +++ b/docs/migration/ravendb-to-sql-migration-overview.md @@ -1,5 +1,8 @@ # Moving data from RavenDB to SQL +> [!IMPORTANT] +> **This page describes the designed behaviour, not what you can run today.** One migration command is built, `--migration-source-report`, and two of the eighteen categories can be copied. Sections marked **Planned** describe code that has no call path yet. [The instructions](ravendb-to-sql-migration-instructions.md) tell an operator what works today, and [what is built but not yet on a call path](ravendb-to-sql-migration-system-design.md#what-is-built-but-not-yet-on-a-call-path) is the authoritative list for a developer. + ## Purpose The migration moves an error instance's data from RavenDB to SQL Server or PostgreSQL, so a customer can switch persisters and keep their data. @@ -8,7 +11,7 @@ This covers the error instance only. The audit instance has no SQL persister, so ## Strategy -- Switch over first, and copy only what has to be copied. Retention does most of the work: error retention is between 5 and 45 days and event retention defaults to 14 days, so most of the source ages out on its own within weeks. That is why archived and resolved messages are optional rather than required. Retention would have deleted them anyway. +- Switch over first, and copy only what has to be copied. Retention does most of the work. Error retention is between 5 and 45 days, and event retention defaults to 14 days, so most of the source ages out on its own within weeks. That is why archived and resolved messages are optional rather than required. Retention would have deleted them anyway. An operator can raise event retention as far as 200 days, which is longer than error retention can ever be, so check the two settings rather than assuming the event log is the shorter one. - The required set is what the instance needs the moment it opens. Most of it is small, but two parts grow: unresolved failures, and the last 7 days of event log, which on a busy instance holds a row for every failure, retry and submission. The strategy assumes you keep unresolved failures low by resolving and archiving. **A neglected instance breaks that assumption**: unresolved failures can legitimately be months old, and a large backlog of them makes the closed window long rather than short. The dry run is what tells you which case you are in. - Anything not selected simply ages out of RavenDB, and the customer deletes the old database when they are ready. @@ -16,15 +19,15 @@ This covers the error instance only. The audit instance has no SQL persister, so - **Minimal downtime**. Only the required data copies with ServiceControl closed. Optional data copies in the background while it serves traffic. - **All three RavenDB sources are supported**. Embedded, a container, or RavenDB Cloud, on one code path rather than three. -- **No writes through the client**. The copier never changes the source, but RavenDB's own expiration does: the primary database has it configured, and the sweep keeps deleting failed messages and event log items throughout the migration and for as long afterwards as the instance is left running. The old database is a fallback that degrades from the moment you start. -- **Abandonable up to a known point, and only up to that point**. While ServiceControl is closed the copy can be thrown away at no cost, because nothing but the copier has written to SQL and the migration has written nothing to RavenDB: see [the one point you can go back](#the-one-point-you-can-go-back). Once the host opens there is no way back at all. +- **No writes through the client**. The copier never changes the source, but RavenDB's own expiration does. The primary database already has expiration configured, and the sweep keeps deleting failed messages and event log items throughout the migration, and for as long afterwards as the instance is left running. So the old database is a fallback that degrades from the moment you start, and a customer who wants a clean fallback has to back it up first. +- **Abandonable up to a known point, and only up to that point**. While ServiceControl is closed the copy can be thrown away at no cost, because nothing but the copier has written to SQL and the migration has written nothing to RavenDB. See [the free abort, and the moment it closes](#the-free-abort-and-the-moment-it-closes). Once the host opens there is no way back at all. - **No duplicates and no gaps**. Rows and the resume cursor, a marker of the last row copied, commit in one transaction, so a crash needs no reconciliation. - **Every identifier anything depends on is carried across**. The event log, historic retry operations and pending integration events are renumbered, because nothing references their keys. - **Refuse rather than half-migrate**. Every check runs before the first row moves, and a failure is a host that will not start. -- **No silent loss**. A migration cannot end with a selected category still in progress or halted, only with each one finished or explicitly abandoned. Abandoning is a deliberate choice, and an abandoned category lets the host open. Skipped rows are counted and reported. The one exception is rows RavenDB's own expiration deletes while the copy runs: they are never read, so they are not skips, and the recount is what tells them apart from a real loss. See [what does not come across](#what-does-not-come-across). -- **Bounded impact on a live instance**. The background copy waits a fixed pause between batches, and reads are streamed, so memory does not track the size of the database. +- **No silent loss**. A migration cannot end with a selected category still in progress or halted. Each one must be finished or explicitly abandoned. Abandoning is a deliberate choice, and an abandoned category lets the host open. Skipped rows are counted and reported. The one exception is rows RavenDB's own expiration deletes while the copy runs: they are never read, so they are not skips, and a planned recount will tell them apart from a real loss. See [what does not come across](#what-does-not-come-across). +- **Bounded impact on a live instance**. The copy is streamed, so memory does not track the size of the database, and a configurable pause sits between the batches of an optional category. - **Known before it starts, visible while it runs**. A dry run reports what will move and how long ServiceControl is closed, and every category transition is reported as it happens. -- **Use existing functionality where possible**. Progress goes through custom checks and the activity feed, so no new client or screen is needed. +- **Use existing functionality where possible**. Progress is planned to go through custom checks and the activity feed, so no new client or screen is needed. ## Non-goals @@ -32,7 +35,11 @@ This covers the error instance only. The audit instance has no SQL persister, so - **Reversible once ServiceControl opens.** Nothing copies SQL rows back to RavenDB, so once the host has served traffic there is no rollback of any kind. - **Steerable while running.** No pause or resume, and no abort command. Going back during the closed window means stopping and reconfiguring, and changing anything else means editing configuration and restarting. - **A general-purpose migration tool.** The source is always RavenDB and the target is always a ServiceControl EF Core persister, both at versions this build can read. -- **Custom migration UI via ServicePulse.** Custom checks and the event log report progress, and migration configuration and control are not available in the UI. +- **Custom migration UI via ServicePulse.** Custom checks and the event log will report progress, and migration configuration and control are not available in the UI. + +## What works today + +Two categories have both a reader and a writer, `KnownEndpoints` and `EndpointSettings`. The other sixteen throw on both sides, and `RavenMigrationSource.ReadBody` throws for every category, so no message body has ever been copied. `EveryRequiredCategoryCanBeCopiedCheck` therefore refuses every real migration before a row moves. One command is built, `--migration-source-report`, and it prints the source facts plus a document count per RavenDB collection. Everything else on this page is design. ## Supported migration scenarios @@ -55,48 +62,50 @@ The copier runs inside the ServiceControl host, so every row and every message b **Infrastructure requirements:** - The ServiceControl host needs network access to the RavenDB source, the SQL target and the body store simultaneously -- Both RavenDB databases, primary and throughput, on one server or cluster -- A SQL Server target must have Full-Text Search installed. `--setup` checks `SERVERPROPERTY('IsFullTextInstalled')` and fails if it is absent, because message search is not optional. A stock SQL Server container image does not include it. PostgreSQL needs nothing extra, since its index is a GIN over `to_tsvector` -- A managed target's transient failures are survivable: retry on failure is on by default and there is no setting to turn it off +- Both RavenDB databases, primary and throughput, on one server or cluster. The migration opens both through a single `IDocumentStore`, and so does the licensing component (`LicensingDataStore.cs:36`) +- A SQL Server target must have Full-Text Search installed. The `AddFullTextSearch` EF Core migration checks `SERVERPROPERTY('IsFullTextInstalled')` and fails if it is absent, because message search is not optional. An EF Core migration runs once, so this is checked the first time `--setup` brings a database up to that migration, not on every `--setup`. A stock SQL Server container image does not include Full-Text Search. PostgreSQL needs nothing extra, since its index is a GIN over `to_tsvector` +- A managed target's transient failures are already survivable: retry on failure is on by default and no settings reader binds the flag that would turn it off **Not supported:** - Any server-to-server copy: no backup and restore, no RavenDB ETL or replication into SQL, no external data pipeline - A host that can reach only one of the two databases at a time, so no staged move by way of an offline copy -- **An embedded RavenDB source when ServiceControl runs in a container.** Reading an embedded database means starting a RavenDB server process, and the container image does not carry one: `ServiceControl.Persistence.RavenDB.csproj:36` excludes the `RavenDBServer` directory from the artifact, and the copy that would restore it at `:44` is conditional on `CI` not being set, which the Dockerfile sets. A containerised instance migrating away from embedded RavenDB has to point at an external RavenDB server rather than at a data directory. Windows installations are unaffected: the installer unzips the server unconditionally +- **An embedded RavenDB source when ServiceControl runs in a container.** Reading an embedded database means starting a RavenDB server process, and the container image does not carry one: `ServiceControl.Persistence.RavenDB.csproj:36` excludes the `RavenDBServer` directory from the artifact, and the copy that would restore it at `:44` is skipped when `CI` or `WindowsSelfContained` is set. The Dockerfile sets `CI`. A containerised instance migrating away from embedded RavenDB has to point at an external RavenDB server rather than at a data directory. Windows installations are unaffected: the installer unzips the server unconditionally - Primary and throughput RavenDB databases in different locations - Anything but RavenDB as the source, or anything but a ServiceControl EF Core persister as the target ## Migration workflow +**Planned.** Of the steps below, only 1, 3 and 5 can be carried out today, and step 5 ends in a refusal. Steps 4, 8 and 9 name commands and background work that do not exist. + 1. Upgrade ServiceControl as normal, still on RavenDB. -2. Set four things in configuration: the new `PersistenceType`, its connection string, `MigrationMode=true`, and whether you want the one [optional](#optional) category, archived and resolved messages, copied. +2. Set four things in configuration: the new `ServiceControl/PersistenceType`, its connection string, `ServiceControl/Migration/Enabled=true`, and whether you want the one [optional](#optional) category, archived and resolved messages, copied. 3. Run `--setup` to create the SQL schema. It fails against a SQL Server instance without Full-Text Search installed. 4. Run the [dry run](#dry-run). It reports what it resolved as a source, what each category holds, and an estimate of how long ServiceControl will be closed. Read [what the dry run reports](#dry-run) before booking an outage around its estimate. -5. Start ServiceControl (`MigrationMode=true`). +5. Start ServiceControl with `ServiceControl/Migration/Enabled=true`. 6. Every check runs before a single row moves. If one fails the host does not start and names which, having copied nothing, so a wrong database name or unconfigured body storage costs a restart rather than a half-finished migration. -7. The copying of [required data](#required) starts, with ServiceControl still closed: the copy runs inside that same start, before the API begins listening and before any background service runs. How long it takes depends on how many unresolved failures you have and how busy the last 7 days were, and the [dry run](#dry-run) gives you an estimate. If a required category halts, the host stays closed until you fix the cause and restart, or abandon that category. +7. The copying of [required data](#required) starts, with ServiceControl still closed. The copy runs inside that same start, before the API begins listening and before any background service runs. How long it takes depends on how many unresolved failures you have and how busy the last 7 days were, and the [dry run](#dry-run) gives you an estimate. If a required category halts, the host stays closed until you fix the cause and restart, or abandon that category. 8. ServiceControl opens by itself the moment the required copy finishes, with no second restart to perform, and whatever [optional data](#optional) you asked for is copied in the background while the instance runs normally. You can watch it from ServicePulse custom checks and events, but not steer it. -9. You run the verification pass once the background job has completed, which reports row counts on both sides category by category, accounting for deliberate skips so a difference is explained rather than reported as a fault, then set `MigrationMode=false` and restart. Counts can differ in both directions without anything being wrong. SQL can hold more rows, because RavenDB keeps expiring rows the copier already took. SQL can also hold fewer, because once ServiceControl opens it sends pending integration events, removes group comments whose group has no failed messages left, and removes failed error imports once they are imported again. +9. You run the verification pass once the background copy has completed. It reports row counts on both sides category by category, accounting for deliberate skips so a difference is explained rather than reported as a fault. You then set `ServiceControl/Migration/Enabled=false` and restart. Counts can differ in both directions without anything being wrong. SQL can hold more rows, because RavenDB keeps expiring rows the copier already took. SQL can also hold fewer, because once ServiceControl opens it sends pending integration events, removes group comments whose group has no failed messages left, and removes failed error imports once they are imported again. 10. RavenDB data can be removed. -- If `MigrationMode=false` is set while a selected category is still incomplete, the startup is gated: it refuses and names exactly what is outstanding, or, where the [free abort](#the-one-point-you-can-go-back) is still open, starts with a warning that says so. See [turning migration mode off is a gated startup too](#turning-migration-mode-off-is-a-gated-startup-too). -- A category that ended *complete with errors* counts as complete and does not block, though its skipped count is printed so the loss is stated rather than silent. -- An explicit override exists for a customer who has changed their mind and accepts leaving data behind. It marks the outstanding categories as abandoned, which is a deliberate end state rather than a failure, so the progress check settles and the guard stays armed for any later migration. +- If `ServiceControl/Migration/Enabled=false` is set while a selected category is still incomplete, the startup is gated: it refuses and names exactly what is outstanding, or, where the [free abort](#the-free-abort-and-the-moment-it-closes) is still open, starts with a warning that says so. See [turning migration mode off is a gated startup too](#turning-migration-mode-off-is-a-gated-startup-too). +- A category that ended `CompleteWithErrors` counts as complete and does not block, though its skipped count is printed so the loss is stated rather than silent. +- `ServiceControl/Migration/AllowIncompleteExit` exists for a customer who has changed their mind and accepts leaving data behind. It marks the outstanding categories as abandoned, which is a deliberate end state rather than a failure, so the progress check settles and the guard stays armed for any later migration. - While a selected category is still unfinished, ServiceControl pauses its own clean-up: the retention sweep, the purge API, the heartbeat settings sync and throughput collection. Your SQL database grows until the copy finishes, and these restart on their own once it does. -- **Steps 5 to 7 are the abort window**, which is not the override above: see [the one point you can go back](#the-one-point-you-can-go-back). -- A category that stops because too many rows failed is *halted*, and it stays that way until someone acts: fix the cause and restart to carry on from where it stopped, or abandon it deliberately if you accept the loss. See [a halt stops one category, and clearing it is a restart](#a-halt-stops-one-category-and-clearing-it-is-a-restart). +- **Steps 5 to 7 are the free abort window**, which is not the same thing as `ServiceControl/Migration/AllowIncompleteExit`. See [the free abort, and the moment it closes](#the-free-abort-and-the-moment-it-closes). +- A category that stops because too many rows failed is `Halted`, and it stays that way until someone acts. Fix the cause and restart to carry on from where it stopped, or abandon it deliberately if you accept the loss. See [a halt stops one category, and clearing it is a restart](#a-halt-stops-one-category-and-clearing-it-is-a-restart). ## Architecture ```mermaid flowchart TB cfg["Configuration + restart
the only way to change anything"] - checks["Custom checks + activity feed
progress, with no new client needed"] + checks["Custom checks + activity feed
planned: progress is log output today"] - subgraph host["One ServiceControl host process, started with MigrationMode = true"] + subgraph host["One ServiceControl host process, started with ServiceControl/Migration/Enabled = true"] direction LR - raven["RavenDB persister
own AssemblyLoadContext
read-only lifecycle"] + raven["RavenDB persister
own AssemblyLoadContext
read-only source lifecycle"] engine["MigrationEngine
categories, throttle,
dry run, verification"] target["EF Core persister
SQL Server or PostgreSQL
own AssemblyLoadContext"] raven -->|"IMigrationSource"| engine @@ -115,68 +124,85 @@ flowchart TB ``` - **The engine and the host know no store.** They deal in categories, cursors and counts. The source maps a category to what it reads and describes itself as labelled facts, the target maps a category to where it writes and how to count it, and each contributes its own startup checks. RavenDB to SQL is the only supported pair, and the engine does not depend on it. -- **The source reads the instance's own RavenDB settings**, so an existing customer sets nothing new. Leave them in place when switching `PersistenceType`. -- **Both persisters load into the same process**, each into its own `AssemblyLoadContext`. +- **The source reads the instance's own RavenDB settings**, so an existing customer sets nothing new. Leave them in place when switching `ServiceControl/PersistenceType`. +- **Both persisters load into the same process**, each into its own `AssemblyLoadContext`, and both are live at once during the copy. - **The engine references neither assembly.** It knows only `IMigrationSource` and `IMigrationTarget`, and treats the resume cursor as an opaque value it passes from one to the other, so it can be tested against fakes on either side. ## Startup sequence ```mermaid flowchart TB - A["ServiceControl starts on SQL"] --> M{"MigrationMode?"} + A["ServiceControl starts on SQL"] --> T{"Can the checkpoint
table be read?"} + T -->|"No: the schema predates this feature"| U["Host does not start.
Says to run --setup first."] + T -->|"Yes"| M{"Migration/Enabled?"} M -->|"On"| B["Open the SQL target
and the RavenDB source, read only"] B --> D{"All checks pass?"} D -->|"No"| E["Host does not start.
Names the failed check.
Nothing is copied."] D -->|"Yes"| F["Copy the required categories.
The API is not listening."] - F -->|"A required category halts"| T["Host stays closed.
Fix the cause and restart,
or abandon the category."] + F -->|"A required category halts"| V["Host stays closed.
Fix the cause and restart,
or abandon the category."] F -->|"Required categories settled"| G["ServiceControl opens.
New failed messages go to SQL."] - G --> H["Copy the selected optional categories
in the background"] + G --> H["Copy the optional category, if selected,
in the background, a pause between batches"] + H --> I["Verify row counts on both sides,
category by category,
then set Migration/Enabled = false and restart,
which comes back through this same gate"] M -->|"Off"| N{"Any category
outstanding?"} N -->|"No"| L["ServiceControl opens.
RavenDB is not opened."] - N -->|"Yes"| O{"Override set?"} + N -->|"Yes"| O{"AllowIncompleteExit set?"} O -->|"Yes"| P["Mark each outstanding category abandoned,
log what it leaves behind,
and open."] O -->|"No"| Q{"Has this instance
ever opened on SQL?"} Q -->|"No"| R["Open with a warning: point PersistenceType
back at RavenDB, or carry on
and lose the way back."] Q -->|"Yes"| S["Host does not start.
Names every outstanding category,
its counts, and every route out."] ``` -**Checked before a single row moves:** +The `Off` branch of that diagram, from node `N` down, is **planned**. Today `ServiceControl/Migration/Enabled=false` means only that the copy is not registered on the main host. The one start that already inspects the checkpoint rows is an `--error-ingestion-only` worker's: it refuses while any category is unfinished, unless `ServiceControl/Migration/AllowIncompleteExit` is set. + +**Checked before a single row moves,** in this order: +- The checkpoint table can be read at all. This one runs on every SQL start, migrating or not, and it refuses a database whose schema predates the feature +- The source and target are a supported pair. This is answered from settings alone, before anything connects +- This build can copy every required category. Today it cannot, so this is where a real migration stops +- The selected categories are valid +- `ServiceControl/RetryHistoryDepth` is greater than zero. At zero or less, the first completed retry after the migration deletes the entire copied retry history, and no row count would ever show it - The SQL schema is current - Message body storage is writable - Both RavenDB databases are reachable - The client certificate is valid, where the source is an external server - The source is at a version this build can read -- The selected categories are valid -- `RetryHistoryDepth` is greater than zero. At zero or less, the first completed retry after the migration deletes the entire copied retry history, and no row count would ever show it ### Turning migration mode off is a gated startup too +**Planned.** None of this section runs today. `ServiceControl/Migration/Enabled=false` currently means only that the copy is not registered: nothing reads `HasHostOpened`, the main host does not read `ServiceControl/Migration/AllowIncompleteExit`, and no code path sets a category to `Abandoned`. The only reader of that setting today is the `--error-ingestion-only` start gate described above. What is already wired is the marker the gate will need, which is stamped on any host start over a database that holds checkpoint rows. + *The right-hand branch above is the half a customer meets last and expects least, so it is worth reading before the migration starts rather than at the end of one.* -Every startup on a SQL Server or PostgreSQL instance looks at the checkpoint table before ServiceControl opens, whether `MigrationMode` is on or off. That is what stops a migration ending by accident, and it costs nothing on an instance with nothing outstanding: one that has never migrated holds no checkpoint rows, and one whose categories all settled has none left open. Both start normally. A RavenDB instance never reaches the gate. +Every startup on a SQL Server or PostgreSQL instance will look at the checkpoint table before ServiceControl opens, whether `ServiceControl/Migration/Enabled` is on or off. That is what stops a migration ending by accident, and it costs nothing on an instance with nothing outstanding: one that has never migrated holds no checkpoint rows, and one whose categories all settled has none left open. Both start normally. A RavenDB instance never reaches the gate. A SQL instance whose schema predates the feature is the one case that does not start, and the fix is to run `--setup`. + +With `ServiceControl/Migration/Enabled` off and at least one category still outstanding, one of three things happens, and each is said out loud at startup rather than discovered weeks later: -With `MigrationMode` off and at least one category still outstanding, one of three things happens, and each is said out loud at startup rather than discovered weeks later: +- **`ServiceControl/Migration/AllowIncompleteExit` is set.** Every outstanding category is recorded as abandoned, with its copied and skipped counts left as they are, and the host starts. Each one is logged saying what state it was in, how much it had copied, and that whatever it had not copied stays only in RavenDB. Abandoning is final: selecting that category in a later migration does not copy it again. +- **It is not set, and this instance has never opened on SQL.** This is the [free abort](#the-free-abort-and-the-moment-it-closes), so the host starts and warns rather than refusing. The warning names the two moves: stop now and point `ServiceControl/PersistenceType` back at RavenDB, which discards the partial copy and costs nothing else, or carry on, which opens ServiceControl on a partly copied database and ends the free abort. It deliberately does not mention `ServiceControl/Migration/AllowIncompleteExit`, because at that moment nothing is lost yet. +- **It is not set, and this instance has already opened on SQL.** The host does not start. The error names every outstanding category, its state, its copied and skipped counts and its last error. It then names the three routes out. Restart with `ServiceControl/Migration/Enabled=true` to let the copy finish, or to resume a halted category once its cause is fixed. Or set `ServiceControl/Migration/AllowIncompleteExit` to abandon what is outstanding and start without it. Or, if RavenDB is already gone, abandon, because that is the only exit left. -- **The override is set.** Every outstanding category is recorded as abandoned, with its copied and skipped counts left as they are, and the host starts. Each one is logged saying what state it was in, how much it had copied, and that whatever it had not copied stays only in RavenDB. Abandoning is final: selecting that category in a later migration does not copy it again. -- **The override is not set, and this instance has never opened on SQL.** This is the [free abort](#the-one-point-you-can-go-back), so the host starts and warns rather than refusing. The warning names the two moves: stop now and point `PersistenceType` back at RavenDB, which discards the partial copy and costs nothing else, or carry on, which opens ServiceControl on a partly copied database and ends the free abort. It deliberately does not mention the override, because at that moment nothing is lost yet. -- **The override is not set, and this instance has already opened on SQL.** The host does not start. The error names every outstanding category, its state, its copied and skipped counts and its last error, and then the three routes out: restart with `MigrationMode=true` to let the copy finish or to resume a halted category once its cause is fixed, set the override to abandon what is outstanding and start without it, or, if RavenDB is already gone, abandon, because that is the only exit left. +**Which is why the source stays until verification passes.** A customer who decommissions RavenDB while a category is outstanding has both doors shut: `ServiceControl/Migration/Enabled=true` cannot start, because it opens the source before it copies anything, and `ServiceControl/Migration/Enabled=false` refuses. Abandoning is then the only way to start the instance, and it is a real loss whose size is the counts in that message. -**Which is why the source stays until verification passes.** A customer who decommissions RavenDB while a category is outstanding has both doors shut: `MigrationMode=true` cannot start, because it opens the source before it copies anything, and `MigrationMode=false` refuses. Abandoning is then the only way to start the instance, and it is a real loss whose size is the counts in that message. +## Which class does what, and where it is called from + +Everything above is what the migration does. Which type does it, who calls it, and what is wired but not yet on a call path, is in [how the migration is put together](ravendb-to-sql-migration-system-design.md). That page is for someone changing the migration code rather than running a migration. ## Data to be migrated (Categories) +There are eighteen categories, seventeen required and one optional. Today only `KnownEndpoints` and `EndpointSettings` can be copied. + ### Required - Unresolved **and retry-issued** failed messages, with their bodies. Attempt history collapses to the newest attempt, because the SQL model has no attempts table. Retry-issued messages are required for the same reason unresolved ones are: issuing a retry deletes the expiry, so they never age out. Leaving one behind means the retry confirmation arrives with no row to mark resolved, and the message stays missing from the customer's list while the retry actually succeeded - Message redirects -- Endpoint settings +- Endpoint settings, which are copied after known endpoints - Known endpoints, including the monitored flag. One category, because the flag is a property of the endpoint row and cannot be copied without it - Notification settings - The licence trial end date -- Throughput history +- Licensing endpoint records, the per-endpoint rows in the throughput database +- Throughput history, which is copied after the licensing endpoint records, because each day's throughput row hangs off one of them - Retry operations, unacknowledged and historic. One category, because RavenDB holds both lists in a single document - Licensing report masks - The uploaded licensed endpoint details file, which nothing recomputes: skipping it means the customer re-downloads it from the licence portal and uploads it again @@ -189,7 +215,7 @@ With `MigrationMode` off and at least one category still outstanding, one of thr ### Optional -- Archived and resolved failed messages: the biggest category by far, and most of the copying time +- Archived and resolved failed messages, with their bodies: the biggest category by far, and most of the copying time ### Not migrated @@ -203,59 +229,64 @@ With `MigrationMode` off and at least one category still outstanding, one of thr ## What does not come across +**Planned.** Sixteen of the eighteen categories have no reader and no writer, so most of the rules below describe a design rather than running code. Where a rule is already enforced, it is marked. + **Whole categories are never copied.** Which ones, and why nothing needs them, is the [not migrated](#not-migrated) list above. Anything in an optional category you did not select is also never copied, and nothing later goes back for it. Neither is an event log item raised before the start of the 7-day window, which falls outside the [required](#required) event log window rather than being skipped. -**Rows skipped one at a time, and counted.** Each of these shows up in the skipped count for its category, broken out by reason, so you can see how much went and why: +**Rows skipped one at a time, and counted.** Each of these shows up in the skipped count for its category, broken out by reason, so you can see how much went and why. Only three skip reasons exist in code today: a body that could not be read, a row missing a value SQL requires, and settings for an unknown endpoint. Each of the other rules below needs a new reason value before it can be written at all, because the checkpoint refuses a batch whose skips do not add up. - A failed message whose `UniqueMessageId` is not a GUID. The target column is a `uniqueidentifier` and the value is never regenerated, because it is simultaneously the primary key, the ServicePulse URL, the retry correlation key and the body lookup key. - A failed message with no processing attempts recorded against it. The SQL model keeps the newest attempt and derives the failure time, the failing endpoint and the exception from it, all of which are required columns, so a message with nothing to derive them from cannot be written at all rather than being written blank. -- A failed message whose body cannot be read after three attempts. **The whole message is skipped, not just its body**, because a message with no body is worse than no message. +- A failed message whose body cannot be read after three attempts. **The whole message is skipped, not just its body**, because a message with no body is worse than no message. Enforced today, by the engine. - A subscription whose message type or transport address exceeds 200 characters. The target key columns are capped at 200 characters, so it cannot be stored at all. -- An archived or resolved failed message, or an event log item, already past its retention period. SQL's retention clean-up would delete it on its first pass, so it is counted rather than copied only to be deleted. -- Endpoint settings for an endpoint ServiceControl does not know. ServiceControl removes those settings shortly after it starts. -- A row missing a value SQL requires, such as a known endpoint with no name or host, or a failed message with no failing endpoint address. An empty group comment is left behind the same way: RavenDB can store one, but SQL never does. +- An archived or resolved failed message, or an event log item, already past its retention period. SQL's retention clean-up would delete it on its first pass, so it is counted rather than copied only to be deleted. The `PastRetention` reason exists and nothing writes it yet. Note that it is not currently treated as benign, so when a writer does start using it, those skips will count toward a halt unless that changes too. +- Endpoint settings for an endpoint ServiceControl does not know. ServiceControl removes those settings shortly after it starts. Enforced today, by `EndpointSettingsWriter`. +- A row missing a value SQL requires, such as a known endpoint with no name or host, or a failed message with no failing endpoint address. An empty group comment is left behind the same way: RavenDB can store one, but SQL never does. Enforced today, by `KnownEndpointsWriter`. **Things that change shape, and are not counted as skips at all.** The dry run counts the two merges before anything moves. Attempt history and renumbering apply to every row of their kind, so there is nothing to count. They are also the ones to read twice: - **Processing attempt history collapses to the newest attempt.** The SQL model has no attempts table. This affects every failed message that failed more than once, whether it is unresolved, archived or resolved. A message that failed five times arrives showing one attempt, and the other four are gone. - **Subscriptions that differ only in message-type version merge onto one row**, because the target key carries the type name without the version. -- **Endpoint settings for two endpoint names that differ only in case merge onto one row on SQL Server**, because SQL Server's default collation compares names without case, so one of the two settings is kept. PostgreSQL keeps both, and so does a SQL Server database created with a case-sensitive collation. The dry run counts this one too, by asking SQL Server how the name column compares, though for unusual characters its count can differ from what the copy does. +- **Endpoint settings for two endpoint names that differ only in case merge onto one row on SQL Server**, because the collation of the name column decides the comparison and the default collation compares names without case, so one of the two settings is kept. It is the column's own collation that decides, not the database default, so a case-sensitive database whose name column was given a case-insensitive collation still merges. PostgreSQL keeps both. The dry run counts this one too, by asking SQL Server how the name column compares, though for unusual characters its count can differ from what the copy does. - **Event log items, historic retry operations and pending integration events are renumbered.** Their keys are database identities and nothing references them, so this is safe, but the old numbers do not survive. -**Rows RavenDB deletes while the copy is running are an absence, not a skip.** Expiration only deletes a document carrying `@expires`, and only two kinds ever get one: a resolved or archived failed message, and an event log item (`ExpirationManager.cs:34,41`). A failed message loses its expiry whenever it becomes unresolved or retry-issued again: when it fails again, when it is unarchived, or when a retry is issued. The exception is a message that failed again after being archived or resolved while the instance ran version 6.18 or earlier: it kept its old expiry, and upgrading does not remove it, so a few unresolved messages can still expire during the copy. Apart from those, only the archived and resolved messages category and the event log can shrink underneath the copier. Archived and resolved messages copy in the background, where the window is longest. The event log's 7 days copy while ServiceControl is closed, and at the default 14-day event retention even the oldest of them is a week from expiring on a source that stopped recently, so the sweep reaches the window only on a source left stopped for days before the move. A document the sweep removes before the stream reaches it is never read, so it is counted nowhere. The copier counts each category before it starts, and if the copy comes up short of that count it counts the source again: rows that no longer exist were removed by RavenDB and are an absence, while rows that still exist but were never read halt the category. The dry run's count is a snapshot rather than a promise, and the recount is what shows that RavenDB removed rows while the copy ran. +**Rows RavenDB deletes while the copy is running are an absence, not a skip.** Expiration only deletes a document carrying `@expires`, and only two kinds ever get one: a resolved or archived failed message, and an event log item (`ExpirationManager.cs:34,41`). A failed message loses its expiry whenever it becomes unresolved or retry-issued again: when it fails again, when it is unarchived, or when a retry is issued. The exception is a message that failed again after being archived or resolved while the instance ran version 6.18 or earlier: it kept its old expiry, and upgrading does not remove it, so a few unresolved messages can still expire during the copy. Apart from those, only the archived and resolved messages category and the event log can shrink underneath the copier. Archived and resolved messages copy in the background, where the window is longest. The event log's 7 days copy while ServiceControl is closed, and at the default 14-day event retention even the oldest of them is a week from expiring on a source that stopped recently, so the sweep reaches the window only on a source left stopped for days before the move. A document the sweep removes before the stream reaches it is never read, so it is counted nowhere. The copier counts each category before it starts. **Planned:** if the copy comes up short of that count it will count the source again: rows that no longer exist were removed by RavenDB and are an absence, while rows that still exist but were never read halt the category. Today any shortfall halts the category. The dry run's count is a snapshot rather than a promise, and the recount is what will show that RavenDB removed rows while the copy ran. -**A category can finish with a small amount of loss and still count as complete.** A few skipped rows in a large table leave the category in a *complete with errors* state, which blocks nothing. Its skipped count is printed and the ids of the skipped rows are written to the log, so while the RavenDB database still exists you can go and look at exactly what did not make it. +**A category can finish with a small amount of loss and still count as complete.** A few skipped rows in a large table leave the category in `CompleteWithErrors`, which blocks nothing. Its skipped count is printed and the ids of the skipped rows are written to the log, so while the RavenDB database still exists you can go and look at exactly what did not make it. -## The one point you can go back +## The free abort, and the moment it closes -While ServiceControl is closed and the required copy is running, nothing except the copier has written to SQL, and the migration has written nothing to RavenDB, which is still authoritative. RavenDB's own expiration still runs, though: unless you disabled it, it keeps deleting expired failed messages and event log items, as [Goals](#goals) describes. If you need your instance back, set `MigrationMode=false`, point `PersistenceType` back at RavenDB, and start. You lose the copy, not your data. To start again later, start from an empty SQL database: drop it, create it, and run `--setup` again. Do not reuse the old copy, because a second attempt skips every category the first one finished, so anything that reached RavenDB since is left behind, and integration events RavenDB has since sent would be sent again. +While ServiceControl is closed and the required copy is running, nothing except the copier has written to SQL, and the migration has written nothing to RavenDB, which is still authoritative. RavenDB's own expiration still runs, though: unless you disabled it, it keeps deleting expired failed messages and event log items, as [Goals](#goals) describes. If you need your instance back, set `ServiceControl/Migration/Enabled=false`, point `ServiceControl/PersistenceType` back at RavenDB, and start. You lose the copy, not your data. To start again later, start from an empty SQL database: drop it, create it, and run `--setup` again. Do not reuse the old copy, because a second attempt skips every category the first one finished, so anything that reached RavenDB since is left behind, and integration events RavenDB has since sent would be sent again. That window closes the moment ServiceControl opens. From then on new failed messages are ingesting into SQL, RavenDB is no longer current, and there is no rollback: nothing copies SQL rows back. The choice at that point is to finish the migration or to accept losing whatever has not been copied. ## Reading from RavenDB - A dedicated read-only RavenDB lifecycle opens the source: connect, check the version, stop. It never calls `DatabaseSetup.Execute`. -- Both source databases must be on the same server or cluster (`LicensingDataStore.cs:35`). -- The source has to be at a ServiceControl version this build can read. ServiceControl stamps a version marker into the database on upgrade, because the RavenDB server version says nothing about which ServiceControl version wrote the data. A source without a marker, or one from a newer major version, is refused by name rather than misread. -- Duration scales with distance to the source. The copier already holds the document from the stream, so each body costs **one** round trip rather than two, but it is one per message and they are not batched. Egress out of RavenDB Cloud is billed to the customer. See [batching and throttling](#batching-and-throttling). +- Both source databases must be on the same server or cluster. The migration opens each of them through one `IDocumentStore`. +- The source has to be at a ServiceControl version this build can read. A marker is stamped into both databases on every RavenDB startup, and the source check refuses four cases by name: no marker at all, a marker it cannot parse, a newer major version, and an older major version. The only other version check compares the RavenDB server version to the RavenDB client version, and runs only for an external source. +- **Planned.** Duration scales with distance to the source. The copier already holds the document from the stream, so each body is designed to cost **one** round trip rather than two, but it is one per message and they are not batched. Egress out of RavenDB Cloud is billed to the customer. No body read exists yet: `ReadBody` throws for every category, so the one-round-trip figure is a target rather than a measurement. See [batching and throttling](#batching-and-throttling). ## Writing to SQL +**Planned.** Only the known endpoints and endpoint settings writers exist, so every bullet below except the body-storage one describes a writer that has not been built. The target schema each bullet relies on is real, and was checked. + - A whole `FailedMessage` is written with its stored status intact. - `UniqueMessageId` keeps its value, but converts type: the source holds a string and the target column is a `uniqueidentifier`. It is the primary key, the ServicePulse URL, the retry correlation key and the body lookup key at once. - `StatusChangedAt` is reconstructed from `@expires` for resolved and archived messages, which is the only place RavenDB sets it. Unresolved and retry-issued messages normally have no `@expires` (see [the exception](#what-does-not-come-across) for messages from version 6.18 or earlier), so the copier uses the newest processing attempt's timestamp. The column is `NOT NULL`, so it cannot be left empty, but the value is harmless for those two: the retention sweep only considers resolved and archived rows, so an unresolved message never ages out whatever is written here. -- Message bodies go through `IBodyStoragePersistence`, which owns the compression threshold and the choice of filesystem, Azure Blob or S3. The copier applies the 102,400-byte inline threshold itself, because that threshold lives on the ingestion path rather than in `IBodyStoragePersistence`. -- Throughput rows are written directly rather than through the collector, and the write sets each day's count rather than adding to it. Throughput is a required category, so it copies while ServiceControl is closed, before any collector has written to SQL. Setting is what makes the category safe to resume after a crash, where adding would double-count. Copying the rows is also what stops the audit and broker collectors re-gathering the same days when the host opens, because `LastCollectedDate` is derived from the newest throughput row rather than stored (`LicensingDataStore.cs:45`). The checkpoint is what stops a second pass overwriting days the collectors have written since. +- Message bodies go through `IBodyStoragePersistence`, which owns the compression threshold and the choice of filesystem, Azure Blob or S3. The copier applies the inline threshold itself, because that threshold lives on the ingestion path rather than in `IBodyStoragePersistence`. It defaults to 102,400 bytes and is overridable by `ServiceControl/MaxBodySizeToStore`. +- Throughput rows are written directly rather than through the collector, and the write sets each day's count rather than adding to it. Throughput is a required category, so it copies while ServiceControl is closed, before any collector has written to SQL. Setting is what makes the category safe to resume after a crash, where adding would double-count, and the collector's own path does add (`LicensingDataStore.cs:211`). Copying the rows is also what stops the audit and broker collectors re-gathering the same days when the host opens, because `LastCollectedDate` is derived from the newest throughput row rather than stored (`LicensingDataStore.cs:45`). The checkpoint is what stops a second pass overwriting days the collectors have written since. - Identifiers narrow on the way across, and the dry run counts every kind. What narrows, merges or cannot be stored at all is in [what does not come across](#what-does-not-come-across). ## Batching and throttling - Batch size comes from the provider: SQL Server divides its own parameter budget by the column count, PostgreSQL uses a flat 50 rows. -- The throttle is a configurable pause between batches, defaulting to 100 ms. Raising it slows the background copy and eases the load on production. Turning `MigrationMode` off is not a remedy: while a selected category is unfinished the host refuses to start, unless you abandon that category. +- The throttle is a pause between batches, defaulting to 100 ms and set by `ServiceControl/Migration/ThrottlePauseMilliseconds`. It applies to the optional category only, and not to its first batch. A required category is never throttled, because it runs with ServiceControl closed and nothing is competing with it. +- **Raising** the pause is what relieves a copy competing with production, because the pause is how long the engine waits between batches. Turning `ServiceControl/Migration/Enabled` off is not a remedy: while a selected category is unfinished the host refuses to start, unless you abandon that category. Because both implemented categories are required, the pause does nothing at all today. ## Checkpointing and resume -A copy that runs for hours will be interrupted at some point: a restart, a dropped connection, a machine reboot. The checkpoint is what makes an interruption cost only the batch that was in flight. It is one row per category, kept on the target and created by `--setup` along with the rest of the schema, and it is written in the same database transaction as the rows it describes. Only the copier writes to it; the status and verify commands read it. +A copy that runs for hours will be interrupted at some point: a restart, a dropped connection, a machine reboot. The checkpoint is what makes an interruption cost only the batch that was in flight. It is one row per category, kept on the target and created by `--setup` along with the rest of the schema, and it is written in the same database transaction as the rows it describes. Only the copier writes to it. The status and verify commands will read it. **What one row holds:** the category it tracks, its state, the resume cursor, how many rows were copied, skipped and already present, a count per skip reason, how many rows the source held when the category started, when it started, when it last made progress, when it settled, the last error, and a version number used to spot a second writer. @@ -293,22 +324,25 @@ The thing to read twice is that the counts never travel back through the engine - **Progress never gets ahead of the data.** The rows and the cursor commit together, so a restart cannot skip past rows that were never written. - **A crash costs the batch in flight and nothing else.** The next run reads from the committed cursor. -- **Re-reading a batch cannot double-count it.** Unreadable bodies stay off the checkpoint until the write commits, so a batch that is read twice is counted once, and rows the earlier attempt did write come back as *already present* rather than as fresh copies. -- **Every skipped row has a reason, or the save is refused.** The checkpoint rejects a batch reporting more skips than it explains, because verification has to account for each one rather than report a healthy migration as broken. +- **Re-reading a batch cannot double-count it.** Unreadable bodies stay off the checkpoint until the write commits, so a batch that is read twice is counted once, and rows the earlier attempt did write come back as `AlreadyPresent` rather than as fresh copies. +- **Every skipped row has a reason, or the save is refused.** The checkpoint rejects a batch whose per-reason counts do not sum exactly to its skipped count, in either direction, because verification has to account for each one rather than report a healthy migration as broken. The engine adds two more guards of the same shape: it refuses a commit whose deltas disagree with the target's own counts, and one whose benign skips exceed its total skips. - **Each category resumes independently**, so a half-copied category picks up where it stopped while its neighbours are untouched. - **The halt counters are per run and deliberately not stored.** If the skips that tripped a halt stayed on the row, a restart with the cause fixed would re-trip it on its first batch. - **A second writer is caught rather than merged.** Each save carries the version it read, and a save against a row that has moved on is refused, so two hosts pointed at one target cannot quietly interleave their progress. -- **If a message is already in SQL the SQL row wins and the copier skips it**, which is what makes every category safe to run twice. +- **If a message is already in SQL the SQL row wins and the copier leaves it alone.** It is counted as already present rather than as a skip, which is what makes every category safe to run twice without moving it toward a halt. ## Error handling - Which rows are skipped, and why, is in [what does not come across](#what-does-not-come-across). What follows is the mechanics around those rules. - A body is read up to three times before the message is skipped whole, and the exhausted attempts count toward the halt threshold. - Deciding whether a row is past the target's retention cutoff needs two retention periods: the source's reverses `@expires` back into the status-change instant, and the target's current one decides whether that instant is past the cutoff. -- A bad row does not stop the copy. Its category finishes in a separate complete-with-errors state. -- The halt threshold is proportional, 5 percent by default, with an absolute floor, and skips halt a category only when both are exceeded. The one exception is a category smaller than the floor, which halts if it loses more than half its rows. Proportional alone halts a three-row category on one bad row; absolute alone halts a five-million-row table on its 101st failure at the default floor of 100. Together, a large category keeps going through losses under the percentage and finishes complete with errors, so ten thousand skipped rows out of five million do not halt it. -- Rows left behind because SQL would remove them anyway (past retention, settings for unknown endpoints) are counted and reported, but never halt a category. The target reports them apart from its real failures, so they land in the skipped count and the log without moving the category toward a halt. -- The percentage is measured against what the run has processed so far rather than against the category's total, so a run that starts badly looks worse than it is. The floor is what keeps that harmless, since fewer than 101 skipped rows never consults the percentage at all. More than that, bunched at the start, does halt a category whose overall rate would have been fine, and the cost is one restart: the skipped rows commit with the cursor, so the next run resumes past them with its counters back at zero. +- A bad row does not stop the copy. Its category finishes `CompleteWithErrors`. +- The halt threshold is proportional, 5 percent by default, with an absolute floor, and a category halts only when both are exceeded. Proportional alone halts a three-row category on one bad row. Absolute alone halts a five-million-row table on its 101st failure at the default floor of 100. Together, a large category keeps going through losses under the percentage and finishes `CompleteWithErrors`, so ten thousand skipped rows out of five million do not halt it. +- **A small category is not protected by the floor.** A separate rule halts any category that lost more than half its rows, whatever the floor says, because a category smaller than the floor would otherwise never reach it however much of it was lost. +- **A category also halts if it reaches the end of the source short.** If copied, skipped and already-present rows together come to less than the source count taken at the start, the category halts even though no threshold was crossed. **Planned:** before halting, the copier counts the source again, and halts only if the missing rows still exist in RavenDB, because rows RavenDB expired during the copy are an absence rather than a loss. +- Rows left behind because SQL would remove them anyway are counted and reported, but never halt a category. Today this exemption covers exactly one reason, settings for an unknown endpoint, and it withdraws itself: if the known endpoints copy skipped anything, an unknown endpoint can be this migration's own doing, so those skips start counting toward a halt again. +- The percentage is measured against what the run has processed so far rather than against the category's total, so a run that starts badly looks worse than it is. The floor is what keeps that harmless in a large category, since fewer than 101 skipped rows never consults the percentage at all. More than that, bunched at the start, does halt a category whose overall rate would have been fine, and the cost is one restart: the skipped rows commit with the cursor, so the next run resumes past them with its counters back at zero. +- The number compared against the floor is this run's fault skips, which is the skipped count less the benign ones, so it is not the same number the operator sees reported. - A source therefore must not read a category in an order that puts the rows most likely to be skipped at the front of it. - Verification therefore cannot treat any count difference as a fault. It accounts for every skip rule, or it reports every successful migration as broken. @@ -319,13 +353,13 @@ A halt is the copy refusing to keep going on one category because something is w ```mermaid stateDiagram-v2 [*] --> NotStarted: nothing has run yet - NotStarted --> InProgress: the host starts with MigrationMode = true + NotStarted --> InProgress: the host starts with Migration/Enabled = true NotStarted --> Blocked: the category it must follow has not settled Blocked --> InProgress: that category settles, then the next restart InProgress --> InProgress: the host was stopped mid-copy,
so the next start resumes from the cursor InProgress --> Complete: every row reached, none skipped InProgress --> CompleteWithErrors: every row reached, some skipped - InProgress --> Halted: too many rows skipped, most of a small category lost,
an error it did not expect, or a shortfall the recount confirms + InProgress --> Halted: too many rows skipped in this run,
most of the category lost,
the source count not reached,
or an unexpected error Halted --> InProgress: fix the cause, restart,
carry on from the cursor Halted --> Abandoned: accept the loss, deliberately InProgress --> Abandoned: accept the loss, deliberately @@ -334,30 +368,25 @@ stateDiagram-v2 Abandoned --> [*] ``` -**Four things halt a category.** +**Four things halt a category.** The skipped rows in this run pass both the percentage and the floor, which says the failures are systematic rather than incidental. Or more than half the category was lost, which catches a category too small to reach the floor. Or the category reached the end of the source with fewer rows accounted for than the source held when it started. Or the copy hits an error it did not expect, in which case the error type and the cursor it stopped at are recorded. A host being shut down is none of these: it leaves the category in progress, to be picked up from the cursor next time. Nor is a second host writing to the same checkpoint, which is refused so that the other host's progress stands. Separately from all four, a copy that commits nothing for 30 minutes is cancelled by a stall watchdog, which does not mark the category halted but does stop the copy and keep the host closed. -- **Too many skips.** The skipped rows in this run pass both the percentage and the floor, which says the failures are systematic rather than incidental. -- **Most of a small category lost.** A category with fewer rows than the floor loses more than half of them. -- **An error the copy did not expect.** The error type and the cursor it stopped at are recorded. -- **A shortfall that survives the recount.** The copy came up short of the starting count, and the missing rows still exist in RavenDB but were never read. +**A halt stops that category and nothing else.** The remaining categories still run, with one exception: a category that must follow the halted one goes to `Blocked` rather than running early, which is how group comments stay behind the unresolved failed messages their groups are built from. A blocked category is not a failure and needs no separate action, since clearing the halt clears the block on the next restart. -A host being shut down is none of these: it leaves the category in progress, to be picked up from the cursor next time. Nor is a second host writing to the same checkpoint, which is refused so that the other host's progress stands. - -**A halt stops that category and nothing else.** The remaining categories still run, with one exception: a category that must follow the halted one goes to blocked rather than running early, which is how group comments stay behind the unresolved failed messages their groups are built from. A blocked category is not a failure and needs no separate action, since clearing the halt clears the block on the next restart. - -**What it costs depends on which category halted.** A halted optional category means the instance keeps serving traffic and that one slice of history is missing until it is resumed. A halted required category means the host stays closed, so the outage carries on until the halt is cleared or the category is abandoned. That is deliberate: opening the host is the point of no return, and it should not happen with required data left behind by accident. +**What it costs depends on which category halted.** A halted optional category means the instance keeps serving traffic and that one slice of history is missing until it is resumed. A halted required category means the host stays closed, so the outage carries on until the halt is cleared or the category is abandoned. That is deliberate: opening the host is the point of no return, and it must not happen with required data left behind by accident. **Clearing it:** 1. Read the reason on the category, in the custom check or the status command. It names the count that tripped the threshold, or the error, and the cursor either way. 2. Fix the cause. It is usually outside the migration: the body store unreachable, a certificate expired, the source or the target down, or the disk full. -3. Restart the host with `MigrationMode=true`. The category picks up at its cursor, its run counters start again at zero, and the skips already recorded stay on the row so the totals still add up at the end. +3. Restart the host with `ServiceControl/Migration/Enabled=true`. The category picks up at its cursor, its run counters start again at zero, and the skips already recorded stay on the row so the totals still add up at the end. 4. Repeat only if it halts again. A restart that halts at the same point is telling you the cause is still there, and a restart that gets further has made real progress, because the rows it skipped are committed and will not be read again. **Or abandon it, on purpose.** Abandoning marks the category as deliberately given up rather than failed, which lets the host open and lets the migration end. It is the right answer when the data is not worth the outage, and the wrong one if it was picked by accident, because nothing goes back for an abandoned category afterwards. What it leaves behind is stated in the counts rather than guessed at. ## Dry run +**Planned.** There is no `--migration-dry-run`. Everything in this section describes a command that has not been built. + Runnable before anything starts, and again later against whatever is still outstanding. It never writes to RavenDB. What it resolves and reports: @@ -379,21 +408,23 @@ It counts, before anything moves, the rows the target says it would skip or merg - Subscriptions whose message type or transport address exceeds the 200-character key limit - Rows already past their retention period, and rows missing a value SQL requires -It reports no duration for the optional categories, and nothing about load on the source. +It reports no duration for the optional category, and nothing about load on the source. ### When you can run the read-only commands +**Planned, except for the source report.** `--migration-verify`, `--migration-dry-run` and `--migration-status` do not exist. `--migration-source-report` does, and it prints the source facts plus a document count per RavenDB collection. It does not run the startup checks, does not count rows per category, does not measure body volume and does not estimate a duration. + `--migration-source-report`, `--migration-verify` and `--migration-dry-run` all open the RavenDB source. **On an embedded source that means stopping the ServiceControl service first**, because a second RavenDB process cannot attach to a data directory the first one holds. Plan the dry run as part of the outage rather than as something you run the day before while the instance keeps serving traffic. On an external source, a container or RavenDB Cloud, all three run against a live instance with no interruption. -A containerised instance runs all three as a one-off `docker run` of the same image with the command's flag, against an external RavenDB server, as the [instructions](ravendb-to-sql-migration-instructions.md#report-on-the-source) show for the source report. It cannot use an embedded source, because the image does not ship the RavenDB server. +A containerised instance runs all three as a one-off `docker run` of the same image with the command's flag, against an external RavenDB server, as the [instructions](ravendb-to-sql-migration-instructions.md#make-the-source-report) show for the source report. It cannot use an embedded source, for the reason given under [supported migration scenarios](#supported-migration-scenarios). -`--migration-status` is the exception and is deliberately so: it reads only the checkpoint table in SQL and never opens the source, so it works on every source shape at any time, including during the background copy. It is the command to use for watching progress. +`--migration-status` is the exception and is deliberately so: it reads only the checkpoint table in SQL and never opens the source, so it will work on every source shape at any time, including during the background copy. It is the command to use for watching progress. ## Configuration and control -- You set `MigrationMode`, and next to it whether to copy the one optional category, archived and resolved messages. A status command and a custom check report back. +- You set `ServiceControl/Migration/Enabled`, and next to it whether to copy the one optional category, archived and resolved messages. A status command and a custom check will report back. **Planned:** neither exists yet, and progress today reaches an operator only as log output. - Categories are read fresh at every startup. Adding one copies it on the next restart, removing one deletes nothing. -- There is no HTTP API, no pause, no resume, no abort command, and no way to add a category to a running instance. All of those mean editing configuration and restarting. Going back during the closed window means stopping and reconfiguring, as [the one point you can go back](#the-one-point-you-can-go-back) describes. +- There is no HTTP API, no pause, no resume, no abort command, and no way to add a category to a running instance. All of those mean editing configuration and restarting. Going back during the closed window means stopping and reconfiguring, as [the free abort, and the moment it closes](#the-free-abort-and-the-moment-it-closes) describes. - The checkpoint table is a record of what happened, not a control channel. - Stopping a copy takes a restart, so it cannot be stopped in ten seconds. diff --git a/docs/migration/ravendb-to-sql-migration-system-design.md b/docs/migration/ravendb-to-sql-migration-system-design.md new file mode 100644 index 0000000000..f1b93bc51d --- /dev/null +++ b/docs/migration/ravendb-to-sql-migration-system-design.md @@ -0,0 +1,127 @@ +# How the migration is put together + +*Written for someone about to change the migration code, not for someone running a migration. [The overview](ravendb-to-sql-migration-overview.md) says what the migration does and what a customer sees. This page says which class does it, who calls it, and why it sits where it does, so anyone moving one piece can see what else moves with it. Read it before you reorder the startup checks or add a category, because in both of those the order is the design. The operator steps are in [the instructions](ravendb-to-sql-migration-instructions.md).* + +Setting keys on this page are written as the code writes them, relative to the `ServiceControl` settings root. An operator sets `ServiceControl/Migration/Enabled`, which the code calls `Migration/Enabled`. + +## Everything at once, before the detail + +![The migration's contracts, the engine between them, and the two persisters on either side. Purple bars mark the parts that are designed but have no call path in this build.](migration-system-design-diagram.png) + +## Four assemblies, and what each one is allowed to know + +| Assembly | What it holds | Why there | +| --- | --- | --- | +| `ServiceControl.Persistence` | `MigrationEngine` and the contracts it drives: `IMigrationSource`, `IMigrationTarget`, `IMigrationCheckpointStore`, `IMigrationStartupCheck`, `IMigrationTargetReadiness`, `IMigrationState`, `IMigrationSourceFactory`, plus `MigrationCategoryIds`, `MigrationCategoryRegistry`, `MigrationCheckpoint`, `MigrationBatch`, `MigrationWriteResult`, `MigrationSourceDescription`, `MigrationSkipReason`, `MigrationCategoryStateExtensions`, `CheckpointMigrationState`, `MigrationCheckpointConflictException`, `MigrationEngineOptions`, `MigrationSettings` and `HaltThreshold` | Both persisters and the host reference it, and it references neither persister. That is what lets the engine be driven entirely by fakes in `ServiceControl.UnitTests/Migration` | +| `ServiceControl.Persistence.RavenDB` | `RavenMigrationSource`, `RavenReadOnlySourceLifecycle`, `RavenDocumentStream`, one `IMigrationCategoryReader` per category, `SourceDataVersionIsReadableCheck`, `RavenDataVersion` | Everything that knows a document id prefix, a RavenDB session or an embedded server lives on the source side | +| `ServiceControl.Persistence.EFCore` | `EFCoreMigrationTarget`, `EFCoreMigrationTargetReadiness`, one `IMigrationCategoryWriter` per category, `IMigrationSqlDialect`, `MigrationInsert`, `PreparedBatch`, `EFMigrationCheckpointStore`, `MigrationCheckpointExtensions`, the target's three checks, and the two hosted services `CheckpointTableIsReadable` and `RecordHostOpenedOnTarget` | Everything that knows a table, a column width or a provider's parameter ceiling lives on the target side | +| `ServiceControl` (the host) | `RequiredCopyBeforeTheHostOpens`, `MigrationStartup` with its nested `ClosedWindowProgress`, `MigrationStartupCheckRunner`, `MigrationPairIsSupportedCheck`, `EveryRequiredCategoryCanBeCopiedCheck`, `SelectedCategoriesAreCoherentCheck`, `AllowIncompleteCategorySet`, `FinishedCopyBeforeAnIngestionNodeOpens`, `MigrationSourceReportCommand`, `PersistenceFactory.CreateMigrationSource` | The only place that names both persisters at once, because deciding whether this pair is supported is the one question neither side can answer alone | + +Two provider assemblies sit below the EF Core one and hold the only provider-specific migration code: `SqlServerMigrationSqlDialect` and `PostgreSqlMigrationSqlDialect`, each implementing `IMigrationSqlDialect`. They are where a parameter ceiling and a column collation are known. + +## The wiring is in place on every SQL instance, migrating or not + +Three registrations happen while the host is being built, long before anything decides whether a migration is running, because resolving them is how the copy finds them later: + +- **`BasePersistence.RegisterDataStores`**, reached through each provider's `AddPersistence`, registers `IMigrationCheckpointStore` as `EFMigrationCheckpointStore`, `IMigrationTargetReadiness` as `EFCoreMigrationTargetReadiness` and `IMigrationTarget` as `EFCoreMigrationTarget`. All three are singletons. Only the target is genuinely never constructed on an instance that does not migrate: the other two are resolved on every SQL start by the two hosted services registered beside them, `CheckpointTableIsReadable` and `RecordHostOpenedOnTarget`. The RavenDB persister registers none of the three, which is exactly what makes RavenDB a source and never a target. +- **`HostApplicationBuilderExtensions.AddServiceControl`** registers `IMigrationState` as a `CheckpointMigrationState` wrapping whatever `IMigrationCheckpointStore` the persister registered, or none at all on RavenDB. It is a `TryAddSingleton`, so a test can put its own in first. +- **`PersistenceFactory.Create`** copies `Settings.RetryHistoryDepth` onto `PersistenceSettings.RetryHistoryDepth`. Without that one line `RetryHistoryDepthIsSafeCheck` cannot see the value it exists to refuse, because the check runs inside the persister and the setting is read outside it. + +**`CheckpointTableIsReadable` is the first migration code to touch the database on any start, migrating or not.** It runs in `StartingAsync`, reads the checkpoint table, and turns any failure into a refusal naming `--setup`. `AddPersistence` is registered before the copy is, and `StartingAsync` runs in registration order, so this refusal comes ahead of every check listed below. It is not one of the target's contributed checks: it is a hosted service on every SQL start, and it exists so that a database whose schema predates this feature stops the host rather than failing later inside a copy. + +**`FinishedCopyBeforeAnIngestionNodeOpens` keeps an `--error-ingestion-only` worker out of a database a copy has not finished filling.** `ErrorIngestionOnlyCommand.BuildHost` registers it on every ingestion-only host. In `StartingAsync` it reads every checkpoint row and refuses if any category is unfinished, because ingesting writes to the target and stamps it as opened, which turns abandoning the copy from a clean rollback into a loss. `Migration/AllowIncompleteExit=true` skips the check, and the refusal names that setting. The worker never runs the copy itself: one host does, and a second copier would race it. + +## Starting the host with the migration on runs the required copy inside host start + +*The sequence below is what the code does when the copy can run. On a real build it stops at `EveryRequiredCategoryCanBeCopiedCheck`, because only two of the eighteen categories have both a reader and a writer. Read the whole section before you assume any step after that one has ever executed outside a test.* + +```mermaid +sequenceDiagram + participant Run as RunCommand + participant Copy as RequiredCopyBeforeTheHostOpens + participant Start as MigrationStartup + participant Checks as MigrationStartupCheckRunner + participant Factory as PersistenceFactory + participant Source as RavenMigrationSource + participant Target as EFCoreMigrationTarget + participant Engine as MigrationEngine + + Run->>Run: registers Copy as a hosted service, then hostBuilder.Build() + Run->>Run: app.RunAsync + Run->>Copy: StartingAsync, before any other hosted service starts + Note over Copy: CheckpointTableIsReadable has already read the
checkpoint table, earlier in the same start + Copy->>Start: RunRequiredCopy(app.Services, settings) + Start->>Checks: MigrationPairIsSupportedCheck + Start->>Factory: CreateMigrationSource(settings) + Factory-->>Start: a source object, nothing connected yet + Start->>Start: CopyableCategoryIds(source, target) + Start->>Checks: EveryRequiredCategoryCanBeCopiedCheck,
then SelectedCategoriesAreCoherentCheck,
then the target's three checks + Checks->>Target: Open + Checks->>Source: Open + Start->>Checks: source.ContributedChecks() + Checks->>Source: SourceDataVersionIsReadableCheck + Start->>Engine: RunCategories(the required categories this build can copy) + Engine-->>Start: one checkpoint per category + Start->>Start: ReportWhatTheCopyLeftBehind + Start->>Start: RefuseIfAnyCategoryDidNotComplete + Start->>Start: CheckpointMigrationState.Seed + Start-->>Copy: returns, or throws and the host never opens + Note over Run: RecordHostOpenedOnTarget, an EF Core hosted service,
stamps the target in StartedAsync +``` + +The order is not incidental, so each position is worth the sentence: + +- **`RequiredCopyBeforeTheHostOpens` is why the copy runs inside host start rather than before it.** `RunCommand` registers it only when `Migration/Enabled` is true, before `hostBuilder.Build()`, and `app.RunAsync` then drives its `StartingAsync`. A Windows service that has reported nothing for 30 seconds is killed by the Service Control Manager, and a copy that ran before `RunAsync` would be exactly that silence. +- **`MigrationPairIsSupportedCheck` runs first among the copy's own checks**, because it is the only one answerable from settings without touching a database. The source is always `PersistenceFactory.MigrationSourcePersistenceType`, so it resolves only `PersistenceType` through `PersistenceManifestLibrary` and compares it against `PersistenceFactory.SqlPersistenceNames`. An unsupported pair costs a message rather than a connection attempt. +- **`PersistenceFactory.CreateMigrationSource` builds the source without opening it.** It resolves the source persistence by name and casts its configuration to `IMigrationSourceFactory`, which only `RavenPersistenceConfiguration` implements, and refuses by name when the cast fails. What comes back is a `RavenMigrationSource` holding a `RavenReadOnlySourceLifecycle` that has connected to nothing. +- **`CopyableCategoryIds` intersects the two registries.** The source's `SupportedCategoryIds` is the key set of its reader dictionary, and the target's is the key set of its writer dictionary, built inside `EFCoreMigrationTarget`. A category needs both a reader and a writer, so a half-built one is never handed to the engine. +- **`EveryRequiredCategoryCanBeCopiedCheck` then refuses a partial build outright.** Copying part of the required set and opening the host anyway would commit the instance to SQL with required data that nothing goes back for. Today's readers and writers cover `KnownEndpoints` and `EndpointSettings` only, so on a real build this check fails and no migration starts at all. `AllowIncompleteCategorySet` is how the acceptance tests get past it, and nothing in the product registers it. It is a DI marker class, not a setting, and it is not `Migration/AllowIncompleteExit`, which is a setting and does something else entirely. +- **`SelectedCategoriesAreCoherentCheck` reads the options as a check rather than in a constructor.** It calls `MigrationEngineOptions.FromSettings`, which throws when `Migration/OptionalCategories` names a category that does not exist or is not optional. Running it as a step means that typo is reported by name beside the other failures, instead of surfacing as an unrelated startup crash. +- **The target's own checks run before either store is opened.** `EFCoreMigrationTargetReadiness.ContributedChecks` returns `RetryHistoryDepthIsSafeCheck`, `SchemaIsCurrentCheck` and `BodyStorageIsWritableCheck`, in that order. That is the one that needs nothing, then the one that reads the schema, then the one that writes a probe body. The probe is deleted again, so the target's body count does not end up one higher than the source's. +- **The source opens last of all the checks**, because it is the only step that can start a process. `RavenReadOnlySourceLifecycle.Open` does four things. It starts the embedded server or connects to the external one. It hangs `RefuseWrite` on every request, so nothing in the copy can write to the database it is reading. It rejects a client certificate that is expired or not yet valid before the first request rather than after it. It then waits for both databases to load, with a five-minute budget on an embedded one. +- **`SourceDataVersionIsReadableCheck` needs a second round of checks** because it needs a session, and a session needs the source open. It reads the `ServiceControl/DataVersion` document that `DatabaseSetup.StampDataVersion` writes on every RavenDB startup, and refuses four cases by name: no stamp at all, a stamp it cannot parse, a newer major version, and an older major version. Reading those documents into shapes this build understands differently is the loss it is there to prevent. +- **`ClosedWindowProgress` wraps the run, not the engine.** It polls the checkpoint store every 30 seconds, logs each running category's copied, skipped and cursor, and cancels the copy when a category has committed nothing for 30 minutes. It is host-side deliberately: the engine cannot know how long a batch takes, and what follows a stall is a message about an outage rather than about a copy. +- **Two catch blocks around `RunCategories` rewrite the failure before it is reported.** A stall re-reads the checkpoints on the caller's token, so the operator gets the stall message rather than a bare cancellation. A `MigrationCheckpointConflictException` is reported as a second instance pointed at the same database. +- **`RefuseIfAnyCategoryDidNotComplete` is what keeps the host closed.** Any category not `Complete`, `CompleteWithErrors` or `Abandoned`, and any attempted category that reported no checkpoint at all, throws with the counts, the last error and the way back to RavenDB. +- **`CheckpointMigrationState.Seed` runs last, before the host opens.** It reads every checkpoint once and works out whether any selected category is unfinished. A snapshot rather than a live query, because the services meant to consult it sit on the request path and cannot afford a database read each time. +- **`RecordHostOpenedOnTarget` stamps the target, and it is not part of the copy.** It is a hosted service the EF Core persister registers, and it runs in `StartedAsync` on any host start, not only `RunCommand`. It writes the `Migration/HostOpenedOnTarget` setting the first time the host opens on a database that already holds checkpoint rows, and leaves it alone afterwards. That gate on existing checkpoints is the whole reason the marker means "a host has opened since the copy began" rather than "a host ran here once". + +## One batch, and every class it passes through + +`MigrationEngine.RunCategoryAsync` is the whole copy for one category. Each step below is delegated, and the reason is always the same shape: the engine must not learn anything about either store. + +| Step | Where the work happens | Why not in the engine | +| --- | --- | --- | +| Read the category's saved row | `EFMigrationCheckpointStore.Read` | The checkpoint table lives in the target's database, so only the target's assembly can query it | +| Pass over a category that is done | `MigrationCategoryStateExtensions.IsFinished` | One definition of "finished" shared by the engine, the host's refusal and the state seed, so the three cannot drift | +| Hold a category behind another | `MigrationCategory.MustFollow`, from `MigrationCategoryRegistry.All` | Ordering is data, not code: group comments follow archived messages because the registry says so | +| Choose the batch size | `EFCoreMigrationTarget.BatchSizeFor`, which asks the category's `IMigrationCategoryWriter` | Only the target knows its provider's parameter ceiling and its own column count | +| Count the source once, on the first run | `RavenMigrationSource.Count` | It streams the category rather than reading collection statistics, because the count has to be over exactly the rows `Read` will hand back. A total that included rows `Read` leaves out would halt the category for a shortfall that never happened | +| Read a batch after the cursor | the category's `IMigrationCategoryReader`, through `RavenDocumentStream.ByPrefix` | The stream opens a no-tracking session, refuses a cursor whose document no longer exists (RavenDB would otherwise silently start after a non-existent id and skip rows), and advances the cursor on every document it sees, projected or not | +| Fetch message bodies | `MigrationEngine.FetchBodiesWithRetry`, calling `IMigrationSource.ReadBody` | The retry policy is store-neutral: three attempts, a backoff between them, and `IsDefect` exceptions such as `NotSupportedException` not retried at all | +| Turn documents into rows | the category's `IMigrationCategoryWriter.Prepare`, returning a `PreparedBatch` | Only the writer knows which columns are `NOT NULL`, which keys are capped, and what the running product would delete anyway. It writes nothing: the insert it returns runs later, inside the target's transaction | +| Insert the rows and save the checkpoint | `EFCoreMigrationTarget.Write`: `PreparedBatch.Insert` calls the provider's `IMigrationSqlDialect.InsertMissing`, built on `MigrationInsert`, then `AlreadyPresentIn`, `MigrationCheckpoint.Extend` and `MigrationCheckpointExtensions.UpsertCheckpoint`, all inside one transaction under the provider's execution strategy | This is the guarantee in [one batch, and why nothing provisional is ever saved](ravendb-to-sql-migration-overview.md#one-batch-and-why-nothing-provisional-is-ever-saved). The counts are added to the checkpoint inside the same transaction as the rows they describe, so nothing travels back through the engine to be saved later | +| Decide whether to halt | `HaltThreshold.Exceeded` and `HaltThreshold.MostOfItWasLost`, on counters the engine keeps per run | Both the percentage and the floor must be exceeded, and `MostOfItWasLost` catches a category too small to reach the floor. The counters are deliberately not stored, so a restart with the cause fixed does not re-trip on its first batch | +| Settle the category | `MigrationEngine.Settle` | The engine logs before it settles, because the checkpoint store shares the target's database and a failed save would otherwise hide the cause | + +Two details in that chain are worth following because they show how a rule travels between classes: + +- **`MigrationCheckpoint.Extend` refuses a batch whose skips are not all explained.** The target passes a count per `MigrationSkipReason`, and the save throws unless those sum exactly to the skipped count, in either direction. `EFCoreMigrationTarget.Write` makes the same bargain in the other direction with `AlreadyPresentIn`: every row in the batch is copied, skipped or already present, and a miscount throws instead of quietly shrinking the halt threshold's denominator. +- **`EndpointSettingsWriter` reads the `KnownEndpoints` checkpoint row before it decides anything.** It skips settings for an endpoint the target does not know, because the heartbeat sync would delete them shortly after startup, and those skips are benign. But if the `KnownEndpoints` copy itself dropped rows, an unknown endpoint can be this migration's own doing rather than the source's, so `PreparedBatch.SkipsReflectTheSource` goes false, `BenignSkipCount` returns zero, and those skips start counting toward a halt again. `EndpointNotKnown` is the only reason `MigrationSkipReasonExtensions.IsBenign` accepts, so that one writer is where the whole benign-skip exemption lives. + +## The read-only commands use the source and nothing else + +`--migration-source-report` runs `MigrationSourceReportCommand`, which calls `PersistenceFactory.OpenMigrationSource` and then `IMigrationSource.Describe` and `Inventory`. Those two members exist for that command alone: the copy never calls either. `Describe` returns `MigrationSourceFact` values, and each carries the setting key its value came from wherever there is one, which is what lets the report tell an operator not just that the database name is wrong but which setting to change. The `Mode` fact has no setting key and prints without one. `Inventory` reports every collection in both source databases, including data no category copies, so the operator sees the whole source rather than the part we intend to move. + +## What is built but not yet on a call path + +*Listed because a reader tracing these classes will otherwise assume the behaviour above them is live.* + +- **Sixteen of the eighteen categories have no reader and no writer.** `RavenMigrationSource` holds `KnownEndpointsReader` and `EndpointSettingsReader`, and `EFCoreMigrationTarget` holds the matching two writers. Every other category throws `NotSupportedException` on both sides, and `EveryRequiredCategoryCanBeCopiedCheck` therefore refuses every real migration. +- **The `Migration/Enabled = false` exit gate is not wired.** `IMigrationTargetReadiness.HasHostOpened` exists and nothing reads it, and the main host does not read `Settings.MigrationAllowIncompleteExit`, so the branch in [turning migration mode off is a gated startup too](ravendb-to-sql-migration-overview.md#turning-migration-mode-off-is-a-gated-startup-too) does not run yet and an instance with an outstanding category simply starts. `RecordHostOpenedOnTarget` does run, so the marker that gate will need is already being written. Nothing anywhere sets `MigrationCategoryState.Abandoned`. The one reader of `MigrationAllowIncompleteExit` is `FinishedCopyBeforeAnIngestionNodeOpens`, described [above](#the-wiring-is-in-place-on-every-sql-instance-migrating-or-not). +- **`IMigrationState.AnyCategoryIncomplete` is seeded and has no production reader.** The services that are meant to stand down while a copy is outstanding still run as normal. +- **`RavenMigrationSource.ReadBody` throws for every category**, so no `CarriesBodies` category can be copied yet, which includes the required failed messages. `NotSupportedException` is on the engine's non-retryable list, so a body-carrying category would not make three attempts and would not skip the message: it would halt the category. +- **The background copy of optional categories has no caller.** `MigrationStartup.RunRequiredCopy` selects the optional ids only so `CheckpointMigrationState.Seed` knows which categories count as outstanding, and nothing runs them. The engine's throttle pause applies only to optional categories, so nothing in the product ever waits between batches today. +- **`CheckpointMigrationState.Recompute` is public for that background copy to call as each category settles.** Today only `Seed` calls it. +- **`MigrationSkipReason.PastRetention` has no writer.** The enum value exists and `IsBenign` does not accept it, so the first writer to use it will count past-retention skips toward a halt unless that is changed too. +- **There is no migration custom check and no migration domain event.** Progress reaches an operator only through `ILogger` output. diff --git a/src/ServiceControl.AcceptanceTests.PostgreSql/ServiceControl.AcceptanceTests.PostgreSql.csproj b/src/ServiceControl.AcceptanceTests.PostgreSql/ServiceControl.AcceptanceTests.PostgreSql.csproj index 718b22bc78..a55fcba55b 100644 --- a/src/ServiceControl.AcceptanceTests.PostgreSql/ServiceControl.AcceptanceTests.PostgreSql.csproj +++ b/src/ServiceControl.AcceptanceTests.PostgreSql/ServiceControl.AcceptanceTests.PostgreSql.csproj @@ -12,6 +12,9 @@ + + + @@ -20,6 +23,7 @@ + @@ -31,6 +35,9 @@ + + + diff --git a/src/ServiceControl.AcceptanceTests.SqlServer/ServiceControl.AcceptanceTests.SqlServer.csproj b/src/ServiceControl.AcceptanceTests.SqlServer/ServiceControl.AcceptanceTests.SqlServer.csproj index cbc66d5b4a..0d67353174 100644 --- a/src/ServiceControl.AcceptanceTests.SqlServer/ServiceControl.AcceptanceTests.SqlServer.csproj +++ b/src/ServiceControl.AcceptanceTests.SqlServer/ServiceControl.AcceptanceTests.SqlServer.csproj @@ -12,6 +12,9 @@ + + + @@ -20,6 +23,7 @@ + @@ -31,6 +35,9 @@ + + + diff --git a/src/ServiceControl.AcceptanceTests/Recoverability/When_hosting_error_ingestion_only.cs b/src/ServiceControl.AcceptanceTests/Recoverability/When_hosting_error_ingestion_only.cs index 092652e7d9..72d5790024 100644 --- a/src/ServiceControl.AcceptanceTests/Recoverability/When_hosting_error_ingestion_only.cs +++ b/src/ServiceControl.AcceptanceTests/Recoverability/When_hosting_error_ingestion_only.cs @@ -29,10 +29,9 @@ namespace ServiceControl.AcceptanceTests.Recoverability using ServiceControl.Infrastructure; using ServiceControl.MessageFailures; using ServiceControl.Operations; + using ServiceControl.Persistence.DataMigration; using ServiceControl.Persistence.EFCore.DbContexts; using ServiceControl.Persistence.EFCore.Entities; - using ServiceControl.Persistence.EFCore.Infrastructure; - using ServiceControl.Recoverability; using ServiceControl.Transports; class When_hosting_error_ingestion_only : AcceptanceTest @@ -97,9 +96,38 @@ public async Task Should_default_the_concurrency_to_10_when_none_is_configured() "InternalCustomChecksHostedService", // reports this node's ingestion health to the database "MetricsReporterHostedService", "HealthCheckPublisherHostedService", // inert, no IHealthCheckPublisher is registered - "ExternalIntegrationRequestsDataStore" // its drain is inert here, nothing calls Subscribe + "ExternalIntegrationRequestsDataStore", // its drain is inert here, nothing calls Subscribe + "CheckpointTableIsReadable", // reads one table and refuses a start against a schema older than the build + "RecordHostOpenedOnTarget", // this node writes to the target, so the stamp belongs here, and it upserts one settings row + "FinishedCopyBeforeAnIngestionNodeOpens" // keeps this node out of a database a copy has not finished filling ]; + [Test] + public async Task Should_refuse_to_start_while_a_copy_into_the_database_is_unfinished() + { + var settings = await CreateSettings(); + + await new SetupCommand().Execute(new HostArguments([]), settings); + + var host = ErrorIngestionOnlyCommand.BuildHost(settings); + + try + { + await host.Services.GetRequiredService().Upsert( + new MigrationCheckpoint(MigrationCategoryIds.KnownEndpoints, MigrationCategoryState.InProgress, null, 0, 0, null, null, null, null, null, null)); + + var exception = Assert.ThrowsAsync(() => host.StartAsync()); + + Assert.That(exception.Message, Does.Contain("has not finished") + .And.Contain(MigrationCategoryIds.KnownEndpoints) + .And.Contain(MigrationSettings.AllowIncompleteExitKey)); + } + finally + { + await host.DisposeAsync(); + } + } + [Test] public void Should_refuse_to_start_against_unsupported_storage() { @@ -204,6 +232,7 @@ public async Task Should_serve_health_over_https_with_the_configured_certificate await new SetupCommand().Execute(new HostArguments([]), settings); host = ErrorIngestionOnlyCommand.BuildHost(settings); + host.Urls.Clear(); host.Urls.Add("https://127.0.0.1:0"); await host.StartAsync(); diff --git a/src/ServiceControl.Migration.AcceptanceTests/.editorconfig b/src/ServiceControl.Migration.AcceptanceTests/.editorconfig new file mode 100644 index 0000000000..da44eb13fb --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/.editorconfig @@ -0,0 +1,6 @@ +[*.cs] + +# Justification: Test project +dotnet_diagnostic.CA2007.severity = none +dotnet_diagnostic.PS0013.severity = none +dotnet_diagnostic.PS0018.severity = none diff --git a/src/ServiceControl.Migration.AcceptanceTests/MigrationAcceptanceTest.cs b/src/ServiceControl.Migration.AcceptanceTests/MigrationAcceptanceTest.cs new file mode 100644 index 0000000000..9ce2d1de75 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/MigrationAcceptanceTest.cs @@ -0,0 +1,472 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.Collections.Generic; +using System.Globalization; +using System.Linq; +using System.Net.Http; +using System.Net.Http.Json; +using System.Reflection; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.AspNetCore.Builder; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Logging.Abstractions; +using NServiceBus.Extensibility; +using NServiceBus.Transport; +using NUnit.Framework; +using ServiceControl.Persistence.EFCore; +using Particular.ServiceControl.Hosting; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.AcceptanceTesting.InfrastructureConfig; +using ServiceControl.Hosting.Commands; +using ServiceControl.Infrastructure.WebApi; +using ServiceControl.Migration.Checks; +using ServiceControl.Operations; +using ServiceControl.Persistence; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.Abstractions; +using ServiceControl.Persistence.EFCore.DbContexts; +using ServiceControl.Persistence.EFCore.Entities; +using ServiceControl.Persistence.EFCore.Infrastructure; +using ServiceControl.Persistence.Infrastructure; +using ServiceControl.Persistence.Tests; +using ServiceControl.Persistence.UnitOfWork; +using TestHelper; +// Aliased rather than imported: Raven.Client.Documents carries LINQ extensions that collide with EF Core's. +using IDocumentStore = Raven.Client.Documents.IDocumentStore; + +// No base class: the persistence test bases sit in projects that cannot see RavenDB. +abstract class MigrationAcceptanceTest +{ + protected const string EndpointSettingsUrl = "api/endpointssettings"; + + readonly AcceptanceTestStorageConfiguration StorageConfiguration = new(); + readonly List setVariables = []; + CancellationTokenSource hostCancellation; + Task runningHost; + + ServiceProvider targetServices; + IMigrationCheckpointStore targetCheckpointStore; + + protected Settings Settings { get; private set; } + protected HttpClient HttpClient { get; private set; } + protected (string ServerUrl, string PrimaryDatabase, string ThroughputDatabase) Source { get; private set; } + protected IDocumentStore SourceStore { get; private set; } + + protected IMigrationTarget Target { get; private set; } + + protected InMemoryBodyStoragePersistence RecordedBodies { get; } = new(); + + [SetUp] + public async Task SetUp() + { + Source = await MigrationSourceServer.CreateDatabases(); + SourceStore = await (await MigrationSourceServer.GetInstance()).Connect(); + + await SeedSourceDataVersion(); + + SetSourceVariable("SERVICECONTROL_RAVENDB_CONNECTIONSTRING", Source.ServerUrl); + SetSourceVariable("SERVICECONTROL_RAVENDB_DATABASENAME", Source.PrimaryDatabase); + SetSourceVariable("LICENSINGCOMPONENT_RAVENDB_THROUGHPUTDATABASENAME", Source.ThroughputDatabase); + SetSourceVariable("SERVICECONTROL_ERRORRETENTIONPERIOD", "10.00:00:00"); + SetSourceVariable("SERVICECONTROL_MIGRATION_ENABLED", "true"); + + // Settings.Port has no public setter, so the port has to come in as the environment variable a customer would set. + SetSourceVariable("SERVICECONTROL_PORT", PortUtility.GetAssignedOrAvailablePort(33500).ToString(CultureInfo.InvariantCulture)); + + var transport = new ConfigureEndpointLearningTransport(); + + Settings = new Settings( + transportType: transport.TypeName, + persisterType: StorageConfiguration.PersistenceType, + forwardErrorMessages: false, + errorRetentionPeriod: TimeSpan.FromDays(10)) + { + TransportConnectionString = transport.ConnectionString + }; + + await StorageConfiguration.CustomizeSettings(Settings); + await new SetupCommand().Execute(new HostArguments([]), Settings); + + HttpClient = new HttpClient { BaseAddress = new Uri(Settings.RootUrl) }; + + var targetServiceCollection = new ServiceCollection(); + targetServiceCollection.AddLogging(); + targetServiceCollection.AddPersistence(Settings); + targetServiceCollection.AddSingleton(RecordedBodies); + targetServices = targetServiceCollection.BuildServiceProvider(); + + Target = targetServices.GetRequiredService(); + targetCheckpointStore = targetServices.GetRequiredService(); + await Target.Open(); + } + + // The variables go first because they are process wide: a failure in the cleanup below must not leave them set for the next test. + [TearDown] + public async Task TearDown() + { + foreach (var name in setVariables) + { + Environment.SetEnvironmentVariable(name, null); + } + + try + { + await StopHost(); + } + finally + { + HttpClient?.Dispose(); + SourceStore?.Dispose(); + + if (targetServices is not null) + { + await targetServices.DisposeAsync(); + } + + await StorageConfiguration.Cleanup(); + } + } + + protected void SetSourceVariable(string name, string value) + { + Environment.SetEnvironmentVariable(name, value); + setVariables.Add(name); + } + + protected async Task SeedSource(string database, params (string Id, object Document)[] documents) + { + using var session = SourceStore.OpenAsyncSession(database); + + foreach (var (id, document) in documents) + { + await session.StoreAsync(document, id); + } + + await session.SaveChangesAsync(); + } + + protected Task SeedSource(params (string Id, object Document)[] documents) => + SeedSource(Source.PrimaryDatabase, documents); + + protected Task SeedSourceEndpointSettings(params (string Name, bool TrackInstances)[] settings) => + SeedSource([.. settings.Select(setting => ($"EndpointSettings/{DeterministicGuid.MakeId(setting.Name)}", (object)new EndpointSettings { Name = setting.Name, TrackInstances = setting.TrackInstances }))]); + + // The target copies an endpoint setting only when that endpoint is known, so seed the endpoint too. + protected Task SeedSourceKnownEndpoints(params string[] names) => + SeedSource([.. names + .Select(name => new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = name, HostId = Guid.NewGuid(), Host = "HOST01" } }) + .Select(endpoint => ($"KnownEndpoints/{endpoint.EndpointDetails.GetDeterministicId()}", (object)endpoint))]); + + // What DatabaseSetup.StampDataVersion writes on a real instance, in both databases because the + // check reads both. Without it every startup check refuses. + protected async Task SeedSourceDataVersion(string version = null) + { + foreach (var database in new[] { Source.PrimaryDatabase, Source.ThroughputDatabase }) + { + await SeedSource(database, (RavenDataVersionDocumentId, new SourceDataVersion { Version = version ?? ThisBuildVersion, StampedAt = DateTime.UtcNow })); + } + } + + // Spelled out rather than referenced: the acceptance projects cannot see the RavenDB persister's types. + const string RavenDataVersionDocumentId = "ServiceControl/DataVersion"; + + static readonly string ThisBuildVersion = + typeof(Settings).Assembly.GetCustomAttribute()?.InformationalVersion.Split('+')[0] + ?? typeof(Settings).Assembly.GetName().Version!.ToString(3); + + protected Task OpenSource(CancellationToken cancellationToken = default) => + PersistenceFactory.OpenMigrationSource(Settings, cancellationToken); + + static IReadOnlyCollection sourceSupportedCategoryIds; + + // SupportedCategoryIds answers before Open and cannot change across it, so one source built once for the + // whole run answers it for every test instead of loading the persister assembly again per test. + protected async Task> SourceSupportedCategoryIds() + { + if (sourceSupportedCategoryIds is null) + { + await using var source = PersistenceFactory.CreateMigrationSource(Settings); + sourceSupportedCategoryIds = source.SupportedCategoryIds; + } + + return sourceSupportedCategoryIds; + } + + + // This build cannot copy every required category yet, so without the marker every host test is refused at startup. + protected static Action AllowingAnIncompleteCategorySet(Action customize = null) => + builder => + { + builder.Services.AddSingleton(); + customize?.Invoke(builder); + }; + + protected async Task RunHostUntilTheApiAnswers(Action customize = null) + { + await StopHost(); + + hostCancellation = new CancellationTokenSource(); + runningHost = RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(customize), hostCancellation.Token); + + await WaitForEndpointSettingsResponse(TimeSpan.FromMinutes(2)); + } + + protected Settings SettingsWithMigrationEnabled(string optionalCategories = null, Action customize = null) + { + SetSourceVariable("SERVICECONTROL_MIGRATION_ENABLED", "true"); + SetSourceVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", optionalCategories); + customize?.Invoke(Settings); + return Settings; + } + + protected async Task RunRequiredCopy( + Action customize = null, + bool leaveOptionalIncomplete = false, + CancellationToken cancellationToken = default) + { + SettingsWithMigrationEnabled(optionalCategories: leaveOptionalIncomplete ? "EventLog" : null); + + await RunHostUntilTheApiAnswers(builder => customize?.Invoke(builder.Services)); + + var copyable = MigrationStartup.CopyableCategoryIds(await SourceSupportedCategoryIds(), Target.SupportedCategoryIds); + + var copied = MigrationCategoryRegistry.All + .Where(category => category.Kind == MigrationCategoryKind.Required && copyable.Contains(category.Id)) + .Select(category => category.Id) + .ToArray(); + + await WaitUntil( + async () => (await targetCheckpointStore.ReadAll(cancellationToken)) + .Count(checkpoint => copied.Contains(checkpoint.CategoryId) && checkpoint.State.IsFinished()) == copied.Length, + "the required copy finished"); + + if (leaveOptionalIncomplete) + { + // The guarded services read IMigrationState.AnyCategoryIncomplete, which any selected category that has not finished holds true. Halted is one of those states. + await targetCheckpointStore.Upsert( + new MigrationCheckpoint("EventLog", MigrationCategoryState.Halted, null, 0, 0, null, null, null, null, null, + "Halted by the test fixture so the guarded services see an unfinished category."), + cancellationToken); + } + } + + protected async Task> WaitForEndpointSettingsResponse(TimeSpan timeout) + { + IReadOnlyList settings = null; + + await WaitUntil(async () => + { + // A host that refused to start surfaces its own exception here instead of waiting out the timeout. + if (runningHost is { IsFaulted: true }) + { + await runningHost; + } + + try + { + settings = await GetEndpointSettings(); + return true; + } + catch (HttpRequestException) + { + return false; + } + }, $"GET {EndpointSettingsUrl} answered", timeout); + + return settings; + } + + protected async Task> GetEndpointSettings() => + await HttpClient.GetFromJsonAsync>(EndpointSettingsUrl, SerializerOptions.Default); + + async Task StopHost() + { + if (runningHost is null) + { + return; + } + + await hostCancellation.CancelAsync(); + await runningHost; + hostCancellation.Dispose(); + runningHost = null; + } + + // Copied from PersistenceTestBase.WaitUntil, which this project cannot compile. + protected static async Task WaitUntil(Func> conditionChecker, string condition, TimeSpan timeout = default) + { + timeout = timeout == default ? TimeSpan.FromSeconds(10) : timeout; + + var start = DateTime.UtcNow; + + while (DateTime.UtcNow - start < timeout) + { + if (await conditionChecker()) + { + return; + } + + await Task.Delay(TimeSpan.FromMilliseconds(500)); + } + + throw new Exception($"{condition} has not been meet in defined timespan: {timeout})"); + } + + // Copied from RavenMigrationSourceTestBase.CollectBatches, which this project cannot compile. + protected static async Task> CollectBatches(IMigrationSource source, MigrationCategory category, string resumeAfter = null, int batchSize = 100) + { + var batches = new List(); + + await foreach (var batch in source.Read(category, resumeAfter, batchSize, TestContext.CurrentContext.CancellationToken)) + { + batches.Add(batch); + } + + return batches; + } + + protected IEndpointSettingsStore EndpointSettingsStore => targetServices.GetRequiredService(); + protected IMonitoringDataStore MonitoringDataStore => targetServices.GetRequiredService(); + + protected async Task CopyCategory(string categoryId, CancellationToken cancellationToken = default) + { + var category = MigrationCategoryRegistry.Find(categoryId) ?? throw new ArgumentException($"'{categoryId}' is not a migration category.", nameof(categoryId)); + var before = await targetCheckpointStore.Read(categoryId, cancellationToken); + var target = new RecordingMigrationTarget(Target); + var options = new MigrationEngineOptions(TimeSpan.Zero, MigrationSettings.DefaultHaltThresholdPercent, MigrationSettings.DefaultHaltThresholdMinimum, []) { BodyRetryBackoff = TimeSpan.Zero }; + + await using var source = await OpenSource(cancellationToken); + var engine = new MigrationEngine(source, target, targetCheckpointStore, TimeProvider.System, options, NullLogger.Instance); + + var after = await engine.RunCategoryAsync(category, cancellationToken); + + if (!after.State.IsFinished()) + { + throw new InvalidOperationException($"Copying '{categoryId}' ended {after.State}: {after.LastError}"); + } + + return new MigrationWriteResult( + after, + (int)(after.CopiedCount - (before?.CopiedCount ?? 0)), + (int)(after.SkippedCount - (before?.SkippedCount ?? 0)), + target.SkippedIds, + (int)(after.AlreadyPresentCount - (before?.AlreadyPresentCount ?? 0))); + } + + + // Makes the target look like a database a migration has touched, which is the only state the host-opened marker is stamped in. + protected async Task SeedCheckpoint(string categoryId, MigrationCategoryState state = MigrationCategoryState.Complete) + { + await using var scope = targetServices.CreateAsyncScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + await dbContext.UpsertCheckpoint(new MigrationCheckpoint(categoryId, state, null, 0, 0, null, null, null, null, null, null)); + } + + protected async Task DeleteCheckpoint(string categoryId) + { + await using var scope = targetServices.CreateAsyncScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + await dbContext.MigrationCheckpoints.Where(checkpoint => checkpoint.CategoryId == categoryId).ExecuteDeleteAsync(); + } + + protected async Task QueryTarget(Func> query) + { + using var scope = targetServices.CreateScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + return await query(dbContext); + } + + protected async Task GetFailedMessage(Guid uniqueMessageId) + { + var row = await QueryTarget(dbContext => dbContext.FailedMessages.AsNoTracking().SingleOrDefaultAsync(m => m.UniqueMessageId == uniqueMessageId)); + + Assert.That(row, Is.Not.Null, $"No failed message row for {uniqueMessageId}"); + + return row; + } + + protected Task FindFailedMessage(Guid uniqueMessageId) => + QueryTarget(dbContext => dbContext.FailedMessages.AsNoTracking().SingleOrDefaultAsync(m => m.UniqueMessageId == uniqueMessageId)); + + protected EFPersisterSettings EFSettings => targetServices.GetRequiredService(); + + protected Task Ingest(params IngestedFailure[] failures) => + InBatch(async unitOfWork => + { + foreach (var failure in failures) + { + await unitOfWork.Recoverability.RecordFailedProcessingAttempt(failure.Context, failure.ProcessingAttempt, failure.Groups); + } + }); + + // Ingestion takes the body and its content type from the message context, so only the context changes. + protected Task IngestWithBody(IngestedFailure failure, byte[] body, string contentType) => + InBatch(unitOfWork => + { + var headers = new Dictionary(failure.Headers) { [NServiceBus.Headers.ContentType] = contentType }; + var context = new MessageContext(failure.MessageId, headers, body, new TransportTransaction(), "receiveAddress", new ContextBag()); + + return unitOfWork.Recoverability.RecordFailedProcessingAttempt(context, failure.ProcessingAttempt, failure.Groups); + }); + + async Task InBatch(Func record) + { + await using var unitOfWork = await targetServices.GetRequiredService().StartNew(); + + await record(unitOfWork); + + await unitOfWork.Complete(TestContext.CurrentContext.CancellationToken); + } + + protected Task ReadCheckpoint(string categoryId) => + QueryTarget(async dbContext => + { + var row = await dbContext.MigrationCheckpoints.AsNoTracking().SingleOrDefaultAsync(checkpoint => checkpoint.CategoryId == categoryId); + + return row is null ? null : new MigrationCheckpoint( + row.CategoryId, row.State, row.Cursor, + row.CopiedCount, row.SkippedCount, row.SourceTotal, row.SkipReasons, + row.StartedAt, row.LastProgressAt, row.SettledAt, row.LastError, + row.AlreadyPresentCount, row.Version); + }); + + protected Task ReadSetting(string key) => + QueryTarget(dbContext => dbContext.Settings.AsNoTracking() + .Where(setting => setting.Key == key) + .Select(setting => setting.Value) + .SingleOrDefaultAsync()); + + sealed class RecordingMigrationTarget(IMigrationTarget inner) : IMigrationTarget + { + public List SkippedIds { get; } = []; + + public Task Open(CancellationToken cancellationToken = default) => inner.Open(cancellationToken); + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => inner.BatchSizeFor(category, cancellationToken); + + public async Task Write(MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default) + { + var result = await inner.Write(category, batch, checkpointToExtend, cancellationToken); + SkippedIds.AddRange(result.SkippedIds); + return result; + } + + public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => inner.Count(category, cancellationToken); + + public IReadOnlyCollection SupportedCategoryIds => inner.SupportedCategoryIds; + } + + // The stamp the RavenDB persister reads back, spelled out because this project cannot see its type. + sealed class SourceDataVersion + { + public string Version { get; set; } + + public DateTime StampedAt { get; set; } + } +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/MigrationSourceServer.cs b/src/ServiceControl.Migration.AcceptanceTests/MigrationSourceServer.cs new file mode 100644 index 0000000000..88a33b0003 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/MigrationSourceServer.cs @@ -0,0 +1,82 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.IO; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Hosting.Internal; +using Microsoft.Extensions.Logging.Abstractions; +using NUnit.Framework; +using Raven.Client.ServerWide; +using Raven.Client.ServerWide.Operations; +using ServiceControl.RavenDB; +using TestHelper; + +// Its own copy rather than SharedEmbeddedServer: that one uses the RavenDB persister's own settings type, which this project cannot reference because it loads the persister from its manifest instead. +static class MigrationSourceServer +{ + public static async Task GetInstance(CancellationToken cancellationToken = default) + { + await startLock.WaitAsync(cancellationToken); + try + { + if (server != null) + { + return server; + } + + var dbPath = Path.Combine(TestContext.CurrentContext.WorkDirectory, "Tests", "MigrationSource"); + var logPath = Path.Combine(TestContext.CurrentContext.WorkDirectory, "Logs", "MigrationSource"); + var port = PortUtility.GetAssignedOrAvailablePort(33350); + + var configuration = new EmbeddedDatabaseConfiguration($"http://localhost:{port}", "primary", dbPath, logPath, "Operations") { RunInMemory = true }; + + server = EmbeddedDatabase.Start(configuration, lifetime); + return server; + } + finally + { + startLock.Release(); + } + } + + public static async Task Stop() + { + await startLock.WaitAsync(); + try + { + if (server is null) + { + return; + } + + await server.Stop(CancellationToken.None); + server.Dispose(); + server = null; + lifetime.StopApplication(); + } + finally + { + startLock.Release(); + } + } + + public static async Task<(string ServerUrl, string PrimaryDatabase, string ThroughputDatabase)> CreateDatabases(CancellationToken cancellationToken = default) + { + var instance = await GetInstance(cancellationToken); + var primary = $"sc_src_{Guid.NewGuid():n}"; + var throughput = $"{primary}-throughput"; + + using var store = await instance.Connect(cancellationToken); + foreach (var name in new[] { primary, throughput }) + { + await store.Maintenance.Server.SendAsync(new CreateDatabaseOperation(new DatabaseRecord(name)), cancellationToken); + } + + return (instance.ServerUrl, primary, throughput); + } + + static EmbeddedDatabase server; + static readonly ApplicationLifetime lifetime = new(new NullLogger()); + static readonly SemaphoreSlim startLock = new(1, 1); +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/StopMigrationSourceServer.cs b/src/ServiceControl.Migration.AcceptanceTests/StopMigrationSourceServer.cs new file mode 100644 index 0000000000..b42d2671f2 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/StopMigrationSourceServer.cs @@ -0,0 +1,11 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System.Threading.Tasks; +using NUnit.Framework; + +[SetUpFixture] +public class StopMigrationSourceServer +{ + [OneTimeTearDown] + public Task Teardown() => MigrationSourceServer.Stop(); +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/TestMigrationTargets.cs b/src/ServiceControl.Migration.AcceptanceTests/TestMigrationTargets.cs new file mode 100644 index 0000000000..e07d2f4c09 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/TestMigrationTargets.cs @@ -0,0 +1,82 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.AspNetCore.Builder; +using Microsoft.Extensions.DependencyInjection; +using ServiceControl.Persistence.DataMigration; + +static class TestMigrationTargets +{ + public static void ParkFirstMigrationWrite(this WebApplicationBuilder builder, TaskCompletionSource parked, Task release) => + builder.DecorateMigrationTarget(inner => new ParkingMigrationTarget(inner, parked, release)); + + public static void FailTheSecondMigrationWrite(this WebApplicationBuilder builder) => + builder.DecorateMigrationTarget(inner => new FaultingMigrationTarget(inner, failingWrite: 2)); + + public static void HaltTheEndpointSettingsCategory(this WebApplicationBuilder builder) => + builder.DecorateMigrationTarget(inner => new FaultingMigrationTarget(inner, failingWrite: 1)); + + public static void DecorateMigrationTarget(this WebApplicationBuilder builder, Func decorate) + { + var registered = builder.Services.Last(service => service.ServiceType == typeof(IMigrationTarget)); + + if (registered.ImplementationType is null) + { + throw new InvalidOperationException( + "IMigrationTarget is registered by a factory rather than by type, so the test decorator cannot rebuild the inner target. Register it as AddSingleton() or give the decorator a different seam."); + } + + builder.Services.AddSingleton(provider => + decorate((IMigrationTarget)ActivatorUtilities.CreateInstance(provider, registered.ImplementationType))); + } +} + +// Throws on one EndpointSettings write, which is the only way to see what a copy does after a write has failed. +sealed class FaultingMigrationTarget(IMigrationTarget inner, int failingWrite) : IMigrationTarget +{ + int endpointSettingsWrites; + + public Task Open(CancellationToken cancellationToken = default) => inner.Open(cancellationToken); + + // Two rows a batch, so five settings take three writes and a failure on the second leaves exactly one batch committed. + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => + category.Id == MigrationCategoryIds.EndpointSettings ? Task.FromResult(2) : inner.BatchSizeFor(category, cancellationToken); + + public Task Write(MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default) => + category.Id == MigrationCategoryIds.EndpointSettings && ++endpointSettingsWrites == failingWrite + ? throw new Exception("injected mid-category failure") + : inner.Write(category, batch, checkpointToExtend, cancellationToken); + + public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => inner.Count(category, cancellationToken); + + public IReadOnlyCollection SupportedCategoryIds => inner.SupportedCategoryIds; +} + +// Holds the copy open at a moment a test can observe, which is the only way to ask what the API answers mid-copy. +sealed class ParkingMigrationTarget(IMigrationTarget inner, TaskCompletionSource parked, Task release) : IMigrationTarget +{ + int writes; + + public Task Open(CancellationToken cancellationToken = default) => inner.Open(cancellationToken); + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => inner.BatchSizeFor(category, cancellationToken); + + public async Task Write(MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default) + { + if (Interlocked.Exchange(ref writes, 1) == 0) + { + parked.SetResult(); + await release.WaitAsync(cancellationToken); + } + + return await inner.Write(category, batch, checkpointToExtend, cancellationToken); + } + + public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => inner.Count(category, cancellationToken); + + public IReadOnlyCollection SupportedCategoryIds => inner.SupportedCategoryIds; +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/When_a_copy_is_killed_mid_category.cs b/src/ServiceControl.Migration.AcceptanceTests/When_a_copy_is_killed_mid_category.cs new file mode 100644 index 0000000000..8c3b213893 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/When_a_copy_is_killed_mid_category.cs @@ -0,0 +1,70 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.EntityFrameworkCore; +using NUnit.Framework; +using ServiceControl.Hosting.Commands; + +[TestFixture] +// Mandatory, not stylistic: this assembly is Parallelizable(ParallelScope.All) and these fixtures set +// process-global environment variables. One fixture added without it makes the whole suite intermittent. +[NonParallelizable] +class When_a_copy_is_killed_mid_category : MigrationAcceptanceTest +{ + [Test] + public async Task It_resumes_rather_than_starting_over() + { + await SeedSourceKnownEndpoints("A", "B", "C", "D", "E"); + await SeedSourceEndpointSettings(("A", true), ("B", true), ("C", true), ("D", true), ("E", true)); + + // Bounded rather than None: a copy that did not fail would otherwise hold this call open forever. + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + + var killed = Assert.ThrowsAsync(async () => + await RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(builder => builder.FailTheSecondMigrationWrite()), cancellation.Token)); + + Assert.That(killed, Is.Not.Null); + Assert.That(await TargetEndpointSettingsCount(), Is.EqualTo(2), "the first batch and its cursor should have committed together"); + + await RunHostUntilTheApiAnswers(); + + var checkpoint = await ReadCheckpoint("EndpointSettings"); + + using (Assert.EnterMultipleScope()) + { + Assert.That(await GetEndpointSettingsNames(), Is.EquivalentTo(new[] { "A", "B", "C", "D", "E" }), "no gaps"); + Assert.That(checkpoint.CopiedCount, Is.EqualTo(5), "no row copied twice"); + Assert.That(checkpoint.SkippedCount, Is.Zero); + Assert.That(checkpoint.AlreadyPresentCount, Is.Zero, "a resumed run must not re-read rows the first run committed"); + } + } + + // Proves the already-present count in the test above can fail: both runs end with five correct rows, so that count is the only thing telling a resume from a fresh copy. + [Test] + public async Task Without_its_checkpoint_the_second_run_re_reads_what_the_first_already_wrote() + { + await SeedSourceKnownEndpoints("A", "B", "C", "D", "E"); + await SeedSourceEndpointSettings(("A", true), ("B", true), ("C", true), ("D", true), ("E", true)); + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + + Assert.ThrowsAsync(async () => + await RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(builder => builder.FailTheSecondMigrationWrite()), cancellation.Token)); + + await DeleteCheckpoint("EndpointSettings"); + await RunHostUntilTheApiAnswers(); + + var checkpoint = await ReadCheckpoint("EndpointSettings"); + + Assert.That(checkpoint.AlreadyPresentCount, Is.EqualTo(2)); + } + + Task TargetEndpointSettingsCount() => + QueryTarget(dbContext => dbContext.EndpointSettings.CountAsync()); + + Task GetEndpointSettingsNames() => + QueryTarget(dbContext => dbContext.EndpointSettings.Select(settings => settings.Name).ToArrayAsync()); +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/When_a_startup_check_fails.cs b/src/ServiceControl.Migration.AcceptanceTests/When_a_startup_check_fails.cs new file mode 100644 index 0000000000..f7afe5455b --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/When_a_startup_check_fails.cs @@ -0,0 +1,230 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.IO; +using System.Linq; +using System.Net.Http; +using System.Security.Cryptography; +using System.Security.Cryptography.X509Certificates; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Infrastructure; +using Microsoft.EntityFrameworkCore.Migrations; +using NUnit.Framework; +using ServiceControl.Hosting.Commands; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.Abstractions; + +[TestFixture] +// Mandatory, not stylistic: this assembly is Parallelizable(ParallelScope.All) and these fixtures set +// process-global environment variables. One fixture added without it makes the whole suite intermittent. +[NonParallelizable] +class When_a_startup_check_fails : MigrationAcceptanceTest +{ + const string HostOpenedSetting = "Migration/HostOpenedOnTarget"; + const string DataVersionDocumentId = "ServiceControl/DataVersion"; + + [Test] + public async Task An_unknown_category_name_is_refused() + { + SetSourceVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "NotACategory"); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the selected categories are coherent").And.Contain("NotACategory")); + } + + [Test] + public async Task A_retry_history_depth_that_empties_the_migrated_table_is_refused() + { + // Assigned rather than set as an environment variable: Settings reads RetryHistoryDepth once, in the constructor the fixture has already run. + Settings.RetryHistoryDepth = 0; + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the retry history depth will not empty a migrated table") + .And.Contain("RetryHistoryDepth") + .And.Contain("HistoricRetryOperations")); + } + + [Test] + public async Task A_target_whose_schema_migrations_were_never_applied_is_refused() + { + await QueryTarget(async dbContext => + { + // EF's own history repository, because it is the only thing that knows where the history table is once the persister is given a schema. + var history = dbContext.GetService(); + var applied = await history.GetAppliedMigrationsAsync(); + + foreach (var row in applied) + { + await dbContext.Database.ExecuteSqlRawAsync(history.GetDeleteScript(row.MigrationId)); + } + + return applied.Count; + }); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the target schema is current").And.Contain("--setup")); + } + + [Test] + public async Task Message_body_storage_that_cannot_be_written_to_is_refused() + { + var parentThatIsAFile = Path.Combine(Path.GetTempPath(), $"sc-not-a-directory-{Guid.NewGuid():n}"); + File.WriteAllText(parentThatIsAFile, string.Empty); + + try + { + ((FileSystemBodyStorageSettings)EFSettings.BodyStorage).StoragePath = Path.Combine(parentThatIsAFile, "bodies"); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("message body storage is writable")); + } + finally + { + File.Delete(parentThatIsAFile); + } + } + + [Test] + public async Task A_client_certificate_that_has_expired_is_refused() + { + var notBefore = new DateTimeOffset(2020, 1, 1, 0, 0, 0, TimeSpan.Zero); + var notAfter = new DateTimeOffset(2021, 1, 1, 0, 0, 0, TimeSpan.Zero); + + SetSourceVariable("SERVICECONTROL_RAVENDB_CLIENTCERTIFICATEBASE64", ExpiredClientCertificate(notBefore, notAfter)); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the migration source opens") + .And.Contain($"{notBefore.UtcDateTime:u} to {notAfter.UtcDateTime:u}")); + } + + [Test] + public async Task A_throughput_database_that_does_not_exist_is_refused() + { + var missing = $"{Source.ThroughputDatabase}-does-not-exist"; + SetSourceVariable("LICENSINGCOMPONENT_RAVENDB_THROUGHPUTDATABASENAME", missing); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the migration source opens") + .And.Contain(missing) + .And.Contain("LicensingComponent/RavenDB/ThroughputDatabaseName")); + } + + [Test] + public async Task A_source_that_carries_no_data_version_stamp_is_refused() + { + await DeleteSourceDataVersion(); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the source is at a data version this build can read") + .And.Contain("carries no ServiceControl data version stamp")); + } + + // The case a customer reaches by upgrading their binaries and turning the migration on without running the new build against RavenDB, which is what restamps the version. + [Test] + public async Task A_source_last_written_by_an_older_major_version_is_refused() + { + await DeleteSourceDataVersion(); + await SeedSourceDataVersion("1.0.0"); + + var refusal = await RefusedStartup(); + + Assert.That(refusal.Message, Does.Contain("the source is at a data version this build can read") + .And.Contain("1.0.0") + .And.Contain("brings the source up to date and restamps it")); + } + + // Without this check a customer on an intermediate build fills a database they can never finish. + [Test] + public async Task A_build_that_cannot_copy_every_required_category_is_refused() + { + SetSourceVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "NotACategory"); + + var copyable = MigrationStartup.CopyableCategoryIds(await SourceSupportedCategoryIds(), Target.SupportedCategoryIds); + var stillMissing = MigrationCategoryRegistry.All + .Where(category => category.Kind == MigrationCategoryKind.Required && !copyable.Contains(category.Id)) + .Select(category => category.Id) + .FirstOrDefault(); + + Assert.That(stillMissing, Is.Not.Null, "this build can now copy every required category, so this test has outlived its subject and should be deleted"); + + var refusal = await RefusedStartup(allowIncompleteCategorySet: false); + + Assert.That(refusal.Message, Does.Contain("this build can copy every required category") + .And.Contain(stillMissing) + .And.Not.Contain("the selected categories are coherent"), + "this refusal arrives before every check that follows it, including the one the broken setting above would trip"); + } + + [Test] + public async Task A_halted_required_category_stops_the_host_rather_than_opening_it() + { + await SeedSourceKnownEndpoints("Sales"); + await SeedSourceEndpointSettings(("Sales", true)); + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + + var exception = Assert.ThrowsAsync(async () => + await RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(builder => builder.HaltTheEndpointSettingsCategory()), cancellation.Token)); + + Assert.Multiple(() => + { + Assert.That(exception.Message, Does.Contain("EndpointSettings").And.Contain("Halted")); + Assert.That(exception.Message, Does.Contain("nothing has been lost")); + Assert.That(exception.Message, Does.Not.Contain("AllowIncompleteExit"), "the clean abort is the answer here, not the flag that accepts loss"); + Assert.That(async () => await HttpClient.GetAsync(EndpointSettingsUrl), Throws.InstanceOf(), "Kestrel bound its socket behind a halted required category"); + }); + } + + // Every refusal is asserted against a source that has rows to copy: an empty target proves nothing otherwise. + // Only the test about the required-set check itself withholds the marker; every other refusal has to get past it. + async Task RefusedStartup(bool allowIncompleteCategorySet = true) + { + await SeedSourceKnownEndpoints("Sales"); + await SeedSourceEndpointSettings(("Sales", true)); + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + + var refusal = Assert.ThrowsAsync(async () => + await RunCommand.Run(Settings, allowIncompleteCategorySet ? AllowingAnIncompleteCategorySet() : null, cancellation.Token)); + + using (Assert.EnterMultipleScope()) + { + Assert.That(refusal.Message, Does.Contain("nothing has been copied")); + Assert.That(await TargetEndpointSettingsCount(), Is.Zero, "a check that fires after rows have moved is worse than no check"); + Assert.That(await ReadSetting(HostOpenedSetting), Is.Null, "a refused startup has opened on nothing, so the abort is still free"); + } + + return refusal; + } + + Task TargetEndpointSettingsCount() => + QueryTarget(dbContext => dbContext.EndpointSettings.CountAsync()); + + // The fixture stamps the source in its SetUp, so the two data version refusals start by removing that stamp. + async Task DeleteSourceDataVersion() + { + using var session = SourceStore.OpenAsyncSession(Source.PrimaryDatabase); + + session.Delete(DataVersionDocumentId); + + await session.SaveChangesAsync(); + } + + static string ExpiredClientCertificate(DateTimeOffset notBefore, DateTimeOffset notAfter) + { + using var key = RSA.Create(2048); + var request = new CertificateRequest("CN=sc-migration-source-expired", key, HashAlgorithmName.SHA256, RSASignaturePadding.Pkcs1); + using var certificate = request.CreateSelfSigned(notBefore, notAfter); + + return Convert.ToBase64String(certificate.Export(X509ContentType.Pkcs12)); + } +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/When_categories_change_between_restarts.cs b/src/ServiceControl.Migration.AcceptanceTests/When_categories_change_between_restarts.cs new file mode 100644 index 0000000000..c9a1d2c675 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/When_categories_change_between_restarts.cs @@ -0,0 +1,36 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System.Threading.Tasks; +using NUnit.Framework; + +[TestFixture] +// Mandatory, not stylistic: this assembly is Parallelizable(ParallelScope.All) and these fixtures set +// process-global environment variables. One fixture added without it makes the whole suite intermittent. +[NonParallelizable] +class When_categories_change_between_restarts : MigrationAcceptanceTest +{ + [Test] + public async Task Changing_the_selection_between_restarts_deletes_nothing() + { + await SeedSourceKnownEndpoints("Sales"); + await SeedSourceEndpointSettings(("Sales", true)); + + await RunHostUntilTheApiAnswers(); + Assert.That(await ReadCheckpoint("EventLog"), Is.Null, "a category no run has copied has no row, and no column anywhere says it was selected"); + + SetSourceVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "EventLog"); + await RunHostUntilTheApiAnswers(); + Assert.That(await ReadCheckpoint("EventLog"), Is.Null, "the required copy runs required categories only, so selecting an optional one starts nothing here"); + + SetSourceVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", string.Empty); + await RunHostUntilTheApiAnswers(); + + var endpointSettings = await ReadCheckpoint("EndpointSettings"); + + Assert.Multiple(() => + { + Assert.That(endpointSettings.CopiedCount, Is.EqualTo(1), "a finished category must not be copied again when the selection changes around it"); + Assert.That(endpointSettings.Cursor, Is.Not.Null, "changing the selection must not delete or reset what an earlier run recorded"); + }); + } +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/When_endpoint_settings_are_migrated.cs b/src/ServiceControl.Migration.AcceptanceTests/When_endpoint_settings_are_migrated.cs new file mode 100644 index 0000000000..dcc1428c13 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/When_endpoint_settings_are_migrated.cs @@ -0,0 +1,45 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Time.Testing; +using NUnit.Framework; + +[TestFixture] +// Mandatory, not stylistic: this assembly is Parallelizable(ParallelScope.All) and these fixtures set +// process-global environment variables. One fixture added without it makes the whole suite intermittent. +[NonParallelizable] +class When_endpoint_settings_are_migrated : MigrationAcceptanceTest +{ + [Test] + public async Task The_source_server_starts_and_creates_both_databases() + { + var source = await MigrationSourceServer.CreateDatabases(); + + Assert.That(source.ServerUrl, Does.StartWith("http://localhost:")); + Assert.That(source.PrimaryDatabase, Is.Not.Empty); + Assert.That(source.ThroughputDatabase, Is.EqualTo($"{source.PrimaryDatabase}-throughput")); + } + + [Test] + public async Task They_are_readable_through_the_product_read_api() + { + await SeedSourceKnownEndpoints("Sales", "Billing"); + await SeedSourceEndpointSettings(("Sales", true), ("Billing", false), ("", true)); + + // A clock that never moves keeps HeartbeatEndpointSettingsSyncHostedService's twenty second + // delay from elapsing and rewriting these rows while the assertions read them. + await RunHostUntilTheApiAnswers(builder => builder.Services.AddSingleton(new FakeTimeProvider())); + + var settings = await GetEndpointSettings(); + + Assert.Multiple(() => + { + Assert.That(settings.Single(row => row.Name == "Sales").TrackInstances, Is.True); + Assert.That(settings.Single(row => row.Name == "Billing").TrackInstances, Is.False); + Assert.That(settings.Single(row => row.Name == string.Empty).TrackInstances, Is.True); + }); + } +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/When_the_api_is_called_during_the_required_copy.cs b/src/ServiceControl.Migration.AcceptanceTests/When_the_api_is_called_during_the_required_copy.cs new file mode 100644 index 0000000000..8753f52057 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/When_the_api_is_called_during_the_required_copy.cs @@ -0,0 +1,42 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.Linq; +using System.Net.Http; +using System.Threading; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Hosting.Commands; + +[TestFixture] +// Mandatory, not stylistic: this assembly is Parallelizable(ParallelScope.All) and these fixtures set +// process-global environment variables. One fixture added without it makes the whole suite intermittent. +[NonParallelizable] +class When_the_api_is_called_during_the_required_copy : MigrationAcceptanceTest +{ + [Test] + public async Task The_api_does_not_answer_until_the_required_copy_has_finished() + { + await SeedSourceKnownEndpoints("Sales", "Billing", "Shipping"); + await SeedSourceEndpointSettings(("Sales", true), ("Billing", true), ("Shipping", true)); + + var copyIsParked = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + var releaseTheCopy = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + var host = RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(builder => builder.ParkFirstMigrationWrite(copyIsParked, releaseTheCopy.Task)), cancellation.Token); + + await copyIsParked.Task.WaitAsync(TimeSpan.FromMinutes(2)); + Assert.That(releaseTheCopy.Task.IsCompleted, Is.False, "the copy must still be parked for this assertion to mean anything"); + Assert.That(async () => await HttpClient.GetAsync(EndpointSettingsUrl), Throws.InstanceOf(), + "Kestrel had already bound its socket while the required copy was still running"); + + releaseTheCopy.SetResult(); + var settings = await WaitForEndpointSettingsResponse(TimeSpan.FromMinutes(2)); + + Assert.That(settings.Select(row => row.Name), Is.SupersetOf(new[] { "Sales", "Billing", "Shipping" })); + + await cancellation.CancelAsync(); + await host; + } +} diff --git a/src/ServiceControl.Migration.AcceptanceTests/When_the_host_opens_on_the_target.cs b/src/ServiceControl.Migration.AcceptanceTests/When_the_host_opens_on_the_target.cs new file mode 100644 index 0000000000..6f444bb978 --- /dev/null +++ b/src/ServiceControl.Migration.AcceptanceTests/When_the_host_opens_on_the_target.cs @@ -0,0 +1,106 @@ +namespace ServiceControl.Migration.AcceptanceTests; + +using System; +using System.Threading; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Hosting.Commands; + +[TestFixture] +// Mandatory, not stylistic: this assembly is Parallelizable(ParallelScope.All) and these fixtures set +// process-global environment variables. One fixture added without it makes the whole suite intermittent. +[NonParallelizable] +class When_the_host_opens_on_the_target : MigrationAcceptanceTest +{ + const string HostOpenedSetting = "Migration/HostOpenedOnTarget"; + + [Test] + public async Task The_row_does_not_exist_while_the_required_copy_is_still_running() + { + await SeedSourceKnownEndpoints("Sales"); + await SeedSourceEndpointSettings(("Sales", true)); + + var copyIsParked = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + var releaseTheCopy = new TaskCompletionSource(TaskCreationOptions.RunContinuationsAsynchronously); + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + var host = RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(builder => builder.ParkFirstMigrationWrite(copyIsParked, releaseTheCopy.Task)), cancellation.Token); + + await copyIsParked.Task.WaitAsync(TimeSpan.FromMinutes(2)); + Assert.That(await ReadSetting(HostOpenedSetting), Is.Null, "the abort is still free at this moment, and this row is what says it is not"); + + releaseTheCopy.SetResult(); + await WaitForEndpointSettingsResponse(TimeSpan.FromMinutes(2)); + + Assert.That(await ReadSetting(HostOpenedSetting), Is.Not.Null); + + await cancellation.CancelAsync(); + await host; + } + + // A customer whose migration never started must not be told they have passed the point of no return. + [Test] + public async Task The_row_does_not_exist_after_a_startup_check_refused() + { + SetSourceVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "NoSuchCategory"); + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + + Assert.That(async () => await RunCommand.Run(Settings, AllowingAnIncompleteCategorySet(), cancellation.Token), + Throws.Exception.With.Message.Contains("NoSuchCategory")); + + Assert.That(await ReadSetting(HostOpenedSetting), Is.Null, "a refused startup has opened on nothing"); + } + + // Ingestion-only hosts write failed messages into the same database a migration targets, so a rollback + // gate that had not seen them would discard everything they ingested. + [Test] + public async Task An_ingestion_only_host_records_the_row_too() + { + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + + // The ingestion-only host runs a real transport, which the migration fixture does not otherwise configure. + Settings.MaximumConcurrencyLevel = 1; + + await SeedCheckpoint(MigrationCategoryIds.KnownEndpoints); + + var app = ErrorIngestionOnlyCommand.BuildHost(Settings, AllowingAnIncompleteCategorySet()); + + await app.StartAsync(cancellation.Token); + + try + { + Assert.That(await ReadSetting(HostOpenedSetting), Is.Not.Null, "only RunCommand used to record this, so a scaled-out ingestion host wrote to the target and left no trace of having opened on it"); + } + finally + { + await app.StopAsync(cancellation.Token); + await app.DisposeAsync(); + } + } + + + // A customer running on SQL who has never migrated must not be told the way back is gone. + [Test] + public async Task A_database_no_migration_has_touched_is_not_stamped() + { + Settings.MaximumConcurrencyLevel = 1; + + using var cancellation = new CancellationTokenSource(TimeSpan.FromMinutes(3)); + var app = ErrorIngestionOnlyCommand.BuildHost(Settings, AllowingAnIncompleteCategorySet()); + + await app.StartAsync(cancellation.Token); + + try + { + Assert.That(await ReadSetting(HostOpenedSetting), Is.Null, "no checkpoint row exists, so no migration has ever run against this database"); + } + finally + { + await app.StopAsync(cancellation.Token); + await app.DisposeAsync(); + } + } + +} diff --git a/src/ServiceControl.Migration.Tests/MigrationCategoryCoverageTests.cs b/src/ServiceControl.Migration.Tests/MigrationCategoryCoverageTests.cs new file mode 100644 index 0000000000..1d8eaebca7 --- /dev/null +++ b/src/ServiceControl.Migration.Tests/MigrationCategoryCoverageTests.cs @@ -0,0 +1,76 @@ +namespace ServiceControl.Migration.Tests; + +using System; +using System.Collections.Generic; +using System.IO; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Persistence; +using ServiceControl.Persistence.DataMigration; + +[TestFixture] +[NonParallelizable] +class MigrationCategoryCoverageTests +{ + Settings settings; + + [SetUp] + public void SetUp() + { + // Both persistence configurations read ErrorRetentionPeriod through the settings reader rather + // than off the Settings object, so the constructor argument below does not satisfy either of them. + SetVariable("SERVICECONTROL_ERRORRETENTIONPERIOD", "10.00:00:00"); + // Registering the SQL Server persistence never connects, so any connection string satisfies it. + SetVariable("SERVICECONTROL_DATABASE_CONNECTIONSTRING", "Server=localhost;Database=ServiceControl;Trusted_Connection=True;TrustServerCertificate=True"); + SetVariable("SERVICECONTROL_MESSAGEBODY_STORAGETYPE", "FileSystem"); + SetVariable("SERVICECONTROL_MESSAGEBODY_FILESYSTEM_STORAGEPATH", + Path.Combine(TestContext.CurrentContext.WorkDirectory, "Bodies", Guid.NewGuid().ToString("n"))); + + settings = new Settings(persisterType: "SQLServer", forwardErrorMessages: false, errorRetentionPeriod: TimeSpan.FromDays(10)); + } + + [TearDown] + public void TearDown() + { + foreach (var name in variables) + { + Environment.SetEnvironmentVariable(name, null); + } + + variables.Clear(); + } + + // A reader with no writer, or the reverse, is a category the intersection would otherwise drop in silence. + [Test] + public async Task Every_reader_has_a_writer_and_the_reverse() + { + await using var source = PersistenceFactory.CreateMigrationSource(settings); + var sourceCategoryIds = source.SupportedCategoryIds; + + var services = new ServiceCollection(); + services.AddLogging(); + services.AddPersistence(settings); + await using var provider = services.BuildServiceProvider(); + var target = provider.GetRequiredService(); + + var readerOnly = sourceCategoryIds.Except(target.SupportedCategoryIds).ToArray(); + var writerOnly = target.SupportedCategoryIds.Except(sourceCategoryIds).ToArray(); + + using (Assert.EnterMultipleScope()) + { + Assert.That(readerOnly, Is.Empty, $"the reader claims a category the writer does not: {string.Join(", ", readerOnly)}"); + Assert.That(writerOnly, Is.Empty, $"the writer claims a category the reader does not: {string.Join(", ", writerOnly)}"); + } + } + + void SetVariable(string name, string value) + { + Environment.SetEnvironmentVariable(name, value); + variables.Add(name); + } + + readonly List variables = []; +} diff --git a/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlMigrationSqlDialect.cs b/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlMigrationSqlDialect.cs new file mode 100644 index 0000000000..fdac0674c0 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlMigrationSqlDialect.cs @@ -0,0 +1,43 @@ +namespace ServiceControl.Persistence.EFCore.PostgreSql; + +using ServiceControl.Persistence.EFCore.DbContexts; +using ServiceControl.Persistence.EFCore.Infrastructure; + +/// +/// The migration's PostgreSQL statements. An insert that does nothing on a conflicting key is all it takes to +/// add only what is absent, and the keys it inserted come back from the same statement. +/// +class PostgreSqlMigrationSqlDialect : PostgreSqlDialect, IMigrationSqlDialect +{ + // Deterministic collations compare keys byte for byte, so there is no schema fact to read and none to name in the statement. + public Task Open(ServiceControlDbContext dbContext, CancellationToken cancellationToken = default) => Task.CompletedTask; + + public IEqualityComparer KeyComparer(Type entityType, string propertyName) => StringComparer.Ordinal; + + // A fixed chunk, because PostgreSQL's parameter limit is far away and the row width does not bring it closer. + public int RowsPerStatement(int parametersPerRow) => MaxRowsPerStatement; + + public async Task> InsertMissing(ServiceControlDbContext dbContext, IReadOnlyList rows, CancellationToken cancellationToken = default) where TEntity : class + { + var insert = MigrationInsert.For(dbContext); + var keyColumns = string.Join(", ", insert.KeyColumns); + var inserted = new List(rows.Count); + + foreach (var chunk in rows.Chunk(RowsPerStatement(insert.Columns.Count))) + { + inserted.AddRange(await insert.Execute( + dbContext, + $""" + INSERT INTO {insert.Table} ({string.Join(", ", insert.Columns)}) + VALUES + {ParameterRows(chunk.Length, insert.Columns.Count)} + ON CONFLICT ({keyColumns}) DO NOTHING + RETURNING {keyColumns} + """, + chunk, + cancellationToken)); + } + + return inserted; + } +} diff --git a/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlPersistence.cs b/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlPersistence.cs index 3cbe6c1549..898a4a6bd4 100644 --- a/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlPersistence.cs +++ b/src/ServiceControl.Persistence.EFCore.PostgreSql/PostgreSqlPersistence.cs @@ -18,6 +18,7 @@ public void AddPersistence(IServiceCollection services) RegisterDataStores(services, settings); services.AddSingleton(); + services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); diff --git a/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerMigrationSqlDialect.cs b/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerMigrationSqlDialect.cs new file mode 100644 index 0000000000..01264a6d1e --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerMigrationSqlDialect.cs @@ -0,0 +1,119 @@ +namespace ServiceControl.Persistence.EFCore.SqlServer; + +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Infrastructure; +using Microsoft.EntityFrameworkCore.Metadata; +using Microsoft.EntityFrameworkCore.Storage; +using ServiceControl.Persistence.EFCore.DbContexts; +using ServiceControl.Persistence.EFCore.Infrastructure; + +/// +/// The migration's SQL Server statements. Two keys that differ only in case can be one row here, because a +/// column's collation decides what counts as equal, so the statements compare keys under the collation the +/// schema gives them and the batch is de-duplicated the same way before it is offered to the table. +/// +class SqlServerMigrationSqlDialect : SqlServerDialect, IMigrationSqlDialect +{ + const int IgnoreCaseStyle = 1; + + OpenedKeys? OpenedState { get; set; } + + public async Task Open(ServiceControlDbContext dbContext, CancellationToken cancellationToken = default) + { + var sql = dbContext.GetService(); + var collations = new Dictionary<(string Table, string Column), string>(); + var comparers = new Dictionary<(Type EntityType, string Property), IEqualityComparer>(); + + foreach (var entityType in dbContext.Model.GetEntityTypes()) + { + if (entityType.GetTableName() is not { } tableName || entityType.FindPrimaryKey() is not { } key) + { + continue; + } + + var storeObject = StoreObjectIdentifier.Table(tableName, entityType.GetSchema()); + var table = sql.DelimitIdentifier(tableName, entityType.GetSchema()); + + foreach (var property in key.Properties.Where(property => property.ClrType == typeof(string))) + { + // A key property with no column of its own is refused by name when MigrationInsert builds the statement. + if (property.GetColumnName(storeObject) is not { } column) + { + continue; + } + + var collation = await dbContext.Database + .SqlQuery($""" + SELECT c.collation_name AS [Name], + CONVERT(int, COLLATIONPROPERTY(c.collation_name, 'ComparisonStyle')) AS [ComparisonStyle] + FROM sys.columns AS c + WHERE c.object_id = OBJECT_ID({table}) AND c.name = {column} AND c.collation_name IS NOT NULL + """) + .SingleOrDefaultAsync(cancellationToken) + ?? throw new InvalidOperationException($"The migration target found no collation for {tableName}.{column}, so the SQL Server schema is missing or not current. Run ServiceControl with --setup against this database first."); + + collations[(table, sql.DelimitIdentifier(column))] = collation.Name; + comparers[(entityType.ClrType, property.Name)] = (collation.ComparisonStyle & IgnoreCaseStyle) == 0 ? StringComparer.Ordinal : StringComparer.OrdinalIgnoreCase; + } + } + + OpenedState = new OpenedKeys(collations, comparers); + } + + public IEqualityComparer KeyComparer(Type entityType, string propertyName) => + Keys.Comparers.TryGetValue((entityType, propertyName), out var comparer) + ? comparer + : throw new ArgumentException($"{entityType.Name}.{propertyName} is not a string primary-key column, so it has no collation to compare under."); + + public int RowsPerStatement(int parametersPerRow) => MaxRowsPerStatement(parametersPerRow); + + public async Task> InsertMissing(ServiceControlDbContext dbContext, IReadOnlyList rows, CancellationToken cancellationToken = default) where TEntity : class + { + var insert = MigrationInsert.For(dbContext); + var columns = string.Join(", ", insert.Columns); + // The collation is an identifier sys.columns returned, and COLLATE takes no parameter. + var partitionBy = string.Join(", ", insert.KeyColumns.Select(column => + Keys.Collations.TryGetValue((insert.Table, column), out var collation) ? $"{column} COLLATE {collation}" : column)); + var inserted = new List(rows.Count); + + foreach (var chunk in rows.Chunk(RowsPerStatement(insert.Columns.Count))) + { + // HOLDLOCK, or two writers can both find a key missing and both insert it. + inserted.AddRange(await insert.Execute( + dbContext, + $""" + MERGE {insert.Table} WITH (HOLDLOCK) AS t + USING ( + SELECT {columns} + FROM ( + SELECT {columns}, ROW_NUMBER() OVER (PARTITION BY {partitionBy} ORDER BY [MigrationOrdinal]) AS [MigrationDuplicate] + FROM (VALUES + {ParameterRowsWithOrdinal(chunk.Length, insert.Columns.Count)} + ) AS v ({columns}, [MigrationOrdinal]) + ) AS deduplicated + WHERE [MigrationDuplicate] = 1 + ) AS s ({columns}) + ON {string.Join(" AND ", insert.KeyColumns.Select(column => $"t.{column} = s.{column}"))} + WHEN NOT MATCHED THEN INSERT ({columns}) VALUES ({string.Join(", ", insert.Columns.Select(column => $"s.{column}"))}) + OUTPUT {string.Join(", ", insert.KeyColumns.Select(column => $"inserted.{column}"))}; + """, + chunk, + cancellationToken)); + } + + return inserted; + } + + OpenedKeys Keys => OpenedState ?? throw new InvalidOperationException($"The SQL Server migration dialect is not open. Call {nameof(Open)} first."); + + // SqlServerDialect.ParameterRows numbers the parameters the same way, but has no room for the position each row arrived in, which the de-duplication orders by. + static string ParameterRowsWithOrdinal(int rowCount, int columnCount) => + string.Join(",\n", Enumerable.Range(0, rowCount).Select(row => + $"({string.Join(", ", Enumerable.Range(0, columnCount).Select(column => $"@p{(row * columnCount) + column}"))}, {row})")); + + sealed record OpenedKeys( + IReadOnlyDictionary<(string Table, string Column), string> Collations, + IReadOnlyDictionary<(Type EntityType, string Property), IEqualityComparer> Comparers); + + sealed record KeyColumnCollation(string Name, int ComparisonStyle); +} diff --git a/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerPersistence.cs b/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerPersistence.cs index dead62d3cc..b508cfd7d1 100644 --- a/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerPersistence.cs +++ b/src/ServiceControl.Persistence.EFCore.SqlServer/SqlServerPersistence.cs @@ -18,6 +18,7 @@ public void AddPersistence(IServiceCollection services) RegisterDataStores(services, settings); services.AddSingleton(); + services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); services.AddSingleton(); diff --git a/src/ServiceControl.Persistence.EFCore/Abstractions/BasePersistence.cs b/src/ServiceControl.Persistence.EFCore/Abstractions/BasePersistence.cs index 6b61a58b40..8d97a84d2c 100644 --- a/src/ServiceControl.Persistence.EFCore/Abstractions/BasePersistence.cs +++ b/src/ServiceControl.Persistence.EFCore/Abstractions/BasePersistence.cs @@ -7,6 +7,7 @@ namespace ServiceControl.Persistence.EFCore.Abstractions; using Particular.LicensingComponent.Persistence; using ServiceControl.CustomChecks; using ServiceControl.Operations.BodyStorage; +using ServiceControl.Persistence.EFCore.DataMigration; using ServiceControl.Persistence.EFCore.Implementation; using ServiceControl.Persistence.EFCore.Implementation.BodyStorage; using ServiceControl.Persistence.EFCore.Implementation.Recoverability; @@ -40,6 +41,10 @@ protected static void RegisterDataStores(IServiceCollection services, EFPersiste services.AddHostedService(p => p.GetRequiredService()); services.AddSingleton(); + services.AddSingleton(); + services.AddSingleton(); + services.AddHostedService(); + services.AddHostedService(); if (settings.RunRetentionSweep) { diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/BodyStorageIsWritableCheck.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/BodyStorageIsWritableCheck.cs new file mode 100644 index 0000000000..5a5c116bdf --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/BodyStorageIsWritableCheck.cs @@ -0,0 +1,35 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using System.Text; +using Infrastructure; +using ServiceControl.Persistence.DataMigration; + +/// +/// Writes, reads and deletes one probe body, so a body store that is misconfigured stops the copy before it +/// starts. Bodies can live on a file share, in Azure Blob Storage or in S3, and without this the first category +/// that carries bodies would find out part way through and skip every message it could not write. +/// +sealed class BodyStorageIsWritableCheck(IBodyStoragePersistence bodyStorage) : IMigrationStartupCheck +{ + const string ProbeBodyId = "migration-writable-probe"; + + public string Name => "message body storage is writable"; + + public async Task Run(CancellationToken cancellationToken = default) + { + await bodyStorage.WriteBody(ProbeBodyId, Encoding.UTF8.GetBytes(ProbeBodyId), "text/plain", cancellationToken); + + try + { + var probe = await bodyStorage.ReadBody(ProbeBodyId, cancellationToken) + ?? throw new Exception($"Message body storage accepted the probe body '{ProbeBodyId}' and then did not return it."); + + await probe.Stream.DisposeAsync(); + } + finally + { + // The probe is not migrated data, so it goes even when the read fails: a probe left behind is a body the source never had. + await bodyStorage.DeleteBodyIfExists(ProbeBodyId, cancellationToken); + } + } +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/CheckpointTableIsReadable.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/CheckpointTableIsReadable.cs new file mode 100644 index 0000000000..6b0d063422 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/CheckpointTableIsReadable.cs @@ -0,0 +1,40 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using Microsoft.Extensions.Hosting; +using ServiceControl.Persistence.DataMigration; + +/// +/// Refuses the start when the migration checkpoint table cannot be read, which means the database was upgraded to this build without --setup. +/// +sealed class CheckpointTableIsReadable(IMigrationCheckpointStore checkpointStore) : IHostedLifecycleService +{ + // StartingAsync, so the refusal lands before the web server binds. RecordHostOpenedOnTarget reads the same + // table for the stamp, but it has to wait until StartedAsync to see what the copy wrote, and by then a + // Windows service has already told the Service Control Manager it is running. + public async Task StartingAsync(CancellationToken cancellationToken = default) + { + try + { + await checkpointStore.ReadAll(cancellationToken); + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + catch (Exception ex) + { + throw new InvalidOperationException( + "ServiceControl could not read its migration checkpoint table, so this database's schema is older than this build. Run ServiceControl with --setup against it before starting, whether or not you intend to migrate.", ex); + } + } + + public Task StartAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StartedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppingAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StopAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/EFCoreMigrationTarget.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/EFCoreMigrationTarget.cs new file mode 100644 index 0000000000..15672c4124 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/EFCoreMigrationTarget.cs @@ -0,0 +1,107 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using System.Collections.Frozen; +using Implementation; +using Infrastructure; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Logging; +using ServiceControl.Persistence.DataMigration; +using Writers; + +/// +/// Writes a migration into SQL Server or PostgreSQL. It holds one writer per category and knows nothing about +/// any of them beyond that, so a category this build cannot write is simply absent from +/// and never reaches the engine. +/// +sealed class EFCoreMigrationTarget(IServiceScopeFactory scopeFactory, IMigrationSqlDialect migrationDialect, ILogger logger) : DataStoreBase(scopeFactory), IMigrationTarget +{ + readonly FrozenDictionary writers = new IMigrationCategoryWriter[] + { + new KnownEndpointsWriter(migrationDialect), + new EndpointSettingsWriter(migrationDialect) + }.ToFrozenDictionary(writer => writer.CategoryId, StringComparer.Ordinal); + + public Task Open(CancellationToken cancellationToken = default) => + ExecuteWithDbContext((dbContext, token) => migrationDialect.Open(dbContext, token), cancellationToken); + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => + ExecuteWithDbContext((dbContext, _) => Task.FromResult(WriterFor(category).BatchSize(dbContext)), cancellationToken); + + public Task Write( + MigrationCategory category, + MigrationBatch batch, + MigrationCheckpoint checkpointToExtend, + CancellationToken cancellationToken = default) => + ExecuteWithDbContext(async (dbContext, writeToken) => + { + var prepared = await WriterFor(category).Prepare(dbContext, batch, writeToken); + + AccountForEveryRow(category, batch, prepared); + + var skipReasons = prepared.Skips.Count == 0 ? null : prepared.Skips.GroupBy(skip => skip.Reason).ToDictionary(group => group.Key, group => group.LongCount()); + + var (copied, alreadyPresent, saved) = await dbContext.Database.CreateExecutionStrategy().ExecuteAsync(async token => + { + // This block is retried, and an attempt that failed leaves the checkpoint it added still tracked, so the next attempt would insert it a second time. + dbContext.ChangeTracker.Clear(); + + await using var transaction = await dbContext.Database.BeginTransactionAsync(token); + + var inserted = await prepared.Insert(dbContext, token); + var present = AlreadyPresentIn(category, batch, inserted, prepared.Skips.Count); + var stored = await dbContext.UpsertCheckpoint( + checkpointToExtend.Extend(inserted, prepared.Skips.Count, present, skipReasons), + token); + await transaction.CommitAsync(token); + + return (inserted, present, stored); + }, writeToken); + + foreach (var (sourceId, reason, detail) in prepared.Skips) + { + logger.LogWarning("Skipped {SourceId} in category {CategoryId} as {SkipReason}: {Detail}", sourceId, category.Id, reason, detail); + } + + return new MigrationWriteResult( + saved, + copied, + prepared.Skips.Count, + [.. prepared.Skips.Select(skip => skip.SourceId)], + AlreadyPresent: alreadyPresent, + SkipReasons: skipReasons, + BenignSkipped: prepared.BenignSkipCount); + }, cancellationToken); + + // Checked before the insert, because the subtraction below can only catch a writer that over-reports. + internal static void AccountForEveryRow(MigrationCategory category, MigrationBatch batch, PreparedBatch prepared) + { + if (prepared.PreparedRowCount + prepared.Skips.Count != batch.Rows.Count) + { + throw new InvalidOperationException($"The {category.Id} writer prepared {prepared.PreparedRowCount} rows and skipped {prepared.Skips.Count} of the {batch.Rows.Count} rows in the batch. Every row must be one or the other, or the rows it dropped would be counted as rows the target already held."); + } + } + + // Without the throw, a miscount would quietly shrink the halt threshold's denominator instead of failing. + internal static int AlreadyPresentIn(MigrationCategory category, MigrationBatch batch, int copied, int skipped) + { + var alreadyPresent = batch.Rows.Count - copied - skipped; + + if (alreadyPresent < 0) + { + throw new InvalidOperationException($"The target copied {copied} and skipped {skipped} of the {batch.Rows.Count} rows in category {category.Id}, leaving {alreadyPresent} already present. Every row is copied, skipped or already present."); + } + + return alreadyPresent; + } + + public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => + ExecuteWithDbContext((dbContext, token) => WriterFor(category).Count(dbContext, token), cancellationToken); + + public IReadOnlyCollection SupportedCategoryIds => writers.Keys; + + IMigrationCategoryWriter WriterFor(MigrationCategory category) => + writers.TryGetValue(category.Id, out var writer) + ? writer + : throw new NotSupportedException($"The migration target cannot yet write the '{category.Id}' category."); +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/EFCoreMigrationTargetReadiness.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/EFCoreMigrationTargetReadiness.cs new file mode 100644 index 0000000000..fde62f5d2c --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/EFCoreMigrationTargetReadiness.cs @@ -0,0 +1,41 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using Abstractions; +using Implementation; +using Infrastructure; +using Microsoft.Extensions.DependencyInjection; +using ServiceControl.Persistence.DataMigration; + +/// +/// What a SQL Server or PostgreSQL database has to be before a copy writes to it, and the stamp saying a host +/// has opened on it since. The stamp is a row in the settings table, so it survives a restart and belongs to +/// the database rather than to the instance. +/// +public class EFCoreMigrationTargetReadiness( + IServiceScopeFactory scopeFactory, + IBodyStoragePersistence bodyStorage, + EFPersisterSettings settings, + TimeProvider timeProvider) : DataStoreBase(scopeFactory), IMigrationTargetReadiness +{ + // Cheapest first: the setting is already in memory, the schema costs one query, and the body store costs a + // round trip to a file share or a cloud service. + public IReadOnlyList ContributedChecks() => + [ + new RetryHistoryDepthIsSafeCheck(settings.RetryHistoryDepth), + new SchemaIsCurrentCheck(scopeFactory), + new BodyStorageIsWritableCheck(bodyStorage) + ]; + + public Task RecordHostOpened(CancellationToken cancellationToken = default) => + ExecuteWithDbContext(async (dbContext, token) => + { + if (await dbContext.GetSetting(SettingKeys.MigrationHostOpenedOnTarget, token) is null) + { + await dbContext.StoreSetting(SettingKeys.MigrationHostOpenedOnTarget, timeProvider.GetUtcNow().UtcDateTime, token); + } + }, cancellationToken); + + public Task HasHostOpened(CancellationToken cancellationToken = default) => + ExecuteWithDbContext(async (dbContext, token) => + await dbContext.GetSetting(SettingKeys.MigrationHostOpenedOnTarget, token) is not null, cancellationToken); +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/IMigrationCategoryWriter.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/IMigrationCategoryWriter.cs new file mode 100644 index 0000000000..bbc9f0ce25 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/IMigrationCategoryWriter.cs @@ -0,0 +1,32 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using DbContexts; +using ServiceControl.Persistence.DataMigration; + +/// +/// Everything the target needs to know about one category: how big a batch is, how its documents +/// become rows, and how many rows the target holds for it. +/// +interface IMigrationCategoryWriter +{ + /// The category this writer handles, named as in . + string CategoryId { get; } + + /// + /// The most rows the target will take from the source in one batch. It comes from how many values each row + /// carries and how many parameters one statement can hold, so a wider table takes fewer rows. The context is read for its model only. + /// + int BatchSize(ServiceControlDbContext dbContext); + + /// + /// Turns one batch of source documents into the rows to insert and the rows to skip, writing nothing: the insert it returns runs later, inside the target's transaction. + /// Every batch row produces exactly one row or exactly one skip, never both and never neither, because the target works out how many were already there by subtracting both from the batch size. + /// + Task Prepare(ServiceControlDbContext dbContext, MigrationBatch batch, CancellationToken cancellationToken = default); + + /// + /// How many rows the target holds for this category. It counts only this category, even where two of them + /// share a table. + /// + Task Count(ServiceControlDbContext dbContext, CancellationToken cancellationToken = default); +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/PreparedBatch.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/PreparedBatch.cs new file mode 100644 index 0000000000..66a5c49598 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/PreparedBatch.cs @@ -0,0 +1,21 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using DbContexts; +using ServiceControl.Persistence.DataMigration; + +/// +/// One batch turned into rows, ready for the target to insert inside its own transaction. Nothing here has +/// touched the database yet. +/// +/// Runs the insert and returns how many rows it added. The target calls it inside the transaction that also saves the checkpoint. +/// How many rows the writer built, which is more than the insert adds when the table already holds some of their keys. +/// One entry per row the writer will not insert, with the reason and a detail line for the log. +/// False once an earlier category dropped rows, because a skip here may then be this migration's own doing rather than something the source never had. +sealed record PreparedBatch( + Func> Insert, + int PreparedRowCount, + IReadOnlyList<(string SourceId, MigrationSkipReason Reason, string Detail)> Skips, + bool SkipsReflectTheSource = true) +{ + public int BenignSkipCount => SkipsReflectTheSource ? Skips.Count(skip => skip.Reason.IsBenign()) : 0; +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/RecordHostOpenedOnTarget.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/RecordHostOpenedOnTarget.cs new file mode 100644 index 0000000000..b422105903 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/RecordHostOpenedOnTarget.cs @@ -0,0 +1,32 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using Microsoft.Extensions.Hosting; +using ServiceControl.Persistence.DataMigration; + +/// +/// Calls when the host starts, on a database a migration has already written to. +/// +sealed class RecordHostOpenedOnTarget(IMigrationTargetReadiness readiness, IMigrationCheckpointStore checkpointStore) : IHostedLifecycleService +{ + // StartedAsync, not StartAsync: the web server binds its port during StartAsync, and a start that dies + // there served nothing. Every command that runs a host reaches it, not only RunCommand. + public async Task StartedAsync(CancellationToken cancellationToken = default) + { + // Only a database a migration has written to gets the stamp, so the marker answers whether a host + // has opened since the copy began rather than whether one ever ran on this database. + if ((await checkpointStore.ReadAll(cancellationToken)).Count > 0) + { + await readiness.RecordHostOpened(cancellationToken); + } + } + + public Task StartingAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StartAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppingAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StopAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/RetryHistoryDepthIsSafeCheck.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/RetryHistoryDepthIsSafeCheck.cs new file mode 100644 index 0000000000..2f6f39b7c7 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/RetryHistoryDepthIsSafeCheck.cs @@ -0,0 +1,24 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using ServiceControl.Persistence.DataMigration; + +/// +/// Refuses a copy that the instance's own retry history setting would throw away. At a depth of zero the +/// persister empties the history table whenever a retry completes, and the migrated rows would go with it. +/// +sealed class RetryHistoryDepthIsSafeCheck(int retryHistoryDepth) : IMigrationStartupCheck +{ + public string Name => "the retry history depth will not empty a migrated table"; + + public Task Run(CancellationToken cancellationToken = default) + { + // The depth at which RetryHistoryDataStore.TrimHistory deletes every row rather than trimming. + if (retryHistoryDepth <= 0) + { + throw new Exception( + $"ServiceControl/RetryHistoryDepth is {retryHistoryDepth}, and at that depth this persister deletes every row of HistoricRetryOperations when a retry completes. The RetryOperations category copies that table, so the first completed retry after cutover would silently discard the migrated retry history. Set ServiceControl/RetryHistoryDepth to a positive number, or clear it to use the default of 10, before setting {MigrationSettings.EnabledKey}."); + } + + return Task.CompletedTask; + } +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/SchemaIsCurrentCheck.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/SchemaIsCurrentCheck.cs new file mode 100644 index 0000000000..24f64d1de2 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/SchemaIsCurrentCheck.cs @@ -0,0 +1,26 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration; + +using Implementation; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.DependencyInjection; +using ServiceControl.Persistence.DataMigration; + +/// +/// Refuses a copy into a database whose schema is behind this build. The writers insert into tables by name, so +/// a missing schema migration turns into a failure part way through the first category instead of a refusal. +/// +sealed class SchemaIsCurrentCheck(IServiceScopeFactory scopeFactory) : DataStoreBase(scopeFactory), IMigrationStartupCheck +{ + public string Name => "the target schema is current"; + + public Task Run(CancellationToken cancellationToken = default) => + ExecuteWithDbContext(async (dbContext, token) => + { + string[] pending = [.. await dbContext.Database.GetPendingMigrationsAsync(token)]; + + if (pending.Length > 0) + { + throw new Exception($"The database is missing {pending.Length} schema migration(s): {string.Join(", ", pending)}. Run ServiceControl with --setup before starting a migration."); + } + }, cancellationToken); +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/Writers/EndpointSettingsWriter.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/Writers/EndpointSettingsWriter.cs new file mode 100644 index 0000000000..ab67ab4f29 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/Writers/EndpointSettingsWriter.cs @@ -0,0 +1,67 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration.Writers; + +using DbContexts; +using Entities; +using Infrastructure; +using Microsoft.EntityFrameworkCore; +using ServiceControl.Persistence.DataMigration; + +/// +/// Writes the per-endpoint settings, and only for endpoints the target already knows. A setting for an unknown +/// endpoint is dropped by the heartbeat sync soon after the instance starts, so copying it would be work the +/// product undoes. That judgment holds only while the endpoints themselves copied cleanly, which is why this +/// writer reads the KnownEndpoints checkpoint before it decides a skip was no loss. +/// +sealed class EndpointSettingsWriter(IMigrationSqlDialect migrationDialect) : IMigrationCategoryWriter +{ + public string CategoryId => MigrationCategoryIds.EndpointSettings; + + public int BatchSize(ServiceControlDbContext dbContext) => migrationDialect.RowsPerStatement(MigrationInsert.For(dbContext).Columns.Count); + + public Task Count(ServiceControlDbContext dbContext, CancellationToken cancellationToken = default) => + dbContext.EndpointSettings.LongCountAsync(cancellationToken); + + public async Task Prepare(ServiceControlDbContext dbContext, MigrationBatch batch, CancellationToken cancellationToken = default) + { + // Only the names this batch asks about: the whole table is a scan per batch, and a large instance + // has as many settings as endpoints, so the cost grows with the square of the endpoint count. + string[] batchNames = [.. batch.Rows.Select(row => ((EndpointSettings)row.Document).Name).Distinct(StringComparer.Ordinal)]; + + // Ordinal, because HeartbeatEndpointSettingsSyncHostedService matches names in a default HashSet. + var knownNames = (await dbContext.KnownEndpoints.AsNoTracking() + .Select(endpoint => endpoint.Name) + .Where(name => batchNames.Contains(name)) + .ToListAsync(cancellationToken)).ToHashSet(StringComparer.Ordinal); + + // The names come from the target's own table, so an endpoint the KnownEndpoints copy dropped looks the same here as one the source never had. + var knownEndpointsSkipped = await dbContext.MigrationCheckpoints + .AsNoTracking() + .Where(checkpoint => checkpoint.CategoryId == MigrationCategoryIds.KnownEndpoints) + .Select(checkpoint => checkpoint.SkippedCount) + .SingleOrDefaultAsync(cancellationToken); + + var rows = new List(batch.Rows.Count); + var skips = new List<(string SourceId, MigrationSkipReason Reason, string Detail)>(); + + foreach (var row in batch.Rows) + { + var settings = (EndpointSettings)row.Document; + + // The empty name is the row holding the default for every endpoint, which the sync keeps whatever endpoints are known. + if (settings.Name != string.Empty && !knownNames.Contains(settings.Name)) + { + skips.Add((row.SourceId, MigrationSkipReason.EndpointNotKnown, $"no known endpoint is named '{settings.Name}', so the heartbeat settings sync would delete this setting")); + continue; + } + + rows.Add(new EndpointSettingsEntity { Name = settings.Name, TrackInstances = settings.TrackInstances }); + } + + // The source yields document-id order, and the statement keeps the first of any keys the database treats as one. + return new PreparedBatch( + async (context, token) => (await migrationDialect.InsertMissing(context, rows, token)).Count, + rows.Count, + skips, + SkipsReflectTheSource: knownEndpointsSkipped == 0); + } +} diff --git a/src/ServiceControl.Persistence.EFCore/DataMigration/Writers/KnownEndpointsWriter.cs b/src/ServiceControl.Persistence.EFCore/DataMigration/Writers/KnownEndpointsWriter.cs new file mode 100644 index 0000000000..67e8095494 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/DataMigration/Writers/KnownEndpointsWriter.cs @@ -0,0 +1,53 @@ +namespace ServiceControl.Persistence.EFCore.DataMigration.Writers; + +using DbContexts; +using Entities; +using Infrastructure; +using Microsoft.EntityFrameworkCore; +using ServiceControl.Persistence.DataMigration; + +/// +/// Writes the endpoints ServiceControl has heard from. The row's key is worked out from the endpoint's own +/// details rather than carried over, so the same endpoint lands on the same row whichever side wrote it first. +/// +sealed class KnownEndpointsWriter(IMigrationSqlDialect migrationDialect) : IMigrationCategoryWriter +{ + public string CategoryId => MigrationCategoryIds.KnownEndpoints; + + public int BatchSize(ServiceControlDbContext dbContext) => migrationDialect.RowsPerStatement(MigrationInsert.For(dbContext).Columns.Count); + + public Task Count(ServiceControlDbContext dbContext, CancellationToken cancellationToken = default) => + dbContext.KnownEndpoints.LongCountAsync(cancellationToken); + + public Task Prepare(ServiceControlDbContext dbContext, MigrationBatch batch, CancellationToken cancellationToken = default) + { + var rows = new List(batch.Rows.Count); + var skips = new List<(string SourceId, MigrationSkipReason Reason, string Detail)>(); + + foreach (var row in batch.Rows) + { + var endpoint = (KnownEndpoint)row.Document; + + if (endpoint.EndpointDetails?.Name is null || endpoint.EndpointDetails.Host is null) + { + var column = endpoint.EndpointDetails?.Name is null ? nameof(KnownEndpointEntity.Name) : nameof(KnownEndpointEntity.Host); + skips.Add((row.SourceId, MigrationSkipReason.RequiredValueMissing, $"KnownEndpoints.{column} is NOT NULL and the document has no value for it")); + continue; + } + + rows.Add(new KnownEndpointEntity + { + Id = endpoint.EndpointDetails.GetDeterministicId(), + Name = endpoint.EndpointDetails.Name, + HostId = endpoint.EndpointDetails.HostId, + Host = endpoint.EndpointDetails.Host, + Monitored = endpoint.Monitored + }); + } + + return Task.FromResult(new PreparedBatch( + async (context, token) => (await migrationDialect.InsertMissing(context, rows, token)).Count, + rows.Count, + skips)); + } +} diff --git a/src/ServiceControl.Persistence.EFCore/Infrastructure/IMigrationSqlDialect.cs b/src/ServiceControl.Persistence.EFCore/Infrastructure/IMigrationSqlDialect.cs new file mode 100644 index 0000000000..864562c443 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/Infrastructure/IMigrationSqlDialect.cs @@ -0,0 +1,31 @@ +namespace ServiceControl.Persistence.EFCore.Infrastructure; + +using ServiceControl.Persistence.EFCore.DbContexts; + +/// +/// The migration's provider-specific SQL. Statements run inside the caller's transaction and insert only what is absent. +/// +public interface IMigrationSqlDialect +{ + /// + /// Reads, once, whatever the statements need from the schema. On SQL Server that is the collation of every + /// string column of every primary key, composite keys included. + /// + Task Open(ServiceControlDbContext dbContext, CancellationToken cancellationToken = default); + + /// + /// How the database compares one string key column, so a caller can work out which of its own rows the insert will treat as one. The insert compares in the database itself; this only predicts what it will find. + /// Call first, and name a property that is a string column of the entity's primary key. + /// + IEqualityComparer KeyComparer(Type entityType, string propertyName); + + /// + /// How many rows the provider will take in one statement, for rows carrying this many values each. + /// + int RowsPerStatement(int parametersPerRow); + + /// + /// Inserts the rows whose key the table does not hold, merges any whose keys the database treats as one, and returns the rows it inserted. + /// + Task> InsertMissing(ServiceControlDbContext dbContext, IReadOnlyList rows, CancellationToken cancellationToken = default) where TEntity : class; +} diff --git a/src/ServiceControl.Persistence.EFCore/Infrastructure/MigrationInsert.cs b/src/ServiceControl.Persistence.EFCore/Infrastructure/MigrationInsert.cs new file mode 100644 index 0000000000..37ac476786 --- /dev/null +++ b/src/ServiceControl.Persistence.EFCore/Infrastructure/MigrationInsert.cs @@ -0,0 +1,141 @@ +namespace ServiceControl.Persistence.EFCore.Infrastructure; + +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Infrastructure; +using Microsoft.EntityFrameworkCore.Metadata; +using Microsoft.EntityFrameworkCore.Storage; +using ServiceControl.Persistence.EFCore.DbContexts; + +/// +/// The table, columns and primary key maps to, and the run of one provider's insert statement over a chunk of those rows. +/// +public sealed class MigrationInsert where TEntity : class +{ + static readonly ExactKey KeyEquality = new(); + + readonly IReadOnlyList properties; + readonly IReadOnlyList keyProperties; + + MigrationInsert(string table, IReadOnlyList properties, IReadOnlyList keyProperties, Func columnName) + { + Table = table; + this.properties = properties; + this.keyProperties = keyProperties; + Columns = [.. properties.Select(columnName)]; + KeyColumns = [.. keyProperties.Select(columnName)]; + } + + /// The table name, quoted for this provider and carrying its schema. + public string Table { get; } + + /// Every column the insert writes, quoted, in the order the parameters of one row follow. + public IReadOnlyList Columns { get; } + + /// The primary key columns, quoted, in the order the statement has to return them. + public IReadOnlyList KeyColumns { get; } + + /// + /// Reads the table, columns and key out of the EF Core model for this entity. + /// + /// The entity is not mapped, has no primary key, or has a column this insert cannot write, such as one the database generates. Rows like that go through EF in the caller's context instead. + public static MigrationInsert For(ServiceControlDbContext dbContext) + { + var entityType = dbContext.Model.FindEntityType(typeof(TEntity)) + ?? throw new InvalidOperationException($"{typeof(TEntity).Name} is not an entity in the EF Core model, so the migration cannot tell which table it lands in."); + var tableName = entityType.GetTableName() + ?? throw new InvalidOperationException($"{typeof(TEntity).Name} is not mapped to a table."); + var key = entityType.FindPrimaryKey() + ?? throw new InvalidOperationException($"{typeof(TEntity).Name} has no primary key, so there is nothing for a row to be absent by."); + IProperty[] properties = [.. entityType.GetProperties()]; + + if (properties.Any(property => property.ValueGenerated != ValueGenerated.Never) + || entityType.GetComplexProperties().Any() + || entityType.GetNavigations().Any(navigation => navigation.TargetEntityType.IsOwned())) + { + throw new InvalidOperationException($"{typeof(TEntity).Name} has a store-generated, complex or owned property, whose columns InsertMissing does not write. Add these rows through EF in the caller's context instead."); + } + + var storeObject = StoreObjectIdentifier.Table(tableName, entityType.GetSchema()); + var sql = dbContext.GetService(); + + return new MigrationInsert( + sql.DelimitIdentifier(tableName, entityType.GetSchema()), + properties, + key.Properties, + property => sql.DelimitIdentifier(property.GetColumnName(storeObject) + ?? throw new InvalidOperationException($"{typeof(TEntity).Name}.{property.Name} has no column in {tableName}."))); + } + + /// + /// Runs one insert statement over a chunk of rows and returns the rows of that chunk it inserted. + /// The statement reads parameters named @p0 upward, row after row in order, and must return the of each row it inserted, in that order. + /// It runs on the caller's open transaction, and throws when there is none or when a returned key matches no row of the chunk. + /// + public async Task> Execute(ServiceControlDbContext dbContext, string sql, TEntity[] chunk, CancellationToken cancellationToken = default) + { + await using var command = dbContext.Database.GetDbConnection().CreateCommand(); + command.Transaction = (dbContext.Database.CurrentTransaction + ?? throw new InvalidOperationException("InsertMissing must run inside the caller's transaction, so its rows commit with the checkpoint.")).GetDbTransaction(); + command.CommandText = sql; + // A raw command starts at the provider's own 30 seconds, where SaveChangesAsync took the configured value. + command.CommandTimeout = dbContext.Database.GetCommandTimeout() ?? command.CommandTimeout; + + var index = 0; + foreach (var row in chunk) + { + foreach (var property in properties) + { + // The mapping applies the value converter and the provider type, as EF's own inserts do. + command.Parameters.Add(property.GetRelationalTypeMapping().CreateParameter(command, $"@p{index++}", property.GetGetter().GetClrValue(row), property.IsNullable)); + } + } + + var rowsByKey = new Dictionary(KeyEquality); + foreach (var row in chunk) + { + // A key repeated exactly is one row to every database, so the first row given carries it. + rowsByKey.TryAdd(KeyOf(row), row); + } + + var inserted = new List(chunk.Length); + + await using var reader = await command.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + object?[] key = [.. keyProperties.Select((property, column) => FromProvider(property, reader.GetValue(column)))]; + + // A key can come back as a different CLR type than it went in as, a date column read back as + // DateTime say, and a plain lookup would report that as a bare KeyNotFoundException. + inserted.Add(rowsByKey.TryGetValue(key, out var row) + ? row + : throw new InvalidOperationException($"The {Table} insert returned the key [{string.Join(", ", key.Select(Describe))}], which matches no row in the batch that produced it. Compare it against the key types the batch carried: [{string.Join(", ", keyProperties.Select(property => property.ClrType.Name))}].")); + } + + return inserted; + } + + object?[] KeyOf(TEntity row) => [.. keyProperties.Select(property => property.GetGetter().GetClrValue(row))]; + + static string Describe(object? value) => value is null ? "null" : $"{value} ({value.GetType().Name})"; + + static object? FromProvider(IProperty property, object value) => + property.GetRelationalTypeMapping().Converter is { } converter ? converter.ConvertFromProvider(value) : value; + + // Exact, because the database decided which keys were one before the statement ran, and OUTPUT and RETURNING echo the value inserted. + sealed class ExactKey : IEqualityComparer + { + public bool Equals(object?[]? x, object?[]? y) => x!.SequenceEqual(y!); + + public int GetHashCode(object?[] key) + { + var hash = new HashCode(); + + foreach (var value in key) + { + hash.Add(value); + } + + return hash.ToHashCode(); + } + } +} diff --git a/src/ServiceControl.Persistence.EFCore/Infrastructure/SettingKeys.cs b/src/ServiceControl.Persistence.EFCore/Infrastructure/SettingKeys.cs index 7a8e403170..7da4cc4bfe 100644 --- a/src/ServiceControl.Persistence.EFCore/Infrastructure/SettingKeys.cs +++ b/src/ServiceControl.Persistence.EFCore/Infrastructure/SettingKeys.cs @@ -9,4 +9,7 @@ static class SettingKeys public const string ReportMasks = "ReportMasks"; public const string LicensedEndpointDetails = "LicensedEndpointDetails"; public const string NotificationEmails = "NotificationEmails"; + // Written the first time a host starts on a database a migration has already written to. Once it is set, + // going back to RavenDB loses everything ServiceControl has written here since. + public const string MigrationHostOpenedOnTarget = "Migration/HostOpenedOnTarget"; } diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/IMigrationCategoryReader.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/IMigrationCategoryReader.cs new file mode 100644 index 0000000000..3e765d2a82 --- /dev/null +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/IMigrationCategoryReader.cs @@ -0,0 +1,19 @@ +#nullable enable + +namespace ServiceControl.Persistence.RavenDB.DataMigration; + +using System.Collections.Generic; +using System.Threading; +using ServiceControl.Persistence.DataMigration; + +/// +/// Which documents one category is made of, and how they become rows. Read keeps the contract +/// sets out, for this one category. +/// +interface IMigrationCategoryReader +{ + /// The category this reader handles, named as in . + string CategoryId { get; } + + IAsyncEnumerable Read(string? resumeAfter, int batchSize, CancellationToken cancellationToken = default); +} diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenDataVersion.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenDataVersion.cs new file mode 100644 index 0000000000..4b7f708c96 --- /dev/null +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenDataVersion.cs @@ -0,0 +1,25 @@ +namespace ServiceControl.Persistence.RavenDB.DataMigration; + +using System; +using System.Reflection; + +/// +/// The document that records which ServiceControl build last wrote a RavenDB database. A migration reads it to +/// decide whether this build reads those documents the same way, which nothing else in the database says. +/// +public class RavenDataVersion +{ + public const string DocumentId = "ServiceControl/DataVersion"; + + // The ServiceControl release version, put on this assembly by MinVer. The "+hash" suffix comes off so that a + // rebuild of the same release does not read as a new data version. + public static string Current { get; } = + typeof(RavenDataVersion).Assembly.GetCustomAttribute()?.InformationalVersion.Split('+')[0] + ?? typeof(RavenDataVersion).Assembly.GetName().Version!.ToString(3); + + /// The release that stamped the database. It can be null on a document written before this property existed, because RavenDB deserializes without honoring the required modifier. + public required string Version { get; set; } + + /// When the stamp was last raised, in UTC. + public DateTime StampedAt { get; set; } +} diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenDocumentStream.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenDocumentStream.cs new file mode 100644 index 0000000000..892641c1f5 --- /dev/null +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenDocumentStream.cs @@ -0,0 +1,82 @@ +#nullable enable + +namespace ServiceControl.Persistence.RavenDB.DataMigration; + +using System; +using System.Collections.Generic; +using System.Collections.ObjectModel; +using System.Runtime.CompilerServices; +using System.Threading; +using Raven.Client.Documents.Commands; +using ServiceControl.Persistence.DataMigration; + +/// +/// The one way every category reads RavenDB: stream the documents whose id starts with a prefix, in id order, +/// and hand them back in batches. Id order is what makes the cursor work, because a resume asks RavenDB to +/// start after an id rather than to skip a count. +/// +static class RavenDocumentStream +{ + /// + /// Streams one collection and yields it in batches of at most . + /// + /// The document id prefix that selects the collection, such as "KnownEndpoints/". + /// The cursor a previous run saved, or null to start at the beginning. + /// Turns one document into a row, or into null for a document this category does not copy. The cursor still moves past a document that becomes null. + /// The source holds no document with the id in , which means the cursor and the database no longer belong together. + public static async IAsyncEnumerable ByPrefix( + RavenReadOnlySourceLifecycle lifecycle, + string categoryId, + string databaseName, + string prefix, + string? resumeAfter, + int batchSize, + Func, MigrationRow?> project, + [EnumeratorCancellation] CancellationToken cancellationToken = default) where TDocument : class + { + using var session = lifecycle.OpenSession(databaseName); + + // RavenDB starts a stream after whatever id it is given, even one that does not exist, so a stale cursor would skip rows nothing has copied. + if (resumeAfter is not null && !await session.Advanced.ExistsAsync(resumeAfter, cancellationToken)) + { + throw new InvalidOperationException( + $"The migration cannot resume the '{categoryId}' category after '{resumeAfter}', because the RavenDB database '{databaseName}' holds no document with that id. The cursor is saved in the target database and names a source document, so restoring RavenDB from a backup or changing which database it reads separates the two. Point the migration back at the RavenDB database this copy started from and restart."); + } + + await using var enumerator = await session.Advanced.StreamAsync( + prefix, startAfter: resumeAfter, token: cancellationToken); + + var rows = new List(batchSize); + var cursor = resumeAfter; + var lastYielded = resumeAfter; + + while (await enumerator.MoveNextAsync()) + { + // The cursor moves on every document the stream hands back, even one that becomes no row, so a resume never walks it again. + cursor = enumerator.Current.Id; + + if (project(enumerator.Current) is { } row) + { + rows.Add(row); + } + + if (rows.Count == batchSize) + { + yield return new MigrationBatch(rows, cursor); + rows = new List(batchSize); + lastYielded = cursor; + } + } + + // A tail whose documents all became no row still goes out, empty, so that the cursor past them is saved. + if (rows.Count > 0 || cursor != lastYielded) + { + yield return new MigrationBatch(rows, cursor!); + } + } + + public static MigrationRow WholeDocument(StreamResult result) where TDocument : class => + new(result.Id, result.Document, NoMetadata); + + static readonly IReadOnlyDictionary NoMetadata = ReadOnlyDictionary.Empty; +} diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenMigrationSource.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenMigrationSource.cs index cc3c59a998..37d458a7c4 100644 --- a/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenMigrationSource.cs +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenMigrationSource.cs @@ -3,6 +3,7 @@ namespace ServiceControl.Persistence.RavenDB.DataMigration; using System; +using System.Collections.Frozen; using System.Collections.Generic; using System.Linq; using System.Threading; @@ -11,11 +12,25 @@ namespace ServiceControl.Persistence.RavenDB.DataMigration; using Raven.Client.Documents.Operations; using Raven.Client.ServerWide.Operations; using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.RavenDB.DataMigration.Readers; +/// +/// Reads a migration out of RavenDB. It holds one reader per category and knows nothing about any of them +/// beyond that, so a category this build cannot read is simply absent from +/// and never reaches the engine. +/// sealed class RavenMigrationSource(RavenReadOnlySourceLifecycle lifecycle) : IMigrationSource { + readonly FrozenDictionary readers = new IMigrationCategoryReader[] + { + new KnownEndpointsReader(lifecycle), + new EndpointSettingsReader(lifecycle) + }.ToFrozenDictionary(reader => reader.CategoryId, StringComparer.Ordinal); + public Task Open(CancellationToken cancellationToken = default) => lifecycle.Open(cancellationToken); + public IReadOnlyList ContributedChecks() => [new SourceDataVersionIsReadableCheck(lifecycle)]; + public async Task Describe(CancellationToken cancellationToken = default) { var settings = lifecycle.Settings; @@ -47,18 +62,36 @@ public async Task> Inventory(Cancel return entries; } - public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => - throw new NotSupportedException($"The RavenDB migration source cannot count category {category.Id} yet"); + // Counted by streaming the same documents Read walks, not from RavenDB's collection statistics: a total that + // counted anything Read leaves out would halt the category for a shortfall that never happened. + public async Task Count(MigrationCategory category, CancellationToken cancellationToken = default) + { + var total = 0L; + + await foreach (var batch in Read(category, resumeAfter: null, batchSize: CountBatchSize, cancellationToken)) + { + total += batch.Rows.Count; + } + + return total; + } public IAsyncEnumerable Read( MigrationCategory category, string? resumeAfter, int batchSize, CancellationToken cancellationToken = default) => - throw new NotSupportedException($"The RavenDB migration source cannot read category {category.Id} yet"); + readers.TryGetValue(category.Id, out var reader) + ? reader.Read(resumeAfter, batchSize, cancellationToken) + : throw new NotSupportedException($"The migration source cannot yet read the '{category.Id}' category."); public Task ReadBody(MigrationCategory category, string sourceId, CancellationToken cancellationToken = default) => throw new NotSupportedException($"The RavenDB migration source cannot read bodies for category {category.Id} yet"); + public IReadOnlyCollection SupportedCategoryIds => readers.Keys; + public ValueTask DisposeAsync() => lifecycle.DisposeAsync(); + + // Counting only adds up row counts, so this size changes nothing but how often the stream stops to hand one back. + const int CountBatchSize = 1024; } diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenReadOnlySourceLifecycle.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenReadOnlySourceLifecycle.cs index 3b6f32ff55..901c76f6ac 100644 --- a/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenReadOnlySourceLifecycle.cs +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/RavenReadOnlySourceLifecycle.cs @@ -5,6 +5,7 @@ namespace ServiceControl.Persistence.RavenDB.DataMigration; using System; using System.Diagnostics; using System.Net.Http; +using System.Security.Cryptography.X509Certificates; using System.Threading; using System.Threading.Tasks; using Microsoft.Extensions.Hosting; @@ -18,6 +19,11 @@ namespace ServiceControl.Persistence.RavenDB.DataMigration; using ServiceControl.Configuration; using ServiceControl.RavenDB; +/// +/// The connection to the RavenDB database a migration reads from, and the embedded server it starts when that +/// database is a data directory rather than a URL. Nothing here writes, so the old database stays exactly as it +/// was and the copy can still be thrown away. +/// sealed class RavenReadOnlySourceLifecycle(RavenPersisterSettings settings, SettingsRootNamespace settingsRoot) : IAsyncDisposable { public RavenPersisterSettings Settings => settings; @@ -97,7 +103,7 @@ async Task EnsureReadable(string databaseName, string settingKey, CancellationTo } catch (Exception e) when (e is not DatabaseLoadTimeoutException) { - throw new InvalidOperationException($"The RavenDB migration source at {Located()} has a database named '{databaseName}', from the '{settingKey}' setting, but could not load it.", e); + throw new InvalidOperationException($"The RavenDB migration source at {Located()} has a database named '{databaseName}', from the '{settingKey}' setting, but could not load it: {e.Message}", e); } } } @@ -110,8 +116,8 @@ internal static string Located(RavenPersisterSettings sourceSettings, SettingsRo async Task StartEmbedded(CancellationToken cancellationToken) { - // A dynamic query is a POST to /queries, which the request guard allows and which builds an auto-index - // on the customer's fallback database. This makes the server refuse it rather than trusting every reader. + // A dynamic query is a POST to /queries, which the read-only guard allows, and it builds an auto-index on the + // customer's database. Nothing on the client can tell those queries apart, so the server is told to refuse them. var configuration = new EmbeddedDatabaseConfiguration(settings.ServerUrl, settings.DatabaseName, settings.DatabasePath, settings.LogPath, settings.LogsMode) { DisableAutoIndexCreation = true }; embedded = EmbeddedDatabase.Start(configuration, lifetime); @@ -126,7 +132,7 @@ async Task StartEmbedded(CancellationToken cancellationToken) } catch (Exception e) { - throw new InvalidOperationException($"The RavenDB migration source could not start a server for the embedded database at {Located()}. A ServiceControl instance still running against that data directory is the usual cause: stop it, run the report, then start it again.", e); + throw new InvalidOperationException($"The RavenDB migration source could not start a server for the embedded database at {Located()}: {e.Message} A ServiceControl instance still holding that data directory is one cause, and stopping it lets the report run; a port already in use or a missing RavenDB server are others, which the message above tells apart.", e); } } @@ -142,6 +148,7 @@ IDocumentStore Connect(string serverUrl) if (!settings.UseEmbeddedServer) { store.Certificate = RavenClientCertificate.FindClientCertificate(settings); + RefuseUnusableCertificate(store.Certificate); } store.OnBeforeRequest += RefuseWrite; @@ -149,6 +156,30 @@ IDocumentStore Connect(string serverUrl) return store.Initialize(); } + // An unusable certificate otherwise surfaces on the first request as a refused connection, which says nothing about the cause. + void RefuseUnusableCertificate(X509Certificate2? certificate) + { + if (certificate is null) + { + if (settings.ConnectionString.StartsWith("https://", StringComparison.OrdinalIgnoreCase)) + { + throw new InvalidOperationException($"The RavenDB migration source at '{settings.ConnectionString}' is secured but no client certificate is configured. Set '{settingsRoot}/{RavenBootstrapper.ClientCertificatePathKey}' or '{settingsRoot}/{RavenBootstrapper.ClientCertificateBase64Key}'."); + } + + return; + } + + // X509Certificate2 reports both bounds in local time, so they move to UTC before the comparison. + var notBefore = certificate.NotBefore.ToUniversalTime(); + var notAfter = certificate.NotAfter.ToUniversalTime(); + var now = DateTime.UtcNow; + + if (now < notBefore || now > notAfter) + { + throw new InvalidOperationException($"The RavenDB client certificate '{certificate.Subject}' is valid from {notBefore:u} to {notAfter:u}, which does not include now."); + } + } + static void RefuseWrite(object? sender, BeforeRequestEventArgs e) { if (IsRead(e.Request.Method, new Uri(e.Url).AbsolutePath)) diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/Readers/EndpointSettingsReader.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/Readers/EndpointSettingsReader.cs new file mode 100644 index 0000000000..03b8189c3d --- /dev/null +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/Readers/EndpointSettingsReader.cs @@ -0,0 +1,27 @@ +#nullable enable + +namespace ServiceControl.Persistence.RavenDB.DataMigration.Readers; + +using System.Collections.Generic; +using System.Threading; +using ServiceControl.Persistence.DataMigration; + +/// +/// Reads the per-endpoint settings, whole, out of the primary database. The collection holds one document per +/// endpoint plus one with an empty name, which carries the default for every endpoint. +/// +sealed class EndpointSettingsReader(RavenReadOnlySourceLifecycle lifecycle) : IMigrationCategoryReader +{ + public string CategoryId => MigrationCategoryIds.EndpointSettings; + + public IAsyncEnumerable Read(string? resumeAfter, int batchSize, CancellationToken cancellationToken = default) => + RavenDocumentStream.ByPrefix( + lifecycle, + CategoryId, + lifecycle.Settings.DatabaseName, + EndpointSettingsStore.CollectionName + "/", + resumeAfter, + batchSize, + RavenDocumentStream.WholeDocument, + cancellationToken); +} diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/Readers/KnownEndpointsReader.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/Readers/KnownEndpointsReader.cs new file mode 100644 index 0000000000..1a7d7b92ed --- /dev/null +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/Readers/KnownEndpointsReader.cs @@ -0,0 +1,26 @@ +#nullable enable + +namespace ServiceControl.Persistence.RavenDB.DataMigration.Readers; + +using System.Collections.Generic; +using System.Threading; +using ServiceControl.Persistence.DataMigration; + +/// +/// Reads the endpoints ServiceControl has heard from, whole, out of the primary database. +/// +sealed class KnownEndpointsReader(RavenReadOnlySourceLifecycle lifecycle) : IMigrationCategoryReader +{ + public string CategoryId => MigrationCategoryIds.KnownEndpoints; + + public IAsyncEnumerable Read(string? resumeAfter, int batchSize, CancellationToken cancellationToken = default) => + RavenDocumentStream.ByPrefix( + lifecycle, + CategoryId, + lifecycle.Settings.DatabaseName, + RavenMonitoringDataStore.KnownEndpointsCollectionName + "/", + resumeAfter, + batchSize, + RavenDocumentStream.WholeDocument, + cancellationToken); +} diff --git a/src/ServiceControl.Persistence.RavenDB/DataMigration/SourceDataVersionIsReadableCheck.cs b/src/ServiceControl.Persistence.RavenDB/DataMigration/SourceDataVersionIsReadableCheck.cs new file mode 100644 index 0000000000..dbdb7a1977 --- /dev/null +++ b/src/ServiceControl.Persistence.RavenDB/DataMigration/SourceDataVersionIsReadableCheck.cs @@ -0,0 +1,58 @@ +#nullable enable + +namespace ServiceControl.Persistence.RavenDB.DataMigration; + +using System; +using System.Threading; +using System.Threading.Tasks; +using ServiceControl.Persistence.DataMigration; + +/// +/// Refuses a copy out of a RavenDB database that a different major version of ServiceControl last wrote. The +/// documents would still load, and the readers would interpret shapes that have since changed, which copies +/// rows that look right and are wrong. Both databases are checked, the primary one and the throughput one. +/// +sealed class SourceDataVersionIsReadableCheck(RavenReadOnlySourceLifecycle lifecycle) : IMigrationStartupCheck +{ + public string Name => "the source is at a data version this build can read"; + + public async Task Run(CancellationToken cancellationToken = default) + { + await Verify(lifecycle.Settings.DatabaseName, cancellationToken); + await Verify(lifecycle.Settings.ThroughputDatabaseName, cancellationToken); + } + + async Task Verify(string databaseName, CancellationToken cancellationToken) + { + using var session = lifecycle.OpenSession(databaseName); + var stamp = await session.LoadAsync(RavenDataVersion.DocumentId, cancellationToken); + + if (stamp is null) + { + throw new Exception( + $"The RavenDB database '{databaseName}' carries no ServiceControl data version stamp. Start this instance once on RavenDB with version {RavenDataVersion.Current} before setting {MigrationSettings.EnabledKey}, so the source is brought up to date and stamped."); + } + + // RavenDB deserializes with Newtonsoft, which ignores the required modifier, so a document saved without + // the property loads with Version null. + if (string.IsNullOrWhiteSpace(stamp.Version) + || !Version.TryParse(stamp.Version.Split('-')[0], out var stamped) + || !Version.TryParse(RavenDataVersion.Current.Split('-')[0], out var thisBuild)) + { + throw new Exception( + $"The RavenDB database '{databaseName}' carries the data version stamp '{stamp.Version}' and this build reports '{RavenDataVersion.Current}', and one of them is not a version this check can compare. It cannot tell whether the source is readable, so it refuses rather than guessing."); + } + + if (stamped.Major > thisBuild.Major) + { + throw new Exception( + $"The RavenDB database '{databaseName}' was last written by ServiceControl {stamp.Version}, which is newer than this build ({RavenDataVersion.Current}). Migrate with the newer version instead."); + } + + if (stamped.Major < thisBuild.Major) + { + throw new Exception( + $"The RavenDB database '{databaseName}' was last written by ServiceControl {stamp.Version}, and this build is {RavenDataVersion.Current}. Its documents may be in shapes this build reads differently, which would copy rows that look correct and are wrong. Start this instance once on RavenDB with {RavenDataVersion.Current} before setting {MigrationSettings.EnabledKey}, which brings the source up to date and restamps it."); + } + } +} diff --git a/src/ServiceControl.Persistence.RavenDB/DatabaseSetup.cs b/src/ServiceControl.Persistence.RavenDB/DatabaseSetup.cs index 73c6812e09..170967dcc6 100644 --- a/src/ServiceControl.Persistence.RavenDB/DatabaseSetup.cs +++ b/src/ServiceControl.Persistence.RavenDB/DatabaseSetup.cs @@ -10,6 +10,7 @@ namespace ServiceControl.Persistence.RavenDB using Raven.Client.ServerWide; using Raven.Client.ServerWide.Operations; using Raven.Client.ServerWide.Operations.Configuration; + using ServiceControl.Persistence.RavenDB.DataMigration; using ServiceControl.RavenDB; class DatabaseSetup(RavenPersisterSettings settings, IDocumentStore documentStore) @@ -29,6 +30,37 @@ public async Task Execute(CancellationToken cancellationToken = default) await LicenseStatusCheck.WaitForLicenseOrThrow(documentStore, cancellationToken); await ConfigureExpiration(settings, cancellationToken); + await StampDataVersion(settings.DatabaseName, cancellationToken); + await StampDataVersion(settings.ThroughputDatabaseName, cancellationToken); + } + + // Records which ServiceControl build last wrote this database, so a migration can tell whether it reads the + // documents the same way. Only ever raised: an older build running again must not hide what a newer one wrote. + async Task StampDataVersion(string databaseName, CancellationToken cancellationToken) + { + using var session = documentStore.OpenAsyncSession(databaseName); + var stamp = await session.LoadAsync(RavenDataVersion.DocumentId, cancellationToken); + + // A version that will not parse is replaced rather than kept, because a migration refuses to start on one it cannot compare. + if (stamp is not null && !string.IsNullOrWhiteSpace(stamp.Version) + && Version.TryParse(stamp.Version.Split('-')[0], out var stamped) + && Version.TryParse(RavenDataVersion.Current.Split('-')[0], out var thisBuild) + && stamped >= thisBuild) + { + return; + } + + if (stamp is null) + { + await session.StoreAsync(new RavenDataVersion { Version = RavenDataVersion.Current, StampedAt = DateTime.UtcNow }, RavenDataVersion.DocumentId, cancellationToken); + } + else + { + stamp.Version = RavenDataVersion.Current; + stamp.StampedAt = DateTime.UtcNow; + } + + await session.SaveChangesAsync(cancellationToken); } async Task CreateDatabase(string databaseName, CancellationToken cancellationToken) diff --git a/src/ServiceControl.Persistence.RavenDB/EndpointSettingsStore.cs b/src/ServiceControl.Persistence.RavenDB/EndpointSettingsStore.cs index 0792bcd8ff..5c88f09c5a 100644 --- a/src/ServiceControl.Persistence.RavenDB/EndpointSettingsStore.cs +++ b/src/ServiceControl.Persistence.RavenDB/EndpointSettingsStore.cs @@ -47,5 +47,5 @@ public async Task UpdateEndpointSettings(EndpointSettings settings, Cancellation await session.SaveChangesAsync(cancellationToken); } - const string CollectionName = "EndpointSettings"; + internal const string CollectionName = "EndpointSettings"; } \ No newline at end of file diff --git a/src/ServiceControl.Persistence.Tests.InMemory/PersistenceTestsContext.cs b/src/ServiceControl.Persistence.Tests.InMemory/PersistenceTestsContext.cs index a23b266f1b..dc2bcb5fc6 100644 --- a/src/ServiceControl.Persistence.Tests.InMemory/PersistenceTestsContext.cs +++ b/src/ServiceControl.Persistence.Tests.InMemory/PersistenceTestsContext.cs @@ -24,6 +24,8 @@ public Task Setup(IHostApplicationBuilder hostBuilder) return Task.CompletedTask; } + public Task InstallSchema(IHost host) => Task.CompletedTask; + public Task PostSetup(IHost host) => Task.CompletedTask; public Task TearDown() => Task.CompletedTask; diff --git a/src/ServiceControl.Persistence.Tests.PostgreSql/PersistenceTestsContext.cs b/src/ServiceControl.Persistence.Tests.PostgreSql/PersistenceTestsContext.cs index 4ce91496d5..b99a642473 100644 --- a/src/ServiceControl.Persistence.Tests.PostgreSql/PersistenceTestsContext.cs +++ b/src/ServiceControl.Persistence.Tests.PostgreSql/PersistenceTestsContext.cs @@ -51,7 +51,7 @@ public async Task Setup(IHostApplicationBuilder hostBuilder) hostBuilder.Services.AddSingleton(FakeTime); } - public async Task PostSetup(IHost host) + public async Task InstallSchema(IHost host) { this.host = host; @@ -59,6 +59,8 @@ public async Task PostSetup(IHost host) await scope.ServiceProvider.GetRequiredService().ApplyMigrations(); } + public Task PostSetup(IHost host) => Task.CompletedTask; + public async Task TearDown() { DeleteBodyStorage(); diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/KnownEndpointSourceTests.cs b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/KnownEndpointSourceTests.cs new file mode 100644 index 0000000000..36c3b20731 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/KnownEndpointSourceTests.cs @@ -0,0 +1,41 @@ +namespace ServiceControl.Persistence.Tests.RavenDB.DataMigration; + +using System; +using System.Linq; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Operations; +using ServiceControl.Persistence; +using ServiceControl.Persistence.DataMigration; + +class KnownEndpointSourceTests : RavenMigrationSourceTestBase +{ + [Test] + public async Task Reads_known_endpoints_by_the_plural_prefix_in_batches_with_their_monitored_flag() + { + using (var session = await SessionProvider.OpenSession()) + { + foreach (var (name, monitored) in new[] { ("Sales.Orders", true), ("Billing", false), ("Shipping", false) }) + { + var endpoint = new KnownEndpoint + { + EndpointDetails = new EndpointDetails { Name = name, HostId = Guid.NewGuid(), Host = "HOST01" }, + HostDisplayName = "HOST01", + Monitored = monitored + }; + + await session.StoreAsync(endpoint, $"KnownEndpoints/{endpoint.EndpointDetails.GetDeterministicId()}"); + } + + await session.SaveChangesAsync(); + } + + await using var source = await OpenMigrationSource(); + var category = MigrationCategoryRegistry.All.Single(entry => entry.Id == MigrationCategoryIds.KnownEndpoints); + + var batches = await CollectBatches(source, category, batchSize: 2); + + Assert.That(batches.Select(batch => batch.Rows.Count), Is.EqualTo(new[] { 2, 1 })); + Assert.That(batches.SelectMany(batch => batch.Rows).Count(row => ((KnownEndpoint)row.Document).Monitored), Is.EqualTo(1)); + } +} diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/MigrationCategoryReadersTests.cs b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/MigrationCategoryReadersTests.cs new file mode 100644 index 0000000000..e9e5fd1cb9 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/MigrationCategoryReadersTests.cs @@ -0,0 +1,21 @@ +namespace ServiceControl.Persistence.Tests.RavenDB.DataMigration; + +using System.Linq; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.RavenDB.DataMigration; + +[TestFixture] +class MigrationCategoryReadersTests : RavenMigrationSourceTestBase +{ + [Test] + public async Task Every_reader_claims_a_category_the_registry_knows() + { + await using var source = (RavenMigrationSource)await OpenMigrationSource(); + + var unknown = source.SupportedCategoryIds.Where(id => MigrationCategoryRegistry.Find(id) is null).ToArray(); + + Assert.That(unknown, Is.Empty, "a reader for an id no category has is dead code the engine can never reach"); + } +} diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenDataVersionTests.cs b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenDataVersionTests.cs new file mode 100644 index 0000000000..bce5ff6deb --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenDataVersionTests.cs @@ -0,0 +1,234 @@ +namespace ServiceControl.Persistence.Tests.RavenDB.DataMigration; + +using System; +using System.Linq; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.RavenDB; +using ServiceControl.Persistence.RavenDB.DataMigration; + +class RavenDataVersionTests : RavenMigrationSourceTestBase +{ + [OneTimeSetUp] + public static void TheVersionStringIsUsable() + { + Assert.That(Version.TryParse(RavenDataVersion.Current.Split('-')[0], out var version), Is.True, + $"RavenDataVersion.Current is '{RavenDataVersion.Current}', which does not parse once the pre-release suffix is cut, so the data version check can neither pass nor refuse for the right reason."); + + Assert.That(version.Major, Is.GreaterThan(1), + "This is a Debug build, which MinVer stamps 1.0.0, so this instance will refuse any database stamped by a release build and its own stamp will satisfy every later check. Run these tests in Release."); + } + + [Test] + public async Task Starting_the_persister_stamps_the_build_that_wrote_the_database() + { + using var session = DocumentStore.OpenAsyncSession(); + + var stamp = await session.LoadAsync(RavenDataVersion.DocumentId, TestContext.CurrentContext.CancellationToken); + + Assert.That(stamp, Is.Not.Null, "Without a stamp a source read three years after it was last written looks identical to one this build wrote a minute ago."); + Assert.Multiple(() => + { + Assert.That(stamp.Version, Is.EqualTo(RavenDataVersion.Current)); + Assert.That(stamp.StampedAt, Is.EqualTo(DateTime.UtcNow).Within(TimeSpan.FromMinutes(1))); + }); + } + + [Test] + public async Task Starting_the_persister_stamps_the_throughput_database_as_well() + { + using var session = DocumentStore.OpenAsyncSession(ThroughputDatabaseName); + + var stamp = await session.LoadAsync(RavenDataVersion.DocumentId, TestContext.CurrentContext.CancellationToken); + + Assert.That(stamp, Is.Not.Null, "The migration reads the throughput database too, so an unstamped one is a source the version check has nothing to judge."); + Assert.Multiple(() => + { + Assert.That(stamp.Version, Is.EqualTo(RavenDataVersion.Current)); + Assert.That(stamp.StampedAt, Is.EqualTo(DateTime.UtcNow).Within(TimeSpan.FromMinutes(1))); + }); + } + + [Test] + public async Task Starting_the_persister_never_lowers_a_stamp_a_newer_build_wrote() + { + // This is an ordinary rollback: a newer build runs once, then an older one starts. + await Stamp("999.0.0"); + + await new DatabaseSetup((RavenPersisterSettings)PersistenceSettings, DocumentStore).Execute(TestContext.CurrentContext.CancellationToken); + + using var session = DocumentStore.OpenAsyncSession(); + var stamp = await session.LoadAsync(RavenDataVersion.DocumentId, TestContext.CurrentContext.CancellationToken); + + Assert.That(stamp.Version, Is.EqualTo("999.0.0"), + "A lowered stamp makes the source check compare equal and pass on a database a newer build has written."); + } + + [Test] + public async Task Starting_the_persister_replaces_a_stamp_nothing_can_compare() + { + // The check refuses any stamp it cannot parse, so if startup left one in place no restart could get this instance past the check. + await Stamp("not-a-version"); + + await new DatabaseSetup((RavenPersisterSettings)PersistenceSettings, DocumentStore).Execute(TestContext.CurrentContext.CancellationToken); + + using var session = DocumentStore.OpenAsyncSession(); + var stamp = await session.LoadAsync(RavenDataVersion.DocumentId, TestContext.CurrentContext.CancellationToken); + + Assert.That(stamp.Version, Is.EqualTo(RavenDataVersion.Current)); + } + + [Test] + public async Task The_source_check_accepts_a_database_this_build_stamped() + { + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + Assert.DoesNotThrowAsync(() => check.Run(TestContext.CurrentContext.CancellationToken), + "A database the running build just stamped is the one case the check has to let through."); + } + + [Test] + public async Task The_source_check_refuses_a_database_carrying_no_stamp() + { + await DeleteStamp(DatabaseName); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(DatabaseName)); + Assert.That(refusal.Message, Does.Contain(RavenDataVersion.Current)); + Assert.That(refusal.Message, Does.Contain(MigrationSettings.EnabledKey)); + }); + } + + [Test] + public async Task The_source_check_refuses_a_database_stamped_by_a_newer_major() + { + await Stamp("999.0.0"); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(DatabaseName)); + Assert.That(refusal.Message, Does.Contain("999.0.0")); + Assert.That(refusal.Message, Does.Contain(RavenDataVersion.Current)); + Assert.That(refusal.Message, Does.Contain("Migrate with the newer version instead")); + }); + } + + [Test] + public async Task The_source_check_refuses_a_database_stamped_by_an_older_major() + { + await Stamp("1.0.0"); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(DatabaseName)); + Assert.That(refusal.Message, Does.Contain("1.0.0")); + Assert.That(refusal.Message, Does.Contain(RavenDataVersion.Current)); + Assert.That(refusal.Message, Does.Contain("Start this instance once on RavenDB"), + "An upgrade that never restarted on RavenDB is the case this check exists for, so its refusal has to name the restart that fixes it."); + }); + } + + [Test] + public async Task The_source_check_refuses_a_database_stamped_with_something_it_cannot_compare() + { + await Stamp("not-a-version"); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(DatabaseName)); + Assert.That(refusal.Message, Does.Contain("not-a-version")); + Assert.That(refusal.Message, Does.Contain(RavenDataVersion.Current)); + Assert.That(refusal.Message, Does.Contain("refuses rather than guessing"), + "Passing on a stamp it cannot read is the one outcome a version check must never produce, and it would copy the whole source unchecked."); + }); + } + + [Test] + public async Task The_source_check_refuses_a_database_whose_stamp_carries_no_version_at_all() + { + // Newtonsoft does not enforce required, so a document written without the property loads as null. + await Stamp(null); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.That(refusal.Message, Does.Contain("refuses rather than guessing"), + "A NullReferenceException names neither the database nor what to do about it."); + } + + [Test] + public async Task The_source_check_refuses_a_throughput_database_carrying_no_stamp() + { + await DeleteStamp(ThroughputDatabaseName); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(ThroughputDatabaseName), + "Restarting on RavenDB is the remedy for whichever database is out of date, so the refusal has to say which one it is."); + Assert.That(refusal.Message, Does.Contain(RavenDataVersion.Current)); + Assert.That(refusal.Message, Does.Contain(MigrationSettings.EnabledKey)); + }); + } + + [Test] + public async Task The_source_check_refuses_a_throughput_database_stamped_by_an_older_major() + { + await Stamp(ThroughputDatabaseName, "1.0.0"); + await using var source = await OpenMigrationSource(); + var check = source.ContributedChecks().Single(); + + var refusal = Assert.ThrowsAsync(() => check.Run(TestContext.CurrentContext.CancellationToken)); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(ThroughputDatabaseName)); + Assert.That(refusal.Message, Does.Contain("1.0.0")); + Assert.That(refusal.Message, Does.Contain(RavenDataVersion.Current)); + Assert.That(refusal.Message, Does.Contain("Start this instance once on RavenDB")); + }); + } + + string DatabaseName => ((RavenPersisterSettings)PersistenceSettings).DatabaseName; + + string ThroughputDatabaseName => ((RavenPersisterSettings)PersistenceSettings).ThroughputDatabaseName; + + Task Stamp(string version) => Stamp(DatabaseName, version); + + async Task Stamp(string databaseName, string version) + { + using var session = DocumentStore.OpenAsyncSession(databaseName); + await session.StoreAsync(new RavenDataVersion { Version = version, StampedAt = DateTime.UtcNow }, RavenDataVersion.DocumentId, TestContext.CurrentContext.CancellationToken); + await session.SaveChangesAsync(TestContext.CurrentContext.CancellationToken); + } + + async Task DeleteStamp(string databaseName) + { + using var session = DocumentStore.OpenAsyncSession(databaseName); + session.Delete(RavenDataVersion.DocumentId); + await session.SaveChangesAsync(TestContext.CurrentContext.CancellationToken); + } +} diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenMigrationSourceTestBase.cs b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenMigrationSourceTestBase.cs new file mode 100644 index 0000000000..358c3f9943 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenMigrationSourceTestBase.cs @@ -0,0 +1,67 @@ +namespace ServiceControl.Persistence.Tests.RavenDB.DataMigration; + +using System; +using System.Collections.Generic; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Configuration; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.RavenDB; + +abstract class RavenMigrationSourceTestBase : RavenPersistenceTestBase +{ + static readonly SettingsRootNamespace SettingsRoot = new("ServiceControl"); + static readonly object SettingsGate = new(); + + protected async Task OpenMigrationSource() + { + var settings = (RavenPersisterSettings)PersistenceSettings; + (string Name, string Value)[] variables = + [ + ("SERVICECONTROL_RAVENDB_CONNECTIONSTRING", settings.ConnectionString), + ("SERVICECONTROL_RAVENDB_DATABASENAME", settings.DatabaseName), + ("SERVICECONTROL_ERRORRETENTIONPERIOD", settings.ErrorRetentionPeriod.ToString()), + ("LICENSINGCOMPONENT_RAVENDB_THROUGHPUTDATABASENAME", settings.ThroughputDatabaseName) + ]; + + IMigrationSource source; + + // Environment variables are process wide and these tests run in parallel, so without the gate a source + // reads whichever test set them last and opens another test's database. + lock (SettingsGate) + { + try + { + foreach (var (name, value) in variables) + { + Environment.SetEnvironmentVariable(name, value); + } + + source = new RavenPersistenceConfiguration().CreateSource(SettingsRoot); + } + finally + { + foreach (var (name, _) in variables) + { + Environment.SetEnvironmentVariable(name, null); + } + } + } + + await source.Open(); + + return source; + } + + protected static async Task> CollectBatches(IMigrationSource source, MigrationCategory category, string resumeAfter = null, int batchSize = 100) + { + var batches = new List(); + + await foreach (var batch in source.Read(category, resumeAfter, batchSize, TestContext.CurrentContext.CancellationToken)) + { + batches.Add(batch); + } + + return batches; + } +} diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenMigrationSourceTests.cs b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenMigrationSourceTests.cs new file mode 100644 index 0000000000..c34641d535 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/RavenMigrationSourceTests.cs @@ -0,0 +1,145 @@ +namespace ServiceControl.Persistence.Tests.RavenDB.DataMigration; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using NUnit.Framework; +using Raven.Client.Documents.Operations.Indexes; +using ServiceControl.Operations; +using ServiceControl.Persistence.DataMigration; + +class RavenMigrationSourceTests : RavenMigrationSourceTestBase +{ + [Test] + public async Task Reads_endpoint_settings_in_document_id_order_in_batches_of_the_requested_size() + { + await SeedEndpointSettings(5); + await using var source = await OpenMigrationSource(); + + var batches = await CollectBatches(source, EndpointSettingsCategory, batchSize: 2); + + Assert.Multiple(() => + { + Assert.That(batches.Select(batch => batch.Rows.Count), Is.EqualTo(new[] { 2, 2, 1 }), "A batch never exceeds the requested size, and the trailing partial batch is still yielded."); + Assert.That(batches.SelectMany(batch => batch.Rows).Select(row => row.SourceId), Is.Ordered); + }); + } + + [Test] + public async Task Every_row_is_identified_by_the_document_id_the_persister_wrote_it_under() + { + await SeedEndpointSettings(5); + await using var source = await OpenMigrationSource(); + + var rows = (await CollectBatches(source, EndpointSettingsCategory)).SelectMany(batch => batch.Rows).ToList(); + + Assert.Multiple(() => + { + Assert.That(rows, Has.Count.EqualTo(5)); + Assert.That(rows.Select(row => row.SourceId), Has.All.StartWith("EndpointSettings/"), "The cursor the engine checkpoints is a source document id, so a row identified by anything else cannot be resumed after."); + }); + } + + [Test] + public async Task Resuming_after_a_cursor_yields_only_what_follows_it() + { + await SeedEndpointSettings(5); + await using var source = await OpenMigrationSource(); + + var firstBatch = (await CollectBatches(source, EndpointSettingsCategory, batchSize: 2))[0]; + var resumed = (await CollectBatches(source, EndpointSettingsCategory, resumeAfter: firstBatch.Cursor, batchSize: 2)).SelectMany(batch => batch.Rows).ToList(); + + Assert.That(resumed, Has.Count.EqualTo(3)); + Assert.That(resumed.Select(row => row.SourceId), Has.No.Member(firstBatch.Rows[0].SourceId)); + Assert.That(resumed.Select(row => row.SourceId), Is.Ordered); + } + + [Test] + public async Task Resuming_after_a_cursor_this_source_never_issued_refuses_instead_of_starting_past_it() + { + // The cursor is saved in the target database, so a restored backup or a corrected database name leaves + // one naming a document this source does not have. + await SeedEndpointSettings(3); + await using var source = await OpenMigrationSource(); + + var refusal = Assert.ThrowsAsync(() => CollectBatches(source, EndpointSettingsCategory, resumeAfter: "EndpointSettings/zzz-from-another-database")); + + Assert.Multiple(() => + { + Assert.That(refusal.Message, Does.Contain(EndpointSettingsCategory.Id)); + Assert.That(refusal.Message, Does.Contain("EndpointSettings/zzz-from-another-database")); + Assert.That(refusal.Message, Does.Contain(DatabaseName), "the operator has to be told which database the cursor was looked for in"); + }); + } + + [Test] + public async Task Counts_every_row_the_category_would_read() + { + await SeedEndpointSettings(5); + await using var source = await OpenMigrationSource(); + + Assert.That(await source.Count(EndpointSettingsCategory), Is.EqualTo(5)); + } + + [Test] + public async Task Counts_a_category_with_nothing_in_it_as_zero() + { + await using var source = await OpenMigrationSource(); + + Assert.That(await source.Count(EndpointSettingsCategory), Is.Zero, "A category with nothing to copy has to be told apart from one this source declines to count."); + } + + [Test] + public async Task Reading_every_category_creates_no_index() + { + await using var source = await OpenMigrationSource(); + + Assert.That(source.SupportedCategoryIds, Is.SubsetOf(Seeds.Keys), "A category read with nothing in it cannot show whether its reader builds an index, so a new reader needs its seed adding here."); + + foreach (var categoryId in source.SupportedCategoryIds) + { + await Seeds[categoryId](this); + } + + var before = await IndexNames(); + + foreach (var categoryId in source.SupportedCategoryIds) + { + await CollectBatches(source, MigrationCategoryRegistry.Find(categoryId)); + } + + var after = await IndexNames(); + + Assert.Multiple(() => + { + Assert.That(after, Is.EquivalentTo(before), "A reader that queries without naming a static index has RavenDB build one and index the whole collection on the customer's live database."); + Assert.That(after, Has.None.StartWith("Auto/")); + }); + } + + // Keyed by category id so a reader added without seed data fails the index test by name instead of passing + // over an empty collection. + static readonly Dictionary> Seeds = new() + { + [MigrationCategoryIds.EndpointSettings] = tests => tests.SeedEndpointSettings(2), + [MigrationCategoryIds.KnownEndpoints] = tests => tests.MonitoringDataStore.CreateIfNotExists(new EndpointDetails { Name = "Sales.Orders", HostId = Guid.NewGuid(), Host = "HOST01" }) + }; + + string DatabaseName => ((RavenPersisterSettings)PersistenceSettings).DatabaseName; + + async Task IndexNames() => + await DocumentStore.Maintenance.SendAsync(new GetIndexNamesOperation(0, IndexNamePageSize), TestContext.CurrentContext.CancellationToken); + + const int IndexNamePageSize = 1024; + + static readonly MigrationCategory EndpointSettingsCategory = MigrationCategoryRegistry.Find("EndpointSettings"); + + async Task SeedEndpointSettings(int count) + { + for (var index = 0; index < count; index++) + { + await EndpointSettingsStore.UpdateEndpointSettings(new EndpointSettings { Name = $"Endpoint{index}", TrackInstances = true }); + } + } +} diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/ReadOnlySourceLifecycleTests.cs b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/ReadOnlySourceLifecycleTests.cs index 859e02c505..ba3f892ffe 100644 --- a/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/ReadOnlySourceLifecycleTests.cs +++ b/src/ServiceControl.Persistence.Tests.RavenDB/DataMigration/ReadOnlySourceLifecycleTests.cs @@ -88,6 +88,20 @@ public async Task Opening_the_source_creates_no_database() Assert.That(await bootstrapStore.Maintenance.Server.SendAsync(new GetDatabaseRecordOperation(absentThroughput)), Is.Null, "Opening a migration source must not create a database that was missing."); } + [Test] + public async Task Opening_the_source_refuses_a_missing_primary_database() + { + var absentPrimary = $"{databaseName}-absent"; + sourceSettings.DatabaseName = absentPrimary; + + await using var lifecycle = new RavenReadOnlySourceLifecycle(sourceSettings, SettingsRoot); + + var exception = Assert.ThrowsAsync(async () => await lifecycle.Open()); + + Assert.That(exception.Message, Does.Contain(absentPrimary).And.Contain("ServiceControl/RavenDB/DatabaseName")); + Assert.That(await bootstrapStore.Maintenance.Server.SendAsync(new GetDatabaseRecordOperation(absentPrimary)), Is.Null, "Opening a migration source must not create a primary database that was missing."); + } + [Test] public async Task Opening_the_source_writes_no_database_settings() { diff --git a/src/ServiceControl.Persistence.Tests.RavenDB/PersistenceTestsContext.cs b/src/ServiceControl.Persistence.Tests.RavenDB/PersistenceTestsContext.cs index 5bcae71f20..aca5f06236 100644 --- a/src/ServiceControl.Persistence.Tests.RavenDB/PersistenceTestsContext.cs +++ b/src/ServiceControl.Persistence.Tests.RavenDB/PersistenceTestsContext.cs @@ -50,6 +50,8 @@ public async Task Setup(IHostApplicationBuilder hostBuilder) persistence.AddInstaller(hostBuilder.Services); } + public Task InstallSchema(IHost host) => Task.CompletedTask; + public async Task PostSetup(IHost host) { DocumentStore = await host.Services.GetRequiredService().GetDocumentStore(); diff --git a/src/ServiceControl.Persistence.Tests.SqlServer/EndpointSettingsKeyCollationTests.cs b/src/ServiceControl.Persistence.Tests.SqlServer/EndpointSettingsKeyCollationTests.cs new file mode 100644 index 0000000000..a00f39cad2 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.SqlServer/EndpointSettingsKeyCollationTests.cs @@ -0,0 +1,77 @@ +namespace ServiceControl.Persistence.Tests; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Infrastructure; +using Microsoft.EntityFrameworkCore.Storage; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceControl.Operations; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.DbContexts; +using ServiceControl.Persistence.EFCore.Entities; +using ServiceControl.Persistence.EFCore.Infrastructure; + +class EndpointSettingsKeyCollationTests : PersistenceTestBase +{ + [Test] + public async Task The_key_columns_own_collation_decides_a_merge_when_the_database_default_disagrees() + { + bool databaseIgnoresCase; + + using (var scope = ServiceProvider.CreateScope()) + { + var dbContext = scope.ServiceProvider.GetRequiredService(); + var database = dbContext.Database; + + databaseIgnoresCase = await database + .SqlQuery($"SELECT CONVERT(int, DATABASEPROPERTYEX(DB_NAME(), 'ComparisonStyle')) & 1 AS [Value]") + .SingleAsync() == 1; + + // Named from the model, because the table sits in the configured schema when the persister has one. + var entityType = dbContext.Model.FindEntityType(typeof(EndpointSettingsEntity)); + var table = dbContext.GetService().DelimitIdentifier(entityType.GetTableName(), entityType.GetSchema()); + + // The column is given the opposite of the database default, because a test where the two agree cannot show which one decided. + var recollate = $""" + DECLARE @columnCollation sysname = IIF(CONVERT(int, DATABASEPROPERTYEX(DB_NAME(), 'ComparisonStyle')) & 1 = 1, N'Latin1_General_CS_AS', N'Latin1_General_CI_AS'); + ALTER TABLE {table} DROP CONSTRAINT [PK_EndpointSettings]; + EXEC (N'ALTER TABLE {table} ALTER COLUMN [Name] nvarchar(450) COLLATE ' + @columnCollation + N' NOT NULL'); + ALTER TABLE {table} ADD CONSTRAINT [PK_EndpointSettings] PRIMARY KEY ([Name]); + """; + + await database.ExecuteSqlRawAsync(recollate); + } + + foreach (var name in new[] { "Sales", "sales" }) + { + await MonitoringDataStore.CreateIfNotExists(new EndpointDetails { Name = name, HostId = Guid.NewGuid(), Host = "HOST01" }); + } + + var target = ServiceProvider.GetRequiredService(); + await target.Open(); + + var category = MigrationCategoryRegistry.All.Single(entry => entry.Id == MigrationCategoryIds.EndpointSettings); + var batch = new MigrationBatch( + [ + new MigrationRow("EndpointSettings/1", new EndpointSettings { Name = "Sales", TrackInstances = true }, new Dictionary()), + new MigrationRow("EndpointSettings/2", new EndpointSettings { Name = "sales", TrackInstances = false }, new Dictionary()) + ], + "EndpointSettings/2"); + var checkpointToExtend = new MigrationCheckpoint(category.Id, MigrationCategoryState.InProgress, batch.Cursor, 0, 0, null, null, null, null, null, null); + + var result = await target.Write(category, batch, checkpointToExtend); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.EqualTo(databaseIgnoresCase ? 2 : 1), "the column's own collation decides: two keys where it respects case, one where it ignores case, whatever the database default says"); + Assert.That( + ServiceProvider.GetRequiredService().KeyComparer(typeof(EndpointSettingsEntity), nameof(EndpointSettingsEntity.Name)).Equals("Sales", "sales"), + Is.EqualTo(!databaseIgnoresCase), + "the dry run predicts merges with this comparer, so it follows the same column the statement does"); + } + } +} diff --git a/src/ServiceControl.Persistence.Tests.SqlServer/MigrationSettingsWiringTests.cs b/src/ServiceControl.Persistence.Tests.SqlServer/MigrationSettingsWiringTests.cs new file mode 100644 index 0000000000..4c9c5e60aa --- /dev/null +++ b/src/ServiceControl.Persistence.Tests.SqlServer/MigrationSettingsWiringTests.cs @@ -0,0 +1,37 @@ +// ReSharper disable once CheckNamespace +namespace ServiceControl.Persistence.Tests; + +using System; +using System.IO; +using System.Runtime.Loader; +using NUnit.Framework; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Persistence; +using ServiceControl.Persistence.EFCore.Abstractions; +using ServiceControl.Persistence.EFCore.SqlServer; + +class MigrationSettingsWiringTests +{ + [Test] + public void The_retry_history_depth_is_carried_onto_the_persister() + { + var settings = new Settings(transportType: "LearningTransport", persisterType: "SQLServer", errorRetentionPeriod: TimeSpan.FromDays(10)) + { + // This project references the persister already, so nothing has to be loaded from a persistence manifest. + AssemblyLoadContextResolver = static _ => AssemblyLoadContext.Default, + RetryHistoryDepth = 42, + PersisterSpecificSettings = new SqlServerPersisterSettings + { + ConnectionString = "Server=.;Database=no-connection-is-opened-by-this-test", + BodyStorage = new FileSystemBodyStorageSettings { StoragePath = Path.GetTempPath() } + } + }; + + PersistenceFactory.Create(settings); + + Assert.That( + settings.PersisterSpecificSettings.RetryHistoryDepth, + Is.EqualTo(42), + "EFCoreMigrationTargetReadiness builds RetryHistoryDepthIsSafeCheck out of this value, so losing the copy refuses every migrating instance over a setting the customer never touched"); + } +} diff --git a/src/ServiceControl.Persistence.Tests.SqlServer/PersistenceTestsContext.cs b/src/ServiceControl.Persistence.Tests.SqlServer/PersistenceTestsContext.cs index 9ded3182bc..2dab62abe4 100644 --- a/src/ServiceControl.Persistence.Tests.SqlServer/PersistenceTestsContext.cs +++ b/src/ServiceControl.Persistence.Tests.SqlServer/PersistenceTestsContext.cs @@ -50,7 +50,7 @@ public async Task Setup(IHostApplicationBuilder hostBuilder) hostBuilder.Services.AddSingleton(FakeTime); } - public async Task PostSetup(IHost host) + public async Task InstallSchema(IHost host) { this.host = host; @@ -58,6 +58,8 @@ public async Task PostSetup(IHost host) await scope.ServiceProvider.GetRequiredService().ApplyMigrations(); } + public Task PostSetup(IHost host) => Task.CompletedTask; + public async Task TearDown() { DeleteBodyStorage(); diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/CheckpointTableIsReadableTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/CheckpointTableIsReadableTests.cs new file mode 100644 index 0000000000..ccba0cb1e7 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/CheckpointTableIsReadableTests.cs @@ -0,0 +1,85 @@ +#nullable enable +namespace ServiceControl.Persistence.Tests; + +using System; +using System.Collections.Generic; +using System.Threading; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.DataMigration; + +// The only thing that tells an operator who upgraded without --setup that the schema is older than the build. +// Without it they get a raw EF error from whichever component touches the table first. +[TestFixture] +class CheckpointTableIsReadableTests +{ + [Test] + public void A_checkpoint_table_this_build_cannot_read_stops_the_start_and_names_what_fixes_it() + { + var tableMissing = new InvalidOperationException("Invalid object name 'MigrationCheckpoints'."); + var check = new CheckpointTableIsReadable(new ProbedCheckpointStore(tableMissing)); + + var refusal = Assert.ThrowsAsync(() => check.StartingAsync()); + + using (Assert.EnterMultipleScope()) + { + Assert.That(refusal!.Message, Does.Contain("--setup"), "a refusal that does not say what to run leaves the operator with a database they cannot start"); + Assert.That(refusal.Message, Does.Contain("older than this build"), "the raw provider error says the table is missing, not that the schema is behind"); + Assert.That(refusal.InnerException, Is.SameAs(tableMissing), "the provider's own error is the only thing saying which table and which database"); + } + } + + [Test] + public void A_checkpoint_table_that_reads_lets_the_start_go_on() => + Assert.DoesNotThrowAsync(() => new CheckpointTableIsReadable(new ProbedCheckpointStore()).StartingAsync(), + "a check that refuses a healthy database stops every instance from starting"); + + [Test] + public async Task A_host_being_shut_down_is_not_reported_as_a_schema_that_is_behind() + { + using var cancellation = new CancellationTokenSource(); + await cancellation.CancelAsync(); + var check = new CheckpointTableIsReadable(new ProbedCheckpointStore(new OperationCanceledException(cancellation.Token))); + + Assert.ThrowsAsync(() => check.StartingAsync(cancellation.Token), + "a stop during startup would otherwise send the operator to run --setup against a database that is fine"); + } + + [Test] + public async Task The_table_is_read_before_the_web_server_binds_rather_than_after() + { + var store = new ProbedCheckpointStore(); + var check = new CheckpointTableIsReadable(store); + + await check.StartAsync(); + await check.StartedAsync(); + + Assert.That(store.ReadAllCalls, Is.Zero, "reading any later than StartingAsync lets a Windows service tell the Service Control Manager it is running on a schema this build cannot read"); + + await check.StartingAsync(); + + Assert.That(store.ReadAllCalls, Is.EqualTo(1)); + } + + // ReadAll is the only member the check calls, so the other two are here to satisfy the interface. + sealed class ProbedCheckpointStore(Exception? readAllFailure = null) : IMigrationCheckpointStore + { + public int ReadAllCalls { get; private set; } + + public Task> ReadAll(CancellationToken cancellationToken = default) + { + ReadAllCalls++; + + return readAllFailure is null + ? Task.FromResult>([]) + : Task.FromException>(readAllFailure); + } + + public Task Read(string categoryId, CancellationToken cancellationToken = default) => + throw new NotSupportedException("The startup check reads the whole table and never one category."); + + public Task Upsert(MigrationCheckpoint checkpoint, CancellationToken cancellationToken = default) => + throw new NotSupportedException("The startup check never writes."); + } +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/EndpointSettingsMigrationTargetTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/EndpointSettingsMigrationTargetTests.cs new file mode 100644 index 0000000000..08c22cffbd --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/EndpointSettingsMigrationTargetTests.cs @@ -0,0 +1,222 @@ +namespace ServiceControl.Persistence.Tests; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceControl.Operations; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.DbContexts; + +class EndpointSettingsMigrationTargetTests : PersistenceTestBase +{ + [SetUp] + public Task OpenTarget() => Target.Open(); + + IMigrationTarget Target => ServiceProvider.GetRequiredService(); + + IMigrationCheckpointStore CheckpointStore => ServiceProvider.GetRequiredService(); + + static readonly MigrationCategory EndpointSettingsCategory = MigrationCategoryRegistry.All.Single(category => category.Id == MigrationCategoryIds.EndpointSettings); + + [Test] + public async Task Every_setting_in_a_batch_arrives_with_its_track_instances_value() + { + await SeedKnownEndpoints("Sales", "Billing", "Shipping"); + + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf( + ("EndpointSettings/1", new EndpointSettings { Name = "Sales", TrackInstances = true }), + ("EndpointSettings/2", new EndpointSettings { Name = "Billing", TrackInstances = false }), + ("EndpointSettings/3", new EndpointSettings { Name = "Shipping", TrackInstances = true })), + CheckpointAfter("EndpointSettings/3")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.EqualTo(3), "Copied is what the insert added, not the size of the batch"); + Assert.That(await TrackInstancesFor("Sales"), Is.True); + Assert.That(await TrackInstancesFor("Billing"), Is.False); + Assert.That(await TrackInstancesFor("Shipping"), Is.True); + } + } + + [Test] + public async Task A_name_already_in_the_target_is_left_alone_and_counted_as_already_present() + { + await SeedKnownEndpoints("Sales"); + await EndpointSettingsStore.UpdateEndpointSettings(new EndpointSettings { Name = "Sales", TrackInstances = true }); + + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf(("EndpointSettings/7", new EndpointSettings { Name = "Sales", TrackInstances = false })), + CheckpointAfter("EndpointSettings/7")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.Zero); + Assert.That(result.AlreadyPresent, Is.EqualTo(1)); + Assert.That(result.Skipped, Is.Zero, "a row the target already holds lost nothing, so it is not a skip"); + Assert.That(await TrackInstancesFor("Sales"), Is.True); + } + } + + [Test] + public async Task A_named_setting_whose_endpoint_is_not_known_is_skipped_as_endpoint_not_known() + { + await SeedKnownEndpoints("Sales"); + + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf( + ("EndpointSettings/1", new EndpointSettings { Name = "Retired", TrackInstances = true }), + ("EndpointSettings/2", new EndpointSettings { Name = "Sales", TrackInstances = true })), + CheckpointAfter("EndpointSettings/2")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.EqualTo(1)); + Assert.That(result.SkippedIds, Is.EqualTo(new[] { "EndpointSettings/1" })); + Assert.That(result.SkipReasons[MigrationSkipReason.EndpointNotKnown], Is.EqualTo(1), "the heartbeat settings sync deletes this row twenty seconds after the host opens, so verify has to see it as a skip"); + Assert.That(result.BenignSkipped, Is.EqualTo(1), "a setting left behind on purpose is reported, but a source full of them must not halt a required category"); + Assert.That((await EndpointSettingsStore.GetAllEndpointSettings().ToListAsync()).Select(settings => settings.Name), Is.EqualTo(new[] { "Sales" })); + } + } + + [Test] + public async Task An_unknown_endpoint_stays_a_benign_skip_while_known_endpoints_dropped_nothing() + { + await SeedKnownEndpoints("Sales"); + await SaveKnownEndpointsCheckpoint(skipped: 0); + + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf(("EndpointSettings/1", new EndpointSettings { Name = "Retired", TrackInstances = true })), + CheckpointAfter("EndpointSettings/1")); + + Assert.That(result.BenignSkipped, Is.EqualTo(1), "the endpoint is unknown in the source too, so the sync would delete this setting whatever the migration did"); + } + + [Test] + public async Task An_unknown_endpoint_stops_being_a_benign_skip_once_known_endpoints_dropped_a_row() + { + await SeedKnownEndpoints("Sales"); + await SaveKnownEndpointsCheckpoint(skipped: 1); + + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf(("EndpointSettings/1", new EndpointSettings { Name = "Retired", TrackInstances = true })), + CheckpointAfter("EndpointSettings/1")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Skipped, Is.EqualTo(1)); + Assert.That(result.BenignSkipped, Is.Zero, "the endpoint may be unknown only because KnownEndpoints dropped it, and a setting the sync would have kept has to reach the halt threshold"); + } + } + + [Test] + public async Task A_batch_of_nothing_but_skips_saves_the_cursor_and_writes_no_rows() + { + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf( + ("EndpointSettings/1", new EndpointSettings { Name = "Retired", TrackInstances = true }), + ("EndpointSettings/2", new EndpointSettings { Name = "Decommissioned", TrackInstances = false })), + CheckpointAfter("EndpointSettings/2")); + + var stored = await CheckpointStore.Read(MigrationCategoryIds.EndpointSettings); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.Zero); + Assert.That(result.AlreadyPresent, Is.Zero, "no name was looked up, so no row may be counted as one the target already held"); + Assert.That(result.Skipped, Is.EqualTo(2)); + Assert.That(await EndpointSettingsStore.GetAllEndpointSettings().ToListAsync(), Is.Empty); + Assert.That(stored.Cursor, Is.EqualTo("EndpointSettings/2"), "a batch that copied nothing still has to commit the cursor past it, or the restart reads the same rows forever"); + } + } + + [Test] + public async Task The_empty_name_default_is_copied_when_no_endpoint_is_known() + { + var result = await Target.Write(EndpointSettingsCategory, BatchOf(("EndpointSettings/1", new EndpointSettings { Name = string.Empty, TrackInstances = true })), CheckpointAfter("EndpointSettings/1")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.EqualTo(1)); + Assert.That(result.Skipped, Is.Zero, "the sync keeps the default whatever endpoints are known"); + Assert.That(await TrackInstancesFor(string.Empty), Is.True); + } + } + + [Test] + public async Task Every_mapped_column_is_set_from_a_fully_populated_document() + { + await SeedKnownEndpoints("Sales"); + await Target.Write(EndpointSettingsCategory, BatchOf(("EndpointSettings/1", new EndpointSettings { Name = "Sales", TrackInstances = true })), CheckpointAfter("EndpointSettings/1")); + + using var scope = ServiceProvider.CreateScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + MigrationEntityCoverage.AssertEveryMappedPropertyIsSet(dbContext.Model, await dbContext.EndpointSettings.AsNoTracking().SingleAsync()); + } + + static MigrationBatch BatchOf(params (string Id, object Document)[] rows) => + new([.. rows.Select(row => new MigrationRow(row.Id, row.Document, new Dictionary()))], rows[^1].Id); + + static MigrationCheckpoint CheckpointAfter(string cursor) => + new(MigrationCategoryIds.EndpointSettings, MigrationCategoryState.InProgress, cursor, 0, 0, null, null, null, null, null, null); + + async Task TrackInstancesFor(string name) => + (await EndpointSettingsStore.GetAllEndpointSettings().ToListAsync()).Single(settings => settings.Name == name).TrackInstances; + + // The target reads this checkpoint to tell an endpoint the source never had from one KnownEndpoints dropped. + Task SaveKnownEndpointsCheckpoint(long skipped) => + CheckpointStore.Upsert(new MigrationCheckpoint( + MigrationCategoryIds.KnownEndpoints, + skipped == 0 ? MigrationCategoryState.Complete : MigrationCategoryState.CompleteWithErrors, + "KnownEndpoints/9", + CopiedCount: 9, + SkippedCount: skipped, + SourceTotal: null, + SkipReasons: skipped == 0 ? null : new Dictionary { [MigrationSkipReason.RequiredValueMissing] = skipped }, + StartedAt: null, + LastProgressAt: null, + SettledAt: null, + LastError: null)); + + // The target copies a named setting only when its endpoint is known, because the heartbeat settings sync keeps only those. + async Task SeedKnownEndpoints(params string[] names) + { + foreach (var name in names) + { + await MonitoringDataStore.CreateIfNotExists(new EndpointDetails { Name = name, HostId = Guid.NewGuid(), Host = "HOST01" }); + } + } + + [Test] + public async Task Two_names_in_one_batch_differing_only_in_case_are_each_copied_or_already_present() + { + await SeedKnownEndpoints("Sales", "sales"); + + var result = await Target.Write( + EndpointSettingsCategory, + BatchOf( + ("EndpointSettings/1", new EndpointSettings { Name = "Sales", TrackInstances = true }), + ("EndpointSettings/2", new EndpointSettings { Name = "sales", TrackInstances = false })), + CheckpointAfter("EndpointSettings/2")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied + result.AlreadyPresent, Is.EqualTo(2), "SQL Server's default collation makes these one key and PostgreSQL makes them two; either way neither row may throw or go uncounted"); + Assert.That(result.Skipped, Is.Zero, "a merge onto a row the batch writes loses nothing the target lacks, and a skip would count toward the halt threshold"); + Assert.That(await Target.Count(EndpointSettingsCategory), Is.EqualTo(result.Copied), "every row reported as copied is a row in the table"); + Assert.That(await TrackInstancesFor("Sales"), Is.True, "the first row in document-id order wins where the key makes the two one"); + } + } + +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/KnownEndpointsMigrationTargetTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/KnownEndpointsMigrationTargetTests.cs new file mode 100644 index 0000000000..40ab4305e2 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/KnownEndpointsMigrationTargetTests.cs @@ -0,0 +1,200 @@ +namespace ServiceControl.Persistence.Tests; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.EntityFrameworkCore; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceControl.Operations; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.DataMigration; +using ServiceControl.Persistence.EFCore.DbContexts; + +class KnownEndpointsMigrationTargetTests : PersistenceTestBase +{ + [SetUp] + public Task OpenTarget() => Target.Open(); + + IMigrationTarget Target => ServiceProvider.GetRequiredService(); + + static readonly MigrationCategory KnownEndpointsCategory = MigrationCategoryRegistry.All.Single(category => category.Id == MigrationCategoryIds.KnownEndpoints); + + [Test] + public async Task A_known_endpoint_is_written_with_its_monitored_flag() + { + var endpoint = new KnownEndpoint + { + EndpointDetails = new EndpointDetails { Name = "Sales.Orders", HostId = Guid.NewGuid(), Host = "SALES01" }, + HostDisplayName = "SALES01", + Monitored = true + }; + var sourceId = $"KnownEndpoints/{endpoint.EndpointDetails.GetDeterministicId()}"; + + var result = await Target.Write(KnownEndpointsCategory, BatchOf((sourceId, endpoint)), CheckpointAfter(sourceId)); + + var stored = (await MonitoringDataStore.GetAllKnownEndpoints()).Single(); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.EqualTo(1)); + Assert.That(stored.Monitored, Is.True); + Assert.That(stored.EndpointDetails.Name, Is.EqualTo("Sales.Orders")); + } + } + + [Test] + public async Task A_known_endpoint_already_in_the_target_keeps_its_flag_and_is_counted_as_already_present() + { + var details = new EndpointDetails { Name = "Sales.Orders", HostId = Guid.NewGuid(), Host = "SALES01" }; + await MonitoringDataStore.CreateIfNotExists(details); + + var endpoint = new KnownEndpoint { EndpointDetails = details, HostDisplayName = "SALES01", Monitored = true }; + var sourceId = $"KnownEndpoints/{details.GetDeterministicId()}"; + + var result = await Target.Write(KnownEndpointsCategory, BatchOf((sourceId, endpoint)), CheckpointAfter(sourceId)); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.Zero); + Assert.That(result.AlreadyPresent, Is.EqualTo(1)); + Assert.That((await MonitoringDataStore.GetAllKnownEndpoints()).Single().Monitored, Is.False, "insert-if-absent never updates, so the flag the target already had stays"); + } + } + + [Test] + public async Task A_known_endpoint_with_no_name_or_no_host_is_skipped_as_required_value_missing() + { + var named = new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = "Sales.Orders", HostId = Guid.NewGuid(), Host = "SALES01" } }; + var nameless = new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = null, HostId = Guid.NewGuid(), Host = "SALES02" } }; + var hostless = new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = "Billing", HostId = Guid.NewGuid(), Host = null } }; + + var result = await Target.Write(KnownEndpointsCategory, BatchOf(("KnownEndpoints/1", named), ("KnownEndpoints/2", nameless), ("KnownEndpoints/3", hostless)), CheckpointAfter("KnownEndpoints/3")); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.EqualTo(1)); + Assert.That(result.SkippedIds, Is.EquivalentTo(new[] { "KnownEndpoints/2", "KnownEndpoints/3" })); + Assert.That(result.SkipReasons[MigrationSkipReason.RequiredValueMissing], Is.EqualTo(2), "KnownEndpoints.Name and Host are NOT NULL, and a throw here would halt a required category"); + Assert.That((await MonitoringDataStore.GetAllKnownEndpoints()).Select(endpoint => endpoint.EndpointDetails.Name), Is.EqualTo(new[] { "Sales.Orders" })); + } + } + + [Test] + public async Task A_batch_of_nothing_but_skips_saves_the_cursor_and_writes_no_rows() + { + var nameless = new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = null, HostId = Guid.NewGuid(), Host = "SALES02" } }; + var hostless = new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = "Billing", HostId = Guid.NewGuid(), Host = null } }; + + var result = await Target.Write(KnownEndpointsCategory, BatchOf(("KnownEndpoints/1", nameless), ("KnownEndpoints/2", hostless)), CheckpointAfter("KnownEndpoints/2")); + + var stored = await ServiceProvider.GetRequiredService().Read(MigrationCategoryIds.KnownEndpoints); + + using (Assert.EnterMultipleScope()) + { + Assert.That(result.Copied, Is.Zero); + Assert.That(result.AlreadyPresent, Is.Zero, "no key was looked up, so no row may be counted as one the target already held"); + Assert.That(result.Skipped, Is.EqualTo(2)); + Assert.That(result.BenignSkipped, Is.Zero, "a missing NOT NULL column is a fault, and faults have to reach the halt threshold"); + Assert.That(await MonitoringDataStore.GetAllKnownEndpoints(), Is.Empty); + Assert.That(stored.Cursor, Is.EqualTo("KnownEndpoints/2"), "a batch that copied nothing still has to commit the cursor past it, or the restart reads the same rows forever"); + } + } + + [Test] + public void A_batch_reporting_more_copied_and_skipped_rows_than_it_held_is_refused() + { + // The guard only counts rows, so what the documents hold cannot change its answer. + var batch = BatchOf(("KnownEndpoints/1", new object()), ("KnownEndpoints/2", new object())); + + var exception = Assert.Throws(() => EFCoreMigrationTarget.AlreadyPresentIn(KnownEndpointsCategory, batch, copied: 2, skipped: 1)); + + Assert.That(exception.Message, Does.Contain("copied 2").And.Contain("skipped 1").And.Contain(MigrationCategoryIds.KnownEndpoints)); + } + + [Test] + public async Task A_batch_whose_checkpoint_save_fails_leaves_no_rows_behind() + { + // The rows save first and the checkpoint second, so this is the only order in which the two can part + // company: a stale version fails the checkpoint after the endpoint row is already in the transaction. + var store = ServiceProvider.GetRequiredService(); + await store.Upsert(CheckpointAfter("KnownEndpoints/0")); + + var endpoint = new KnownEndpoint + { + EndpointDetails = new EndpointDetails { Name = "Sales.Orders", HostId = Guid.NewGuid(), Host = "SALES01" }, + Monitored = true + }; + var sourceId = $"KnownEndpoints/{endpoint.EndpointDetails.GetDeterministicId()}"; + + Assert.ThrowsAsync(async () => + await Target.Write(KnownEndpointsCategory, BatchOf((sourceId, endpoint)), CheckpointAfter(sourceId))); + + using (Assert.EnterMultipleScope()) + { + Assert.That(await MonitoringDataStore.GetAllKnownEndpoints(), Is.Empty, "the rows and the checkpoint commit together, so a failed checkpoint takes the rows with it"); + Assert.That((await store.Read(MigrationCategoryIds.KnownEndpoints)).Cursor, Is.EqualTo("KnownEndpoints/0"), "the stored cursor must still describe the rows the target actually holds"); + } + } + + [Test] + public async Task Every_mapped_column_is_set_from_a_fully_populated_document() + { + var endpoint = new KnownEndpoint + { + EndpointDetails = new EndpointDetails { Name = "Sales.Orders", HostId = Guid.NewGuid(), Host = "SALES01" }, + HostDisplayName = "SALES01", + Monitored = true + }; + var sourceId = $"KnownEndpoints/{endpoint.EndpointDetails.GetDeterministicId()}"; + + await Target.Write(KnownEndpointsCategory, BatchOf((sourceId, endpoint)), CheckpointAfter(sourceId)); + + using var scope = ServiceProvider.CreateScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + MigrationEntityCoverage.AssertEveryMappedPropertyIsSet(dbContext.Model, await dbContext.KnownEndpoints.AsNoTracking().SingleAsync()); + } + + + [Test] + public async Task A_batch_spanning_several_statements_counts_what_each_statement_inserted() + { + var rows = Enumerable.Range(0, 500) + .Select(index => new KnownEndpoint { EndpointDetails = new EndpointDetails { Name = $"Endpoint{index}", HostId = Guid.NewGuid(), Host = "HOST01" }, Monitored = index % 2 == 0 }) + .Select(endpoint => ($"KnownEndpoints/{endpoint.EndpointDetails.GetDeterministicId()}", (object)endpoint)) + .ToArray(); + var batch = BatchOf(rows); + + var first = await Target.Write(KnownEndpointsCategory, batch, CheckpointAfter(batch.Cursor)); + // The engine carries the committed checkpoint into the next write, and the version guard refuses anything else. + var second = await Target.Write(KnownEndpointsCategory, batch, first.Saved with { Cursor = batch.Cursor }); + + using (Assert.EnterMultipleScope()) + { + Assert.That(first.Copied, Is.EqualTo(500)); + Assert.That(second.Copied, Is.Zero, "every key is present on the second write, so no statement may report a row it did not insert"); + Assert.That(second.AlreadyPresent, Is.EqualTo(500)); + } + } + + + [Test] + public void A_batch_whose_writer_neither_prepared_nor_skipped_a_row_is_refused() + { + // The already-present count is a subtraction, so it only catches a writer that over-reports. A row dropped in silence under-reports, and without this guard it is counted as a row the target already held. + var batch = BatchOf(("KnownEndpoints/1", new object()), ("KnownEndpoints/2", new object())); + var prepared = new PreparedBatch((_, _) => Task.FromResult(1), PreparedRowCount: 1, Skips: []); + + var exception = Assert.Throws(() => EFCoreMigrationTarget.AccountForEveryRow(KnownEndpointsCategory, batch, prepared)); + + Assert.That(exception.Message, Does.Contain("prepared 1").And.Contain("skipped 0").And.Contain(MigrationCategoryIds.KnownEndpoints)); + } + + static MigrationBatch BatchOf(params (string Id, object Document)[] rows) => + new([.. rows.Select(row => new MigrationRow(row.Id, row.Document, new Dictionary()))], rows[^1].Id); + + static MigrationCheckpoint CheckpointAfter(string cursor) => + new(MigrationCategoryIds.KnownEndpoints, MigrationCategoryState.InProgress, cursor, 0, 0, null, null, null, null, null, null); +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationCategoryWritersTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationCategoryWritersTests.cs new file mode 100644 index 0000000000..9e8956328c --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationCategoryWritersTests.cs @@ -0,0 +1,19 @@ +namespace ServiceControl.Persistence.Tests; + +using System.Linq; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; + +class MigrationCategoryWritersTests : PersistenceTestBase +{ + [Test] + public void Every_writer_claims_a_category_the_registry_knows() + { + var unknown = ServiceProvider.GetRequiredService().SupportedCategoryIds + .Where(id => MigrationCategoryRegistry.Find(id) is null) + .ToArray(); + + Assert.That(unknown, Is.Empty, "a writer for an id no category has is dead code the engine can never reach"); + } +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationEntityCoverage.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationEntityCoverage.cs new file mode 100644 index 0000000000..d3ede8f8ec --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationEntityCoverage.cs @@ -0,0 +1,34 @@ +namespace ServiceControl.Persistence.Tests; + +using System; +using System.Collections; +using System.Linq; +using Microsoft.EntityFrameworkCore.Metadata; +using NUnit.Framework; + +static class MigrationEntityCoverage +{ + public static void AssertEveryMappedPropertyIsSet(IModel model, TEntity entity, params string[] legitimatelyDefault) + { + var entityType = model.FindEntityType(typeof(TEntity)) + ?? throw new ArgumentException($"{typeof(TEntity).Name} is not an entity in the EF Core model.", nameof(entity)); + + // Reflection rather than a new() constraint, which an entity with required members cannot satisfy. + var defaults = Activator.CreateInstance(typeof(TEntity)); + + // Every property in this model that is not ValueGenerated.Never is an identity column the database assigns. + var unset = entityType.GetProperties() + .Where(property => property.ValueGenerated == ValueGenerated.Never && !legitimatelyDefault.Contains(property.Name)) + .Where(property => SameValue(property.GetGetter().GetClrValue(entity), property.GetGetter().GetClrValue(defaults))) + .Select(property => property.Name) + .ToArray(); + + Assert.That(unset, Is.Empty, $"{typeof(TEntity).Name} has mapped properties equal to a default-constructed {typeof(TEntity).Name}. Set each from the source document, feed the test a document whose values all differ from those defaults, or name the property in legitimatelyDefault if RavenDB genuinely holds that value."); + } + + // Collections compare by item, and an empty one is unset whether or not the entity initialises it. + static bool SameValue(object actual, object defaultValue) => + actual is IEnumerable items and not string + ? !items.Cast().Any() || (defaultValue is IEnumerable defaultItems && items.Cast().SequenceEqual(defaultItems.Cast())) + : Equals(actual, defaultValue); +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationEntityCoverageClaimTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationEntityCoverageClaimTests.cs new file mode 100644 index 0000000000..d5d0fedf42 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationEntityCoverageClaimTests.cs @@ -0,0 +1,33 @@ +namespace ServiceControl.Persistence.Tests; + +using System.Linq; +using Microsoft.EntityFrameworkCore.Metadata; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceControl.Persistence.EFCore.DbContexts; + +class MigrationEntityCoverageClaimTests : PersistenceTestBase +{ + [Test] + public void Every_property_the_database_fills_in_is_a_key_it_assigns_on_insert() + { + using var scope = ServiceProvider.CreateScope(); + var model = scope.ServiceProvider.GetRequiredService().Model; + + var databaseFilled = model.GetEntityTypes() + .SelectMany(entityType => entityType.GetProperties()) + .Where(property => property.ValueGenerated != ValueGenerated.Never) + .ToArray(); + + var notKeysAssignedOnInsert = databaseFilled + .Where(property => !property.IsPrimaryKey() || property.ValueGenerated != ValueGenerated.OnAdd) + .Select(property => $"{property.DeclaringType.ClrType.Name}.{property.Name} is {property.ValueGenerated}") + .ToArray(); + + using (Assert.EnterMultipleScope()) + { + Assert.That(databaseFilled, Is.Not.Empty, "the model no longer has a single property the database fills in, so this test is watching nothing. Either the identity keys have gone, or the model was never built."); + Assert.That(notKeysAssignedOnInsert, Is.Empty, "MigrationEntityCoverage checks only the properties that are ValueGenerated.Never, on the claim that every other one is a key the database assigns on insert. These properties break that claim, so the coverage check silently stops watching them and a migration can copy a row with them left unset."); + } + } +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationTargetCoverageTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationTargetCoverageTests.cs new file mode 100644 index 0000000000..da87ec7dcd --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationTargetCoverageTests.cs @@ -0,0 +1,26 @@ +namespace ServiceControl.Persistence.Tests; + +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.DependencyInjection; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; + +class MigrationTargetCoverageTests : PersistenceTestBase +{ + [Test] + public async Task Every_category_the_target_supports_has_a_batch_size_and_a_count() + { + var target = ServiceProvider.GetRequiredService(); + + Assert.That(target.SupportedCategoryIds, Is.SupersetOf(new[] { MigrationCategoryIds.KnownEndpoints, MigrationCategoryIds.EndpointSettings }), "the test proves nothing if the target supports no category"); + + foreach (var id in target.SupportedCategoryIds) + { + var category = MigrationCategoryRegistry.Find(id)!; + + Assert.That(await target.BatchSizeFor(category, CancellationToken.None), Is.Positive, $"{id}: every supported category has a batch size"); + Assert.That(await target.Count(category, CancellationToken.None), Is.Zero, id); + } + } +} diff --git a/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationTargetReadinessTests.cs b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationTargetReadinessTests.cs new file mode 100644 index 0000000000..cec366b9c6 --- /dev/null +++ b/src/ServiceControl.Persistence.Tests/EFCore/Migration/MigrationTargetReadinessTests.cs @@ -0,0 +1,166 @@ +namespace ServiceControl.Persistence.Tests; + +using System; +using System.IO; +using System.Linq; +using System.Threading.Tasks; +using Microsoft.EntityFrameworkCore; +using Microsoft.EntityFrameworkCore.Infrastructure; +using Microsoft.EntityFrameworkCore.Migrations; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Hosting; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.Persistence.EFCore.Abstractions; +using ServiceControl.Persistence.EFCore.DataMigration; +using ServiceControl.Persistence.EFCore.DbContexts; +using ServiceControl.Persistence.EFCore.Implementation.BodyStorage; +using ServiceControl.Persistence.EFCore.Infrastructure; + +class MigrationTargetReadinessTests : PersistenceTestBase +{ + IMigrationTargetReadiness Readiness => ServiceProvider.GetRequiredService(); + + [Test] + public void The_target_contributes_the_three_checks_only_it_can_make() => + Assert.That( + Readiness.ContributedChecks().Select(check => check.GetType()), + Is.EqualTo(new[] { typeof(RetryHistoryDepthIsSafeCheck), typeof(SchemaIsCurrentCheck), typeof(BodyStorageIsWritableCheck) })); + + [Test] + public void Every_contributed_check_passes_against_a_migrated_database() + { + // PersistenceFactory.Create is what carries the host's depth onto the persister, and the test container never runs it. + PersistenceSettings.RetryHistoryDepth = 10; + + using (Assert.EnterMultipleScope()) + { + foreach (var check in Readiness.ContributedChecks()) + { + Assert.DoesNotThrowAsync(() => check.Run(), $"a healthy target must pass every contributed check, and it failed '{check.Name}'"); + } + } + } + + [Test] + public async Task The_schema_check_refuses_a_database_whose_migrations_have_not_been_applied() + { + using var scope = ServiceProvider.CreateScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + // EF's own history repository, because it is the only thing that knows where the history table is once the persister is given a schema. + var history = dbContext.GetService(); + + foreach (var row in await history.GetAppliedMigrationsAsync()) + { + await dbContext.Database.ExecuteSqlRawAsync(history.GetDeleteScript(row.MigrationId)); + } + + var exception = Assert.ThrowsAsync(() => Readiness.ContributedChecks().OfType().Single().Run()); + + Assert.That(exception.Message, Does.Contain(dbContext.Database.GetMigrations().First()).And.Contain("--setup")); + } + + [Test] + public async Task The_body_storage_check_leaves_no_probe_body_behind() + { + var bodyStorage = ServiceProvider.GetRequiredService(); + + await Readiness.ContributedChecks().Single(check => check.Name == "message body storage is writable").Run(); + + Assert.That(await bodyStorage.ReadBody("migration-writable-probe"), Is.Null, "a probe left in the store is a body the source never had, which a count comparison reads as an extra rather than a loss"); + } + + [Test] + public void The_body_storage_check_refuses_a_store_that_cannot_write() + { + var parentThatIsAFile = Path.Combine(Path.GetTempPath(), $"sc-not-a-directory-{Guid.NewGuid():n}"); + File.WriteAllText(parentThatIsAFile, string.Empty); + + try + { + var check = new BodyStorageIsWritableCheck(new FileSystemBodyStoragePersistence( + new FileSystemBodyStorageSettings { StoragePath = Path.Combine(parentThatIsAFile, "bodies") })); + + Assert.CatchAsync(() => check.Run()); + } + finally + { + File.Delete(parentThatIsAFile); + } + } + + [Test] + public void The_retry_history_depth_check_passes_at_the_default_depth() => + Assert.DoesNotThrowAsync(() => new RetryHistoryDepthIsSafeCheck(10).Run()); + + [Test] + public void The_retry_history_depth_check_refuses_a_depth_that_empties_the_table() + { + var exception = Assert.ThrowsAsync(() => new RetryHistoryDepthIsSafeCheck(0).Run()); + + Assert.That(exception.Message, Does.Contain("RetryHistoryDepth").And.Contain("HistoricRetryOperations")); + } + + // The copy runs before any hosted service starts, so a target that only works once one has started is broken exactly when the migration needs it. + [Test] + public async Task The_target_answers_from_a_container_whose_hosted_services_have_not_started() + { + var (host, context) = await BuildHostWithoutStarting(); + + try + { + await using var scope = host.Services.GetRequiredService().CreateAsyncScope(); + + Assert.That(await scope.ServiceProvider.GetRequiredService().EndpointSettings.LongCountAsync(), Is.Zero); + Assert.DoesNotThrowAsync(() => host.Services.GetRequiredService().ContributedChecks().Single(check => check.Name == "message body storage is writable").Run()); + } + finally + { + await context.TearDown(); + host.Dispose(); + } + } + + // PersistenceTestBase already started a host in SetUp, so this builds a second one over its own test database. + static async Task<(IHost Host, PersistenceTestsContext Context)> BuildHostWithoutStarting() + { + var context = new PersistenceTestsContext(); + var hostBuilder = Host.CreateApplicationBuilder(); + + await context.Setup(hostBuilder); + var host = hostBuilder.Build(); + await context.InstallSchema(host); + + return (host, context); + } + + [Test] + public async Task The_host_opened_record_is_absent_until_written_and_stays_after_a_second_write() + { + var readiness = ServiceProvider.GetRequiredService(); + + Assert.That(await readiness.HasHostOpened(), Is.False, "a target nothing has opened on must not claim otherwise"); + + await readiness.RecordHostOpened(); + var firstOpened = Now; + + AdvanceClock(TimeSpan.FromDays(7)); + await readiness.RecordHostOpened(); + + using (Assert.EnterMultipleScope()) + { + Assert.That(await readiness.HasHostOpened(), Is.True); + // The stored instant is what tells an operator the clean abort is over, so a later start must not move it forward. + Assert.That(await ReadHostOpenedAt(), Is.EqualTo(firstOpened)); + } + } + + async Task ReadHostOpenedAt() + { + using var scope = ServiceProvider.CreateScope(); + var dbContext = scope.ServiceProvider.GetRequiredService(); + + return await dbContext.GetSetting(SettingKeys.MigrationHostOpenedOnTarget); + } +} diff --git a/src/ServiceControl.Persistence.Tests/IPersistenceTestsContext.cs b/src/ServiceControl.Persistence.Tests/IPersistenceTestsContext.cs index 15d9b76981..5e238f2d22 100644 --- a/src/ServiceControl.Persistence.Tests/IPersistenceTestsContext.cs +++ b/src/ServiceControl.Persistence.Tests/IPersistenceTestsContext.cs @@ -10,6 +10,13 @@ public interface IPersistenceTestsContext { Task Setup(IHostApplicationBuilder hostBuilder); + /// + /// Puts the schema in place on the built host, before it is started, the way --setup does in + /// production. The EF Core persisters refuse to start against a database whose schema predates the + /// build, and that check runs before any hosted service, so migrating after the start is too late. + /// + Task InstallSchema(IHost host); + Task PostSetup(IHost host); Task TearDown(); diff --git a/src/ServiceControl.Persistence.Tests/PersistenceTestBase.cs b/src/ServiceControl.Persistence.Tests/PersistenceTestBase.cs index f065c190f6..dc5ae1178e 100644 --- a/src/ServiceControl.Persistence.Tests/PersistenceTestBase.cs +++ b/src/ServiceControl.Persistence.Tests/PersistenceTestBase.cs @@ -53,6 +53,7 @@ public async Task SetUp() host = hostBuilder.Build(); + await PersistenceTestsContext.InstallSchema(host); await host.StartAsync(); await PersistenceTestsContext.PostSetup(host); } diff --git a/src/ServiceControl.Persistence/DataMigration/CheckpointMigrationState.cs b/src/ServiceControl.Persistence/DataMigration/CheckpointMigrationState.cs new file mode 100644 index 0000000000..54735fa2db --- /dev/null +++ b/src/ServiceControl.Persistence/DataMigration/CheckpointMigrationState.cs @@ -0,0 +1,42 @@ +namespace ServiceControl.Persistence.DataMigration; + +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; + +/// +/// Answers from the saved checkpoints, read once before the host opens rather than queried live. +/// +public sealed class CheckpointMigrationState(IMigrationCheckpointStore? checkpointStore = null) : IMigrationState +{ + volatile bool anyCategoryIncomplete; + IReadOnlyCollection selectedCategoryIds = []; + + public bool AnyCategoryIncomplete => anyCategoryIncomplete; + + /// + /// Reads every checkpoint and fixes the answer for the life of the host. + /// + /// The categories this instance was asked to copy. They are passed in because the store holds a row only for a category some run has already started. + public async Task Seed(IReadOnlyCollection selectedIds, CancellationToken cancellationToken = default) + { + // Only a persister that can be migrated into keeps checkpoints, so no store means no migration has run. + if (checkpointStore is null) + { + return; + } + + selectedCategoryIds = selectedIds; + + Recompute(await checkpointStore.ReadAll(cancellationToken)); + } + + /// + /// Works the answer out again from the checkpoints given, without reading the store. + /// + /// Every checkpoint the store holds. A selected category with no checkpoint among them has not started, which is not finished. + public void Recompute(IReadOnlyList checkpoints) => + anyCategoryIncomplete = selectedCategoryIds.Any(id => + checkpoints.FirstOrDefault(checkpoint => checkpoint.CategoryId == id)?.State.IsFinished() != true); +} diff --git a/src/ServiceControl.Persistence/DataMigration/HaltThreshold.cs b/src/ServiceControl.Persistence/DataMigration/HaltThreshold.cs index 92d6a73ce7..e2d686ea15 100644 --- a/src/ServiceControl.Persistence/DataMigration/HaltThreshold.cs +++ b/src/ServiceControl.Persistence/DataMigration/HaltThreshold.cs @@ -1,8 +1,19 @@ namespace ServiceControl.Persistence.DataMigration; +/// +/// The two rules that decide a category has lost too much to keep copying. +/// public static class HaltThreshold { - // Both must be exceeded: the floor ignores a few bad rows in a small category, the percentage a small share of a large one. + /// + /// Whether the skipped rows are enough to stop the category. Both limits have to be passed, so the floor + /// ignores a few bad rows in a small category and the percentage ignores a small share of a large one. + /// + /// Rows skipped as faults. A skip the product would have dropped anyway does not belong here. + /// Rows dealt with over the same stretch as , copied, skipped and already present alike. + /// The share of skipped rows, as a percentage, that has to be passed. + /// The number of skipped rows that has to be passed before the percentage counts at all. + /// True when both limits are passed, which means the category stops. public static bool Exceeded(long skippedCount, long totalCount, int percentThreshold, int minimumFloor) { if (skippedCount <= minimumFloor || totalCount == 0) @@ -13,4 +24,15 @@ public static bool Exceeded(long skippedCount, long totalCount, int percentThres var percent = skippedCount * 100m / totalCount; return percent > percentThreshold; } + + /// + /// Whether more than half the rows were skipped, which catches a category too small ever to reach + /// 's floor. Ask it only at the end of a run, over that run or over every run + /// together. Part way through, the same ratio is no more than a bad first batch. + /// + /// Rows skipped as faults. + /// Rows dealt with over the same stretch as . + /// True when the skipped rows are more than half, which means the category stops. + public static bool MostOfItWasLost(long skippedCount, long totalCount) => + totalCount > 0 && skippedCount * 2 > totalCount; } diff --git a/src/ServiceControl.Persistence/DataMigration/IMigrationCheckpointStore.cs b/src/ServiceControl.Persistence/DataMigration/IMigrationCheckpointStore.cs index bc71283d01..309e908d1c 100644 --- a/src/ServiceControl.Persistence/DataMigration/IMigrationCheckpointStore.cs +++ b/src/ServiceControl.Persistence/DataMigration/IMigrationCheckpointStore.cs @@ -6,18 +6,45 @@ namespace ServiceControl.Persistence.DataMigration; using System.Threading; using System.Threading.Tasks; +/// +/// Where one category has got to. says which of these +/// let the host open, and a restart picks up every category that is not one of those. +/// public enum MigrationCategoryState { + /// No run has read a row of this category yet. NotStarted, + + /// A run is copying this category, or a run stopped without settling it. InProgress, + + /// The category reached the end of the source with nothing skipped. Complete, + + /// The category reached the end of the source, but some rows were skipped and stay only in the old database. CompleteWithErrors, + + /// Too many rows failed, so the category stopped and waits for a person. Fix the cause and restart to carry on from the cursor. Halted, + + /// A person accepted the loss and stopped copying this category. Nothing copies it again, even in a later migration. Abandoned, + + /// This category has to follow another one, and that one is not finished, so it did not run. A restart clears it once the other one settles. Blocked } -/// A category's saved progress: where a restart carries on from, and what the status and verify commands report. +/// +/// A category's saved progress: where a restart carries on from, and what the status and verify commands report. +/// Every count is the total across every run, not this run alone. +/// +/// The point the last committed batch reached. A restart reads the source after it. Null means nothing has been read. +/// How many rows the source held when the category first started. It is captured once, so a source that has grown since does not move it. +/// How many rows each reason skipped. The counts here add up to . +/// The moment the category stopped running, whatever state it stopped in. Read it beside , because a halt settles too. +/// Why the category halted, in the words the operator is shown. Null when it has not halted. +/// Rows the target already held, so they were neither copied nor skipped. They still count as accounted for when the run checks the category against . +/// The optimistic concurrency token, which is the guard against two writers. A store sets it on save and refuses a checkpoint carrying a value the stored row no longer holds. public sealed record MigrationCheckpoint( string CategoryId, MigrationCategoryState State, @@ -28,14 +55,21 @@ public sealed record MigrationCheckpoint( IReadOnlyDictionary? SkipReasons, DateTime? StartedAt, DateTime? LastProgressAt, - // The moment the category stopped running, whatever state it stopped in. Read it beside State: a halt settles too. DateTime? SettledAt, string? LastError, long AlreadyPresentCount = 0, - // The optimistic concurrency token. A store sets it on save and refuses one carrying a value the stored row no longer holds. long Version = 0) { - /// Adds one batch's outcome to this checkpoint. A target calls it inside the transaction that writes the rows, so the saved counts are the real ones. + /// + /// Adds one batch's outcome to this checkpoint. A target calls it inside the transaction that writes the rows, + /// so the saved counts are the real ones. + /// + /// Rows this batch wrote. + /// Rows this batch could not write. Every one of them needs a reason. + /// Rows this batch found the target already held. + /// How many rows each reason skipped in this batch. + /// A copy with this batch's counts and reasons added to the totals. + /// The reasons do not add up to , which would leave the verify command unable to account for a row. public MigrationCheckpoint Extend(int copied, int skipped, int alreadyPresent, IReadOnlyDictionary? skipReasons) { var explained = skipReasons?.Values.Sum() ?? 0; @@ -70,11 +104,26 @@ public MigrationCheckpoint Extend(int copied, int skipped, int alreadyPresent, I } } +/// +/// Where the checkpoints are kept. The target database holds them, so progress and the rows it describes +/// commit together. +/// public interface IMigrationCheckpointStore { + /// + /// Every checkpoint the store holds. A category no run has started has no row, so it is absent rather than + /// returned as not started. + /// Task> ReadAll(CancellationToken cancellationToken = default); + + /// + /// One category's checkpoint, or null when no run has started it. + /// Task Read(string categoryId, CancellationToken cancellationToken = default); - /// Saves the checkpoint and returns it as stored, carrying the version the save landed on. Throws when the stored row has moved on. + /// + /// Saves the checkpoint and returns it as stored, carrying the version the save landed on. + /// + /// The stored row has moved on, which means another writer saved it, so this save is refused. Task Upsert(MigrationCheckpoint checkpoint, CancellationToken cancellationToken = default); } diff --git a/src/ServiceControl.Persistence/DataMigration/IMigrationSource.cs b/src/ServiceControl.Persistence/DataMigration/IMigrationSource.cs index bd19e623aa..fef0c5a64a 100644 --- a/src/ServiceControl.Persistence/DataMigration/IMigrationSource.cs +++ b/src/ServiceControl.Persistence/DataMigration/IMigrationSource.cs @@ -5,21 +5,41 @@ namespace ServiceControl.Persistence.DataMigration; using System.Threading; using System.Threading.Tasks; +/// +/// Implemented by a persister that can be read as the old database a migration copies from. Nothing here +/// writes to the source, because the old database has to stay usable if the migration is thrown away. +/// public interface IMigrationSource : IAsyncDisposable { - /// Connects to the source read-only. Every other member throws until this has run. + /// + /// Connects to the source read-only. Every other member except throws + /// until this has run. + /// Task Open(CancellationToken cancellationToken = default); - /// What the source report and dry run print about the source. + /// + /// The checks this source wants run before the copy starts, in the order they must run. Call it after Open. + /// + IReadOnlyList ContributedChecks(); + + /// + /// What the source report and the dry run print about the source. + /// Task Describe(CancellationToken cancellationToken = default); - /// A count of everything the source holds, including data no category copies. + /// + /// A count of everything the source holds, including data no category copies. + /// Task> Inventory(CancellationToken cancellationToken = default); - /// How many rows the source holds for one category, for progress and verify. + /// + /// How many rows the source holds for one category, for progress and verify. + /// Task Count(MigrationCategory category, CancellationToken cancellationToken = default); - /// Reads a category in batches, after the checkpoint cursor if provided, or from the start when it is null. Throws on a cursor it never issued. + /// + /// Reads a category in batches, after the checkpoint cursor, or from the start when it is null. Throws on a cursor it never issued. + /// /// A ceiling, not a target: returning fewer costs nothing, returning more fails the target's write. IAsyncEnumerable Read( MigrationCategory category, @@ -27,6 +47,14 @@ IAsyncEnumerable Read( int batchSize, CancellationToken cancellationToken = default); - /// Reads one row's message body when Read did not attach it. Returns null when the row has no body. + /// + /// Reads one row's message body when Read did not attach it. Returns null when the row has no body. + /// Task ReadBody(MigrationCategory category, string sourceId, CancellationToken cancellationToken = default); + + /// + /// The categories this source can read. A category outside this set is never handed to the engine. + /// Answers before Open and does not change across it. + /// + IReadOnlyCollection SupportedCategoryIds { get; } } diff --git a/src/ServiceControl.Persistence/DataMigration/IMigrationStartupCheck.cs b/src/ServiceControl.Persistence/DataMigration/IMigrationStartupCheck.cs new file mode 100644 index 0000000000..bd08c1df27 --- /dev/null +++ b/src/ServiceControl.Persistence/DataMigration/IMigrationStartupCheck.cs @@ -0,0 +1,21 @@ +namespace ServiceControl.Persistence.DataMigration; + +using System.Threading; +using System.Threading.Tasks; + +/// +/// One check the host runs before the copy starts. The host, the source and the target each contribute their own. +/// +public interface IMigrationStartupCheck +{ + /// + /// What the check is called in the refusal the host prints, so it is phrased to finish the sentence + /// "Migration startup check '...' failed". + /// + string Name { get; } + + /// + /// Runs the check. Throws when it fails, which stops the host before anything is copied. + /// + Task Run(CancellationToken cancellationToken = default); +} diff --git a/src/ServiceControl.Persistence/DataMigration/IMigrationState.cs b/src/ServiceControl.Persistence/DataMigration/IMigrationState.cs new file mode 100644 index 0000000000..93260cd7ca --- /dev/null +++ b/src/ServiceControl.Persistence/DataMigration/IMigrationState.cs @@ -0,0 +1,12 @@ +namespace ServiceControl.Persistence.DataMigration; + +/// +/// Whether this instance is still part way through copying its old database in. +/// +public interface IMigrationState +{ + /// + /// True while any selected category is unfinished. Services that delete or overwrite copied data stand down while it is true. + /// + bool AnyCategoryIncomplete { get; } +} diff --git a/src/ServiceControl.Persistence/DataMigration/IMigrationTarget.cs b/src/ServiceControl.Persistence/DataMigration/IMigrationTarget.cs index 78026fc571..a961f99783 100644 --- a/src/ServiceControl.Persistence/DataMigration/IMigrationTarget.cs +++ b/src/ServiceControl.Persistence/DataMigration/IMigrationTarget.cs @@ -4,24 +4,54 @@ namespace ServiceControl.Persistence.DataMigration; using System.Threading; using System.Threading.Tasks; -/// Implemented by a persister that can be the new database a migration copies into. +/// +/// Implemented by a persister that can be the new database a migration copies into. +/// public interface IMigrationTarget { - /// The most rows a source may return in one batch for this category. The target picks it because its own database sets the limit, and a batch over it fails the write. - int BatchSizeFor(MigrationCategory category); + /// + /// Makes the target ready to write. The host calls it before the copy and after the target's own checks have passed. + /// + Task Open(CancellationToken cancellationToken = default); - /// Saves the batch's rows and the checkpoint in one transaction, so progress never gets ahead of the data. Extend checkpointToExtend with this batch's own outcome through and save the result, so what lands is the real split rather than a guess the next save has to correct. - /// Prior totals, the cursor this batch reached, and any rows the engine itself skipped. Not yet counting anything the target does. + /// + /// The most rows a source may return in one batch for this category. + /// The target picks it because its own database sets the limit, and a bigger batch fails the write. + /// + Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default); + + /// + /// Saves the batch's rows and the checkpoint in one transaction, so progress never gets ahead of the data. + /// Add this batch's own outcome with and save that, so the stored counts are never a guess. + /// + /// Prior totals, the cursor this batch reached, and any rows the engine itself skipped. Nothing the target does is counted yet. Task Write( MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default); - /// How many rows the target holds for one category, for progress and verify. Counts only that category, even where two categories share a table. + /// + /// How many rows the target holds for one category, for progress and verify. + /// Counts only that category, even where two categories share a table. + /// Task Count(MigrationCategory category, CancellationToken cancellationToken = default); + + /// + /// The categories this target can write. A category outside this set is never handed to the engine. + /// Answers before Open and does not change across it. + /// + IReadOnlyCollection SupportedCategoryIds { get; } } -/// What the target did with one batch, and the checkpoint it committed alongside the rows. Every skipped row must have a reason in SkipReasons. -/// How many of Skipped the target would have deleted anyway, such as a row already past retention. Counted and reported like any skip, but never counted toward the halt threshold. +/// +/// What the target did with one batch, and the checkpoint it committed alongside the rows. Every skipped row must have a reason in SkipReasons. +/// +/// The checkpoint as the target stored it, carrying the version that save landed on. The engine carries on from this one, never from the one it passed in. +/// Rows this batch wrote. The same number must show up as the rise in the saved copied count, because the halt threshold reads one and the end-of-run reconciliation the other. +/// Rows this batch could not write, benign ones included. +/// The source ids of those rows, so the engine can name each one in the log. +/// Rows the target already held. They were neither copied nor skipped, and they still count as accounted for. +/// How many rows each reason skipped. The counts have to add up to Skipped, or the checkpoint refuses the batch. +/// How many of Skipped the target would have deleted anyway, such as a row already past retention. Reported like any other skip, but never counted toward the halt threshold. public sealed record MigrationWriteResult(MigrationCheckpoint Saved, int Copied, int Skipped, IReadOnlyList SkippedIds, int AlreadyPresent = 0, IReadOnlyDictionary? SkipReasons = null, int BenignSkipped = 0); diff --git a/src/ServiceControl.Persistence/DataMigration/IMigrationTargetReadiness.cs b/src/ServiceControl.Persistence/DataMigration/IMigrationTargetReadiness.cs new file mode 100644 index 0000000000..8cde500f8f --- /dev/null +++ b/src/ServiceControl.Persistence/DataMigration/IMigrationTargetReadiness.cs @@ -0,0 +1,26 @@ +namespace ServiceControl.Persistence.DataMigration; + +using System.Collections.Generic; +using System.Threading; +using System.Threading.Tasks; + +/// +/// The target's own startup checks, plus a marker kept in the target database saying a host has already started on it. +/// +public interface IMigrationTargetReadiness +{ + /// + /// The checks this target wants run before the copy starts, in the order they must run. + /// + IReadOnlyList ContributedChecks(); + + /// + /// Stamps the marker the first time a host opens on the target, and leaves that first stamp alone on every start after it. + /// + Task RecordHostOpened(CancellationToken cancellationToken = default); + + /// + /// Whether a host has already opened on the target, which is what makes discarding a partial copy a loss rather than a clean abort. + /// + Task HasHostOpened(CancellationToken cancellationToken = default); +} diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationBatch.cs b/src/ServiceControl.Persistence/DataMigration/MigrationBatch.cs index 431730ba82..7a36324b40 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationBatch.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationBatch.cs @@ -3,9 +3,17 @@ namespace ServiceControl.Persistence.DataMigration; using System; using System.Collections.Generic; +/// +/// When a category is copied. Required categories are copied with ServiceControl closed, because the host must +/// not serve a half copied instance. Optional ones are copied in the background once it is open. +/// public enum MigrationCategoryKind { Required, Optional } /// One kind of data, such as endpoint settings, copied as a unit and resumed from its own cursor. +/// The name in , which is also the key of its checkpoint row. +/// Whether its rows have message bodies, which the engine fetches separately when the source does not attach them. +/// The order within one kind, counting from 1. Required and optional both start at 1, so the two lists are never sorted together. +/// The category that has to finish first, or null when nothing has to. A category whose predecessor is unfinished is recorded as instead of running. public sealed record MigrationCategory( string Id, MigrationCategoryKind Kind, @@ -16,7 +24,11 @@ public sealed record MigrationCategory( /// A message body read from the source, as bytes plus its content type. public sealed record MigrationBody(ReadOnlyMemory Content, string ContentType); -/// One item read from the source. Body is null unless the source attached it or the engine fetched it. +/// One item read from the source. +/// The row's identifier in the old database. It is what the engine logs when the row is skipped, and what takes. +/// The row itself, as the object the source read. The target's writer for that category is what knows the type. +/// Facts about the row that the document does not carry. Empty when the source has none to add. +/// The message body, or null when the source did not attach one and the engine has not fetched it. public sealed record MigrationRow( string SourceId, object Document, @@ -24,4 +36,6 @@ public sealed record MigrationRow( MigrationBody? Body = null); /// Rows read from the source together, plus the cursor to resume after them. +/// The rows, which can be empty when every document in this stretch turned into no row. The cursor past them still has to be saved. +/// Where the source got to. A later read handed this back carries on after it. public sealed record MigrationBatch(IReadOnlyList Rows, string Cursor); diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationCategoryIds.cs b/src/ServiceControl.Persistence/DataMigration/MigrationCategoryIds.cs index 6f15bcdf24..3a21f36ddd 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationCategoryIds.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationCategoryIds.cs @@ -1,5 +1,10 @@ namespace ServiceControl.Persistence.DataMigration; +/// +/// The name of every category a migration can copy. The name is what the checkpoint row is keyed on and what an +/// operator writes in , so renaming one strands the progress +/// already saved under the old name. says when each one is copied. +/// public static class MigrationCategoryIds { public const string KnownEndpoints = nameof(KnownEndpoints); diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationCategoryRegistry.cs b/src/ServiceControl.Persistence/DataMigration/MigrationCategoryRegistry.cs index 69665e34e1..fbda3cef1f 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationCategoryRegistry.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationCategoryRegistry.cs @@ -4,9 +4,23 @@ namespace ServiceControl.Persistence.DataMigration; using System.Linq; using ServiceControl.MessageFailures; +/// +/// Every category a migration can copy, and when each one is copied. This list is the whole of what a migration +/// covers: a source or a target that cannot handle a category says so through its own supported ids, and the +/// engine never invents one. +/// public static class MigrationCategoryRegistry { + /// + /// The failed message statuses that are copied before the host opens. A message in one of these is waiting on + /// somebody, so an operator who cannot see it after the cutover has lost work. + /// public static readonly IReadOnlyList UnresolvedAndRetryIssuedStatuses = [FailedMessageStatus.Unresolved, FailedMessageStatus.RetryIssued]; + + /// + /// The failed message statuses that are copied in the background. These are history, so the instance is usable + /// while they are still arriving. + /// public static readonly IReadOnlyList ArchivedAndResolvedStatuses = [FailedMessageStatus.Archived, FailedMessageStatus.Resolved]; public static readonly IReadOnlyList All = diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationCategoryStateExtensions.cs b/src/ServiceControl.Persistence/DataMigration/MigrationCategoryStateExtensions.cs new file mode 100644 index 0000000000..dd3911c7b1 --- /dev/null +++ b/src/ServiceControl.Persistence/DataMigration/MigrationCategoryStateExtensions.cs @@ -0,0 +1,23 @@ +namespace ServiceControl.Persistence.DataMigration; + +public static class MigrationCategoryStateExtensions +{ + /// + /// Whether the category is done with, so a restart passes over it and it no longer holds the host back. + /// is deliberately not finished. A halt says "stopped, and here + /// is why", and a restart once the cause is fixed has to be able to pick the category up again. + /// + public static bool IsFinished(this MigrationCategoryState state) => + state is MigrationCategoryState.Complete + or MigrationCategoryState.CompleteWithErrors + or MigrationCategoryState.Abandoned; +} + +public static class MigrationSkipReasonExtensions +{ + /// + /// Whether the running product would have dropped this row anyway, which makes the skip no loss. The halt + /// threshold never counts a benign skip, so no number of them can stop a category. + /// + public static bool IsBenign(this MigrationSkipReason reason) => reason is MigrationSkipReason.EndpointNotKnown; +} diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationEngine.cs b/src/ServiceControl.Persistence/DataMigration/MigrationEngine.cs index 3503321796..6be3813c9b 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationEngine.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationEngine.cs @@ -7,6 +7,13 @@ namespace ServiceControl.Persistence.DataMigration; using System.Threading.Tasks; using Microsoft.Extensions.Logging; +/// +/// Copies one category at a time from the source to the target, and records where it got to after every batch. +/// The engine knows nothing about either database: what a row is, how it is written and how big a batch can be +/// all come from the source and the target. A category that fails comes back as a halted checkpoint rather than +/// an exception. A shutdown and a checkpoint conflict are the two things that do come out as exceptions, because +/// neither is the category's fault. +/// public sealed class MigrationEngine( IMigrationSource source, IMigrationTarget target, @@ -15,8 +22,13 @@ public sealed class MigrationEngine( MigrationEngineOptions options, ILogger logger) { + /// How many times a message body is read before the row is skipped as unreadable. public const int MaxBodyReadAttempts = 3; + /// + /// The categories of one kind to copy, in the order to copy them. Optional categories the operator did not + /// ask for are left out. + /// public IReadOnlyList SelectCategories(MigrationCategoryKind kind) => MigrationCategoryRegistry.All .Where(c => c.Kind == kind) @@ -24,6 +36,10 @@ public IReadOnlyList SelectCategories(MigrationCategoryKind k .OrderBy(c => c.Order) .ToArray(); + /// + /// Copies the categories one after another and returns where each one ended, in the same order. A category + /// that halts does not stop the ones after it. + /// // Runs in the order given without re-sorting: required and optional orders both start at 1, so // sorting a mixed list would put an optional category in front of a required one. public async Task> RunCategories( @@ -40,12 +56,18 @@ public async Task> RunCategories( return results; } + /// + /// Copies one category, carrying on from its saved cursor, and returns the checkpoint it ended on. A category + /// already finished is returned untouched without reading the source. + /// + /// The stored checkpoint, whose state says how it ended and whose LastError says why it halted. + /// Another writer saved this category's checkpoint, which means a second instance is copying into the same database. The state is left as that writer set it. + /// The host is shutting down. The last committed batch saved its own counts, so the stored checkpoint is already correct and a restart carries on from it. public async Task RunCategoryAsync(MigrationCategory category, CancellationToken cancellationToken = default) { var checkpoint = await checkpointStore.Read(category.Id, cancellationToken) ?? NotStarted(category); - // Halted is deliberately not one of them: a halt says "stopped, and here is why", and a restart after the cause is fixed has to be able to pick it up again. - if (checkpoint.State is MigrationCategoryState.Complete or MigrationCategoryState.CompleteWithErrors or MigrationCategoryState.Abandoned) + if (checkpoint.State.IsFinished()) { return checkpoint; } @@ -53,7 +75,7 @@ public async Task RunCategoryAsync(MigrationCategory catego if (category.MustFollow is { } mustFollowId) { var predecessor = await checkpointStore.Read(mustFollowId, cancellationToken); - if (predecessor is not { State: MigrationCategoryState.Complete or MigrationCategoryState.CompleteWithErrors or MigrationCategoryState.Abandoned }) + if (predecessor?.State.IsFinished() != true) { var predecessorState = predecessor?.State.ToString() ?? "not started"; var blocked = checkpoint with @@ -61,8 +83,10 @@ public async Task RunCategoryAsync(MigrationCategory catego State = MigrationCategoryState.Blocked, LastError = $"Blocked: {category.Id} must follow {mustFollowId}, which is {predecessorState}" }; + logger.LogWarning("Category {CategoryId} did not run: it must follow {PredecessorId}, which is {PredecessorState}", category.Id, mustFollowId, predecessorState); + return await checkpointStore.Upsert(blocked, cancellationToken); } } @@ -73,6 +97,7 @@ public async Task RunCategoryAsync(MigrationCategory catego { State = MigrationCategoryState.InProgress, StartedAt = checkpoint.StartedAt ?? timeProvider.GetUtcNow().UtcDateTime, + LastProgressAt = timeProvider.GetUtcNow().UtcDateTime, SettledAt = null, LastError = null }, cancellationToken); @@ -86,7 +111,21 @@ public async Task RunCategoryAsync(MigrationCategory catego try { - var batchSize = target.BatchSizeFor(category); + var batchSize = await target.BatchSizeFor(category, cancellationToken); + + // Captured once so a restart keeps the total its first start saw, and saved straight away so a run + // killed during the first batch does not walk the whole category again to recount it. + if (checkpoint.SourceTotal is null) + { + var countedRows = await source.Count(category, cancellationToken); + + checkpoint = await checkpointStore.Upsert(checkpoint with + { + SourceTotal = countedRows, + LastProgressAt = timeProvider.GetUtcNow().UtcDateTime + }, cancellationToken); + } + await foreach (var batch in source.Read(category, checkpoint.Cursor, batchSize, cancellationToken).WithCancellation(cancellationToken)) { if (!isFirstBatch && category.Kind == MigrationCategoryKind.Optional) @@ -96,8 +135,8 @@ public async Task RunCategoryAsync(MigrationCategory catego isFirstBatch = false; var batchToWrite = batch; - // Stays off checkpoint until the write commits: the catch persists checkpoint, and a restart - // re-reads an uncommitted batch and would count these skips again. + // Kept out of checkpoint until the write commits: the catch below saves checkpoint, and a + // restart re-reads the uncommitted batch and would count these skips twice. var bodySkips = 0; if (category.CarriesBodies) { @@ -111,18 +150,30 @@ public async Task RunCategoryAsync(MigrationCategory catego } } - // Prior totals, the new cursor, and the rows this engine already skipped. The target adds its own - // outcome inside the transaction that writes the rows, so nothing provisional is ever stored. + // The target adds its own outcome inside the transaction that writes the rows, + // so nothing provisional is ever stored. var checkpointToExtend = checkpoint with { Cursor = batch.Cursor, + LastProgressAt = timeProvider.GetUtcNow().UtcDateTime, SkippedCount = checkpoint.SkippedCount + bodySkips, SkipReasons = MigrationCheckpoint.AddSkipReasons(checkpoint.SkipReasons, bodySkips == 0 ? null : new Dictionary { [MigrationSkipReason.BodyUnreadable] = bodySkips }) }; var result = await target.Write(category, batchToWrite, checkpointToExtend, cancellationToken); + + // The target has committed this row, so the engine takes it before anything below can throw. A halt + // settling from the older version would be refused by the store as a conflict and lose its reason. checkpoint = result.Saved; + // The result states the batch's outcome twice, as its own counts and as deltas on the checkpoint + // it committed. The halt threshold reads the first and the end-of-run reconciliation the second. + var committed = (result.Saved.CopiedCount - checkpointToExtend.CopiedCount, result.Saved.SkippedCount - checkpointToExtend.SkippedCount, result.Saved.AlreadyPresentCount - checkpointToExtend.AlreadyPresentCount); + if (committed != (result.Copied, result.Skipped, result.AlreadyPresent)) + { + throw new InvalidOperationException($"The target reported copying {result.Copied}, skipping {result.Skipped} and finding {result.AlreadyPresent} already present in category {category.Id}, but the checkpoint it committed moved by {committed}. The halt threshold judges the first and the end-of-run reconciliation the second, so they cannot differ."); + } + foreach (var id in result.SkippedIds) { logger.LogWarning("Skipped {SourceId} in category {CategoryId}", id, category.Id); @@ -135,7 +186,6 @@ public async Task RunCategoryAsync(MigrationCategory catego } processedThisRun += bodySkips + result.Copied + result.Skipped + result.AlreadyPresent; - // Rows the target would have deleted anyway are not faults, so they never halt a category. skippedThisRun += bodySkips + result.Skipped - result.BenignSkipped; if (HaltThreshold.Exceeded(skippedThisRun, processedThisRun, options.HaltThresholdPercent, options.HaltThresholdMinimum)) @@ -146,8 +196,8 @@ public async Task RunCategoryAsync(MigrationCategory catego } } } - // A shutdown is not a halt, and there is nothing to reconcile: the last committed batch stored its - // real split with its own rows, so the row on disk is already correct and resumable. + // A shutdown is not a halt: the last committed batch saved its counts with its own rows, + // so the checkpoint on disk is already correct and resumable. catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) { throw; @@ -165,9 +215,44 @@ public async Task RunCategoryAsync(MigrationCategory catego return await Settle(checkpoint with { State = MigrationCategoryState.Halted, LastError = reason }, cancellationToken); } + var accountedFor = checkpoint.CopiedCount + checkpoint.SkippedCount + checkpoint.AlreadyPresentCount; + + // The counts on the row are cumulative, so a resumed run is judged on the whole category. Only a + // shortfall is a fault: a source that has grown since it was counted is the instance still running. + if (checkpoint.SourceTotal is { } sourceTotal && accountedFor < sourceTotal) + { + var reason = $"Halted: {category.Id} reached the end of the source having accounted for {accountedFor} of the {sourceTotal} rows the source reported: {checkpoint.CopiedCount} copied, {checkpoint.SkippedCount} skipped, {checkpoint.AlreadyPresentCount} already present. A cursor that no longer matches the source is the usual cause. Fix the cause and restart, or abandon the category to accept the loss."; + logger.LogError("Category {CategoryId} halted at cursor {Cursor}: {LastError}", category.Id, checkpoint.Cursor, reason); + return await Settle(checkpoint with { State = MigrationCategoryState.Halted, LastError = reason }, cancellationToken); + } + + // The per-batch threshold cannot see this: a category smaller than the halt floor never reaches the + // floor however much of it is lost, so losing most of it reads as complete with a few errors. + if (HaltThreshold.MostOfItWasLost(skippedThisRun, processedThisRun)) + { + var reason = $"Halted: {category.Id} reached the end of the source having skipped {skippedThisRun} of the {processedThisRun} rows this run processed, which is most of them. A category this small never reaches the threshold of {options.HaltThresholdMinimum} rows, so losing most of it is judged on its own. Fix the cause and restart to resume from the cursor, or abandon the category to accept the loss."; + logger.LogError("Category {CategoryId} halted at cursor {Cursor}: {LastError}", category.Id, checkpoint.Cursor, reason); + return await Settle(checkpoint with { State = MigrationCategoryState.Halted, LastError = reason }, cancellationToken); + } + + // A halted category resumes from a cursor already at the end of the source, so the restart it was told + // to do reads nothing and the per-run rule above can judge nothing. Judge it on every run's totals + // instead, or a category that lost all of its rows settles as finished the moment it is restarted. + var lostRows = FaultSkips(checkpoint); + if (processedThisRun == 0 && HaltThreshold.MostOfItWasLost(lostRows, accountedFor)) + { + var reason = $"Halted: {category.Id} has accounted for {accountedFor} rows across every run, {lostRows} of them skipped as faults, which is most of them. This run reached the end of the source without reading a row, so restarting again cannot change it. Fix the cause and copy the category to a fresh target database, or abandon it to accept the loss."; + logger.LogError("Category {CategoryId} stays halted at cursor {Cursor}: {LastError}", category.Id, checkpoint.Cursor, reason); + return await Settle(checkpoint with { State = MigrationCategoryState.Halted, LastError = reason }, cancellationToken); + } + return await Settle(checkpoint with { State = checkpoint.SkippedCount > 0 ? MigrationCategoryState.CompleteWithErrors : MigrationCategoryState.Complete }, cancellationToken); } + // Skips of rows the product would have removed anyway are not losses, so they cannot make a category look lost. + static long FaultSkips(MigrationCheckpoint checkpoint) => + checkpoint.SkipReasons is null ? 0 : checkpoint.SkipReasons.Where(reason => !reason.Key.IsBenign()).Sum(reason => reason.Value); + // Halts log before settling: the store shares the target's database, so a failed save would hide the cause. Task Settle(MigrationCheckpoint settled, CancellationToken cancellationToken) => checkpointStore.Upsert(settled with { SettledAt = timeProvider.GetUtcNow().UtcDateTime }, cancellationToken); @@ -230,8 +315,8 @@ Task Settle(MigrationCheckpoint settled, CancellationToken static bool IsDefect(Exception exception) => exception is NotSupportedException or NotImplementedException or InvalidOperationException or ArgumentException or NullReferenceException or InvalidCastException; - // A configured pause of zero means "do not throttle", and a timer that is never going to be - // waited on is worse than no timer: against a fake clock nobody advances, it never completes. + // Zero means no throttling, and it must return without waiting: against a fake clock + // nobody advances, even a zero-length wait never finishes. Task Pause(TimeSpan duration, CancellationToken cancellationToken) => duration <= TimeSpan.Zero ? Task.CompletedTask : Task.Delay(duration, timeProvider, cancellationToken); diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationEngineOptions.cs b/src/ServiceControl.Persistence/DataMigration/MigrationEngineOptions.cs index 8433836893..b53a5404a2 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationEngineOptions.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationEngineOptions.cs @@ -6,6 +6,10 @@ namespace ServiceControl.Persistence.DataMigration; using ServiceControl.Configuration; /// The engine's tuning: the pause between optional batches, when a category halts, and which optional categories to copy. +/// How long to wait between batches of an optional category, which is what keeps the background copy off the instance's back. Zero waits not at all. +/// The share of skipped rows, as a percentage, that stops a category. +/// The number of skipped rows that has to be passed before the percentage counts, so a handful of bad rows in a small category is not a halt. +/// The optional categories the operator asked for. An optional category outside this list is never copied. public sealed record MigrationEngineOptions( TimeSpan ThrottlePause, int HaltThresholdPercent, @@ -17,7 +21,11 @@ public sealed record MigrationEngineOptions( /// How long to wait before trying again to read a message body that failed. Not read from settings. public TimeSpan BodyRetryBackoff { get; init; } = DefaultBodyRetryBackoff; - /// Reads the options from settings. Throws if the optional categories setting names one that doesn't exist or isn't optional. + /// + /// Reads the options from the instance's settings. Anything not set takes the default from + /// . + /// + /// The optional categories setting names a category that does not exist or is not optional, which is a typo the operator has to see before the copy starts. public static MigrationEngineOptions FromSettings(SettingsRootNamespace settingsRootNamespace) { var throttleMilliseconds = SettingsReader.Read(settingsRootNamespace, MigrationSettings.ThrottlePauseMillisecondsKey, MigrationSettings.DefaultThrottlePauseMilliseconds); diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationSettings.cs b/src/ServiceControl.Persistence/DataMigration/MigrationSettings.cs index 0de8ddb892..bb7812fee6 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationSettings.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationSettings.cs @@ -1,15 +1,30 @@ namespace ServiceControl.Persistence.DataMigration; -/// The migration's setting names, relative to the instance's settings root, and their defaults. explains the tuning ones. +/// +/// The migration's setting names, relative to the instance's settings root, and their defaults. +/// explains the tuning ones. +/// public static class MigrationSettings { public const string ThrottlePauseMillisecondsKey = "Migration/ThrottlePauseMilliseconds"; public const string HaltThresholdPercentKey = "Migration/HaltThresholdPercent"; public const string HaltThresholdMinimumKey = "Migration/HaltThresholdMinimum"; - /// A comma-separated list of the optional categories to copy, such as "EventLog, CustomChecks". + /// + /// A comma-separated list of the optional categories to copy, such as "EventLog, CustomChecks". + /// public const string OptionalCategoriesKey = "Migration/OptionalCategories"; + /// + /// Turns the migration on: the required copy runs before the host opens. + /// + public const string EnabledKey = "Migration/Enabled"; + /// + /// Accepts permanent loss on an instance that has already served traffic on the target. + /// + public const string AllowIncompleteExitKey = "Migration/AllowIncompleteExit"; public const int DefaultThrottlePauseMilliseconds = 100; public const int DefaultHaltThresholdPercent = 5; public const int DefaultHaltThresholdMinimum = 100; + public const bool DefaultEnabled = false; + public const bool DefaultAllowIncompleteExit = false; } diff --git a/src/ServiceControl.Persistence/DataMigration/MigrationSkipReason.cs b/src/ServiceControl.Persistence/DataMigration/MigrationSkipReason.cs index 4129eb939d..6cf45de09b 100644 --- a/src/ServiceControl.Persistence/DataMigration/MigrationSkipReason.cs +++ b/src/ServiceControl.Persistence/DataMigration/MigrationSkipReason.cs @@ -1,9 +1,27 @@ namespace ServiceControl.Persistence.DataMigration; +/// +/// Why one row was not copied. Every skipped row carries one, so the verify command can account for the +/// difference between the two databases. says which of +/// these are no loss. +/// public enum MigrationSkipReason { + /// The message body could not be read from the source, after the engine had tried more than once. BodyUnreadable, + + /// The row is older than the retention period, so the instance would have deleted it soon anyway. PastRetention, - // Never written by a copier. A database a newer build wrote still reads rather than throwing where the host decides whether to start. + + /// The source row has no value for something the target column requires. + RequiredValueMissing, + + /// The row names an endpoint the target does not know, so what it holds would be deleted after the cutover anyway. + EndpointNotKnown, + + /// + /// Nothing ever writes this. A reason only a newer build knows reads back as Unknown, so an older host still + /// starts instead of throwing. + /// Unknown } diff --git a/src/ServiceControl.Persistence/PersistenceSettings.cs b/src/ServiceControl.Persistence/PersistenceSettings.cs index 08a73eb1c3..411ada68a2 100644 --- a/src/ServiceControl.Persistence/PersistenceSettings.cs +++ b/src/ServiceControl.Persistence/PersistenceSettings.cs @@ -8,25 +8,50 @@ namespace ServiceControl.Persistence /// public abstract class PersistenceSettings { + /// + /// Whether this host starts the database and nothing else, so an operator can repair or inspect it. + /// Only the RavenDB persister supports it. There the embedded server and RavenDB Studio start as + /// usual, and the host registers nothing that ingests or serves data. On any other persister the + /// host refuses to start in maintenance mode. + /// public bool MaintenanceMode { get; set; } - //HINT: This needs to be here so that ServerControl instance can add an instance specific metadata to tweak the DatabasePath value + + /// + /// The directory that holds the database files, or null when this host holds none. The RavenDB + /// persister fills it in from its DbPath setting, and its disk space checks measure the drive that + /// this path names. The SQL persisters leave it null, because their files live on the database server. + /// public string? DatabasePath { get; set; } /// - /// Whether this host owns the background deletion of data past its retention period. Only one - /// host in a deployment should, so error ingestion only hosts turn it off. + /// Whether this host deletes data that is past its retention period. Only one host in a deployment + /// must do this, so a host that only ingests errors turns it off. The SQL persisters start a + /// background sweeper when it is true. RavenDB expires documents on its own and does not read this. /// public bool RunRetentionSweep { get; set; } = true; + /// + /// Whether message bodies are indexed for search. The RavenDB persister decides this as it ingests a + /// message, so a change only reaches messages ingested after it. The SQL persisters always index the + /// body, so the value does not change what they do. + /// public bool EnableFullTextSearchOnBodies { get; set; } = true; /// - /// Wall-clock limit for the message view queries, see . + /// How many completed retry operations the history keeps, copied from the ServiceControl/RetryHistoryDepth + /// setting. At 0 it keeps none, so the next retry to complete clears the history. + /// + public int RetryHistoryDepth { get; set; } + + /// + /// The wall clock limit for a message query, see . It is also how long + /// this instance waits for a remote instance to answer. /// public TimeSpan QueryTimeout { get; set; } = QueryTimeLimit.Default; /// - /// The setting is read from, as named in the timeout error. + /// The name of the setting that comes from. The timeout error names it, so + /// an operator can see which setting to change. /// public const string QueryTimeoutSettingName = "ServiceControl/" + QueryTimeLimit.SettingName; } diff --git a/src/ServiceControl.UnitTests/ApprovalFiles/APIApprovals.PlatformSampleSettings.approved.txt b/src/ServiceControl.UnitTests/ApprovalFiles/APIApprovals.PlatformSampleSettings.approved.txt index abd4c98313..c7f9253639 100644 --- a/src/ServiceControl.UnitTests/ApprovalFiles/APIApprovals.PlatformSampleSettings.approved.txt +++ b/src/ServiceControl.UnitTests/ApprovalFiles/APIApprovals.PlatformSampleSettings.approved.txt @@ -63,6 +63,8 @@ "VirtualDirectory": "", "HeartbeatGracePeriod": "00:00:40", "TransportType": "ServiceControl.Transports.Learning.LearningTransportCustomization, ServiceControl.Transports.Learning", + "MigrationEnabled": false, + "MigrationAllowIncompleteExit": false, "ErrorLogQueue": "error.log", "ErrorQueue": "error", "ForwardErrorMessages": false, diff --git a/src/ServiceControl.UnitTests/Migration/CheckpointMigrationStateTests.cs b/src/ServiceControl.UnitTests/Migration/CheckpointMigrationStateTests.cs new file mode 100644 index 0000000000..5db301bf83 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/CheckpointMigrationStateTests.cs @@ -0,0 +1,94 @@ +#nullable enable +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.UnitTests.Migration.Fakes; + +[TestFixture] +class CheckpointMigrationStateTests +{ + [Test] + public async Task An_instance_that_has_never_migrated_has_no_store_and_nothing_incomplete() + { + var state = new CheckpointMigrationState(); + + await state.Seed(["EndpointSettings"], CancellationToken.None); + + Assert.That(state.AnyCategoryIncomplete, Is.False); + } + + [Test] + public async Task Selected_categories_that_all_finished_leave_the_guard_off() + { + var store = new InMemoryMigrationCheckpointStore(); + await store.Upsert(Checkpoint("KnownEndpoints", MigrationCategoryState.Complete)); + await store.Upsert(Checkpoint("EndpointSettings", MigrationCategoryState.CompleteWithErrors)); + + var state = new CheckpointMigrationState(store); + await state.Seed(["KnownEndpoints", "EndpointSettings"], CancellationToken.None); + + Assert.That(state.AnyCategoryIncomplete, Is.False); + } + + [TestCase(MigrationCategoryState.InProgress)] + [TestCase(MigrationCategoryState.NotStarted)] + [TestCase(MigrationCategoryState.Halted)] + [TestCase(MigrationCategoryState.Blocked)] + public async Task One_selected_category_short_of_finished_holds_the_guard_on(MigrationCategoryState unfinished) + { + var store = new InMemoryMigrationCheckpointStore(); + await store.Upsert(Checkpoint("KnownEndpoints", MigrationCategoryState.Complete)); + await store.Upsert(Checkpoint("EndpointSettings", unfinished)); + + var state = new CheckpointMigrationState(store); + await state.Seed(["KnownEndpoints", "EndpointSettings"], CancellationToken.None); + + Assert.That(state.AnyCategoryIncomplete, Is.True); + } + + [Test] + public async Task A_selected_category_with_no_row_at_all_has_not_started() + { + var store = new InMemoryMigrationCheckpointStore(); + await store.Upsert(Checkpoint("KnownEndpoints", MigrationCategoryState.Complete)); + + var state = new CheckpointMigrationState(store); + await state.Seed(["KnownEndpoints", "EndpointSettings"], CancellationToken.None); + + Assert.That(state.AnyCategoryIncomplete, Is.True); + } + + [Test] + public async Task An_unselected_category_that_never_ran_does_not_hold_the_guard_on_forever() + { + var store = new InMemoryMigrationCheckpointStore(); + await store.Upsert(new MigrationCheckpoint("EndpointSettings", MigrationCategoryState.Complete, null, 3, 0, 3, null, null, null, null, null)); + await store.Upsert(new MigrationCheckpoint("EventLog", MigrationCategoryState.NotStarted, null, 0, 0, null, null, null, null, null, null)); + + var state = new CheckpointMigrationState(store); + await state.Seed(["EndpointSettings"], CancellationToken.None); + + Assert.That(state.AnyCategoryIncomplete, Is.False); + } + + [Test] + public void Exactly_complete_complete_with_errors_and_abandoned_are_finished() => + Assert.That( + Enum.GetValues().Where(state => state.IsFinished()), + Is.EquivalentTo(new[] { MigrationCategoryState.Complete, MigrationCategoryState.CompleteWithErrors, MigrationCategoryState.Abandoned })); + + [Test] + public void Exactly_EndpointNotKnown_is_benign_and_every_other_skip_reason_counts_as_a_loss() => + Assert.That( + Enum.GetValues().Where(reason => reason.IsBenign()), + Is.EquivalentTo(new[] { MigrationSkipReason.EndpointNotKnown }), + "a benign reason is exempt from the halt threshold, so any number of rows lost to one settles the category as complete"); + + static MigrationCheckpoint Checkpoint(string categoryId, MigrationCategoryState state) => + new(categoryId, state, null, 0, 0, null, null, null, null, null, null); +} diff --git a/src/ServiceControl.UnitTests/Migration/ClosedWindowProgressTests.cs b/src/ServiceControl.UnitTests/Migration/ClosedWindowProgressTests.cs new file mode 100644 index 0000000000..f09df8b572 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/ClosedWindowProgressTests.cs @@ -0,0 +1,263 @@ +#nullable enable +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Collections.Concurrent; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Logging; +using NUnit.Framework; +using ServiceControl.Migration; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.UnitTests.Migration.Fakes; + +// No acceptance test can reach the watchdog: the host copies on the real clock and nobody waits half an hour. +[TestFixture] +class ClosedWindowProgressTests +{ + static readonly TimeSpan PollInterval = MigrationStartup.ClosedWindowProgress.PollInterval; + static readonly TimeSpan StallLimit = MigrationStartup.ClosedWindowProgress.StallLimit; + + [Test] + public async Task A_category_that_commits_nothing_for_the_stall_limit_has_its_copy_stopped() + { + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit + PollInterval); + + Assert.That(progress.Token.WaitHandle.WaitOne(TimeSpan.FromSeconds(10)), Is.True, "a copy that committed nothing for longer than the limit was never stopped"); + await progress.DisposeAsync(); + Assert.That(progress.StalledCategoryId, Is.EqualTo(MigrationCategoryIds.KnownEndpoints), "the refusal message names the category from this"); + } + + [Test] + public async Task A_resumed_row_carrying_the_previous_runs_stamp_is_not_a_stall() + { + // The row a killed run left behind keeps its last stamp, so an operator restarting hours later + // would otherwise have a healthy copy cancelled on the first tick, every time. + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: clock.GetUtcNow().UtcDateTime - TimeSpan.FromHours(2))); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(PollInterval); + await WaitForAPollAtTheCurrentTime(store, clock); + await progress.DisposeAsync(); + + Assert.That(progress.StalledCategoryId, Is.Null, "the window runs from when this watch began, not from a stamp the previous run left"); + } + + [Test] + public async Task A_row_left_running_by_another_run_is_not_this_copys_to_stop() + { + // ReadAll returns every row in the target, including ones this run never attempted. + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(InProgress(MigrationCategoryIds.EndpointSettings, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit + PollInterval); + await WaitForAPollAtTheCurrentTime(store, clock); + await progress.DisposeAsync(); + + Assert.That(progress.StalledCategoryId, Is.Null, "a category this run never attempted was named as the reason its copy stopped"); + } + + [Test] + public async Task Committing_nothing_for_exactly_the_stall_limit_is_not_a_stall() + { + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit); + await WaitForAPollAtTheCurrentTime(store, clock); + await progress.DisposeAsync(); + + Assert.That(progress.StalledCategoryId, Is.Null, "the limit is the point at which a copy has not yet stalled"); + } + + [Test] + public async Task A_category_that_has_committed_nothing_since_this_watch_began_is_stopped() + { + // The engine stamps a category when it marks it running, so the window covers counting the source + // and the first read, which is where a copy that never gets going actually hangs. + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: clock.GetUtcNow().UtcDateTime, startedAt: clock.GetUtcNow().UtcDateTime - TimeSpan.FromDays(3))); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit + PollInterval); + + Assert.That(progress.Token.WaitHandle.WaitOne(TimeSpan.FromSeconds(10)), Is.True, "a category that has committed nothing since the watch began was never stopped"); + await progress.DisposeAsync(); + Assert.That(progress.StalledCategoryId, Is.EqualTo(MigrationCategoryIds.KnownEndpoints)); + } + + [Test] + public async Task A_row_an_older_build_left_without_a_stamp_is_still_watched() + { + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: null, startedAt: clock.GetUtcNow().UtcDateTime - TimeSpan.FromDays(3))); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit + PollInterval); + + Assert.That(progress.Token.WaitHandle.WaitOne(TimeSpan.FromSeconds(10)), Is.True, "an unstamped row was left unwatched for ever"); + await progress.DisposeAsync(); + Assert.That(progress.StalledCategoryId, Is.EqualTo(MigrationCategoryIds.KnownEndpoints)); + } + + [Test] + public async Task A_category_this_run_finished_is_not_stopped_for_having_gone_quiet() + { + // Required categories run one at a time, so a small one settles early and its stamp then ages + // for as long as the next category takes. + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(Settled(MigrationCategoryIds.KnownEndpoints, MigrationCategoryState.Complete, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit + PollInterval); + await WaitForAPollAtTheCurrentTime(store, clock); + await progress.DisposeAsync(); + + Assert.That(progress.StalledCategoryId, Is.Null, "a category that finished was named as the reason the copy stopped"); + } + + [Test] + public async Task A_halted_category_this_run_attempted_is_not_stopped_for_having_gone_quiet() + { + // A halt settles the row and leaves its stamp behind. The gate is what refuses the host over it, not the watchdog. + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + await store.Upsert(Settled(MigrationCategoryIds.KnownEndpoints, MigrationCategoryState.Halted, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + clock.Advance(StallLimit + PollInterval); + await WaitForAPollAtTheCurrentTime(store, clock); + await progress.DisposeAsync(); + + Assert.That(progress.StalledCategoryId, Is.Null, "a halted category was named as the reason the copy stopped"); + } + + // The watch has no total limit, so a copy that keeps committing outlives any length of run. + [Test] + public async Task A_category_still_committing_batches_is_never_stopped_however_long_it_takes() + { + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + var committed = await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints); + + // Four times the limit, committing a batch every poll, which a total timeout would have killed long ago. + for (var elapsed = TimeSpan.Zero; elapsed < StallLimit * 4; elapsed += PollInterval) + { + clock.Advance(PollInterval); + await WaitForAPollAtTheCurrentTime(store, clock); + committed = await store.Upsert(committed with { LastProgressAt = clock.GetUtcNow().UtcDateTime }); + } + + await progress.DisposeAsync(); + + Assert.That(progress.StalledCategoryId, Is.Null, "a copy committing a batch every poll was stopped, so the watch is a deadline rather than a stall detector"); + } + + [Test] + public async Task A_poll_that_fails_leaves_the_watch_running_and_says_so() + { + var clock = new TimerRecordingTimeProvider(); + var store = new PollObservingCheckpointStore(clock); + var storeFailure = new InvalidOperationException("the checkpoint store is unreachable"); + store.FailReadAll(1, storeFailure); + await store.Upsert(InProgress(MigrationCategoryIds.KnownEndpoints, lastProgressAt: clock.GetUtcNow().UtcDateTime)); + var logger = new CapturingLogger(); + var progress = await StartWatching(store, clock, MigrationCategoryIds.KnownEndpoints, logger); + + clock.Advance(PollInterval); + await WaitForAPollAtTheCurrentTime(store, clock); + clock.Advance(StallLimit + PollInterval); + + Assert.That(progress.Token.WaitHandle.WaitOne(TimeSpan.FromSeconds(10)), Is.True, "a watch that stopped at the first blip never noticed the stall that followed"); + await progress.DisposeAsync(); + using (Assert.EnterMultipleScope()) + { + Assert.That(progress.StalledCategoryId, Is.EqualTo(MigrationCategoryIds.KnownEndpoints)); + Assert.That(logger.Entries.Where(entry => entry.Level == LogLevel.Error).Select(entry => entry.Exception), Has.Member(storeFailure), "a watch that has gone deaf has to say so"); + } + } + + static async Task StartWatching(PollObservingCheckpointStore store, TimerRecordingTimeProvider clock, string attemptedCategoryId, ILogger? logger = null) + { + var progress = new MigrationStartup.ClosedWindowProgress(store, clock, logger ?? new CapturingLogger(), [attemptedCategoryId]); + + Assert.That(await clock.TimerCreated.WaitAsync(TimeSpan.FromSeconds(10)), Is.True, "the watch never started its timer, so advancing the clock would tick nothing"); + + return progress; + } + + // Without this, a test asserting the watch did nothing would just be asking before the watch had looked. + static async Task WaitForAPollAtTheCurrentTime(PollObservingCheckpointStore store, TimerRecordingTimeProvider clock) + { + var now = clock.GetUtcNow().UtcDateTime; + + while (true) + { + Assert.That(await store.Polled.WaitAsync(TimeSpan.FromSeconds(10)), Is.True, $"no poll read the checkpoints at {now:O}; a watch that stopped polling never reaches one"); + + if (store.PolledAt.TryDequeue(out var polledAt) && polledAt == now) + { + return; + } + } + } + + static MigrationCheckpoint InProgress(string categoryId, DateTime? lastProgressAt, DateTime? startedAt = null) => + new(categoryId, MigrationCategoryState.InProgress, null, 0, 0, null, null, startedAt ?? lastProgressAt, lastProgressAt, null, null); + + static MigrationCheckpoint Settled(string categoryId, MigrationCategoryState state, DateTime lastProgressAt) => + new(categoryId, state, null, 0, 0, null, null, lastProgressAt, lastProgressAt, lastProgressAt, null); + + // Records when each poll read the rows, so a test can wait for a poll that saw the state it is judging. + sealed class PollObservingCheckpointStore(TimeProvider clock) : IMigrationCheckpointStore + { + readonly InMemoryMigrationCheckpointStore inner = new(); + Exception? readAllFailure; + int readAllFailuresLeft; + + public SemaphoreSlim Polled { get; } = new(0); + + public ConcurrentQueue PolledAt { get; } = new(); + + public void FailReadAll(int times, Exception failure) + { + readAllFailuresLeft = times; + readAllFailure = failure; + } + + public Task> ReadAll(CancellationToken cancellationToken = default) + { + PolledAt.Enqueue(clock.GetUtcNow().UtcDateTime); + Polled.Release(); + + if (readAllFailuresLeft > 0) + { + readAllFailuresLeft--; + return Task.FromException>(readAllFailure!); + } + + return inner.ReadAll(cancellationToken); + } + + public Task Read(string categoryId, CancellationToken cancellationToken = default) => inner.Read(categoryId, cancellationToken); + + public Task Upsert(MigrationCheckpoint checkpoint, CancellationToken cancellationToken = default) => inner.Upsert(checkpoint, cancellationToken); + } +} diff --git a/src/ServiceControl.UnitTests/Migration/EveryRequiredCategoryCanBeCopiedCheckTests.cs b/src/ServiceControl.UnitTests/Migration/EveryRequiredCategoryCanBeCopiedCheckTests.cs new file mode 100644 index 0000000000..1b68299de7 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/EveryRequiredCategoryCanBeCopiedCheckTests.cs @@ -0,0 +1,48 @@ +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Linq; +using NUnit.Framework; +using ServiceControl.Migration; +using ServiceControl.Migration.Checks; +using ServiceControl.Persistence.DataMigration; + +[TestFixture] +class EveryRequiredCategoryCanBeCopiedCheckTests +{ + static readonly string[] EveryRequiredCategory = + [.. MigrationCategoryRegistry.All.Where(category => category.Kind == MigrationCategoryKind.Required).Select(category => category.Id)]; + + [Test] + public void A_build_that_cannot_copy_the_whole_required_set_refuses_MigrationEnabled() + { + var copyable = EveryRequiredCategory.Except([MigrationCategoryIds.RetryOperations]).ToArray(); + + var exception = Assert.ThrowsAsync(() => new EveryRequiredCategoryCanBeCopiedCheck(copyable).Run()); + + Assert.That(exception.Message, Does.Contain(MigrationCategoryIds.RetryOperations).And.Contain(MigrationSettings.EnabledKey), + "a customer who cannot see which categories are missing, or which setting stranded them, has nothing to act on"); + } + + [Test] + public void The_acceptance_fixtures_marker_lets_the_copy_this_build_can_do_run() + { + Assert.DoesNotThrowAsync(() => new EveryRequiredCategoryCanBeCopiedCheck([], new AllowIncompleteCategorySet()).Run()); + } + + [Test] + public void A_build_that_copies_every_required_category_passes_without_the_marker() + { + Assert.DoesNotThrowAsync(() => new EveryRequiredCategoryCanBeCopiedCheck(EveryRequiredCategory).Run()); + } + + [Test] + public void A_category_only_one_side_supports_counts_as_not_copyable() + { + var source = new[] { MigrationCategoryIds.KnownEndpoints, MigrationCategoryIds.EndpointSettings }; + var target = new[] { MigrationCategoryIds.KnownEndpoints }; + + Assert.That(MigrationStartup.CopyableCategoryIds(source, target), Is.EquivalentTo(new[] { MigrationCategoryIds.KnownEndpoints }), + "a category with a reader and no writer would otherwise be attempted and fail partway through a customer's copy"); + } +} diff --git a/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationSource.cs b/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationSource.cs index 66341cbf20..0c44819509 100644 --- a/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationSource.cs +++ b/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationSource.cs @@ -13,6 +13,7 @@ public sealed class InMemoryMigrationSource : IMigrationSource { readonly Dictionary> rowsByCategory = []; readonly Dictionary bodyFailures = []; + readonly Dictionary bodyFailuresInTurn = []; readonly Dictionary bodyReadAttempts = []; readonly Dictionary bodies = []; @@ -24,13 +25,22 @@ public sealed class InMemoryMigrationSource : IMigrationSource public void FailBodyReads(string sourceId, int times, Exception failure) => bodyFailures[sourceId] = (times, failure); - /// Makes the body read for this row behave like a host shutting down: the token is cancelled and the read throws. + /// + /// Fails the body reads for this row with one exception per attempt, in order, so a test can tell one attempt's error from another's. An attempt past the end of the list reads the body normally. + /// + public void FailBodyReadsInTurn(string sourceId, params Exception[] failuresInAttemptOrder) => bodyFailuresInTurn[sourceId] = failuresInAttemptOrder; + + /// + /// Makes the body read for this row behave like a host shutting down: the token is cancelled and the read throws. + /// public (string SourceId, CancellationTokenSource Source)? StopOnBodyRead { get; set; } public int BodyReadAttempts(string sourceId) => bodyReadAttempts.GetValueOrDefault(sourceId); public Task Open(CancellationToken cancellationToken = default) => Task.CompletedTask; + public IReadOnlyList ContributedChecks() => []; + public Task Describe(CancellationToken cancellationToken = default) => Task.FromResult(Description); public Task> Inventory(CancellationToken cancellationToken = default) => @@ -38,7 +48,7 @@ public Task> Inventory(Cancellation [.. rowsByCategory.Select(pair => new MigrationSourceInventoryEntry("memory", pair.Key, pair.Value.Count))]); public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => - Task.FromResult((long)(rowsByCategory.TryGetValue(category.Id, out var rows) ? rows.Count : 0)); + Task.FromResult(rowsByCategory.TryGetValue(category.Id, out var rows) ? rows.Count : 0); public async IAsyncEnumerable Read( MigrationCategory category, @@ -85,8 +95,17 @@ public async IAsyncEnumerable Read( throw failures.Failure; } + if (bodyFailuresInTurn.TryGetValue(sourceId, out var failuresInTurn) && attempt <= failuresInTurn.Length) + { + throw failuresInTurn[attempt - 1]; + } + return bodies.GetValueOrDefault(sourceId); } + // Every registered category, not just the seeded ones: Seed writes rowsByCategory after construction, + // and the interface promises an answer that does not change across Open. + public IReadOnlyCollection SupportedCategoryIds => [.. MigrationCategoryRegistry.All.Select(category => category.Id)]; + public ValueTask DisposeAsync() => ValueTask.CompletedTask; } diff --git a/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationTarget.cs b/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationTarget.cs index cdfaff9777..1ef4ed74b8 100644 --- a/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationTarget.cs +++ b/src/ServiceControl.UnitTests/Migration/Fakes/InMemoryMigrationTarget.cs @@ -2,6 +2,7 @@ namespace ServiceControl.UnitTests.Migration.Fakes; using System; using System.Collections.Generic; +using System.Linq; using System.Threading; using System.Threading.Tasks; using ServiceControl.Persistence.DataMigration; @@ -10,6 +11,7 @@ public sealed class InMemoryMigrationTarget(IMigrationCheckpointStore checkpoint { readonly Dictionary> writtenKeysByCategory = []; readonly Dictionary> writtenRowsByCategory = []; + readonly Dictionary> rowsHandedToWriteByCategory = []; readonly HashSet preExistingKeys = []; readonly Dictionary rejectedKeys = []; @@ -17,11 +19,23 @@ public sealed class InMemoryMigrationTarget(IMigrationCheckpointStore checkpoint public string NoBatchSizeFor { get; set; } public int? FailOnCallNumber { get; set; } - /// What FailOnCallNumber throws, when the default simulated failure is the wrong shape for the test. + /// + /// What FailOnCallNumber throws, when the default simulated failure is the wrong shape for the test. + /// public Exception FailWith { get; set; } - /// Cancels the token on this call and then writes normally, so the stop surfaces from the source's next batch. + /// + /// Runs at the start of every write, before any simulated failure, with the checkpoint the engine is extending, so a test can watch a stamp move between batches. + /// + public Action BeforeWrite { get; set; } + + /// + /// Cancels the token on this call and then writes normally, so the stop surfaces from the source's next batch. + /// public (int CallNumber, CancellationTokenSource Source)? CancelOnCall { get; set; } + /// + /// Cancels the token on this call and throws instead of writing, so the stop surfaces from the write itself. + /// public (int CallNumber, CancellationTokenSource Source)? StopOnCall { get; set; } int callCount; @@ -32,8 +46,16 @@ public sealed class InMemoryMigrationTarget(IMigrationCheckpointStore checkpoint public IReadOnlyList WrittenRows(string categoryId) => writtenRowsByCategory.TryGetValue(categoryId, out var rows) ? rows : []; - public int BatchSizeFor(MigrationCategory category) => - category.Id == NoBatchSizeFor ? throw new InvalidOperationException($"No batch size is mapped for category {category.Id}") : DefaultBatchSize; + /// + /// Every row the engine sent to a write that ran, in the order it sent them, including the rows this fake then skipped or found already present. It is the only way to see the same row sent twice, because WrittenRows de-duplicates as the real targets do. Empty for a category never written to. + /// + public IReadOnlyList RowsHandedToWrite(string categoryId) => + rowsHandedToWriteByCategory.TryGetValue(categoryId, out var rows) ? rows : []; + + public Task Open(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => + category.Id == NoBatchSizeFor ? throw new InvalidOperationException($"No batch size is mapped for category {category.Id}") : Task.FromResult(DefaultBatchSize); public async Task Write( MigrationCategory category, @@ -42,6 +64,7 @@ public async Task Write( CancellationToken cancellationToken = default) { callCount++; + BeforeWrite?.Invoke(checkpointToExtend); if (StopOnCall is { } stop && stop.CallNumber == callCount) { @@ -62,6 +85,11 @@ public async Task Write( var keys = writtenKeysByCategory.TryGetValue(category.Id, out var existingKeys) ? existingKeys : writtenKeysByCategory[category.Id] = []; var rows = writtenRowsByCategory.TryGetValue(category.Id, out var existingRows) ? existingRows : writtenRowsByCategory[category.Id] = []; + // Recorded here rather than at the top of the method: a write that threw above never committed, so + // the restart that sends its batch again is doing the right thing. + var handedOver = rowsHandedToWriteByCategory.TryGetValue(category.Id, out var existingHandedOver) ? existingHandedOver : rowsHandedToWriteByCategory[category.Id] = []; + handedOver.AddRange(batch.Rows); + var copied = 0; var alreadyPresent = 0; var benignSkipped = 0; @@ -91,8 +119,8 @@ public async Task Write( copied++; } - // Extended and saved in the same operation as the rows, as the real targets do, so what lands - // is this batch's real split rather than a provisional one the next save has to correct. + // The real targets extend and save the checkpoint in the transaction that writes the rows, + // so this fake saves it here too. var saved = await checkpointStore.Upsert( checkpointToExtend.Extend(copied, skippedIds.Count, alreadyPresent, skipReasons), cancellationToken); @@ -102,4 +130,6 @@ public async Task Write( public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult((long)(writtenRowsByCategory.TryGetValue(category.Id, out var rows) ? rows.Count : 0)); + + public IReadOnlyCollection SupportedCategoryIds => [.. MigrationCategoryRegistry.All.Select(category => category.Id)]; } diff --git a/src/ServiceControl.UnitTests/Migration/HaltThresholdTests.cs b/src/ServiceControl.UnitTests/Migration/HaltThresholdTests.cs index fbe2bebe97..51cbfa5678 100644 --- a/src/ServiceControl.UnitTests/Migration/HaltThresholdTests.cs +++ b/src/ServiceControl.UnitTests/Migration/HaltThresholdTests.cs @@ -61,6 +61,42 @@ public void A_hair_past_the_proportion_halts_once_the_floor_is_behind_it() Assert.That(exceeded, Is.True); } + [Test] + public void A_large_category_losing_a_small_share_does_not_halt_once_past_the_floor() + { + // 101 of 5,000,000 is 0.002%. The floor is already behind it, so only the proportion can stop a halt + // here, and a rule that halted on the floor alone would stop a healthy copy of a huge category. + var exceeded = HaltThreshold.Exceeded(skippedCount: 101, totalCount: 5_000_000, percentThreshold: 5, minimumFloor: 100); + + Assert.That(exceeded, Is.False); + } + + [Test] + public void Losing_every_row_of_a_run_is_losing_most_of_it() + { + Assert.That(HaltThreshold.MostOfItWasLost(skippedCount: 90, totalCount: 90), Is.True); + } + + [Test] + public void Losing_all_but_one_row_of_a_run_is_losing_most_of_it() + { + Assert.That(HaltThreshold.MostOfItWasLost(skippedCount: 99, totalCount: 100), Is.True); + } + + [Test] + public void Losing_exactly_half_a_run_is_not_losing_most_of_it() + { + // The rule needs strictly more than half, so an exact half is the one input that separates it from a + // rule that halted at half as well. + Assert.That(HaltThreshold.MostOfItWasLost(skippedCount: 50, totalCount: 100), Is.False); + } + + [Test] + public void A_run_that_processed_nothing_lost_nothing() + { + Assert.That(HaltThreshold.MostOfItWasLost(skippedCount: 0, totalCount: 0), Is.False); + } + [Test] public void A_skip_count_with_nothing_processed_never_halts_and_never_divides_by_zero() { diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEnabledSettingsTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEnabledSettingsTests.cs new file mode 100644 index 0000000000..67deb29537 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/MigrationEnabledSettingsTests.cs @@ -0,0 +1,67 @@ +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Linq; +using System.Reflection; +using NUnit.Framework; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Configuration; +using ServiceControl.Persistence.DataMigration; + +[TestFixture] +[NonParallelizable] +class MigrationEnabledSettingsTests +{ + static Settings NewSettings() => + new(transportType: "LearningTransport", persisterType: "RavenDB", errorRetentionPeriod: TimeSpan.FromDays(10)); + + [TearDown] + public void ClearEnvironmentVariables() + { + Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_ENABLED", null); + Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_ALLOWINCOMPLETEEXIT", null); + } + + [Test] + public void Both_default_to_off() + { + var settings = NewSettings(); + + Assert.Multiple(() => + { + Assert.That(settings.MigrationEnabled, Is.False); + Assert.That(settings.MigrationAllowIncompleteExit, Is.False); + }); + } + + [Test] + public void MigrationEnabled_is_read_from_the_environment() + { + Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_ENABLED", "true"); + + Assert.That(NewSettings().MigrationEnabled, Is.True); + } + + [Test] + public void AllowIncompleteExit_is_read_from_the_environment() + { + Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_ALLOWINCOMPLETEEXIT", "true"); + + Assert.That(NewSettings().MigrationAllowIncompleteExit, Is.True); + } + + [Test] + public void Only_MigrationEngineOptions_parses_the_engine_settings_keys() + { + var parsers = typeof(Settings).Assembly.GetTypes() + .Concat(typeof(MigrationEngineOptions).Assembly.GetTypes()) + .Where(type => type.GetMethods(BindingFlags.Public | BindingFlags.NonPublic | BindingFlags.Static) + .Any(method => method.Name is "FromSettings" or "Read" or "Resolve" + && method.GetParameters() is [{ ParameterType.Name: nameof(SettingsRootNamespace) }])) + .Select(type => type.Name) + .ToArray(); + + Assert.That(parsers, Is.EquivalentTo(new[] { nameof(MigrationEngineOptions) }), + "Migration/ThrottlePauseMilliseconds and its three neighbours have exactly one parser. A second one drifts its defaults and its refusal message away from this one, and nothing fails until a customer types a category name wrong."); + } +} diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineBodyRetryTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineBodyRetryTests.cs index dc938d6543..dbc94ad95a 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineBodyRetryTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineBodyRetryTests.cs @@ -114,10 +114,11 @@ public async Task The_configured_backoff_is_waited_between_body_read_attempts() // millisecond and the message is skipped for an outage it would have survived. var category = MigrationCategoryRegistry.Find("UnresolvedAndRetryIssuedFailedMessages")!; var source = new InMemoryMigrationSource(); - source.Seed(category.Id, Row("msg-1")); - var body = new MigrationBody(new byte[] { 1 }, "text/plain"); - source.SetBody("msg-1", body); - source.FailBodyReads("msg-1", MigrationEngine.MaxBodyReadAttempts - 1, new TimeoutException("body store unreachable")); + source.Seed(category.Id, Row("msg-1"), Row("msg-2")); + // Every attempt on msg-1 fails, so the guard against a wait after the final one is the only thing + // keeping the third timer from being created. + source.FailBodyReads("msg-1", MigrationEngine.MaxBodyReadAttempts, new TimeoutException("body store unreachable")); + source.SetBody("msg-2", new MigrationBody(new byte[] { 2 }, "text/plain")); var checkpointStore = new InMemoryMigrationCheckpointStore(); var target = new InMemoryMigrationTarget(checkpointStore); var clock = new TimerRecordingTimeProvider(); @@ -127,7 +128,8 @@ public async Task The_configured_backoff_is_waited_between_body_read_attempts() var runTask = engine.RunCategoryAsync(category); - // Two failures, so a wait after each before the attempt that succeeds. + // A wait after the first two failures only. A third would never complete, because the clock is + // not advanced again and the run below would time out waiting for it. for (var waitNumber = 1; waitNumber <= MigrationEngine.MaxBodyReadAttempts - 1; waitNumber++) { Assert.That(await clock.TimerCreated.WaitAsync(TimeSpan.FromSeconds(5)), Is.True, $"backoff {waitNumber} never started"); @@ -139,20 +141,24 @@ public async Task The_configured_backoff_is_waited_between_body_read_attempts() using (Assert.EnterMultipleScope()) { - Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.Complete)); + Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.CompleteWithErrors)); + Assert.That(checkpoint.SkippedCount, Is.EqualTo(1), "msg-1 was given up on, so all three attempts ran"); Assert.That(clock.DueTimes, Is.EqualTo(new[] { backoff, backoff }), "no wait after the final attempt, which has nothing left to retry"); - Assert.That(target.WrittenRows(category.Id).Single().Body, Is.EqualTo(body)); } } [Test] public async Task The_skip_warning_for_an_unreadable_body_carries_the_last_attempts_exception() { + // The first failure of a retry run is usually a transient connect error. The last one says what the + // body store was doing when the message was given up on, and it is all the operator gets. var category = MigrationCategoryRegistry.Find("UnresolvedAndRetryIssuedFailedMessages")!; var source = new InMemoryMigrationSource(); source.Seed(category.Id, Row("msg-1")); - var failure = new TimeoutException("body store unreachable"); - source.FailBodyReads("msg-1", MigrationEngine.MaxBodyReadAttempts, failure); + var failures = Enumerable.Range(1, MigrationEngine.MaxBodyReadAttempts) + .Select(attempt => new TimeoutException($"body store unreachable on attempt {attempt}")) + .ToArray(); + source.FailBodyReadsInTurn("msg-1", failures); var checkpointStore = new InMemoryMigrationCheckpointStore(); var target = new InMemoryMigrationTarget(checkpointStore); var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []) { BodyRetryBackoff = TimeSpan.Zero }; @@ -161,7 +167,7 @@ public async Task The_skip_warning_for_an_unreadable_body_carries_the_last_attem await engine.RunCategoryAsync(category); - Assert.That(logger.Entries.Single(e => e.Message.StartsWith("Skipped msg-1")).Exception, Is.SameAs(failure)); + Assert.That(logger.Entries.Single(e => e.Message.StartsWith("Skipped msg-1")).Exception, Is.SameAs(failures[^1])); } [Test] diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineCategorySelectionTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineCategorySelectionTests.cs index 9c74a7b605..d41169cd71 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineCategorySelectionTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineCategorySelectionTests.cs @@ -3,6 +3,7 @@ namespace ServiceControl.UnitTests.Migration; using System; using System.Collections.Generic; using System.Linq; +using System.Threading.Tasks; using Microsoft.Extensions.Logging.Abstractions; using Microsoft.Extensions.Time.Testing; using NUnit.Framework; @@ -13,9 +14,9 @@ namespace ServiceControl.UnitTests.Migration; [TestFixture] class MigrationEngineCategorySelectionTests { - static MigrationEngine BuildEngine(IReadOnlyCollection selectedOptionalIds, out InMemoryMigrationCheckpointStore checkpointStore) + static MigrationEngine BuildEngine(IReadOnlyCollection selectedOptionalIds) { - checkpointStore = new InMemoryMigrationCheckpointStore(); + var checkpointStore = new InMemoryMigrationCheckpointStore(); var target = new InMemoryMigrationTarget(checkpointStore); var options = new MigrationEngineOptions(TimeSpan.FromSeconds(1), 5, 100, selectedOptionalIds); return new MigrationEngine(new InMemoryMigrationSource(), target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); @@ -33,7 +34,7 @@ public void Every_category_runs_in_a_fixed_order() .Where(category => category.Kind == MigrationCategoryKind.Optional) .Select(category => category.Id) .ToArray(); - var engine = BuildEngine(everyOptionalId, out _); + var engine = BuildEngine(everyOptionalId); var runOrder = engine.SelectCategories(MigrationCategoryKind.Required) .Concat(engine.SelectCategories(MigrationCategoryKind.Optional)) @@ -46,7 +47,7 @@ public void Every_category_runs_in_a_fixed_order() public void Only_configured_optional_categories_are_selected() { string[] configured = [MigrationCategoryIds.GroupComments, MigrationCategoryIds.EventLog, MigrationCategoryIds.ArchivedAndResolvedFailedMessages]; - var engine = BuildEngine(configured, out _); + var engine = BuildEngine(configured); var selected = engine.SelectCategories(MigrationCategoryKind.Optional); var left = MigrationCategoryRegistry.All @@ -63,18 +64,27 @@ .. selected.Select(Describe), } [Test] - public void A_category_removed_from_configuration_leaves_its_checkpoint_row_untouched() + public async Task A_category_removed_from_configuration_leaves_its_checkpoint_row_untouched() { - var engine = BuildEngine([], out var checkpointStore); - var previousRun = new MigrationCheckpoint("EventLog", MigrationCategoryState.CompleteWithErrors, "cursor-99", 40, 2, 42, null, DateTime.UtcNow, DateTime.UtcNow, DateTime.UtcNow, null); - checkpointStore.Upsert(previousRun).GetAwaiter().GetResult(); + // Dropping a category from the configuration must not restart, reset or delete what it already + // copied: those rows are in the target, and a row wound back to the start copies every one again. + var stillConfigured = MigrationCategoryIds.CustomChecks; + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var source = new InMemoryMigrationSource(); + source.Seed(stillConfigured, new MigrationRow("check-1", new object(), new Dictionary())); + var target = new InMemoryMigrationTarget(checkpointStore); + var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, [stillConfigured]); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + var previousRun = new MigrationCheckpoint(MigrationCategoryIds.EventLog, MigrationCategoryState.CompleteWithErrors, "cursor-99", 40, 2, 42, null, DateTime.UtcNow, DateTime.UtcNow, DateTime.UtcNow, null); + await checkpointStore.Upsert(previousRun); - var selected = engine.SelectCategories(MigrationCategoryKind.Optional); + var results = await engine.RunCategories(engine.SelectCategories(MigrationCategoryKind.Optional)); + var deselected = await checkpointStore.Read(MigrationCategoryIds.EventLog); using (Assert.EnterMultipleScope()) { - Assert.That(selected.Select(c => c.Id), Does.Not.Contain("EventLog")); - Assert.That(checkpointStore.Read("EventLog").GetAwaiter().GetResult(), Is.EqualTo(previousRun with { Version = 1 })); + Assert.That(results.Select(c => c.CategoryId), Is.EqualTo(new[] { stillConfigured }), "the run touched only the configured category"); + Assert.That(deselected, Is.EqualTo(previousRun with { Version = 1 }), "the row is still at the version the seeding save left it"); } } } diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineCopyTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineCopyTests.cs index aafb9d4935..c4e7898d97 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineCopyTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineCopyTests.cs @@ -40,6 +40,109 @@ public async Task Copies_every_row_and_finishes_Complete_when_nothing_was_skippe } } + [Test] + public async Task A_first_run_captures_the_source_total_and_every_batch_moves_LastProgressAt() + { + var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, Row("a"), Row("b"), Row("c"), Row("d")); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 2 }; + var startedAt = new DateTimeOffset(2026, 3, 4, 5, 6, 7, TimeSpan.Zero); + var betweenBatches = TimeSpan.FromMinutes(1); + var clock = new FakeTimeProvider(startedAt); + var stamps = new List(); + // The engine stamps the checkpoint before the target sees it, so moving the clock in here is what + // makes a stamp that never moves show up as two identical entries. + target.BeforeWrite = extending => + { + stamps.Add(extending.LastProgressAt); + clock.Advance(betweenBatches); + }; + var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, clock, options, NullLogger.Instance); + + var checkpoint = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(checkpoint.SourceTotal, Is.EqualTo(4), "the progress line and the verify percentage both read this"); + Assert.That(stamps, Is.EqualTo(new DateTime?[] { startedAt.UtcDateTime, (startedAt + betweenBatches).UtcDateTime }), "the stall watchdog stops a copy whose LastProgressAt stops moving, so every batch has to move it"); + Assert.That(checkpoint.LastProgressAt, Is.EqualTo((startedAt + betweenBatches).UtcDateTime), "the saved row carries the last batch's stamp"); + } + } + + [Test] + public async Task A_category_that_read_fewer_rows_than_the_source_holds_halts_instead_of_settling_complete() + { + // The row was counted at 10 against a source that yields 2, so without the shortfall check the + // category settles Complete with 8 rows never copied. + var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, Row("d"), Row("e")); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + await checkpointStore.Upsert(new MigrationCheckpoint(category.Id, MigrationCategoryState.InProgress, null, 0, 0, 10, null, DateTime.UtcNow, DateTime.UtcNow, null, null)); + var target = new InMemoryMigrationTarget(checkpointStore); + var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var checkpoint = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.Halted)); + Assert.That(checkpoint.LastError, Does.Contain("KnownEndpoints").And.Contain("10").And.Contain("2 copied")); + Assert.That(checkpoint.SettledAt, Is.Not.Null); + } + } + + [Test] + public async Task A_resumed_category_is_reconciled_on_the_totals_the_row_carries_not_the_rows_this_run_copied() + { + // Judging this run on the two rows it copies, rather than the four the row ends up counting, + // would halt every copy that was ever restarted. + var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, Row("a"), Row("b"), Row("c"), Row("d")); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + await checkpointStore.Upsert(new MigrationCheckpoint(category.Id, MigrationCategoryState.InProgress, "b", 2, 0, 4, null, DateTime.UtcNow, DateTime.UtcNow, null, null)); + var target = new InMemoryMigrationTarget(checkpointStore); + var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var checkpoint = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.Complete)); + Assert.That(checkpoint.CopiedCount, Is.EqualTo(4)); + } + } + + [Test] + public async Task A_source_that_grew_since_it_was_counted_still_settles_complete() + { + // The row was counted at 2 and the source now holds 3: the source instance was still ingesting + // while the copy ran. + var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, Row("a"), Row("b"), Row("c")); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + await checkpointStore.Upsert(new MigrationCheckpoint(category.Id, MigrationCategoryState.InProgress, null, 0, 0, 2, null, DateTime.UtcNow, DateTime.UtcNow, null, null)); + var target = new InMemoryMigrationTarget(checkpointStore); + var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var checkpoint = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.Complete)); + Assert.That(checkpoint.CopiedCount, Is.EqualTo(3)); + Assert.That(checkpoint.SourceTotal, Is.EqualTo(2), "re-counting would rebase the total the shortfall halt compares against, and the second run is the one a stale cursor is found on"); + } + } + [Test] public async Task A_category_with_no_rows_finishes_Complete_without_a_cursor() { diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineFailurePathTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineFailurePathTests.cs index fb26628421..a384994fcc 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineFailurePathTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineFailurePathTests.cs @@ -91,7 +91,7 @@ public async Task A_halted_category_is_re_attempted_on_the_next_run_and_resumes_ var halted = await firstRun.RunCategoryAsync(category); Assert.That(halted.State, Is.EqualTo(MigrationCategoryState.Halted)); - // Stopped on its first write, so the saved row is the restarted one rather than the completed one. + // Call 3 is the restarted run's first write, so the row read back next is a resumed copy, not a finished one. target.FailOnCallNumber = null; using var stopping = new CancellationTokenSource(); target.StopOnCall = (3, stopping); @@ -109,9 +109,9 @@ public async Task A_halted_category_is_re_attempted_on_the_next_run_and_resumes_ Assert.That(finished.State, Is.EqualTo(MigrationCategoryState.Complete)); Assert.That(finished.LastError, Is.Null, "a cleared halt does not leave a stale error on the row"); Assert.That(finished.CopiedCount, Is.EqualTo(4)); - var writtenIds = target.WrittenRows(category.Id).Select(r => r.SourceId).ToArray(); - Assert.That(writtenIds, Is.EquivalentTo(new[] { "a", "b", "c", "d" })); - Assert.That(writtenIds.Distinct().Count(), Is.EqualTo(writtenIds.Length), "no duplicates across the halt"); + Assert.That(target.WrittenRows(category.Id).Select(r => r.SourceId), Is.EquivalentTo(new[] { "a", "b", "c", "d" })); + // The target de-duplicates, as the real ones do, so what it kept can never show a row sent twice. + Assert.That(target.RowsHandedToWrite(category.Id).Select(r => r.SourceId), Is.Unique, "no duplicates across the halt"); } } @@ -191,7 +191,7 @@ public async Task A_target_with_no_batch_size_for_a_category_halts_it_and_the_ne } [Test] - public void A_halt_whose_save_fails_still_logs_the_exception_that_caused_it() + public void A_write_failure_halt_logs_its_exception_before_it_settles() { var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; var source = new InMemoryMigrationSource(); @@ -207,7 +207,7 @@ public void A_halt_whose_save_fails_still_logs_the_exception_that_caused_it() } [Test] - public void A_threshold_halt_whose_save_fails_still_logs_why_it_halted() + public void A_threshold_halt_logs_its_reason_before_it_settles() { var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; var source = new InMemoryMigrationSource(); @@ -225,6 +225,24 @@ public void A_threshold_halt_whose_save_fails_still_logs_why_it_halted() Assert.That(logger.Entries.Where(e => e.Level == LogLevel.Error).Select(e => e.Message), Has.Some.Contains("Halted: 1 of 1 rows skipped")); } + [Test] + public async Task A_shortfall_halt_logs_its_reason_before_it_settles() + { + var category = MigrationCategoryRegistry.Find("KnownEndpoints")!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, Row("d"), Row("e")); + var checkpointStore = new HaltSaveFailsCheckpointStore(); + // The row is counted at 10 against a source holding 2, so the copy reaches the end short of its total. + await checkpointStore.Upsert(new MigrationCheckpoint(category.Id, MigrationCategoryState.InProgress, null, 0, 0, 10, null, DateTime.UtcNow, DateTime.UtcNow, null, null)); + var target = new InMemoryMigrationTarget(checkpointStore); + var logger = new CapturingLogger(); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []), logger); + + Assert.ThrowsAsync(() => engine.RunCategoryAsync(category)); + + Assert.That(logger.Entries.Where(e => e.Level == LogLevel.Error).Select(e => e.Message), Has.Some.Contains("accounted for 2 of the 10 rows"), "the save that records the reason is the one that failed, so a halt that settles before it logs leaves an operator a copy that stopped with nothing saying why"); + } + [Test] public async Task A_checkpoint_conflict_leaves_the_other_writer_alone_instead_of_halting_over_it() { diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineHaltTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineHaltTests.cs index 47a77f85d7..fc4467fb80 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineHaltTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineHaltTests.cs @@ -186,4 +186,159 @@ public async Task A_restart_after_a_threshold_halt_counts_only_its_own_skips_and Assert.That((finished.CopiedCount, finished.SkippedCount), Is.EqualTo((880L, 120L)), "copied, skipped at the end"); } } + + [Test] + public async Task A_category_smaller_than_the_floor_that_loses_every_row_halts_rather_than_completing() + { + // Ninety rows is under the hundred-row floor, so the threshold the engine checks after every batch + // can never fire, however many rows are lost. + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 90).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 30 }; + foreach (var i in Enumerable.Range(1, 90)) + { + target.RejectKey($"row-{i}", MigrationSkipReason.RequiredValueMissing); + } + var options = new MigrationEngineOptions(TimeSpan.Zero, HaltThresholdPercent: 5, HaltThresholdMinimum: 100, []); + + var checkpoint = await new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance).RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.Halted), "a required category that copied nothing must not let the host open"); + Assert.That(checkpoint.CopiedCount, Is.Zero); + Assert.That(checkpoint.LastError, Does.Contain("most of them")); + } + } + + [Test] + public async Task A_category_that_lost_every_row_stays_halted_when_it_is_restarted_with_nothing_fixed() + { + // The halt tells the operator to restart, and the restart resumes from a cursor already at the end, + // so the run that clears the halt is the one that reads nothing and can judge nothing. + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 90).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 30 }; + foreach (var i in Enumerable.Range(1, 90)) + { + target.RejectKey($"row-{i}", MigrationSkipReason.RequiredValueMissing); + } + var options = new MigrationEngineOptions(TimeSpan.Zero, HaltThresholdPercent: 5, HaltThresholdMinimum: 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var halted = await engine.RunCategoryAsync(category); + var restarted = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(halted.State, Is.EqualTo(MigrationCategoryState.Halted)); + Assert.That(restarted.State, Is.EqualTo(MigrationCategoryState.Halted), "a restart that copied nothing reported the category finished, and the host would open on an empty table"); + Assert.That(restarted.CopiedCount, Is.Zero); + } + } + + // A transient failure on the last read halts a category that copied everything. The restart reads nothing, + // so a rule that asks only whether this run read rows would hold it halted with nothing left to fix. + [Test] + public async Task A_category_that_copied_every_row_before_it_halted_completes_on_the_restart() + { + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 30).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 30 }; + var options = new MigrationEngineOptions(TimeSpan.Zero, HaltThresholdPercent: 5, HaltThresholdMinimum: 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var copied = await engine.RunCategoryAsync(category); + await checkpointStore.Upsert(copied with { State = MigrationCategoryState.Halted, LastError = "the source connection reset on the last read" }); + + var restarted = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(restarted.State, Is.EqualTo(MigrationCategoryState.Complete), "a category holding every one of its rows was left halted with nothing an operator could fix"); + Assert.That(restarted.CopiedCount, Is.EqualTo(30)); + } + } + + // The halt lives on the row, and the run that clears it is the one that settles. A start killed in between + // must not leave the row saying the category is fine. + [Test] + public async Task A_restart_killed_before_it_settles_does_not_let_the_next_one_report_the_category_finished() + { + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 90).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 30 }; + foreach (var i in Enumerable.Range(1, 90)) + { + target.RejectKey($"row-{i}", MigrationSkipReason.RequiredValueMissing); + } + var options = new MigrationEngineOptions(TimeSpan.Zero, HaltThresholdPercent: 5, HaltThresholdMinimum: 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var halted = await engine.RunCategoryAsync(category); + // What a start killed after the InProgress save but before the settle leaves behind. + await checkpointStore.Upsert(halted with { State = MigrationCategoryState.InProgress, LastError = null, SettledAt = null }); + + var restarted = await engine.RunCategoryAsync(category); + + Assert.That(restarted.State, Is.EqualTo(MigrationCategoryState.Halted), "an interrupted restart erased the halt, so the next one reported an empty category finished and the host would open"); + } + + [Test] + public async Task A_small_category_losing_rows_the_product_would_drop_anyway_still_completes() + { + // Benign skips are rows the target would have deleted anyway, so no number of them may halt a + // category, and the most-of-the-run rule must not be the exception that brings that back. + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 90).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 30 }; + foreach (var i in Enumerable.Range(1, 90)) + { + target.RejectKey($"row-{i}", MigrationSkipReason.EndpointNotKnown, benign: true); + } + var options = new MigrationEngineOptions(TimeSpan.Zero, HaltThresholdPercent: 5, HaltThresholdMinimum: 100, []); + + var checkpoint = await new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance).RunCategoryAsync(category); + + Assert.That(checkpoint.State, Is.EqualTo(MigrationCategoryState.CompleteWithErrors)); + } + + // The restart after a halt reads nothing, so the whole row is judged at once rather than this run alone. + // Sixty of these ninety rows are gone and none of them is a loss, so there is nothing to stay halted for. + [Test] + public async Task A_category_halted_after_losing_only_rows_the_product_would_drop_anyway_completes_on_the_restart() + { + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 90).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 30 }; + foreach (var i in Enumerable.Range(1, 90).Where(i => i % 3 != 0)) + { + target.RejectKey($"row-{i}", MigrationSkipReason.EndpointNotKnown, benign: true); + } + var options = new MigrationEngineOptions(TimeSpan.Zero, HaltThresholdPercent: 5, HaltThresholdMinimum: 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var copied = await engine.RunCategoryAsync(category); + await checkpointStore.Upsert(copied with { State = MigrationCategoryState.Halted, LastError = "the source connection reset on the last read" }); + + var restarted = await engine.RunCategoryAsync(category); + + using (Assert.EnterMultipleScope()) + { + Assert.That(restarted.State, Is.EqualTo(MigrationCategoryState.CompleteWithErrors), "a category that lost nothing the product wanted was left halted for ever, and the host never opens"); + Assert.That((restarted.CopiedCount, restarted.SkippedCount), Is.EqualTo((30L, 60L))); + } + } } diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineOptionsTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineOptionsTests.cs index 0ede2f09b3..90d4e802f2 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineOptionsTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineOptionsTests.cs @@ -61,14 +61,13 @@ public void Refuses_an_unknown_optional_category_id() var ex = Assert.Throws(() => MigrationEngineOptions.FromSettings(Namespace)); - Assert.That(ex.Message, Does.Contain("NoSuchCategory")); + Assert.That(ex.Message, Does.Contain("NoSuchCategory").And.Contain(MigrationSettings.OptionalCategoriesKey), "the refusal has to name both the typo and the setting holding it for the customer to fix it"); } [Test] public void Refuses_a_required_category_id_named_as_optional() { - // EndpointSettings is required, not optional: naming it here is a customer mistake, not - // a valid way to force it. Required categories are never a matter of configuration. + // EndpointSettings is required, not optional: naming it here is a customer mistake, not a way to force it. Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "EndpointSettings"); var ex = Assert.Throws(() => MigrationEngineOptions.FromSettings(Namespace)); diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineResumeTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineResumeTests.cs index d21760171f..278d2f23f0 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineResumeTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineResumeTests.cs @@ -82,9 +82,9 @@ public async Task Restarting_after_a_mid_category_stop_produces_no_duplicates_an { Assert.That(finalCheckpoint.State, Is.EqualTo(MigrationCategoryState.Complete)); Assert.That(finalCheckpoint.CopiedCount, Is.EqualTo(6)); - var writtenIds = target.WrittenRows(category.Id).Select(r => r.SourceId).ToArray(); - Assert.That(writtenIds, Is.EquivalentTo(allIds), "no gaps"); - Assert.That(writtenIds.Distinct().Count(), Is.EqualTo(writtenIds.Length), "no duplicates"); + Assert.That(target.WrittenRows(category.Id).Select(r => r.SourceId), Is.EquivalentTo(allIds), "no gaps"); + // The target de-duplicates, as the real ones do, so what it kept can never show a row sent twice. + Assert.That(target.RowsHandedToWrite(category.Id).Select(r => r.SourceId), Is.Unique, "no duplicates: the restart resumes past the rows the first run committed"); } } diff --git a/src/ServiceControl.UnitTests/Migration/MigrationEngineSkipReasonTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationEngineSkipReasonTests.cs index 2e07501603..a2609c9390 100644 --- a/src/ServiceControl.UnitTests/Migration/MigrationEngineSkipReasonTests.cs +++ b/src/ServiceControl.UnitTests/Migration/MigrationEngineSkipReasonTests.cs @@ -25,10 +25,10 @@ public async Task Reasons_the_target_reports_add_up_across_batches_on_the_checkp source.Seed(category.Id, Row("a"), Row("b"), Row("c"), Row("d")); var checkpointStore = new InMemoryMigrationCheckpointStore(); var target = new InMemoryMigrationTarget(checkpointStore) { DefaultBatchSize = 3 }; - // One reason exists, so this pins the total rather than the split between reasons. - target.RejectKey("a", MigrationSkipReason.BodyUnreadable); - target.RejectKey("b", MigrationSkipReason.BodyUnreadable); - target.RejectKey("d", MigrationSkipReason.BodyUnreadable); + // Two reasons over two batches: a and b land in the first, d in the second, so the saved map has to merge both. + target.RejectKey("a", MigrationSkipReason.RequiredValueMissing); + target.RejectKey("b", MigrationSkipReason.RequiredValueMissing); + target.RejectKey("d", MigrationSkipReason.EndpointNotKnown); var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []); var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); @@ -37,8 +37,8 @@ public async Task Reasons_the_target_reports_add_up_across_batches_on_the_checkp using (Assert.EnterMultipleScope()) { Assert.That(checkpoint.SkippedCount, Is.EqualTo(3)); - Assert.That(checkpoint.SkipReasons, Is.EquivalentTo(new Dictionary { [MigrationSkipReason.BodyUnreadable] = 3 })); - Assert.That((await checkpointStore.Read(category.Id))!.SkipReasons, Is.EquivalentTo(new Dictionary { [MigrationSkipReason.BodyUnreadable] = 3 })); + Assert.That(checkpoint.SkipReasons, Is.EquivalentTo(new Dictionary { [MigrationSkipReason.RequiredValueMissing] = 2, [MigrationSkipReason.EndpointNotKnown] = 1 })); + Assert.That((await checkpointStore.Read(category.Id))!.SkipReasons, Is.EquivalentTo(new Dictionary { [MigrationSkipReason.RequiredValueMissing] = 2, [MigrationSkipReason.EndpointNotKnown] = 1 })); } } @@ -131,9 +131,34 @@ public async Task A_target_counting_more_benign_skips_than_skipped_rows_halts_th } } + [Test] + public async Task A_target_whose_reported_counts_disagree_with_the_checkpoint_it_committed_never_finishes_the_category() + { + var category = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + var source = new InMemoryMigrationSource(); + source.Seed(category.Id, [.. Enumerable.Range(1, 300).Select(i => Row($"row-{i}"))]); + var checkpointStore = new InMemoryMigrationCheckpointStore(); + var target = new MiscountedCopyTarget(checkpointStore); + var options = new MigrationEngineOptions(TimeSpan.Zero, 5, 100, []); + var engine = new MigrationEngine(source, target, checkpointStore, new FakeTimeProvider(), options, NullLogger.Instance); + + var settled = await engine.RunCategoryAsync(category); + + var stored = await checkpointStore.Read(category.Id); + + using (Assert.EnterMultipleScope()) + { + Assert.That(settled.State, Is.EqualTo(MigrationCategoryState.Halted)); + Assert.That(stored!.LastError, Does.Contain("but the checkpoint it committed moved by"), "the halt has to name which two accounts disagreed, or the operator goes looking for the wrong problem"); + Assert.That(stored.State.IsFinished(), Is.False, "the halt threshold never saw the 150 rows the target dropped, so the category settled finished and the host opened on half a category"); + } + } + sealed class OverCountedBenignTarget(IMigrationCheckpointStore checkpointStore) : IMigrationTarget { - public int BatchSizeFor(MigrationCategory category) => 10; + public Task Open(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult(10); public async Task Write(MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default) { @@ -143,11 +168,36 @@ public async Task Write(MigrationCategory category, Migrat } public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult(0L); + + public IReadOnlyCollection SupportedCategoryIds => [.. MigrationCategoryRegistry.All.Select(category => category.Id)]; + } + + // Commits half of every batch as skipped and reports the whole batch copied. The saved counts still add up + // to the source total, so nothing later in the run can notice, and the halt threshold sees a clean copy. + sealed class MiscountedCopyTarget(IMigrationCheckpointStore checkpointStore) : IMigrationTarget + { + public Task Open(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult(10); + + public async Task Write(MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default) + { + var skipped = batch.Rows.Count / 2; + var reasons = new Dictionary { [MigrationSkipReason.RequiredValueMissing] = skipped }; + var saved = await checkpointStore.Upsert(checkpointToExtend.Extend(batch.Rows.Count - skipped, skipped, 0, reasons), cancellationToken); + return new MigrationWriteResult(saved, batch.Rows.Count, 0, []); + } + + public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult(0L); + + public IReadOnlyCollection SupportedCategoryIds => [.. MigrationCategoryRegistry.All.Select(category => category.Id)]; } sealed class UnexplainedSkipTarget(IMigrationCheckpointStore checkpointStore) : IMigrationTarget { - public int BatchSizeFor(MigrationCategory category) => 10; + public Task Open(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task BatchSizeFor(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult(10); public async Task Write(MigrationCategory category, MigrationBatch batch, MigrationCheckpoint checkpointToExtend, CancellationToken cancellationToken = default) { @@ -156,5 +206,7 @@ public async Task Write(MigrationCategory category, Migrat } public Task Count(MigrationCategory category, CancellationToken cancellationToken = default) => Task.FromResult(0L); + + public IReadOnlyCollection SupportedCategoryIds => [.. MigrationCategoryRegistry.All.Select(category => category.Id)]; } } diff --git a/src/ServiceControl.UnitTests/Migration/MigrationPairIsSupportedCheckTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationPairIsSupportedCheckTests.cs new file mode 100644 index 0000000000..c25bf44d0c --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/MigrationPairIsSupportedCheckTests.cs @@ -0,0 +1,35 @@ +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Linq; +using NUnit.Framework; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Migration.Checks; +using ServiceControl.Persistence; + +[TestFixture] +class MigrationPairIsSupportedCheckTests +{ + [OneTimeSetUp] + public static void TheTreeIsBuilt() => + Assert.That( + PersistenceFactory.SqlPersistenceNames.Where(name => PersistenceManifestLibrary.Find(name) is null), + Is.Empty, + "These persistence manifests did not resolve, so this fixture cannot tell a supported pair from an unsupported one. Build the tree first with: dotnet build src --configuration Release -graph"); + + [TestCase("SQLServer")] + [TestCase("PostgreSQL")] + public void RavenDB_to_a_SQL_persister_is_supported(string target) => + Assert.DoesNotThrowAsync(() => Check(target).Run()); + + [Test] + public void A_RavenDB_target_is_refused_naming_the_target_setting() + { + var exception = Assert.ThrowsAsync(() => Check("RavenDB").Run()); + + Assert.That(exception.Message, Does.Contain("ServiceControl/PersistenceType")); + } + + static MigrationPairIsSupportedCheck Check(string target) => + new(new Settings(transportType: "LearningTransport", persisterType: target, errorRetentionPeriod: TimeSpan.FromDays(10))); +} diff --git a/src/ServiceControl.UnitTests/Migration/MigrationStartupCheckRunnerTests.cs b/src/ServiceControl.UnitTests/Migration/MigrationStartupCheckRunnerTests.cs new file mode 100644 index 0000000000..2767013bb3 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/MigrationStartupCheckRunnerTests.cs @@ -0,0 +1,75 @@ +#nullable enable +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Collections.Generic; +using System.Threading; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Migration; +using ServiceControl.Persistence.DataMigration; + +[TestFixture] +class MigrationStartupCheckRunnerTests +{ + [Test] + public async Task Checks_run_in_the_order_they_were_given() + { + var ran = new List(); + + await MigrationStartupCheckRunner.Run( + [ + Check("the first", _ => ran.Add("first")), + Check("the second", _ => ran.Add("second")), + Check("the third", _ => ran.Add("third")) + ]); + + Assert.That(ran, Is.EqualTo(new[] { "first", "second", "third" }), "the databases are opened by checks that sit behind the ones refusing the configuration outright"); + } + + [Test] + public void The_first_check_to_refuse_stops_the_ones_behind_it() + { + var ran = new List(); + var failure = new InvalidOperationException("ServiceControl/RetryHistoryDepth is 0"); + + var exception = Assert.ThrowsAsync(() => MigrationStartupCheckRunner.Run( + [ + Check("the first", _ => ran.Add("first")), + Check("the migration target is ready", _ => throw failure), + Check("the last", _ => ran.Add("last")) + ])); + + using (Assert.EnterMultipleScope()) + { + Assert.That(ran, Is.EqualTo(new[] { "first" }), "a check behind a refusal would run against a configuration already known to be wrong, and the two that open a database are at the end of the list"); + Assert.That(exception!.Message, Does.Contain("the migration target is ready").And.Contain(failure.Message), "the operator needs to know which check refused and why"); + Assert.That(exception.InnerException, Is.SameAs(failure), "the original carries the stack and the detail the summary leaves out"); + } + } + + [Test] + public void A_host_being_stopped_is_not_a_check_refusing() + { + using var stopping = new CancellationTokenSource(); + stopping.Cancel(); + + var exception = Assert.ThrowsAsync(() => MigrationStartupCheckRunner.Run( + [Check("the first", token => token.ThrowIfCancellationRequested())], stopping.Token)); + + Assert.That(exception!.Message, Does.Not.Contain("Migration startup check"), "a stop dressed up as a refusal sends the operator after a setting that was never wrong"); + } + + static IMigrationStartupCheck Check(string name, Action run) => new FakeCheck(name, run); + + sealed class FakeCheck(string name, Action run) : IMigrationStartupCheck + { + public string Name => name; + + public Task Run(CancellationToken cancellationToken = default) + { + run(cancellationToken); + return Task.CompletedTask; + } + } +} diff --git a/src/ServiceControl.UnitTests/Migration/RequiredCopyGateTests.cs b/src/ServiceControl.UnitTests/Migration/RequiredCopyGateTests.cs new file mode 100644 index 0000000000..42707844e2 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/RequiredCopyGateTests.cs @@ -0,0 +1,184 @@ +#nullable enable +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Collections.Generic; +using System.Linq; +using Microsoft.Extensions.Logging; +using NUnit.Framework; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Migration; +using ServiceControl.Persistence.DataMigration; +using ServiceControl.UnitTests.Migration.Fakes; + +[TestFixture] +class RequiredCopyGateTests +{ + [TestCase(MigrationCategoryState.Complete)] + [TestCase(MigrationCategoryState.CompleteWithErrors)] + public void A_finished_required_category_lets_the_host_open(MigrationCategoryState state) => + Assert.DoesNotThrow(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete([EndpointSettings], [Checkpoint(state)], null, NewSettings())); + + [TestCase(MigrationCategoryState.Halted)] + [TestCase(MigrationCategoryState.InProgress)] + [TestCase(MigrationCategoryState.NotStarted)] + // Blocked happens because EndpointSettings must follow KnownEndpoints. + [TestCase(MigrationCategoryState.Blocked)] + public void A_required_category_that_is_not_finished_keeps_the_host_closed(MigrationCategoryState state) + { + var exception = Assert.Throws(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete([EndpointSettings], [Checkpoint(state)], null, NewSettings())); + + Assert.That(exception.Message, Does.Contain(MigrationCategoryIds.EndpointSettings).And.Contain(state.ToString())); + } + + [Test] + public void A_category_that_reported_no_checkpoint_at_all_keeps_the_host_closed() + { + var exception = Assert.Throws(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete([EndpointSettings], [], null, NewSettings())); + + Assert.That(exception.Message, Does.Contain(MigrationCategoryIds.EndpointSettings), + "an empty result reads as success unless the gate checks what it asked for against what came back"); + } + + [Test] + public void Of_two_attempted_categories_the_one_that_reported_nothing_is_the_one_named() + { + var exception = Assert.Throws(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete( + [KnownEndpoints, EndpointSettings], + [Checkpoint(MigrationCategoryState.Complete, MigrationCategoryIds.KnownEndpoints)], + null, + NewSettings())); + + Assert.That(exception.Message, Does.Contain($"{MigrationCategoryIds.EndpointSettings} reported no checkpoint at all").And.Not.Contain(MigrationCategoryIds.KnownEndpoints), + "comparing counts instead of ids would let a two-category run through with one category's fate unknown"); + } + + [Test] + public void The_refusal_reads_to_its_end_as_a_sentence_and_says_how_to_get_back() + { + var exception = Assert.Throws(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete([EndpointSettings], [Checkpoint(MigrationCategoryState.Halted)], null, NewSettings())); + + Assert.That(exception.Message, Does.Contain("the copy resumes from its last committed batch. Nothing has opened on SQLServer yet") + .And.Contain($"setting {MigrationSettings.EnabledKey}=false and pointing PersistenceType back at RavenDB"), + "the operator reads the whole message, and the rollback instructions are at the end of it"); + } + + [Test] + public void A_stall_is_reported_with_the_limit_that_was_actually_measured() + { + var exception = Assert.Throws(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete( + [EndpointSettings], [Checkpoint(MigrationCategoryState.InProgress)], MigrationCategoryIds.EndpointSettings, NewSettings())); + + Assert.That(exception.Message, Does.Contain($"committed nothing for {MigrationStartup.ClosedWindowProgress.StallLimit.TotalMinutes:0.#} minutes") + .And.Contain("looking for a setting to change. Nothing has opened on SQLServer yet"), + "prose spelling the limit out goes stale the moment the constant moves"); + } + + [Test] + public void A_stall_on_a_category_that_finished_is_not_blamed_for_what_is_outstanding() + { + var exception = Assert.Throws(() => MigrationStartup.RefuseIfAnyCategoryDidNotComplete( + [KnownEndpoints, EndpointSettings], + [Checkpoint(MigrationCategoryState.Complete, MigrationCategoryIds.KnownEndpoints), Checkpoint(MigrationCategoryState.Halted)], + MigrationCategoryIds.KnownEndpoints, + NewSettings())); + + Assert.That(exception.Message, Does.Not.Contain("committed nothing for").And.Contain("Fix the cause and restart"), + "blaming a stall for a halt it had nothing to do with sends the operator after the wrong thing"); + } + + // The last thing said before the cutover is one way, and the only place a skipped total appears at all. + [Test] + public void A_category_that_skipped_rows_is_reported_as_rows_nothing_will_come_back_for() + { + var logger = new CapturingLogger(); + + MigrationStartup.ReportWhatTheCopyLeftBehind( + [Settled(MigrationCategoryState.CompleteWithErrors, copied: 7, skipped: 5, new Dictionary + { + [MigrationSkipReason.EndpointNotKnown] = 3, + [MigrationSkipReason.RequiredValueMissing] = 2 + })], + logger, + RunStartedAt); + + var entry = logger.Entries.Single(); + + Assert.Multiple(() => + { + Assert.That(entry.Level, Is.EqualTo(LogLevel.Warning), "rows that are never coming back are not an informational matter"); + Assert.That(entry.Message, Does.Contain("5 skipped"), "the total is the number the operator decides on"); + Assert.That(entry.Message, Does.Contain("EndpointNotKnown 3").And.Contain("RequiredValueMissing 2"), "a total with no reasons cannot be acted on"); + Assert.That(entry.Message, Does.Contain("no later run will fetch them"), "without this the operator waits for a copy that is already over"); + }); + } + + [Test] + public void A_category_that_skipped_nothing_says_so_without_raising_a_warning() + { + var logger = new CapturingLogger(); + + MigrationStartup.ReportWhatTheCopyLeftBehind([Settled(MigrationCategoryState.Complete, copied: 7, skipped: 0, null)], logger, RunStartedAt); + + var entry = logger.Entries.Single(); + + Assert.Multiple(() => + { + Assert.That(entry.Level, Is.EqualTo(LogLevel.Information), "a clean copy warning about nothing trains the operator to ignore the warning that matters"); + Assert.That(entry.Message, Does.Contain("nothing skipped")); + }); + } + + // Restarting a migrated instance used to reprint the whole copy summary in the present tense, so an operator + // restarting to change a setting read "7 copied" and had no way to tell the copy had not run again. + [Test] + public void A_category_an_earlier_run_finished_is_not_reported_as_copied_again() + { + var logger = new CapturingLogger(); + + var alreadyDone = Settled(MigrationCategoryState.Complete, copied: 7, skipped: 0, null) + with + { SettledAt = RunStartedAt.AddMinutes(-5) }; + + MigrationStartup.ReportWhatTheCopyLeftBehind([alreadyDone], logger, RunStartedAt); + + var entry = logger.Entries.Single(); + + Assert.Multiple(() => + { + Assert.That(entry.Message, Does.Contain("already finished before this start"), "without this the line is indistinguishable from a copy that just ran"); + Assert.That(entry.Message, Does.Contain("This start copied nothing"), "the operator needs to know nothing was written to a target that is already serving"); + Assert.That(entry.Message, Does.Contain("7"), "the historical total is still worth stating, just not as this run's work"); + Assert.That(entry.Level, Is.EqualTo(LogLevel.Information)); + }); + } + + // A category this run actually settled must keep the ordinary wording, or the fix above would silence every report. + [Test] + public void A_category_this_run_finished_is_still_reported_as_copied() + { + var logger = new CapturingLogger(); + + var justDone = Settled(MigrationCategoryState.Complete, copied: 7, skipped: 0, null) + with + { SettledAt = RunStartedAt.AddSeconds(2) }; + + MigrationStartup.ReportWhatTheCopyLeftBehind([justDone], logger, RunStartedAt); + + Assert.That(logger.Entries.Single().Message, Does.Contain("7 copied").And.Not.Contain("already finished")); + } + + static readonly DateTime RunStartedAt = new(2026, 9, 20, 12, 0, 0, DateTimeKind.Utc); + + static MigrationCheckpoint Settled(MigrationCategoryState state, long copied, long skipped, IReadOnlyDictionary? skipReasons) => + new(MigrationCategoryIds.EndpointSettings, state, null, copied, skipped, null, skipReasons, null, null, null, null); + + static readonly MigrationCategory EndpointSettings = MigrationCategoryRegistry.Find(MigrationCategoryIds.EndpointSettings)!; + static readonly MigrationCategory KnownEndpoints = MigrationCategoryRegistry.Find(MigrationCategoryIds.KnownEndpoints)!; + + static MigrationCheckpoint Checkpoint(MigrationCategoryState state, string categoryId = MigrationCategoryIds.EndpointSettings) => + new(categoryId, state, null, 0, 0, null, null, null, null, null, null); + + static Settings NewSettings() => + new(transportType: "LearningTransport", persisterType: "SQLServer", errorRetentionPeriod: TimeSpan.FromDays(10)); +} diff --git a/src/ServiceControl.UnitTests/Migration/SelectedCategoriesAreCoherentCheckTests.cs b/src/ServiceControl.UnitTests/Migration/SelectedCategoriesAreCoherentCheckTests.cs new file mode 100644 index 0000000000..1a4e502844 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/SelectedCategoriesAreCoherentCheckTests.cs @@ -0,0 +1,51 @@ +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceControl.Migration.Checks; +using ServiceControl.Persistence.DataMigration; + +[TestFixture] +[NonParallelizable] +class SelectedCategoriesAreCoherentCheckTests +{ + [TearDown] + public void ClearOptionalCategories() => Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", null); + + [Test] + public async Task No_optional_categories_selected_passes_and_selects_none() + { + var check = new SelectedCategoriesAreCoherentCheck(); + + await check.Run(); + + Assert.That(check.Options.SelectedOptionalCategoryIds, Is.Empty); + } + + [Test] + public async Task The_selected_ids_are_exposed_and_the_spacing_is_tolerated() + { + Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "EventLog, CustomChecks"); + var check = new SelectedCategoriesAreCoherentCheck(); + + await check.Run(); + + Assert.That(check.Options.SelectedOptionalCategoryIds, Is.EquivalentTo(new[] { MigrationCategoryIds.EventLog, MigrationCategoryIds.CustomChecks })); + } + + [Test] + public void An_unknown_optional_category_is_refused_and_the_typo_is_named() + { + Environment.SetEnvironmentVariable("SERVICECONTROL_MIGRATION_OPTIONALCATEGORIES", "EventLogs"); + + var exception = Assert.ThrowsAsync(async () => + await new SelectedCategoriesAreCoherentCheck().Run()); + + using (Assert.EnterMultipleScope()) + { + Assert.That(exception.Message, Does.Contain("EventLogs")); + Assert.That(exception.Message, Does.Contain(MigrationSettings.OptionalCategoriesKey)); + } + } +} diff --git a/src/ServiceControl.UnitTests/Migration/StoppedCopyExplanationTests.cs b/src/ServiceControl.UnitTests/Migration/StoppedCopyExplanationTests.cs new file mode 100644 index 0000000000..230e4fa926 --- /dev/null +++ b/src/ServiceControl.UnitTests/Migration/StoppedCopyExplanationTests.cs @@ -0,0 +1,203 @@ +#nullable enable +namespace ServiceControl.UnitTests.Migration; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using NUnit.Framework; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Migration; +using ServiceControl.Persistence.DataMigration; + +// The copy can stop three ways that are not the engine's to report: the stall watchdog cancelling it, the +// checkpoint store going down with the stall, and a second instance writing to the same database. +[TestFixture] +class StoppedCopyExplanationTests +{ + [Test] + public async Task A_copy_the_stall_watchdog_stopped_comes_back_with_what_the_store_holds() + { + var store = Holding(InProgress(MigrationCategoryIds.KnownEndpoints), Finished(MigrationCategoryIds.EndpointSettings)); + + var finished = await MigrationStartup.CopyOrExplainWhyItStopped( + CancelledCopy, + () => MigrationCategoryIds.KnownEndpoints, + store, + Attempted(MigrationCategoryIds.KnownEndpoints, MigrationCategoryIds.EndpointSettings), + NewSettings()); + + Assert.That(finished.Select(checkpoint => checkpoint.CategoryId).ToArray(), Is.EquivalentTo(new[] { MigrationCategoryIds.KnownEndpoints, MigrationCategoryIds.EndpointSettings }), + "the refusal names what is outstanding from these, so a stall that lets the cancellation out instead tells the operator only that something stopped"); + } + + [Test] + public async Task A_category_this_run_never_attempted_is_left_out_of_what_comes_back() + { + var store = Holding(InProgress(MigrationCategoryIds.KnownEndpoints), InProgress(MigrationCategoryIds.EndpointSettings)); + + var finished = await MigrationStartup.CopyOrExplainWhyItStopped( + CancelledCopy, + () => MigrationCategoryIds.KnownEndpoints, + store, + Attempted(MigrationCategoryIds.KnownEndpoints), + NewSettings()); + + Assert.That(finished.Select(checkpoint => checkpoint.CategoryId).ToArray(), Is.EqualTo(new[] { MigrationCategoryIds.KnownEndpoints }), + "a row an earlier run left in progress would otherwise keep this host closed over a category it never tried to copy"); + } + + [Test] + public void A_cancellation_with_no_stall_behind_it_stays_a_cancellation() + { + var store = Holding(InProgress(MigrationCategoryIds.KnownEndpoints)); + + Assert.That(async () => await MigrationStartup.CopyOrExplainWhyItStopped(CancelledCopy, NoStall, store, Attempted(MigrationCategoryIds.KnownEndpoints), NewSettings()), + Throws.InstanceOf(), + "only the watchdog turns a cancellation into a refusal, and it did not fire here"); + + Assert.That(store.Reads, Is.Zero, "reading the store means the recovery ran for a stall that never happened"); + } + + [Test] + public void A_host_shutting_down_while_a_category_is_stalled_stays_a_cancellation() + { + using var shuttingDown = new CancellationTokenSource(); + shuttingDown.Cancel(); + var store = Holding(InProgress(MigrationCategoryIds.KnownEndpoints)); + + Assert.That(async () => await MigrationStartup.CopyOrExplainWhyItStopped( + CancelledCopy, + () => MigrationCategoryIds.KnownEndpoints, + store, + Attempted(MigrationCategoryIds.KnownEndpoints), + NewSettings(), + shuttingDown.Token), + Throws.InstanceOf(), + "a shutdown is not a failure, so it must not come out as a refusal telling the operator to go looking at the source and the target"); + } + + [Test] + public void A_stall_the_store_cannot_be_read_after_reports_the_stall_first_and_the_store_second() + { + var unreachable = new TimeoutException("checkpoint store unreachable"); + + var exception = Assert.ThrowsAsync(async () => await MigrationStartup.CopyOrExplainWhyItStopped( + CancelledCopy, + () => MigrationCategoryIds.KnownEndpoints, + Failing(unreachable), + Attempted(MigrationCategoryIds.KnownEndpoints), + NewSettings())); + + Assert.Multiple(() => + { + Assert.That(exception!.Message, Does.Contain($"committed nothing for {MigrationStartup.ClosedWindowProgress.StallLimit.TotalMinutes:0.#} minutes"), + "the stall is why the copy stopped, and a store error reported in its place sends the operator after the wrong thing"); + Assert.That(exception.Message, Does.Contain("the checkpoint store is unreachable as well: checkpoint store unreachable"), + "without the store's own words there is nothing to act on"); + Assert.That(exception.Message, Does.Contain("Nothing has opened on SQLServer yet"), "every other refusal says how to go back, and this one is the worst to be stuck in"); + Assert.That(exception.InnerException, Is.SameAs(unreachable)); + }); + } + + [Test] + public void A_shutdown_that_lands_during_the_read_back_is_not_reported_as_an_unreachable_store() + { + using var shuttingDown = new CancellationTokenSource(); + var store = Failing(new OperationCanceledException(), shuttingDown.Cancel); + + Assert.That(async () => await MigrationStartup.CopyOrExplainWhyItStopped( + CancelledCopy, + () => MigrationCategoryIds.KnownEndpoints, + store, + Attempted(MigrationCategoryIds.KnownEndpoints), + NewSettings(), + shuttingDown.Token), + Throws.InstanceOf(), + "the host is stopping, so the store is not unreachable and saying it is would have an operator checking a database that is fine"); + } + + [Test] + public void A_checkpoint_saved_by_another_instance_says_which_instance_to_stop() + { + var conflict = new MigrationCheckpointConflictException("Checkpoint KnownEndpoints was saved from version 3, but the stored row is at version 4."); + + var exception = Assert.ThrowsAsync(async () => await MigrationStartup.CopyOrExplainWhyItStopped( + Task.FromException>(conflict), + NoStall, + Holding(), + Attempted(MigrationCategoryIds.KnownEndpoints), + NewSettings())); + + Assert.Multiple(() => + { + Assert.That(exception!.Message, Does.Contain("a second ServiceControl pointed at the same SQLServer database"), + "the store's own message names a version, which tells nobody what is actually wrong"); + Assert.That(exception.Message, Does.Contain($"Stop the other instance, then restart with {MigrationSettings.EnabledKey} still on"), + "this is the only refusal in the subsystem where doing nothing makes it worse"); + Assert.That(exception.Message, Does.Contain(conflict.Message).And.Contain("Nothing has opened on SQLServer yet")); + Assert.That(exception.InnerException, Is.SameAs(conflict)); + }); + } + + [Test] + public async Task A_copy_that_finished_is_reported_as_it_returned_without_the_store_being_asked() + { + var copied = new[] { Finished(MigrationCategoryIds.KnownEndpoints) }; + var store = Holding(InProgress(MigrationCategoryIds.EndpointSettings)); + + var finished = await MigrationStartup.CopyOrExplainWhyItStopped( + Task.FromResult>(copied), + NoStall, + store, + Attempted(MigrationCategoryIds.KnownEndpoints), + NewSettings()); + + Assert.Multiple(() => + { + Assert.That(finished, Is.SameAs(copied), "what the engine returned is what the run did; the store is only for a copy that did not get to return anything"); + Assert.That(store.Reads, Is.Zero); + }); + } + + static Task> CancelledCopy => Task.FromCanceled>(new CancellationToken(canceled: true)); + + static string NoStall() => null!; + + static IReadOnlySet Attempted(params string[] categoryIds) => categoryIds.ToHashSet(StringComparer.Ordinal); + + static MigrationCheckpoint InProgress(string categoryId) => + new(categoryId, MigrationCategoryState.InProgress, "a", 3, 0, null, null, null, null, null, null); + + static MigrationCheckpoint Finished(string categoryId) => + new(categoryId, MigrationCategoryState.Complete, null, 3, 0, null, null, null, null, null, null); + + static ReadAllCheckpointStore Holding(params MigrationCheckpoint[] checkpoints) => new(_ => checkpoints); + + static ReadAllCheckpointStore Failing(Exception failure, Action? firstDo = null) => new(_ => + { + firstDo?.Invoke(); + throw failure; + }); + + static Settings NewSettings() => + new(transportType: "LearningTransport", persisterType: "SQLServer", errorRetentionPeriod: TimeSpan.FromDays(10)); + + // Only ReadAll is reachable from here, and the store shares the target's database, so a target that has gone + // away takes this read with it. + sealed class ReadAllCheckpointStore(Func> readAll) : IMigrationCheckpointStore + { + public int Reads { get; private set; } + + public Task> ReadAll(CancellationToken cancellationToken = default) + { + Reads++; + return Task.FromResult(readAll(cancellationToken)); + } + + public Task Read(string categoryId, CancellationToken cancellationToken = default) => throw new NotSupportedException(); + + public Task Upsert(MigrationCheckpoint checkpoint, CancellationToken cancellationToken = default) => throw new NotSupportedException(); + } +} diff --git a/src/ServiceControl.UnitTests/ScatterGather/IncompleteResultsTests.cs b/src/ServiceControl.UnitTests/ScatterGather/IncompleteResultsTests.cs index 744fa7bf19..627fee7dee 100644 --- a/src/ServiceControl.UnitTests/ScatterGather/IncompleteResultsTests.cs +++ b/src/ServiceControl.UnitTests/ScatterGather/IncompleteResultsTests.cs @@ -190,12 +190,10 @@ public async Task A_remote_timeout_does_not_cut_the_slower_remotes_short() factory.Register(settings.RemoteInstances[3], Delayed(delay, Healthy("msg-4"))); var api = new TestApi(settings, factory, Local("local-msg")); - var started = DateTime.UtcNow; var result = await api.Execute(Context(), "/api/messages"); - Assert.That(DateTime.UtcNow - started, Is.GreaterThanOrEqualTo(delay), "the composite must wait for the remotes that are still answering"); - Assert.That(result.Results.Select(m => m.MessageId), Is.EquivalentTo(["local-msg", "msg-2", "msg-3", "msg-4"])); + Assert.That(result.Results.Select(m => m.MessageId), Is.EquivalentTo(["local-msg", "msg-2", "msg-3", "msg-4"]), "the delay lives in the fake, so these three cannot arrive unless the composite waited for the remotes that are still answering"); Assert.That(result.IncompleteInstances, Is.EqualTo([new IncompleteInstance(settings.RemoteInstances[0].InstanceId, QueryFailure.TimedOut)])); Assert.That(result.QueryStats.TotalCount, Is.EqualTo(4)); } diff --git a/src/ServiceControl/HostApplicationBuilderExtensions.cs b/src/ServiceControl/HostApplicationBuilderExtensions.cs index cc43874a16..03d0d421c8 100644 --- a/src/ServiceControl/HostApplicationBuilderExtensions.cs +++ b/src/ServiceControl/HostApplicationBuilderExtensions.cs @@ -17,6 +17,7 @@ using global::ServiceControl.Operations.Metrics; using global::ServiceControl.Recoverability.Retrying.Metrics; using global::ServiceControl.Persistence; + using global::ServiceControl.Persistence.DataMigration; using global::ServiceControl.Transports; using Licensing; using Microsoft.AspNetCore.HttpLogging; @@ -103,6 +104,11 @@ public static void AddServiceControl(this IHostApplicationBuilder hostBuilder, S services.AddSingleton(provider => new Lazy(provider.GetRequiredService)); services.AddPersistence(settings); + + // The checkpoint store is optional: EF Core registers one, RavenDB does not. + services.TryAddSingleton(provider => + new CheckpointMigrationState(provider.GetService())); + services.AddMetrics(settings.PrintMetrics); hostBuilder.AddTelemetry(settings); services.AddServiceControlHealthChecks(); diff --git a/src/ServiceControl/Hosting/Commands/ErrorIngestionOnlyCommand.cs b/src/ServiceControl/Hosting/Commands/ErrorIngestionOnlyCommand.cs index 00d38bb0b3..f81f856435 100644 --- a/src/ServiceControl/Hosting/Commands/ErrorIngestionOnlyCommand.cs +++ b/src/ServiceControl/Hosting/Commands/ErrorIngestionOnlyCommand.cs @@ -5,6 +5,7 @@ namespace ServiceControl.Hosting.Commands using System.Threading; using System.Threading.Tasks; using Microsoft.AspNetCore.Builder; + using Microsoft.Extensions.DependencyInjection; using NServiceBus; using Particular.ServiceControl; using Particular.ServiceControl.Hosting; @@ -13,8 +14,10 @@ namespace ServiceControl.Hosting.Commands using ServiceControl.ExternalIntegrations; using ServiceControl.Hosting.Https; using ServiceControl.Infrastructure.Health; + using ServiceControl.Migration; using ServiceControl.Monitoring; using ServiceControl.Persistence; + using ServiceControl.Persistence.DataMigration; using ServiceControl.Recoverability; /// @@ -25,15 +28,13 @@ namespace ServiceControl.Hosting.Commands /// class ErrorIngestionOnlyCommand : AbstractCommand { - static readonly string[] SupportedStorageNames = ["SQLServer", "PostgreSQL"]; - public override async Task Execute(HostArguments args, Settings settings, CancellationToken cancellationToken = default) { EnsureStorageCanScaleOut(settings); var app = BuildHost(settings); - await app.RunAsync(settings.RootUrl); + await app.RunAsync(); } internal static WebApplication BuildHost(Settings settings, Action customize = null) @@ -47,12 +48,20 @@ internal static WebApplication BuildHost(Settings settings, Action + new FinishedCopyBeforeAnIngestionNodeOpens(provider.GetRequiredService(), settings)); + customize?.Invoke(hostBuilder); var app = hostBuilder.Build(); app.MapServiceControlHealthChecks(); + // Set here rather than passed to RunAsync, so a caller that starts the host itself gets the + // configured address instead of Kestrel's default port. + app.Urls.Clear(); + app.Urls.Add(settings.RootUrl); + return app; } @@ -60,7 +69,7 @@ static void EnsureStorageCanScaleOut(Settings settings) { var manifest = PersistenceManifestLibrary.Find(settings.PersistenceType); - if (manifest == null || !SupportedStorageNames.Contains(manifest.Name, StringComparer.OrdinalIgnoreCase)) + if (manifest == null || !PersistenceFactory.SqlPersistenceNames.Contains(manifest.Name, StringComparer.OrdinalIgnoreCase)) { throw new Exception( $"--error-ingestion-only requires SQL Server or PostgreSQL storage, but this instance is configured to use '{settings.PersistenceType}'. Scaling out error ingestion is not supported for this storage type."); diff --git a/src/ServiceControl/Hosting/Commands/RunCommand.cs b/src/ServiceControl/Hosting/Commands/RunCommand.cs index 33a6361511..68a15864ef 100644 --- a/src/ServiceControl/Hosting/Commands/RunCommand.cs +++ b/src/ServiceControl/Hosting/Commands/RunCommand.cs @@ -1,9 +1,12 @@ namespace ServiceControl.Hosting.Commands { + using System; using System.Threading; using System.Threading.Tasks; using Infrastructure.WebApi; using Microsoft.AspNetCore.Builder; + using Microsoft.Extensions.DependencyInjection; + using Microsoft.Extensions.Hosting; using NServiceBus; using Particular.ServiceControl; using Particular.ServiceControl.Hosting; @@ -11,11 +14,19 @@ using ServiceControl; using ServiceControl.Hosting.Auth; using ServiceControl.Hosting.Https; + using ServiceControl.Migration; using ServicePulse; class RunCommand : AbstractCommand { - public override async Task Execute(HostArguments args, Settings settings, CancellationToken cancellationToken = default) + public override Task Execute(HostArguments args, Settings settings, CancellationToken cancellationToken = default) => + Run(settings, customize: null, cancellationToken); + + /// + /// Builds and runs the full instance. When the migration is turned on, the required copy is added as the + /// first thing the host starts, so the instance opens only once the copy has finished. + /// + internal static async Task Run(Settings settings, Action customize, CancellationToken cancellationToken = default) { var endpointConfiguration = new EndpointConfiguration(settings.InstanceName); var assemblyScanner = endpointConfiguration.AssemblyScanner(); @@ -31,7 +42,16 @@ public override async Task Execute(HostArguments args, Settings settings, Cancel hostBuilder.AddServiceControl(settings, endpointConfiguration); hostBuilder.AddServiceControlApi(settings.CorsSettings); - var app = hostBuilder.Build(); + customize?.Invoke(hostBuilder); + + if (settings.MigrationEnabled) + { + hostBuilder.Services.AddHostedService(provider => new RequiredCopyBeforeTheHostOpens(provider, settings)); + } + + // A start that refuses would otherwise leave everything the host built undisposed. + await using var app = hostBuilder.Build(); + app.UseServiceControl(settings.ForwardedHeadersSettings, settings.HttpsSettings); if (settings.EnableIntegratedServicePulse) { @@ -39,7 +59,23 @@ public override async Task Execute(HostArguments args, Settings settings, Cancel } app.UseServiceControlAuthentication(settings.OpenIdConnectSettings.Enabled); - await app.RunAsync(settings.RootUrl); + // WebApplication's RunAsync(url) takes no cancellation token, so set the url here and call IHost's RunAsync below. + app.Urls.Clear(); + app.Urls.Add(settings.RootUrl); + + // Starting the host is what marks the target as opened, through the EF Core persistence's + // RecordHostOpenedOnTarget hosted service, so nothing here does it. + try + { + await app.RunAsync(cancellationToken); + } + // Stopping the service during the required copy cancels it inside host start, and that is a stop + // rather than a failure: every committed batch is durable and the next start resumes from the cursor. +#pragma warning disable PS0020 // The host cancels on its own lifetime token, not the caller's, so that is the one to filter on + catch (OperationCanceledException) when (app.Lifetime.ApplicationStopping.IsCancellationRequested) +#pragma warning restore PS0020 + { + } } } } diff --git a/src/ServiceControl/Infrastructure/Settings/Settings.cs b/src/ServiceControl/Infrastructure/Settings/Settings.cs index af52f4bf88..f998288e38 100644 --- a/src/ServiceControl/Infrastructure/Settings/Settings.cs +++ b/src/ServiceControl/Infrastructure/Settings/Settings.cs @@ -16,6 +16,7 @@ using ServiceControl.Infrastructure.Settings; using ServiceControl.Infrastructure.WebApi; using ServiceControl.Persistence; + using ServiceControl.Persistence.DataMigration; using ServiceControl.Transports; using ServicePulse; using JsonSerializer = System.Text.Json.JsonSerializer; @@ -185,6 +186,8 @@ public string InstanceId public string TransportType { get; set; } public string PersistenceType { get; private set; } + public bool MigrationEnabled => SettingsReader.Read(SettingsRootNamespace, MigrationSettings.EnabledKey, MigrationSettings.DefaultEnabled); + public bool MigrationAllowIncompleteExit => SettingsReader.Read(SettingsRootNamespace, MigrationSettings.AllowIncompleteExitKey, MigrationSettings.DefaultAllowIncompleteExit); public string ErrorLogQueue { get; set; } public string ErrorQueue { get; set; } diff --git a/src/ServiceControl/Migration/Checks/EveryRequiredCategoryCanBeCopiedCheck.cs b/src/ServiceControl/Migration/Checks/EveryRequiredCategoryCanBeCopiedCheck.cs new file mode 100644 index 0000000000..4b5388efd5 --- /dev/null +++ b/src/ServiceControl/Migration/Checks/EveryRequiredCategoryCanBeCopiedCheck.cs @@ -0,0 +1,39 @@ +namespace ServiceControl.Migration.Checks; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using ServiceControl.Persistence.DataMigration; + +/// +/// Registered by a test host that needs a copy to run on a build which cannot yet write every required category. +/// +class AllowIncompleteCategorySet; + +/// +/// Refuses to start the copy when this build cannot write every required category. A partial required copy +/// would open ServiceControl on the target and commit the instance to it, with the missing categories left in +/// the old database and no way back. +/// +class EveryRequiredCategoryCanBeCopiedCheck(IReadOnlyCollection copyableCategoryIds, AllowIncompleteCategorySet allowIncompleteCategorySet = null) : IMigrationStartupCheck +{ + public string Name => "this build can copy every required category"; + + public Task Run(CancellationToken cancellationToken = default) + { + var missing = MigrationCategoryRegistry.All + .Where(category => category.Kind == MigrationCategoryKind.Required && !copyableCategoryIds.Contains(category.Id)) + .Select(category => category.Id) + .ToArray(); + + if (allowIncompleteCategorySet is null && missing.Length > 0) + { + throw new Exception( + $"This build of ServiceControl cannot yet copy {missing.Length} of the required categories ({string.Join(", ", missing)}), so setting {MigrationSettings.EnabledKey} would copy part of the required set, open ServiceControl on the target and commit this instance to it with those categories never copied. Upgrade to a build that copies all of them."); + } + + return Task.CompletedTask; + } +} diff --git a/src/ServiceControl/Migration/Checks/MigrationPairIsSupportedCheck.cs b/src/ServiceControl/Migration/Checks/MigrationPairIsSupportedCheck.cs new file mode 100644 index 0000000000..e3ffed392a --- /dev/null +++ b/src/ServiceControl/Migration/Checks/MigrationPairIsSupportedCheck.cs @@ -0,0 +1,31 @@ +namespace ServiceControl.Migration.Checks; + +using System; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Persistence; +using ServiceControl.Persistence.DataMigration; + +/// +/// Refuses any target but SQL Server or PostgreSQL, since the source is always RavenDB. The seams underneath are +/// general enough to copy between any two persisters, and this is the check that says which pair is actually supported. +/// +class MigrationPairIsSupportedCheck(Settings settings) : IMigrationStartupCheck +{ + public string Name => "the source and target are the supported pair"; + + public Task Run(CancellationToken cancellationToken = default) + { + var targetName = PersistenceManifestLibrary.Find(settings.PersistenceType)?.Name; + + if (!PersistenceFactory.SqlPersistenceNames.Contains(targetName, StringComparer.OrdinalIgnoreCase)) + { + throw new Exception( + $"Migrating from '{PersistenceFactory.MigrationSourcePersistenceType}' to '{settings.PersistenceType}' is not supported. The only supported migration is from RavenDB to SQL Server or PostgreSQL, so set {Settings.SettingsRootNamespace}/PersistenceType to {string.Join(" or ", PersistenceFactory.SqlPersistenceNames)} before setting {MigrationSettings.EnabledKey}."); + } + + return Task.CompletedTask; + } +} diff --git a/src/ServiceControl/Migration/Checks/SelectedCategoriesAreCoherentCheck.cs b/src/ServiceControl/Migration/Checks/SelectedCategoriesAreCoherentCheck.cs new file mode 100644 index 0000000000..f85dd117e5 --- /dev/null +++ b/src/ServiceControl/Migration/Checks/SelectedCategoriesAreCoherentCheck.cs @@ -0,0 +1,26 @@ +namespace ServiceControl.Migration.Checks; + +using System.Threading; +using System.Threading.Tasks; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Persistence.DataMigration; + +/// +/// Reads the engine options out of the settings, which refuses a typo in the optional category list before +/// anything is copied rather than part way through the copy. +/// +class SelectedCategoriesAreCoherentCheck : IMigrationStartupCheck +{ + public string Name => "the selected categories are coherent"; + + /// + /// The options parsed from the settings, which is null until has returned. + /// + public MigrationEngineOptions Options { get; private set; } + + public Task Run(CancellationToken cancellationToken = default) + { + Options = MigrationEngineOptions.FromSettings(Settings.SettingsRootNamespace); + return Task.CompletedTask; + } +} diff --git a/src/ServiceControl/Migration/FinishedCopyBeforeAnIngestionNodeOpens.cs b/src/ServiceControl/Migration/FinishedCopyBeforeAnIngestionNodeOpens.cs new file mode 100644 index 0000000000..3d5491b38f --- /dev/null +++ b/src/ServiceControl/Migration/FinishedCopyBeforeAnIngestionNodeOpens.cs @@ -0,0 +1,52 @@ +namespace ServiceControl.Migration; + +using System; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Hosting; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Persistence.DataMigration; + +/// +/// Refuses to start an error ingestion only host while a copy into its database is unfinished. +/// A refusal throws, which fails the start and stops the host. +/// +sealed class FinishedCopyBeforeAnIngestionNodeOpens(IMigrationCheckpointStore checkpointStore, Settings settings) : IHostedLifecycleService +{ + // An ingestion node never runs the copy: one host does that, and a second copier would race it. It stays out + // until that copy finishes because ingesting writes to the target and stamps it as opened, which is what + // turns abandoning a part-copied database from a clean rollback into permanent loss. + public async Task StartingAsync(CancellationToken cancellationToken = default) + { + if (settings.MigrationAllowIncompleteExit) + { + return; + } + + var unfinished = (await checkpointStore.ReadAll(cancellationToken)) + .Where(checkpoint => !checkpoint.State.IsFinished()) + .ToArray(); + + if (unfinished.Length == 0) + { + return; + } + + var detail = string.Join("; ", unfinished.Select(checkpoint => $"{checkpoint.CategoryId} is {checkpoint.State}")); + + throw new Exception( + $"A copy into this database has not finished, so this error ingestion only host will not start and nothing has been lost. {detail}. " + + $"Let the instance running the copy finish it and start this host again, or set {Settings.SettingsRootNamespace}/{MigrationSettings.AllowIncompleteExitKey}=true to ingest into a part-copied database and accept that it can no longer be abandoned without loss."); + } + + public Task StartAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StartedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppingAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StopAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; +} diff --git a/src/ServiceControl/Migration/MigrationStartup.cs b/src/ServiceControl/Migration/MigrationStartup.cs new file mode 100644 index 0000000000..b593e16648 --- /dev/null +++ b/src/ServiceControl/Migration/MigrationStartup.cs @@ -0,0 +1,394 @@ +namespace ServiceControl.Migration; + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.DependencyInjection; +using Microsoft.Extensions.Logging; +using ServiceBus.Management.Infrastructure.Settings; +using ServiceControl.Migration.Checks; +using ServiceControl.Persistence; +using ServiceControl.Persistence.DataMigration; + +/// +/// The required copy, from the checks that decide whether it can run at all to the refusal it throws when a +/// category did not finish. Everything here happens with ServiceControl closed, which is the only window in +/// which the copy can be thrown away at no cost. +/// +static class MigrationStartup +{ + // A category needs both a reader and a writer, so a half-implemented one is never attempted. + internal static IReadOnlyCollection CopyableCategoryIds(IReadOnlyCollection sourceSupports, IReadOnlyCollection targetSupports) => + [.. sourceSupports.Intersect(targetSupports, StringComparer.Ordinal)]; + + /// + /// Runs the startup checks, copies every required category this build can copy, and then seeds the + /// the host reads. Throws when a check refuses or a category does not finish, + /// with a message telling the operator what to do and how to go back; the caller must let that stop the host. + /// Call it after the host is built and before it starts, because the copy has to finish before anything else + /// opens on the target. + /// + /// The built host's services, which is where the target, the checkpoint store and the migration state come from. + /// The instance settings, read for the source and target persistence types. + /// Cancelled when the host is shutting down, which ends the copy without a refusal message. + public static async Task RunRequiredCopy(IServiceProvider services, Settings settings, CancellationToken cancellationToken = default) + { + await MigrationStartupCheckRunner.Run( + [ + new MigrationPairIsSupportedCheck(settings) + ], cancellationToken); + + var loggerFactory = services.GetRequiredService(); + var logger = loggerFactory.CreateLogger(typeof(MigrationStartup)); + var target = services.GetRequiredService(); + var checkpointStore = services.GetRequiredService(); + var timeProvider = services.GetRequiredService(); + + await using var source = PersistenceFactory.CreateMigrationSource(settings); + + var copyable = CopyableCategoryIds(source.SupportedCategoryIds, target.SupportedCategoryIds); + + var options = await RunChecksAndOpen(services, source, copyable, cancellationToken); + + var engine = new MigrationEngine( + source, + target, + checkpointStore, + timeProvider, + options, + loggerFactory.CreateLogger()); + + var selected = engine.SelectCategories(MigrationCategoryKind.Required) + .Concat(engine.SelectCategories(MigrationCategoryKind.Optional)) + .Select(category => category.Id) + .ToArray(); + + var toCopy = engine.SelectCategories(MigrationCategoryKind.Required) + .Where(category => copyable.Contains(category.Id)) + .ToArray(); + + var deferred = engine.SelectCategories(MigrationCategoryKind.Required) + .Where(category => !copyable.Contains(category.Id)) + .Select(category => category.Id) + .ToArray(); + + logger.LogInformation( + "Migration mode: copying {CopyCount} required categories before ServiceControl opens ({DeferredCount} not yet implemented: {Deferred})", + toCopy.Length, deferred.Length, string.Join(", ", deferred)); + + var attempted = toCopy.Select(category => category.Id).ToHashSet(StringComparer.Ordinal); + + // Captured before the copy so the report can tell a category this run finished from one an earlier run did. + var runStartedAt = timeProvider.GetUtcNow().UtcDateTime; + + await using var progress = new ClosedWindowProgress(checkpointStore, timeProvider, logger, attempted, cancellationToken); + + var finished = await CopyOrExplainWhyItStopped( + engine.RunCategories(toCopy, progress.Token), + () => progress.StalledCategoryId, + checkpointStore, + attempted, + settings, + cancellationToken); + + ReportWhatTheCopyLeftBehind(finished, logger, runStartedAt); + + RefuseIfAnyCategoryDidNotComplete(toCopy, finished, progress.StalledCategoryId, settings); + + if (services.GetRequiredService() is CheckpointMigrationState state) + { + await state.Seed(selected, cancellationToken); + } + } + + /// + /// Runs every startup check in the order they have to run, and opens the target and the source as two of + /// them. The order is what the operator sees: a check that costs nothing comes before one that connects to a + /// database, and the source's own checks run last because they need it open. + /// + /// The categories both ends can handle, which is what the check on this build's coverage is given. + /// The options read from the settings, which the coherence check parsed on its way past. + /// A check refused. The message names the check and says what to do, and nothing has been copied. + public static async Task RunChecksAndOpen(IServiceProvider services, IMigrationSource source, IReadOnlyCollection copyableCategoryIds, CancellationToken cancellationToken = default) + { + var target = services.GetRequiredService(); + var readiness = services.GetRequiredService(); + var categories = new SelectedCategoriesAreCoherentCheck(); + + await MigrationStartupCheckRunner.Run( + [ + new EveryRequiredCategoryCanBeCopiedCheck(copyableCategoryIds, services.GetService()), + categories, + .. readiness.ContributedChecks(), + new Step("the migration target opens", target.Open), + new Step("the migration source opens", source.Open) + ], cancellationToken); + + await MigrationStartupCheckRunner.Run(source.ContributedChecks(), cancellationToken); + + return categories.Options; + } + + /// + /// Runs the copy and turns the two ways it can stop without finishing into a refusal the operator can act on: + /// a category that stalled, and a second instance writing checkpoints to the same database. + /// + /// The copy, already running. It has to have been started under the stall watchdog's token, because cancelling that token is how a stalled copy is stopped. + /// Reads which category stalled, or null when none has. It is read after the copy stops, because the watchdog sets it while the copy is still running. + /// Read after a stall, to find out how far the attempted categories got. + /// The category ids this run tried to copy. A checkpoint for any other category is left out of the result. + /// Read for the persistence types the refusals name. + /// The host's token. Cancelling it ends the copy as a plain cancellation with no refusal, because a shutdown is not a failure. + /// What the copy returned, or what the checkpoint store holds for the attempted categories when a stall stopped it. + /// A stall whose outstanding categories could not be read back, or a checkpoint saved by another instance. Both messages say what to do and how to go back. + /// The host is shutting down. + internal static async Task> CopyOrExplainWhyItStopped( + Task> copy, + Func stalledCategoryId, + IMigrationCheckpointStore checkpointStore, + IReadOnlySet attempted, + Settings settings, + CancellationToken cancellationToken = default) + { + try + { + return await copy; + } + // Without this, a stalled copy just ends as a cancellation and the operator never sees the refusal telling them what to do. + // The stall cancels the linked progress token, not the caller's, so filtering on the caller's would never match. +#pragma warning disable PS0020 + catch (OperationCanceledException) when (stalledCategoryId() is not null && !cancellationToken.IsCancellationRequested) + { + try + { + // The caller's token, because the stall has already cancelled the progress one and a read on that would fail. + return [.. (await checkpointStore.ReadAll(cancellationToken)).Where(checkpoint => attempted.Contains(checkpoint.CategoryId))]; + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + // Letting this out would replace the stall with a store error and send the operator after the wrong thing. + catch (Exception exception) + { + throw new Exception( + $"The required copy did not finish, so ServiceControl will not start and nothing has been lost. {StallExplanation(stalledCategoryId())}" + + $"Which categories are outstanding could not be read back, because the checkpoint store is unreachable as well: {exception.Message} " + + RollbackAdvice(settings), exception); + } + } +#pragma warning restore PS0020 + // The engine never turns a conflict into a halt, so without this the copy ends on the checkpoint store's + // own message and none of the advice every other refusal carries. + catch (MigrationCheckpointConflictException exception) + { + throw new Exception( + $"The required copy stopped because another writer saved a migration checkpoint for this instance, which is a second ServiceControl pointed at the same {settings.PersistenceType} database. ServiceControl will not start. {exception.Message} Stop the other instance, then restart with {MigrationSettings.EnabledKey} still on; the copy resumes from its last committed batch. " + + RollbackAdvice(settings), exception); + } + } + + /// + /// Throws unless every attempted category finished. A category that reported no checkpoint at all counts as + /// outstanding too, because nothing says how much of it was copied. + /// + /// The category the watchdog stopped, or null. The refusal blames the stall only when that category is one of the outstanding ones. + /// A category did not finish. The message names each one with its state and counts, says how to carry on, and says how to go back. + internal static void RefuseIfAnyCategoryDidNotComplete( + IReadOnlyList attempted, + IReadOnlyList finished, + string stalledCategoryId, + Settings settings) + { + var outstanding = finished + .Where(checkpoint => !checkpoint.State.IsFinished()) + .Select(checkpoint => (checkpoint.CategoryId, Detail: $"{checkpoint.CategoryId} is {checkpoint.State} after copying {checkpoint.CopiedCount} and skipping {checkpoint.SkippedCount}{(checkpoint.LastError is null ? "" : $": {checkpoint.LastError}")}")) + .ToList(); + + var reported = finished.Select(checkpoint => checkpoint.CategoryId).ToHashSet(StringComparer.Ordinal); + + outstanding.AddRange(attempted + .Where(category => !reported.Contains(category.Id)) + .Select(category => (CategoryId: category.Id, Detail: $"{category.Id} reported no checkpoint at all, so whether it copied anything is unknown"))); + + if (outstanding.Count == 0) + { + return; + } + + var detail = string.Join("; ", outstanding.Select(item => item.Detail)); + // If the stalled category finished anyway, the stall is not why these are outstanding, so don't blame it. + var stallExplainsIt = outstanding.Any(item => string.Equals(item.CategoryId, stalledCategoryId, StringComparison.Ordinal)); + + throw new Exception( + $"The required copy did not finish, so ServiceControl will not start and nothing has been lost. {detail}. " + + (stallExplainsIt + ? StallExplanation(stalledCategoryId) + : $"Fix the cause and restart with {MigrationSettings.EnabledKey} still on; the copy resumes from its last committed batch. ") + + RollbackAdvice(settings)); + } + + + /// + /// Logs one line per category saying what it copied and what it left behind. Skipped rows are warned about + /// one at a time while the copy runs, and nothing else states the total or says they are never coming. + /// + /// When this start began, which is what tells a category this run finished from one an earlier run did. + internal static void ReportWhatTheCopyLeftBehind(IReadOnlyList finished, ILogger logger, DateTime runStartedAt) + { + foreach (var checkpoint in finished) + { + // A category already finished when this run began is returned without being run, so its counts are + // an earlier run's. Reporting them in the same words as a fresh copy reads as a second copy against + // a target that is already serving traffic. + if (checkpoint.SettledAt is { } settledAt && settledAt < runStartedAt) + { + logger.LogInformation( + "{CategoryId}: already finished before this start, by a run that copied {Copied} and skipped {Skipped}. This start copied nothing.", + checkpoint.CategoryId, checkpoint.CopiedCount, checkpoint.SkippedCount); + continue; + } + + if (checkpoint.SkippedCount == 0) + { + logger.LogInformation("{CategoryId}: {Copied} copied, {AlreadyPresent} already present, nothing skipped", + checkpoint.CategoryId, checkpoint.CopiedCount, checkpoint.AlreadyPresentCount); + continue; + } + + var reasons = checkpoint.SkipReasons is { Count: > 0 } counts + ? string.Join(", ", counts.OrderByDescending(reason => reason.Value).Select(reason => $"{reason.Key} {reason.Value}")) + : "no reason recorded"; + + logger.LogWarning( + "{CategoryId}: {Copied} copied, {AlreadyPresent} already present, {Skipped} skipped ({Reasons}). The skipped rows were not copied and no later run will fetch them: they stay only in the source database.", + checkpoint.CategoryId, checkpoint.CopiedCount, checkpoint.AlreadyPresentCount, checkpoint.SkippedCount, reasons); + } + } + + static string StallExplanation(string stalledCategoryId) => + $"The copy was stopped because '{stalledCategoryId}' committed nothing for {ClosedWindowProgress.StallLimit.TotalMinutes:0.#} minutes, which is not a configurable limit: check that the source and the target are both responding rather than looking for a setting to change. "; + + static string RollbackAdvice(Settings settings) => + $"Nothing has opened on {settings.PersistenceType} yet, so setting {MigrationSettings.EnabledKey}=false and pointing PersistenceType back at {PersistenceFactory.MigrationSourcePersistenceType} discards the partial copy and returns the instance to {PersistenceFactory.MigrationSourcePersistenceType} with no loss."; + + /// + /// Makes opening the target or the source look like a startup check, so a failure to connect is reported in + /// the same words as a check that refused, and in its place in the order. + /// + sealed class Step(string name, Func run) : IMigrationStartupCheck + { + public string Name => name; + + public Task Run(CancellationToken cancellationToken = default) => run(cancellationToken); + } + + /// + /// Logs how far each category has got while the copy runs, and stops the copy when one of them commits + /// nothing for . Without it a copy that is waiting on a database nobody is watching + /// holds the instance closed for as long as the operator leaves it. + /// + internal sealed class ClosedWindowProgress : IAsyncDisposable + { + // Not configurable, and not a total timeout: a deadline would kill a copy that is working. + internal static readonly TimeSpan PollInterval = TimeSpan.FromSeconds(30); + internal static readonly TimeSpan StallLimit = TimeSpan.FromMinutes(30); + + readonly CancellationTokenSource cancellation; + readonly HashSet attempted; + readonly DateTime watchStartedAt; + readonly Task polling; + + public ClosedWindowProgress(IMigrationCheckpointStore checkpointStore, TimeProvider timeProvider, ILogger logger, IReadOnlyCollection attemptedCategoryIds, CancellationToken cancellationToken = default) + { + cancellation = CancellationTokenSource.CreateLinkedTokenSource(cancellationToken); + attempted = attemptedCategoryIds.ToHashSet(StringComparer.Ordinal); + watchStartedAt = timeProvider.GetUtcNow().UtcDateTime; + polling = Poll(checkpointStore, timeProvider, logger, cancellation.Token); + } + + // The copy must run under this token, because cancelling it is how the watchdog stops a stalled copy. + public CancellationToken Token => cancellation.Token; + + /// + /// The category that stalled, or null when none has. The watchdog sets it before it cancels the token, + /// so read it after the copy has stopped. + /// + public string StalledCategoryId { get; private set; } + + async Task Poll(IMigrationCheckpointStore checkpointStore, TimeProvider timeProvider, ILogger logger, CancellationToken cancellationToken) + { + using var timer = new PeriodicTimer(PollInterval, timeProvider); + + try + { + while (await timer.WaitForNextTickAsync(cancellationToken)) + { + try + { + // Only this run's categories, because a row left in progress by an earlier run is not a stall in this one. + var running = (await checkpointStore.ReadAll(cancellationToken)) + .Where(checkpoint => checkpoint.State == MigrationCategoryState.InProgress && attempted.Contains(checkpoint.CategoryId)) + .ToArray(); + + foreach (var checkpoint in running) + { + logger.LogInformation( + "{CategoryId}: {Copied} of {Total} copied, {Skipped} skipped, cursor {Cursor}", + checkpoint.CategoryId, + checkpoint.CopiedCount, + checkpoint.SourceTotal is { } total ? total.ToString() : "an unknown number of", + checkpoint.SkippedCount, + checkpoint.Cursor ?? "the start"); + } + + // A resumed row carries the previous run's stamp, so the window starts at whichever is + // later: that stamp, or the moment this watch began. + var stalled = running.FirstOrDefault(checkpoint => + timeProvider.GetUtcNow().UtcDateTime + - (checkpoint.LastProgressAt is { } lastProgress && lastProgress > watchStartedAt ? lastProgress : watchStartedAt) > StallLimit); + + if (stalled is not null) + { + StalledCategoryId = stalled.CategoryId; + logger.LogError( + "{CategoryId} has committed nothing for {StallLimit}, so the copy is being stopped. Every committed batch is durable and the next start resumes from the cursor.", + stalled.CategoryId, StallLimit); + await cancellation.CancelAsync(); + } + } + // The copy finished or the host is stopping: the outer catch ends the poll. + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + // Log, don't throw: an error here would come out of DisposeAsync and hide what the copy itself failed with. + catch (Exception exception) + { + logger.LogError(exception, "The stall watchdog's poll failed, so a stalled copy will not be noticed until a later poll succeeds. The copy itself is unaffected and is still running."); + } + } + } + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + // Either the copy finished and disposal cancelled the poll, or the host is shutting down. + } + } + + public async ValueTask DisposeAsync() + { + await cancellation.CancelAsync(); + + try + { + await polling; + } + finally + { + cancellation.Dispose(); + } + } + } +} diff --git a/src/ServiceControl/Migration/MigrationStartupCheckRunner.cs b/src/ServiceControl/Migration/MigrationStartupCheckRunner.cs new file mode 100644 index 0000000000..daa17e956d --- /dev/null +++ b/src/ServiceControl/Migration/MigrationStartupCheckRunner.cs @@ -0,0 +1,39 @@ +namespace ServiceControl.Migration; + +using System; +using System.Collections.Generic; +using System.Threading; +using System.Threading.Tasks; +using ServiceControl.Persistence.DataMigration; + +/// +/// Runs startup checks in order and stops at the first one that refuses. +/// +static class MigrationStartupCheckRunner +{ + /// + /// Runs each check in turn, and wraps whatever a failing one throws in a message naming the check and + /// saying that nothing has been copied. + /// + /// A check failed. The check's own message is kept, and its exception is the inner one. + public static async Task Run(IReadOnlyList checks, CancellationToken cancellationToken = default) + { + foreach (var check in checks) + { + try + { + await check.Run(cancellationToken); + } + // A shutdown is not a check failing, and saying it was would send the customer after the wrong thing. + catch (OperationCanceledException) when (cancellationToken.IsCancellationRequested) + { + throw; + } + catch (Exception exception) + { + throw new Exception( + $"Migration startup check '{check.Name}' failed, so ServiceControl will not start and nothing has been copied. {exception.Message}", exception); + } + } + } +} diff --git a/src/ServiceControl/Migration/RequiredCopyBeforeTheHostOpens.cs b/src/ServiceControl/Migration/RequiredCopyBeforeTheHostOpens.cs new file mode 100644 index 0000000000..9ca9388246 --- /dev/null +++ b/src/ServiceControl/Migration/RequiredCopyBeforeTheHostOpens.cs @@ -0,0 +1,30 @@ +namespace ServiceControl.Migration; + +using System; +using System.Threading; +using System.Threading.Tasks; +using Microsoft.Extensions.Hosting; +using ServiceBus.Management.Infrastructure.Settings; + +/// +/// Runs the required copy as the host starts, before any hosted service of its own does. +/// A refusal throws, which fails the start and stops the host, as the copy requires. +/// +sealed class RequiredCopyBeforeTheHostOpens(IServiceProvider services, Settings settings) : IHostedLifecycleService +{ + // Inside the host rather than before it, because a Windows service reports itself started only once the host + // starts, and the Service Control Manager kills a process that has said nothing for 30 seconds. Every + // StartingAsync runs before any hosted service, so nothing has bound a port or begun ingesting. + public Task StartingAsync(CancellationToken cancellationToken = default) => + MigrationStartup.RunRequiredCopy(services, settings, cancellationToken); + + public Task StartAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StartedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppingAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StopAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; + + public Task StoppedAsync(CancellationToken cancellationToken = default) => Task.CompletedTask; +} diff --git a/src/ServiceControl/Persistence/PersistenceFactory.cs b/src/ServiceControl/Persistence/PersistenceFactory.cs index 59981ba691..9ef994351b 100644 --- a/src/ServiceControl/Persistence/PersistenceFactory.cs +++ b/src/ServiceControl/Persistence/PersistenceFactory.cs @@ -10,6 +10,12 @@ namespace ServiceControl.Persistence static class PersistenceFactory { + /// + /// The manifest names of the two persisters built on EF Core. Anything that is true of both of them and + /// of neither RavenDB nor a future persister is decided by this list. + /// + public static readonly string[] SqlPersistenceNames = ["SQLServer", "PostgreSQL"]; + public static IPersistence Create(Settings settings, bool maintenanceMode = false) { var persistenceConfiguration = CreatePersistenceConfiguration(settings.PersistenceType, settings); @@ -23,6 +29,7 @@ public static IPersistence Create(Settings settings, bool maintenanceMode = fals settings.PersisterSpecificSettings ??= persistenceConfiguration.CreateSettings(Settings.SettingsRootNamespace); settings.PersisterSpecificSettings.MaintenanceMode = maintenanceMode; settings.PersisterSpecificSettings.RunRetentionSweep = !settings.ErrorIngestionOnly; + settings.PersisterSpecificSettings.RetryHistoryDepth = settings.RetryHistoryDepth; var persistence = persistenceConfiguration.Create(settings.PersisterSpecificSettings); return persistence; diff --git a/src/ServiceControl/Program.cs b/src/ServiceControl/Program.cs index b74bdc8488..96fb2359df 100644 --- a/src/ServiceControl/Program.cs +++ b/src/ServiceControl/Program.cs @@ -52,7 +52,13 @@ LoggingConfigurator.ConfigureNLog("bootstrap.txt", "./", NLog.LogLevel.Fatal); NLog.LogManager.GetCurrentClassLogger().Fatal(ex, "Unrecoverable error"); } - throw; + + // The message goes to the console on its own, because the log above already holds the whole exception. + // Rethrowing instead would print the stack trace a second time and fire the unhandled-exception handler + // for a third, which buries a configuration mistake in what reads like a crash. + await Console.Error.WriteLineAsync(ex.Message); + + return 1; } finally {