From 4d46c9b70d2aefc2f9ba41f406a0f4fa0eb04e84 Mon Sep 17 00:00:00 2001 From: Rick Bonkestoter Date: Wed, 30 Sep 2026 09:34:28 +0200 Subject: [PATCH 1/2] feat(mssql): fail over an availability group from its context menu MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Fail Over… on an availability group builds the plan from cluster_type_desc: one FAILOVER (or FORCE_FAILOVER_ALLOW_DATA_LOSS) on the target for WSFC, the documented multi-instance sequence for read-scale (NONE), and a refusal with the pcs/crm commands for Pacemaker (EXTERNAL). The plan is data, shown in full before anything runs, re-read and compared at confirm time, and every picked connection must prove it is the replica it was picked for. The old primary's SET (ROLE = SECONDARY) is retried for up to 30 s: issued the moment the promotion returns it fails while that replica is still resolving, seen against a real read-scale group. IToolUiContext gains OpenConnection so the view can read the primary's state when opened on a secondary (additive, folded into tool API 8). --- changelog.d/SE-247.added.md | 18 + changelog.d/SE-284.added.md | 3 +- .../ViewModels/ToolDialogViewModel.cs | 3 + src/DataTray.Sdk/Tools/ToolHostApi.cs | 4 + src/DataTray.Sdk/Ui/IToolUiContext.cs | 5 + .../AvailabilityGroupFailover.cs | 297 ++++++++++++++++ .../AvailabilityGroupQueries.cs | 114 +++++++ .../DataTray.Tools.MsSqlAdmin.csproj | 7 + .../FailoverAvailabilityGroupTool.cs | 279 +++++++++++++++ .../FailoverAvailabilityGroupView.cs | 272 +++++++++++++++ .../Lang/strings.json | 5 +- .../Lang/strings.nl.json | 5 +- src/DataTray.Tools.MsSqlAdmin/plugin.json | 2 +- .../Mcp/McpSqlClassifierTests.cs | 10 + .../AvailabilityGroupFailoverTests.cs | 317 ++++++++++++++++++ 15 files changed, 1337 insertions(+), 4 deletions(-) create mode 100644 changelog.d/SE-247.added.md create mode 100644 src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupFailover.cs create mode 100644 src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupQueries.cs create mode 100644 src/DataTray.Tools.MsSqlAdmin/FailoverAvailabilityGroupTool.cs create mode 100644 src/DataTray.Tools.MsSqlAdmin/FailoverAvailabilityGroupView.cs create mode 100644 tests/DataTray.Tools.MsSqlAdmin.Tests/AvailabilityGroupFailoverTests.cs diff --git a/changelog.d/SE-247.added.md b/changelog.d/SE-247.added.md new file mode 100644 index 00000000..7601e5aa --- /dev/null +++ b/changelog.d/SE-247.added.md @@ -0,0 +1,18 @@ +- **An availability group can be failed over from its context menu: *Fail Over…*.** Pick the replica to + fail over to, point DataTray at the saved connections for the instances involved, and it shows the exact + plan — every statement and the instance it runs on — before anything runs. Which plan you get follows the + group's cluster type: + - **WSFC**: one statement on the target. A plain `FAILOVER` when the target is synchronous and every + database is failover-ready; otherwise `FORCE_FAILOVER_ALLOW_DATA_LOSS`, with a red warning, how far + behind each database is, and what to do about the old primary afterwards. + - **Read-scale (`CLUSTER_TYPE = NONE`)**: the documented sequence across both instances — make both + replicas synchronous, wait until the target is `SYNCHRONIZED`, take the group offline on the primary, + promote the target, demote and resume the old primary, re-create the listener. It stops before taking + the group offline if the target never catches up. + - **Pacemaker (`CLUSTER_TYPE = EXTERNAL`)**: DataTray refuses, because the cluster manager owns the + primary role, and shows the `pcs`/`crm` commands with the target node filled in to run yourself. + + A forced or read-scale failover asks you to type the group name. Each picked connection is checked to + really be the replica it was picked for, and when you confirm, the state is read again: if the plan + changed in the meantime (a target that fell behind turns a planned failover into a forced one), nothing + runs. Failing over is not reachable from the MCP server. diff --git a/changelog.d/SE-284.added.md b/changelog.d/SE-284.added.md index 299e71b6..317cac9a 100644 --- a/changelog.d/SE-284.added.md +++ b/changelog.d/SE-284.added.md @@ -6,4 +6,5 @@ "behind by" column — the primary's last commit time minus the replica's, in seconds, next to the DMVs' own log-send/redo queue sizes in kilobytes. It is the only node dialog in DataTray that polls (every 10 seconds): queue depth moves while you are looking at it, unlike everything else this dialog family shows. - Creating or reconfiguring a group, and failing one over, stay out of scope for now (SE-247). + Failing a group over is its own action on the group (*Fail Over…*); creating or reconfiguring a group + is not in yet. diff --git a/src/DataTray.App/ViewModels/ToolDialogViewModel.cs b/src/DataTray.App/ViewModels/ToolDialogViewModel.cs index 9fd8ec94..4ac233f1 100644 --- a/src/DataTray.App/ViewModels/ToolDialogViewModel.cs +++ b/src/DataTray.App/ViewModels/ToolDialogViewModel.cs @@ -503,6 +503,9 @@ public Task QueryAsync(string sql, CancellationToken ct) => // DatabasePicker fields do. IReadOnlyList IToolUiContext.ListConnections() => ((IToolHost)this).ListConnections(); + ToolConnection? IToolUiContext.OpenConnection(string connectionId, string? database) => + ((IToolHost)this).OpenConnection(connectionId, database); + Task> IToolUiContext.ListDatabasesAsync(string connectionId, CancellationToken ct) => ((IToolHost)this).ListDatabasesAsync(connectionId, ct); diff --git a/src/DataTray.Sdk/Tools/ToolHostApi.cs b/src/DataTray.Sdk/Tools/ToolHostApi.cs index 15f1811a..f617ddf2 100644 --- a/src/DataTray.Sdk/Tools/ToolHostApi.cs +++ b/src/DataTray.Sdk/Tools/ToolHostApi.cs @@ -118,6 +118,10 @@ public static class ToolHostApi // default interface members — an older host simply returns empty and a view that reads // it says what is missing — so this is a fold-in by the SE-253 test ("does it add // types?"), not a v9. + // also in v8 (2026-09-30): IToolUiContext.OpenConnection(id, database), mirroring IToolHost's (SE-247). + // The availability group failover view reads the group's state from its primary before + // the run, and the primary is often not the connection the tool was launched on. A + // default interface member returning null — no new types — so a fold-in by the same test. public const int Version = 8; /// Oldest plugin ABI this host still loads. Every bump has been additive (v2 tool defaults, v3 diff --git a/src/DataTray.Sdk/Ui/IToolUiContext.cs b/src/DataTray.Sdk/Ui/IToolUiContext.cs index b76640a2..d54cf494 100644 --- a/src/DataTray.Sdk/Ui/IToolUiContext.cs +++ b/src/DataTray.Sdk/Ui/IToolUiContext.cs @@ -64,6 +64,11 @@ public interface IToolUiContext Task> ListDatabasesAsync(string connectionId, CancellationToken ct) => Task.FromResult>([]); + /// Open one of those connections, the same way does for the + /// running tool, so a view can show live data from a second instance before the run — the availability + /// group failover reads the group's state from its primary. Null default for an older host. + ToolConnection? OpenConnection(string connectionId, string? database = null) => null; + // ── Lifecycle-owning views (IToolDialogLifecycle) ───────────────────────────────────────────────── /// The plugin's own localizer, so a custom view can translate its labels the same way the diff --git a/src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupFailover.cs b/src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupFailover.cs new file mode 100644 index 00000000..b6927b83 --- /dev/null +++ b/src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupFailover.cs @@ -0,0 +1,297 @@ +using System.Text; +using DataTray.Providers.MsSql; + +namespace DataTray.Tools.MsSqlAdmin; + +/// What a failover to one replica turns out to be, decided by the group's cluster type and the +/// target's state — never by the operating system either side runs on. +internal enum FailoverKind +{ + /// WSFC, target synchronous-commit and failover-ready: one FAILOVER, no data loss. + Planned, + + /// WSFC, target not failover-ready: FORCE_FAILOVER_ALLOW_DATA_LOSS. + Forced, + + /// CLUSTER_TYPE = NONE: an ordered sequence across two instances (Msg 47122 refuses a + /// plain FAILOVER there). + ReadScale, + + /// Nothing DataTray will run — says why. + Refused +} + +internal sealed record AgReplica(string Name, bool IsSynchronousCommit, bool? IsConnected); + +/// One database on one replica. and are null when +/// the instance the state was read from cannot see that replica's rows (a secondary only sees its own). +internal sealed record AgDatabase(string Name, string Replica, string? SyncState, bool IsFailoverReady, DateTime? LastCommit); + +/// A listener, and the static addresses it was created with. No addresses means DHCP. +internal sealed record AgListener(string DnsName, int Port, IReadOnlyList<(string Address, string Mask)> Addresses); + +/// Everything the planner needs, read from the group's DMVs (see ). +/// is null when the instance it was read from does not know — a WSFC primary that is +/// down, say. +internal sealed record AgTopology( + string Group, + string? ClusterType, + string? Primary, + int RequiredSynchronizedSecondaries, + IReadOnlyList Replicas, + IReadOnlyList Databases, + AgListener? Listener); + +/// One statement in a plan, and the replica whose instance runs it. A wait step +/// () is a query polled until it returns 0, and aborts the plan if it never does. A +/// step is re-run for a short while when it fails, for a statement the instance only +/// accepts once it has caught up with the step before it. +internal sealed record AgStep(string Replica, string Sql, string Purpose, bool IsWait = false, bool Retries = false); + +internal sealed record FailoverPlan( + FailoverKind Kind, + IReadOnlyList Steps, + IReadOnlyList Warnings, + string? RefusalReason = null, + string? ExternalCommands = null) +{ + /// Whether the plan can lose committed transactions. Only a forced WSFC failover can: the + /// read-scale sequence uses the same statement, but only after its wait step proved the target is + /// SYNCHRONIZED with the primary already offline. + public bool CanLoseData => Kind == FailoverKind.Forced; + + /// Whether running it needs the user to type the group name. Everything but a planned failover: + /// a read-scale failover takes the group offline for its duration, a forced one may lose data. + public bool NeedsTypedConfirmation => Kind is FailoverKind.Forced or FailoverKind.ReadScale; + + /// The plan as the script the user reviews — also what ExecuteAsync compares against the + /// plan it rebuilds from fresh state, so a plan that changed since review is refused, not run. + public string Script() + { + var sb = new StringBuilder(); + for (var i = 0; i < Steps.Count; i++) + { + var step = Steps[i]; + sb.Append("-- ").Append(i + 1).Append(". on ").Append(step.Replica).Append(": ").AppendLine(step.Purpose); + sb.AppendLine(step.Sql.TrimEnd()); + sb.AppendLine(); + } + + return sb.ToString().TrimEnd(); + } +} + +/// +/// Builds the failover plan for moving an availability group's primary role to target (SE-247). +/// Pure — the plan is data, so every branch is tested without a server. Which branch is taken follows +/// cluster_type_desc only: +/// +/// WSFC — one statement on the target: FAILOVER when it is synchronous-commit and every +/// database is failover-ready, otherwise FORCE_FAILOVER_ALLOW_DATA_LOSS. +/// NONE (read-scale) — the documented sequence, every step visible. +/// EXTERNAL (Pacemaker) — refused: the clustermanager owns the role and DataTray does not run +/// shell commands on cluster nodes (Rick's decision, 2026-08-13). The exact command is handed over instead. +/// +/// +internal static class AvailabilityGroupFailover +{ + public static FailoverPlan Plan(AgTopology t, string target) + { + var replica = t.Replicas.FirstOrDefault(r => SameName(r.Name, target)); + if (replica is null) + { + return Refuse($"{target} is not a replica of {t.Group}."); + } + + if (t.Primary is not null && SameName(t.Primary, target)) + { + return Refuse($"{target} is already the primary replica of {t.Group}."); + } + + // Case-insensitive on purpose: sys.availability_groups returns this column lowercase ("none") where + // the documentation shows uppercase — see Codebase gotchas and AvailabilityGroupStatus. + return t.ClusterType?.ToUpperInvariant() switch + { + "WSFC" => Wsfc(t, replica), + "NONE" => ReadScale(t, replica), + "EXTERNAL" => External(t, replica), + _ => Refuse($"Unknown cluster type '{t.ClusterType}'. DataTray only fails over WSFC and read-scale (NONE) groups.") + }; + } + + private static FailoverPlan Wsfc(AgTopology t, AgReplica target) + { + var databases = t.Databases.Where(d => SameName(d.Replica, target.Name)).ToList(); + // A planned failover needs both ends synchronous-commit; is_failover_ready covers the target's state. + var primary = t.Primary is null ? null : t.Replicas.FirstOrDefault(r => SameName(r.Name, t.Primary)); + var ready = target.IsSynchronousCommit && primary?.IsSynchronousCommit != false + && databases.Count > 0 && databases.All(d => d.IsFailoverReady); + if (ready) + { + return new FailoverPlan(FailoverKind.Planned, + [new AgStep(target.Name, $"ALTER AVAILABILITY GROUP {Id(t.Group)} FAILOVER;", + "planned manual failover — the target is synchronized, nothing is lost")], + [$"Sessions on {t.Primary ?? "the current primary"} are disconnected. Clients that connect by instance name rather than through the listener do not follow the new primary."]); + } + + var warnings = new List + { + target.IsSynchronousCommit + ? $"{target.Name} is not failover-ready: at least one database is not SYNCHRONIZED. Transactions the primary committed but {target.Name} has not received are lost." + : $"{target.Name} uses asynchronous commit, so it can trail the primary. Transactions the primary committed but {target.Name} has not received are lost." + }; + warnings.AddRange(LossPerDatabase(t, target.Name)); + warnings.Add("The Windows cluster must have quorum, or the failover is refused."); + warnings.Add("Afterwards every secondary database is suspended, the old primary's included, and has to be resumed on its own replica. Resuming rolls back the transactions the new primary never got — if you might need them, create a database snapshot of each suspended database before you resume it."); + + return new FailoverPlan(FailoverKind.Forced, + [new AgStep(target.Name, $"ALTER AVAILABILITY GROUP {Id(t.Group)} FORCE_FAILOVER_ALLOW_DATA_LOSS;", + "forced failover — ALLOWS DATA LOSS")], + warnings); + } + + private static FailoverPlan ReadScale(AgTopology t, AgReplica target) + { + if (t.Primary is null) + { + return Refuse("The primary replica could not be determined. A read-scale failover takes the primary offline first, so it needs a working connection to it."); + } + + if (t.Replicas.Count != 2) + { + // ponytail: two replicas only — the documented sequence, and what the lab verifies. With more, + // every other secondary needs its own role change and resume; add when a real 3-replica read-scale + // group asks for it. + return Refuse($"{t.Group} has {t.Replicas.Count} replicas. DataTray only fails over a read-scale group of exactly two."); + } + + if (target.IsConnected == false) + { + return Refuse($"{target.Name} is not connected to the primary, so it can never become SYNCHRONIZED. Fix the connection first."); + } + + var group = Id(t.Group); + var primary = t.Primary; + var steps = new List(); + + foreach (var r in t.Replicas.Where(r => !r.IsSynchronousCommit)) + { + steps.Add(new AgStep(primary, + $"ALTER AVAILABILITY GROUP {group} MODIFY REPLICA ON {Lit(r.Name)} WITH (AVAILABILITY_MODE = SYNCHRONOUS_COMMIT);", + $"make {r.Name} synchronous-commit")); + } + + steps.Add(new AgStep(primary, WaitForSynchronizedSql(t.Group, target.Name), + $"wait until every database on {target.Name} is SYNCHRONIZED (0 = ready)", IsWait: true)); + steps.Add(new AgStep(primary, + $"ALTER AVAILABILITY GROUP {group} SET (REQUIRED_SYNCHRONIZED_SECONDARIES_TO_COMMIT = 1);", + "refuse commits the secondary has not hardened")); + steps.Add(new AgStep(primary, $"ALTER AVAILABILITY GROUP {group} OFFLINE;", + "take the group offline on the primary — writes stop here")); + steps.Add(new AgStep(target.Name, $"ALTER AVAILABILITY GROUP {group} FORCE_FAILOVER_ALLOW_DATA_LOSS;", + "promote the target (the only failover CLUSTER_TYPE = NONE accepts; nothing is lost after the wait above)")); + // Retried: issued the moment the promotion returns, the old primary is still RESOLVING and fails it + // with "the availability group resource did not come online" — seen against the lab, and the same + // statement succeeds a few seconds later. + steps.Add(new AgStep(primary, $"ALTER AVAILABILITY GROUP {group} SET (ROLE = SECONDARY);", + "demote the old primary (retried while it is still resolving)", Retries: true)); + // On the old primary, now a secondary. The read-scale page says "on the primary", but "Resume an + // availability database" says a locally suspended secondary database is resumed on its own replica — + // and the lab agrees: the old primary's databases are the suspended ones, and resuming them there works. + foreach (var db in t.Databases.Select(d => d.Name).Distinct(StringComparer.OrdinalIgnoreCase).Order(StringComparer.OrdinalIgnoreCase)) + { + steps.Add(new AgStep(primary, $"ALTER DATABASE {Id(db)} SET HADR RESUME;", + $"resume data movement for {db}")); + } + + if (t.Listener is { } listener) + { + steps.Add(new AgStep(target.Name, $"ALTER AVAILABILITY GROUP {group} REMOVE LISTENER {Lit(listener.DnsName)};", + "drop the listener, which no cluster moves for a read-scale group")); + steps.Add(new AgStep(target.Name, AddListenerSql(t.Group, listener), + "re-create it on the new primary")); + } + + var warnings = new List + { + $"The group is offline from step \"OFFLINE\" until {target.Name} is promoted: no writes are accepted anywhere in between.", + "The plan stops before OFFLINE if the target does not reach SYNCHRONIZED within the wait, and changes nothing after that point.", + $"Both replicas stay synchronous-commit and REQUIRED_SYNCHRONIZED_SECONDARIES_TO_COMMIT stays 1 afterwards (it was {t.RequiredSynchronizedSecondaries}). With it at 1, the new primary stops accepting commits whenever the secondary is disconnected." + }; + + return new FailoverPlan(FailoverKind.ReadScale, steps, warnings); + } + + private static FailoverPlan External(AgTopology t, AgReplica target) => new( + FailoverKind.Refused, [], [], + $"{t.Group} is managed by Pacemaker (CLUSTER_TYPE = EXTERNAL). The cluster manager owns which replica is primary, and a failover through T-SQL would fight it — DataTray does not run commands on cluster nodes. Run one of these on any cluster node:", + PacemakerCommands(target.Name)); + + /// The Pacemaker commands from "Always On availability group failover on Linux" (Microsoft Learn, + /// failover-high-availability), with the target filled in. Neither the resource name nor the Pacemaker node + /// name is visible from SQL Server, so the resource stays the documentation's ag_cluster and the node + /// is the replica's name — both marked as the things to check before running anything. + internal static string PacemakerCommands(string node) => + $""" + # Check first: the AG resource name (the docs use ag_cluster) and the node name ({node} is the replica's + # name, not necessarily Pacemaker's) — see: sudo pcs status / crm status + # The target must be a synchronous-commit replica. + + # RHEL 8+ / Ubuntu, promotable clone (ag_cluster-clone): + sudo pcs resource move ag_cluster-clone --master {node} && sleep 30 && sudo pcs resource clear ag_cluster-clone + + # RHEL 7 / older Ubuntu (ag_cluster-master): + sudo pcs resource move ag_cluster-master {node} --master --lifetime=30S + sudo pcs resource clear ag_cluster-master + + # SLES (not supported from SQL Server 2025 on): + sudo crm resource migrate ag_cluster {node} --lifetime=30S + sudo crm configure delete cli-prefer-ms-ag_cluster + """; + + /// 0 when every database in the group is SYNCHRONIZED on . Counts the + /// group's databases rather than the non-synchronized rows, so a database with no row for the target at + /// all (never joined there) counts as not ready instead of vanishing from the check. + internal static string WaitForSynchronizedSql(string group, string replica) => + $""" + SELECT COUNT(*) FROM sys.availability_databases_cluster adc + JOIN sys.availability_groups ag ON ag.group_id = adc.group_id + WHERE ag.name = {Lit(group)} AND NOT EXISTS ( + SELECT 1 FROM sys.dm_hadr_database_replica_states drs + JOIN sys.availability_replicas ar ON ar.replica_id = drs.replica_id + WHERE ar.replica_server_name = {Lit(replica)} AND drs.group_database_id = adc.group_database_id + AND drs.synchronization_state_desc = 'SYNCHRONIZED'); + """; + + private static string AddListenerSql(string group, AgListener listener) + { + var ip = listener.Addresses.Count == 0 + ? "DHCP" + // An IPv6 address has no mask and is written as a one-element tuple. + : $"IP ({string.Join(", ", listener.Addresses.Select(a => a.Mask.Length == 0 ? $"({Lit(a.Address)})" : $"({Lit(a.Address)}, {Lit(a.Mask)})"))})"; + return $"ALTER AVAILABILITY GROUP {Id(group)} ADD LISTENER {Lit(listener.DnsName)} (WITH {ip}, PORT = {listener.Port});"; + } + + /// Per database, how far the target trails the primary — the "how bad" number the user needs + /// before accepting data loss. Same derivation the SE-284 dashboard shows (). + private static IEnumerable LossPerDatabase(AgTopology t, string target) + { + foreach (var db in t.Databases.Where(d => SameName(d.Replica, target))) + { + var primary = t.Primary is null ? null : t.Databases.FirstOrDefault(d => SameName(d.Replica, t.Primary) && SameName(d.Name, db.Name)); + var behind = AvailabilityGroupStatus.BehindBy(false, primary?.LastCommit, db.LastCommit); + var state = db.SyncState ?? "state unknown"; + yield return behind is null + ? $"{db.Name}: {state}, amount behind unknown (the primary's last commit time is not visible from here)." + : $"{db.Name}: {state}, {behind} behind the primary."; + } + } + + private static FailoverPlan Refuse(string reason) => new(FailoverKind.Refused, [], [], reason); + + private static bool SameName(string a, string b) => string.Equals(a, b, StringComparison.OrdinalIgnoreCase); + + internal static string Id(string name) => $"[{name.Replace("]", "]]")}]"; + + internal static string Lit(string value) => $"N'{value.Replace("'", "''")}'"; +} diff --git a/src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupQueries.cs b/src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupQueries.cs new file mode 100644 index 00000000..64c6f47f --- /dev/null +++ b/src/DataTray.Tools.MsSqlAdmin/AvailabilityGroupQueries.cs @@ -0,0 +1,114 @@ +using static DataTray.Tools.MsSqlAdmin.AvailabilityGroupFailover; + +namespace DataTray.Tools.MsSqlAdmin; + +/// +/// The four reads the failover planner is built from, and the parsing of their rows into an +/// . Shaped after the SE-284 dashboard's queries (AvailabilityGroupDashboardView +/// in the MSSQL provider) and carrying the same lessons: _desc values compared case-insensitively, +/// small integer DMV columns read through , and database names taken from +/// dm_hadr_database_replica_cluster_states because database_id is instance-local. They cannot be +/// literally shared: that view talks to SqlConnection with parameters, and this plugin reaches SQL +/// Server only through the host's provider — hence the group name as an escaped literal here. +/// +/// +/// Every query tolerates being run on a secondary, which only sees its own rows in the replica/database +/// state DMVs: the joins to them are LEFT joins, and whatever is missing stays null in the topology rather +/// than being guessed. +/// +internal static class AvailabilityGroupQueries +{ + public static string Group(string group) => + $""" + SELECT ag.cluster_type_desc, ag.required_synchronized_secondaries_to_commit, ags.primary_replica + FROM sys.availability_groups ag + LEFT JOIN sys.dm_hadr_availability_group_states ags ON ags.group_id = ag.group_id + WHERE ag.name = {Lit(group)} + """; + + public static string Replicas(string group) => + $""" + SELECT ar.replica_server_name, ar.availability_mode_desc, ars.connected_state_desc, ars.is_local + FROM sys.availability_replicas ar + JOIN sys.availability_groups ag ON ag.group_id = ar.group_id + LEFT JOIN sys.dm_hadr_availability_replica_states ars ON ars.replica_id = ar.replica_id + WHERE ag.name = {Lit(group)} + ORDER BY ar.replica_server_name + """; + + public static string Databases(string group) => + $""" + SELECT dcs.database_name, ar.replica_server_name, drs.synchronization_state_desc, + dcs.is_failover_ready, drs.last_commit_time + FROM sys.dm_hadr_database_replica_cluster_states dcs + JOIN sys.availability_replicas ar ON ar.replica_id = dcs.replica_id + JOIN sys.availability_groups ag ON ag.group_id = ar.group_id + LEFT JOIN sys.dm_hadr_database_replica_states drs + ON drs.replica_id = dcs.replica_id AND drs.group_database_id = dcs.group_database_id + WHERE ag.name = {Lit(group)} + ORDER BY dcs.database_name, ar.replica_server_name + """; + + public static string Listener(string group) => + $""" + SELECT l.dns_name, l.port, ip.ip_address, ip.ip_subnet_mask, ip.is_dhcp + FROM sys.availability_group_listeners l + JOIN sys.availability_groups ag ON ag.group_id = l.group_id + LEFT JOIN sys.availability_group_listener_ip_addresses ip ON ip.listener_id = l.listener_id + WHERE ag.name = {Lit(group)} + """; + + /// The instance's own name as a replica name has to match it — the check that a saved + /// connection really points at the replica it was picked for. + public const string ServerName = "SELECT CAST(SERVERPROPERTY('ServerName') AS nvarchar(256))"; + + /// The replica name the launch connection is, or null when it hosts none of the group's replicas. + public static string? LocalReplica(IReadOnlyList replicaRows) => + replicaRows.FirstOrDefault(r => Bool(r[3]) == true) is { } row ? Str(row[0]) : null; + + public static AgTopology Parse( + string group, + IReadOnlyList groupRows, + IReadOnlyList replicaRows, + IReadOnlyList databaseRows, + IReadOnlyList listenerRows) + { + if (groupRows.Count == 0) + { + throw new InvalidOperationException($"Availability group {group} does not exist on this instance."); + } + + var g = groupRows[0]; + var replicas = replicaRows + .Select(r => new AgReplica( + Req(r[0]), + string.Equals(Str(r[1]), "SYNCHRONOUS_COMMIT", StringComparison.OrdinalIgnoreCase), + Str(r[2]) is { } connected ? string.Equals(connected, "CONNECTED", StringComparison.OrdinalIgnoreCase) : null)) + .ToList(); + var databases = databaseRows + .Select(r => new AgDatabase(Req(r[0]), Req(r[1]), Str(r[2])?.ToUpperInvariant(), Bool(r[3]) == true, r[4] as DateTime?)) + .ToList(); + + AgListener? listener = null; + if (listenerRows.Count > 0) + { + var first = listenerRows[0]; + var addresses = listenerRows + .Where(r => r[2] is not null && Bool(r[4]) != true) + .Select(r => (Req(r[2]), Str(r[3]) ?? string.Empty)) + .ToList(); + listener = new AgListener(Req(first[0]), Convert.ToInt32(first[1]), addresses); + } + + return new AgTopology(group, Str(g[0]), Str(g[2]), g[1] is null ? 0 : Convert.ToInt32(g[1]), + replicas, databases, listener); + } + + // For columns the catalog declares NOT NULL (names); a null there means the query is wrong, not the data. + private static string Req(object? value) => + Str(value) ?? throw new InvalidOperationException("An availability group catalog column that is never NULL came back NULL."); + + private static string? Str(object? value) => value is null or DBNull ? null : Convert.ToString(value); + + private static bool? Bool(object? value) => value is null or DBNull ? null : Convert.ToBoolean(value); +} diff --git a/src/DataTray.Tools.MsSqlAdmin/DataTray.Tools.MsSqlAdmin.csproj b/src/DataTray.Tools.MsSqlAdmin/DataTray.Tools.MsSqlAdmin.csproj index 453a5e30..1a0a1833 100644 --- a/src/DataTray.Tools.MsSqlAdmin/DataTray.Tools.MsSqlAdmin.csproj +++ b/src/DataTray.Tools.MsSqlAdmin/DataTray.Tools.MsSqlAdmin.csproj @@ -20,6 +20,13 @@ + + + + +