diff --git a/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs b/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs
index da9093c3f9..ed6b760e81 100644
--- a/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs
+++ b/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs
@@ -66,7 +66,7 @@ private static (string Name, bool Optional)[] McpParams(string toolName)
}
[Theory]
- [InlineData("get_ag_health", "server_name")]
+ [InlineData("get_ag_health", "server_name,limit")]
public void ParamContract_MatchesContract(string toolName, string expectedCsv)
{
Assert.Equal(expectedCsv.Split(','), McpParams(toolName).Select(p => p.Name).ToArray());
@@ -505,6 +505,115 @@ await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) =
}
}
+ ///
+ /// #4471: the cap. Seeds 42 HEALTHY single-replica groups, then one CRITICAL group (a suspended database)
+ /// LAST in insertion order, so a naive "first N inserted" cap would return only healthy groups and a naive
+ /// name sort would bury the critical one under "AG_A0" through "AG_A9". Asserts the default call (no
+ /// limit argument) returns exactly the default page, flags the cut with the right counts, and puts
+ /// the critical group FIRST — the reason this cap exists at all.
+ ///
+ [Fact]
+ public async Task AgHealth_DefaultLimitCapsMostSevereFirstAndFlagsTruncation_AgainstDevPostgres()
+ {
+ var cs = ConnectionString;
+ Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live AG-health test.");
+
+ var ct = TestContext.Current.CancellationToken;
+ using var connection = new NpgsqlConnection(cs);
+ await connection.OpenAsync(ct);
+ await PgMigrations.MigrateAsync(connection, ct);
+ await DeleteRowsAsync(connection, ct);
+ await using var postgres = NpgsqlDataSource.Create(cs!);
+
+ var bodySucceeded = false;
+ try
+ {
+ await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct);
+ var when = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-5);
+
+ const int HealthyGroupCount = 42;
+ for (var i = 0; i < HealthyGroupCount; i++)
+ {
+ await InsertReplicaAsync(connection, ct, when, $"AG_A{i:D2}", $"NODE{i:D2}", "PRIMARY", "HEALTHY");
+ }
+
+ /* The one unhealthy group, planted LAST — after every healthy one both by insertion order and by
+ name ("AG_ZZZ" sorts after every "AG_A.."), so only severity-first ordering can put it first. */
+ await InsertReplicaAsync(connection, ct, when, "AG_ZZZ_CRITICAL", "CRITNODE", "PRIMARY", "HEALTHY");
+ await DarlingMcpTestData.ExecAsync(connection, ct,
+ @"INSERT INTO ag_database_replica_states (collection_id, collection_time, server_id, server_name, ag_name, database_name, replica_server_name, is_local, synchronization_state_desc, last_hardened_lsn, last_commit_lsn, log_send_queue_size, redo_queue_size, log_send_rate, redo_rate, is_suspended, suspend_reason_desc, availability_mode_desc, secondary_lag_seconds)
+VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,$19)",
+ CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(when), ServerId, ServerName, "AG_ZZZ_CRITICAL", "StoppedDb", "CRITNODE", false,
+ "NOT SYNCHRONIZING", "0x0003", "0x0004", 9000L, 0L, 0L, 0L, true, "SUSPEND_FROM_USER", "SYNCHRONOUS_COMMIT", 0L);
+
+ const int TotalGroupCount = HealthyGroupCount + 1;
+
+ /* No limit argument — the default (DarlingMcpAgTools.DefaultGroupLimit) must cap the response. */
+ var json = await DarlingMcpAgTools.GetAgHealth(postgres, ServerName);
+ Assert.False(McpHelpers.IsErrorEnvelope(json), $"tool returned an error: {json}");
+
+ using var doc = JsonDocument.Parse(json);
+ var root = doc.RootElement;
+ var groups = root.GetProperty("availability_groups").EnumerateArray().ToList();
+
+ Assert.Equal(DarlingMcpAgTools.DefaultGroupLimit, groups.Count);
+ Assert.Equal(DarlingMcpAgTools.DefaultGroupLimit, root.GetProperty("groups_returned").GetInt32());
+ Assert.Equal(TotalGroupCount, root.GetProperty("groups_total").GetInt32());
+ Assert.True(root.GetProperty("groups_truncated").GetBoolean());
+ Assert.Contains("TRUNCATED", root.GetProperty("groups_truncated_note").GetString());
+
+ /* THE reason this cap exists: the one critical group is first, ahead of every healthy one. */
+ Assert.Equal("AG_ZZZ_CRITICAL", groups[0].GetProperty("ag_name").GetString());
+ Assert.Equal("Critical", groups[0].GetProperty("severity").GetString());
+
+ /* limit >= total: no cut at all. */
+ var uncappedJson = await DarlingMcpAgTools.GetAgHealth(postgres, ServerName, limit: TotalGroupCount);
+ using var uncappedDoc = JsonDocument.Parse(uncappedJson);
+ var uncappedRoot = uncappedDoc.RootElement;
+ Assert.Equal(TotalGroupCount, uncappedRoot.GetProperty("availability_groups").GetArrayLength());
+ Assert.False(uncappedRoot.GetProperty("groups_truncated").GetBoolean());
+ Assert.Equal(JsonValueKind.Null, uncappedRoot.GetProperty("groups_truncated_note").ValueKind);
+
+ bodySucceeded = true;
+ }
+ finally
+ {
+ await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) =>
+ await DeleteRowsAsync(cleanup, cleanupCt));
+ }
+ }
+
+ ///
+ /// #4471: an out-of-range limit is REFUSED through the shared ,
+ /// never clamped or ignored, the same rule get_alert_history and get_blocking_snapshots apply
+ /// to their own limit. No topology needs to be planted: the refusal happens before any read.
+ ///
+ [Fact]
+ public async Task AgHealth_RefusesOutOfRangeLimit_AgainstDevPostgres()
+ {
+ var cs = ConnectionString;
+ Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live AG-health test.");
+
+ var ct = TestContext.Current.CancellationToken;
+ using var connection = new NpgsqlConnection(cs);
+ await connection.OpenAsync(ct);
+ await PgMigrations.MigrateAsync(connection, ct);
+ await using var postgres = NpgsqlDataSource.Create(cs!);
+
+ foreach (var badLimit in new[] { 0, -1, 1001 })
+ {
+ var refused = await DarlingMcpAgTools.GetAgHealth(postgres, limit: badLimit);
+ Assert.True(McpHelpers.IsRefusalEnvelope(refused), $"limit {badLimit} must be refused: {refused}");
+ Assert.Equal("limit", JsonDocument.Parse(refused).RootElement.GetProperty("hints").GetProperty("parameter").GetString());
+ }
+
+ foreach (var okLimit in new[] { 1, 1000 })
+ {
+ var accepted = await DarlingMcpAgTools.GetAgHealth(postgres, limit: okLimit);
+ Assert.False(McpHelpers.IsErrorEnvelope(accepted), $"limit {okLimit} must be accepted: {accepted}");
+ }
+ }
+
private static async Task DeleteRowsAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct)
{
var sql = string.Join(" ", new[] { "ag_replica_states", "ag_database_replica_states" }
diff --git a/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt b/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt
index e9347eda4f..ea676e5281 100644
--- a/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt
+++ b/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt
@@ -1,3 +1,4 @@
# DarlingMcpAgTools: tools/list budget for #3898. Ceilings only go down; see McpToolsListBudgetTests. One block per tool, blank line between blocks.
tool get_ag_health 616
+param get_ag_health.limit 155
param get_ag_health.server_name 118
diff --git a/Darling/Darling.Tests/McpToolsListBudgetTests.cs b/Darling/Darling.Tests/McpToolsListBudgetTests.cs
index 278d2e0a35..ce2e06944f 100644
--- a/Darling/Darling.Tests/McpToolsListBudgetTests.cs
+++ b/Darling/Darling.Tests/McpToolsListBudgetTests.cs
@@ -85,6 +85,9 @@ public McpToolsListBudgetTests(ITestOutputHelper output)
/// here.
///
///
+ /* #4471: +199 bytes for get_ag_health's new limit parameter (155 bytes for its own served description;
+ the served head is unchanged, since the size-budget guidance for the cap moved to the tool's
+ <> tail, which is not served in tools/list and so is not counted here). */
/* #4198/#4199 (M2b): +839 bytes for get_fleet_overview's worst_only/band filters (2 new params) and
get_collection_log's fleet-form server_name/limit descriptions, after trimming both to the D2 200-char
parameter cap and moving the rest to each tool's tail (get_tool_guide), which is not served in
@@ -187,7 +190,7 @@ shared 32 KB budget and is not a served description either. */
// #4452 (merge): re-measured on the tree combining dev's #4442 get_read_latency addition with this
// branch's scheduler-issues tool description growth (+95). Constant set to the value
// McpToolsListBudgetTests itself measured on the merged tree, not the two deltas added by hand.
- private const int TotalCeilingBytes = 176_277;
+ private const int TotalCeilingBytes = 176_476;
diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs
index 62f8cf34a1..3b3273b4ca 100644
--- a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs
+++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs
@@ -2373,7 +2373,7 @@ private static CatalogRead R(string category, string description, params Catalog
["get_daily_summary_range"] = R(CatOverview, "One daily health summary per collected day over a span of days - the Performance Calendar's month grid.", PServer(), PInt("days_back", 30), PAsOf()),
["get_fleet_overview"] = R(CatOverview, "The banded cross-server fleet roll-up.", PHours(DefaultFleetHours), PTextDefault("detail", "summary"), PBool("worst_only", false), PText("band")),
["get_sweep_reports"] = R(CatOverview, "The scheduled Fleet Sweep Reports: the sweep timeline for the window, the newest sweep in full, and the watch-item worklist - or one sweep by sweep_id (a string; the ids do not survive a JSON number round trip).", PHours(1), PAsOf(), PText("sweep_id"), PText("watch_state")),
- ["get_ag_health"] = R(CatOverview, "Availability Group topology: replicas and per-database secondary state.", PServer()),
+ ["get_ag_health"] = R(CatOverview, "Availability Group topology: replicas and per-database secondary state.", PServer(), PLimit(DarlingMcpAgTools.DefaultGroupLimit)),
["get_store_metrics"] = R(CatOverview, "The monitoring store's own size/compression/growth (self-metrics): a summary by default, object_kind to list one kind, an exact object_name for one object's daily series.", PInt("days_back", 30), PText("object_kind"), PText("object_name"), PLimit(DarlingMcpStoreMetricsTools.DefaultLimit)),
["get_store_log"] = R(CatOverview, "What the monitoring store's OWN PostgreSQL server log recorded - a per-class census with the capture denominator beside it, not the lines. Deliberately unbanded.", PHours(24), PLimit(DarlingMcpStoreLogTools.DefaultRetainedLimit), PAsOf()),
["get_store_query_stats"] = R(CatOverview, "The monitoring store's OWN SQL statements ranked by server-side cost (pg_stat_statements), split by the role that ran them (on a managed store: the web viewer, MCP tools, the Darling Viewer, or the service itself).", PText("role"), PText("order_by"), PTop(DarlingMcpStoreQueryStatsTools.DefaultTop), PBool("full_text", false)),
@@ -3279,7 +3279,7 @@ sizing the points itself (the OptionalDouble rule). */
["get_daily_summary"] = (c, pg, an) => DarlingMcpHealthTools.GetDailySummary(pg, Server(c), Str(c, "summary_date"), c.RequestAborted),
["get_daily_summary_range"] = (c, pg, an) => DarlingMcpHealthTools.GetDailySummaryRange(pg, Server(c), QueryInt(c, "days_back", null, 30), AsOf(c), c.RequestAborted),
["get_fleet_overview"] = (c, pg, an) => DarlingMcpFleetTools.GetFleetOverview(pg, Hours(c, DefaultFleetHours), Str(c, "detail") ?? "summary", QueryBool(c, "worst_only", false), Str(c, "band"), c.RequestAborted),
- ["get_ag_health"] = (c, pg, an) => DarlingMcpAgTools.GetAgHealth(pg, Server(c), c.RequestAborted),
+ ["get_ag_health"] = (c, pg, an) => DarlingMcpAgTools.GetAgHealth(pg, Server(c), Rows(c, "limit", DarlingMcpAgTools.DefaultGroupLimit), c.RequestAborted),
["get_store_metrics"] = (c, pg, an) => DarlingMcpStoreMetricsTools.GetStoreMetrics(pg, QueryInt(c, "days_back", null, 30), Str(c, "object_kind"), Str(c, "object_name"), Rows(c, "limit", DarlingMcpStoreMetricsTools.DefaultLimit), c.RequestAborted),
["get_store_log"] = (c, pg, an) => DarlingMcpStoreLogTools.GetStoreLog(pg, Hours(c, 24), Rows(c, "limit", DarlingMcpStoreLogTools.DefaultRetainedLimit), AsOf(c), c.RequestAborted),
["get_store_query_stats"] = (c, pg, an) => DarlingMcpStoreQueryStatsTools.GetStoreQueryStats(pg, Str(c, "role"), Str(c, "order_by") ?? "total_time", Rows(c, "top", DarlingMcpStoreQueryStatsTools.DefaultTop), QueryBool(c, "full_text", false), c.RequestAborted),
diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs
index e825873a6a..1102f6b1ed 100644
--- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs
+++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs
@@ -87,7 +87,8 @@ public static async Task GetAgHealthAsync(
NpgsqlDataSource postgres,
int? serverIdFilter = null,
DateTime? nowUtc = null,
- CancellationToken cancellationToken = default)
+ CancellationToken cancellationToken = default,
+ int? limit = null)
{
var effectiveNow = nowUtc ?? DateTime.UtcNow;
var replicas = await ReadReplicasAsync(postgres, serverIdFilter, effectiveNow, cancellationToken);
@@ -99,7 +100,7 @@ costs exactly one indexed read that returns nothing — this read runs on every
? new List()
: await ReadDatabasesAsync(postgres, serverIdFilter, effectiveNow, cancellationToken);
- return Build(replicas, databases, effectiveNow);
+ return Build(replicas, databases, effectiveNow, limit);
}
///
@@ -127,7 +128,8 @@ public static Task GetAvailabilityGroupCountAsync(
internal static AgHealthResult Build(
IReadOnlyList replicas,
IReadOnlyList databases,
- DateTime nowUtc)
+ DateTime nowUtc,
+ int? limit = null)
{
/* Group key is (server_id, ag_name) — one card per reporting server's view of an AG. A NULL ag_name is
possible under quorum loss (the catalog views fall back to cached metadata); it groups under an empty
@@ -160,6 +162,18 @@ NULLs a non-local or quorum-lost replica reports never mask a real Healthy/Warni
worst = Worse(worst, database.SynchronizationStateSeverity);
}
+ /* The tie-break magnitude for #4471's cap: the worst (largest) queue-drain signal anywhere in the
+ group — secondary lag if any database reports it, else the bigger of the two queue depths. Used
+ only to order groups that TIE on severity (severity is not banded on lag/queue by design, see the
+ class doc's "What the badge deliberately does NOT band" paragraph), so a page cut by limit drops
+ the least-lagging tied groups first rather than an arbitrary alphabetical tail. */
+ long worstMagnitude = 0;
+ foreach (var database in databaseViews)
+ {
+ worstMagnitude = Math.Max(worstMagnitude, database.SecondaryLagSeconds ?? 0);
+ worstMagnitude = Math.Max(worstMagnitude, Math.Max(database.LogSendQueueKb ?? 0, database.RedoQueueKb ?? 0));
+ }
+
groups.Add(new AvailabilityGroupView
{
ServerId = first.ServerId,
@@ -172,27 +186,50 @@ NULLs a non-local or quorum-lost replica reports never mask a real Healthy/Warni
SeverityLabel = SeverityLabel(worst),
Replicas = replicaViews,
Databases = databaseViews,
+ WorstMagnitude = worstMagnitude,
});
}
- /* Worst-first, then by name — the same "problems surface without scrolling" ordering the fleet roll-up
- uses, and DESC by severity matches the house grid default. */
+ /* Worst-first, then by the largest lag/queue magnitude, then by name — the same "problems surface without
+ scrolling" ordering the fleet roll-up uses (DESC by severity matches the house grid default), with the
+ magnitude tie-break (#4471) so a page a limit cuts drops the LEAST-lagging tied groups, never the most
+ severe. */
groups.Sort(CompareGroups);
+ var totalGroupCount = groups.Count;
+ var pagedGroups = groups;
+ var groupsTruncated = false;
+ if (limit is int effectiveLimit && totalGroupCount > effectiveLimit)
+ {
+ pagedGroups = groups.Take(Math.Max(0, effectiveLimit)).ToList();
+ groupsTruncated = true;
+ }
+
return new AgHealthResult
{
/* Naive UTC, like every other instant the API emits — the browser appends the zone itself (R5). */
GeneratedAt = DateTime.SpecifyKind(nowUtc, DateTimeKind.Unspecified),
- AvailabilityGroupCount = groups.Count,
+ /* #4471: these three roll-up counts (and WorstSeverity below) describe the WHOLE scope, not just the
+ returned page — the same reason findings_truncated's total_finding_count in get_analysis_findings
+ is measured before that tool's own limit cuts. A caller reading distinct_ag_count off a truncated
+ page must still get the fleet's real distinct-AG count, not the page's. */
+ AvailabilityGroupCount = totalGroupCount,
ReportingServerCount = groups.Select(g => g.ServerId).Distinct().Count(),
DistinctAgCount = groups.Select(g => Key(g.AgName)).Distinct(StringComparer.Ordinal).Count(),
WorstSeverity = groups.Count == 0 ? HealthSeverity.Unknown : groups.Max(g => g.Severity),
- AvailabilityGroups = groups,
+ AvailabilityGroups = pagedGroups,
+ GroupsReturned = pagedGroups.Count,
+ GroupsTotal = totalGroupCount,
+ GroupsTruncated = groupsTruncated,
+ GroupsTruncatedNote = groupsTruncated
+ ? $"TRUNCATED: {totalGroupCount} groups were in scope; only the top {pagedGroups.Count} (most severe first, then by the largest lag/queue depth) are returned. Scope by server_name, or raise limit, to see the rest."
+ : null,
};
}
- /// Worst severity first, then AG name, then reporting server — so the several perspectives on one AG
- /// stay adjacent once severity ties.
+ /// Worst severity first, then the largest lag/queue magnitude (#4471's cap tie-break — see the
+ /// group-building loop above), then AG name, then reporting server — so the several perspectives on one AG
+ /// stay adjacent once every other key ties.
private static int CompareGroups(AvailabilityGroupView a, AvailabilityGroupView b)
{
var bySeverity = b.Severity.CompareTo(a.Severity);
@@ -201,6 +238,12 @@ private static int CompareGroups(AvailabilityGroupView a, AvailabilityGroupView
return bySeverity;
}
+ var byMagnitude = b.WorstMagnitude.CompareTo(a.WorstMagnitude);
+ if (byMagnitude != 0)
+ {
+ return byMagnitude;
+ }
+
var byName = string.Compare(a.AgName ?? "", b.AgName ?? "", StringComparison.OrdinalIgnoreCase);
return byName != 0
? byName
@@ -586,6 +629,12 @@ public sealed class AvailabilityGroupView
[JsonPropertyName("severity_label")] public string SeverityLabel { get; init; } = "";
[JsonPropertyName("replicas")] public IReadOnlyList Replicas { get; init; } = Array.Empty();
[JsonPropertyName("databases")] public IReadOnlyList Databases { get; init; } = Array.Empty();
+
+ /// #4471's severity-tie tie-break ONLY — the largest secondary_lag_seconds or queue-depth (KB)
+ /// anywhere in the group, never serialized. Lag and queue depth are deliberately NOT banded into severity
+ /// (see the class doc), so this exists purely to order same-severity groups by that raw magnitude before a
+ /// limit cut, rather than let it fall to an arbitrary name sort.
+ [JsonIgnore] internal long WorstMagnitude { get; init; }
}
/// The full AG topology payload — the /api/ag body and the get_ag_health MCP tool's
@@ -607,6 +656,26 @@ public sealed class AgHealthResult
[JsonPropertyName("worst_severity")] public HealthSeverity WorstSeverity { get; init; }
[JsonPropertyName("availability_groups")] public IReadOnlyList AvailabilityGroups { get; init; } = Array.Empty();
+
+ /// #4471: how many groups this response actually carries in —
+ /// when nothing was cut, or the caller's limit otherwise.
+ [JsonPropertyName("groups_returned")] public int GroupsReturned { get; init; }
+
+ /// #4471: how many (reporting server, AG) groups the scope held BEFORE the limit cut —
+ /// identical to , carried under its own name so a caller reading this
+ /// tool's truncation trio (//)
+ /// never has to cross-reference a differently-named field for the same number.
+ [JsonPropertyName("groups_total")] public int GroupsTotal { get; init; }
+
+ /// #4471: true when exceeded the caller's limit and the tail — the
+ /// least severe, then least-lagging — was cut. Scope by server_name or raise limit to see the
+ /// rest; the most severe groups are always the ones kept.
+ [JsonPropertyName("groups_truncated")] public bool GroupsTruncated { get; init; }
+
+ /// Null unless — the same shape as get_analysis_findings'
+ /// findings_truncated_note, spelling out the cut and the fix instead of leaving a caller to infer one
+ /// from the bare flag.
+ [JsonPropertyName("groups_truncated_note")] public string? GroupsTruncatedNote { get; init; }
}
/// The /api/ag/count body (#4189) — the one field
diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs
index e66abee0b2..00fb289940 100644
--- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs
+++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs
@@ -33,6 +33,17 @@ namespace PerformanceMonitor.Darling.Service.Mcp;
[McpServerToolType]
public sealed class DarlingMcpAgTools
{
+ ///
+ /// #4471: the fleet-wide page cap on groups (one card per reporting server's view of one AG), the same shape
+ /// as get_analysis_findings' limit. An uncapped fleet-wide call on a production fleet (43
+ /// servers, several AGs with many databases) measured 265,794 characters — over 8x the shared
+ /// (32 KB). On a 43-server/3-AG/2-replica/3-database fixture
+ /// (~2.7 KB/group, close to that reported shape), 11 groups measured 30,115 bytes and 12 measured 32,812 —
+ /// so 11 is the largest default that stays under budget on that shape; see DarlingMcpAgToolsTests for
+ /// the before/after this was set from.
+ ///
+ public const int DefaultGroupLimit = 11;
+
[McpServerTool(Name = "get_ag_health"), Description(
"AG health fleet-wide from each server's latest collection: replica role/state and per-database " +
"secondary state (queue KB, rate KB/s, lag sec, drain min, suspended+why). One row per REPLICA's view: " +
@@ -55,12 +66,22 @@ public sealed class DarlingMcpAgTools
"movement is suspended). Each group carries its collection_time: the collectors write NO row for a server " +
"with no AGs, so a server whose AGs were dropped keeps returning its last non-empty snapshot until then — " +
"an old collection_time on a group is that case, not a live reading. Returns an empty result on a fleet " +
- "with no Availability Groups.")]
+ "with no Availability Groups. limit pages the groups, MOST SEVERE FIRST then by the largest " +
+ "secondary_lag_seconds/queue depth in the group, so the cap never hides a problem — an uncapped fleet-wide " +
+ "call measured 265,794 characters on a 43-server production fleet with several many-database AGs, well " +
+ "over an MCP client's typical per-result limit. Default 11 groups; groups_truncated (with " +
+ "groups_truncated_note) flags when the scope held more than that — groups_total says how many, " +
+ "groups_returned says how many came back, and the fix is to scope by server_name (one server's view is " +
+ "rarely more than a handful of groups) or raise limit for the rest.")]
public static async Task GetAgHealth(
NpgsqlDataSource postgres,
[Description("Server name or display name to limit the topology to one monitored server's view. Optional — omit for the whole fleet.")] string? server_name = null,
+ [Description("Maximum groups to return, most severe first, then by the largest lag/queue depth in the group. Default 11, range 1-1000. groups_truncated flags a cut here.")] int limit = DefaultGroupLimit,
CancellationToken cancellationToken = default)
{
+ var limitError = McpHelpers.ValidateTop(limit);
+ if (limitError != null) return limitError;
+
/* Fleet-wide by default: only resolve when a name was actually supplied. The shared resolver auto-selects
a sole registered server for an omitted name, which is right for a per-server tool and wrong here — it
would silently narrow the fleet view on a one-server store. */
@@ -76,7 +97,7 @@ would silently narrow the fleet view on a one-server store. */
try
{
- var result = await DarlingAgReader.GetAgHealthAsync(postgres, serverIdFilter, cancellationToken: cancellationToken);
+ var result = await DarlingAgReader.GetAgHealthAsync(postgres, serverIdFilter, cancellationToken: cancellationToken, limit: limit);
if (result.AvailabilityGroupCount == 0)
{