diff --git a/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs b/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs index da9093c3f9..ed6b760e81 100644 --- a/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs +++ b/Darling/Darling.Tests/DarlingMcpAgToolsTests.cs @@ -66,7 +66,7 @@ private static (string Name, bool Optional)[] McpParams(string toolName) } [Theory] - [InlineData("get_ag_health", "server_name")] + [InlineData("get_ag_health", "server_name,limit")] public void ParamContract_MatchesContract(string toolName, string expectedCsv) { Assert.Equal(expectedCsv.Split(','), McpParams(toolName).Select(p => p.Name).ToArray()); @@ -505,6 +505,115 @@ await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) = } } + /// + /// #4471: the cap. Seeds 42 HEALTHY single-replica groups, then one CRITICAL group (a suspended database) + /// LAST in insertion order, so a naive "first N inserted" cap would return only healthy groups and a naive + /// name sort would bury the critical one under "AG_A0" through "AG_A9". Asserts the default call (no + /// limit argument) returns exactly the default page, flags the cut with the right counts, and puts + /// the critical group FIRST — the reason this cap exists at all. + /// + [Fact] + public async Task AgHealth_DefaultLimitCapsMostSevereFirstAndFlagsTruncation_AgainstDevPostgres() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live AG-health test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await DeleteRowsAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + var bodySucceeded = false; + try + { + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + var when = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-5); + + const int HealthyGroupCount = 42; + for (var i = 0; i < HealthyGroupCount; i++) + { + await InsertReplicaAsync(connection, ct, when, $"AG_A{i:D2}", $"NODE{i:D2}", "PRIMARY", "HEALTHY"); + } + + /* The one unhealthy group, planted LAST — after every healthy one both by insertion order and by + name ("AG_ZZZ" sorts after every "AG_A.."), so only severity-first ordering can put it first. */ + await InsertReplicaAsync(connection, ct, when, "AG_ZZZ_CRITICAL", "CRITNODE", "PRIMARY", "HEALTHY"); + await DarlingMcpTestData.ExecAsync(connection, ct, + @"INSERT INTO ag_database_replica_states (collection_id, collection_time, server_id, server_name, ag_name, database_name, replica_server_name, is_local, synchronization_state_desc, last_hardened_lsn, last_commit_lsn, log_send_queue_size, redo_queue_size, log_send_rate, redo_rate, is_suspended, suspend_reason_desc, availability_mode_desc, secondary_lag_seconds) +VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,$19)", + CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(when), ServerId, ServerName, "AG_ZZZ_CRITICAL", "StoppedDb", "CRITNODE", false, + "NOT SYNCHRONIZING", "0x0003", "0x0004", 9000L, 0L, 0L, 0L, true, "SUSPEND_FROM_USER", "SYNCHRONOUS_COMMIT", 0L); + + const int TotalGroupCount = HealthyGroupCount + 1; + + /* No limit argument — the default (DarlingMcpAgTools.DefaultGroupLimit) must cap the response. */ + var json = await DarlingMcpAgTools.GetAgHealth(postgres, ServerName); + Assert.False(McpHelpers.IsErrorEnvelope(json), $"tool returned an error: {json}"); + + using var doc = JsonDocument.Parse(json); + var root = doc.RootElement; + var groups = root.GetProperty("availability_groups").EnumerateArray().ToList(); + + Assert.Equal(DarlingMcpAgTools.DefaultGroupLimit, groups.Count); + Assert.Equal(DarlingMcpAgTools.DefaultGroupLimit, root.GetProperty("groups_returned").GetInt32()); + Assert.Equal(TotalGroupCount, root.GetProperty("groups_total").GetInt32()); + Assert.True(root.GetProperty("groups_truncated").GetBoolean()); + Assert.Contains("TRUNCATED", root.GetProperty("groups_truncated_note").GetString()); + + /* THE reason this cap exists: the one critical group is first, ahead of every healthy one. */ + Assert.Equal("AG_ZZZ_CRITICAL", groups[0].GetProperty("ag_name").GetString()); + Assert.Equal("Critical", groups[0].GetProperty("severity").GetString()); + + /* limit >= total: no cut at all. */ + var uncappedJson = await DarlingMcpAgTools.GetAgHealth(postgres, ServerName, limit: TotalGroupCount); + using var uncappedDoc = JsonDocument.Parse(uncappedJson); + var uncappedRoot = uncappedDoc.RootElement; + Assert.Equal(TotalGroupCount, uncappedRoot.GetProperty("availability_groups").GetArrayLength()); + Assert.False(uncappedRoot.GetProperty("groups_truncated").GetBoolean()); + Assert.Equal(JsonValueKind.Null, uncappedRoot.GetProperty("groups_truncated_note").ValueKind); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) => + await DeleteRowsAsync(cleanup, cleanupCt)); + } + } + + /// + /// #4471: an out-of-range limit is REFUSED through the shared , + /// never clamped or ignored, the same rule get_alert_history and get_blocking_snapshots apply + /// to their own limit. No topology needs to be planted: the refusal happens before any read. + /// + [Fact] + public async Task AgHealth_RefusesOutOfRangeLimit_AgainstDevPostgres() + { + var cs = ConnectionString; + Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live AG-health test."); + + var ct = TestContext.Current.CancellationToken; + using var connection = new NpgsqlConnection(cs); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + await using var postgres = NpgsqlDataSource.Create(cs!); + + foreach (var badLimit in new[] { 0, -1, 1001 }) + { + var refused = await DarlingMcpAgTools.GetAgHealth(postgres, limit: badLimit); + Assert.True(McpHelpers.IsRefusalEnvelope(refused), $"limit {badLimit} must be refused: {refused}"); + Assert.Equal("limit", JsonDocument.Parse(refused).RootElement.GetProperty("hints").GetProperty("parameter").GetString()); + } + + foreach (var okLimit in new[] { 1, 1000 }) + { + var accepted = await DarlingMcpAgTools.GetAgHealth(postgres, limit: okLimit); + Assert.False(McpHelpers.IsErrorEnvelope(accepted), $"limit {okLimit} must be accepted: {accepted}"); + } + } + private static async Task DeleteRowsAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct) { var sql = string.Join(" ", new[] { "ag_replica_states", "ag_database_replica_states" } diff --git a/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt b/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt index e9347eda4f..ea676e5281 100644 --- a/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt +++ b/Darling/Darling.Tests/McpToolsListBudget/DarlingMcpAgTools.txt @@ -1,3 +1,4 @@ # DarlingMcpAgTools: tools/list budget for #3898. Ceilings only go down; see McpToolsListBudgetTests. One block per tool, blank line between blocks. tool get_ag_health 616 +param get_ag_health.limit 155 param get_ag_health.server_name 118 diff --git a/Darling/Darling.Tests/McpToolsListBudgetTests.cs b/Darling/Darling.Tests/McpToolsListBudgetTests.cs index 278d2e0a35..ce2e06944f 100644 --- a/Darling/Darling.Tests/McpToolsListBudgetTests.cs +++ b/Darling/Darling.Tests/McpToolsListBudgetTests.cs @@ -85,6 +85,9 @@ public McpToolsListBudgetTests(ITestOutputHelper output) /// here. /// /// + /* #4471: +199 bytes for get_ag_health's new limit parameter (155 bytes for its own served description; + the served head is unchanged, since the size-budget guidance for the cap moved to the tool's + <> tail, which is not served in tools/list and so is not counted here). */ /* #4198/#4199 (M2b): +839 bytes for get_fleet_overview's worst_only/band filters (2 new params) and get_collection_log's fleet-form server_name/limit descriptions, after trimming both to the D2 200-char parameter cap and moving the rest to each tool's tail (get_tool_guide), which is not served in @@ -187,7 +190,7 @@ shared 32 KB budget and is not a served description either. */ // #4452 (merge): re-measured on the tree combining dev's #4442 get_read_latency addition with this // branch's scheduler-issues tool description growth (+95). Constant set to the value // McpToolsListBudgetTests itself measured on the merged tree, not the two deltas added by hand. - private const int TotalCeilingBytes = 176_277; + private const int TotalCeilingBytes = 176_476; diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs index 62f8cf34a1..3b3273b4ca 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingWebEndpoints.cs @@ -2373,7 +2373,7 @@ private static CatalogRead R(string category, string description, params Catalog ["get_daily_summary_range"] = R(CatOverview, "One daily health summary per collected day over a span of days - the Performance Calendar's month grid.", PServer(), PInt("days_back", 30), PAsOf()), ["get_fleet_overview"] = R(CatOverview, "The banded cross-server fleet roll-up.", PHours(DefaultFleetHours), PTextDefault("detail", "summary"), PBool("worst_only", false), PText("band")), ["get_sweep_reports"] = R(CatOverview, "The scheduled Fleet Sweep Reports: the sweep timeline for the window, the newest sweep in full, and the watch-item worklist - or one sweep by sweep_id (a string; the ids do not survive a JSON number round trip).", PHours(1), PAsOf(), PText("sweep_id"), PText("watch_state")), - ["get_ag_health"] = R(CatOverview, "Availability Group topology: replicas and per-database secondary state.", PServer()), + ["get_ag_health"] = R(CatOverview, "Availability Group topology: replicas and per-database secondary state.", PServer(), PLimit(DarlingMcpAgTools.DefaultGroupLimit)), ["get_store_metrics"] = R(CatOverview, "The monitoring store's own size/compression/growth (self-metrics): a summary by default, object_kind to list one kind, an exact object_name for one object's daily series.", PInt("days_back", 30), PText("object_kind"), PText("object_name"), PLimit(DarlingMcpStoreMetricsTools.DefaultLimit)), ["get_store_log"] = R(CatOverview, "What the monitoring store's OWN PostgreSQL server log recorded - a per-class census with the capture denominator beside it, not the lines. Deliberately unbanded.", PHours(24), PLimit(DarlingMcpStoreLogTools.DefaultRetainedLimit), PAsOf()), ["get_store_query_stats"] = R(CatOverview, "The monitoring store's OWN SQL statements ranked by server-side cost (pg_stat_statements), split by the role that ran them (on a managed store: the web viewer, MCP tools, the Darling Viewer, or the service itself).", PText("role"), PText("order_by"), PTop(DarlingMcpStoreQueryStatsTools.DefaultTop), PBool("full_text", false)), @@ -3279,7 +3279,7 @@ sizing the points itself (the OptionalDouble rule). */ ["get_daily_summary"] = (c, pg, an) => DarlingMcpHealthTools.GetDailySummary(pg, Server(c), Str(c, "summary_date"), c.RequestAborted), ["get_daily_summary_range"] = (c, pg, an) => DarlingMcpHealthTools.GetDailySummaryRange(pg, Server(c), QueryInt(c, "days_back", null, 30), AsOf(c), c.RequestAborted), ["get_fleet_overview"] = (c, pg, an) => DarlingMcpFleetTools.GetFleetOverview(pg, Hours(c, DefaultFleetHours), Str(c, "detail") ?? "summary", QueryBool(c, "worst_only", false), Str(c, "band"), c.RequestAborted), - ["get_ag_health"] = (c, pg, an) => DarlingMcpAgTools.GetAgHealth(pg, Server(c), c.RequestAborted), + ["get_ag_health"] = (c, pg, an) => DarlingMcpAgTools.GetAgHealth(pg, Server(c), Rows(c, "limit", DarlingMcpAgTools.DefaultGroupLimit), c.RequestAborted), ["get_store_metrics"] = (c, pg, an) => DarlingMcpStoreMetricsTools.GetStoreMetrics(pg, QueryInt(c, "days_back", null, 30), Str(c, "object_kind"), Str(c, "object_name"), Rows(c, "limit", DarlingMcpStoreMetricsTools.DefaultLimit), c.RequestAborted), ["get_store_log"] = (c, pg, an) => DarlingMcpStoreLogTools.GetStoreLog(pg, Hours(c, 24), Rows(c, "limit", DarlingMcpStoreLogTools.DefaultRetainedLimit), AsOf(c), c.RequestAborted), ["get_store_query_stats"] = (c, pg, an) => DarlingMcpStoreQueryStatsTools.GetStoreQueryStats(pg, Str(c, "role"), Str(c, "order_by") ?? "total_time", Rows(c, "top", DarlingMcpStoreQueryStatsTools.DefaultTop), QueryBool(c, "full_text", false), c.RequestAborted), diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs index e825873a6a..1102f6b1ed 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingAgReader.cs @@ -87,7 +87,8 @@ public static async Task GetAgHealthAsync( NpgsqlDataSource postgres, int? serverIdFilter = null, DateTime? nowUtc = null, - CancellationToken cancellationToken = default) + CancellationToken cancellationToken = default, + int? limit = null) { var effectiveNow = nowUtc ?? DateTime.UtcNow; var replicas = await ReadReplicasAsync(postgres, serverIdFilter, effectiveNow, cancellationToken); @@ -99,7 +100,7 @@ costs exactly one indexed read that returns nothing — this read runs on every ? new List() : await ReadDatabasesAsync(postgres, serverIdFilter, effectiveNow, cancellationToken); - return Build(replicas, databases, effectiveNow); + return Build(replicas, databases, effectiveNow, limit); } /// @@ -127,7 +128,8 @@ public static Task GetAvailabilityGroupCountAsync( internal static AgHealthResult Build( IReadOnlyList replicas, IReadOnlyList databases, - DateTime nowUtc) + DateTime nowUtc, + int? limit = null) { /* Group key is (server_id, ag_name) — one card per reporting server's view of an AG. A NULL ag_name is possible under quorum loss (the catalog views fall back to cached metadata); it groups under an empty @@ -160,6 +162,18 @@ NULLs a non-local or quorum-lost replica reports never mask a real Healthy/Warni worst = Worse(worst, database.SynchronizationStateSeverity); } + /* The tie-break magnitude for #4471's cap: the worst (largest) queue-drain signal anywhere in the + group — secondary lag if any database reports it, else the bigger of the two queue depths. Used + only to order groups that TIE on severity (severity is not banded on lag/queue by design, see the + class doc's "What the badge deliberately does NOT band" paragraph), so a page cut by limit drops + the least-lagging tied groups first rather than an arbitrary alphabetical tail. */ + long worstMagnitude = 0; + foreach (var database in databaseViews) + { + worstMagnitude = Math.Max(worstMagnitude, database.SecondaryLagSeconds ?? 0); + worstMagnitude = Math.Max(worstMagnitude, Math.Max(database.LogSendQueueKb ?? 0, database.RedoQueueKb ?? 0)); + } + groups.Add(new AvailabilityGroupView { ServerId = first.ServerId, @@ -172,27 +186,50 @@ NULLs a non-local or quorum-lost replica reports never mask a real Healthy/Warni SeverityLabel = SeverityLabel(worst), Replicas = replicaViews, Databases = databaseViews, + WorstMagnitude = worstMagnitude, }); } - /* Worst-first, then by name — the same "problems surface without scrolling" ordering the fleet roll-up - uses, and DESC by severity matches the house grid default. */ + /* Worst-first, then by the largest lag/queue magnitude, then by name — the same "problems surface without + scrolling" ordering the fleet roll-up uses (DESC by severity matches the house grid default), with the + magnitude tie-break (#4471) so a page a limit cuts drops the LEAST-lagging tied groups, never the most + severe. */ groups.Sort(CompareGroups); + var totalGroupCount = groups.Count; + var pagedGroups = groups; + var groupsTruncated = false; + if (limit is int effectiveLimit && totalGroupCount > effectiveLimit) + { + pagedGroups = groups.Take(Math.Max(0, effectiveLimit)).ToList(); + groupsTruncated = true; + } + return new AgHealthResult { /* Naive UTC, like every other instant the API emits — the browser appends the zone itself (R5). */ GeneratedAt = DateTime.SpecifyKind(nowUtc, DateTimeKind.Unspecified), - AvailabilityGroupCount = groups.Count, + /* #4471: these three roll-up counts (and WorstSeverity below) describe the WHOLE scope, not just the + returned page — the same reason findings_truncated's total_finding_count in get_analysis_findings + is measured before that tool's own limit cuts. A caller reading distinct_ag_count off a truncated + page must still get the fleet's real distinct-AG count, not the page's. */ + AvailabilityGroupCount = totalGroupCount, ReportingServerCount = groups.Select(g => g.ServerId).Distinct().Count(), DistinctAgCount = groups.Select(g => Key(g.AgName)).Distinct(StringComparer.Ordinal).Count(), WorstSeverity = groups.Count == 0 ? HealthSeverity.Unknown : groups.Max(g => g.Severity), - AvailabilityGroups = groups, + AvailabilityGroups = pagedGroups, + GroupsReturned = pagedGroups.Count, + GroupsTotal = totalGroupCount, + GroupsTruncated = groupsTruncated, + GroupsTruncatedNote = groupsTruncated + ? $"TRUNCATED: {totalGroupCount} groups were in scope; only the top {pagedGroups.Count} (most severe first, then by the largest lag/queue depth) are returned. Scope by server_name, or raise limit, to see the rest." + : null, }; } - /// Worst severity first, then AG name, then reporting server — so the several perspectives on one AG - /// stay adjacent once severity ties. + /// Worst severity first, then the largest lag/queue magnitude (#4471's cap tie-break — see the + /// group-building loop above), then AG name, then reporting server — so the several perspectives on one AG + /// stay adjacent once every other key ties. private static int CompareGroups(AvailabilityGroupView a, AvailabilityGroupView b) { var bySeverity = b.Severity.CompareTo(a.Severity); @@ -201,6 +238,12 @@ private static int CompareGroups(AvailabilityGroupView a, AvailabilityGroupView return bySeverity; } + var byMagnitude = b.WorstMagnitude.CompareTo(a.WorstMagnitude); + if (byMagnitude != 0) + { + return byMagnitude; + } + var byName = string.Compare(a.AgName ?? "", b.AgName ?? "", StringComparison.OrdinalIgnoreCase); return byName != 0 ? byName @@ -586,6 +629,12 @@ public sealed class AvailabilityGroupView [JsonPropertyName("severity_label")] public string SeverityLabel { get; init; } = ""; [JsonPropertyName("replicas")] public IReadOnlyList Replicas { get; init; } = Array.Empty(); [JsonPropertyName("databases")] public IReadOnlyList Databases { get; init; } = Array.Empty(); + + /// #4471's severity-tie tie-break ONLY — the largest secondary_lag_seconds or queue-depth (KB) + /// anywhere in the group, never serialized. Lag and queue depth are deliberately NOT banded into severity + /// (see the class doc), so this exists purely to order same-severity groups by that raw magnitude before a + /// limit cut, rather than let it fall to an arbitrary name sort. + [JsonIgnore] internal long WorstMagnitude { get; init; } } /// The full AG topology payload — the /api/ag body and the get_ag_health MCP tool's @@ -607,6 +656,26 @@ public sealed class AgHealthResult [JsonPropertyName("worst_severity")] public HealthSeverity WorstSeverity { get; init; } [JsonPropertyName("availability_groups")] public IReadOnlyList AvailabilityGroups { get; init; } = Array.Empty(); + + /// #4471: how many groups this response actually carries in — + /// when nothing was cut, or the caller's limit otherwise. + [JsonPropertyName("groups_returned")] public int GroupsReturned { get; init; } + + /// #4471: how many (reporting server, AG) groups the scope held BEFORE the limit cut — + /// identical to , carried under its own name so a caller reading this + /// tool's truncation trio (//) + /// never has to cross-reference a differently-named field for the same number. + [JsonPropertyName("groups_total")] public int GroupsTotal { get; init; } + + /// #4471: true when exceeded the caller's limit and the tail — the + /// least severe, then least-lagging — was cut. Scope by server_name or raise limit to see the + /// rest; the most severe groups are always the ones kept. + [JsonPropertyName("groups_truncated")] public bool GroupsTruncated { get; init; } + + /// Null unless — the same shape as get_analysis_findings' + /// findings_truncated_note, spelling out the cut and the fix instead of leaving a caller to infer one + /// from the bare flag. + [JsonPropertyName("groups_truncated_note")] public string? GroupsTruncatedNote { get; init; } } /// The /api/ag/count body (#4189) — the one field diff --git a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs index e66abee0b2..00fb289940 100644 --- a/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs +++ b/Darling/PerformanceMonitor.Darling.Service/Mcp/DarlingMcpAgTools.cs @@ -33,6 +33,17 @@ namespace PerformanceMonitor.Darling.Service.Mcp; [McpServerToolType] public sealed class DarlingMcpAgTools { + /// + /// #4471: the fleet-wide page cap on groups (one card per reporting server's view of one AG), the same shape + /// as get_analysis_findings' limit. An uncapped fleet-wide call on a production fleet (43 + /// servers, several AGs with many databases) measured 265,794 characters — over 8x the shared + /// (32 KB). On a 43-server/3-AG/2-replica/3-database fixture + /// (~2.7 KB/group, close to that reported shape), 11 groups measured 30,115 bytes and 12 measured 32,812 — + /// so 11 is the largest default that stays under budget on that shape; see DarlingMcpAgToolsTests for + /// the before/after this was set from. + /// + public const int DefaultGroupLimit = 11; + [McpServerTool(Name = "get_ag_health"), Description( "AG health fleet-wide from each server's latest collection: replica role/state and per-database " + "secondary state (queue KB, rate KB/s, lag sec, drain min, suspended+why). One row per REPLICA's view: " + @@ -55,12 +66,22 @@ public sealed class DarlingMcpAgTools "movement is suspended). Each group carries its collection_time: the collectors write NO row for a server " + "with no AGs, so a server whose AGs were dropped keeps returning its last non-empty snapshot until then — " + "an old collection_time on a group is that case, not a live reading. Returns an empty result on a fleet " + - "with no Availability Groups.")] + "with no Availability Groups. limit pages the groups, MOST SEVERE FIRST then by the largest " + + "secondary_lag_seconds/queue depth in the group, so the cap never hides a problem — an uncapped fleet-wide " + + "call measured 265,794 characters on a 43-server production fleet with several many-database AGs, well " + + "over an MCP client's typical per-result limit. Default 11 groups; groups_truncated (with " + + "groups_truncated_note) flags when the scope held more than that — groups_total says how many, " + + "groups_returned says how many came back, and the fix is to scope by server_name (one server's view is " + + "rarely more than a handful of groups) or raise limit for the rest.")] public static async Task GetAgHealth( NpgsqlDataSource postgres, [Description("Server name or display name to limit the topology to one monitored server's view. Optional — omit for the whole fleet.")] string? server_name = null, + [Description("Maximum groups to return, most severe first, then by the largest lag/queue depth in the group. Default 11, range 1-1000. groups_truncated flags a cut here.")] int limit = DefaultGroupLimit, CancellationToken cancellationToken = default) { + var limitError = McpHelpers.ValidateTop(limit); + if (limitError != null) return limitError; + /* Fleet-wide by default: only resolve when a name was actually supplied. The shared resolver auto-selects a sole registered server for an omitted name, which is right for a per-server tool and wrong here — it would silently narrow the fleet view on a one-server store. */ @@ -76,7 +97,7 @@ would silently narrow the fleet view on a one-server store. */ try { - var result = await DarlingAgReader.GetAgHealthAsync(postgres, serverIdFilter, cancellationToken: cancellationToken); + var result = await DarlingAgReader.GetAgHealthAsync(postgres, serverIdFilter, cancellationToken: cancellationToken, limit: limit); if (result.AvailabilityGroupCount == 0) {