Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
111 changes: 110 additions & 1 deletion Darling/Darling.Tests/DarlingMcpAgToolsTests.cs
Original file line number Diff line number Diff line change
Expand Up @@ -66,7 +66,7 @@ private static (string Name, bool Optional)[] McpParams(string toolName)
}

[Theory]
[InlineData("get_ag_health", "server_name")]
[InlineData("get_ag_health", "server_name,limit")]
public void ParamContract_MatchesContract(string toolName, string expectedCsv)
{
Assert.Equal(expectedCsv.Split(','), McpParams(toolName).Select(p => p.Name).ToArray());
Expand Down Expand Up @@ -505,6 +505,115 @@ await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) =
}
}

/// <summary>
/// #4471: the cap. Seeds 42 HEALTHY single-replica groups, then one CRITICAL group (a suspended database)
/// LAST in insertion order, so a naive "first N inserted" cap would return only healthy groups and a naive
/// name sort would bury the critical one under "AG_A0" through "AG_A9". Asserts the default call (no
/// <c>limit</c> argument) returns exactly the default page, flags the cut with the right counts, and puts
/// the critical group FIRST — the reason this cap exists at all.
/// </summary>
[Fact]
public async Task AgHealth_DefaultLimitCapsMostSevereFirstAndFlagsTruncation_AgainstDevPostgres()
{
var cs = ConnectionString;
Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live AG-health test.");

var ct = TestContext.Current.CancellationToken;
using var connection = new NpgsqlConnection(cs);
await connection.OpenAsync(ct);
await PgMigrations.MigrateAsync(connection, ct);
await DeleteRowsAsync(connection, ct);
await using var postgres = NpgsqlDataSource.Create(cs!);

var bodySucceeded = false;
try
{
await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct);
var when = DarlingMcpTestData.TruncateToSeconds(DateTime.UtcNow).AddMinutes(-5);

const int HealthyGroupCount = 42;
for (var i = 0; i < HealthyGroupCount; i++)
{
await InsertReplicaAsync(connection, ct, when, $"AG_A{i:D2}", $"NODE{i:D2}", "PRIMARY", "HEALTHY");
}

/* The one unhealthy group, planted LAST — after every healthy one both by insertion order and by
name ("AG_ZZZ" sorts after every "AG_A.."), so only severity-first ordering can put it first. */
await InsertReplicaAsync(connection, ct, when, "AG_ZZZ_CRITICAL", "CRITNODE", "PRIMARY", "HEALTHY");
await DarlingMcpTestData.ExecAsync(connection, ct,
@"INSERT INTO ag_database_replica_states (collection_id, collection_time, server_id, server_name, ag_name, database_name, replica_server_name, is_local, synchronization_state_desc, last_hardened_lsn, last_commit_lsn, log_send_queue_size, redo_queue_size, log_send_rate, redo_rate, is_suspended, suspend_reason_desc, availability_mode_desc, secondary_lag_seconds)
VALUES ($1,$2,$3,$4,$5,$6,$7,$8,$9,$10,$11,$12,$13,$14,$15,$16,$17,$18,$19)",
CollectionIdGenerator.Next(), DarlingMcpTestData.Naive(when), ServerId, ServerName, "AG_ZZZ_CRITICAL", "StoppedDb", "CRITNODE", false,
"NOT SYNCHRONIZING", "0x0003", "0x0004", 9000L, 0L, 0L, 0L, true, "SUSPEND_FROM_USER", "SYNCHRONOUS_COMMIT", 0L);

const int TotalGroupCount = HealthyGroupCount + 1;

/* No limit argument — the default (DarlingMcpAgTools.DefaultGroupLimit) must cap the response. */
var json = await DarlingMcpAgTools.GetAgHealth(postgres, ServerName);
Assert.False(McpHelpers.IsErrorEnvelope(json), $"tool returned an error: {json}");

using var doc = JsonDocument.Parse(json);
var root = doc.RootElement;
var groups = root.GetProperty("availability_groups").EnumerateArray().ToList();

Assert.Equal(DarlingMcpAgTools.DefaultGroupLimit, groups.Count);
Assert.Equal(DarlingMcpAgTools.DefaultGroupLimit, root.GetProperty("groups_returned").GetInt32());
Assert.Equal(TotalGroupCount, root.GetProperty("groups_total").GetInt32());
Assert.True(root.GetProperty("groups_truncated").GetBoolean());
Assert.Contains("TRUNCATED", root.GetProperty("groups_truncated_note").GetString());

/* THE reason this cap exists: the one critical group is first, ahead of every healthy one. */
Assert.Equal("AG_ZZZ_CRITICAL", groups[0].GetProperty("ag_name").GetString());
Assert.Equal("Critical", groups[0].GetProperty("severity").GetString());

/* limit >= total: no cut at all. */
var uncappedJson = await DarlingMcpAgTools.GetAgHealth(postgres, ServerName, limit: TotalGroupCount);
using var uncappedDoc = JsonDocument.Parse(uncappedJson);
var uncappedRoot = uncappedDoc.RootElement;
Assert.Equal(TotalGroupCount, uncappedRoot.GetProperty("availability_groups").GetArrayLength());
Assert.False(uncappedRoot.GetProperty("groups_truncated").GetBoolean());
Assert.Equal(JsonValueKind.Null, uncappedRoot.GetProperty("groups_truncated_note").ValueKind);

bodySucceeded = true;
}
finally
{
await LiveStoreCleanup.RunAsync(cs!, bodySucceeded, async (cleanup, cleanupCt) =>
await DeleteRowsAsync(cleanup, cleanupCt));
}
}

/// <summary>
/// #4471: an out-of-range <c>limit</c> is REFUSED through the shared <see cref="McpHelpers.ValidateTop"/>,
/// never clamped or ignored, the same rule <c>get_alert_history</c> and <c>get_blocking_snapshots</c> apply
/// to their own <c>limit</c>. No topology needs to be planted: the refusal happens before any read.
/// </summary>
[Fact]
public async Task AgHealth_RefusesOutOfRangeLimit_AgainstDevPostgres()
{
var cs = ConnectionString;
Assert.SkipWhen(string.IsNullOrEmpty(cs), "Set DARLING_TEST_PG to a Postgres connection string to run the live AG-health test.");

var ct = TestContext.Current.CancellationToken;
using var connection = new NpgsqlConnection(cs);
await connection.OpenAsync(ct);
await PgMigrations.MigrateAsync(connection, ct);
await using var postgres = NpgsqlDataSource.Create(cs!);

foreach (var badLimit in new[] { 0, -1, 1001 })
{
var refused = await DarlingMcpAgTools.GetAgHealth(postgres, limit: badLimit);
Assert.True(McpHelpers.IsRefusalEnvelope(refused), $"limit {badLimit} must be refused: {refused}");
Assert.Equal("limit", JsonDocument.Parse(refused).RootElement.GetProperty("hints").GetProperty("parameter").GetString());
}

foreach (var okLimit in new[] { 1, 1000 })
{
var accepted = await DarlingMcpAgTools.GetAgHealth(postgres, limit: okLimit);
Assert.False(McpHelpers.IsErrorEnvelope(accepted), $"limit {okLimit} must be accepted: {accepted}");
}
}

private static async Task DeleteRowsAsync(NpgsqlConnection connection, System.Threading.CancellationToken ct)
{
var sql = string.Join(" ", new[] { "ag_replica_states", "ag_database_replica_states" }
Expand Down
Original file line number Diff line number Diff line change
@@ -1,3 +1,4 @@
# DarlingMcpAgTools: tools/list budget for #3898. Ceilings only go down; see McpToolsListBudgetTests. One block per tool, blank line between blocks.
tool get_ag_health 616
param get_ag_health.limit 155
param get_ag_health.server_name 118
5 changes: 4 additions & 1 deletion Darling/Darling.Tests/McpToolsListBudgetTests.cs
Original file line number Diff line number Diff line change
Expand Up @@ -85,6 +85,9 @@ public McpToolsListBudgetTests(ITestOutputHelper output)
/// here.</item>
/// </list>
/// </summary>
/* #4471: +199 bytes for get_ag_health's new limit parameter (155 bytes for its own served description;
the served head is unchanged, since the size-budget guidance for the cap moved to the tool's
<<GUIDE>> tail, which is not served in tools/list and so is not counted here). */
/* #4198/#4199 (M2b): +839 bytes for get_fleet_overview's worst_only/band filters (2 new params) and
get_collection_log's fleet-form server_name/limit descriptions, after trimming both to the D2 200-char
parameter cap and moving the rest to each tool's tail (get_tool_guide), which is not served in
Expand Down Expand Up @@ -187,7 +190,7 @@ shared 32 KB budget and is not a served description either. */
// #4452 (merge): re-measured on the tree combining dev's #4442 get_read_latency addition with this
// branch's scheduler-issues tool description growth (+95). Constant set to the value
// McpToolsListBudgetTests itself measured on the merged tree, not the two deltas added by hand.
private const int TotalCeilingBytes = 176_277;
private const int TotalCeilingBytes = 176_476;



Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -2373,7 +2373,7 @@ private static CatalogRead R(string category, string description, params Catalog
["get_daily_summary_range"] = R(CatOverview, "One daily health summary per collected day over a span of days - the Performance Calendar's month grid.", PServer(), PInt("days_back", 30), PAsOf()),
["get_fleet_overview"] = R(CatOverview, "The banded cross-server fleet roll-up.", PHours(DefaultFleetHours), PTextDefault("detail", "summary"), PBool("worst_only", false), PText("band")),
["get_sweep_reports"] = R(CatOverview, "The scheduled Fleet Sweep Reports: the sweep timeline for the window, the newest sweep in full, and the watch-item worklist - or one sweep by sweep_id (a string; the ids do not survive a JSON number round trip).", PHours(1), PAsOf(), PText("sweep_id"), PText("watch_state")),
["get_ag_health"] = R(CatOverview, "Availability Group topology: replicas and per-database secondary state.", PServer()),
["get_ag_health"] = R(CatOverview, "Availability Group topology: replicas and per-database secondary state.", PServer(), PLimit(DarlingMcpAgTools.DefaultGroupLimit)),
["get_store_metrics"] = R(CatOverview, "The monitoring store's own size/compression/growth (self-metrics): a summary by default, object_kind to list one kind, an exact object_name for one object's daily series.", PInt("days_back", 30), PText("object_kind"), PText("object_name"), PLimit(DarlingMcpStoreMetricsTools.DefaultLimit)),
["get_store_log"] = R(CatOverview, "What the monitoring store's OWN PostgreSQL server log recorded - a per-class census with the capture denominator beside it, not the lines. Deliberately unbanded.", PHours(24), PLimit(DarlingMcpStoreLogTools.DefaultRetainedLimit), PAsOf()),
["get_store_query_stats"] = R(CatOverview, "The monitoring store's OWN SQL statements ranked by server-side cost (pg_stat_statements), split by the role that ran them (on a managed store: the web viewer, MCP tools, the Darling Viewer, or the service itself).", PText("role"), PText("order_by"), PTop(DarlingMcpStoreQueryStatsTools.DefaultTop), PBool("full_text", false)),
Expand Down Expand Up @@ -3279,7 +3279,7 @@ sizing the points itself (the OptionalDouble rule). */
["get_daily_summary"] = (c, pg, an) => DarlingMcpHealthTools.GetDailySummary(pg, Server(c), Str(c, "summary_date"), c.RequestAborted),
["get_daily_summary_range"] = (c, pg, an) => DarlingMcpHealthTools.GetDailySummaryRange(pg, Server(c), QueryInt(c, "days_back", null, 30), AsOf(c), c.RequestAborted),
["get_fleet_overview"] = (c, pg, an) => DarlingMcpFleetTools.GetFleetOverview(pg, Hours(c, DefaultFleetHours), Str(c, "detail") ?? "summary", QueryBool(c, "worst_only", false), Str(c, "band"), c.RequestAborted),
["get_ag_health"] = (c, pg, an) => DarlingMcpAgTools.GetAgHealth(pg, Server(c), c.RequestAborted),
["get_ag_health"] = (c, pg, an) => DarlingMcpAgTools.GetAgHealth(pg, Server(c), Rows(c, "limit", DarlingMcpAgTools.DefaultGroupLimit), c.RequestAborted),
["get_store_metrics"] = (c, pg, an) => DarlingMcpStoreMetricsTools.GetStoreMetrics(pg, QueryInt(c, "days_back", null, 30), Str(c, "object_kind"), Str(c, "object_name"), Rows(c, "limit", DarlingMcpStoreMetricsTools.DefaultLimit), c.RequestAborted),
["get_store_log"] = (c, pg, an) => DarlingMcpStoreLogTools.GetStoreLog(pg, Hours(c, 24), Rows(c, "limit", DarlingMcpStoreLogTools.DefaultRetainedLimit), AsOf(c), c.RequestAborted),
["get_store_query_stats"] = (c, pg, an) => DarlingMcpStoreQueryStatsTools.GetStoreQueryStats(pg, Str(c, "role"), Str(c, "order_by") ?? "total_time", Rows(c, "top", DarlingMcpStoreQueryStatsTools.DefaultTop), QueryBool(c, "full_text", false), c.RequestAborted),
Expand Down
Loading
Loading