From b4691dd6d37c4d4cdb91f90357cc2120a91bfa3c Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 10:39:03 +0300 Subject: [PATCH 01/12] Fill the dashboard's alerts panel when no sink is registered The dispatcher skipped subscribing entirely when no IAlertSink was present. That was right while sinks were the only consumer of an alert; it is not right now that the dashboard reads the same history, because an application that installs alerting for the panel and configures delivery later gets a panel that can never fill. Also loads uptime for the checker the board selects on load. That selection is made by MarkLoaded rather than by a click, so it never went through the path that reads it, and the panel opened missing the columns every later selection showed. --- .../Healthie.Sample.BlazorUI.csproj | 3 + samples/Healthie.Sample.BlazorUI/Program.cs | 24 +++ .../packages.lock.json | 134 ++++++++++-- src/Healthie.AI/DiagnosisInsights.cs | 23 ++ src/Healthie.AI/StartupExtensions.cs | 6 + .../Insights/IHealthieInsights.cs | 129 ++++++++++++ src/Healthie.Alerting/AlertDispatcher.cs | 28 ++- src/Healthie.Alerting/AlertHistory.cs | 73 +++++++ src/Healthie.Alerting/HealthieAlertOptions.cs | 10 + src/Healthie.Alerting/StartupExtensions.cs | 8 + .../Components/HealthieDashboard.Insights.cs | 160 ++++++++++++++ .../Components/HealthieDashboard.razor | 111 +++++++++- .../Components/HealthieDashboard.razor.cs | 26 ++- src/Healthie.Dashboard/wwwroot/healthie.css | 162 ++++++++++++++ .../LeadershipInsights.cs | 24 +++ .../StartupExtensions.cs | 7 + .../RelationalDialect.cs | 13 +- .../RelationalStateProviderInitializer.cs | 7 +- src/Healthie.Uptime/StartupExtensions.cs | 6 + src/Healthie.Uptime/UptimeInsights.cs | 82 ++++++++ tests/Healthie.Tests.Unit/AlertingTests.cs | 45 ++++ tests/Healthie.Tests.Unit/InsightsTests.cs | 198 ++++++++++++++++++ 22 files changed, 1247 insertions(+), 32 deletions(-) create mode 100644 src/Healthie.AI/DiagnosisInsights.cs create mode 100644 src/Healthie.Abstractions/Insights/IHealthieInsights.cs create mode 100644 src/Healthie.Alerting/AlertHistory.cs create mode 100644 src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs create mode 100644 src/Healthie.LeaderElection/LeadershipInsights.cs create mode 100644 src/Healthie.Uptime/UptimeInsights.cs create mode 100644 tests/Healthie.Tests.Unit/InsightsTests.cs diff --git a/samples/Healthie.Sample.BlazorUI/Healthie.Sample.BlazorUI.csproj b/samples/Healthie.Sample.BlazorUI/Healthie.Sample.BlazorUI.csproj index 05d1a53..23caa16 100644 --- a/samples/Healthie.Sample.BlazorUI/Healthie.Sample.BlazorUI.csproj +++ b/samples/Healthie.Sample.BlazorUI/Healthie.Sample.BlazorUI.csproj @@ -13,6 +13,9 @@ + + + diff --git a/samples/Healthie.Sample.BlazorUI/Program.cs b/samples/Healthie.Sample.BlazorUI/Program.cs index 7a2b6b3..aedf66e 100644 --- a/samples/Healthie.Sample.BlazorUI/Program.cs +++ b/samples/Healthie.Sample.BlazorUI/Program.cs @@ -2,7 +2,11 @@ using Healthie.Sample.BlazorUI.Components; using Healthie.Scheduling.Quartz; using Healthie.StateProviding.CosmosDb; +using Healthie.Abstractions.Enums; +using Healthie.Alerting; using Healthie.Dashboard; +using Healthie.LeaderElection; +using Healthie.Uptime; using Microsoft.Azure.Cosmos; var builder = WebApplication.CreateBuilder(args); @@ -23,6 +27,26 @@ options.AllowMutations = builder.Configuration.GetValue("Healthie:AllowMutations", true); }); +// The dashboard shows a panel per feature package the application installs, and shows none of them +// otherwise. Installing them here is what makes this sample demonstrate the whole board rather than +// the core of it. +builder.Services + .AddHealthieAlerts(options => + { + // Suspicious as well as unhealthy, and a short window, so the alerts panel actually has + // something in it while somebody is looking at the sample. + options.MinimumSeverity = PulseCheckerHealth.Suspicious; + options.DeduplicationWindow = TimeSpan.FromSeconds(20); + }) + .AddHealthieUptime(); + +// Leader election off by default: with one replica it is always the leader, and the badge would be +// a permanent reassurance about a problem nobody has. Healthie:LeaderElection=true shows it. +if (builder.Configuration.GetValue("Healthie:LeaderElection", false)) +{ + builder.Services.AddHealthieLeaderElection(); +} + // Swap the scheduler with Healthie:Scheduler=Quartz. AddHealthie has already registered the // built-in timer, and the last registration wins, so this overrides it. if (string.Equals(builder.Configuration["Healthie:Scheduler"], "Quartz", StringComparison.OrdinalIgnoreCase)) diff --git a/samples/Healthie.Sample.BlazorUI/packages.lock.json b/samples/Healthie.Sample.BlazorUI/packages.lock.json index 4a5073c..c55b3da 100644 --- a/samples/Healthie.Sample.BlazorUI/packages.lock.json +++ b/samples/Healthie.Sample.BlazorUI/packages.lock.json @@ -33,21 +33,50 @@ "resolved": "1.1.0", "contentHash": "J2G1k+u5unBV+aYcwxo94ip16Rkp65pgWFb0R6zwJipzWNMgvqlWeuI7/+R+e8bob66LnSG+llLJ+z8wI94cHg==" }, + "Microsoft.Extensions.Configuration": { + "type": "Transitive", + "resolved": "10.0.10", + "contentHash": "plJWK2zpWuuyxI8F8s2scx6Je7N1Ajjs6HvYUGKwRnDMWIVIz9FHwAkiT7ASgrvAOd10T0FPVlh9BzAJJME+jg==", + "dependencies": { + "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", + "Microsoft.Extensions.Primitives": "10.0.10" + } + }, "Microsoft.Extensions.Configuration.Abstractions": { "type": "Transitive", - "resolved": "8.0.0", - "contentHash": "3lE/iLSutpgX1CC0NOW70FJoGARRHbyKmG7dc0klnUZ9Dd9hS6N/POPWhKhMLCEuNN5nXEY5agmlFtH562vqhQ==", + "resolved": "10.0.10", + "contentHash": "5Vnd2I75DmZCVEjSynIdJ/0EGafgnLQwgR3t2C2/fkjx/nRG+cLwxLLdInoHeCEpkD5K4Ov/g9ZCRYrl4TRsaA==", "dependencies": { - "Microsoft.Extensions.Primitives": "8.0.0" + "Microsoft.Extensions.Primitives": "10.0.10" + } + }, + "Microsoft.Extensions.Configuration.Binder": { + "type": "Transitive", + "resolved": "10.0.10", + "contentHash": "GqmN2o1CkJvk7uWp+p4CwBYW0w/zfoEbvsiFDbO2G8l1Uz+mrDAbAcZiXhU2lufKPby1cjAUdd5GTWpebYOkOA==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.10", + "Microsoft.Extensions.Configuration.Abstractions": "10.0.10" + } + }, + "Microsoft.Extensions.Diagnostics": { + "type": "Transitive", + "resolved": "10.0.10", + "contentHash": "Kr/e7lUf4+N8tacbqJ2Ctwe/HarKdAc9ZkgKVVqvtJDBKbez+T/KnUwu82KSlnBp/SrpBcxc7u7xkE2oUZT/5Q==", + "dependencies": { + "Microsoft.Extensions.Configuration": "10.0.10", + "Microsoft.Extensions.Diagnostics.Abstractions": "10.0.10", + "Microsoft.Extensions.Options.ConfigurationExtensions": "10.0.10" } }, "Microsoft.Extensions.Diagnostics.Abstractions": { "type": "Transitive", - "resolved": "8.0.1", - "contentHash": "elH2vmwNmsXuKmUeMQ4YW9ldXiF+gSGDgg1vORksob5POnpaI6caj1Hu8zaYbEuibhqCoWg0YRWDazBY3zjBfg==", + "resolved": "10.0.10", + "contentHash": "9uWiKpeOVac355STyChWR/pliFX/5CeLqChW9kKsaxyDH4EUTZxMkT4Jwp/J/peLm0GBFmSX5c0WCse3yCnq1Q==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "8.0.2", - "Microsoft.Extensions.Options": "8.0.2" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", + "Microsoft.Extensions.Options": "10.0.10", + "System.Diagnostics.DiagnosticSource": "10.0.10" } }, "Microsoft.Extensions.FileProviders.Abstractions": { @@ -58,27 +87,50 @@ "Microsoft.Extensions.Primitives": "8.0.0" } }, + "Microsoft.Extensions.Logging": { + "type": "Transitive", + "resolved": "10.0.10", + "contentHash": "Tf6z5HsL0VDYRTfvsoNrTGHGheCwkTsZBA2FFh5ATJUbkAwug+FFNISJK2gjpUNemlAOoWllAK52HOWCjto3EQ==", + "dependencies": { + "Microsoft.Extensions.DependencyInjection": "10.0.10", + "Microsoft.Extensions.Logging.Abstractions": "10.0.10", + "Microsoft.Extensions.Options": "10.0.10" + } + }, "Microsoft.Extensions.Logging.Abstractions": { "type": "Transitive", - "resolved": "8.0.3", - "contentHash": "dL0QGToTxggRLMYY4ZYX5AMwBb+byQBd/5dMiZE07Nv73o6I5Are3C7eQTh7K2+A4ct0PVISSr7TZANbiNb2yQ==", + "resolved": "10.0.10", + "contentHash": "zkFxGYUvdxAvIKTyXHrmW+Sux53D4SezD9dMyZ6hrwwzPQJNuwCRy1f5W7AvYTqacEGhWF2XderRQG1OvbV8og==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "8.0.2" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", + "System.Diagnostics.DiagnosticSource": "10.0.10" } }, "Microsoft.Extensions.Options": { "type": "Transitive", - "resolved": "8.0.2", - "contentHash": "dWGKvhFybsaZpGmzkGCbNNwBD1rVlWzrZKANLW/CcbFJpCEceMCGzT7zZwHOGBCbwM0SzBuceMj5HN1LKV1QqA==", + "resolved": "10.0.10", + "contentHash": "srnhnk7nE8krBiIXp71LvBmKBtraBONWSRzdjJgRv1Ko9Mp8IVNqv4vIS9hGeVteBig8aQkva9ZG+sC+o5sVcA==", "dependencies": { - "Microsoft.Extensions.DependencyInjection.Abstractions": "8.0.0", - "Microsoft.Extensions.Primitives": "8.0.0" + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", + "Microsoft.Extensions.Primitives": "10.0.10" + } + }, + "Microsoft.Extensions.Options.ConfigurationExtensions": { + "type": "Transitive", + "resolved": "10.0.10", + "contentHash": "tnBmu/LwF25ZQK+HBNCu2xrwnkKoB/XEbJyooGGoYxHrhvxbSKi7eOFiJ4AXBy/QU4vtCvCJfoi8k9Ej72qzOQ==", + "dependencies": { + "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", + "Microsoft.Extensions.Configuration.Binder": "10.0.10", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", + "Microsoft.Extensions.Options": "10.0.10", + "Microsoft.Extensions.Primitives": "10.0.10" } }, "Microsoft.Extensions.Primitives": { "type": "Transitive", - "resolved": "8.0.0", - "contentHash": "bXJEZrW9ny8vjMF1JV253WeLhpEVzFo1lyaZu1vQ4ZxWUlVvknZ/+ftFgVheLubb4eZPSwwxBeqS1JkCOjxd8g==" + "resolved": "10.0.10", + "contentHash": "5wu/GrYVd8mG2DVUw3vFJzF+O336TyTGg/Kmcgw9bfwYhCoFiV5lR5QeEmKecJyrW4W54nMfD3p3589E8a7czQ==" }, "Microsoft.Win32.SystemEvents": { "type": "Transitive", @@ -123,8 +175,8 @@ }, "System.Diagnostics.DiagnosticSource": { "type": "Transitive", - "resolved": "8.0.1", - "contentHash": "vaoWjvkG1aenR2XdjaVivlCV9fADfgyhW5bZtXT23qaEea0lWiUljdQuze4E31vKM7ZWJaSUsbYIKE3rnzfZUg==" + "resolved": "10.0.10", + "contentHash": "cjtKi6ERMYWp6b9UTVPcwDT29PjKDtlM3W9OwnWL5abRsI8ku42Q2wqZoLIIXJnT/XF2s2CjuK8Nl4a3mmTxQQ==" }, "System.Drawing.Common": { "type": "Transitive", @@ -217,6 +269,13 @@ "Microsoft.Extensions.Hosting.Abstractions": "[8.0.1, )" } }, + "Healthie.NET.Alerting": { + "type": "Project", + "dependencies": { + "Healthie.NET.Abstractions": "[1.0.0, )", + "Microsoft.Extensions.Http": "[10.0.10, )" + } + }, "Healthie.NET.CosmosDb": { "type": "Project", "dependencies": { @@ -238,6 +297,12 @@ "Microsoft.Extensions.Diagnostics.HealthChecks": "[8.0.29, )" } }, + "Healthie.NET.LeaderElection": { + "type": "Project", + "dependencies": { + "Healthie.NET.Abstractions": "[1.0.0, )" + } + }, "Healthie.NET.Quartz": { "type": "Project", "dependencies": { @@ -247,6 +312,12 @@ "Quartz.Extensions.Hosting": "[3.19.1, )" } }, + "Healthie.NET.Uptime": { + "type": "Project", + "dependencies": { + "Healthie.NET.Abstractions": "[1.0.0, )" + } + }, "Cronos": { "type": "CentralTransitive", "requested": "[0.13.0, )", @@ -273,11 +344,20 @@ "System.ValueTuple": "4.5.0" } }, + "Microsoft.Extensions.DependencyInjection": { + "type": "CentralTransitive", + "requested": "[8.0.1, )", + "resolved": "10.0.10", + "contentHash": "ANyvsgkNBRvcJh2XLgn8veGmajf+8m0AbKK+HPWdRL1yraSNVVSmQhFntLtdz/C795jxqqup+k05cs/3jZQPOA==", + "dependencies": { + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10" + } + }, "Microsoft.Extensions.DependencyInjection.Abstractions": { "type": "CentralTransitive", "requested": "[8.0.2, )", - "resolved": "8.0.2", - "contentHash": "3iE7UF7MQkCv1cxzCahz+Y/guQbTqieyxyaWKhrRO91itI9cOKO76OHeQDahqG4MmW5umr3CcCvGmK92lWNlbg==" + "resolved": "10.0.10", + "contentHash": "z/2xXlFw2aLGjHyEm6E0tQ+In6VfzQzTrtArbQ2c0TQE16ZbyDCMGPvaUT9I0s8rgy9sRWlU2P9waW37qV04qA==" }, "Microsoft.Extensions.Diagnostics.HealthChecks": { "type": "CentralTransitive", @@ -310,6 +390,20 @@ "Microsoft.Extensions.Logging.Abstractions": "8.0.2" } }, + "Microsoft.Extensions.Http": { + "type": "CentralTransitive", + "requested": "[10.0.10, )", + "resolved": "10.0.10", + "contentHash": "DuEBLw2y7ZBfilnaJtlge8f2M49932v43t7j1InceXVjcZzxNrSEkobYXdgZkXuPFnFbKcbWpAS+fZgYEW1BhA==", + "dependencies": { + "Microsoft.Extensions.Configuration.Abstractions": "10.0.10", + "Microsoft.Extensions.DependencyInjection.Abstractions": "10.0.10", + "Microsoft.Extensions.Diagnostics": "10.0.10", + "Microsoft.Extensions.Logging": "10.0.10", + "Microsoft.Extensions.Logging.Abstractions": "10.0.10", + "Microsoft.Extensions.Options": "10.0.10" + } + }, "Quartz.Extensions.DependencyInjection": { "type": "CentralTransitive", "requested": "[3.19.1, )", diff --git a/src/Healthie.AI/DiagnosisInsights.cs b/src/Healthie.AI/DiagnosisInsights.cs new file mode 100644 index 0000000..c1ac77e --- /dev/null +++ b/src/Healthie.AI/DiagnosisInsights.cs @@ -0,0 +1,23 @@ +using Healthie.Abstractions.Insights; + +namespace Healthie.AI; + +/// +/// Lets the dashboard ask for an explanation of why a checker has been failing. +/// +/// +/// A string rather than the full , because the board shows prose and +/// this keeps the shared contract free of this package's own types. Asked for on demand: the call +/// goes to a language model, so nothing should make it on a board that redraws every second. +/// +/// The diagnostician that talks to the model. +internal sealed class DiagnosisInsights(IPulseDiagnostician diagnostician) : IDiagnosisInsights +{ + /// + public async Task ExplainAsync(string checkerName, CancellationToken cancellationToken = default) + { + var diagnosis = await diagnostician.DiagnoseAsync(checkerName, cancellationToken).ConfigureAwait(false); + + return diagnosis.Summary; + } +} diff --git a/src/Healthie.AI/StartupExtensions.cs b/src/Healthie.AI/StartupExtensions.cs index 7323fe9..37147ea 100644 --- a/src/Healthie.AI/StartupExtensions.cs +++ b/src/Healthie.AI/StartupExtensions.cs @@ -1,3 +1,4 @@ +using Healthie.Abstractions.Insights; using Microsoft.Extensions.DependencyInjection; using Microsoft.Extensions.DependencyInjection.Extensions; @@ -37,6 +38,11 @@ public static IServiceCollection AddHealthieAI(this IServiceCollection services) services.TryAddSingleton(); + // What the dashboard offers when this package is installed: a button that asks why. Not + // referenced from there, so a dashboard does not pull a model client in with it. + services.TryAddSingleton(provider => + new DiagnosisInsights(provider.GetRequiredService())); + return services; } } diff --git a/src/Healthie.Abstractions/Insights/IHealthieInsights.cs b/src/Healthie.Abstractions/Insights/IHealthieInsights.cs new file mode 100644 index 0000000..0d51943 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/IHealthieInsights.cs @@ -0,0 +1,129 @@ +using Healthie.Abstractions.Enums; + +namespace Healthie.Abstractions.Insights; + +/// +/// The read-only views the dashboard renders when a feature package is installed. +/// +/// +/// +/// The dashboard is a Razor class library that references this package and nothing else, and that is +/// worth keeping: referencing alerting, uptime, leader election and the AI package to display them +/// would put all four on everyone who installs a dashboard. +/// +/// +/// So each feature package registers its own implementation of the small contract below, and the +/// dashboard asks the container what it has. A panel appears because a service is registered, not +/// because a flag was set, and an application that installed none of them sees the board it saw +/// before. +/// +/// +/// Reads only. Nothing here changes a checker: mutation stays on the interfaces that already own it, +/// which is what lets the dashboard show all of this in read-only mode. +/// +/// +public interface IUptimeInsights +{ + /// + /// The share of a window a checker spent healthy. + /// + /// The checker to report on. + /// How far back to look. + /// A token to monitor for cancellation requests. + /// The report, or null if nothing is recorded for that checker. + /// + /// Distinct from the percentage the board already shows, which is the share of the runs still in + /// the rolling history -- a few minutes at a fast interval. This is measured over real time and + /// survives history being trimmed. + /// + Task GetUptimeAsync( + string checkerName, + TimeSpan window, + CancellationToken cancellationToken = default); +} + +/// +/// What a checker's uptime looked like over a window. +/// +/// The share of the window spent healthy, from 0 to 100. +/// The window measured. +/// The longest unbroken unhealthy stretch, or null if there was none. +public sealed record UptimeInsight(double Percentage, TimeSpan Window, TimeSpan? LongestOutage); + +/// +/// The alerts that have been raised, and whether they are being delivered. +/// +public interface IAlertInsights +{ + /// + /// The most recent alerts, newest first. + /// + /// How many to return. + /// A token to monitor for cancellation requests. + /// The alerts, newest first. + Task> GetRecentAlertsAsync( + int limit, + CancellationToken cancellationToken = default); + + /// + /// How many alerts were dropped because the queue was full. + /// + /// + /// Worth showing rather than only logging: a dropped alert is one nobody was told about, and the + /// board is where somebody would look to find out that alerting itself is behind. + /// + int DroppedCount { get; } +} + +/// +/// One alert, as the dashboard shows it. +/// +/// The checker the alert is about. +/// That checker's display name. +/// The health before, or null if it had never run. +/// The health that raised the alert. +/// The check's own message. +/// When it was raised, in UTC. +/// Whether every sink accepted it. +public sealed record AlertInsight( + string CheckerName, + string DisplayName, + PulseCheckerHealth? PreviousHealth, + PulseCheckerHealth CurrentHealth, + string Message, + DateTime OccurredAt, + bool Delivered); + +/// +/// Whether this replica is the one running the checks. +/// +/// +/// With leader election on, only one replica runs a checker on each interval. A board that did not +/// say so would show a follower with everything idle and look broken. +/// +public interface ILeadershipInsights +{ + /// Whether this replica currently holds the lease. + bool IsLeader { get; } + + /// Something identifying this replica, for a board an operator is comparing across tabs. + string ReplicaId { get; } +} + +/// +/// An explanation of why a checker has been failing. +/// +public interface IDiagnosisInsights +{ + /// + /// Explains a checker's recent failures. + /// + /// The checker to explain. + /// A token to monitor for cancellation requests. + /// The explanation. + /// + /// Asked for rather than shown: this goes to a language model, which costs money and takes + /// seconds, so nothing should call it on a board that redraws every second. + /// + Task ExplainAsync(string checkerName, CancellationToken cancellationToken = default); +} diff --git a/src/Healthie.Alerting/AlertDispatcher.cs b/src/Healthie.Alerting/AlertDispatcher.cs index 9aee9d6..bc2da8b 100644 --- a/src/Healthie.Alerting/AlertDispatcher.cs +++ b/src/Healthie.Alerting/AlertDispatcher.cs @@ -39,17 +39,26 @@ public sealed class AlertDispatcher : BackgroundService private long _dropped; + private readonly AlertHistory? _history; + /// Initializes a new instance of the class. /// Every registered pulse checker. /// Every registered alert sink. /// Which changes alert, and how hard to try. + /// + /// Keeps the last few alerts for the dashboard. Optional, so an application without one carries + /// no history at all. + /// /// An optional logger for diagnostic output. public AlertDispatcher( IEnumerable checkers, IEnumerable sinks, HealthieAlertOptions options, + AlertHistory? history = null, ILogger? logger = null) { + _history = history; + ArgumentNullException.ThrowIfNull(checkers); ArgumentNullException.ThrowIfNull(sinks); @@ -84,11 +93,12 @@ public AlertDispatcher( /// public override Task StartAsync(CancellationToken cancellationToken) { - // Nothing to deliver to, so there is no reason to subscribe or to hold a queue. + // Still subscribes with no sink registered: the dashboard's alerts panel reads the same + // history this fills, so "nowhere to deliver" is no longer "nothing to do". if (_sinks.Count == 0) { - _logger?.LogInformation("Alerting is registered but no sink is; no alerts will be sent."); - return Task.CompletedTask; + _logger?.LogInformation( + "Alerting is registered but no sink is; alerts will show on the dashboard and be sent nowhere."); } Subscribe(); @@ -172,6 +182,8 @@ private void OnStateChanged(IPulseChecker checker, PulseCheckerStateChangedEvent /// private void OnAlertDropped(Alert alert) { + _history?.RecordDropped(); + var dropped = Interlocked.Increment(ref _dropped); _logger?.LogWarning( @@ -226,6 +238,10 @@ private bool IsWithinDeduplicationWindow(string checkerName) /// private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) { + // "Raised" and "delivered" are different facts, and the board shows both: an alert that + // fired and reached nobody is the failure worth seeing. + var delivered = true; + foreach (var sink in _sinks) { using var timeout = CancellationTokenSource.CreateLinkedTokenSource(stoppingToken); @@ -237,6 +253,8 @@ private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) } catch (OperationCanceledException) when (!stoppingToken.IsCancellationRequested) { + delivered = false; + _logger?.LogWarning( "Alert sink {Sink} did not deliver the alert for '{CheckerName}' within {Timeout}.", sink.GetType().Name, @@ -249,6 +267,8 @@ private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) } catch (Exception ex) { + delivered = false; + _logger?.LogError( ex, "Alert sink {Sink} failed to deliver the alert for '{CheckerName}'.", @@ -256,6 +276,8 @@ private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) alert.CheckerName); } } + + _history?.Record(alert, delivered); } /// diff --git a/src/Healthie.Alerting/AlertHistory.cs b/src/Healthie.Alerting/AlertHistory.cs new file mode 100644 index 0000000..ebbcdda --- /dev/null +++ b/src/Healthie.Alerting/AlertHistory.cs @@ -0,0 +1,73 @@ +using Healthie.Abstractions.Insights; + +namespace Healthie.Alerting; + +/// +/// The last few alerts, kept so the dashboard can show them. +/// +/// +/// +/// Alerting is fire-and-forget by design: an alert goes to its sinks and is gone. That is right for +/// delivery and wrong for the one screen an operator looks at, where "has anything fired recently, +/// and did it get through" is the first question. This keeps just enough to answer it. +/// +/// +/// Bounded and in memory on purpose. It is a window onto what just happened, not a record -- the +/// record is wherever the sinks deliver to. Nothing here is worth a round trip to a database or +/// worth surviving a restart. +/// +/// +/// How many alerts to keep. +public sealed class AlertHistory(int capacity) : IAlertInsights +{ + private readonly Queue _recent = new(capacity); + // A plain object, not System.Threading.Lock: this package targets net8.0 as well. + private readonly object _gate = new(); + + private int _dropped; + + /// + public int DroppedCount => Volatile.Read(ref _dropped); + + /// Records an alert and whether every sink took it. + /// The alert that was raised. + /// Whether every sink accepted it. + public void Record(Alert alert, bool delivered) + { + var insight = new AlertInsight( + alert.CheckerName, + alert.DisplayName, + alert.PreviousHealth, + alert.CurrentHealth, + alert.Message, + alert.OccurredAt, + delivered); + + lock (_gate) + { + if (_recent.Count == capacity) + { + _recent.Dequeue(); + } + + _recent.Enqueue(insight); + } + } + + /// Records that an alert never reached the queue. + public void RecordDropped() => Interlocked.Increment(ref _dropped); + + /// + public Task> GetRecentAlertsAsync( + int limit, + CancellationToken cancellationToken = default) + { + lock (_gate) + { + IReadOnlyList newestFirst = + [.. _recent.Reverse().Take(Math.Max(limit, 0))]; + + return Task.FromResult(newestFirst); + } + } +} diff --git a/src/Healthie.Alerting/HealthieAlertOptions.cs b/src/Healthie.Alerting/HealthieAlertOptions.cs index 5c7af0b..b385e79 100644 --- a/src/Healthie.Alerting/HealthieAlertOptions.cs +++ b/src/Healthie.Alerting/HealthieAlertOptions.cs @@ -55,4 +55,14 @@ public sealed class HealthieAlertOptions /// leak in the process being monitored. A dropped alert is counted and logged. /// public int QueueCapacity { get; set; } = 1024; + + /// + /// How many recent alerts the dashboard can show. Defaults to 50. + /// + /// + /// A window onto what just happened rather than a record; the record is wherever the sinks + /// deliver to. Kept in memory and bounded, so it costs nothing to leave on. + /// + public int HistoryLength { get; set; } = 50; + } diff --git a/src/Healthie.Alerting/StartupExtensions.cs b/src/Healthie.Alerting/StartupExtensions.cs index f0e7aec..3e19d09 100644 --- a/src/Healthie.Alerting/StartupExtensions.cs +++ b/src/Healthie.Alerting/StartupExtensions.cs @@ -1,3 +1,4 @@ +using Healthie.Abstractions.Insights; using Microsoft.Extensions.DependencyInjection; using Microsoft.Extensions.DependencyInjection.Extensions; using Microsoft.Extensions.Hosting; @@ -32,6 +33,13 @@ public static IServiceCollection AddHealthieAlerts( configure?.Invoke(options); services.TryAddSingleton(options); + + // What the dashboard renders when this package is installed: the last few alerts and whether + // they were delivered. Registered here rather than referenced there, so installing a + // dashboard does not drag alerting in with it. + services.TryAddSingleton(new AlertHistory(options.HistoryLength)); + services.TryAddSingleton(p => p.GetRequiredService()); + services.AddHostedService(); return services; diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs b/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs new file mode 100644 index 0000000..5d1c57f --- /dev/null +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs @@ -0,0 +1,160 @@ +using Healthie.Abstractions.Insights; +using Microsoft.AspNetCore.Components; +using Microsoft.Extensions.DependencyInjection; + +namespace Healthie.Dashboard.Components; + +/// +/// The parts of the board that appear only when a feature package is installed. +/// +/// +/// +/// Each of these is resolved from the container rather than injected, because [Inject] throws +/// when a service is absent and absent is the normal case: an application with no alerting installed +/// should get a board without an alerts panel, not an exception. +/// +/// +/// All of them read. Nothing here changes a checker, which is what lets the whole set show in +/// read-only mode -- the one exception is asking a language model to explain a failure, which is +/// still a read but costs money on someone else's account, so it is treated as an action. +/// +/// +public sealed partial class HealthieDashboard +{ + /// How far back the uptime panel looks. + /// + /// A day, because that is the window an operator asks about first and the one an overnight + /// incident falls inside. The board's own percentage covers the last few minutes. + /// + private static readonly TimeSpan UptimeWindow = TimeSpan.FromHours(24); + + [Inject] + private IServiceProvider Services { get; set; } = default!; + + private IUptimeInsights? _uptimeInsights; + private IAlertInsights? _alertInsights; + private ILeadershipInsights? _leadershipInsights; + private IDiagnosisInsights? _diagnosisInsights; + + private UptimeInsight? _uptime; + private string? _uptimeFor; + + private IReadOnlyList _recentAlerts = []; + private bool _alertsOpen; + + private string? _diagnosis; + private bool _diagnosing; + + /// Whether anything installed has something to add to the board. + private bool HasInsights => + _uptimeInsights is not null + || _alertInsights is not null + || _leadershipInsights is not null + || _diagnosisInsights is not null; + + /// Whether this replica is the one running the checks. + /// + /// True when nothing is elected: a single replica runs everything, and saying "follower" there + /// would be a warning about a situation that does not exist. + /// + private bool IsLeader => _leadershipInsights?.IsLeader ?? true; + + /// Picks up whatever feature packages the application installed. + private void ResolveInsights() + { + _uptimeInsights = Services.GetService(); + _alertInsights = Services.GetService(); + _leadershipInsights = Services.GetService(); + _diagnosisInsights = Services.GetService(); + } + + /// Reads the uptime for the selected checker. + /// + /// Only on selection, not on the clock tick: it is a query over recorded segments, and the board + /// redraws every second. + /// + private async Task LoadUptimeAsync(string? checkerName) + { + if (_uptimeInsights is null || checkerName is null) + { + _uptime = null; + _uptimeFor = null; + return; + } + + if (_uptimeFor == checkerName) + { + return; + } + + _uptimeFor = checkerName; + _uptime = await _uptimeInsights.GetUptimeAsync(checkerName, UptimeWindow); + } + + /// Reads the recent alerts, newest first. + private async Task LoadAlertsAsync() + { + if (_alertInsights is null) + { + return; + } + + _recentAlerts = await _alertInsights.GetRecentAlertsAsync(AlertsShown); + } + + /// How many alerts the panel lists. + private const int AlertsShown = 20; + + private async Task ToggleAlertsAsync() + { + _alertsOpen = !_alertsOpen; + + if (_alertsOpen) + { + await LoadAlertsAsync(); + } + } + + /// Asks the model why a checker has been failing. + /// + /// Guarded against a second click while the first is in flight: the call takes seconds and costs + /// money, and a button that looks unresponsive invites exactly that. + /// + private async Task ExplainAsync(string checkerName) + { + if (_diagnosisInsights is null || _diagnosing) + { + return; + } + + _diagnosing = true; + _diagnosis = null; + + try + { + _diagnosis = await _diagnosisInsights.ExplainAsync(checkerName); + } + catch (Exception ex) + { + // Shown rather than swallowed: a model that is misconfigured or out of quota should say + // so on the board, not leave a button that appears to do nothing. + _diagnosis = $"Could not explain this: {ex.Message}"; + } + finally + { + _diagnosing = false; + } + } + + /// Formats an uptime percentage the way the rest of the board formats one. + private static string Percent(double value) => $"{value:0.##}%"; + + /// A short, human length: "4m", "2h 10m". + private static string Duration(TimeSpan span) => span switch + { + { TotalSeconds: < 60 } => $"{span.TotalSeconds:0}s", + { TotalMinutes: < 60 } => $"{span.TotalMinutes:0}m", + { TotalHours: < 24 } => $"{(int)span.TotalHours}h {span.Minutes}m", + _ => $"{(int)span.TotalDays}d {span.Hours}h", + }; +} diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor b/src/Healthie.Dashboard/Components/HealthieDashboard.razor index 92b5b0e..a76e675 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor @@ -51,6 +51,34 @@ value="@_searchFilter" @oninput="OnSearchInput" /> + @* + Only when something is elected. A single-replica application is always the leader and + does not need telling. + *@ + @if (_leadershipInsights is not null) + { + + @(IsLeader ? "LEADER" : "FOLLOWER") + + } + + @if (_alertInsights is not null) + { + + } +
@if (Options.AllowMutations) { @@ -105,6 +133,52 @@
+ @* + Reads only, so it shows in read-only mode too: knowing what fired and whether it landed + is exactly what a read-only board is for. + *@ + @if (_alertsOpen && _alertInsights is not null) + { +
+
+ RECENT ALERTS + @if (_alertInsights.DroppedCount > 0) + { + + @_alertInsights.DroppedCount DROPPED + + } +
+ +
+ + @if (_recentAlerts.Count == 0) + { +
Nothing has alerted yet. Alerts appear here as checkers change health.
+ } + else + { +
    + @foreach (var alert in _recentAlerts) + { +
  • + @Relative(alert.OccurredAt) + @alert.DisplayName + + @(alert.PreviousHealth?.ToString() ?? "new") → @alert.CurrentHealth + + @alert.Message + @if (!alert.Delivered) + { + NOT DELIVERED + } +
  • + } +
+ } +
+ } + @* The event log at full size. The sidebar copy is a glance; this is the one to read when @@ -236,7 +310,7 @@
UPTIME
@Uptime(_selected!)
+ @* Beside the run-based figure above, so the two are read against each other. *@ + @if (_uptime is { } uptime) + { +
+
24H
+
@Percent(uptime.Percentage)
+
+ @if (uptime.LongestOutage is { } outage) + { +
+
WORST
+
@Duration(outage)
+
+ } + }
FAILS
@@ -457,6 +546,26 @@ legible elsewhere: the interval is the rate on the row, the threshold is the denominator in FAILS, and the group and tags are chips on the row. *@ + @* + Explaining is a read, but it spends money on the host's account and takes + seconds, so it sits with the actions rather than with the readings. + *@ + @if (_diagnosisInsights is not null && Options.AllowMutations) + { +
+ + @if (_diagnosis is { } diagnosis) + { +

@diagnosis

+ } +
+ } + @if (Options.AllowMutations) {
diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs b/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs index 8f9cec4..728f6a5 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs @@ -169,6 +169,9 @@ protected override async Task OnInitializedAsync() _initialized = true; _isDarkMode = ThemeState.IsDarkMode; + // Whatever feature packages the application installed. Absent is the normal case. + ResolveInsights(); + // Hand the states the prerender read across to the interactive render. Without this the // circuit starts from an empty board and reads every checker again; the board it renders in // the meantime -- momentarily empty, then repopulating -- replaces the one the prerender @@ -194,6 +197,10 @@ protected override async Task OnInitializedAsync() await LoadAsync(); } + // The first checker is selected by MarkLoaded rather than by a click, so nothing has read its + // uptime yet -- without this the panel opens missing the columns every later selection shows. + await LoadUptimeAsync(_selected); + _clockLoop = RunClockAsync(); } @@ -434,14 +441,29 @@ private void OnTagFilterChanged(ChangeEventArgs args) Refresh(); } - private void OnRowKeyDown(KeyboardEventArgs args, string name) + private async Task OnRowKeyDown(KeyboardEventArgs args, string name) { if (args.Key is "Enter" or " ") { - _selected = name; + await SelectAsync(name); } } + /// + /// Selects a checker and reads anything the feature packages can add about it. + /// + /// + /// On selection rather than on the clock tick: uptime is a query over recorded segments and the + /// board redraws every second. + /// + private async Task SelectAsync(string? name) + { + _selected = name; + _diagnosis = null; + + await LoadUptimeAsync(name); + } + private void ToggleAbout() => _showAbout = !_showAbout; private void ToggleLog() => _showLog = !_showLog; diff --git a/src/Healthie.Dashboard/wwwroot/healthie.css b/src/Healthie.Dashboard/wwwroot/healthie.css index ed9565e..fd9ae42 100644 --- a/src/Healthie.Dashboard/wwwroot/healthie.css +++ b/src/Healthie.Dashboard/wwwroot/healthie.css @@ -1537,3 +1537,165 @@ from { opacity: 0; transform: translate(-50%, calc(-50% - 6px)); } to { opacity: 1; transform: translate(-50%, -50%); } } + +/* --------------------------------------------------------------------------------------------- + Panels that appear only when a feature package is installed: leadership, alerts, and the + explanation. Built from the same variables as everything above, so they inherit both themes and + need no colours of their own. + --------------------------------------------------------------------------------------------- */ + +.hpm-replica { + padding: 4px 9px; + border: 1px solid var(--hpm-line2); + border-radius: 999px; + font-family: var(--hpm-mono); + font-size: 10.5px; + font-weight: 700; + letter-spacing: 0.08em; + white-space: nowrap; +} + +.hpm-replica--leader { + color: var(--hpm-ok); + border-color: var(--hpm-ok); + background: var(--hpm-ok-bg); +} + +/* A follower is not a fault, so it reads as information rather than as a warning. */ +.hpm-replica--follower { + color: var(--hpm-faint); +} + +.hpm-btn--alerts-dropped { + color: var(--hpm-crit); + border-color: var(--hpm-crit); +} + +.hpm-alerts { + border-bottom: 1px solid var(--hpm-line); + background: var(--hpm-panel); +} + +.hpm-alerts-head { + display: flex; + align-items: center; + gap: 10px; + padding: 10px 16px; + border-bottom: 1px solid var(--hpm-line); +} + +.hpm-alerts-title { + font-family: var(--hpm-mono); + font-size: 11px; + font-weight: 700; + letter-spacing: 0.08em; + color: var(--hpm-faint); +} + +.hpm-alerts-dropped { + font-family: var(--hpm-mono); + font-size: 11px; + font-weight: 700; + color: var(--hpm-crit); +} + +.hpm-alerts-empty { + padding: 16px; + color: var(--hpm-faint); + font-size: 13px; +} + +.hpm-alerts-list { + margin: 0; + padding: 0; + list-style: none; + max-height: 260px; + overflow-y: auto; +} + +.hpm-alert { + display: grid; + grid-template-columns: 72px 160px 150px 1fr auto; + gap: 12px; + align-items: baseline; + padding: 8px 16px; + border-bottom: 1px solid var(--hpm-line); + font-size: 12.5px; +} + +.hpm-alert:last-child { + border-bottom: 0; +} + +.hpm-alert-when, +.hpm-alert-move { + font-family: var(--hpm-mono); + font-size: 11.5px; + color: var(--hpm-faint); +} + +.hpm-alert-name { + font-weight: 600; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.hpm-alert-msg { + color: var(--hpm-faint); + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +/* An alert that fired and reached nobody is the one worth seeing from across the room. */ +.hpm-alert--undelivered { + background: var(--hpm-crit-bg, rgba(255, 90, 100, 0.08)); +} + +.hpm-alert-flag { + font-family: var(--hpm-mono); + font-size: 10px; + font-weight: 700; + letter-spacing: 0.06em; + color: var(--hpm-crit); + white-space: nowrap; +} + +.hpm-explain { + margin: 12px 0 4px; +} + +.hpm-sel-btn--explain { + border-color: var(--hpm-line2); +} + +.hpm-sel-btn--explain[disabled] { + opacity: 0.6; + cursor: default; +} + +.hpm-explain-text { + margin: 10px 0 0; + padding: 10px 12px; + border-left: 2px solid var(--hpm-line2); + background: var(--hpm-panel2); + color: var(--hpm-text); + font-size: 13px; + line-height: 1.5; + white-space: pre-wrap; +} + +/* The alert row is five columns of fixed intent; below the detail breakpoint it stacks instead of + squeezing every one of them to nothing. */ +@media (max-width: 860px) { + .hpm-alert { + grid-template-columns: 1fr; + gap: 4px; + } + + .hpm-alert-msg, + .hpm-alert-name { + white-space: normal; + } +} diff --git a/src/Healthie.LeaderElection/LeadershipInsights.cs b/src/Healthie.LeaderElection/LeadershipInsights.cs new file mode 100644 index 0000000..61302b5 --- /dev/null +++ b/src/Healthie.LeaderElection/LeadershipInsights.cs @@ -0,0 +1,24 @@ +using Healthie.Abstractions.Insights; + +namespace Healthie.LeaderElection; + +/// +/// Tells the dashboard whether this replica is the one running the checks. +/// +/// +/// Without it, a board served by a follower shows every checker sitting still and reads as broken: +/// nothing is running here, and nothing is meant to be. The replica id is there because an operator +/// comparing two tabs needs to know which is which. +/// +/// The scheduler that holds, or does not hold, the lease. +/// Where the replica's own identity comes from. +internal sealed class LeadershipInsights( + LeaderElectedPulseScheduler scheduler, + LeaderElectionOptions options) : ILeadershipInsights +{ + /// + public bool IsLeader => scheduler.IsLeader; + + /// + public string ReplicaId => options.HolderId; +} diff --git a/src/Healthie.LeaderElection/StartupExtensions.cs b/src/Healthie.LeaderElection/StartupExtensions.cs index 8f0f9c8..ad0acd4 100644 --- a/src/Healthie.LeaderElection/StartupExtensions.cs +++ b/src/Healthie.LeaderElection/StartupExtensions.cs @@ -1,4 +1,5 @@ using Healthie.Abstractions.Scheduling; +using Healthie.Abstractions.Insights; using Microsoft.Extensions.DependencyInjection; using Microsoft.Extensions.DependencyInjection.Extensions; @@ -52,6 +53,12 @@ public static IServiceCollection AddHealthieLeaderElection( services.AddSingleton(provider => provider.GetRequiredService()); services.AddHostedService(); + // So a board served by a follower says so, instead of showing everything idle and looking + // broken. + services.TryAddSingleton(provider => new LeadershipInsights( + provider.GetRequiredService(), + provider.GetRequiredService())); + return services; } diff --git a/src/Healthie.StateProviding.Relational/RelationalDialect.cs b/src/Healthie.StateProviding.Relational/RelationalDialect.cs index 0f5da4e..8bcae47 100644 --- a/src/Healthie.StateProviding.Relational/RelationalDialect.cs +++ b/src/Healthie.StateProviding.Relational/RelationalDialect.cs @@ -29,7 +29,9 @@ namespace Healthie.StateProviding.Relational; /// /// /// Statement adding the version column to a table that predates it, with {0} for the table -/// name. Run only when the column is missing. +/// name. Run only when the column is missing. Optional: a dialect that does not supply one is left +/// alone by the migration, which is right for a table this library created and wrong only for one +/// that predates versioning, where the statement is the whole point. /// /// /// Statement inserting one row only if the name is not taken, reporting the outcome through rows @@ -40,7 +42,7 @@ public sealed record RelationalDialect( string Name, string CreateTableFormat, string UpsertFormat, - string AddVersionColumnFormat, + string? AddVersionColumnFormat = null, string? InsertIfAbsentFormat = null) { /// @@ -199,8 +201,11 @@ internal static void ValidateTableName(string tableName) /// A plain ALTER, run only when the column is genuinely missing -- the initializer checks first /// rather than relying on an IF NOT EXISTS that SQLite does not have for ADD COLUMN. /// - internal string AddVersionColumn(string tableName) => - Format(AddVersionColumnFormat, tableName); + /// + /// The statement adding the version column, or null when the dialect does not supply one. + /// + internal string? AddVersionColumn(string tableName) => + AddVersionColumnFormat is null ? null : Format(AddVersionColumnFormat, tableName); internal static string Select(string tableName) => Format(SelectFormat, tableName); diff --git a/src/Healthie.StateProviding.Relational/RelationalStateProviderInitializer.cs b/src/Healthie.StateProviding.Relational/RelationalStateProviderInitializer.cs index 3396501..00ce165 100644 --- a/src/Healthie.StateProviding.Relational/RelationalStateProviderInitializer.cs +++ b/src/Healthie.StateProviding.Relational/RelationalStateProviderInitializer.cs @@ -26,7 +26,7 @@ public sealed class RelationalStateProviderInitializer( private readonly string _createTableSql = (dialect ?? throw new ArgumentNullException(nameof(dialect))) .CreateTable(Validated(tableName)); - private readonly string _addVersionColumnSql = dialect.AddVersionColumn(Validated(tableName)); + private readonly string? _addVersionColumnSql = dialect.AddVersionColumn(Validated(tableName)); private readonly string _tableName = Validated(tableName); private static string Validated(string tableName) @@ -73,7 +73,10 @@ public async Task InitializeAsync(CancellationToken cancellationToken = default) /// private async Task AddVersionColumnIfMissingAsync(DbConnection connection, CancellationToken cancellationToken) { - if (await HasVersionColumnAsync(connection, cancellationToken).ConfigureAwait(false)) + // Nothing to add, and nothing to add it with: a hand-built dialect that supplied no + // statement is telling us its table already has the column, or that it is not our business. + if (_addVersionColumnSql is null + || await HasVersionColumnAsync(connection, cancellationToken).ConfigureAwait(false)) { return; } diff --git a/src/Healthie.Uptime/StartupExtensions.cs b/src/Healthie.Uptime/StartupExtensions.cs index e6fdf84..2d46192 100644 --- a/src/Healthie.Uptime/StartupExtensions.cs +++ b/src/Healthie.Uptime/StartupExtensions.cs @@ -1,3 +1,4 @@ +using Healthie.Abstractions.Insights; using Microsoft.Extensions.DependencyInjection; using Microsoft.Extensions.DependencyInjection.Extensions; @@ -25,6 +26,11 @@ public static IServiceCollection AddHealthieUptime(this IServiceCollection servi services.TryAddSingleton(); services.AddHostedService(); + // What the dashboard renders when this package is installed. Registered here rather than + // referenced there, so installing a dashboard does not drag uptime in with it. + services.TryAddSingleton(provider => + new UptimeInsights(provider.GetRequiredService())); + return services; } } diff --git a/src/Healthie.Uptime/UptimeInsights.cs b/src/Healthie.Uptime/UptimeInsights.cs new file mode 100644 index 0000000..5f31772 --- /dev/null +++ b/src/Healthie.Uptime/UptimeInsights.cs @@ -0,0 +1,82 @@ +using Healthie.Abstractions.Enums; +using Healthie.Abstractions.Insights; + +namespace Healthie.Uptime; + +/// +/// Reports this package's uptime to the dashboard, without the dashboard having to reference it. +/// +/// +/// The board already shows a percentage of its own: the share of the runs still in the rolling +/// history, which at a one-second interval is the last couple of minutes. This one is measured over +/// real time and survives history being trimmed, which is what makes it worth showing separately +/// rather than replacing the other. +/// +internal sealed class UptimeInsights(IUptimeStore store, TimeProvider? timeProvider = null) : IUptimeInsights +{ + private readonly TimeProvider _time = timeProvider ?? TimeProvider.System; + + /// + public async Task GetUptimeAsync( + string checkerName, + TimeSpan window, + CancellationToken cancellationToken = default) + { + ArgumentException.ThrowIfNullOrWhiteSpace(checkerName); + + var to = _time.GetUtcNow().UtcDateTime; + var from = to - window; + + var segments = await store.GetSegmentsAsync(checkerName, from, to, cancellationToken) + .ConfigureAwait(false); + + if (segments.Count == 0) + { + return null; + } + + var report = UptimeCalculator.Calculate(checkerName, segments, from, to); + + // Null when nothing was observed, which is not the same as nothing being recorded: the + // dashboard shows the difference rather than printing a confident zero. + if (report.UptimePercentage is not { } percentage) + { + return null; + } + + return new UptimeInsight(percentage, window, LongestOutage(segments, from, to)); + } + + /// + /// The longest unbroken unhealthy stretch inside the window. + /// + /// + /// A percentage alone cannot tell a hundred one-second blips from one long outage, and those are + /// very different mornings. Segments are clipped to the window so a stretch that started before + /// it is counted only for the part that falls inside. + /// + private static TimeSpan? LongestOutage(IReadOnlyList segments, DateTime from, DateTime to) + { + TimeSpan? longest = null; + + foreach (var segment in segments.Where(s => s.Health == PulseCheckerHealth.Unhealthy)) + { + var start = segment.StartedAt < from ? from : segment.StartedAt; + var end = segment.EndedAt is { } ended && ended < to ? ended : to; + + if (end <= start) + { + continue; + } + + var length = end - start; + + if (longest is null || length > longest) + { + longest = length; + } + } + + return longest; + } +} diff --git a/tests/Healthie.Tests.Unit/AlertingTests.cs b/tests/Healthie.Tests.Unit/AlertingTests.cs index 6a223d9..f4899df 100644 --- a/tests/Healthie.Tests.Unit/AlertingTests.cs +++ b/tests/Healthie.Tests.Unit/AlertingTests.cs @@ -1,4 +1,5 @@ using Healthie.Abstractions.Enums; +using Healthie.Abstractions.Insights; using Healthie.Alerting; using Microsoft.Extensions.Hosting; @@ -98,6 +99,50 @@ public async Task AFailure_ReachesTheSink() } } + /// + /// An application can install alerting for the dashboard's panel alone and wire up delivery + /// later. The dispatcher used to skip subscribing when no sink was registered -- correct while + /// sinks were the only consumer, and the reason that panel could never fill once it was one. + /// + [Fact] + public async Task WithNoSinkRegistered_AlertsStillReachTheHistoryTheDashboardReads() + { + var checker = new FakePulseChecker("alerting-target"); + var history = new AlertHistory(capacity: 4); + var dispatcher = new AlertDispatcher( + [checker], + [], + new HealthieAlertOptions { DeduplicationWindow = TimeSpan.Zero }, + history); + + await ((IHostedService)dispatcher).StartAsync(Ct); + + try + { + Assert.Equal(1, checker.SubscriberCount); + checker.RaiseStateChanged(PulseCheckerHealth.Unhealthy); + + IReadOnlyList recent = []; + var deadline = DateTime.UtcNow.AddSeconds(5); + + while (recent.Count == 0 && DateTime.UtcNow < deadline) + { + recent = await history.GetRecentAlertsAsync(10, Ct); + + if (recent.Count == 0) + { + await Task.Delay(20, Ct); + } + } + + Assert.Equal(PulseCheckerHealth.Unhealthy, Assert.Single(recent).CurrentHealth); + } + finally + { + await ((IHostedService)dispatcher).StopAsync(CancellationToken.None); + } + } + /// /// A change carrying no result says nothing about health, so it must not alert. This is the one /// branch of the health-change test that the other tests here cannot reach: every other way of diff --git a/tests/Healthie.Tests.Unit/InsightsTests.cs b/tests/Healthie.Tests.Unit/InsightsTests.cs new file mode 100644 index 0000000..21af29a --- /dev/null +++ b/tests/Healthie.Tests.Unit/InsightsTests.cs @@ -0,0 +1,198 @@ +using Healthie.Abstractions.Enums; +using Healthie.Abstractions.Insights; +using Healthie.Alerting; +using Healthie.DependencyInjection; +using Healthie.LeaderElection; +using Healthie.Uptime; +using Microsoft.Extensions.DependencyInjection; + +namespace Healthie.Tests.Unit; + +/// +/// The read-only views the dashboard renders when a feature package is installed. +/// +/// +/// The point of the contract is that the dashboard shows a panel because a service is registered, +/// not because a flag was set -- so what is asserted here is mostly presence and absence, and that +/// nothing appears in an application that installed nothing. +/// +public class InsightsTests +{ + private static CancellationToken Ct => TestContext.Current.CancellationToken; + + private static ServiceProvider Build(Action configure) + { + var services = new ServiceCollection(); + services.AddHealthie(typeof(InsightsTests).Assembly); + configure(services); + + return services.BuildServiceProvider(); + } + + /// + /// The case that matters most: an application with none of the feature packages must resolve + /// none of the contracts, so the board renders exactly what it rendered before. + /// + [Fact] + public void WithNoFeaturePackages_NoInsightsAreRegistered() + { + using var provider = Build(_ => { }); + + Assert.Null(provider.GetService()); + Assert.Null(provider.GetService()); + Assert.Null(provider.GetService()); + Assert.Null(provider.GetService()); + } + + [Fact] + public void AddHealthieUptime_RegistersUptimeInsightsAndNothingElse() + { + using var provider = Build(services => services.AddHealthieUptime()); + + Assert.NotNull(provider.GetService()); + Assert.Null(provider.GetService()); + Assert.Null(provider.GetService()); + } + + [Fact] + public void AddHealthieAlerts_RegistersAlertInsights() + { + using var provider = Build(services => services.AddHealthieAlerts()); + + Assert.NotNull(provider.GetService()); + Assert.Null(provider.GetService()); + } + + [Fact] + public void AddHealthieLeaderElection_RegistersLeadershipInsights() + { + using var provider = Build(services => services.AddHealthieLeaderElection()); + + var leadership = provider.GetService(); + + Assert.NotNull(leadership); + Assert.False(string.IsNullOrWhiteSpace(leadership.ReplicaId)); + } + + /// + /// Nothing recorded is not the same as nothing observed, and neither is a zero. A checker that + /// has never run reports no uptime rather than a confident 0%. + /// + [Fact] + public async Task UptimeInsights_ForACheckerThatNeverRan_ReportsNothing() + { + using var provider = Build(services => services.AddHealthieUptime()); + + var uptime = provider.GetRequiredService(); + + Assert.Null(await uptime.GetUptimeAsync("never-ran", TimeSpan.FromHours(24), Ct)); + } + + [Fact] + public async Task UptimeInsights_ReportsTheShareOfTheWindowSpentHealthy() + { + using var provider = Build(services => services.AddHealthieUptime()); + + var store = provider.GetRequiredService(); + var now = DateTime.UtcNow; + + // Transitions, which is how the recorder writes them: each one closes the segment before it. + // Healthy for an hour, then unhealthy for the hour up to now, inside a four-hour window. + await store.RecordAsync("split", PulseCheckerHealth.Healthy, now.AddHours(-2), Ct); + await store.RecordAsync("split", PulseCheckerHealth.Unhealthy, now.AddHours(-1), Ct); + + var uptime = await provider.GetRequiredService() + .GetUptimeAsync("split", TimeSpan.FromHours(4), Ct); + + Assert.NotNull(uptime); + + // Half of the observed time, not of the window: the two hours nothing was recorded are time + // nobody was watching, and counting them either way would be a claim about nothing. + Assert.Equal(50, uptime.Percentage, 0); + + Assert.NotNull(uptime.LongestOutage); + Assert.Equal(1, uptime.LongestOutage.Value.TotalHours, 1); + } + + /// + /// A percentage cannot tell a hundred blips from one long outage, which is why the longest + /// stretch is reported beside it. + /// + [Fact] + public async Task UptimeInsights_ReportsTheLongestOutageNotTheTotal() + { + using var provider = Build(services => services.AddHealthieUptime()); + + var store = provider.GetRequiredService(); + var now = DateTime.UtcNow; + + await store.RecordAsync("blips", PulseCheckerHealth.Unhealthy, now.AddMinutes(-50), Ct); + await store.RecordAsync("blips", PulseCheckerHealth.Healthy, now.AddMinutes(-45), Ct); + await store.RecordAsync("blips", PulseCheckerHealth.Unhealthy, now.AddMinutes(-30), Ct); + await store.RecordAsync("blips", PulseCheckerHealth.Healthy, now.AddMinutes(-10), Ct); + + var uptime = await provider.GetRequiredService() + .GetUptimeAsync("blips", TimeSpan.FromHours(1), Ct); + + // 25 minutes unhealthy in total, but the longest single stretch is 20. + Assert.Equal(20, uptime!.LongestOutage!.Value.TotalMinutes, 1); + } + + [Fact] + public async Task AlertHistory_KeepsTheNewestAndDropsTheOldest() + { + var history = new AlertHistory(capacity: 2); + + foreach (var i in Enumerable.Range(1, 3)) + { + history.Record(Alert($"checker-{i}"), delivered: true); + } + + var recent = await history.GetRecentAlertsAsync(10, Ct); + + Assert.Equal(2, recent.Count); + Assert.Equal("checker-3", recent[0].CheckerName); + Assert.Equal("checker-2", recent[1].CheckerName); + } + + /// + /// An alert that fired and reached nobody is the failure worth seeing, so it is recorded as + /// raised-but-undelivered rather than not recorded at all. + /// + [Fact] + public async Task AlertHistory_RemembersWhetherAnAlertWasDelivered() + { + var history = new AlertHistory(capacity: 4); + + history.Record(Alert("delivered"), delivered: true); + history.Record(Alert("failed"), delivered: false); + + var recent = await history.GetRecentAlertsAsync(10, Ct); + + Assert.False(recent[0].Delivered); + Assert.True(recent[1].Delivered); + } + + [Fact] + public void AlertHistory_CountsWhatNeverReachedTheQueue() + { + var history = new AlertHistory(capacity: 4); + + Assert.Equal(0, history.DroppedCount); + + history.RecordDropped(); + history.RecordDropped(); + + Assert.Equal(2, history.DroppedCount); + } + + private static Alert Alert(string name) => new( + name, + name, + Group: null, + Tags: [], + PulseCheckerHealth.Healthy, + PulseCheckerHealth.Unhealthy, + "down", + DateTime.UtcNow); +} From 6d29c34e9d666631e6c0c8d031a0cf7067cc5930 Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 10:51:55 +0300 Subject: [PATCH 02/12] Release notes and docs for 4.0.0 Turns the Unreleased section into 4.0.0, reordered to Keep a Changelog's Added/Fixed/Security and with the two Fixed blocks merged. The intro says plainly that the major number is about the size of the release rather than a break in it, because every entry under it is additive. Documents the panels the board grows per feature package, in both the repository README and the dashboard package's own, including which of them survive read-only mode and why EXPLAIN does not. --- CHANGELOG.md | 80 ++++++++++++++++++++------------ README.md | 6 ++- src/Healthie.Dashboard/README.md | 22 +++++++++ 3 files changed, 76 insertions(+), 32 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 10f1de6..2ea72fc 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,37 +7,16 @@ and this project adheres to [Semantic Versioning](https://semver.org/). ## [Unreleased] -Eleven new packages, and the schedule model that several of them needed. Everything here is -additive: no public API was removed or changed, and an application that upgrades without touching -its code behaves exactly as it did. +## [4.0.0] - 2026-07-30 -### Fixed +Eleven new packages, the schedule model several of them needed, and optimistic concurrency on the +state contract. -- **The dashboard's own page did not encode its title.** `MapHealthieUI` builds that one page as a - string rather than through Razor, which encodes every interpolation for you, so a - `HealthieUIOptions.DashboardTitle` built from anything the host did not write itself could close - the title element and open a script one. -- **A dashboard component left its handler behind when it was disposed.** The service is scoped to a - circuit, so it was assumed the circuit ending released everything; that holds only while the - dashboard is mounted once per circuit, and a host that routes to it inside its own layout builds a - new one every time the user navigates back. `IHealthieDashboardService.UnsubscribeFromStateChangesAsync` - is the missing half, and the component calls it. -- **Two callers scheduling one checker at once could leave a timer nobody could stop.** Installing a - schedule was "cancel the old one, then start the new one", and only the last one stored was - reachable; the other kept triggering, could not be unscheduled, and held a linked - `CancellationTokenSource` that was never disposed. -- **`HealthieMcpOptions.MaxHistoryPageSize` was documented and never read.** `get_check_history` - clamped to a hard-coded 200 instead, so a host that lowered the option got no such thing. -- **The relational table-name guard admitted a name with a trailing newline.** In .NET `$` matches - immediately before one, so a table name ending in one passed a check whose own error message says - it does not. Nothing could be smuggled through it -- a lone trailing newline is whitespace to - every engine here -- but the guard now ends at `\z` and means what it says. -- **The release workflow granted `contents: write` to every job in it.** Only the one job that - creates the GitHub release needs it, and the top level now grants nothing but read, so a job added - later starts with no write access rather than inheriting it. -- **A checker name from a REST route reached the log with its control characters intact.** The - not-found branch logs a name precisely when it matches nothing, percent-encoded CR and LF arrive - decoded, and a log sink writing plain text writes them as line breaks. +**The major number is about the size of the release, not about a break in it.** Everything here is +additive: nothing public was removed, renamed or changed shape, every new interface member is a +defaulted one, and an application that upgrades without touching its code behaves as it did. The one +thing to read before upgrading is what the relational providers do to an existing table on startup, +under *Optimistic concurrency* below. ### Added @@ -128,8 +107,48 @@ its code behaves exactly as it did. - **State removal.** `IStateProvider.DeleteStateAsync` cleans up after a checker that was renamed or removed. Defaulted to refuse rather than to silently do nothing. +- **The dashboard shows what the feature packages know.** Installing one used to change what the + application did and nothing about the one screen an operator looks at, so uptime, alerts and + leadership were only visible to whatever the host wired up itself. Each now has a small read-only + contract in `Healthie.NET.Abstractions` that its package implements, and the board renders the + panel when the container can resolve it: uptime over the last day and the longest outage in it + beside the run-based percentage, a drawer of recent alerts saying which reached their sinks and + which did not, a badge naming the replica when it is not the one running the checks, and a button + that asks the model why a checker has been failing. An application that installs none of them gets + the board exactly as it was. + + Read-only throughout, so all of it shows under `HealthieUIOptions.AllowMutations = false` -- the + one exception is asking the model, which is still a read but spends money on the host's account, + and is gated with the controls that change things. + ### Fixed +- **The dashboard's own page did not encode its title.** `MapHealthieUI` builds that one page as a + string rather than through Razor, which encodes every interpolation for you, so a + `HealthieUIOptions.DashboardTitle` built from anything the host did not write itself could close + the title element and open a script one. +- **A dashboard component left its handler behind when it was disposed.** The service is scoped to a + circuit, so it was assumed the circuit ending released everything; that holds only while the + dashboard is mounted once per circuit, and a host that routes to it inside its own layout builds a + new one every time the user navigates back. `IHealthieDashboardService.UnsubscribeFromStateChangesAsync` + is the missing half, and the component calls it. +- **Two callers scheduling one checker at once could leave a timer nobody could stop.** Installing a + schedule was "cancel the old one, then start the new one", and only the last one stored was + reachable; the other kept triggering, could not be unscheduled, and held a linked + `CancellationTokenSource` that was never disposed. +- **`HealthieMcpOptions.MaxHistoryPageSize` was documented and never read.** `get_check_history` + clamped to a hard-coded 200 instead, so a host that lowered the option got no such thing. +- **The relational table-name guard admitted a name with a trailing newline.** In .NET `$` matches + immediately before one, so a table name ending in one passed a check whose own error message says + it does not. Nothing could be smuggled through it -- a lone trailing newline is whitespace to + every engine here -- but the guard now ends at `\z` and means what it says. +- **The release workflow granted `contents: write` to every job in it.** Only the one job that + creates the GitHub release needs it, and the top level now grants nothing but read, so a job added + later starts with no write access rather than inheriting it. +- **A checker name from a REST route reached the log with its control characters intact.** The + not-found branch logs a name precisely when it matches nothing, percent-encoded CR and LF arrive + decoded, and a log sink writing plain text writes them as line breaks. + - A cron schedule whose next occurrence was more than about fifty days out **stopped the checker permanently and silently**. `Task.Delay` refuses a longer wait and throws, and that throw is not a cancellation, so the scheduler's loop ended with nothing logged. An annual certificate check @@ -432,6 +451,7 @@ Blazor dashboard, and the CosmosDB and Quartz.NET providers. - Dashboard UI improvements and additional sample pulse checkers. -[Unreleased]: https://github.com/ivanvyd/Healthie.NET/compare/v3.1.4...HEAD +[Unreleased]: https://github.com/ivanvyd/Healthie.NET/compare/v4.0.0...HEAD +[4.0.0]: https://github.com/ivanvyd/Healthie.NET/compare/v3.1.4...v4.0.0 [3.0.0]: https://github.com/ivanvyd/Healthie.NET/compare/v2.3.0...v3.0.0 [2.3.0]: https://github.com/ivanvyd/Healthie.NET/releases/tag/v2.3.0 diff --git a/README.md b/README.md index 2af2d89..a08971d 100644 --- a/README.md +++ b/README.md @@ -446,6 +446,7 @@ The `Healthie.NET.Dashboard` package is a pulse monitor for your services, shipp - **Aggregate pulse trace** -- an EKG across the header, with the combined checks-per-minute of everything you monitor. - **A row per checker** -- status light, name, checks per minute, and a pulse strip of the last N runs, one blip per run, coloured by the health it reported. - **Detail panel** -- select a checker for its uptime, failure streak, last message, and its interval, failure threshold, group and tags, all editable in place. +- **A panel per feature package you install** -- 24-hour uptime and longest outage from `Healthie.NET.Uptime`, a recent-alerts drawer from `Healthie.NET.Alerting`, a replica badge from `Healthie.NET.LeaderElection`, and an `EXPLAIN` button from `Healthie.NET.AI`. Nothing to configure: the panel appears because the package is registered, and an application that installs none of them gets the board unchanged. All read-only, so all of it survives `AllowMutations = false` -- bar `EXPLAIN`, which spends money on your account. - **Groups and tags** -- a checker sits in at most one group and carries any number of tags. Sections the list by group, filters it by tag. Both are declared in code and can be changed here; see [Groups and tags](#groups-and-tags). - **Pin** -- keep the checkers worth watching at the top. A pin is shared, not personal. - **Rows or cards** -- the same list laid out either way, flat or sectioned by group with per-group healthy/suspicious/failing tallies. @@ -822,11 +823,12 @@ Upgrading from v1.x? See the [v1 to v2 migration guide](https://github.com/ivanv ## Roadmap -Shipped since 3.1.4: alerting on transitions, OpenTelemetry metrics and traces, arbitrary intervals +Shipped in 4.0.0: alerting on transitions, OpenTelemetry metrics and traces, arbitrary intervals and cron, PostgreSQL / SQL Server / SQLite state providers, Hangfire / Coravel / Temporal scheduling, ready-made checkers, uptime reporting, leader election, optimistic concurrency on `IStateProvider`, `HealthChanged` on the state-changed event, Slack / Teams / PagerDuty alert -sinks, and a Redis state provider. +sinks, a Redis state provider, and a dashboard that surfaces each of those packages as it is +installed. Every item that was on this list has shipped. Two open questions are decisions rather than features, and both are deliberate as they stand: diff --git a/src/Healthie.Dashboard/README.md b/src/Healthie.Dashboard/README.md index 6259032..4529263 100644 --- a/src/Healthie.Dashboard/README.md +++ b/src/Healthie.Dashboard/README.md @@ -60,6 +60,7 @@ app.MapHealthieUI().RequireAuthorization("AdminPolicy"); // With auth - Per-checker management: start, stop, trigger, reset, change interval, change threshold - Bulk actions: Start All, Stop All, Trigger All - A read-only mode that reports everything and changes nothing — see below +- Panels for the feature packages you install — uptime, alerts, leadership, AI — see below - Groups and tags, both editable here and seeded from code — see below - Pin a checker to the top of the list - Rows or cards, flat or sectioned by group with per-group tallies @@ -99,6 +100,27 @@ builder.Services.AddHealthieUI(options => options.AllowMutations = false); app.MapHealthieUI().RequireAuthorization(); ``` +## Panels for the packages you install + +The board grows a panel for each feature package the application registers, and shows none of them +otherwise. There is nothing to switch on: it renders the panel when the container can resolve the +contract, so installing the package is the whole configuration. + +| Install | What appears | +|---|---| +| `Healthie.NET.Uptime` | `24H` on the selected checker — uptime measured over real time — and `WORST`, the longest unbroken outage inside that window | +| `Healthie.NET.Alerting` | An `ALERTS` button opening the recent alerts, each with its health transition, its message, and a flag when it did not reach every sink | +| `Healthie.NET.LeaderElection` | A badge naming this replica, shown only when it is *not* the one running the checks — a board where nothing is moving is otherwise indistinguishable from a broken one | +| `Healthie.NET.AI` | An `EXPLAIN` button on a failing checker, which asks your `IChatClient` why it has been failing | + +`24H` sits beside the board's own `UPTIME`, which is the share of the runs still in the rolling +history — a hundred results, so at a one-second interval, the last hundred seconds. They answer +different questions and disagree for good reasons, which is why both are shown. + +All of it reads and none of it writes, so it all stays under `AllowMutations = false`. The exception +is `EXPLAIN`: still a read, but it spends money on your account, so it is gated with the controls +that change things. + ## Groups and tags The two look similar and answer different questions. From c0171cddf4a00eabeb82ab5237a791310e77abdc Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 11:20:05 +0300 Subject: [PATCH 03/12] Deslop pass over the v4 diff A zero-capacity alert history threw on the first alert rather than discarding it: trimming ran before the enqueue, so an empty queue was the case on every call. Trimming after the enqueue removes the special case instead of guarding it, and HealthieAlertOptions.HistoryLength now clamps to a minimum of one the way MaxHistoryLength does. The rest is tidying the review found: the six insight contracts split one per file, since IHealthieInsights.cs was the only file in Abstractions holding more than one public type and no type of that name existed; a doc block that ended up with two summaries; an unused HasInsights property; the alerts markup indented a level deeper than its siblings; a CSS variable fallback no other call site carries and no theme leaves undefined; and _history grouped with the other injected fields rather than after an unrelated counter. --- .../Insights/AlertInsight.cs | 26 ++++ .../Insights/IAlertInsights.cs | 30 ++++ .../Insights/IDiagnosisInsights.cs | 23 ++++ .../Insights/IHealthieInsights.cs | 129 ------------------ .../Insights/ILeadershipInsights.cs | 18 +++ .../Insights/IUptimeInsights.cs | 38 ++++++ .../Insights/UptimeInsight.cs | 13 ++ src/Healthie.Alerting/AlertDispatcher.cs | 6 +- src/Healthie.Alerting/AlertHistory.cs | 8 +- src/Healthie.Alerting/HealthieAlertOptions.cs | 14 +- .../Components/HealthieDashboard.Insights.cs | 7 - .../Components/HealthieDashboard.razor | 87 ++++++------ src/Healthie.Dashboard/wwwroot/healthie.css | 2 +- .../RelationalDialect.cs | 6 +- tests/Healthie.Tests.Unit/InsightsTests.cs | 26 ++++ 15 files changed, 237 insertions(+), 196 deletions(-) create mode 100644 src/Healthie.Abstractions/Insights/AlertInsight.cs create mode 100644 src/Healthie.Abstractions/Insights/IAlertInsights.cs create mode 100644 src/Healthie.Abstractions/Insights/IDiagnosisInsights.cs delete mode 100644 src/Healthie.Abstractions/Insights/IHealthieInsights.cs create mode 100644 src/Healthie.Abstractions/Insights/ILeadershipInsights.cs create mode 100644 src/Healthie.Abstractions/Insights/IUptimeInsights.cs create mode 100644 src/Healthie.Abstractions/Insights/UptimeInsight.cs diff --git a/src/Healthie.Abstractions/Insights/AlertInsight.cs b/src/Healthie.Abstractions/Insights/AlertInsight.cs new file mode 100644 index 0000000..806b884 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/AlertInsight.cs @@ -0,0 +1,26 @@ +using Healthie.Abstractions.Enums; + +namespace Healthie.Abstractions.Insights; + +/// +/// One alert, as the dashboard shows it. +/// +/// The checker the alert is about. +/// That checker's display name. +/// The health before, or null if it had never run. +/// The health that raised the alert. +/// The check's own message. +/// When it was raised, in UTC. +/// Whether every sink accepted it. +/// +/// Raised and delivered are separate facts, and both are carried: an alert that fired and reached +/// nobody is the failure worth seeing on the board. +/// +public sealed record AlertInsight( + string CheckerName, + string DisplayName, + PulseCheckerHealth? PreviousHealth, + PulseCheckerHealth CurrentHealth, + string Message, + DateTime OccurredAt, + bool Delivered); diff --git a/src/Healthie.Abstractions/Insights/IAlertInsights.cs b/src/Healthie.Abstractions/Insights/IAlertInsights.cs new file mode 100644 index 0000000..68c2cd6 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/IAlertInsights.cs @@ -0,0 +1,30 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// The alerts that have been raised, and whether they are being delivered. +/// +/// +/// Declared here rather than in the alerting package, so the dashboard can render the panel without +/// referencing it -- see for why that matters. Reads only. +/// +public interface IAlertInsights +{ + /// + /// The most recent alerts, newest first. + /// + /// How many to return. + /// A token to monitor for cancellation requests. + /// The alerts, newest first. + Task> GetRecentAlertsAsync( + int limit, + CancellationToken cancellationToken = default); + + /// + /// How many alerts were dropped because the queue was full. + /// + /// + /// Worth showing rather than only logging: a dropped alert is one nobody was told about, and the + /// board is where somebody would look to find out that alerting itself is behind. + /// + int DroppedCount { get; } +} diff --git a/src/Healthie.Abstractions/Insights/IDiagnosisInsights.cs b/src/Healthie.Abstractions/Insights/IDiagnosisInsights.cs new file mode 100644 index 0000000..dbfbdf7 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/IDiagnosisInsights.cs @@ -0,0 +1,23 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// An explanation of why a checker has been failing. +/// +/// +/// Declared here rather than in the AI package, so the dashboard can offer the button without +/// referencing it -- see for why that matters. +/// +public interface IDiagnosisInsights +{ + /// + /// Explains a checker's recent failures. + /// + /// The checker to explain. + /// A token to monitor for cancellation requests. + /// The explanation. + /// + /// Asked for rather than shown: this goes to a language model, which costs money and takes + /// seconds, so nothing should call it on a board that redraws every second. + /// + Task ExplainAsync(string checkerName, CancellationToken cancellationToken = default); +} diff --git a/src/Healthie.Abstractions/Insights/IHealthieInsights.cs b/src/Healthie.Abstractions/Insights/IHealthieInsights.cs deleted file mode 100644 index 0d51943..0000000 --- a/src/Healthie.Abstractions/Insights/IHealthieInsights.cs +++ /dev/null @@ -1,129 +0,0 @@ -using Healthie.Abstractions.Enums; - -namespace Healthie.Abstractions.Insights; - -/// -/// The read-only views the dashboard renders when a feature package is installed. -/// -/// -/// -/// The dashboard is a Razor class library that references this package and nothing else, and that is -/// worth keeping: referencing alerting, uptime, leader election and the AI package to display them -/// would put all four on everyone who installs a dashboard. -/// -/// -/// So each feature package registers its own implementation of the small contract below, and the -/// dashboard asks the container what it has. A panel appears because a service is registered, not -/// because a flag was set, and an application that installed none of them sees the board it saw -/// before. -/// -/// -/// Reads only. Nothing here changes a checker: mutation stays on the interfaces that already own it, -/// which is what lets the dashboard show all of this in read-only mode. -/// -/// -public interface IUptimeInsights -{ - /// - /// The share of a window a checker spent healthy. - /// - /// The checker to report on. - /// How far back to look. - /// A token to monitor for cancellation requests. - /// The report, or null if nothing is recorded for that checker. - /// - /// Distinct from the percentage the board already shows, which is the share of the runs still in - /// the rolling history -- a few minutes at a fast interval. This is measured over real time and - /// survives history being trimmed. - /// - Task GetUptimeAsync( - string checkerName, - TimeSpan window, - CancellationToken cancellationToken = default); -} - -/// -/// What a checker's uptime looked like over a window. -/// -/// The share of the window spent healthy, from 0 to 100. -/// The window measured. -/// The longest unbroken unhealthy stretch, or null if there was none. -public sealed record UptimeInsight(double Percentage, TimeSpan Window, TimeSpan? LongestOutage); - -/// -/// The alerts that have been raised, and whether they are being delivered. -/// -public interface IAlertInsights -{ - /// - /// The most recent alerts, newest first. - /// - /// How many to return. - /// A token to monitor for cancellation requests. - /// The alerts, newest first. - Task> GetRecentAlertsAsync( - int limit, - CancellationToken cancellationToken = default); - - /// - /// How many alerts were dropped because the queue was full. - /// - /// - /// Worth showing rather than only logging: a dropped alert is one nobody was told about, and the - /// board is where somebody would look to find out that alerting itself is behind. - /// - int DroppedCount { get; } -} - -/// -/// One alert, as the dashboard shows it. -/// -/// The checker the alert is about. -/// That checker's display name. -/// The health before, or null if it had never run. -/// The health that raised the alert. -/// The check's own message. -/// When it was raised, in UTC. -/// Whether every sink accepted it. -public sealed record AlertInsight( - string CheckerName, - string DisplayName, - PulseCheckerHealth? PreviousHealth, - PulseCheckerHealth CurrentHealth, - string Message, - DateTime OccurredAt, - bool Delivered); - -/// -/// Whether this replica is the one running the checks. -/// -/// -/// With leader election on, only one replica runs a checker on each interval. A board that did not -/// say so would show a follower with everything idle and look broken. -/// -public interface ILeadershipInsights -{ - /// Whether this replica currently holds the lease. - bool IsLeader { get; } - - /// Something identifying this replica, for a board an operator is comparing across tabs. - string ReplicaId { get; } -} - -/// -/// An explanation of why a checker has been failing. -/// -public interface IDiagnosisInsights -{ - /// - /// Explains a checker's recent failures. - /// - /// The checker to explain. - /// A token to monitor for cancellation requests. - /// The explanation. - /// - /// Asked for rather than shown: this goes to a language model, which costs money and takes - /// seconds, so nothing should call it on a board that redraws every second. - /// - Task ExplainAsync(string checkerName, CancellationToken cancellationToken = default); -} diff --git a/src/Healthie.Abstractions/Insights/ILeadershipInsights.cs b/src/Healthie.Abstractions/Insights/ILeadershipInsights.cs new file mode 100644 index 0000000..528e3c3 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/ILeadershipInsights.cs @@ -0,0 +1,18 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// Whether this replica is the one running the checks. +/// +/// +/// With leader election on, only one replica runs a checker on each interval. A board that did not +/// say so would show a follower with everything idle and look broken. Declared here rather than in +/// the leader-election package -- see for why. Reads only. +/// +public interface ILeadershipInsights +{ + /// Whether this replica currently holds the lease. + bool IsLeader { get; } + + /// Something identifying this replica, for a board an operator is comparing across tabs. + string ReplicaId { get; } +} diff --git a/src/Healthie.Abstractions/Insights/IUptimeInsights.cs b/src/Healthie.Abstractions/Insights/IUptimeInsights.cs new file mode 100644 index 0000000..fb0bdb4 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/IUptimeInsights.cs @@ -0,0 +1,38 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// How much of a window a checker spent healthy, for the dashboard to show. +/// +/// +/// +/// Declared here rather than in the package that implements it, so the dashboard can render the +/// panel without referencing that package. The dashboard is a Razor class library that references +/// this one and nothing else, and referencing alerting, uptime, leader election and the AI package +/// to display them would put all four on everyone who installs a dashboard. The board asks the +/// container what it has: a panel appears because a service is registered, not because a flag was +/// set. +/// +/// +/// Reads only, as every contract in this namespace does. Mutation stays on the interfaces that +/// already own it, which is what lets the dashboard show all of this in read-only mode. +/// +/// +public interface IUptimeInsights +{ + /// + /// The share of a window a checker spent healthy. + /// + /// The checker to report on. + /// How far back to look. + /// A token to monitor for cancellation requests. + /// The report, or null if nothing is recorded for that checker. + /// + /// Distinct from the percentage the board already shows, which is the share of the runs still in + /// the rolling history -- a few minutes at a fast interval. This is measured over real time and + /// survives history being trimmed. + /// + Task GetUptimeAsync( + string checkerName, + TimeSpan window, + CancellationToken cancellationToken = default); +} diff --git a/src/Healthie.Abstractions/Insights/UptimeInsight.cs b/src/Healthie.Abstractions/Insights/UptimeInsight.cs new file mode 100644 index 0000000..67a0955 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/UptimeInsight.cs @@ -0,0 +1,13 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// What a checker's uptime looked like over a window. +/// +/// The share of the window spent healthy, from 0 to 100. +/// The window measured. +/// The longest unbroken unhealthy stretch, or null if there was none. +/// +/// The longest outage is carried beside the percentage because a percentage alone cannot tell a +/// hundred one-second blips from one long outage, and those are very different mornings. +/// +public sealed record UptimeInsight(double Percentage, TimeSpan Window, TimeSpan? LongestOutage); diff --git a/src/Healthie.Alerting/AlertDispatcher.cs b/src/Healthie.Alerting/AlertDispatcher.cs index bc2da8b..61df573 100644 --- a/src/Healthie.Alerting/AlertDispatcher.cs +++ b/src/Healthie.Alerting/AlertDispatcher.cs @@ -31,6 +31,7 @@ public sealed class AlertDispatcher : BackgroundService private readonly IReadOnlyList _checkers; private readonly IReadOnlyList _sinks; private readonly HealthieAlertOptions _options; + private readonly AlertHistory? _history; private readonly ILogger? _logger; private readonly Channel _queue; @@ -39,8 +40,6 @@ public sealed class AlertDispatcher : BackgroundService private long _dropped; - private readonly AlertHistory? _history; - /// Initializes a new instance of the class. /// Every registered pulse checker. /// Every registered alert sink. @@ -57,14 +56,13 @@ public AlertDispatcher( AlertHistory? history = null, ILogger? logger = null) { - _history = history; - ArgumentNullException.ThrowIfNull(checkers); ArgumentNullException.ThrowIfNull(sinks); _checkers = [.. checkers]; _sinks = [.. sinks]; _options = options ?? throw new ArgumentNullException(nameof(options)); + _history = history; _logger = logger; _queue = Channel.CreateBounded( diff --git a/src/Healthie.Alerting/AlertHistory.cs b/src/Healthie.Alerting/AlertHistory.cs index ebbcdda..1eda498 100644 --- a/src/Healthie.Alerting/AlertHistory.cs +++ b/src/Healthie.Alerting/AlertHistory.cs @@ -45,12 +45,14 @@ public void Record(Alert alert, bool delivered) lock (_gate) { - if (_recent.Count == capacity) + // Trims after enqueuing rather than before. Dropping the oldest first has to special-case + // an empty queue, and a capacity of zero makes every call the empty case. + _recent.Enqueue(insight); + + while (_recent.Count > capacity) { _recent.Dequeue(); } - - _recent.Enqueue(insight); } } diff --git a/src/Healthie.Alerting/HealthieAlertOptions.cs b/src/Healthie.Alerting/HealthieAlertOptions.cs index b385e79..4049a12 100644 --- a/src/Healthie.Alerting/HealthieAlertOptions.cs +++ b/src/Healthie.Alerting/HealthieAlertOptions.cs @@ -56,13 +56,19 @@ public sealed class HealthieAlertOptions /// public int QueueCapacity { get; set; } = 1024; + private int _historyLength = 50; + /// - /// How many recent alerts the dashboard can show. Defaults to 50. + /// How many recent alerts the dashboard can show. Defaults to 50, minimum 1. /// /// /// A window onto what just happened rather than a record; the record is wherever the sinks - /// deliver to. Kept in memory and bounded, so it costs nothing to leave on. + /// deliver to. Kept in memory and bounded, so it costs nothing to leave on. Clamped rather than + /// rejected, as MaxHistoryLength is. /// - public int HistoryLength { get; set; } = 50; - + public int HistoryLength + { + get => _historyLength; + set => _historyLength = Math.Max(value, 1); + } } diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs b/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs index 5d1c57f..351cc14 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs @@ -45,13 +45,6 @@ public sealed partial class HealthieDashboard private string? _diagnosis; private bool _diagnosing; - /// Whether anything installed has something to add to the board. - private bool HasInsights => - _uptimeInsights is not null - || _alertInsights is not null - || _leadershipInsights is not null - || _diagnosisInsights is not null; - /// Whether this replica is the one running the checks. /// /// True when nothing is elected: a single replica runs everything, and saying "follower" there diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor b/src/Healthie.Dashboard/Components/HealthieDashboard.razor index a76e675..edf3c5f 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor @@ -1,4 +1,4 @@ -@* Healthie.NET dashboard -- Pulse Monitor. *@ +@* Healthie.NET dashboard -- Pulse Monitor. *@ @implements IAsyncDisposable @using Healthie.Abstractions.Extensions @using Healthie.Dashboard.Components @@ -133,52 +133,51 @@
- @* - Reads only, so it shows in read-only mode too: knowing what fired and whether it landed - is exactly what a read-only board is for. - *@ - @if (_alertsOpen && _alertInsights is not null) - { -
-
- RECENT ALERTS - @if (_alertInsights.DroppedCount > 0) - { - - @_alertInsights.DroppedCount DROPPED - - } -
- -
- - @if (_recentAlerts.Count == 0) - { -
Nothing has alerted yet. Alerts appear here as checkers change health.
- } - else + @* + Reads only, so it shows in read-only mode too: knowing what fired and whether it landed + is exactly what a read-only board is for. + *@ + @if (_alertsOpen && _alertInsights is not null) + { +
+
+ RECENT ALERTS + @if (_alertInsights.DroppedCount > 0) { -
    - @foreach (var alert in _recentAlerts) - { -
  • - @Relative(alert.OccurredAt) - @alert.DisplayName - - @(alert.PreviousHealth?.ToString() ?? "new") → @alert.CurrentHealth - - @alert.Message - @if (!alert.Delivered) - { - NOT DELIVERED - } -
  • - } -
+ + @_alertInsights.DroppedCount DROPPED + } -
- } +
+ +
+ @if (_recentAlerts.Count == 0) + { +
Nothing has alerted yet. Alerts appear here as checkers change health.
+ } + else + { +
    + @foreach (var alert in _recentAlerts) + { +
  • + @Relative(alert.OccurredAt) + @alert.DisplayName + + @(alert.PreviousHealth?.ToString() ?? "new") → @alert.CurrentHealth + + @alert.Message + @if (!alert.Delivered) + { + NOT DELIVERED + } +
  • + } +
+ } + + } @* The event log at full size. The sidebar copy is a glance; this is the one to read when diff --git a/src/Healthie.Dashboard/wwwroot/healthie.css b/src/Healthie.Dashboard/wwwroot/healthie.css index fd9ae42..0c544d7 100644 --- a/src/Healthie.Dashboard/wwwroot/healthie.css +++ b/src/Healthie.Dashboard/wwwroot/healthie.css @@ -1650,7 +1650,7 @@ /* An alert that fired and reached nobody is the one worth seeing from across the room. */ .hpm-alert--undelivered { - background: var(--hpm-crit-bg, rgba(255, 90, 100, 0.08)); + background: var(--hpm-crit-bg); } .hpm-alert-flag { diff --git a/src/Healthie.StateProviding.Relational/RelationalDialect.cs b/src/Healthie.StateProviding.Relational/RelationalDialect.cs index 8bcae47..8e17c04 100644 --- a/src/Healthie.StateProviding.Relational/RelationalDialect.cs +++ b/src/Healthie.StateProviding.Relational/RelationalDialect.cs @@ -195,15 +195,13 @@ internal static void ValidateTableName(string tableName) internal const string DeleteFormat = "DELETE FROM {0} WHERE name = @name"; /// - /// Adds the version column to a table created before it existed. + /// Adds the version column to a table created before it existed, or null when the + /// dialect does not supply the statement. /// /// /// A plain ALTER, run only when the column is genuinely missing -- the initializer checks first /// rather than relying on an IF NOT EXISTS that SQLite does not have for ADD COLUMN. /// - /// - /// The statement adding the version column, or null when the dialect does not supply one. - /// internal string? AddVersionColumn(string tableName) => AddVersionColumnFormat is null ? null : Format(AddVersionColumnFormat, tableName); diff --git a/tests/Healthie.Tests.Unit/InsightsTests.cs b/tests/Healthie.Tests.Unit/InsightsTests.cs index 21af29a..b5479f9 100644 --- a/tests/Healthie.Tests.Unit/InsightsTests.cs +++ b/tests/Healthie.Tests.Unit/InsightsTests.cs @@ -173,6 +173,32 @@ public async Task AlertHistory_RemembersWhetherAnAlertWasDelivered() Assert.True(recent[1].Delivered); } + /// + /// Trimming used to happen before the enqueue, which made a capacity of zero the empty-queue + /// case on every single call -- so the first alert raised took down the dispatcher's delivery + /// loop rather than being discarded. + /// + [Fact] + public async Task AlertHistory_WithNoRoomAtAll_DiscardsRatherThanThrows() + { + var history = new AlertHistory(capacity: 0); + + history.Record(Alert("nowhere"), delivered: true); + + Assert.Empty(await history.GetRecentAlertsAsync(10, Ct)); + } + + /// + /// And the supported route there cannot reach zero: the option clamps, as MaxHistoryLength does, + /// so a board configured with no history shows the last alert rather than nothing at all. + /// + [Fact] + public void HistoryLength_BelowOne_IsClamped() + { + Assert.Equal(1, new HealthieAlertOptions { HistoryLength = 0 }.HistoryLength); + Assert.Equal(1, new HealthieAlertOptions { HistoryLength = -5 }.HistoryLength); + } + [Fact] public void AlertHistory_CountsWhatNeverReachedTheQueue() { From 522db806d0d61638cc0953343894a6000b3a5a11 Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 11:28:37 +0300 Subject: [PATCH 04/12] Correct the package count in the 4.0.0 notes Twelve packages are new since v3.1.4, not eleven -- counted against the tag rather than from the list in the section, which had drifted. --- CHANGELOG.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 2ea72fc..5a25bd4 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -9,7 +9,7 @@ and this project adheres to [Semantic Versioning](https://semver.org/). ## [4.0.0] - 2026-07-30 -Eleven new packages, the schedule model several of them needed, and optimistic concurrency on the +Twelve new packages, the schedule model several of them needed, and optimistic concurrency on the state contract. **The major number is about the size of the release, not about a break in it.** Everything here is From 6a59014a3c43419675feed7167ae19b950e9d914 Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 11:31:28 +0300 Subject: [PATCH 05/12] Say what actually breaks in 4.0.0 The intro claimed nothing public changed shape. Diffing the public API of the eight pre-3.1.4 packages against the tag says otherwise: HealthieTools' constructor gained a defaulted HealthieMcpOptions parameter when it was fixed to read MaxHistoryPageSize. Source- compatible, binary-breaking, and worth naming -- it is the reason the major number is honest rather than only a size claim. --- CHANGELOG.md | 16 +++++++++++----- 1 file changed, 11 insertions(+), 5 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 5a25bd4..9ad427b 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -12,11 +12,17 @@ and this project adheres to [Semantic Versioning](https://semver.org/). Twelve new packages, the schedule model several of them needed, and optimistic concurrency on the state contract. -**The major number is about the size of the release, not about a break in it.** Everything here is -additive: nothing public was removed, renamed or changed shape, every new interface member is a -defaulted one, and an application that upgrades without touching its code behaves as it did. The one -thing to read before upgrading is what the relational providers do to an existing table on startup, -under *Optimistic concurrency* below. +**Source-compatible: an application that upgrades without touching its code still compiles and still +behaves as it did.** Nothing public was removed or renamed, and every new interface member is a +defaulted one, so a provider or scheduler written against 3.x keeps working. Two things to know +before upgrading: + +- **One binary break.** `HealthieTools`, the MCP read-only tool class, gained a defaulted + `HealthieMcpOptions?` parameter on its constructor so it could read the page-size option it had + been ignoring. Calling code still compiles unchanged; an assembly compiled against 3.1.4 and not + rebuilt does not. Recompiling is enough, and that break is why this is a major rather than a minor. +- **The relational providers alter an existing table on startup** to add the version column + optimistic concurrency needs. See *Optimistic concurrency* below. ### Added From fd1804a2ef75b35176c8f4003523dc467ce62aeb Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 11:34:47 +0300 Subject: [PATCH 06/12] Assert the detail panel's stats by name, not by count The sample installs the uptime package, so the panel grew a 24H cell and the hard-coded count of three failed. A count was the wrong assertion anyway: 24H appears once a segment exists and WORST only after an outage, both of which depend on how long the app has been up. Asserts the three that are always there instead, and that nothing unexpected joins them -- scoped to the stats grid, since the same label class dresses the GROUP and TAGS editors below it. --- tests/Healthie.Tests.E2E/DashboardTests.cs | 14 +++++++++++++- 1 file changed, 13 insertions(+), 1 deletion(-) diff --git a/tests/Healthie.Tests.E2E/DashboardTests.cs b/tests/Healthie.Tests.E2E/DashboardTests.cs index e379f0c..8d766a6 100644 --- a/tests/Healthie.Tests.E2E/DashboardTests.cs +++ b/tests/Healthie.Tests.E2E/DashboardTests.cs @@ -101,7 +101,19 @@ public async Task SelectingAChecker_ShowsItsDetail(ProviderSetup setup) await page.Locator(".hpm-sel-name", new() { HasTextString = TargetChecker }) .WaitForAsync(new() { Timeout = 10_000 }); Assert.Equal(TargetChecker, (await page.Locator(".hpm-sel-name").TextContentAsync())?.Trim()); - Assert.Equal(3, await page.Locator(".hpm-stat").CountAsync()); + + // A count would be a timing assertion, not a detail one: the sample installs the uptime + // package, whose 24H cell appears once a segment has been recorded and whose WORST cell + // appears only after an outage. So: the three that are always there, and nothing unexpected. + // Scoped to the stats grid: the same label class dresses the GROUP and TAGS editors below it. + var labels = await page.Locator(".hpm-stats .hpm-stat-label").AllTextContentsAsync(); + var trimmed = labels.Select(label => label.Trim()).ToList(); + + Assert.Contains("UPTIME", trimmed); + Assert.Contains("FAILS", trimmed); + Assert.Contains("STATE", trimmed); + Assert.Empty(trimmed.Except(["UPTIME", "FAILS", "STATE", "24H", "WORST"])); + browser.AssertNoErrors(page); } From e77c20900ae14d9766063231a446f727318cd0d7 Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 11:54:17 +0300 Subject: [PATCH 07/12] Describe the replica badge as it actually renders Both docs said it appears only on a follower. It appears whenever the leader-election package is installed, reading LEADER or FOLLOWER and naming the replica on hover -- which is the more useful behaviour when the package is opt-in and installed precisely because there is more than one replica. Caught by running a leader-elected instance rather than by reading the markup. --- CHANGELOG.md | 2 +- src/Healthie.Dashboard/README.md | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 9ad427b..1c19a62 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -119,7 +119,7 @@ before upgrading: contract in `Healthie.NET.Abstractions` that its package implements, and the board renders the panel when the container can resolve it: uptime over the last day and the longest outage in it beside the run-based percentage, a drawer of recent alerts saying which reached their sinks and - which did not, a badge naming the replica when it is not the one running the checks, and a button + which did not, a badge saying whether this replica is the one running the checks, and a button that asks the model why a checker has been failing. An application that installs none of them gets the board exactly as it was. diff --git a/src/Healthie.Dashboard/README.md b/src/Healthie.Dashboard/README.md index 4529263..03db1b8 100644 --- a/src/Healthie.Dashboard/README.md +++ b/src/Healthie.Dashboard/README.md @@ -110,7 +110,7 @@ contract, so installing the package is the whole configuration. |---|---| | `Healthie.NET.Uptime` | `24H` on the selected checker — uptime measured over real time — and `WORST`, the longest unbroken outage inside that window | | `Healthie.NET.Alerting` | An `ALERTS` button opening the recent alerts, each with its health transition, its message, and a flag when it did not reach every sink | -| `Healthie.NET.LeaderElection` | A badge naming this replica, shown only when it is *not* the one running the checks — a board where nothing is moving is otherwise indistinguishable from a broken one | +| `Healthie.NET.LeaderElection` | A `LEADER` or `FOLLOWER` badge, hover-naming the replica — on a follower every checker sits still, which is otherwise indistinguishable from a broken board | | `Healthie.NET.AI` | An `EXPLAIN` button on a failing checker, which asks your `IChatClient` why it has been failing | `24H` sits beside the board's own `UPTIME`, which is the share of the runs still in the rolling From d7c6e9485e77e8c4a38961b2d5e45649b32ff5ec Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 12:32:45 +0300 Subject: [PATCH 08/12] Group by default, surface schedules, and add a side menu Grouping is on when the board opens. A group is a partition, so the sectioned view answers what is wrong and where without the reader doing it in their head; checkers with no group collect under one heading rather than vanishing. The stale tag-era names went with it -- _groupByTags and TagGroup partitioned on Group, which is the exact confusion CLAUDE.md warns about, preserved in the identifiers. The board ignored PulseSchedule entirely. It read state.Interval for the rate column, the aggregate CHECKS/MIN and the interval picker, and Interval is documented as ignored once Schedule is set -- so a cron checker advertised a rate it was not running at, was summed into the aggregate at that rate, and offered a picker that stored a field nothing reads. It now reads EffectiveSchedule, shows a cron expression as one, counts cron checkers separately rather than guessing a rate for them, and disables the picker while an expression is in force. A cron expression can also be set from the board. The scheduler judges it before anything is stored, through a new defaulted IPulseScheduler.TryValidateSchedule, because Cronos, Quartz and Temporal do not agree on dialects and the only answer worth having is from the implementation that will run it. SetIntervalAsync now clears the schedule, so choosing an interval takes effect. The side menu is a rail, not a drawer: this package ships no JavaScript, and an overlay drawer needs a focus trap to be honest about keyboard use. Collapsed it keeps its icons and its toggle; below 820px it becomes a scrolling strip above the list. --- .../Pulses/CertificateExpiryPulseChecker.cs | 7 +- src/Healthie.Abstractions/IPulseChecker.cs | 22 ++ src/Healthie.Abstractions/PulseChecker.cs | 39 ++- .../Scheduling/IPulseScheduler.cs | 24 ++ .../Scheduling/IPulsesScheduler.cs | 17 ++ .../Scheduling/PulsesScheduler.cs | 16 ++ .../Controllers/HealthCheckersController.cs | 59 +++++ .../Components/HealthieDashboard.razor | 145 +++++++++-- .../Components/HealthieDashboard.razor.cs | 201 +++++++++++++-- .../Components/HealthieIcons.cs | 3 + .../Services/HealthieDashboardService.cs | 16 ++ .../Services/IHealthieDashboardService.cs | 15 ++ src/Healthie.Dashboard/wwwroot/healthie.css | 231 +++++++++++++++++- .../TimerPulseScheduler.cs | 42 +++- .../QuartzPulseScheduler.cs | 31 +++ .../ScheduleMutationTests.cs | 176 +++++++++++++ .../StubDashboardService.cs | 3 + 17 files changed, 1007 insertions(+), 40 deletions(-) create mode 100644 tests/Healthie.Tests.Unit/ScheduleMutationTests.cs diff --git a/samples/Healthie.Sample.BlazorUI/Pulses/CertificateExpiryPulseChecker.cs b/samples/Healthie.Sample.BlazorUI/Pulses/CertificateExpiryPulseChecker.cs index e9fbdf4..381fbe2 100644 --- a/samples/Healthie.Sample.BlazorUI/Pulses/CertificateExpiryPulseChecker.cs +++ b/samples/Healthie.Sample.BlazorUI/Pulses/CertificateExpiryPulseChecker.cs @@ -1,6 +1,7 @@ using Healthie.Abstractions; using Healthie.Abstractions.Enums; using Healthie.Abstractions.Models; +using Healthie.Abstractions.Scheduling; using Healthie.Abstractions.StateProviding; namespace Healthie.Sample.BlazorUI.Pulses; @@ -9,8 +10,12 @@ public class CertificateExpiryPulseChecker : PulseChecker { private readonly Random _random = new(); + // On a cron expression rather than an interval, because that is what a certificate check + // actually wants and because it is the case the board has to render: no rate a minute, and an + // interval picker that must not pretend to apply. Every two minutes here so the sample shows it + // running rather than waiting until 03:20. public CertificateExpiryPulseChecker(IStateProvider stateProvider) - : base(stateProvider, PulseInterval.Every2Minutes, 1) + : base(stateProvider, PulseSchedule.Cron("*/2 * * * *"), 1) { } diff --git a/src/Healthie.Abstractions/IPulseChecker.cs b/src/Healthie.Abstractions/IPulseChecker.cs index 53ceeee..5bc4a81 100644 --- a/src/Healthie.Abstractions/IPulseChecker.cs +++ b/src/Healthie.Abstractions/IPulseChecker.cs @@ -1,5 +1,6 @@ using Healthie.Abstractions.Enums; using Healthie.Abstractions.Models; +using Healthie.Abstractions.Scheduling; namespace Healthie.Abstractions; @@ -45,6 +46,27 @@ public interface IPulseChecker : IPulse, IState, IAsyncDisposable /// A task that represents the asynchronous operation. Task SetIntervalAsync(PulseInterval interval, CancellationToken cancellationToken = default); + /// + /// Sets the schedule the pulse check runs on, which a may not be + /// able to express. + /// + /// The schedule to run on, or null to go back to the interval. + /// A token to monitor for cancellation requests. + /// A task that represents the asynchronous operation. + /// + /// This implementation does not support being rescheduled. The default throws; anything deriving + /// from PulseChecker overrides it. + /// + /// + /// Defaulted rather than abstract so a checker written against the older interface keeps + /// compiling. It throws rather than doing nothing, because a scheduling call that silently has + /// no effect leaves a checker running on the old schedule and a caller believing otherwise. + /// + Task SetScheduleAsync(PulseSchedule? schedule, CancellationToken cancellationToken = default) => + throw new NotSupportedException( + $"{GetType().Name} does not support being given a {nameof(PulseSchedule)}. Derive from " + + "PulseChecker, or implement this method."); + /// /// Sets the unhealthy threshold for the pulse checker asynchronously. /// diff --git a/src/Healthie.Abstractions/PulseChecker.cs b/src/Healthie.Abstractions/PulseChecker.cs index 1eb8826..85a8959 100644 --- a/src/Healthie.Abstractions/PulseChecker.cs +++ b/src/Healthie.Abstractions/PulseChecker.cs @@ -352,9 +352,46 @@ private async Task UpdateStateAsync(Action apply, Cance } /// + /// + /// Clears any , because that overrides the interval: + /// without this, choosing an interval on a cron-scheduled checker stored a field nothing reads + /// and the checker carried on at its old cadence. + /// public async Task SetIntervalAsync(PulseInterval interval, CancellationToken cancellationToken = default) { - await UpdateStateAsync(state => state.Interval = interval, cancellationToken).ConfigureAwait(false); + await UpdateStateAsync( + state => + { + state.Interval = interval; + state.Schedule = null; + }, + cancellationToken).ConfigureAwait(false); + } + + /// + /// + /// A schedule an interval can express exactly is stored as that interval, so the common case + /// keeps the shape every stored state and every older reader already understands, and only a + /// genuinely custom period or a cron expression occupies . + /// + public async Task SetScheduleAsync(PulseSchedule? schedule, CancellationToken cancellationToken = default) + { + if (schedule is null) + { + // Back to whatever interval is stored, which is what the field is there to hold. + await UpdateStateAsync(state => state.Schedule = null, cancellationToken).ConfigureAwait(false); + + return; + } + + if (schedule.TryToInterval(out var interval)) + { + await SetIntervalAsync(interval, cancellationToken).ConfigureAwait(false); + + return; + } + + await UpdateStateAsync(state => state.Schedule = schedule, cancellationToken).ConfigureAwait(false); } /// diff --git a/src/Healthie.Abstractions/Scheduling/IPulseScheduler.cs b/src/Healthie.Abstractions/Scheduling/IPulseScheduler.cs index 8d36f20..6385f14 100644 --- a/src/Healthie.Abstractions/Scheduling/IPulseScheduler.cs +++ b/src/Healthie.Abstractions/Scheduling/IPulseScheduler.cs @@ -50,6 +50,30 @@ Task ScheduleAsync(IPulseChecker checker, PulseSchedule schedule, CancellationTo $"{nameof(ScheduleAsync)}({nameof(PulseSchedule)}) to use it."); } + /// + /// Judges a schedule before anything is asked to store it. + /// + /// The schedule to judge. + /// Why it was refused, or null when it was accepted. + /// true when this scheduler could run it. + /// + /// Asked of the scheduler because the scheduler is the authority: Cronos, Quartz and Temporal do + /// not agree on cron dialects, so the only implementation whose answer means anything is the one + /// that will run it. This exists so a schedule typed into the dashboard is refused before it is + /// persisted -- storing it first and discovering the problem while rescheduling leaves a checker + /// with a schedule nothing can run. + /// + /// Defaulted to accept, so a scheduler written against the older interface is unaffected and + /// reports the problem when scheduling, as it does today. + /// + /// + bool TryValidateSchedule(PulseSchedule schedule, out string? error) + { + error = null; + + return true; + } + /// /// Unschedules a previously scheduled pulse checker. /// diff --git a/src/Healthie.Abstractions/Scheduling/IPulsesScheduler.cs b/src/Healthie.Abstractions/Scheduling/IPulsesScheduler.cs index 09672e2..d7d5531 100644 --- a/src/Healthie.Abstractions/Scheduling/IPulsesScheduler.cs +++ b/src/Healthie.Abstractions/Scheduling/IPulsesScheduler.cs @@ -33,6 +33,23 @@ public interface IPulsesScheduler : IHostedService /// Thrown when no pulse checker with the specified exists. Task SetIntervalAsync(string name, PulseInterval interval, CancellationToken cancellationToken = default); + /// + /// Sets the schedule a pulse checker runs on, and reschedules it. + /// + /// The name of the pulse checker. + /// The schedule to run it on, or null to go back to its interval. + /// A token to monitor for cancellation requests. + /// A task that represents the asynchronous operation. + /// + /// No pulse checker with the specified exists, or the registered + /// scheduler will not run the schedule. + /// + /// + /// Refused before it is stored rather than after: a schedule the scheduler cannot run, persisted + /// and then failed on, leaves a checker that no longer runs and a store that says it should. + /// + Task SetScheduleAsync(string name, PulseSchedule? schedule, CancellationToken cancellationToken = default); + /// /// Sets the unhealthy threshold for a specific pulse checker. /// diff --git a/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs b/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs index c619b80..418a5d1 100644 --- a/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs +++ b/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs @@ -98,6 +98,22 @@ public async Task SetIntervalAsync(string name, PulseInterval interval, Cancella await ScheduleAsync(pulseChecker, cancellationToken).ConfigureAwait(false); } + /// + public async Task SetScheduleAsync(string name, PulseSchedule? schedule, CancellationToken cancellationToken = default) + { + var pulseChecker = GetCheckerOrThrow(name); + + if (schedule is not null && !_pulseScheduler.TryValidateSchedule(schedule, out var error)) + { + throw new ArgumentException( + $"The registered scheduler will not run '{schedule}' for pulse checker '{name}'. {error}", + nameof(schedule)); + } + + await pulseChecker.SetScheduleAsync(schedule, cancellationToken).ConfigureAwait(false); + await ScheduleAsync(pulseChecker, cancellationToken).ConfigureAwait(false); + } + /// public async Task SetUnhealthyThresholdAsync(string name, uint threshold, CancellationToken cancellationToken = default) { diff --git a/src/Healthie.Api/Controllers/HealthCheckersController.cs b/src/Healthie.Api/Controllers/HealthCheckersController.cs index 0584a65..c5e071f 100644 --- a/src/Healthie.Api/Controllers/HealthCheckersController.cs +++ b/src/Healthie.Api/Controllers/HealthCheckersController.cs @@ -113,6 +113,65 @@ public async Task SetCheckerInterval(string checkerName, [FromQue } } + /// + /// Sets the schedule a pulse checker runs on, for a cadence no interval expresses. + /// + /// The name of the pulse checker. + /// + /// A standard Unix cron expression, in five fields or six with a leading seconds field. Omit it + /// to clear the schedule and go back to the checker's interval. + /// + /// A token to monitor for cancellation requests. + /// + /// 204 No Content on success, 404 if the checker is not found, or 400 if the name is empty or + /// the registered scheduler will not run the expression. + /// + /// + /// The 400 carries the scheduler's own reason, because the schedulers do not agree on cron + /// dialects and "invalid expression" would leave the caller guessing which rule they broke. + /// + [HttpPut("{checkerName}/schedule")] + [ProducesResponseType(StatusCodes.Status204NoContent)] + [ProducesResponseType(StatusCodes.Status404NotFound)] + [ProducesResponseType(StatusCodes.Status400BadRequest)] + public async Task SetCheckerSchedule( + string checkerName, + [FromQuery] string? cron, + CancellationToken cancellationToken) + { + if (string.IsNullOrWhiteSpace(checkerName)) + { + return BadRequest("Checker name cannot be empty."); + } + + try + { + var checkers = await pulsesScheduler.GetPulseCheckersAsync(cancellationToken).ConfigureAwait(false); + if (!checkers.ContainsKey(checkerName)) + { + logger?.LogWarning("Checker '{CheckerName}' not found for setting a schedule.", ForLog(checkerName)); + return NotFound($"Checker '{checkerName}' not found."); + } + + var schedule = string.IsNullOrWhiteSpace(cron) ? null : PulseSchedule.Cron(cron); + + await pulsesScheduler.SetScheduleAsync(checkerName, schedule, cancellationToken).ConfigureAwait(false); + + return NoContent(); + } + catch (ArgumentException ex) + { + // The schedule itself was refused -- by PulseSchedule for its shape, or by the scheduler + // for its dialect. Either way it is the caller's input, not a server fault. + return BadRequest(ex.Message); + } + catch (Exception ex) + { + logger?.LogError(ex, "Error setting the schedule for checker '{CheckerName}'.", ForLog(checkerName)); + return StatusCode(StatusCodes.Status500InternalServerError, "An unexpected error occurred."); + } + } + /// /// Sets the unhealthy threshold for a specific pulse checker. /// diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor b/src/Healthie.Dashboard/Components/HealthieDashboard.razor index edf3c5f..5b8ed21 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor @@ -52,8 +52,9 @@ @* - Only when something is elected. A single-replica application is always the leader and - does not need telling. + Only when something is elected: an application with no leader election installed has + one replica running everything, and a permanent LEADER badge there would reassure + about a problem nobody has. *@ @if (_leadershipInsights is not null) { @@ -129,7 +130,10 @@ -
AGGREGATE PULSE · @ChecksPerMinute CHECKS/MIN
+
+ AGGREGATE PULSE · @ChecksPerMinute CHECKS/MIN@(CronCheckerCount > 0 ? $" · {CronCheckerCount} CRON" : null) +
@@ -337,8 +341,9 @@
- @RatePerMinute(state) - per min + @RateLabel(state) + @RateUnit(state)
@if (history.Count == 0) @@ -385,13 +390,95 @@ ; } -
+
+ + @* + Reads and filters only, so it shows in read-only mode: every entry either scrolls to a + section or opens a panel. Rendered before the list in the DOM as well as to its left, so + tab order matches reading order. + *@ + +
-
@@ -440,21 +527,21 @@ }
} - else if (_groupByTags) + else if (_sectionByGroup) { foreach (var group in GroupedRows()) { - var collapsed = _collapsedGroups.Contains(group.Tag); + var collapsed = _collapsedGroups.Contains(group.Name);
+ @* + Disabled while a cron expression is in force rather than left + live: the schedule overrides the interval, so a picker that + still moved would store a value nothing reads. + *@
- - @foreach (var interval in Intervals) { @@ -589,6 +683,25 @@
+
+ + + @if (_scheduleError is not null) + { + + } + else + { +
Enter to apply, Escape to undo. e.g. 0 6 * * MON-FRI
+ } +
+
GROUP
@if (_isNamingGroup) diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs b/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs index 728f6a5..d3a5ba4 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor.cs @@ -1,6 +1,7 @@ using Healthie.Abstractions.Enums; using Healthie.Abstractions.Extensions; using Healthie.Abstractions.Models; +using Healthie.Abstractions.Scheduling; using Healthie.Dashboard.Services; using Microsoft.AspNetCore.Components; using Microsoft.AspNetCore.Components.Web; @@ -66,11 +67,48 @@ public sealed partial class HealthieDashboard : IAsyncDisposable private bool _showAbout; private bool _showLog; private bool _asCards; - private bool _groupByTags; + + /// + /// Sectioned by group on open, flat on request. + /// + /// + /// A group is a partition -- every checker under exactly one heading, tallies that add up -- so + /// the sectioned view is the one that answers "what is wrong, and where" at a glance. A flat + /// list of forty checkers makes the reader do that grouping in their head. Checkers given no + /// group collect under one heading rather than disappearing, so nothing is hidden by the + /// default. + /// + private bool _sectionByGroup = true; + private bool _isNamingGroup; private string? _tagDraft; private string? _groupDraft; + /// + /// Whether the side menu is expanded. Open on a desktop-width first load. + /// + /// + /// The rail collapses to its icons rather than disappearing, so the toggle is always reachable + /// and the layout does not reflow to a different set of controls. + /// + private bool _navOpen = true; + + /// + /// The group the menu has narrowed the list to, or null for everything. + /// + /// + /// A separate filter from _tagFilter and the search box, and combined with them rather + /// than replacing them: picking a group in the menu should not silently clear a search someone + /// has typed. + /// + private string? _navSection; + + /// What is in the cron box, which is not the stored schedule until it is applied. + private string? _cronDraft; + + /// Why the last schedule was refused, shown beside the box that was typed into. + private string? _scheduleError; + /// Marks the "new group" choice in the group picker, which no real group can collide with. private const string NewGroupOption = "\u0000new"; @@ -149,12 +187,22 @@ private string OverallLabel _ => "var(--hpm-ok)", }; - /// How many checks a minute the active checkers add up to. + /// + /// How many checks a minute the active checkers add up to. + /// + /// + /// Cron checkers are left out rather than guessed at: "every weekday at 06:00" has no rate a + /// minute, and folding in the interval they are not running on is what this used to do. + /// reports them separately so they are not silently missing. + /// private string ChecksPerMinute => _states.Values - .Where(state => state.IsActive) - .Sum(state => 60d / state.Interval.ToTimeSpan().TotalSeconds) + .Where(state => state.IsActive && !state.EffectiveSchedule.IsCron) + .Sum(state => 60d / state.EffectiveSchedule.Period!.Value.TotalSeconds) .ToString("0"); + /// How many active checkers run on a cron expression instead of a rate. + private int CronCheckerCount => _states.Values.Count(state => state.IsActive && state.EffectiveSchedule.IsCron); + /// How many runs the sparkline can show, which is however many are kept. private int HistoryWindow => _states.Count == 0 ? 0 : _states.Values.Max(s => s.History.Count); @@ -197,9 +245,10 @@ protected override async Task OnInitializedAsync() await LoadAsync(); } - // The first checker is selected by MarkLoaded rather than by a click, so nothing has read its - // uptime yet -- without this the panel opens missing the columns every later selection shows. - await LoadUptimeAsync(_selected); + // The first checker is selected by MarkLoaded rather than by a click, so it has not been + // through the path that reads its uptime and fills the editors. Without this the panel opens + // missing the columns and drafts that every later selection has. + await SelectAsync(_selected); _clockLoop = RunClockAsync(); } @@ -361,6 +410,15 @@ private void Refresh() entry.Value.Tags.Contains(_tagFilter, StringComparer.OrdinalIgnoreCase)); } + // Narrowed from the side menu, and combined with the two above rather than replacing them. + if (_navSection is { } section) + { + filtered = filtered.Where(entry => string.Equals( + string.IsNullOrWhiteSpace(entry.Value.Group) ? UngroupedName : entry.Value.Group, + section, + StringComparison.OrdinalIgnoreCase)); + } + // Pinned first, then by name. Pinning is only useful if it survives the sort. _filtered = [ @@ -393,17 +451,17 @@ private void OnSearchInput(ChangeEventArgs args) /// several of them can be on one checker, which would put it under several headings and make /// the tallies count it twice. /// - private IEnumerable GroupedRows() => + private IEnumerable GroupedRows() => _filtered .GroupBy(entry => string.IsNullOrWhiteSpace(entry.Value.Group) ? UngroupedName : entry.Value.Group!, StringComparer.OrdinalIgnoreCase) - .Select(group => new TagGroup(group.Key, [.. group])) + .Select(group => new CheckerGroup(group.Key, [.. group])) // Ungrouped last: it is the leftovers, not a heading anyone chose. - .OrderBy(group => group.Tag == UngroupedName) - .ThenBy(group => group.Tag, StringComparer.OrdinalIgnoreCase); + .OrderBy(group => group.Name == UngroupedName) + .ThenBy(group => group.Name, StringComparer.OrdinalIgnoreCase); /// One group's checkers, and the tallies its header shows. - private sealed record TagGroup(string Tag, List> Rows) + private sealed record CheckerGroup(string Name, List> Rows) { public int Healthy => Rows.Count(r => HealthOf(r.Value) == PulseCheckerHealth.Healthy); @@ -430,8 +488,43 @@ private void ToggleGroup(string tag) private void ToggleGrouping() { - _groupByTags = !_groupByTags; + _sectionByGroup = !_sectionByGroup; + Refresh(); + } + + private void ToggleNav() => _navOpen = !_navOpen; + + /// + /// The groups the menu lists, which are the ones in the store rather than the ones surviving the + /// current filters. + /// + /// + /// Off , not off the filtered rows: a menu that dropped a group because the + /// search box had narrowed it away would take away the means of getting back to it. + /// + private IEnumerable NavGroups() => + _states + .GroupBy(entry => string.IsNullOrWhiteSpace(entry.Value.Group) ? UngroupedName : entry.Value.Group!, + StringComparer.OrdinalIgnoreCase) + .Select(group => new CheckerGroup(group.Key, [.. group])) + .OrderBy(group => group.Name == UngroupedName) + .ThenBy(group => group.Name, StringComparer.OrdinalIgnoreCase); + + private Task ShowEverythingAsync() + { + _navSection = null; + Refresh(); + + return Task.CompletedTask; + } + + /// Narrows the list to one group, or back to everything when it is picked again. + private Task ShowGroupAsync(string group) + { + _navSection = string.Equals(_navSection, group, StringComparison.OrdinalIgnoreCase) ? null : group; Refresh(); + + return Task.CompletedTask; } private void OnTagFilterChanged(ChangeEventArgs args) @@ -461,6 +554,11 @@ private async Task SelectAsync(string? name) _selected = name; _diagnosis = null; + // The cron box follows the selection: left alone it would show one checker's expression + // while Enter applied it to another. + _cronDraft = _selectedState?.Schedule?.CronExpression; + _scheduleError = null; + await LoadUptimeAsync(name); } @@ -619,6 +717,59 @@ private async Task OnIntervalChanged(ChangeEventArgs args) AddEvent("CONF", Status.Paused, $"{DisplayNameOf(_selected)} interval set to {interval}"); } + /// + /// Applies what is in the cron box: a schedule, or none when it has been emptied. + /// + /// + /// On Enter rather than on every keystroke, because a half-typed expression is not a schedule + /// and every attempt reschedules the checker. A refusal is shown where it was typed -- the + /// schedulers disagree about cron dialects, so which rule was broken is the useful part, and the + /// scheduler is asked before anything is stored. + /// + private async Task ApplyCronAsync() + { + if (_selected is null) + { + return; + } + + _scheduleError = null; + + var typed = _cronDraft?.Trim(); + + try + { + if (string.IsNullOrEmpty(typed)) + { + await DashboardService.SetScheduleAsync(_selected, null); + AddEvent("CONF", Status.Paused, $"{DisplayNameOf(_selected)} back on its interval"); + + return; + } + + await DashboardService.SetScheduleAsync(_selected, PulseSchedule.Cron(typed)); + AddEvent("CONF", Status.Paused, $"{DisplayNameOf(_selected)} scheduled on '{typed}'"); + } + catch (ArgumentException ex) + { + _scheduleError = ex.Message; + } + } + + private async Task OnCronKeyDown(KeyboardEventArgs args) + { + switch (args.Key) + { + case "Enter": + await ApplyCronAsync(); + break; + case "Escape": + _cronDraft = _selectedState?.Schedule?.CronExpression; + _scheduleError = null; + break; + } + } + private async Task OnThresholdChanged(ChangeEventArgs args) { if (_selected is null || @@ -682,13 +833,33 @@ private static string StatusWord(PulseCheckerState state) => : state.LastResult is null ? "PENDING" : state.LastResult.Health.ToString().ToUpperInvariant(); - private static string RatePerMinute(PulseCheckerState state) + /// + /// What the rate column reads: a number of runs a minute, or the cron expression when the + /// schedule is one. + /// + /// + /// Off , not off Interval. Interval is + /// documented as ignored once a schedule is set, so a cron checker used to advertise a rate it + /// was not running at -- and the aggregate summed that rate into its total. + /// + private static string RateLabel(PulseCheckerState state) { - var perMinute = 60d / state.Interval.ToTimeSpan().TotalSeconds; + var schedule = state.EffectiveSchedule; + + if (schedule.CronExpression is { } cron) + { + return cron; + } + + var perMinute = 60d / schedule.Period!.Value.TotalSeconds; return perMinute >= 1 ? Math.Round(perMinute).ToString("0") : perMinute.ToString("0.0"); } + /// The unit under the rate, which a cron expression does not have. + private static string RateUnit(PulseCheckerState state) => + state.EffectiveSchedule.IsCron ? "cron" : "per min"; + private static string FailuresLabel(PulseCheckerState state) => state.UnhealthyThreshold > 0 ? $"{state.ConsecutiveFailureCount}/{state.UnhealthyThreshold}" diff --git a/src/Healthie.Dashboard/Components/HealthieIcons.cs b/src/Healthie.Dashboard/Components/HealthieIcons.cs index ba99bb9..4daf8ea 100644 --- a/src/Healthie.Dashboard/Components/HealthieIcons.cs +++ b/src/Healthie.Dashboard/Components/HealthieIcons.cs @@ -58,6 +58,9 @@ internal static class HealthieIcons public static MarkupString X => Svg( ""); + public static MarkupString Menu => Svg( + ""); + public static MarkupString Rows => Svg( ""); diff --git a/src/Healthie.Dashboard/Services/HealthieDashboardService.cs b/src/Healthie.Dashboard/Services/HealthieDashboardService.cs index d4273a2..88cd57b 100644 --- a/src/Healthie.Dashboard/Services/HealthieDashboardService.cs +++ b/src/Healthie.Dashboard/Services/HealthieDashboardService.cs @@ -50,6 +50,22 @@ await pulsesScheduler.SetIntervalAsync(name, interval, cancellationToken) } } + /// + public async Task SetScheduleAsync(string name, PulseSchedule? schedule, + CancellationToken cancellationToken = default) + { + try + { + await pulsesScheduler.SetScheduleAsync(name, schedule, cancellationToken) + .ConfigureAwait(false); + } + catch (Exception ex) + { + logger?.LogError(ex, "Failed to set the schedule for checker '{CheckerName}'.", name); + throw; + } + } + /// public async Task SetThresholdAsync(string name, uint threshold, CancellationToken cancellationToken = default) diff --git a/src/Healthie.Dashboard/Services/IHealthieDashboardService.cs b/src/Healthie.Dashboard/Services/IHealthieDashboardService.cs index 5b97ebf..48754fc 100644 --- a/src/Healthie.Dashboard/Services/IHealthieDashboardService.cs +++ b/src/Healthie.Dashboard/Services/IHealthieDashboardService.cs @@ -1,5 +1,6 @@ using Healthie.Abstractions.Enums; using Healthie.Abstractions.Models; +using Healthie.Abstractions.Scheduling; namespace Healthie.Dashboard.Services; @@ -69,6 +70,20 @@ Task> GetAllStatesAsync( Task SetIntervalAsync(string name, PulseInterval interval, CancellationToken cancellationToken = default); + /// + /// Sets the schedule a pulse checker runs on, for a cadence no + /// expresses. + /// + /// The name of the pulse checker. + /// The schedule, or null to go back to the interval. + /// A token to monitor for cancellation requests. + /// + /// The registered scheduler will not run this schedule. The message is written to be shown to + /// whoever typed it. + /// + Task SetScheduleAsync(string name, PulseSchedule? schedule, + CancellationToken cancellationToken = default); + /// /// Sets the unhealthy threshold for a specific pulse checker. /// diff --git a/src/Healthie.Dashboard/wwwroot/healthie.css b/src/Healthie.Dashboard/wwwroot/healthie.css index 0c544d7..707f7ae 100644 --- a/src/Healthie.Dashboard/wwwroot/healthie.css +++ b/src/Healthie.Dashboard/wwwroot/healthie.css @@ -279,15 +279,21 @@ /* ------------------------------------------------------------------ body -- */ .hpm-body { + --hpm-nav-w: 208px; + max-width: 1500px; margin: 0 auto; padding: 22px 26px 70px; display: grid; - grid-template-columns: minmax(0, 1fr) 400px; + grid-template-columns: var(--hpm-nav-w) minmax(0, 1fr) 400px; gap: 18px; align-items: start; } +.hpm-body--nav-collapsed { + --hpm-nav-w: 54px; +} + .hpm-list { display: flex; flex-direction: column; @@ -722,12 +728,59 @@ /* ------------------------------------------------------------ responsive -- */ @media (max-width: 1100px) { + /* The detail panel drops below the list; the rail keeps its column. */ .hpm-body { - grid-template-columns: minmax(0, 1fr); + grid-template-columns: var(--hpm-nav-w) minmax(0, 1fr); } .hpm-side { position: static; + grid-column: 1 / -1; + } +} + +/* + * Below this a rail costs more width than it earns, so it becomes a strip above the list: one + * horizontally scrolling row of the same items. Still every entry, still no JavaScript, and nothing + * overlays the content -- which is what an off-canvas drawer would need a focus trap to do honestly. + */ +@media (max-width: 820px) { + .hpm-body, + .hpm-body--nav-collapsed { + grid-template-columns: minmax(0, 1fr); + } + + .hpm-nav { + position: static; + flex-direction: row; + align-items: center; + gap: 6px; + overflow-x: auto; + scrollbar-width: thin; + } + + .hpm-nav-toggle, + .hpm-nav-heading { + display: none; + } + + .hpm-nav-item { + width: auto; + flex: 0 0 auto; + border-left: 0; + border-bottom: 2px solid transparent; + border-radius: 6px; + } + + .hpm-nav-item--current { + border-left-color: transparent; + border-bottom-color: var(--hpm-accent, var(--hpm-ok)); + } + + /* The labels come back here even when the rail was left collapsed on a wider screen. */ + .hpm-body--nav-collapsed .hpm-nav-text, + .hpm-body--nav-collapsed .hpm-nav-count { + display: inline; } } @@ -1699,3 +1752,177 @@ white-space: normal; } } + +/* --------------------------------------------------------------------------- + Side menu + + A rail rather than a drawer, and no JavaScript, because this package ships + without any: an overlay drawer needs a backdrop and a focus trap to be + honest about keyboard use, and neither is doable in CSS alone. Collapsed it + keeps its icons and its toggle, so nothing becomes unreachable and the page + does not reflow onto a different set of controls. + --------------------------------------------------------------------------- */ + +.hpm-nav { + position: sticky; + top: 14px; + display: flex; + flex-direction: column; + gap: 2px; + min-width: 0; + padding: 8px; + border: 1px solid var(--hpm-line); + border-radius: 8px; + background: var(--hpm-panel); +} + +.hpm-nav-toggle { + display: flex; + align-items: center; + gap: 10px; + width: 100%; + padding: 8px 9px; + border: 0; + border-radius: 6px; + background: none; + color: var(--hpm-faint); + font: inherit; + font-size: 11.5px; + letter-spacing: 0.09em; + cursor: pointer; +} + +.hpm-nav-toggle:hover { + color: var(--hpm-text); + background: var(--hpm-line); +} + +.hpm-nav-burger { + flex: 0 0 auto; + width: 16px; + height: 16px; +} + +.hpm-nav-heading { + margin: 12px 0 4px; + padding: 0 9px; + color: var(--hpm-faint); + font-size: 9.5px; + letter-spacing: 0.16em; +} + +.hpm-nav-item { + display: flex; + align-items: center; + gap: 9px; + width: 100%; + padding: 7px 9px; + border: 0; + border-left: 2px solid transparent; + border-radius: 0 6px 6px 0; + background: none; + color: var(--hpm-text); + font: inherit; + font-size: 11.5px; + letter-spacing: 0.05em; + text-align: left; + cursor: pointer; +} + +.hpm-nav-item:hover { + background: var(--hpm-line); +} + +.hpm-nav-item--current { + border-left-color: var(--hpm-accent, var(--hpm-ok)); + background: var(--hpm-line); +} + +.hpm-nav-dot { + flex: 0 0 auto; + width: 7px; + height: 7px; + border-radius: 50%; + background: var(--hpm-faint); +} + +.hpm-nav-dot.hpm-ok { background: var(--hpm-ok); } +.hpm-nav-dot.hpm-warn { background: var(--hpm-warn); } +.hpm-nav-dot.hpm-crit { background: var(--hpm-crit); } + +.hpm-nav-text { + flex: 1 1 auto; + min-width: 0; + overflow: hidden; + text-overflow: ellipsis; + white-space: nowrap; +} + +.hpm-nav-count { + flex: 0 0 auto; + color: var(--hpm-faint); + font-size: 10px; +} + +.hpm-nav-count--warn { + color: var(--hpm-crit); +} + +/* Collapsed: the labels go, the icons and the toggle stay. */ +.hpm-body--nav-collapsed .hpm-nav-text, +.hpm-body--nav-collapsed .hpm-nav-count, +.hpm-body--nav-collapsed .hpm-nav-label, +.hpm-body--nav-collapsed .hpm-nav-heading { + display: none; +} + +.hpm-body--nav-collapsed .hpm-nav-item, +.hpm-body--nav-collapsed .hpm-nav-toggle { + justify-content: center; + padding-inline: 0; +} + +.hpm-body--nav-collapsed .hpm-nav-dot { + width: 9px; + height: 9px; +} + +/* The cron box sits under the interval and threshold pair, full width. */ +.hpm-field--cron { + margin-top: 10px; +} + +.hpm-field--cron input { + font-family: var(--hpm-mono); +} + +.hpm-field--cron input[aria-invalid="true"] { + border-color: var(--hpm-crit); +} + +.hpm-field-hint, +.hpm-field-error { + margin-top: 5px; + font-size: 10.5px; + line-height: 1.45; +} + +.hpm-field-hint { + color: var(--hpm-faint); +} + +.hpm-field-hint code { + font-family: var(--hpm-mono); + color: var(--hpm-muted); +} + +.hpm-field-error { + color: var(--hpm-crit); +} + +/* A cron expression is text, not a number, so it does not get the big numeric treatment. */ +.hpm-rate--cron { + font-family: var(--hpm-mono); + font-size: 11px; + letter-spacing: 0; +} diff --git a/src/Healthie.DependencyInjection/TimerPulseScheduler.cs b/src/Healthie.DependencyInjection/TimerPulseScheduler.cs index f1e29a7..8e504d2 100644 --- a/src/Healthie.DependencyInjection/TimerPulseScheduler.cs +++ b/src/Healthie.DependencyInjection/TimerPulseScheduler.cs @@ -252,18 +252,50 @@ private async Task TriggerAsync(IPulseChecker checker, CancellationToken token) } } + /// + /// + /// Cronos is what actually drives the timer, so asking Cronos is the only answer worth giving. + /// + public bool TryValidateSchedule(PulseSchedule schedule, out string? error) + { + ArgumentNullException.ThrowIfNull(schedule); + + error = null; + + if (schedule.CronExpression is not { } expression) + { + return true; + } + + try + { + CronExpression.Parse(expression, CronFormatFor(expression)); + + return true; + } + catch (CronFormatException ex) + { + error = $"Expected standard Unix cron -- five fields, or six with a leading seconds " + + $"field. {ex.Message}"; + + return false; + } + } + + /// Six fields or more means the leading one is seconds. + private static CronFormat CronFormatFor(string expression) => + expression.Split(' ', StringSplitOptions.RemoveEmptyEntries).Length >= 6 + ? CronFormat.IncludeSeconds + : CronFormat.Standard; + /// /// Parses a standard Unix cron expression, in five fields or six with a leading seconds field. /// private static CronExpression ParseCron(string expression, string checkerName) { - var format = expression.Split(' ', StringSplitOptions.RemoveEmptyEntries).Length >= 6 - ? CronFormat.IncludeSeconds - : CronFormat.Standard; - try { - return CronExpression.Parse(expression, format); + return CronExpression.Parse(expression, CronFormatFor(expression)); } catch (CronFormatException ex) { diff --git a/src/Healthie.Scheduling.Quartz/QuartzPulseScheduler.cs b/src/Healthie.Scheduling.Quartz/QuartzPulseScheduler.cs index 01bb08e..0587982 100644 --- a/src/Healthie.Scheduling.Quartz/QuartzPulseScheduler.cs +++ b/src/Healthie.Scheduling.Quartz/QuartzPulseScheduler.cs @@ -70,6 +70,37 @@ public Task ScheduleAsync( : ScheduleCoreAsync(checker, cronExpression: null, schedule.Period, cancellationToken); } + /// + /// + /// Answered by the same translation that runs it. Quartz takes a six-field expression with its + /// own day-of-week numbering and refuses to constrain both day fields at once, so an expression + /// Cronos is happy with is not automatically one this scheduler can run. + /// + public bool TryValidateSchedule(PulseSchedule schedule, out string? error) + { + ArgumentNullException.ThrowIfNull(schedule); + + error = null; + + if (schedule.CronExpression is not { } expression) + { + return true; + } + + try + { + CronScheduleBuilder.CronSchedule(UnixCron.ToQuartz(expression)); + + return true; + } + catch (Exception ex) when (ex is NotSupportedException or FormatException or ArgumentException) + { + error = ex.Message; + + return false; + } + } + /// /// Replaces this checker's Quartz job with one on the given trigger. /// diff --git a/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs b/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs new file mode 100644 index 0000000..60360b8 --- /dev/null +++ b/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs @@ -0,0 +1,176 @@ +using Healthie.Abstractions; +using Healthie.Abstractions.Enums; +using Healthie.Abstractions.Models; +using Healthie.Abstractions.Scheduling; +using Healthie.Abstractions.StateProviding; +using Healthie.DependencyInjection; + +namespace Healthie.Tests.Unit; + +/// +/// Setting a schedule from outside the code that declared it. +/// +/// +/// The trap these cover is that is ignored once +/// is set. Anything that writes one without considering the +/// other stores a value nothing reads, and the checker carries on at its old cadence while the caller +/// believes otherwise. +/// +public class ScheduleMutationTests +{ + private static CancellationToken Ct => TestContext.Current.CancellationToken; + + private sealed class Checker(IStateProvider provider, PulseSchedule schedule) + : PulseChecker(provider, schedule) + { + public override string Name => "schedule-target"; + + public override Task CheckAsync(CancellationToken cancellationToken = default) => + Task.FromResult(new PulseCheckerResult(PulseCheckerHealth.Healthy, "ok")); + } + + private static async Task<(Checker Checker, IStateProvider Provider)> DailyCheckerAsync() + { + var provider = new InMemoryStateProvider(); + var checker = new Checker(provider, PulseSchedule.Cron("0 3 * * *")); + + // Seeds the initial state, which is what makes the cron schedule the stored one. + await checker.TriggerAsync(Ct); + + return (checker, provider); + } + + private static async Task StateOf(IStateProvider provider) => + (await provider.GetStateAsync("schedule-target", Ct))!; + + /// + /// The headline bug: choosing an interval on a cron checker used to write a field nothing reads. + /// + [Fact] + public async Task SettingAnInterval_ClearsACronSchedule_SoTheChoiceTakesEffect() + { + var (checker, provider) = await DailyCheckerAsync(); + using var _ = checker; + + Assert.True((await StateOf(provider)).EffectiveSchedule.IsCron); + + await checker.SetIntervalAsync(PulseInterval.Every30Seconds, Ct); + + var state = await StateOf(provider); + + Assert.Null(state.Schedule); + Assert.Equal(PulseInterval.Every30Seconds, state.Interval); + Assert.Equal(TimeSpan.FromSeconds(30), state.EffectiveSchedule.Period); + } + + [Fact] + public async Task SettingACronSchedule_StoresIt() + { + var (checker, provider) = await DailyCheckerAsync(); + using var _ = checker; + + await checker.SetScheduleAsync(PulseSchedule.Cron("*/5 * * * *"), Ct); + + Assert.Equal("*/5 * * * *", (await StateOf(provider)).Schedule?.CronExpression); + } + + /// + /// A schedule the enum can name exactly is stored as that interval, so the common case keeps the + /// shape every stored state and every older reader already understands. + /// + [Fact] + public async Task SettingAScheduleAnIntervalCanExpress_StoresTheInterval_NotTheSchedule() + { + var (checker, provider) = await DailyCheckerAsync(); + using var _ = checker; + + await checker.SetScheduleAsync(PulseSchedule.Every(TimeSpan.FromMinutes(5)), Ct); + + var state = await StateOf(provider); + + Assert.Null(state.Schedule); + Assert.Equal(PulseInterval.Every5Minutes, state.Interval); + } + + [Fact] + public async Task ClearingTheSchedule_GoesBackToTheStoredInterval() + { + var (checker, provider) = await DailyCheckerAsync(); + using var _ = checker; + + await checker.SetScheduleAsync(null, Ct); + + var state = await StateOf(provider); + + Assert.Null(state.Schedule); + Assert.False(state.EffectiveSchedule.IsCron); + } + + /// + /// A period no names has to survive as a schedule rather than being + /// rounded to the nearest one the enum happens to have. + /// + [Fact] + public async Task SettingAnAwkwardPeriod_KeepsItExactly() + { + var (checker, provider) = await DailyCheckerAsync(); + using var _ = checker; + + await checker.SetScheduleAsync(PulseSchedule.Every(TimeSpan.FromSeconds(90)), Ct); + + Assert.Equal(TimeSpan.FromSeconds(90), (await StateOf(provider)).Schedule?.Period); + } + + [Fact] + public void TimerScheduler_AcceptsAValidCronExpression() + { + using var scheduler = new TimerPulseScheduler(); + + Assert.True(scheduler.TryValidateSchedule(PulseSchedule.Cron("0 6 * * MON-FRI"), out var error)); + Assert.Null(error); + } + + /// + /// Refused with a reason, not just refused: the schedulers disagree about cron dialects, so which + /// rule was broken is the part worth showing whoever typed it. + /// + [Theory] + [InlineData("99 99 * * *")] + [InlineData("not a cron expression")] + [InlineData("* * *")] + public void TimerScheduler_RefusesABadCronExpression_WithAReason(string expression) + { + using var scheduler = new TimerPulseScheduler(); + + Assert.False(scheduler.TryValidateSchedule(PulseSchedule.Cron(expression), out var error)); + Assert.False(string.IsNullOrWhiteSpace(error)); + } + + [Fact] + public void TimerScheduler_AcceptsAPeriodWithoutOpinion() + { + using var scheduler = new TimerPulseScheduler(); + + Assert.True(scheduler.TryValidateSchedule(PulseSchedule.Every(TimeSpan.FromSeconds(90)), out _)); + } + + /// + /// Refused before it is stored. Persisting first and failing on the reschedule leaves a checker + /// that no longer runs and a store that says it should. + /// + [Fact] + public async Task PulsesScheduler_RefusingASchedule_LeavesTheStoredOneAlone() + { + var provider = new InMemoryStateProvider(); + using var checker = new Checker(provider, PulseSchedule.Cron("0 3 * * *")); + await checker.TriggerAsync(Ct); + + using var scheduler = new TimerPulseScheduler(); + var pulses = new PulsesScheduler([checker], scheduler, new HealthieOptions()); + + await Assert.ThrowsAsync(() => + pulses.SetScheduleAsync("schedule-target", PulseSchedule.Cron("99 99 * * *"), Ct)); + + Assert.Equal("0 3 * * *", (await StateOf(provider)).Schedule?.CronExpression); + } +} diff --git a/tests/Healthie.Tests.Unit/StubDashboardService.cs b/tests/Healthie.Tests.Unit/StubDashboardService.cs index cce2521..9aac4ef 100644 --- a/tests/Healthie.Tests.Unit/StubDashboardService.cs +++ b/tests/Healthie.Tests.Unit/StubDashboardService.cs @@ -1,5 +1,6 @@ using Healthie.Abstractions.Enums; using Healthie.Abstractions.Models; +using Healthie.Abstractions.Scheduling; using Healthie.Dashboard.Services; namespace Healthie.Tests.Unit; @@ -31,6 +32,8 @@ private static NotSupportedException NotStubbed(string member) => public virtual Task SetIntervalAsync(string name, PulseInterval interval, CancellationToken cancellationToken = default) => throw NotStubbed(nameof(SetIntervalAsync)); + public virtual Task SetScheduleAsync(string name, PulseSchedule? schedule, CancellationToken cancellationToken = default) => throw NotStubbed(nameof(SetScheduleAsync)); + public virtual Task SetThresholdAsync(string name, uint threshold, CancellationToken cancellationToken = default) => throw NotStubbed(nameof(SetThresholdAsync)); public virtual Task SetTagsAsync(string name, IReadOnlyList tags, CancellationToken cancellationToken = default) => throw NotStubbed(nameof(SetTagsAsync)); From ac614be84ea6006e0dc39032cfac72d65123bdfd Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 13:08:26 +0300 Subject: [PATCH 09/12] Cover the new dashboard behaviour, and correct the tests that assumed the old default Three E2E grouping tests clicked the GROUP button to turn grouping on. With it on by default that click turns it off, so they asserted grouped structure on an ungrouped board -- one failed and two passed by luck, which is worse. The one that needed a flat count for comparison now takes the grouped count first and toggles to flat, which is the comparison it always wanted; the other two simply stop toggling. New coverage: the board opens grouped, the side menu narrows to a group and back, the cron editor applies a valid expression and refuses a bad one without disturbing what is stored, and the interval picker is inert for a cron checker and live for an interval one. The cron E2E tests first read the box before the click had round-tripped -- the box is already visible for whichever checker was selected on load, so waiting on it proves nothing. They wait on the panel title instead. Also: the unit test double derived from PulseChecker, so assembly scanning discovered it and could not construct it, which failed twenty registration, MCP and AI tests that have nothing to do with schedules. It takes only a state provider now, like every other checker in that assembly. Caught by running the whole suite rather than the filter. --- CHANGELOG.md | 35 +++++ .../Scheduling/PulsesScheduler.cs | 7 +- .../Components/HealthieDashboard.razor | 2 +- src/Healthie.Dashboard/README.md | 16 ++- src/Healthie.Dashboard/wwwroot/healthie.css | 116 ++++++++++------- .../TimerPulseScheduler.cs | 5 +- .../DashboardGroupingTests.cs | 123 +++++++++++++++++- .../ScheduleMutationTests.cs | 25 +++- 8 files changed, 258 insertions(+), 71 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 1c19a62..62450e3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -126,9 +126,44 @@ before upgrading: Read-only throughout, so all of it shows under `HealthieUIOptions.AllowMutations = false` -- the one exception is asking the model, which is still a read but spends money on the host's account, and is gated with the controls that change things. +- **A side menu on the dashboard**, listing an overview, a section per group with its tally and worst + state, and the alerts, event-log and about views. Picking a group narrows the list and picking it + again is the way back; it combines with the search box and the tag filter rather than clearing + them, and the groups it lists come from the store rather than from the rows currently surviving + those filters -- a menu that dropped a group because a search had narrowed it away would take away + the means of getting back to it. + + A rail rather than an overlay drawer, and collapsing to its icons rather than disappearing. This + package ships no JavaScript of its own, and an off-canvas drawer needs a focus trap to be honest + about keyboard use. Below 820px it becomes a scrolling strip above the list. +- **The dashboard opens sectioned by group** rather than as one flat list. A group is a partition, so + the sectioned view answers what is wrong and where at a glance; a flat list of forty checkers asks + the reader to do that grouping themselves. Checkers with no group collect under one heading, so the + default hides nothing. The `GROUP` button still switches to the flat list. +- **A cron expression can be set from the dashboard**, beside the interval picker, and through + `PUT /healthie/{checkerName}/schedule` on the REST API. The scheduler judges the expression before + anything is stored -- `IPulseScheduler.TryValidateSchedule`, defaulted to accept so an existing + scheduler is unaffected -- because Cronos, Quartz and Temporal do not agree on cron dialects and + the only answer worth having is from the implementation that will run it. A refusal carries that + implementation's own reason and leaves the stored schedule alone; storing first and failing on the + reschedule would leave a checker that no longer runs and a store that says it should. + + `IPulseChecker.SetScheduleAsync` and `IPulsesScheduler.SetScheduleAsync` are the API behind it. A + schedule an interval can express exactly is stored as that interval, so only a genuinely custom + period or a cron expression occupies `PulseCheckerState.Schedule`. ### Fixed +- **The dashboard ignored `PulseSchedule` everywhere it showed a cadence.** It read + `PulseCheckerState.Interval` for the rate column, for the aggregate checks-per-minute, and for the + interval picker -- and that field is documented as ignored once `Schedule` is set. So a checker on + a cron expression advertised a rate it was not running at, was summed into the aggregate at that + rate, and offered a picker whose every change stored a field nothing reads. The board now reads + `EffectiveSchedule`, shows a cron expression as one, counts cron checkers separately rather than + inventing a rate for them, and disables the picker while an expression is in force. +- **`SetIntervalAsync` did nothing to a checker that had a schedule.** The schedule overrides the + interval, so choosing one left the checker at its old cadence with nothing to say so. It now clears + the schedule, which is what choosing an interval means. - **The dashboard's own page did not encode its title.** `MapHealthieUI` builds that one page as a string rather than through Razor, which encodes every interpolation for you, so a `HealthieUIOptions.DashboardTitle` built from anything the host did not write itself could close diff --git a/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs b/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs index 418a5d1..c2b9e9b 100644 --- a/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs +++ b/src/Healthie.Abstractions/Scheduling/PulsesScheduler.cs @@ -105,9 +105,10 @@ public async Task SetScheduleAsync(string name, PulseSchedule? schedule, Cancell if (schedule is not null && !_pulseScheduler.TryValidateSchedule(schedule, out var error)) { - throw new ArgumentException( - $"The registered scheduler will not run '{schedule}' for pulse checker '{name}'. {error}", - nameof(schedule)); + // No parameter name: this message is written to be shown to whoever typed the + // expression, and "(Parameter 'schedule')" is plumbing to everyone but a debugger. The + // caller named the checker, so the message does not repeat it back. + throw new ArgumentException($"'{schedule}' cannot be scheduled. {error}"); } await pulseChecker.SetScheduleAsync(schedule, cancellationToken).ConfigureAwait(false); diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor b/src/Healthie.Dashboard/Components/HealthieDashboard.razor index 5b8ed21..aaa6d9c 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor @@ -688,7 +688,7 @@ diff --git a/src/Healthie.Dashboard/README.md b/src/Healthie.Dashboard/README.md index 03db1b8..2fbfb2c 100644 --- a/src/Healthie.Dashboard/README.md +++ b/src/Healthie.Dashboard/README.md @@ -1,4 +1,4 @@ -![Healthie.NET - Trust your uptime](https://raw.githubusercontent.com/ivanvyd/Healthie.NET/main/healthie.net.banner.png) +![Healthie.NET - Trust your uptime](https://raw.githubusercontent.com/ivanvyd/Healthie.NET/main/healthie.net.banner.png) # Healthie.NET.Dashboard @@ -57,13 +57,14 @@ app.MapHealthieUI().RequireAuthorization("AdminPolicy"); // With auth ## Features - Event-driven real-time updates via `IPulseChecker.StateChanged` (no polling) -- Per-checker management: start, stop, trigger, reset, change interval, change threshold +- Per-checker management: start, stop, trigger, reset, retime by interval or cron expression, change threshold - Bulk actions: Start All, Stop All, Trigger All - A read-only mode that reports everything and changes nothing — see below - Panels for the feature packages you install — uptime, alerts, leadership, AI — see below - Groups and tags, both editable here and seeded from code — see below +- A collapsible left menu: overview, a section per group with its tally, and the alerts, log and about views - Pin a checker to the top of the list -- Rows or cards, flat or sectioned by group with per-group tallies +- Rows or cards, sectioned by group when it opens or flat on request, with per-group tallies - Live event log, with a full-size view behind the expand icon - Legend and about behind the `?` in the header - Dark/light theme toggle @@ -84,10 +85,11 @@ builder.Services.AddHealthieUI(options => options.AllowMutations = false); That leaves a board that only reports. Every state, sparkline, group, tag, and event stays exactly where it was; the controls that would change any of it are not rendered. Nothing is lost to the -reader, because the values behind the editors are on the board already — the interval is the row's -rate, the threshold is the denominator in `FAILS`, and the group and tags are the chips under each -name. Searching, filtering, grouping, switching to cards, opening the event log, and the theme -toggle all still work: they change your view, not the checker. +reader, because the values behind the editors are on the board already — the schedule is the row's +rate, or its cron expression where it runs on one; the threshold is the denominator in `FAILS`; and +the group and tags are the chips under each name. The side menu, searching, filtering, grouping, +switching to cards, opening the event log, and the theme toggle all still work: they change your +view, not the checker. **This is not authorization.** It is one setting for the whole application, applied to every viewer alike, so it cannot hand the controls to an admin and withhold them from everyone else. It answers diff --git a/src/Healthie.Dashboard/wwwroot/healthie.css b/src/Healthie.Dashboard/wwwroot/healthie.css index 707f7ae..8d2f14c 100644 --- a/src/Healthie.Dashboard/wwwroot/healthie.css +++ b/src/Healthie.Dashboard/wwwroot/healthie.css @@ -1,4 +1,4 @@ -/* +/* * Healthie.NET dashboard -- Pulse Monitor. * * No third-party CSS and no web fonts: this ships inside someone else's application, which may run @@ -739,51 +739,6 @@ } } -/* - * Below this a rail costs more width than it earns, so it becomes a strip above the list: one - * horizontally scrolling row of the same items. Still every entry, still no JavaScript, and nothing - * overlays the content -- which is what an off-canvas drawer would need a focus trap to do honestly. - */ -@media (max-width: 820px) { - .hpm-body, - .hpm-body--nav-collapsed { - grid-template-columns: minmax(0, 1fr); - } - - .hpm-nav { - position: static; - flex-direction: row; - align-items: center; - gap: 6px; - overflow-x: auto; - scrollbar-width: thin; - } - - .hpm-nav-toggle, - .hpm-nav-heading { - display: none; - } - - .hpm-nav-item { - width: auto; - flex: 0 0 auto; - border-left: 0; - border-bottom: 2px solid transparent; - border-radius: 6px; - } - - .hpm-nav-item--current { - border-left-color: transparent; - border-bottom-color: var(--hpm-accent, var(--hpm-ok)); - } - - /* The labels come back here even when the rail was left collapsed on a wider screen. */ - .hpm-body--nav-collapsed .hpm-nav-text, - .hpm-body--nav-collapsed .hpm-nav-count { - display: inline; - } -} - @media (max-width: 820px) { .hpm-list-head { display: none; @@ -1926,3 +1881,72 @@ font-size: 11px; letter-spacing: 0; } + +/* + * A disabled control that still looks live invites the click it will ignore -- which is the same + * complaint as the interval picker silently doing nothing, moved one step later. The browser's own + * 0.7 opacity is not enough against this palette. + */ +.hpm-field select:disabled, +.hpm-field input:disabled { + color: var(--hpm-faint); + background: var(--hpm-panel2); + border-style: dashed; + cursor: not-allowed; +} + +.hpm-field:has(select:disabled) > label, +.hpm-field:has(input:disabled) > label { + color: var(--hpm-faint); +} +/* + * Below this a rail costs more width than it earns, so it becomes a strip above the list: one + * horizontally scrolling row of the same items. Still every entry, still no JavaScript, and nothing + * overlays the content -- which is what an off-canvas drawer would need a focus trap to do honestly. + */ +@media (max-width: 820px) { + .hpm-body, + .hpm-body--nav-collapsed { + grid-template-columns: minmax(0, 1fr); + } + + .hpm-nav { + position: static; + padding: 6px; + } + + /* The items live in this wrapper, so this is what has to become the row. */ + .hpm-nav-body { + display: flex; + flex-direction: row; + align-items: center; + gap: 6px; + overflow-x: auto; + scrollbar-width: thin; + scrollbar-color: var(--hpm-line2) transparent; + } + + .hpm-nav-toggle, + .hpm-nav-heading { + display: none; + } + + .hpm-nav-item { + width: auto; + flex: 0 0 auto; + border-left: 0; + border-bottom: 2px solid transparent; + border-radius: 6px; + } + + .hpm-nav-item--current { + border-left-color: transparent; + border-bottom-color: var(--hpm-accent, var(--hpm-ok)); + } + + /* The labels come back here even when the rail was left collapsed on a wider screen. */ + .hpm-body--nav-collapsed .hpm-nav-text, + .hpm-body--nav-collapsed .hpm-nav-count { + display: inline; + } +} diff --git a/src/Healthie.DependencyInjection/TimerPulseScheduler.cs b/src/Healthie.DependencyInjection/TimerPulseScheduler.cs index 8e504d2..6e47593 100644 --- a/src/Healthie.DependencyInjection/TimerPulseScheduler.cs +++ b/src/Healthie.DependencyInjection/TimerPulseScheduler.cs @@ -275,8 +275,9 @@ public bool TryValidateSchedule(PulseSchedule schedule, out string? error) } catch (CronFormatException ex) { - error = $"Expected standard Unix cron -- five fields, or six with a leading seconds " + - $"field. {ex.Message}"; + // Cronos names the field and the range it wanted, which is more use than restating the + // format -- the field this is shown beside already gives an example of one. + error = ex.Message; return false; } diff --git a/tests/Healthie.Tests.E2E/DashboardGroupingTests.cs b/tests/Healthie.Tests.E2E/DashboardGroupingTests.cs index 510ef33..83698ec 100644 --- a/tests/Healthie.Tests.E2E/DashboardGroupingTests.cs +++ b/tests/Healthie.Tests.E2E/DashboardGroupingTests.cs @@ -22,6 +22,9 @@ public class DashboardGroupingTests(BrowserFixture browser) private const string TargetGroup = "Data Stores"; + /// The one checker in the sample that runs on a cron expression rather than an interval. + private const string CronChecker = "TLS Certificate"; + private static ILocator RowFor(IPage page, string displayName) => page.Locator(".hpm-row").Filter(new() { HasTextString = displayName }); @@ -74,16 +77,128 @@ public async Task Dashboard_WhenGrouped_ShowsEveryCheckerExactlyOnce(ProviderSet await using var app = await SampleApp.StartAsync(setup, Ct); var page = await OpenDashboardAsync(app); - var flatCount = await page.Locator(".hpm-row").CountAsync(); + // The board opens grouped, so this is the state under test without touching anything. + await page.Locator(".hpm-group").First.WaitForAsync(); + var groupedCount = await page.Locator(".hpm-row").CountAsync(); + Assert.Equal(1, await RowFor(page, TargetChecker).CountAsync()); + + // The same checkers laid out flat: the button goes the other way now, which makes this the + // comparison the test always wanted rather than a count taken before grouping was applied. await page.GetByRole(AriaRole.Button, new() { Name = "GROUP", Exact = true }).ClickAsync(); - await page.Locator(".hpm-group").First.WaitForAsync(); + await Assertions.Expect(page.Locator(".hpm-group")).ToHaveCountAsync(0); - Assert.Equal(flatCount, await page.Locator(".hpm-row").CountAsync()); + Assert.Equal(groupedCount, await page.Locator(".hpm-row").CountAsync()); Assert.Equal(1, await RowFor(page, TargetChecker).CountAsync()); browser.AssertNoErrors(page); } + /// + /// The board opens sectioned by group rather than as one flat list, and the GROUP button + /// reflects that rather than inviting a click that turns it off unannounced. + /// + [Theory] + [MemberData(nameof(DashboardTests.Setups), MemberType = typeof(DashboardTests))] + public async Task Dashboard_OnOpen_IsSectionedByGroup(ProviderSetup setup) + { + await using var app = await SampleApp.StartAsync(setup, Ct); + var page = await OpenDashboardAsync(app); + + await Assertions.Expect(page.Locator(".hpm-group").First).ToBeVisibleAsync(); + await Assertions.Expect(page.GetByRole(AriaRole.Button, new() { Name = "GROUP", Exact = true })) + .ToHaveAttributeAsync("aria-pressed", "true"); + + browser.AssertNoErrors(page); + } + + /// + /// The side menu narrows the list to one group, and says which one it is narrowed to. + /// + [Theory] + [MemberData(nameof(DashboardTests.Setups), MemberType = typeof(DashboardTests))] + public async Task Dashboard_SideMenu_NarrowsTheListToOneGroup(ProviderSetup setup) + { + await using var app = await SampleApp.StartAsync(setup, Ct); + var page = await OpenDashboardAsync(app); + + var everything = await page.Locator(".hpm-row").CountAsync(); + var inTarget = await GroupFor(page, TargetGroup).Locator(".hpm-row").CountAsync(); + Assert.True(inTarget > 0 && inTarget < everything); + + var entry = page.Locator(".hpm-nav-item").Filter(new() { HasTextString = TargetGroup }); + await entry.ClickAsync(); + + await Assertions.Expect(page.Locator(".hpm-row")).ToHaveCountAsync(inTarget); + await Assertions.Expect(entry).ToHaveAttributeAsync("aria-current", "true"); + + // Picking it again is the way back, so the menu cannot strand anyone in one group. + await entry.ClickAsync(); + await Assertions.Expect(page.Locator(".hpm-row")).ToHaveCountAsync(everything); + + browser.AssertNoErrors(page); + } + + /// + /// A cron expression typed on the board reaches the scheduler, and a bad one is refused without + /// touching what is stored. + /// + /// + /// Ordered deliberately: the refusal is asserted first, so a passing "it applied" cannot be the + /// result of the field simply ignoring everything typed into it. + /// + [Theory] + [MemberData(nameof(DashboardTests.Setups), MemberType = typeof(DashboardTests))] + public async Task Dashboard_CronEditor_AppliesAValidExpressionAndRefusesABadOne(ProviderSetup setup) + { + await using var app = await SampleApp.StartAsync(setup, Ct); + var page = await OpenDashboardAsync(app); + + // Waiting on the panel's title, not on the box: the box is already visible for whichever + // checker was selected on load, so it tells you nothing about whether the click has landed. + await RowFor(page, CronChecker).ClickAsync(); + await Assertions.Expect(page.Locator(".hpm-sel-name")).ToHaveTextAsync(CronChecker); + + var box = page.Locator("#hpm-cron"); + var before = await box.InputValueAsync(); + Assert.False(string.IsNullOrWhiteSpace(before)); + + await box.FillAsync("99 99 * * *"); + await box.PressAsync("Enter"); + + await Assertions.Expect(page.Locator(".hpm-field-error")).ToBeVisibleAsync(); + await Assertions.Expect(RowFor(page, CronChecker).Locator(".hpm-rate")).ToHaveTextAsync(before); + + await box.FillAsync("0 6 * * MON-FRI"); + await box.PressAsync("Enter"); + + await Assertions.Expect(page.Locator(".hpm-field-error")).ToHaveCountAsync(0); + await Assertions.Expect(RowFor(page, CronChecker).Locator(".hpm-rate")).ToHaveTextAsync("0 6 * * MON-FRI"); + + browser.AssertNoErrors(page); + } + + /// + /// The interval picker is inert while a cron expression is in force, because the schedule + /// overrides it -- a live picker would store a value nothing reads. + /// + [Theory] + [MemberData(nameof(DashboardTests.Setups), MemberType = typeof(DashboardTests))] + public async Task Dashboard_ForACronChecker_DisablesTheIntervalPicker(ProviderSetup setup) + { + await using var app = await SampleApp.StartAsync(setup, Ct); + var page = await OpenDashboardAsync(app); + + await RowFor(page, CronChecker).ClickAsync(); + await Assertions.Expect(page.Locator(".hpm-sel-name")).ToHaveTextAsync(CronChecker); + await Assertions.Expect(page.Locator("#hpm-interval")).ToBeDisabledAsync(); + + await RowFor(page, TargetChecker).ClickAsync(); + await Assertions.Expect(page.Locator(".hpm-sel-name")).ToHaveTextAsync(TargetChecker); + await Assertions.Expect(page.Locator("#hpm-interval")).ToBeEnabledAsync(); + + browser.AssertNoErrors(page); + } + /// A group's header reports what is inside it, so its tallies must add up to that. [Theory] [MemberData(nameof(DashboardTests.Setups), MemberType = typeof(DashboardTests))] @@ -92,7 +207,6 @@ public async Task Dashboard_GroupHeader_TalliesMatchTheRowsInsideIt(ProviderSetu await using var app = await SampleApp.StartAsync(setup, Ct); var page = await OpenDashboardAsync(app); - await page.GetByRole(AriaRole.Button, new() { Name = "GROUP", Exact = true }).ClickAsync(); var group = GroupFor(page, TargetGroup); await group.WaitForAsync(); @@ -112,7 +226,6 @@ public async Task Dashboard_CollapsingAGroup_HidesItsRowsAndLeavesTheOthers(Prov await using var app = await SampleApp.StartAsync(setup, Ct); var page = await OpenDashboardAsync(app); - await page.GetByRole(AriaRole.Button, new() { Name = "GROUP", Exact = true }).ClickAsync(); var group = GroupFor(page, TargetGroup); await group.WaitForAsync(); diff --git a/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs b/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs index 60360b8..e0aae4a 100644 --- a/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs +++ b/tests/Healthie.Tests.Unit/ScheduleMutationTests.cs @@ -1,4 +1,4 @@ -using Healthie.Abstractions; +using Healthie.Abstractions; using Healthie.Abstractions.Enums; using Healthie.Abstractions.Models; using Healthie.Abstractions.Scheduling; @@ -20,19 +20,30 @@ public class ScheduleMutationTests { private static CancellationToken Ct => TestContext.Current.CancellationToken; - private sealed class Checker(IStateProvider provider, PulseSchedule schedule) - : PulseChecker(provider, schedule) + /// + /// A real on a cron schedule, because the behaviour under test lives + /// in the base class rather than in the interface. + /// + /// + /// Takes only a state provider, like the other checkers in this assembly: assembly scanning + /// registers every concrete here, so one whose constructor the + /// container cannot satisfy fails every registration test in the suite rather than only its own. + /// + internal sealed class CronScheduledChecker(IStateProvider stateProvider) + : PulseChecker(stateProvider, PulseSchedule.Cron(InitialCron)) { + public const string InitialCron = "0 3 * * *"; + public override string Name => "schedule-target"; public override Task CheckAsync(CancellationToken cancellationToken = default) => Task.FromResult(new PulseCheckerResult(PulseCheckerHealth.Healthy, "ok")); } - private static async Task<(Checker Checker, IStateProvider Provider)> DailyCheckerAsync() + private static async Task<(CronScheduledChecker Checker, IStateProvider Provider)> DailyCheckerAsync() { var provider = new InMemoryStateProvider(); - var checker = new Checker(provider, PulseSchedule.Cron("0 3 * * *")); + var checker = new CronScheduledChecker(provider); // Seeds the initial state, which is what makes the cron schedule the stored one. await checker.TriggerAsync(Ct); @@ -162,7 +173,7 @@ public void TimerScheduler_AcceptsAPeriodWithoutOpinion() public async Task PulsesScheduler_RefusingASchedule_LeavesTheStoredOneAlone() { var provider = new InMemoryStateProvider(); - using var checker = new Checker(provider, PulseSchedule.Cron("0 3 * * *")); + using var checker = new CronScheduledChecker(provider); await checker.TriggerAsync(Ct); using var scheduler = new TimerPulseScheduler(); @@ -171,6 +182,6 @@ public async Task PulsesScheduler_RefusingASchedule_LeavesTheStoredOneAlone() await Assert.ThrowsAsync(() => pulses.SetScheduleAsync("schedule-target", PulseSchedule.Cron("99 99 * * *"), Ct)); - Assert.Equal("0 3 * * *", (await StateOf(provider)).Schedule?.CronExpression); + Assert.Equal(CronScheduledChecker.InitialCron, (await StateOf(provider)).Schedule?.CronExpression); } } From ff6a326e149aac63ff2f5f4f6f98c05e1ca19bd8 Mon Sep 17 00:00:00 2001 From: Ivan Vydrin Date: Thu, 30 Jul 2026 14:06:37 +0300 Subject: [PATCH 10/12] Rebuild alerts as a view, persist the history, and surface sinks and metrics The alerts list was a full-bleed band that shoved the whole board down when it opened, edge to edge with none of the card treatment every other panel has and a CLOSE button spanning half the width. It is now a view in the middle column, switched from the side menu alongside a new metrics view, so opening it changes one column instead of displacing the page. The history is written through the application's own IStateProvider, so a deployment on CosmosDB, Postgres or Redis keeps its alerts across a redeploy and one on the in-memory provider does not. No second storage contract, and no provider had to learn about alerts. Reads are paged -- an operator arriving after a restart is looking for what happened before it, which is exactly what a last-twenty list throws away -- and the view says how many are kept, of what cap, in which provider. Sinks are listed whether or not they have done anything, with delivered and failed counts and the last error. "Nothing has alerted" and "nothing is configured to deliver" looked identical and mean opposite things. A sink that recovers stops being shown as failing, because what matters is whether it is working now. Alerting is configurable from the board: minimum severity, deduplication window, delivery timeout and recovery alerts, which are the four the dispatcher reads on every alert rather than snapshotting. Its queue capacity and history length are fixed when it is built, so they are shown as facts rather than offered as controls that would quietly do nothing. A test alert goes through the real sinks, because the only way to find out a webhook URL is wrong is to use it. Metrics come from a MeterListener on the Healthie.NET meter, opt-in through AddHealthieMetrics: it reads the instruments this library already emits without competing with an OpenTelemetry exporter on the same meter. Overlapped triggers get their own card -- a checker whose check outlasts its interval looks healthy and is quietly running at a fraction of the rate it was asked to, and that is the only place it shows. Three E2E tests interacted with the detail panel straight after clicking a row. The panel's controls are already on screen for whatever was selected on load, so waiting on them proved nothing: the tag test was editing the wrong checker and failing elsewhere. They wait on the panel title now, through one helper. --- samples/Healthie.Sample.BlazorUI/Program.cs | 6 +- .../Insights/AlertPage.cs | 24 + .../Insights/AlertSinkStatus.cs | 23 + .../Insights/IAlertConfiguration.cs | 53 ++ .../Insights/IAlertInsights.cs | 28 +- .../Insights/IMetricsInsights.cs | 23 + .../Insights/MetricsSnapshot.cs | 40 ++ src/Healthie.Alerting/AlertConfiguration.cs | 53 ++ src/Healthie.Alerting/AlertDispatcher.cs | 41 +- src/Healthie.Alerting/AlertHistory.cs | 203 ++++++- src/Healthie.Alerting/StartupExtensions.cs | 23 +- .../Components/HealthieDashboard.Insights.cs | 121 +++- .../Components/HealthieDashboard.razor | 327 +++++++++-- .../Components/HealthieDashboard.razor.cs | 39 ++ src/Healthie.Dashboard/wwwroot/healthie.css | 529 +++++++++--------- .../MeterMetricsInsights.cs | 145 +++++ .../StartupExtensions.cs | 22 + .../DashboardGroupingTests.cs | 17 +- tests/Healthie.Tests.E2E/DashboardTests.cs | 2 +- tests/Healthie.Tests.Unit/AlertingTests.cs | 4 +- tests/Healthie.Tests.Unit/InsightsTests.cs | 114 +++- 21 files changed, 1485 insertions(+), 352 deletions(-) create mode 100644 src/Healthie.Abstractions/Insights/AlertPage.cs create mode 100644 src/Healthie.Abstractions/Insights/AlertSinkStatus.cs create mode 100644 src/Healthie.Abstractions/Insights/IAlertConfiguration.cs create mode 100644 src/Healthie.Abstractions/Insights/IMetricsInsights.cs create mode 100644 src/Healthie.Abstractions/Insights/MetricsSnapshot.cs create mode 100644 src/Healthie.Alerting/AlertConfiguration.cs create mode 100644 src/Healthie.DependencyInjection/MeterMetricsInsights.cs diff --git a/samples/Healthie.Sample.BlazorUI/Program.cs b/samples/Healthie.Sample.BlazorUI/Program.cs index aedf66e..884432c 100644 --- a/samples/Healthie.Sample.BlazorUI/Program.cs +++ b/samples/Healthie.Sample.BlazorUI/Program.cs @@ -1,4 +1,4 @@ -using Healthie.DependencyInjection; +using Healthie.DependencyInjection; using Healthie.Sample.BlazorUI.Components; using Healthie.Scheduling.Quartz; using Healthie.StateProviding.CosmosDb; @@ -38,7 +38,9 @@ options.MinimumSeverity = PulseCheckerHealth.Suspicious; options.DeduplicationWindow = TimeSpan.FromSeconds(20); }) - .AddHealthieUptime(); + .AddHealthieUptime() + // Reads the library's own meter in-process, so the board can show what it has counted. + .AddHealthieMetrics(); // Leader election off by default: with one replica it is always the leader, and the badge would be // a permanent reassurance about a problem nobody has. Healthie:LeaderElection=true shows it. diff --git a/src/Healthie.Abstractions/Insights/AlertPage.cs b/src/Healthie.Abstractions/Insights/AlertPage.cs new file mode 100644 index 0000000..4e856cb --- /dev/null +++ b/src/Healthie.Abstractions/Insights/AlertPage.cs @@ -0,0 +1,24 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// One page of the alert history, newest first. +/// +/// The alerts on this page. +/// How many are held in total, which is what the pager counts against. +/// +/// The name of the state provider the history is kept in, so the board can say where it went and +/// whether it will still be there after a restart. +/// +/// +/// How many alerts are kept before the oldest is discarded. The history is bounded on purpose: it is +/// a window onto what happened, and the record of record is wherever the sinks deliver to. +/// +public sealed record AlertPage( + IReadOnlyList Alerts, + int Total, + string StoredIn, + int Capacity) +{ + /// An empty page, for a provider that has never been written to. + public static AlertPage Empty(string storedIn, int capacity) => new([], 0, storedIn, capacity); +} diff --git a/src/Healthie.Abstractions/Insights/AlertSinkStatus.cs b/src/Healthie.Abstractions/Insights/AlertSinkStatus.cs new file mode 100644 index 0000000..2f24802 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/AlertSinkStatus.cs @@ -0,0 +1,23 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// One place alerts are delivered to, and how it has been going. +/// +/// The sink's type name, which is what identifies it on the board. +/// How many alerts it has accepted. +/// How many it refused, timed out on, or threw over. +/// The most recent failure's message, or null if it has never failed. +/// +/// Counted per sink rather than in total because that is the question being asked: with three sinks +/// configured, "one alert did not get through" is a very different morning from "Slack is down". +/// +public sealed record AlertSinkStatus(string Name, int Delivered, int Failed, string? LastError) +{ + /// Whether this sink is currently getting alerts through. + /// + /// Judged on the last attempt rather than the ratio: a sink that failed a hundred times and is + /// working now is working, and one that has delivered thousands and just started failing is the + /// thing worth showing in red. + /// + public bool IsHealthy => LastError is null; +} diff --git a/src/Healthie.Abstractions/Insights/IAlertConfiguration.cs b/src/Healthie.Abstractions/Insights/IAlertConfiguration.cs new file mode 100644 index 0000000..de5584d --- /dev/null +++ b/src/Healthie.Abstractions/Insights/IAlertConfiguration.cs @@ -0,0 +1,53 @@ +using Healthie.Abstractions.Enums; + +namespace Healthie.Abstractions.Insights; + +/// +/// The alerting settings that can be changed while the application is running. +/// +/// +/// +/// Only the ones that take effect. The dispatcher reads these four on every alert, so a change here +/// applies to the next one; its queue capacity and history length are fixed when it is built and +/// cannot be moved without a restart, so they are not offered. Showing a control that quietly does +/// nothing is the failure this whole release has been fixing. +/// +/// +/// In memory and not persisted: this is the same object the host configured at startup, so a +/// restart returns to whatever AddHealthieAlerts was given. The board says so. +/// +/// +public interface IAlertConfiguration +{ + /// What the dispatcher is using now. + AlertSettings Current { get; } + + /// Applies new settings, from the next alert onwards. + /// The settings to apply. + void Apply(AlertSettings settings); + + /// + /// Sends a test alert to every registered sink and reports what happened. + /// + /// A token to monitor for cancellation requests. + /// Each sink, and whether it took the alert. + /// + /// The only way to find out that a webhook URL is wrong is to use it. Deduplication and the + /// severity threshold are bypassed, because the point is to exercise delivery rather than to + /// decide whether this alert is worth sending. + /// + Task> SendTestAlertAsync(CancellationToken cancellationToken = default); +} + +/// +/// The alerting settings a running application will honour a change to. +/// +/// The least severe health that raises an alert. +/// Whether returning to healthy raises one too. +/// How long the same checker is quiet for after alerting. +/// How long a single sink gets before it is abandoned. +public sealed record AlertSettings( + PulseCheckerHealth MinimumSeverity, + bool SendRecoveries, + TimeSpan DeduplicationWindow, + TimeSpan DeliveryTimeout); diff --git a/src/Healthie.Abstractions/Insights/IAlertInsights.cs b/src/Healthie.Abstractions/Insights/IAlertInsights.cs index 68c2cd6..6b07af5 100644 --- a/src/Healthie.Abstractions/Insights/IAlertInsights.cs +++ b/src/Healthie.Abstractions/Insights/IAlertInsights.cs @@ -10,13 +10,20 @@ namespace Healthie.Abstractions.Insights; public interface IAlertInsights { /// - /// The most recent alerts, newest first. + /// One page of alerts, newest first. /// - /// How many to return. + /// How many of the newest to pass over. + /// How many to return. /// A token to monitor for cancellation requests. - /// The alerts, newest first. - Task> GetRecentAlertsAsync( - int limit, + /// The page, and how many there are in total. + /// + /// Paged rather than capped at a handful because the history outlives the process: an operator + /// arriving after a restart is looking for what happened before it, which is exactly the part a + /// "last twenty" list throws away. + /// + Task GetAlertsAsync( + int skip, + int take, CancellationToken cancellationToken = default); /// @@ -27,4 +34,15 @@ Task> GetRecentAlertsAsync( /// board is where somebody would look to find out that alerting itself is behind. /// int DroppedCount { get; } + + /// + /// Where alerts are being delivered, and how that is going. + /// + /// + /// Empty when nothing is registered to deliver to, which is the case worth surfacing: an + /// application that installed alerting and never configured a sink raises alerts that reach + /// nobody, and a board showing a healthy list of alerts looks exactly like one that is notifying + /// people. Startup logs it once; this is what puts it where somebody is looking. + /// + IReadOnlyList Sinks { get; } } diff --git a/src/Healthie.Abstractions/Insights/IMetricsInsights.cs b/src/Healthie.Abstractions/Insights/IMetricsInsights.cs new file mode 100644 index 0000000..899af84 --- /dev/null +++ b/src/Healthie.Abstractions/Insights/IMetricsInsights.cs @@ -0,0 +1,23 @@ +namespace Healthie.Abstractions.Insights; + +/// +/// What the library's own instruments have counted, for a board to show. +/// +/// +/// +/// The Healthie.NET meter is the real home of this data and OpenTelemetry is the real way to +/// read it -- an APM keeps history, does percentiles properly, and alerts on them. This exists for +/// the case where there is no APM in front of the operator, which is most first runs and most small +/// deployments: the numbers are already being emitted, and nobody was looking at them. +/// +/// +/// So: a live window, not a time series. Nothing here is persisted or survives a restart, and the +/// board says as much beside it. +/// +/// +public interface IMetricsInsights +{ + /// What has been counted since the process started. + /// A token to monitor for cancellation requests. + MetricsSnapshot Snapshot(CancellationToken cancellationToken = default); +} diff --git a/src/Healthie.Abstractions/Insights/MetricsSnapshot.cs b/src/Healthie.Abstractions/Insights/MetricsSnapshot.cs new file mode 100644 index 0000000..edc13db --- /dev/null +++ b/src/Healthie.Abstractions/Insights/MetricsSnapshot.cs @@ -0,0 +1,40 @@ +using Healthie.Abstractions.Enums; + +namespace Healthie.Abstractions.Insights; + +/// +/// What the library's instruments have counted since the process started. +/// +/// Total checks completed. +/// Checks completed, by the health each reported. +/// Checks that produced a state different from the stored one. +/// Triggers that returned immediately because the previous check had not finished. +/// Mean check duration, or null before anything has run. +/// The slowest single check, or null before anything has run. +/// When collection started, which is when the process did. +public sealed record MetricsSnapshot( + long Checks, + IReadOnlyDictionary ResultsByHealth, + long Transitions, + long OverlappedTriggers, + TimeSpan? MeanDuration, + TimeSpan? SlowestDuration, + DateTime Since) +{ + /// + /// The share of completed checks that reported healthy, or null before anything has run. + /// + public double? HealthyShare => Checks == 0 + ? null + : 100d * ResultsByHealth.GetValueOrDefault(PulseCheckerHealth.Healthy) / Checks; + + /// + /// Whether any trigger has been skipped for overlapping the previous one. + /// + /// + /// Called out separately because it is the one number here that means something is wrong rather + /// than something happened: a checker whose check outlasts its own interval looks healthy and is + /// quietly running at a fraction of the rate it was asked to. This is the only place it shows. + /// + public bool HasOverlaps => OverlappedTriggers > 0; +} diff --git a/src/Healthie.Alerting/AlertConfiguration.cs b/src/Healthie.Alerting/AlertConfiguration.cs new file mode 100644 index 0000000..8f32604 --- /dev/null +++ b/src/Healthie.Alerting/AlertConfiguration.cs @@ -0,0 +1,53 @@ +using Healthie.Abstractions.Insights; + +namespace Healthie.Alerting; + +/// +/// Lets the dashboard change the alerting settings that a running dispatcher honours. +/// +/// +/// Writes straight to the options object the dispatcher holds, which is the same singleton the host +/// configured: the dispatcher reads these four on every alert rather than snapshotting them, so a +/// change applies to the next one with nothing to restart or re-register. +/// +/// The options the dispatcher is reading. +/// The dispatcher, for sending a test alert through the real sinks. +internal sealed class AlertConfiguration(HealthieAlertOptions options, AlertDispatcher dispatcher) + : IAlertConfiguration +{ + /// + public AlertSettings Current => new( + options.MinimumSeverity, + options.SendRecoveries, + options.DeduplicationWindow, + options.DeliveryTimeout); + + /// + public void Apply(AlertSettings settings) + { + ArgumentNullException.ThrowIfNull(settings); + + if (settings.DeduplicationWindow < TimeSpan.Zero) + { + throw new ArgumentException( + "A deduplication window cannot be negative. Zero means alert on every change.", + nameof(settings)); + } + + if (settings.DeliveryTimeout <= TimeSpan.Zero) + { + throw new ArgumentException( + "A delivery timeout must be positive, or no sink would ever get long enough to answer.", + nameof(settings)); + } + + options.MinimumSeverity = settings.MinimumSeverity; + options.SendRecoveries = settings.SendRecoveries; + options.DeduplicationWindow = settings.DeduplicationWindow; + options.DeliveryTimeout = settings.DeliveryTimeout; + } + + /// + public Task> SendTestAlertAsync(CancellationToken cancellationToken = default) => + dispatcher.SendTestAlertAsync(cancellationToken); +} diff --git a/src/Healthie.Alerting/AlertDispatcher.cs b/src/Healthie.Alerting/AlertDispatcher.cs index 61df573..e7f07f3 100644 --- a/src/Healthie.Alerting/AlertDispatcher.cs +++ b/src/Healthie.Alerting/AlertDispatcher.cs @@ -1,3 +1,4 @@ +using Healthie.Abstractions.Insights; using Healthie.Abstractions; using Healthie.Abstractions.Enums; using Healthie.Abstractions.Models; @@ -99,6 +100,11 @@ public override Task StartAsync(CancellationToken cancellationToken) "Alerting is registered but no sink is; alerts will show on the dashboard and be sent nowhere."); } + foreach (var sink in _sinks) + { + _history?.Register(sink.GetType().Name); + } + Subscribe(); return base.StartAsync(cancellationToken); @@ -245,17 +251,21 @@ private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) using var timeout = CancellationTokenSource.CreateLinkedTokenSource(stoppingToken); timeout.CancelAfter(_options.DeliveryTimeout); + var name = sink.GetType().Name; + try { await sink.SendAsync(alert, timeout.Token).ConfigureAwait(false); + _history?.RecordDelivery(name, error: null); } catch (OperationCanceledException) when (!stoppingToken.IsCancellationRequested) { delivered = false; + _history?.RecordDelivery(name, $"Did not respond within {_options.DeliveryTimeout}."); _logger?.LogWarning( "Alert sink {Sink} did not deliver the alert for '{CheckerName}' within {Timeout}.", - sink.GetType().Name, + name, alert.CheckerName, _options.DeliveryTimeout); } @@ -266,11 +276,12 @@ private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) catch (Exception ex) { delivered = false; + _history?.RecordDelivery(name, ex.Message); _logger?.LogError( ex, "Alert sink {Sink} failed to deliver the alert for '{CheckerName}'.", - sink.GetType().Name, + name, alert.CheckerName); } } @@ -278,6 +289,32 @@ private async Task DeliverAsync(Alert alert, CancellationToken stoppingToken) _history?.Record(alert, delivered); } + /// + /// Puts one alert through every sink and reports what each did, bypassing the queue. + /// + /// A token to monitor for cancellation requests. + /// Each sink's tally, including this attempt. + /// + /// Delivered directly rather than enqueued so the caller can be told the outcome; the queue + /// exists to keep a slow sink away from the checks, and nothing is checking here. + /// + public async Task> SendTestAlertAsync(CancellationToken cancellationToken = default) + { + var alert = new Alert( + "healthie.test", + "Healthie test alert", + Group: null, + Tags: [], + PulseCheckerHealth.Healthy, + PulseCheckerHealth.Unhealthy, + "Test alert raised from the dashboard. Nothing is wrong.", + DateTime.UtcNow); + + await DeliverAsync(alert, cancellationToken).ConfigureAwait(false); + + return _history?.Sinks ?? []; + } + /// public override void Dispose() { diff --git a/src/Healthie.Alerting/AlertHistory.cs b/src/Healthie.Alerting/AlertHistory.cs index 1eda498..61ace8b 100644 --- a/src/Healthie.Alerting/AlertHistory.cs +++ b/src/Healthie.Alerting/AlertHistory.cs @@ -1,34 +1,125 @@ using Healthie.Abstractions.Insights; +using Healthie.Abstractions.StateProviding; +using Microsoft.Extensions.Logging; namespace Healthie.Alerting; /// -/// The last few alerts, kept so the dashboard can show them. +/// The alerts that have been raised, kept so the dashboard can show them. /// /// /// /// Alerting is fire-and-forget by design: an alert goes to its sinks and is gone. That is right for -/// delivery and wrong for the one screen an operator looks at, where "has anything fired recently, -/// and did it get through" is the first question. This keeps just enough to answer it. +/// delivery and wrong for the one screen an operator looks at, where "what fired, and did it get +/// through" is the first question -- and it is asked most often just after a restart, about what +/// happened before it. /// /// -/// Bounded and in memory on purpose. It is a window onto what just happened, not a record -- the -/// record is wherever the sinks deliver to. Nothing here is worth a round trip to a database or -/// worth surviving a restart. +/// So the log is written through the application's own : a deployment on +/// CosmosDB, Postgres or Redis keeps its alert history across a redeploy, and one left on the +/// in-memory provider does not. There is no second storage contract to configure, and no provider +/// had to learn about alerts. +/// +/// +/// Bounded, and written whole on each alert. Both are affordable because alerts are transitions +/// rather than checks -- a checker running every second raises nothing until its health changes -- and +/// both are deliberate: an unbounded log in a state document would grow without limit, and the record +/// of record is wherever the sinks deliver to. /// /// -/// How many alerts to keep. -public sealed class AlertHistory(int capacity) : IAlertInsights +/// How many alerts to keep before the oldest is discarded. +/// Where to persist the log, or null to keep it in memory only. +/// An optional logger for diagnostic output. +public sealed class AlertHistory( + int capacity, + IStateProvider? stateProvider = null, + ILogger? logger = null) : IAlertInsights { + /// The key the whole log is stored under. + /// + /// Deliberately not a checker's name, and prefixed so it cannot collide with one: a state + /// provider is keyed by checker name, and this is the one entry that is not a checker. + /// + private const string StorageKey = "healthie.alerts.log"; + private readonly Queue _recent = new(capacity); // A plain object, not System.Threading.Lock: this package targets net8.0 as well. private readonly object _gate = new(); + private bool _loaded; + + private readonly Dictionary _sinks = []; + private int _dropped; /// public int DroppedCount => Volatile.Read(ref _dropped); + /// + public IReadOnlyList Sinks + { + get + { + lock (_gate) + { + return [.. _sinks.Select(entry => new AlertSinkStatus( + entry.Key, entry.Value.Delivered, entry.Value.Failed, entry.Value.LastError))]; + } + } + } + + /// Registers a sink so it appears on the board before it has done anything. + /// The sink's type name. + /// + /// Named at startup rather than discovered on first delivery, because "no sinks configured" and + /// "sinks configured, nothing has alerted yet" look identical otherwise and mean opposite things. + /// + public void Register(string name) + { + lock (_gate) + { + _sinks.TryAdd(name, new SinkTally()); + } + } + + /// Records the outcome of one sink's attempt at one alert. + /// The sink's type name. + /// The failure, or null when it was accepted. + public void RecordDelivery(string name, string? error) + { + lock (_gate) + { + if (!_sinks.TryGetValue(name, out var tally)) + { + tally = new SinkTally(); + _sinks[name] = tally; + } + + if (error is null) + { + tally.Delivered++; + + // Cleared on success, so a sink that failed once and recovered stops being shown as + // broken -- what matters is whether it is working now. + tally.LastError = null; + } + else + { + tally.Failed++; + tally.LastError = error; + } + } + } + + private sealed class SinkTally + { + public int Delivered { get; set; } + + public int Failed { get; set; } + + public string? LastError { get; set; } + } + /// Records an alert and whether every sink took it. /// The alert that was raised. /// Whether every sink accepted it. @@ -43,6 +134,8 @@ public void Record(Alert alert, bool delivered) alert.OccurredAt, delivered); + List toPersist; + lock (_gate) { // Trims after enqueuing rather than before. Dropping the oldest first has to special-case @@ -53,23 +146,107 @@ public void Record(Alert alert, bool delivered) { _recent.Dequeue(); } + + toPersist = [.. _recent]; } + + // Outside the lock: this is a round trip to the state store, and holding a lock across it + // would stall every reader of the board for the duration of a database write. + _ = PersistAsync(toPersist); } /// Records that an alert never reached the queue. public void RecordDropped() => Interlocked.Increment(ref _dropped); /// - public Task> GetRecentAlertsAsync( - int limit, + public async Task GetAlertsAsync( + int skip, + int take, CancellationToken cancellationToken = default) { + await EnsureLoadedAsync(cancellationToken).ConfigureAwait(false); + lock (_gate) { - IReadOnlyList newestFirst = - [.. _recent.Reverse().Take(Math.Max(limit, 0))]; + var newestFirst = _recent.Reverse().ToList(); + + IReadOnlyList page = + [.. newestFirst.Skip(Math.Max(skip, 0)).Take(Math.Max(take, 0))]; + + return new AlertPage(page, newestFirst.Count, StoreName, capacity); + } + } + + /// What the board calls the place this history is kept. + private string StoreName => stateProvider?.GetType().Name ?? "memory"; + + /// + /// Reads the stored log once, the first time anything asks for a page. + /// + /// + /// Lazily rather than at startup: the dispatcher subscribes while the host is still starting, and + /// a state provider may not have finished initializing its container or table by then. Nothing + /// needs the history until somebody opens the board. + /// + private async Task EnsureLoadedAsync(CancellationToken cancellationToken) + { + if (_loaded || stateProvider is null) + { + return; + } + + // Set before the read, not after: a failed read must not leave every later page request + // retrying a store that is not answering. + _loaded = true; + + try + { + var stored = await stateProvider + .GetStateAsync>(StorageKey, cancellationToken) + .ConfigureAwait(false); - return Task.FromResult(newestFirst); + if (stored is null or { Count: 0 }) + { + return; + } + + lock (_gate) + { + // In front of anything raised while this was loading, then trimmed: the stored log is + // older by definition, and a restart that raises an alert immediately must not lose it. + var live = _recent.ToList(); + _recent.Clear(); + + foreach (var insight in stored.Concat(live).TakeLast(capacity)) + { + _recent.Enqueue(insight); + } + } + } + catch (Exception ex) + { + // A history that cannot be read is not a reason to fail the board: it shows what this + // process has seen, and says where the rest was meant to be. + logger?.LogWarning(ex, "Could not read the stored alert history from the state provider."); + } + } + + private async Task PersistAsync(List log) + { + if (stateProvider is null) + { + return; + } + + try + { + await stateProvider.SetStateAsync(StorageKey, log).ConfigureAwait(false); + } + catch (Exception ex) + { + // Never propagated. This runs on the alert-delivery path, and a state store that is down + // must not take alerting down with it -- the alert has already reached its sinks. + logger?.LogWarning(ex, "Could not persist the alert history to the state provider."); } } } diff --git a/src/Healthie.Alerting/StartupExtensions.cs b/src/Healthie.Alerting/StartupExtensions.cs index 3e19d09..3d83561 100644 --- a/src/Healthie.Alerting/StartupExtensions.cs +++ b/src/Healthie.Alerting/StartupExtensions.cs @@ -1,4 +1,6 @@ using Healthie.Abstractions.Insights; +using Healthie.Abstractions.StateProviding; +using Microsoft.Extensions.Logging; using Microsoft.Extensions.DependencyInjection; using Microsoft.Extensions.DependencyInjection.Extensions; using Microsoft.Extensions.Hosting; @@ -37,10 +39,25 @@ public static IServiceCollection AddHealthieAlerts( // What the dashboard renders when this package is installed: the last few alerts and whether // they were delivered. Registered here rather than referenced there, so installing a // dashboard does not drag alerting in with it. - services.TryAddSingleton(new AlertHistory(options.HistoryLength)); + // Given the application's own state provider, so the alert log lands wherever checker state + // already does and survives a redeploy on any durable one. GetService, not GetRequiredService: + // alerting can be registered before AddHealthie has put a provider in. + services.TryAddSingleton(provider => new AlertHistory( + options.HistoryLength, + provider.GetService(), + provider.GetService>())); services.TryAddSingleton(p => p.GetRequiredService()); - - services.AddHostedService(); + services.TryAddSingleton(); + + // The dispatcher is resolved as itself and handed to the host, rather than registered + // straight as a hosted service: a test alert has to go through the sinks of the dispatcher + // that is actually running, not a second copy of it. Guarded by hand because + // TryAddEnumerable refuses a factory-built descriptor, and this must stay callable twice. + if (services.All(descriptor => descriptor.ServiceType != typeof(AlertDispatcher))) + { + services.AddSingleton(); + services.AddHostedService(provider => provider.GetRequiredService()); + } return services; } diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs b/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs index 351cc14..3def298 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.Insights.cs @@ -1,3 +1,4 @@ +using Healthie.Abstractions.Enums; using Healthie.Abstractions.Insights; using Microsoft.AspNetCore.Components; using Microsoft.Extensions.DependencyInjection; @@ -31,16 +32,29 @@ public sealed partial class HealthieDashboard [Inject] private IServiceProvider Services { get; set; } = default!; + /// How many alerts a page of the history holds. + private const int AlertsPerPage = 25; + private IUptimeInsights? _uptimeInsights; private IAlertInsights? _alertInsights; private ILeadershipInsights? _leadershipInsights; private IDiagnosisInsights? _diagnosisInsights; + private IMetricsInsights? _metricsInsights; + private IAlertConfiguration? _alertConfiguration; private UptimeInsight? _uptime; private string? _uptimeFor; - private IReadOnlyList _recentAlerts = []; - private bool _alertsOpen; + private AlertPage? _alerts; + private int _alertPage; + private bool _undeliveredOnly; + + private MetricsSnapshot? _metrics; + + private AlertSettings? _settingsDraft; + private string? _settingsError; + private IReadOnlyList? _testResult; + private bool _testing; private string? _diagnosis; private bool _diagnosing; @@ -59,6 +73,9 @@ private void ResolveInsights() _alertInsights = Services.GetService(); _leadershipInsights = Services.GetService(); _diagnosisInsights = Services.GetService(); + _metricsInsights = Services.GetService(); + _alertConfiguration = Services.GetService(); + _settingsDraft = _alertConfiguration?.Current; } /// Reads the uptime for the selected checker. @@ -84,7 +101,7 @@ private async Task LoadUptimeAsync(string? checkerName) _uptime = await _uptimeInsights.GetUptimeAsync(checkerName, UptimeWindow); } - /// Reads the recent alerts, newest first. + /// Reads one page of the alert history, newest first. private async Task LoadAlertsAsync() { if (_alertInsights is null) @@ -92,20 +109,93 @@ private async Task LoadAlertsAsync() return; } - _recentAlerts = await _alertInsights.GetRecentAlertsAsync(AlertsShown); + _alerts = await _alertInsights.GetAlertsAsync(_alertPage * AlertsPerPage, AlertsPerPage); + } + + /// The alerts this page shows, after the undelivered filter. + /// + /// Filtered here rather than in the query: the store pages over everything it holds, and a filter + /// pushed into it would make the page counts disagree with the pager. This narrows what is shown + /// on the page you are on, which is what the toggle says it does. + /// + private IReadOnlyList VisibleAlerts => + _alerts is null + ? [] + : _undeliveredOnly + ? [.. _alerts.Alerts.Where(alert => !alert.Delivered)] + : _alerts.Alerts; + + private int AlertPageCount => + _alerts is null || _alerts.Total == 0 ? 1 : (_alerts.Total + AlertsPerPage - 1) / AlertsPerPage; + + private async Task GoToAlertPageAsync(int page) + { + _alertPage = Math.Clamp(page, 0, AlertPageCount - 1); + + await LoadAlertsAsync(); } - /// How many alerts the panel lists. - private const int AlertsShown = 20; + private async Task ToggleUndeliveredOnlyAsync() + { + _undeliveredOnly = !_undeliveredOnly; + + await LoadAlertsAsync(); + } - private async Task ToggleAlertsAsync() + /// Reads what the library's own instruments have counted. + private void LoadMetrics() => _metrics = _metricsInsights?.Snapshot(); + + /// + /// Applies the alerting settings in the form, from the next alert onwards. + /// + private void ApplyAlertSettings() { - _alertsOpen = !_alertsOpen; + if (_alertConfiguration is null || _settingsDraft is null) + { + return; + } + + _settingsError = null; - if (_alertsOpen) + try + { + _alertConfiguration.Apply(_settingsDraft); + AddEvent("CONF", Status.Paused, "Alerting settings changed"); + } + catch (ArgumentException ex) { + _settingsError = ex.Message; + } + } + + /// + /// Sends one alert through the real sinks, because the only way to find out that a webhook URL is + /// wrong is to use it. + /// + private async Task SendTestAlertAsync() + { + if (_alertConfiguration is null || _testing) + { + return; + } + + _testing = true; + _testResult = null; + + try + { + _testResult = await _alertConfiguration.SendTestAlertAsync(); + AddEvent("TEST", Status.Ok, "Test alert sent to every sink"); await LoadAlertsAsync(); } + catch (Exception ex) + { + _settingsError = $"Could not send the test alert: {ex.Message}"; + } + finally + { + _testing = false; + } } /// Asks the model why a checker has been failing. @@ -142,6 +232,19 @@ private async Task ExplainAsync(string checkerName) /// Formats an uptime percentage the way the rest of the board formats one. private static string Percent(double value) => $"{value:0.##}%"; + /// A check duration, in the unit a person reads durations of checks in. + private static string Milliseconds(TimeSpan span) => + span.TotalMilliseconds < 1000 ? $"{span.TotalMilliseconds:0} ms" : $"{span.TotalSeconds:0.##} s"; + + /// The status class for a health, so an alert row is coloured like everything else. + private static string HealthClass(PulseCheckerHealth? health) => health switch + { + PulseCheckerHealth.Healthy => "hpm-ok", + PulseCheckerHealth.Suspicious => "hpm-warn", + PulseCheckerHealth.Unhealthy => "hpm-crit", + _ => "hpm-paused", + }; + /// A short, human length: "4m", "2h 10m". private static string Duration(TimeSpan span) => span switch { diff --git a/src/Healthie.Dashboard/Components/HealthieDashboard.razor b/src/Healthie.Dashboard/Components/HealthieDashboard.razor index aaa6d9c..1eb3b7e 100644 --- a/src/Healthie.Dashboard/Components/HealthieDashboard.razor +++ b/src/Healthie.Dashboard/Components/HealthieDashboard.razor @@ -74,8 +74,8 @@ title="@(_alertInsights.DroppedCount > 0 ? $"Recent alerts. {_alertInsights.DroppedCount} were dropped because the queue was full -- nobody was told about those." : "Alerts raised recently, and whether they reached their sinks")" - aria-expanded="@_alertsOpen" - @onclick="ToggleAlertsAsync"> + aria-pressed="@(_view == BoardView.Alerts ? "true" : "false")" + @onclick="@(() => ShowViewAsync(BoardView.Alerts))"> ALERTS@(_alertInsights.DroppedCount > 0 ? $" ({_alertInsights.DroppedCount} DROPPED)" : null) } @@ -137,52 +137,6 @@
- @* - Reads only, so it shows in read-only mode too: knowing what fired and whether it landed - is exactly what a read-only board is for. - *@ - @if (_alertsOpen && _alertInsights is not null) - { -
-
- RECENT ALERTS - @if (_alertInsights.DroppedCount > 0) - { - - @_alertInsights.DroppedCount DROPPED - - } -
- -
- - @if (_recentAlerts.Count == 0) - { -
Nothing has alerted yet. Alerts appear here as checkers change health.
- } - else - { -
    - @foreach (var alert in _recentAlerts) - { -
  • - @Relative(alert.OccurredAt) - @alert.DisplayName - - @(alert.PreviousHealth?.ToString() ?? "new") → @alert.CurrentHealth - - @alert.Message - @if (!alert.Delivered) - { - NOT DELIVERED - } -
  • - } -
- } -
- } - @* The event log at full size. The sidebar copy is a glance; this is the one to read when something has gone wrong and the interesting line has already scrolled away. @@ -440,10 +394,10 @@ @if (_alertInsights is not null) { - } + @if (_metricsInsights is not null) + { + + } + +
+ + @* + Where alerts go, before the alerts themselves: an empty list means nothing has + fired, and that reads completely differently depending on whether anything is + configured to deliver. + *@ +
+ @if (_alertInsights.Sinks.Count == 0) + { +
+ +
+
NO SINK CONFIGURED
+
+ Alerts are recorded here and sent nowhere. Add one with + AddHealthieSlackAlerts, AddHealthieWebhookAlerts, + AddHealthieMicrosoftTeamsAlerts or + AddHealthiePagerDutyAlerts. +
+
+
+ } + else + { + @foreach (var sink in _alertInsights.Sinks) + { +
+ +
+
@sink.Name
+
+ @sink.Delivered delivered · @sink.Failed failed + @if (sink.LastError is { } error) + { + — @error + } +
+
+
+ } + } +
+ + @if (VisibleAlerts.Count == 0) + { +
+ @(_undeliveredOnly + ? "Every alert on this page reached its sinks." + : "Nothing has alerted yet. Alerts appear here as checkers change health.") +
+ } + else + { +
    + @foreach (var alert in VisibleAlerts) + { +
  • +
    + @alert.DisplayName + + @(alert.PreviousHealth?.ToString().ToUpperInvariant() ?? "NEW") + + @alert.CurrentHealth.ToString().ToUpperInvariant() + +
    + @if (!alert.Delivered) + { + NOT DELIVERED + } + @Relative(alert.OccurredAt) +
    +
    @alert.Message
    +
  • + } +
+ + @if (AlertPageCount > 1) + { +
+ + PAGE @(_alertPage + 1) OF @AlertPageCount + +
+ } + } + + @* + Only the settings a running dispatcher actually honours. Its queue capacity and + history length are fixed when it is built, so they are shown as read-only facts + above rather than offered as controls that would quietly do nothing. + *@ + @if (_alertConfiguration is not null && Options.AllowMutations && _settingsDraft is { } draft) + { +
+

SETTINGS

+

+ Applies from the next alert. Held in memory, so a restart returns to whatever + AddHealthieAlerts was given. +

+ +
+
+ + +
+
+ + +
+
+ + +
+
+ + + + @if (_settingsError is not null) + { + + } + +
+ + +
+ + @if (_testResult is { } results) + { +
+ @if (results.Count == 0) + { + Nothing to deliver to, so the test alert was recorded and sent nowhere. + } + else + { + @results.Count(r => r.IsHealthy) of @results.Count sink(s) took it. + } +
+ } +
+ } + + } + else if (_view == BoardView.Metrics && _metricsInsights is not null) + { +
+
+
+

METRICS

+

+ @if (_metrics is { } snapshot) + { + Since @snapshot.Since.ToString("u") · this process only + } +

+
+
+ +
+ +

+ Read from the Healthie.NET meter in this process. It is a live count, + not a time series: nothing here survives a restart, and an OpenTelemetry exporter + on the same meter is the way to keep history. +

+ + @if (_metrics is { } m) + { +
+
+
CHECKS RUN
+
@m.Checks
+
+
+
HEALTHY
+
@(m.HealthyShare is { } share ? Percent(share) : "--")
+
+
+
TRANSITIONS
+
@m.Transitions
+
+
+
OVERLAPS
+
@m.OverlappedTriggers
+
+
+
MEAN CHECK
+
@(m.MeanDuration is { } mean ? Milliseconds(mean) : "--")
+
+
+
SLOWEST CHECK
+
@(m.SlowestDuration is { } slowest ? Milliseconds(slowest) : "--")
+
+
+ +

RESULTS BY HEALTH

+
+ @foreach (var health in new[] { PulseCheckerHealth.Healthy, PulseCheckerHealth.Suspicious, PulseCheckerHealth.Unhealthy }) + { +
+
@health.ToString().ToUpperInvariant()
+
@m.ResultsByHealth.GetValueOrDefault(health)
+
+ } +
+ } +
+ } + +