diff --git a/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs b/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs index 2f3945f1b7..6009d0d706 100644 --- a/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs +++ b/Darling/Darling.Tests/DarlingAnomalyBaselineTests.cs @@ -1066,6 +1066,79 @@ await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup } } + /// + /// #4248: IoLatency reads the RAW file_io_stats hypertable at per-file grain over the 30-day window, so its + /// cache key is the UTC DAY, not the hour (PgBaselineProvider.IsDailyCacheMetric) — two calls hours apart on + /// the same UTC day share the one compute, and a call on the next UTC day recomputes. Proven by counting the + /// baseline reads Npgsql actually executes (CommandCapture), #3941's own live-pin technique. + /// + [Fact] + public async Task EndToEnd_IoLatencyArm_TwoCallsHoursApartOnOneDay_ShareOneCompute_NextDayRecomputes_AgainstDevPostgres() + { + var connectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(connectionString), + "Set DARLING_TEST_PG to a Postgres connection string to run the live IO-arm day-cache test."); + + var ct = TestContext.Current.CancellationToken; + const int ioServerId = TestServerId + 5; // own id — this test cleans its own rows + + using var connection = new NpgsqlConnection(connectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + await using (var cleanup = new NpgsqlCommand($"DELETE FROM file_io_stats WHERE server_id = {ioServerId};", connection)) + { + await cleanup.ExecuteNonQueryAsync(ct); + } + + await using var postgres = NpgsqlDataSource.Create(connectionString!); + var bodySucceeded = false; + try + { + var day = DateTime.UtcNow.Date.AddDays(-8); + while (day.DayOfWeek != DayOfWeek.Monday) day = day.AddDays(-1); + var historyStart = DateTime.SpecifyKind(day.AddHours(10), DateTimeKind.Unspecified); + + for (var i = 0; i < 5; i++) + { + await InsertAsync(connection, + "INSERT INTO file_io_stats (collection_id, collection_time, server_id, server_name, delta_reads, delta_writes, delta_stall_read_ms) VALUES ($1, $2, $3, $4, $5, $6, $7)", + (long)(200 + i), historyStart.AddMinutes(5 * i), ioServerId, "IO-DAILY-CACHE", + 10L, 0L, (long)(10 * (i + 1))); + } + + var provider = new PgBaselineProvider(postgres); + var analysisDay = historyStart.AddDays(7).Date; // the 30-day window's end (#4248: midnight, not the hour) + + var (morning, firstReads) = await CommandCapture.CountBaselineReadsAsync( + () => provider.GetBaselineAsync(ioServerId, MetricNames.IoLatency, analysisDay.AddHours(1), ct)); + Assert.Equal(1, firstReads); + Assert.True(morning.SampleCount > 0, "the seed produced no baseline — the comparison would prove nothing"); + + /* Nineteen hours later (over CacheTtl's one hour), same UTC day: the #4248 pin — no second read. */ + var (afternoon, secondReads) = await CommandCapture.CountBaselineReadsAsync( + () => provider.GetBaselineAsync(ioServerId, MetricNames.IoLatency, analysisDay.AddHours(20), ct)); + Assert.Equal(0, secondReads); + Assert.Equal(morning.SampleCount, afternoon.SampleCount); + Assert.Equal(morning.Median, afternoon.Median); + + /* The next UTC day is a different window end (midnight moved), so a fresh compute. */ + var (_, nextDayReads) = await CommandCapture.CountBaselineReadsAsync( + () => provider.GetBaselineAsync(ioServerId, MetricNames.IoLatency, analysisDay.AddDays(1).AddHours(1), ct)); + Assert.Equal(1, nextDayReads); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(connectionString!, bodySucceeded, async (cleanup, cleanupCt) => + { + using var command = new NpgsqlCommand($"DELETE FROM file_io_stats WHERE server_id = {ioServerId};", cleanup); + await command.ExecuteNonQueryAsync(cleanupCt); + }); + } + } + /// /// #3653 (A8, first slice) proven live through the PG detector: the I/O read hands the shared gate the /// per-file-row PEAK and MEAN, and the gate fires only when both clear. History: three Mondays at 10:00, diff --git a/Darling/Darling.Tests/PgTargetClockTests.cs b/Darling/Darling.Tests/PgTargetClockTests.cs index a684e773d2..98a3485353 100644 --- a/Darling/Darling.Tests/PgTargetClockTests.cs +++ b/Darling/Darling.Tests/PgTargetClockTests.cs @@ -66,7 +66,7 @@ public void TheClockRead_IsAProtectedVirtualSeam_OverriddenOnceByThePostgresProv var baseCode = CSharpSourceWalker.StripCommentsAndStrings(baseSource); Assert.Single(Regex.Matches(baseCode, @"await ReadServerClockAsync\(")); Assert.Contains("await ReadServerClockAsync(connection, serverId, AsNaive(windowEnd), cancellationToken)", baseCode, StringComparison.Ordinal); - Assert.Contains("var windowEnd = RoundedHour(analysisTime);", baseCode, StringComparison.Ordinal); + Assert.Contains("var windowEnd = RoundedKeyTime(metricName, analysisTime);", baseCode, StringComparison.Ordinal); Assert.DoesNotContain("DateTime.UtcNow", CSharpSourceWalker.StripCommentsAndStrings( RepoFile.ReadRepoFile("Darling", "PerformanceMonitor.Darling.Analysis", "PgTargetBaselineProvider.Clock.cs")), StringComparison.Ordinal); } diff --git a/Darling/PerformanceMonitor.Darling.Analysis/BaselineCache.cs b/Darling/PerformanceMonitor.Darling.Analysis/BaselineCache.cs index 475dda09d0..67f3abc2e0 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/BaselineCache.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/BaselineCache.cs @@ -41,7 +41,12 @@ namespace PerformanceMonitor.Darling.Analysis; /// Time. after the compute an entry is dead whatever its hour — /// the bound on anything that did move inside a settled window (a late row, a purge, the target's clock changing zone, /// which is re-keyed on the next compute exactly as before). Dead entries are swept, so the tier holds at most one -/// TTL's worth of computes. +/// TTL's worth of computes. Except (#4248): a successful compute of a daily-cache metric +/// () is never eroded by the TTL +/// ( is a 24-hour backstop, not the real bound) — the +/// EntryKey's analysis-day component is, exactly as the hourly case's analysis-hour component always was, so this +/// tier keeps answering every lookup for the rest of that UTC day. See , +/// which this tier's live check and sweep both call, so they never evict a still-fresh daily entry early. /// Failure. Only a SUCCESSFUL compute is shared. A failed one (a timeout, a role that cannot read a /// relation) stays in the failing provider's own tier, where it has always meant "no baseline for this pass" — one /// caller's timeout never blanks another caller's baselines, and a success from any caller beats a local failure. @@ -103,7 +108,7 @@ internal void Invalidate(int serverId) internal void Clear() => _entries.Clear(); private static bool IsLive(PgBaselineProvider.CachedBaseline entry, DateTime nowUtc) - => nowUtc - entry.RealTime < PgBaselineProvider.CacheTtl; + => PgBaselineProvider.IsFresh(entry, nowUtc); /// At most once per quarter of , drops the entries no lookup can /// take any more, so the tier holds little beyond one TTL of computes. Without it the tier would grow for the life of diff --git a/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs b/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs index ee2a1dc8bd..3e0caa75c7 100644 --- a/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs +++ b/Darling/PerformanceMonitor.Darling.Analysis/PgBaselineProvider.cs @@ -354,7 +354,7 @@ private async Task GetOrComputeBaselinesAsync( int serverId, string metricName, DateTime analysisTime, CancellationToken cancellationToken) { var cacheKey = CacheKeyFor(serverId, metricName, key: null); - var roundedHour = RoundedHour(analysisTime); + var roundedHour = RoundedKeyTime(metricName, analysisTime); if (TryGetFresh(serverId, cacheKey, roundedHour, out var cached)) { @@ -363,11 +363,14 @@ private async Task GetOrComputeBaselinesAsync( var (byMember, clock, utcOffsetMinutes, timeZoneId) = await ComputeBaselinesAsync(serverId, metricName, keys: null, analysisTime, cancellationToken); + var buckets = BucketsOf(byMember, UnkeyedMember); + var realTime = DateTime.UtcNow; var entry = new CachedBaseline { ComputedAt = roundedHour, - RealTime = DateTime.UtcNow, - Buckets = BucketsOf(byMember, UnkeyedMember), + RealTime = realTime, + Buckets = buckets, + FreshUntilUtc = buckets is not null && IsDailyCacheMetric(metricName) ? realTime.AddDays(1) : null, Clock = clock, UtcOffsetMinutes = utcOffsetMinutes, TimeZoneId = timeZoneId @@ -397,7 +400,7 @@ private async Task GetOrComputeBaselinesAsync( private async Task> GetOrComputeKeyedBaselinesAsync( int serverId, string metricName, IReadOnlyCollection keys, DateTime analysisTime, CancellationToken cancellationToken) { - var roundedHour = RoundedHour(analysisTime); + var roundedHour = RoundedKeyTime(metricName, analysisTime); var entries = new Dictionary(StringComparer.Ordinal); var misses = new List(); var asked = new HashSet(StringComparer.Ordinal); @@ -435,11 +438,13 @@ private async Task> GetOrComputeKeyedBaseline for (var member = 1; member <= set.Length; member++) { var key = set[member - 1]; + var memberBuckets = BucketsOf(byMember, member); var entry = new CachedBaseline { ComputedAt = roundedHour, RealTime = computedAt, - Buckets = BucketsOf(byMember, member), + Buckets = memberBuckets, + FreshUntilUtc = memberBuckets is not null && IsDailyCacheMetric(metricName) ? computedAt.AddDays(1) : null, Clock = clock, UtcOffsetMinutes = utcOffsetMinutes, TimeZoneId = timeZoneId, @@ -460,6 +465,35 @@ private async Task> GetOrComputeKeyedBaseline internal static DateTime RoundedHour(DateTime analysisTime) => new(analysisTime.Year, analysisTime.Month, analysisTime.Day, analysisTime.Hour, 0, 0); + /// Midnight UTC of 's day (#4248) — the key time and window end for an + /// arm that reads a RAW hypertable at full grain over the 30-day window (), + /// playing 's role at day grain instead of hour grain. The underlying rows move by + /// about 1/720 an hour, so an hourly key bought nothing but 24x the recomputes. + internal static DateTime RoundedDay(DateTime analysisTime) + => new(analysisTime.Year, analysisTime.Month, analysisTime.Day, 0, 0, 0); + + /// #4248: the two arms whose clean CTE reads a RAW hypertable (cpu_utilization_stats, + /// file_io_stats — the #1743 follow-up pair, see 's remarks) rather than a + /// pre-aggregated CREATE MATERIALIZED VIEW ... _baseline supply. Both tables carry their own 30-day + /// service-side retention floor (DarlingRetentionHorizons.BaselineServingRawCollectors), so the 30-day + /// WINDOW does not change here — only the cache KEY's grain does, because a full-grain 30-day read is what made + /// an hourly recompute expensive (measured: ~50 MB of temp per call). The + /// other seven arms and the two event arms read an already-aggregated view — + /// far fewer rows for the same 30 days — and keep the hourly key. PgTargetBaselineProvider's arms all + /// read raw PostgreSQL-target hypertables too (its own remarks say so), but they are NOT in this set: nobody + /// has ruled on a keyed-arm day-long lifetime against 's hourly-turnover + /// calibration, so that file is unchanged here. + internal static bool IsDailyCacheMetric(string metricName) + => metricName is MetricNames.Cpu or MetricNames.IoLatency; + + /// The cache key's time AND the compute's window end (#3941's invariant, restated for #4248): whichever + /// grain uses () — the day for a raw-hypertable + /// arm, the hour for every other one. Both call sites (the key and the window) read this ONE seam so they can + /// never diverge — a key that named a different instant than the window it was computed over would be sharing + /// an answer that is not the answer a fresh compute at that key would give. + internal static DateTime RoundedKeyTime(string metricName, DateTime analysisTime) + => IsDailyCacheMetric(metricName) ? RoundedDay(analysisTime) : RoundedHour(analysisTime); + /// This provider's engine in the shared tier's key (#3941): a SQL Server series and a PostgreSQL-target /// series of one server id are never each other's. private string SharedKind => GetType().FullName ?? GetType().Name; @@ -474,7 +508,7 @@ private bool TryGetFresh(int serverId, string cacheKey, DateTime roundedHour, [N { var local = _cache.TryGetValue(cacheKey, out var own) && own.ComputedAt == roundedHour - && (DateTime.UtcNow - own.RealTime) < CacheTtl; + && IsFresh(own, DateTime.UtcNow); if (local && own!.Buckets is not null) { cached = own; @@ -492,6 +526,22 @@ private bool TryGetFresh(int serverId, string cacheKey, DateTime roundedHour, [N return local; } + /// #4248: is still good to return? A successful compute of a daily-cache + /// metric ( set, a rolling 24 hours from + /// — never eroded by ) stays live for a full day of real time no matter when in the UTC + /// day it ran, so it is never the TTL that ends it: the CALLER'S key (ComputedAt == roundedHour in + /// and ) already stops matching the instant the + /// requested day rolls over, which is what actually bounds an entry to "the rest of the UTC day it was computed + /// in" — the whole point, two calls an hour or more apart on the same day share the one compute. Every + /// hourly-cache metric, and a FAILED compute of ANY metric ( null — a + /// failure never earns the day-long trust), keep the original rolling plus + /// bound, so a timeout still retries within the hour. Shared by 's + /// live check and sweep, so the shared tier never evicts a still-fresh daily entry early. + internal static bool IsFresh(CachedBaseline entry, DateTime nowUtc) + => entry.FreshUntilUtc is DateTime freshUntil + ? nowUtc < freshUntil + : (nowUtc - entry.RealTime) < CacheTtl; + /// Files a compute in this provider's cache and, when it SUCCEEDED, in the shared tier (#3941). A failed /// compute (null buckets) is this caller's "no baseline this pass" and nobody else's. private void Store(int serverId, string cacheKey, CachedBaseline entry) @@ -803,7 +853,7 @@ hour was answered from a window up to 59 minutes off its own. Ending the window in the hour ask for the SAME rows — what lets the process share one compute between the scheduled pass, analyze_server and compare_analysis (BaselineCache) without changing anyone's answer. The lookup still keys the bucket on the analysis instant (LookUp); its hour-of-week is the hour's. */ - var windowEnd = RoundedHour(analysisTime); + var windowEnd = RoundedKeyTime(metricName, analysisTime); var windowStart = windowEnd.AddDays(-BaselineMath.BaselineWindowDays); var clock = LocalClockWindow.Utc(windowEnd); @@ -1508,6 +1558,13 @@ internal sealed class CachedBaseline public DateTime RealTime { get; init; } public Dictionary<(int HourOfDay, int DayOfWeek), BaselineBucket>? Buckets { get; init; } + /// #4248: for a successful compute of a daily-cache metric, plus 24 hours — + /// null for every hourly-cache metric and for a failed compute of any metric. See , + /// the only reader: this bound alone would outlive the metric's own UTC day, but ComputedAt's key + /// match already stops answering the moment that day ends, so in practice this is the "still trustworthy in + /// real time" backstop, not the day boundary itself. + public DateTime? FreshUntilUtc { get; init; } + /// The clock the buckets were keyed with (#3653 Q6) — the lookup must use the SAME one. public LocalClockWindow Clock { get; init; } = LocalClockWindow.Utc(DateTime.MinValue); diff --git a/Lite.Tests/SharedBaselineCacheTests.cs b/Lite.Tests/SharedBaselineCacheTests.cs index 153c96f6d2..47a2368481 100644 --- a/Lite.Tests/SharedBaselineCacheTests.cs +++ b/Lite.Tests/SharedBaselineCacheTests.cs @@ -79,6 +79,39 @@ public void AnEntry_AnswersOnlyItsOwnHour_AndOnlyInsideTheTtl_AndTheSweepDropsTh Assert.Equal(0, cache.Count); } + /// + /// #4248: a daily-cache metric's entry ( set) answers + /// any analysis instant that rounds to the same UTC day, an hour or more after the compute — the CacheTtl the + /// non-daily case above is held to — and stops answering the instant the day rolls over, even well inside + /// CacheTtl. Direct against , so it needs no seed and cannot be confused by which + /// hour-of-day cell a lookup would extract from the buckets. + /// + [Fact] + public void DailyCacheEntry_AnswersForItsWholeUtcDay_NotJustTheTtl_AndStopsAtMidnight() + { + var cache = new BaselineCache(); + var day = new DateTime(2026, 3, 18, 0, 0, 0); // the logical key: the analysis day, not real time + var now = DateTime.UtcNow; // when the compute actually ran + var dailyEntry = new BaselineProvider.CachedBaseline + { + ComputedAt = day, + RealTime = now, + Buckets = new Dictionary<(int HourOfDay, int DayOfWeek), BaselineBucket>(), + Clock = LocalClockWindow.Utc(day), + FreshUntilUtc = now.AddDays(1), + }; + cache.Put(1, "1:cpu", dailyEntry); + + /* Well over CacheTtl (an hour) after the compute, same UTC day as the key (`day`): still a hit, unlike the + hourly entry above, which the sibling test already holds to CacheTtl. */ + Assert.True(cache.TryGet(1, "1:cpu", day, out var lateSameDay)); + Assert.Same(dailyEntry, lateSameDay); + + /* Midnight: a new day, a new key — the shared tier's TryGet is keyed on the ROUNDED DAY the caller asks + for, so this is a natural miss, not a FreshUntilUtc check. */ + Assert.False(cache.TryGet(1, "1:cpu", day.AddDays(1), out _)); + } + [Fact] public void TheTier_IsOnePerStore() { @@ -125,9 +158,17 @@ public async Task ASharedEntry_IsTheFreshAnswer_AndReadsNothing() AssertSameBucket(fresh, stillShared); Assert.Equal(fresh.SampleCount + 1, changed.SampleCount); - /* The next analysis hour is another window: computed, and it sees the new row. */ - var nextHour = await new BaselineProvider(_duckDb, sharedCache: shared).GetBaselineAsync(ServerId, MetricNames.Cpu, Hour.AddMinutes(65)); - Assert.True(nextHour.SampleCount > 0); + /* #4248: Cpu reads its RAW table over the full 30-day window, so since #4248 it is cached by UTC DAY, not + the hour — a lookup later the SAME day (even a different hour-of-day cell, so not compared bucket-for- + bucket against `fresh`) still finds rows, and DailyCacheEntry_AnswersForItsWholeUtcDay below pins the + no-recompute behavior directly against the cache's own liveness check. */ + var nextHourSameDay = await new BaselineProvider(_duckDb, sharedCache: shared).GetBaselineAsync(ServerId, MetricNames.Cpu, Hour.AddMinutes(65)); + Assert.True(nextHourSameDay.SampleCount > 0); + + /* The next UTC DAY is a new window end: computed, and it sees the new row (the window's oldest day also + drops off the far end, so this only checks the compute ran and found rows, not an exact count). */ + var nextDay = await new BaselineProvider(_duckDb, sharedCache: shared).GetBaselineAsync(ServerId, MetricNames.Cpu, Hour.AddDays(1).AddMinutes(50)); + Assert.True(nextDay.SampleCount > 0); } [Fact] diff --git a/Lite/Analysis/BaselineCache.cs b/Lite/Analysis/BaselineCache.cs index 63fe659fbc..56e4647f65 100644 --- a/Lite/Analysis/BaselineCache.cs +++ b/Lite/Analysis/BaselineCache.cs @@ -20,7 +20,11 @@ namespace PerformanceMonitorLite.Analysis; /// computed for, and the compute's window ends AT that hour ( snaps it), so every caller /// in the hour is asking for the same 30 days of rows and the cached answer is the fresh one. Time: an entry is dead /// after its compute whatever its hour — the bound on anything that moved inside -/// the window (a late row, a retention purge, the target's clock changing zone) — and dead entries are swept. Failure: +/// the window (a late row, a retention purge, the target's clock changing zone) — and dead entries are swept. Except +/// (#4248): a successful compute of a daily-cache metric () is never +/// eroded by the TTL — the EntryKey's analysis-day component already stops matching once that UTC day ends, exactly as +/// the hourly case's analysis-hour component always has; see , which this tier's +/// live check and sweep both call. Failure: /// only a SUCCESSFUL compute is shared; a failed one stays the failing provider's "no baseline this pass", and a shared /// success beats it. What does not invalidate: thresholds and settings (nothing configurable reaches the baseline SQL — /// every threshold is applied after the lookup), and a server's removal (its history stays in the store). @@ -78,7 +82,7 @@ internal void Invalidate(int serverId) internal void Clear() => _entries.Clear(); private static bool IsLive(BaselineProvider.CachedBaseline entry, DateTime nowUtc) - => nowUtc - entry.RealTime < BaselineProvider.CacheTtl; + => BaselineProvider.IsFresh(entry, nowUtc); /// At most once per quarter of , drops the entries no lookup can take /// any more, so the tier holds little beyond one TTL of computes. diff --git a/Lite/Analysis/BaselineProvider.cs b/Lite/Analysis/BaselineProvider.cs index c5f3dd651e..3440175d39 100644 --- a/Lite/Analysis/BaselineProvider.cs +++ b/Lite/Analysis/BaselineProvider.cs @@ -220,18 +220,45 @@ public void ClearCache() internal static DateTime RoundedHour(DateTime analysisTime) => new(analysisTime.Year, analysisTime.Month, analysisTime.Day, analysisTime.Hour, 0, 0); + /// Midnight UTC of 's day (#4248) — Darling's PgBaselineProvider.RoundedDay, + /// twinned. The key time and window end for an arm that reads a RAW table at full grain over the 30-day window + /// (), playing 's role at day grain instead of hour grain. + internal static DateTime RoundedDay(DateTime analysisTime) + => new(analysisTime.Year, analysisTime.Month, analysisTime.Day, 0, 0, 0); + + /// #4248: the two arms whose clean CTE reads a RAW table (v_cpu_utilization_stats, + /// v_file_io_stats) rather than an already-aggregated one — Darling's PgBaselineProvider.IsDailyCacheMetric, + /// twinned, so the two products flag the same anomalies off the same cache grain. See that method's remarks for + /// why these two and not the other seven RobustTierScaffold arms. + internal static bool IsDailyCacheMetric(string metricName) + => metricName is MetricNames.Cpu or MetricNames.IoLatency; + + /// The cache key's time AND the compute's window end (#3941, restated for #4248): the day for a + /// raw-table arm (), the hour for every other one — the one seam both read so + /// they can never diverge. + internal static DateTime RoundedKeyTime(string metricName, DateTime analysisTime) + => IsDailyCacheMetric(metricName) ? RoundedDay(analysisTime) : RoundedHour(analysisTime); + + /// #4248: is still good to return? Darling's PgBaselineProvider.IsFresh, + /// twinned — see its remarks for why is a 24-hour backstop and not + /// itself the day boundary. + internal static bool IsFresh(CachedBaseline entry, DateTime nowUtc) + => entry.FreshUntilUtc is DateTime freshUntil + ? nowUtc < freshUntil + : (nowUtc - entry.RealTime) < CacheTtl; + private async Task GetOrComputeBaselinesAsync( int serverId, string metricName, DateTime analysisTime, CancellationToken cancellationToken) { var cacheKey = $"{serverId}:{metricName}"; - var roundedHour = RoundedHour(analysisTime); + var roundedHour = RoundedKeyTime(metricName, analysisTime); /* #3941, Darling's PgBaselineProvider.TryGetFresh twinned: a local SUCCESS; then the store's shared tier (copied in, so the rest of this provider's life is a local hit); then a local FAILURE — "no baseline this pass", as before. A shared success beats a local failure because it is the answer the failed compute was after. */ var local = _cache.TryGetValue(cacheKey, out var cached) && cached.ComputedAt == roundedHour && - (DateTime.UtcNow - cached.RealTime) < CacheTtl; + IsFresh(cached, DateTime.UtcNow); if (local && cached!.Buckets is not null) { return cached; @@ -250,11 +277,13 @@ before. A shared success beats a local failure because it is the answer the fail var (buckets, clock, utcOffsetMinutes, timeZoneId) = await ComputeBaselinesAsync(serverId, metricName, analysisTime, cancellationToken); + var realTime = DateTime.UtcNow; var entry = new CachedBaseline { ComputedAt = roundedHour, - RealTime = DateTime.UtcNow, + RealTime = realTime, Buckets = buckets, + FreshUntilUtc = buckets is not null && IsDailyCacheMetric(metricName) ? realTime.AddDays(1) : null, Clock = clock, UtcOffsetMinutes = utcOffsetMinutes, TimeZoneId = timeZoneId @@ -312,7 +341,7 @@ hour was answered from a window up to 59 minutes off its own. Ending the window in the hour ask for the SAME rows — what lets the store's callers share one compute (BaselineCache) without changing anyone's answer. The lookup still keys the bucket on the analysis instant; its hour-of-week is the hour's. Darling's PgBaselineProvider.ComputeBucketsAsync is the twin. */ - var windowEnd = RoundedHour(analysisTime); + var windowEnd = RoundedKeyTime(metricName, analysisTime); var query = GetBaselineQuery(metricName); var clock = LocalClockWindow.Utc(windowEnd); if (query == null) return (null, clock, null, null); @@ -675,6 +704,11 @@ internal sealed class CachedBaseline public DateTime RealTime { get; init; } public Dictionary<(int HourOfDay, int DayOfWeek), BaselineBucket>? Buckets { get; init; } + /// #4248: for a successful compute of a daily-cache metric, plus 24 hours — + /// null for every hourly-cache metric and for a failed compute of any metric. Darling's + /// PgBaselineProvider.CachedBaseline.FreshUntilUtc, twinned — see . + public DateTime? FreshUntilUtc { get; init; } + /// The clock the buckets were keyed with (#3653 Q6) — the lookup must use the SAME one. public LocalClockWindow Clock { get; init; } = LocalClockWindow.Utc(DateTime.MinValue);