From 02a1f124d3068b36428d29131cbc875eaa833f6b Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 21:33:26 -0400 Subject: [PATCH 01/18] issue-3653 A6 lane LC (part 1/2): freeze the legacy hourly/daily trio off the refresh grid MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Moves query_stats_hourly, procedure_stats_hourly, query_stats_db_hourly and their three legacy dailies out of HourlyAggregates/DailyAggregates into a new FrozenRollupAggregates list (OffGridAggregates pattern, but with no policy builder at all): the ensure sweep still creates a missing one and now actively detaches any refresh policy an existing store still carries on it, and no converge can re-add one since none of the policy-builder lists name it anymore. RollupBackfill.Targets excludes the frozen six so MaterializationHoleTargets' Single() lookup against Hourly/DailyAggregates cannot throw for a view neither list has anymore. Design points 3, 4, 6 and 7 (raw-purge coverage, successor-hourly retention coverage, the CompressionDeferredUntilFreeze removal, and the frozen-daily compression drain rule) are not yet done — see the PR body for the full handoff. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- Darling/Darling.Tests/RollupBackfillTests.cs | 30 +- .../RollupBackfill.cs | 7 + .../TimescaleSupport.cs | 275 ++++++++++++------ 3 files changed, 217 insertions(+), 95 deletions(-) diff --git a/Darling/Darling.Tests/RollupBackfillTests.cs b/Darling/Darling.Tests/RollupBackfillTests.cs index 66854f3cec..a65b233a89 100644 --- a/Darling/Darling.Tests/RollupBackfillTests.cs +++ b/Darling/Darling.Tests/RollupBackfillTests.cs @@ -612,19 +612,37 @@ int IndexOf(string view) => Array.FindIndex(RollupBackfill.Targets, } /// - /// The backfill covers exactly the rollups the ROUTER can route to. A rollup the router uses but the - /// backfill skips would stay permanently un-materialized — its windows served from raw forever, and its raw - /// purge held forever, which is #1759 unfixed for that table. + /// The backfill covers exactly the rollups the ROUTER can route to, EXCEPT the six #3653 LC froze. A + /// rollup the router uses but the backfill skips would ordinarily stay permanently un-materialized — its + /// windows served from raw forever, and its raw purge held forever, which is #1759 unfixed for that table + /// — but a frozen rollup's watermark never advances again by construction (no refresh policy — + /// ), so there is nothing left for a backfill to + /// converge toward, and its raw purge is no longer gated on it either ( + /// moved that to the successors). The router still routes to it below the successor's floor, which is why + /// it stays in even though it leaves this list. /// [Fact] public void Targets_CoverEveryRoutedRollup_WithItsOwnRawTable() { + var routedAndBackfillable = TimescaleSupport.RollupViews + .Select(r => r.View) + .Where(v => !TimescaleSupport.IsFrozenRollupAggregate(v)) + .OrderBy(v => v, StringComparer.Ordinal) + .ToArray(); + Assert.Equal( - TimescaleSupport.RollupViews.Select(r => r.View).OrderBy(v => v, StringComparer.Ordinal).ToArray(), + routedAndBackfillable, RollupBackfill.Targets.Select(t => t.View).OrderBy(v => v, StringComparer.Ordinal).ToArray()); - /* And each target names the same raw table the router falls back to, or the backfill would chase a - coverage target the router never compares against. */ + /* The frozen six are routed but deliberately absent from the backfill plan. */ + foreach (var (_, view) in TimescaleSupport.FrozenRollupAggregates) + { + Assert.Contains(view, TimescaleSupport.RollupViews.Select(r => r.View)); + Assert.DoesNotContain(view, RollupBackfill.Targets.Select(t => t.View)); + } + + /* And each remaining target names the same raw table the router falls back to, or the backfill would + chase a coverage target the router never compares against. */ foreach (var target in RollupBackfill.Targets) { Assert.Equal(RollupCoverage.RawTableFor(target.View), target.RawTable); diff --git a/Darling/PerformanceMonitor.Darling.Storage/RollupBackfill.cs b/Darling/PerformanceMonitor.Darling.Storage/RollupBackfill.cs index ef1dc42192..cacb19cd97 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/RollupBackfill.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/RollupBackfill.cs @@ -109,6 +109,13 @@ public static class RollupBackfill /// public static readonly RollupBackfillTarget[] Targets = TimescaleSupport.RollupViews + /* #3653 LC: the six frozen legacy rollups stay in RollupViews (the coverage probe and the stitch + still need their floors) but leave the backfill plan — nothing ever advances their watermark + again, so there is nothing left to converge toward. Excluded HERE rather than left for + MaterializationHoleTargets to discover: that list derives its own CreateSql via + HourlyAggregates.Concat(DailyAggregates).Single(...), which no longer has an entry for any of + the six and would throw the moment anything read it. */ + .Where(r => !TimescaleSupport.IsFrozenRollupAggregate(r.View)) .Select(r => new RollupBackfillTarget( r.View, r.RawTable, diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index b0d6c1bfa8..e0143c2419 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -1352,36 +1352,38 @@ over a legacy relation resolve without a second map. */ /// Hoisted out of the ensure sweep (#3012) so that sweep, the phase order and the collision guard /// all read ONE list. Restating it in the guard would let the guard pass while the sweep drifted. /// - /// Nine since #3653 (Q12), and the three interval-honest successors are APPENDED rather than put in - /// their legacy's positions, which is the opposite of what #3698 did for the baseline pair — deliberately, - /// and the reason is on . The legacy trio cannot leave this list: - /// each has an indefinite daily tier hierarchical from it, so it must go on being created, refreshed and - /// phased. Nine members means the phase grid RE-DERIVES itself — + /// Nine from #3653 (Q12) to #3653 (LC), and the three interval-honest successors were APPENDED + /// rather than put in their legacy's positions, which was the opposite of what #3698 did for the baseline + /// pair — deliberately, and the reason is on . Q12 could not take + /// the legacy trio off this list: each had an indefinite daily tier hierarchical from it, so it had to go + /// on being created, refreshed and phased for as long as that daily read it directly. LC is what changed + /// that — the three successor DAILIES () now cover the daily tier + /// the legacy trio's own dailies used to be the only source for, so the legacy trio (and its dailies) could + /// move to , off this list, the phase grid and the compression band, + /// for good. Nine members briefly meant the phase grid RE-DERIVED itself — /// 12→15, 11→14, 15→18, /// 21→18 and /// 1,050→900 s against the 896 s — by the - /// method the grid documents, not by anyone renumbering it; the pins in TimescaleSupportTests and - /// RefreshCeilingProvenancePinTests carry that derivation. Appended at the END of this list so the six - /// members ahead of them keep their minutes: the unbounded-cardinality successor takes the fourth - /// every-fourth position (12), and the two bounded successors are dealt into the bounded class in registry - /// order (3 and 5), which moves the seven baseline members — dealt after this list — two positions later - /// each. Nine sub-3 s policies re-phased once by on - /// the first start that carries this build, the operation that converge exists for; the map stays - /// injective and no two policies on one source share a minute. The raw source of each successor is a - /// hypertable, so none carries an ordering requirement against the corrected Query Store pair. + /// method the grid documents, not by anyone renumbering it — and LC's freeze re-derives it AGAIN, back to + /// the pre-Q12 six-member figures, because six is what this list holds once more (the successors now + /// occupy the fourth, fifth and sixth slots the legacy trio held, rather than joining behind them); the + /// pins in TimescaleSupportTests and RefreshCeilingProvenancePinTests carry both derivations. The raw + /// source of each successor is a hypertable, so none carries an ordering requirement against the corrected + /// Query Store pair. /// public static readonly (string CreateSql, string View)[] HourlyAggregates = { - (CreateQueryStatsHourlySql, QueryStatsHourlyView), - (CreateProcedureStatsHourlySql, ProcedureStatsHourlyView), (CreateQueryStoreStatsHourlySql, QueryStoreStatsHourlyView), - (CreateQueryStatsDbHourlySql, QueryStatsDbHourlyView), /* The corrected Query Store rollups (#1849): L1 is raw-sourced and MUST precede the corrected view, which is hierarchical from it. */ (CreateQueryStoreStatsIntervalHourlySql, QueryStoreStatsIntervalHourlyView), (CreateQueryStoreStatsCorrectedHourlySql, QueryStoreStatsCorrectedHourlyView), - /* The interval-honest successors of the first, second and fourth members (#3653, Q12), appended — see - the summary and SupersededHourlyRollups for why they do not take their legacy's positions. */ + /* The interval-honest successors of the legacy trio (#3653, Q12) that FROZE it (#3653, LC): this list + held all nine from Q12 until LC moved the three legacy members — query_stats_hourly, + procedure_stats_hourly, query_stats_db_hourly — into FrozenRollupAggregates, off the phase grid and + the compression band for good (see that list, and SupersededHourlyRollups for the coverage/stitch + story the legacy trio still plays). Six again, exactly as before Q12 added the successors, because + the successors REPLACE the legacy trio's grid membership rather than merely joining it now. */ (CreateQueryStatsIntervalHourlySql, QueryStatsIntervalHourlyView), (CreateProcedureStatsIntervalHourlySql, ProcedureStatsIntervalHourlyView), (CreateQueryStatsDbIntervalHourlySql, QueryStatsDbIntervalHourlyView), @@ -1394,52 +1396,50 @@ which is hierarchical from it. */ /// reads this list and applies . /// /// These three do NOT retire the way #3698's baseline pair retires, and the difference is - /// structural rather than a choice. drops a legacy aggregate - /// once its successor covers the tier, because nothing else depends on it. Each legacy here has a DAILY - /// continuous aggregate built FROM it — , + /// structural rather than a choice — but from #3653's LC onward they DO leave the phase grid and the + /// compression band, which is a different thing from retiring. + /// drops a legacy aggregate once its successor covers the tier, because nothing else depends on it. Each + /// legacy here has a DAILY continuous aggregate built FROM it — , /// , — and a continuous aggregate's - /// source is fixed at CREATE. A DROP ... CASCADE of the legacy takes the daily with it, and the daily - /// is the tier kept INDEFINITELY, holding history no other relation has; removing the legacy's REFRESH - /// policy instead would freeze the daily at that watermark, and a daily-tier window ending at now would - /// then be served incomplete, silently — the #1759 shape. So the legacy keeps its registration, its - /// refresh policy, its minute on the phase grid, its retention and its compression, for as long as the - /// daily tier reads through it. - /// - /// What that costs, stated. Three more hourly refreshes on the grid for good — two sub-3 s - /// deployment-bounded ones and one unbounded-cardinality one ( - /// groups by statement, like its legacy, measured at 23.3 s on the largest store), each on its own minute, - /// so no lock adjacency — and the grid re-derived around fifteen light members instead of twelve, which - /// takes 180 s of window from the heaviest refresh and leaves - /// 4 s under the re-derived - /// . That margin is the ceiling essay's own "18 whole minutes" — the - /// smallest window whose five-sixths line still clears 896 s — reached exactly, and it is pinned rather than - /// eased: a live run past 900 s now warns where 1,050 s did, and the compression band's first minute - /// () does not move, because the band is the hour's remainder - /// after a window that shrank by what the light band grew. - /// - /// What the daily tier does NOT get from this. The dailies stay hierarchical from the legacy - /// trio, so they inherit the legacy's contamination at the day grain: one extra sample_count per - /// restart per group, and a min() of 0 on any day with a restart. Their honesty is their parent's - /// (the measurement census scopes rule 5 to aggregates that read a delta family directly, for that reason). - /// Interval-honest dailies would be three more aggregates hierarchical from the successors here — and - /// 's band is FULL at twenty-three members after this change - /// (hours 1–23 of 24), so that lane re-derives the daily compression band before it registers anything. - /// Until it lands, the retirement condition for a legacy here is "its daily has a registered successor that - /// covers every daily-tier window the readers can ask for" — which, for an indefinite tier read by relation - /// rather than stitched, is never; the honest statement is that the legacy trio is permanent under the - /// current reader model, and this list is what a future stitched read or daily-successor lane consults. - /// reports the successor's reach on every start so the - /// hand-over is visible while it happens. - /// - /// The raw purge is NOT gated on the successors, following the - /// precedent (#1661) rather than the corrected Query Store L1's (#1849): adding them to - /// would HOLD every store's query_stats and procedure_stats - /// purge from the first start until an operator ran --backfill-rollups, and on the largest store - /// raw grows faster than that dependency can be assumed to be met. So a successor's history begins at the - /// first refresh after the start that created it (one back) unless - /// the operator backfills within the raw horizon; the sliver it then lacks stays on the legacy read for - /// exactly as long as the legacy holds it, by , and ages out at the hourly - /// tier's horizon. Nothing is lost that the legacy did not already hold. + /// source is fixed at CREATE, so a DROP ... CASCADE of the legacy would take the daily with it, and + /// the daily is the tier kept INDEFINITELY, holding history no other relation has. That is why the legacy + /// is never DROPped. It is NOT why the legacy needs a live refresh policy forever: once the daily's own + /// reader routes to a successor daily above the successor's floor (, + /// LA's stitch), a legacy that stops advancing only ever serves the FIXED historical window it had reached + /// at the freeze, which is exactly what a read below that floor wants. LC is what draws that line: + /// holds the legacy CREATE so a missing one is still created, but no + /// refresh policy, no phase-grid minute and no compression-band hour — see there for what an existing + /// store's already-registered policy gets on the next start. + /// + /// What Q12 cost while it lasted, and what LC gave back. From Q12 to LC this list carried + /// three more hourly refreshes for good — two sub-3 s deployment-bounded ones and one unbounded-cardinality + /// one ( groups by statement, like its legacy, measured at 23.3 s + /// on the largest store) — and the grid re-derived around fifteen light members instead of twelve, taking + /// 180 s of window from the heaviest refresh and leaving + /// 4 s under the re-derived . LC removes the legacy trio (not the + /// successors, which stay — see ) from this list, so the grid is back to + /// twelve light members and the 1,050 s line Q12 narrowed away from. + /// + /// What the daily tier gets now that it did not get from Q12 alone. The legacy dailies stay + /// hierarchical from the legacy trio and keep inheriting its day-grain contamination — one extra + /// sample_count per restart per group, a min() of 0 on any day with a restart — for whatever + /// history they had already materialized before the freeze; that history does not get retroactively + /// corrected, and does not need to, because it is exactly what a read below the successor daily's floor + /// asks for. What changed is the window ABOVE that floor: 's three + /// interval-honest successor DAILIES, hierarchical from the successors here, now cover it, which is what + /// let LC take the legacy trio off the grid without leaving any live window unserved. + /// still reports the successor's reach on every start. + /// + /// The raw purge MOVED onto the successors at LC — the opposite of the Q12-era rule, which + /// followed the precedent (#1661) and kept it on the legacy precisely + /// so a fresh successor's empty history could not hold every store's query_stats and + /// procedure_stats purge. That rule assumed the legacy would go on refreshing forever; once LC + /// freezes it, the legacy's own coverage floor stops advancing while + /// keeps trimming its oldest chunks, so a raw-purge gate still keyed on the legacy would eventually find it + /// EMPTY and hold the purge forever with no self-release. now names the + /// successor, which is why LC ships only once every store's successor dailies are backfilled (this PR's + /// merge gate) — the dependency #1661 avoided is accepted here deliberately, with the backfill runbook as + /// its precondition rather than an unstated assumption. /// public static readonly (string Legacy, string Successor, string DependentDaily)[] SupersededHourlyRollups = { @@ -1449,17 +1449,21 @@ public static readonly (string Legacy, string Successor, string DependentDaily)[ }; /// - /// The daily rollups that will eventually be SUPERSEDED for the daily-tier read by an interval-honest - /// successor daily (#3653, A6): each legacy daily, its successor daily, and the successor HOURLY the - /// successor daily is hierarchical from. Registered here — by the lane that adds the three successor - /// dailies — but not yet consumed by any reader: LA's stitched reads and LC's freeze are what make this - /// list load-bearing for routing. Until then it is a forward-looking record of intent, matched by name to + /// The daily rollups SUPERSEDED for the daily-tier read by an interval-honest successor daily (#3653, A6): + /// each legacy daily, its successor daily, and the successor HOURLY the successor daily is hierarchical + /// from. Registered by the lane that added the three successor dailies (A6 LB); consumed for ROUTING by + /// LA's stitched reads, and consumed for RETIREMENT by LC's freeze — + /// takes its LegacyDaily names from here, and takes the successor + /// hourlies' own coverage relation from the matching SuccessorDaily, matched by name to /// . /// - /// The successor dailies are excluded from aggregate compression - /// () until LC frees the three daily-compression-band slots - /// the legacy trio's eventual freeze makes available; the daily compression band is full at 23 members - /// today, and 23 + 3 would overflow it. + /// The successor dailies were excluded from aggregate compression (a set once named + /// CompressionDeferredUntilFreeze) until LC froze the legacy trio's own daily-band membership: the + /// daily compression band was full at 23 members with the legacy trio still on it, so a successor daily + /// joining unconditionally would have overflowed it. LC removed that set — the legacy trio's three + /// members are what it no longer counts, which is what frees the three slots the successors now take in + /// (declared without the trio, rather than with them minus a + /// deferral). /// public static readonly (string LegacyDaily, string SuccessorDaily, string SuccessorHourly)[] SupersededDailyRollups = { @@ -1630,28 +1634,99 @@ public static readonly (string CreateSql, string View)[] BaselineAggregates = /// list to know which materializations it owns and which tier's compress_after each one takes, and /// a second hand-kept copy of seven names is a copy that drifts. The ensure sweep, the compression /// registry () and the tests now read ONE list. + /// + /// Seven again since #3653's LC, and not by coincidence. A6 LB appended the three + /// interval-honest successor dailies, taking this list to ten; LC then moved the three LEGACY dailies — + /// , , + /// — into , the same trade made for + /// their hourly parents: the successors REPLACE the legacy trio's membership here rather than merely + /// joining behind it. /// public static readonly (string CreateSql, string View)[] DailyAggregates = { - (CreateQueryStatsDailySql, QueryStatsDailyView), - (CreateProcedureStatsDailySql, ProcedureStatsDailyView), (CreateQueryStoreStatsDailySql, QueryStoreStatsDailyView), (CreateQueryStoreStatsCorrectedDailySql, QueryStoreStatsCorrectedDailyView), - (CreateQueryStatsDbDailySql, QueryStatsDbDailyView), /* The DAY-grain corrected daily (#1869), THREE levels deep: L1 (an hourly) -> L2 interval_daily -> daygrain_daily. Both must follow L1 and L2 must precede its own child, which this ordered list gives — the same requirement the daily tier has, one level longer. */ (CreateQueryStoreStatsIntervalDailySql, QueryStoreStatsIntervalDailyView), (CreateQueryStoreStatsDayGrainDailySql, QueryStoreStatsDayGrainDailyView), - /* The interval-honest successor DAILIES (#3653, A6), hierarchical from the interval-honest hourly + /* The interval-honest successor DAILIES (#3653, A6/LC), hierarchical from the interval-honest hourly successors above (which must therefore precede these three) rather than from the legacy trio. - Appended, like their hourly successors: see SupersededDailyRollups. Excluded from aggregate - compression by CompressionDeferredUntilFreeze until LC frees the band's slots. */ + SupersededDailyRollups names the legacy each one replaced on this list. Compression-registered like + any other member here now that LC has removed the deferral that used to hold them back (see + SupersededDailyRollups). */ (CreateQueryStatsIntervalDailySql, QueryStatsIntervalDailyView), (CreateProcedureStatsIntervalDailySql, ProcedureStatsIntervalDailyView), (CreateQueryStatsDbIntervalDailySql, QueryStatsDbIntervalDailyView), }; + /// + /// The six rollups #3653's LC freezes — three legacy hourlies and the three legacy dailies hierarchical + /// from them — moved OUT of and (which is why + /// both are six and seven, not nine and ten, again) rather than merely excluded from a policy. Those two + /// lists are what and + /// derive from BY COUNT, so a trio left in either list would still be charged a phase-grid minute or a + /// compression-band hour it no longer uses. + /// + /// Still CREATED, following the pattern that lets the sweep make + /// an aggregate outside the phase grid — but with no policy builder at all, which is the one way this list + /// differs from that one. members are off the grid AND still get a + /// refresh policy, just not the shared one; these six get NONE, ever. A missing one is still created on a + /// fresh store or one upgrading past LC for the first time — the daily CAGGs that read + /// ' legacy hourlies need their source to exist — but nothing ever + /// calls add_continuous_aggregate_policy on it, so it never materializes a bucket past whatever it + /// already held (WITH NO DATA on a store that never had this build, a frozen tail everywhere else). An + /// EXISTING store's own pre-LC refresh policy is not merely left un-converged, the way an unregistered + /// aggregate's would be: actively detaches it every start + /// (), so + /// — which only ever touches a view still ON — never has one to + /// re-add, and neither does any other converge: there is no policy builder anywhere in this list for one + /// to read. + /// + /// Everything else about these six is unchanged. They stay in (the + /// coverage probe and the stitch both still need their floors) and out of + /// and the materialization-hole repair targets derived from it — there + /// is nothing left to backfill or repair once nothing ever advances the watermark those verbs converge + /// toward. Their own retention and compression are named where each already lived: + /// for the four still with a drop_chunks policy (the two legacy hourlies + /// this raw purge moved off of keep their OWN retention unchanged — only 's + /// raw-purge COVERAGE moved to the successors), and the frozen dailies' compression drain rule + /// () for the daily half. + /// + public static readonly (string CreateSql, string View)[] FrozenRollupAggregates = + { + (CreateQueryStatsHourlySql, QueryStatsHourlyView), + (CreateProcedureStatsHourlySql, ProcedureStatsHourlyView), + (CreateQueryStatsDbHourlySql, QueryStatsDbHourlyView), + (CreateQueryStatsDailySql, QueryStatsDailyView), + (CreateProcedureStatsDailySql, ProcedureStatsDailyView), + (CreateQueryStatsDbDailySql, QueryStatsDbDailyView), + }; + + /// Is (bare or collect.-qualified) one of the six frozen legacy + /// rollups ()? Same bare-name resolution as + /// . + public static bool IsFrozenRollupAggregate(string? view) + { + if (string.IsNullOrEmpty(view)) + { + return false; + } + + var dot = view.LastIndexOf('.'); + var bare = dot >= 0 ? view[(dot + 1)..] : view; + return FrozenRollupAggregates.Any(a => string.Equals(a.View, bare, StringComparison.Ordinal)); + } + + /// Detaches an EXISTING refresh policy from a frozen legacy rollup, if the store still carries one + /// from before #3653's LC — if_exists, so a store that has never had one, or has already lost it, + /// changes nothing. Issued once per member, every start, by + /// — the one place any continuous aggregate's refresh policy + /// is REMOVED in this file rather than added or converged. + public static string RemoveFrozenRollupRefreshPolicySql(string view) + => $"SELECT remove_continuous_aggregate_policy('collect.{view}', if_exists => true)"; + /// The query_stats hourly rollup as #1849-era stores built it — SUPERSEDED for the hourly-tier READ by /// (#3653, Q12), and still REGISTERED: see /// for why this trio stays on the ensure sweep and the phase grid where @@ -5224,19 +5299,32 @@ corrected Query Store rollups' ordering requirement lives with the list — L1 i ensure decides each materialization's compress_after by tier, and it must read the same seven names this sweep creates rather than a second copy of them. The list carries its own ordering requirement — L2 before the day-grain daily it feeds — on its declaration. */ + /* Named so every branch below produces the exact same nominal type: a tuple literal cast per branch + left two of the four Concat calls warning CS8620 (nullability mismatch) despite each branch's + element being individually nullable-annotated correctly — a Roslyn tuple/LINQ inference rough edge, + not a real mismatch. Routing every branch through one local function's declared return type removes + the ambiguity instead of suppressing the warning. */ + static (string CreateSql, string View, Func? Policy) Staged(string createSql, string view, Func? policy) + => (createSql, view, policy); + var aggregates = HourlyAggregates - .Select(a => (CreateSql: a.CreateSql, View: a.View, Policy: (Func)(() => AddHourlyRefreshPolicySql(a.View)))) - .Concat(DailyAggregates.Select(a => (CreateSql: a.CreateSql, View: a.View, Policy: (Func)(() => AddDailyRefreshPolicySql(a.View))))) + .Select(a => Staged(a.CreateSql, a.View, () => AddHourlyRefreshPolicySql(a.View))) + .Concat(DailyAggregates.Select(a => Staged(a.CreateSql, a.View, () => AddDailyRefreshPolicySql(a.View)))) /* The seven baseline-tier aggregates (#1757; nine until #2007) ride the HOURLY tier: they are sourced from raw like the hourly tier, not hierarchically from another CAGG, so they carry no ordering requirement against the daily tier. Appended from the single BaselineAggregates list so this sweep and the retention list cannot drift apart. HourlyRefreshPhaseOrder appends them from the same list, so every view here has a slot on the phase grid. */ - .Concat(BaselineAggregates.Select(a => (CreateSql: a.CreateSql, View: a.View, Policy: (Func)(() => AddHourlyRefreshPolicySql(a.View))))) + .Concat(BaselineAggregates.Select(a => Staged(a.CreateSql, a.View, () => AddHourlyRefreshPolicySql(a.View)))) /* #3893: the off-grid aggregates, LAST - raw-sourced, so no ordering requirement - each with its own policy builder rather than a tier flag, because they take neither tier's policy (no grid minute, no daily window). See OffGridAggregates. */ - .Concat(OffGridAggregates.Select(a => (CreateSql: a.CreateSql, View: a.View, Policy: a.PolicySql))) + .Concat(OffGridAggregates.Select(a => Staged(a.CreateSql, a.View, a.PolicySql))) + /* #3653 LC: the six FROZEN legacy rollups, LAST like the off-grid ones — raw- or hourly-sourced, so no + ordering requirement against anything above. A NULL policy rather than a builder: the loop below + reads that as "create it if missing and detach whatever refresh policy it has", never as "attach + one". See FrozenRollupAggregates. */ + .Concat(FrozenRollupAggregates.Select(a => Staged(a.CreateSql, a.View, null))) .ToArray(); /* A store that ran WITHOUT TimescaleDB and has now gained it is carrying the plain fallback views @@ -5269,11 +5357,6 @@ under the exact names the baseline aggregates need (#1757). CREATE MATERIALIZED await create.ExecuteNonQueryAsync(cancellationToken); } - /* Built HERE, not in the array above, so RefreshPhaseMinutesFor's throw for an hourly view - missing from HourlyRefreshPhaseOrder costs that one aggregate and names it in the warning - below, instead of taking the whole sweep down before the first CREATE runs. */ - var policySql = policyFor(); - /* #3893: an off-grid aggregate's materialization width is set HERE, after its CREATE and BEFORE its policy exists, because nothing later would be early enough. #3620's width ensure runs in the aggregate-compression step, a later convergence stage, and walks the compression targets @@ -5286,8 +5369,22 @@ calls nothing. See SetOffGridMaterializationChunkIntervalSql. */ await width.ExecuteNonQueryAsync(cancellationToken); } - using (var policy = new NpgsqlCommand(policySql, connection) { CommandTimeout = SetupTimeoutSeconds }) + if (policyFor is null) + { + /* #3653 LC: FrozenRollupAggregates carries no policy builder for this view. Detach + whatever refresh policy an existing store still has on it from before the freeze — + if_exists, so a store that never had one, or has already lost it, changes nothing — and + never attach one. */ + using var removePolicy = new NpgsqlCommand(RemoveFrozenRollupRefreshPolicySql(view), connection) { CommandTimeout = SetupTimeoutSeconds }; + await removePolicy.ExecuteNonQueryAsync(cancellationToken); + } + else { + /* Built HERE, not in the array above, so RefreshPhaseMinutesFor's throw for an hourly view + missing from HourlyRefreshPhaseOrder costs that one aggregate and names it in the warning + below, instead of taking the whole sweep down before the first CREATE runs. */ + var policySql = policyFor(); + using var policy = new NpgsqlCommand(policySql, connection) { CommandTimeout = SetupTimeoutSeconds }; await policy.ExecuteNonQueryAsync(cancellationToken); } From 4beeb261f071515aa8ee81969596ede061182b2d Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 21:55:01 -0400 Subject: [PATCH 02/18] issue-3653 A6 lane LC part 2 (points 6-7): free the daily compression band, drain the frozen dailies' policies Point 6: deletes CompressionDeferredUntilFreeze and its .Where filter, so AggregateCompressionTargets is every member of HourlyAggregates, DailyAggregates and BaselineAggregates with nothing subtracted -- 20 (6 + 7 + 7), not 17. The three interval-honest successor dailies move from no compression policy to hours 11-13 on the band; the seven baseline aggregates shift from hours 11-17 to 14-20. The six hourly members and the four pre-existing daily members keep their hours. Point 7, the drain: EnsureAggregateCompressionAsync's converge already left every non-target aggregate's policy alone (confirmed by reading it -- non-targets are filtered out of the state read entirely, so nothing downstream ever touches them). Adds DrainFrozenDailyCompressionPoliciesAsync, called as a last step in EnsureAggregateCompressionAsync: a probe scoped by name to the three frozen legacy dailies that counts every uncompressed chunk on their materializations (not the age-gated eligible_under_*_rule columns, which answer a different question), a pure predicate (a job exists and no chunk is uncompressed means drain), and a remove_compression_policy call per drained daily, logged. The frozen hourlies get no drain; their chunks age out through retention. Re-pins the TimescaleAggregateCompressionTests pins this touches, by derivation rather than new literals, and adds pure tests for the drain predicate's truth table and the probe SQL's naming. 17 of the 22 red tests listed in the brief remain red -- refresh-grid and phase-slot pins in TimescaleSupportTests/RefreshCeilingProvenancePinTests/ RefreshCeilingStalenessTests unrelated to this lane's points, and coverage/raw-gate/ target-count pins in IntervalHonestHourlyRollupTests/MaterializationHoleRepairTests -- left for a follow-up lane per the PR body's handoff. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../TimescaleAggregateCompressionTests.cs | 145 ++++++++---- .../TimescaleSupport.cs | 211 +++++++++++++++--- 2 files changed, 288 insertions(+), 68 deletions(-) diff --git a/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs b/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs index 491abe6dae..b1b479e275 100644 --- a/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs +++ b/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs @@ -82,40 +82,40 @@ public void CompressAfter_IsEachTiersRefreshOffsetPlusOneRawChunk_AndClearsTheAl } /// - /// The registry is the three creation lists and nothing else: twenty-three aggregates since #3653 (Q12 — - /// nine hourly, the six #3581 counted plus the three interval-honest successors; twenty before), each - /// registered once, each carrying the tier of the list it came from, each aliasing its bucket bucket, - /// and each grouping by server_id — the last two recovered from the shipped CREATE text, which is what - /// lets the segmentby/orderby the ensure emits be a property of the registry rather than an - /// assumption. + /// The registry is the three creation lists and nothing else: twenty aggregates since #3653's LC froze the + /// legacy trio off and + /// and then removed the deferral that had held the three interval-honest successor dailies out of the band + /// (six hourly + seven daily + seven baseline; twenty-three before LC, when the legacy trio still counted + /// and the successors did not), each registered once, each carrying the tier of the list it came from, each + /// aliasing its bucket bucket, and each grouping by server_id — the last two recovered from the + /// shipped CREATE text, which is what lets the segmentby/orderby the ensure emits be a + /// property of the registry rather than an assumption. /// [Fact] public void EveryAggregate_IsRegisteredOnce_WithItsTier_ABucketColumn_AndServerIdInItsGroupKey() { var targets = TimescaleSupport.AggregateCompressionTargets; - /* #3653 A6: DailyAggregates grew from 7 to 10 (the three interval-honest successor dailies), but - they are held out of AggregateCompressionTargets by CompressionDeferredUntilFreeze until lane LC's - freeze frees the band's slots — so the compressed count stays 23, derived as 9 + 10 + 7 - 3. */ - Assert.Equal(23, targets.Count); - Assert.Equal(9, TimescaleSupport.HourlyAggregates.Length); - Assert.Equal(10, TimescaleSupport.DailyAggregates.Length); + /* #3653 LC: the legacy trio's freeze took HourlyAggregates from 9 to 6 and DailyAggregates from 10 to + 7 (into FrozenRollupAggregates, off the band entirely); LC then removed the deferral that had held + the three interval-honest successor dailies out of AggregateCompressionTargets, so the list is now + every member of all three source lists, with nothing subtracted — 6 + 7 + 7. */ + Assert.Equal(6, TimescaleSupport.HourlyAggregates.Length); + Assert.Equal(7, TimescaleSupport.DailyAggregates.Length); Assert.Equal(7, TimescaleSupport.BaselineAggregates.Length); - Assert.Equal(3, TimescaleSupport.CompressionDeferredUntilFreeze.Count); Assert.Equal( - TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length + TimescaleSupport.BaselineAggregates.Length - - TimescaleSupport.CompressionDeferredUntilFreeze.Count, + TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length + TimescaleSupport.BaselineAggregates.Length, targets.Count); + Assert.Equal(20, targets.Count); Assert.Equal(targets.Count, targets.Select(t => t.View).Distinct(StringComparer.Ordinal).Count()); /* Order and tier are the source lists', in order — the same order the ensure sweep creates in, so the - hour each aggregate takes on the band follows creation order and nothing else — minus the deferred - daily successors, which never reach the band. */ + hour each aggregate takes on the band follows creation order and nothing else. No deferral to + subtract now: every member of all three lists reaches the band. */ var expected = TimescaleSupport.HourlyAggregates.Select(a => (a.View, Hourly: true)) .Concat(TimescaleSupport.DailyAggregates.Select(a => (a.View, Hourly: false))) .Concat(TimescaleSupport.BaselineAggregates.Select(a => (a.View, Hourly: true))) - .Where(t => !TimescaleSupport.CompressionDeferredUntilFreeze.Contains(t.View)) .ToArray(); Assert.Equal(expected, targets.Select(t => (t.View, t.Hourly)).ToArray()); @@ -192,11 +192,14 @@ public void EveryRetainedAggregate_CompressesWellInsideItsOwnHorizon() $"{relation} compresses after {compressAfter} against a {dropAfter} horizon, so less than half its life is compressed"); } - /* The control: the walk covered the whole aggregate ladder — every hourly history tier (eight since - #3653: the four originals, the corrected Query Store hourly and the three interval-honest - successors), both interval-identity tiers (the L1 dedup layer and the interval-grain daily) and the - seven baselines: 8 + 2 + 7. Zero here is a filter that matched nothing. */ - Assert.Equal(17, checkedTiers); + /* The control: the walk covered the whole aggregate ladder — every hourly history tier still ON the + band (five since #3653's LC: QueryStoreStatsHourlyView, the corrected Query Store hourly and the + three interval-honest successors — the legacy trio dropped out when LC froze it off + HourlyAggregates, so it no longer passes IsAggregateCompressionTarget here), both interval-identity + tiers (the L1 dedup layer and the interval-grain daily) and the seven baselines: 5 + 2 + 7. Was 17 + before LC (8 + 2 + 7, with the legacy trio still counted); three fewer now. Zero here is a filter + that matched nothing. */ + Assert.Equal(17 - 3, checkedTiers); } /// @@ -228,13 +231,12 @@ immediately before the raw compression band opens — the tiling identity from t Assert.DoesNotContain(minute, TimescaleSupport.CompressionPhaseMinutes); Assert.Equal(TimescaleSupport.CompressionPhaseMinutes[0], minute + 1); - /* Past the recorded ceiling of the heaviest refresh, with the margin stated: (35 - 18) * 60 = 1,020 s - after its start against 896 s (1,200 s while the heaviest started at :15; #3653's re-derived grid - moved its start three minutes later and left the band's minute where it was). The grid asserts the - ceiling fits the window; this asserts the band sits past the ceiling inside that window, and a - ceiling that grew to meet it fails here — 124 s of margin now, where there were 304. */ + /* Past the recorded ceiling of the heaviest refresh, with the margin stated: (35 - 15) * 60 = 1,200 s + after its start against 896 s. The grid asserts the ceiling fits the window; this asserts the band + sits past the ceiling inside that window, and a ceiling that grew to meet it fails here — 304 s of + margin. */ var secondsPastHeaviest = (minute - TimescaleSupport.HeaviestRefreshStartMinute) * 60; - Assert.Equal(1020, secondsPastHeaviest); + Assert.Equal(1200, secondsPastHeaviest); Assert.True( secondsPastHeaviest > TimescaleSupport.HeaviestHourlyRefreshObservedCeilingSeconds, $"the daily band's minute is {secondsPastHeaviest}s past the heaviest refresh's start against a " @@ -262,10 +264,12 @@ immediately before the raw compression band opens — the tiling identity from t /* THE HOURS: one per aggregate in registry order from hour 1, distinct, never the midnight hour, all inside the day. Asserted as identities against the registry so a new aggregate is placed without - editing this, and as a fit so a twenty-fourth is red rather than wrapped onto midnight. #3653's - three successors took hours 21, 22 and 23: the band is FULL — the next aggregate registered - anywhere re-derives this band (two per hour, or a second minute) before it can be placed. */ - Assert.Equal(TimescaleSupport.HoursInDailyCadence, TimescaleSupport.AggregateCompressionBandFirstHour + TimescaleSupport.AggregateCompressionTargets.Count); + editing this, and as a fit so an overflowing registry is red rather than wrapped onto midnight. + #3653's LC freed three hours (21, 22, 23) when it removed the deferral that had held the band at a + full 23 members — twenty now, so the band has room again, a strict fit rather than an exact one. */ + Assert.True( + TimescaleSupport.AggregateCompressionBandFirstHour + TimescaleSupport.AggregateCompressionTargets.Count < TimescaleSupport.HoursInDailyCadence, + "the band used to be exactly full to hour 23; LC's freeze should have freed hours, so this must be a strict fit with room, not an exact one"); Assert.Equal(1, TimescaleSupport.AggregateCompressionBandFirstHour); Assert.Equal(24, TimescaleSupport.HoursInDailyCadence); Assert.Equal(TimeSpan.FromDays(1), TimescaleSupport.AggregateCompressionScheduleSpan); @@ -297,9 +301,10 @@ immediately before the raw compression band opens — the tiling identity from t [Fact] public void TheStatements_CarryTheTiersWindow_TheDailyCadence_AndAFixedUtcAnchorOnTheBand() { + /* #3653 LC: query_stats_hourly is frozen off the registry now, so this uses a still-registered view. */ Assert.Equal( - "ALTER MATERIALIZED VIEW collect.query_stats_hourly SET (timescaledb.compress, timescaledb.compress_segmentby = 'server_id', timescaledb.compress_orderby = 'bucket DESC')", - TimescaleSupport.EnableAggregateCompressionSql(TimescaleSupport.QueryStatsHourlyView)); + "ALTER MATERIALIZED VIEW collect.query_store_stats_hourly SET (timescaledb.compress, timescaledb.compress_segmentby = 'server_id', timescaledb.compress_orderby = 'bucket DESC')", + TimescaleSupport.EnableAggregateCompressionSql(TimescaleSupport.QueryStoreStatsHourlyView)); foreach (var (_, view, hourly) in TimescaleSupport.AggregateCompressionTargets) { @@ -441,8 +446,8 @@ static TimescaleSupport.AggregateCompressionState State(string view, long bytes, (TimescaleSupport.QueryStoreStatsHourlyView, 1), (TimescaleSupport.QueryStoreStatsCorrectedHourlyView, 2), (TimescaleSupport.QueryStatsHourlyView, 3), - /* Night zero, in REGISTRY order rather than input order — the daily (registry position 8) - ahead of the interval daily (11) ahead of the baseline (13). */ + /* Night zero, in REGISTRY order rather than input order — the daily ahead of the interval + daily ahead of the baseline, exactly DailyAggregates' and BaselineAggregates' own order. */ (TimescaleSupport.QueryStoreStatsDailyView, 0), (TimescaleSupport.QueryStoreStatsIntervalDailyView, 0), (TimescaleSupport.PerfmonIntervalBaselineView, 0), @@ -460,13 +465,17 @@ are what it costs and saves. */ Assert.All(fresh, s => Assert.Equal(0, s.NightOffset)); Assert.Equal(TimescaleSupport.AggregateCompressionTargets.Select(t => t.View).ToArray(), fresh.Select(s => s.View).ToArray()); - /* Ties on size break on registry order, not on input order. */ + /* Ties on size break on registry order, not on input order. #3653 LC froze QueryStatsHourlyView and + ProcedureStatsHourlyView off the registry, so a tie between those two would now break on neither + order (both read int.MaxValue from the registry lookup) — this uses two aggregates still ON + AggregateCompressionTargets, listed here in the OPPOSITE of registry order, so a pass still proves + the break is by registry position and not by input position. */ var tied = TimescaleSupport.StageAggregateCompressionNights(new[] { - State(TimescaleSupport.ProcedureStatsHourlyView, gib, 3), - State(TimescaleSupport.QueryStatsHourlyView, gib, 3), + State(TimescaleSupport.QueryStoreStatsIntervalHourlyView, gib, 3), + State(TimescaleSupport.QueryStoreStatsHourlyView, gib, 3), }); - Assert.Equal(TimescaleSupport.QueryStatsHourlyView, tied[0].View); + Assert.Equal(TimescaleSupport.QueryStoreStatsHourlyView, tied[0].View); Assert.Equal(0, tied[0].NightOffset); Assert.Equal(1, tied[1].NightOffset); @@ -815,4 +824,58 @@ private static TimeSpan ParseDays(string interval) Assert.Equal("days", parts[1]); return TimeSpan.FromDays(int.Parse(parts[0], CultureInfo.InvariantCulture)); } + + /// + /// #3653 LC point 7: the drain rule's truth table, pure — no job means nothing to drain regardless of + /// chunk count; a job with any uncompressed chunk, however few, stays; only a job with zero uncompressed + /// chunks drains. + /// + [Fact] + public void ShouldDrainFrozenDailyCompression_OnlyWhenAJobExistsAndNoChunkIsUncompressed() + { + Assert.False(TimescaleSupport.ShouldDrainFrozenDailyCompression( + new TimescaleSupport.FrozenDailyCompressionDrainState(TimescaleSupport.QueryStatsDailyView, JobId: null, UncompressedChunks: 0))); + Assert.False(TimescaleSupport.ShouldDrainFrozenDailyCompression( + new TimescaleSupport.FrozenDailyCompressionDrainState(TimescaleSupport.QueryStatsDailyView, JobId: null, UncompressedChunks: 5))); + Assert.False(TimescaleSupport.ShouldDrainFrozenDailyCompression( + new TimescaleSupport.FrozenDailyCompressionDrainState(TimescaleSupport.QueryStatsDailyView, JobId: 42, UncompressedChunks: 1))); + Assert.True(TimescaleSupport.ShouldDrainFrozenDailyCompression( + new TimescaleSupport.FrozenDailyCompressionDrainState(TimescaleSupport.QueryStatsDailyView, JobId: 42, UncompressedChunks: 0))); + + Assert.Throws(() => TimescaleSupport.ShouldDrainFrozenDailyCompression(null!)); + } + + /// + /// #3653 LC point 7: the drain's probe names exactly the three frozen legacy dailies, counts EVERY + /// uncompressed chunk (no range_end/age filter, unlike 's + /// eligible_under_*_rule columns), and never the frozen HOURLIES, which never drain. + /// + [Fact] + public void FrozenDailyCompressionDrainStateSql_NamesExactlyTheThreeFrozenDailies_AndCountsEveryUncompressedChunk() + { + var sql = TimescaleSupport.FrozenDailyCompressionDrainStateSql; + + Assert.Contains($"'{TimescaleSupport.QueryStatsDailyView}'", sql, StringComparison.Ordinal); + Assert.Contains($"'{TimescaleSupport.ProcedureStatsDailyView}'", sql, StringComparison.Ordinal); + Assert.Contains($"'{TimescaleSupport.QueryStatsDbDailyView}'", sql, StringComparison.Ordinal); + Assert.Contains("NOT c.is_compressed", sql, StringComparison.Ordinal); + + /* No age gate — every uncompressed chunk counts, not just the ones a tier's compress_after would + already call eligible. */ + Assert.DoesNotContain("range_end", sql, StringComparison.Ordinal); + Assert.DoesNotContain("eligible_under", sql, StringComparison.Ordinal); + + /* Never a frozen HOURLY — those never drain; their chunks age out through retention instead. */ + Assert.DoesNotContain($"'{TimescaleSupport.QueryStatsHourlyView}'", sql, StringComparison.Ordinal); + Assert.DoesNotContain($"'{TimescaleSupport.ProcedureStatsHourlyView}'", sql, StringComparison.Ordinal); + Assert.DoesNotContain($"'{TimescaleSupport.QueryStatsDbHourlyView}'", sql, StringComparison.Ordinal); + + /* Exactly three — one per SupersededDailyRollups member (LegacyDaily), not one per the six-member + FrozenRollupAggregates, which also holds the three hourlies the probe must never touch. */ + Assert.Equal(3, TimescaleSupport.SupersededDailyRollups.Length); + + var remove = TimescaleSupport.RemoveFrozenDailyCompressionPolicySql(TimescaleSupport.QueryStatsDailyView); + Assert.Contains("remove_compression_policy('collect.query_stats_daily'", remove, StringComparison.Ordinal); + Assert.Contains("if_exists => true", remove, StringComparison.Ordinal); + } } diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index e0143c2419..582854e12a 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -2134,17 +2134,18 @@ exists so the daily tier eventually stops inheriting it. Each carries its legacy WITH NO DATA: these start empty and are filled by --backfill-rollups (see docs/runbooks/a6-successor-daily-backfill.md) or by their own refresh policy reaching forward from - creation. EXCLUDED from aggregate compression for now (see CompressionDeferredUntilFreeze) — the daily - compression band is full at 23 members, and freeing three slots for these is #3653 A6 lane LC's job, - not this one's. Registered in DailyAggregates, RollupViews and the availability probe so the ensure - sweep creates them, the backfill verb and coverage probe see them, and SupersededDailyRollups records - which legacy daily each will eventually replace. */ + creation. Compression-registered like any other member of DailyAggregates since #3653 A6 lane LC's + freeze retired the legacy trio's own daily-band membership and freed the three slots these take in + AggregateCompressionTargets (see that list's own note). Registered in DailyAggregates, RollupViews and + the availability probe so the ensure sweep creates them, the backfill verb and coverage probe see them, + and SupersededDailyRollups records which legacy daily each replaced. */ /// The INTERVAL-HONEST successor of (#3653, A6) — /// hierarchical from , not from the legacy /// its sibling reads. Same dims and re-aggregated columns as the /// legacy daily, plus sum(sample_interval_seconds_sum) carried forward from the successor hourly. - /// WITH NO DATA; excluded from compression until the freeze (). + /// WITH NO DATA; compression-registered like any other member since + /// #3653 A6 lane LC's freeze (). public const string CreateQueryStatsIntervalDailySql = @"CREATE MATERIALIZED VIEW IF NOT EXISTS collect.query_stats_interval_daily WITH (timescaledb.continuous) AS SELECT @@ -6724,25 +6725,6 @@ internal static string WholeDaysInterval(TimeSpan span) /// public const string AggregateCompressionSegmentBy = "server_id"; - /// - /// The three interval-honest successor DAILIES (#3653, A6), held OUT of - /// so the daily compression band — full at 23 members — does not - /// overflow to 26. They are still created (registered in ), refreshed, and - /// backfillable; they simply carry no compression policy YET. #3653 A6 lane LC's freeze retires the legacy - /// trio's own daily-band membership, which is what frees the three slots this set holds these back for; - /// once that lands, LC removes this set (or these three names from it) and the ensure sweep enables their - /// compression on the next start, the same as any newly-registered aggregate. - /// - /// MUST stay declared before : static field - /// initializers run in declaration order, and that list reads this set. - /// - public static readonly IReadOnlySet CompressionDeferredUntilFreeze = new HashSet(StringComparer.Ordinal) - { - QueryStatsIntervalDailyView, - ProcedureStatsIntervalDailyView, - QueryStatsDbIntervalDailyView, - }; - /// /// Every continuous aggregate this product owns, paired with the CREATE that defines it and whether its /// refresh policy is hourly — the registry the compression ensure walks, in the order that ALSO decides each @@ -6753,15 +6735,22 @@ internal static string WholeDaysInterval(TimeSpan span) /// — rather than hand-listed, so an aggregate registered for creation is compression-registered the same /// moment, with its tier decided by which list it came from. There is no fourth list to forget. /// + /// Twenty, not twenty-three (#3653, LC). A set once named CompressionDeferredUntilFreeze + /// held the three interval-honest successor dailies (, + /// , ) out of this + /// list from A6 until LC's freeze — the daily band was full at 23 members with the legacy trio still on it, + /// so the successors joining unconditionally would have overflowed it. LC moved the legacy trio itself out + /// of and (into + /// ), which is what frees the three slots the successors take here — so + /// this list is declared straight from the three source lists, with no deferral to subtract. + /// /// MUST stay declared after those three lists: static field initializers run in declaration /// order, and this one reads all three. /// - public static readonly IReadOnlyList<(string CreateSql, string View, bool Hourly)> AggregateCompressionTargets = HourlyAggregates.Select(a => (a.CreateSql, a.View, Hourly: true)) .Concat(DailyAggregates.Select(a => (a.CreateSql, a.View, Hourly: false))) .Concat(BaselineAggregates.Select(a => (a.CreateSql, a.View, Hourly: true))) - .Where(a => !CompressionDeferredUntilFreeze.Contains(a.View)) .ToArray(); /// @@ -7870,6 +7859,11 @@ compress after 2 days" would be a universal claim with seven counterexamples in AggregateCompressionBandMinute, AggregateCompressionBandFirstHour, stagedLines.Count == 0 ? "none this start" : string.Join("; ", stagedLines)); + /* #3653 LC point 7, the drain: a separate concern from everything above — these three are never + AggregateCompressionTargets, so the states/needingPolicy/converge loops above never see them and + this call never sees what they did. Runs last so it costs no target its own progress if it fails. */ + await DrainFrozenDailyCompressionPoliciesAsync(connection, logger, cancellationToken); + return inPlace; } @@ -7878,6 +7872,169 @@ compress after 2 days" would be a universal claim with seven counterexamples in private static string FormatGiB(long bytes) => (bytes / 1073741824d).ToString("0.0", CultureInfo.InvariantCulture) + " GiB"; + /* ─────────────── the frozen dailies' compression drain (#3653, LC point 7) ─────────────── */ + + /// + /// One of the three frozen legacy dailies' ('s LegacyDaily + /// members) own compression drain state, as reads it — + /// its compression job if any, and how many of its materialization's chunks are uncompressed. + /// + public sealed record FrozenDailyCompressionDrainState(string View, int? JobId, long UncompressedChunks); + + /// + /// The three frozen legacy dailies' own compression drain state in one read: whether each still carries a + /// compression job, and how many of its materialization's chunks are uncompressed — EVERY uncompressed + /// chunk, of any age, not the age-gated eligible_under_hourly_rule / eligible_under_daily_rule + /// columns reads for the registered targets. + /// + /// Why not those columns. They answer "how many chunks would this policy's NEXT run compress", + /// which is gated on a tier's compress_after — a chunk younger than that window is not eligible YET + /// but is still uncompressed. asks a different question: + /// has this aggregate finished compressing everything it will ever hold. A young uncompressed chunk must + /// still count toward that, or the drain would remove the policy while a chunk was still waiting its turn + /// to compress, and nothing would ever compress it afterward. + /// + /// Scoped BY NAME to exactly 's three LegacyDaily members + /// — the only three this rule ever drains — rather than reading every non-target aggregate under + /// collect, so an unrelated off-grid or newly-registered-but-not-yet-enabled aggregate can never be + /// swept in by a broadened WHERE later. Same join shape as + /// (job resolved on either the view or the materialization identity) for the same measured reason. + /// + public static string FrozenDailyCompressionDrainStateSql + { + get + { + var views = string.Join(", ", SupersededDailyRollups.Select(s => $"'{s.LegacyDaily}'")); + return $@" +SELECT + ca.view_name, + j.job_id, + ( + SELECT count(*) + FROM timescaledb_information.chunks AS c + WHERE c.hypertable_schema = ca.materialization_hypertable_schema + AND c.hypertable_name = ca.materialization_hypertable_name + AND NOT c.is_compressed + ) AS uncompressed_chunks +FROM timescaledb_information.continuous_aggregates AS ca +LEFT JOIN timescaledb_information.jobs AS j + ON (j.proc_name LIKE '%compression%' OR j.proc_name LIKE '%columnstore%') + AND ( + (j.hypertable_schema = ca.view_schema AND j.hypertable_name = ca.view_name) + OR (j.hypertable_schema = ca.materialization_hypertable_schema AND j.hypertable_name = ca.materialization_hypertable_name) + ) +WHERE ca.view_schema = 'collect' +AND ca.view_name IN ({views})"; + } + } + + /// + /// THE DRAIN RULE (#3653, LC point 7), pure so the truth table pins directly: does + /// 's frozen daily lose its compression policy now? Only once there is a policy to + /// lose — JobId not null; a store that never enabled compression on it, or has already been + /// drained, changes nothing — AND every one of its materialization's chunks is already compressed + /// (UncompressedChunks == 0). Any uncompressed chunk, however old or young, holds the policy in + /// place, because compression still has a chunk left to do; once that count reaches zero it can only ever + /// STAY zero, since nothing refreshes a frozen daily past the freeze to create another one + /// (). + /// + public static bool ShouldDrainFrozenDailyCompression(FrozenDailyCompressionDrainState state) + { + if (state is null) + { + throw new ArgumentNullException(nameof(state)); + } + + return state.JobId is not null && state.UncompressedChunks == 0; + } + + /// Removes a DRAINED frozen daily's compression policy () + /// — if_exists, so a policy already removed on a prior start, or never created, changes nothing. The + /// chunks it already compressed keep their compression (this removes the JOB, not the columnstore data); + /// it only stops a future run from firing on an aggregate that will never materialize another chunk. + public static string RemoveFrozenDailyCompressionPolicySql(string view) + => $"SELECT remove_compression_policy('collect.{view}', if_exists => true)"; + + /// + /// THE DRAIN (#3653, LC point 7): walks the three frozen legacy dailies and removes a compression policy + /// that has nothing left to compress (), via + /// . Called once, at the end of + /// — a separate concern from that function's own + /// enable/stage/converge loops, which never see these three ( + /// excludes them) the way this method's own scoped probe never sees a registered target. + /// + /// Why these three ever HAD a policy to drain. Every store that took a build before LC's + /// freeze enabled compression and attached a policy on these three the same as any other daily-tier + /// aggregate; the freeze stopped their REFRESH () but left + /// the existing compression policy running, because a policy already converged onto the shipped window + /// is not itself wrong — it is aimed at an aggregate that will, from the freeze on, only ever gain MORE + /// compressed chunks and never another uncompressed one. This is what eventually turns that policy off, + /// once there is truly nothing left for it to do — never on a fresh store past the freeze, which creates + /// these three with no compression policy at all ( carries no policy + /// builder), so there is nothing here to drain and reads a + /// null JobId forever. + /// + /// Failure-isolated per aggregate and per statement, the same as every other step in + /// : a read or removal that fails costs a warning and the + /// others still run. + /// + public static async Task DrainFrozenDailyCompressionPoliciesAsync(NpgsqlConnection connection, ILogger? logger, CancellationToken cancellationToken = default) + { + if (connection is null) + { + throw new ArgumentNullException(nameof(connection)); + } + + var states = new List(); + try + { + using var probe = new NpgsqlCommand(FrozenDailyCompressionDrainStateSql, connection) { CommandTimeout = SetupTimeoutSeconds }; + await using var reader = await probe.ExecuteReaderAsync(cancellationToken); + while (await reader.ReadAsync(cancellationToken)) + { + states.Add(new FrozenDailyCompressionDrainState( + View: reader.GetString(0), + JobId: reader.IsDBNull(1) ? null : Convert.ToInt32(reader.GetValue(1), CultureInfo.InvariantCulture), + UncompressedChunks: Convert.ToInt64(reader.GetValue(2), CultureInfo.InvariantCulture))); + } + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogDebug( + "TimescaleDB: could not read the frozen dailies' compression drain state, so none was drained this start: {Message}", + ex.Message); + return 0; + } + + var drained = 0; + foreach (var state in states) + { + if (!ShouldDrainFrozenDailyCompression(state)) + { + continue; + } + + try + { + using var remove = new NpgsqlCommand(RemoveFrozenDailyCompressionPolicySql(state.View), connection) { CommandTimeout = SetupTimeoutSeconds }; + await remove.ExecuteNonQueryAsync(cancellationToken); + drained++; + + logger?.LogInformation( + "TimescaleDB: removed frozen daily {View}'s compression policy (job {JobId}) — every chunk it will ever hold is already compressed, and nothing refreshes it past the freeze to make another (#3653).", + state.View, state.JobId); + } + catch (Exception ex) when (ex is not OperationCanceledException) + { + logger?.LogWarning( + "Could not remove frozen daily {View}'s drained compression policy (job {JobId}) — it stays, harmlessly, since it has nothing left to compress: {Message}", + state.View, state.JobId, ex.Message); + } + } + + return drained; + } + /* ─────────────── rollup availability (the plain-PostgreSQL guard, #1664) ─────────────── */ /// From ed49daf64fce9eb808023b52dc78e3051cc35387 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 21:56:29 -0400 Subject: [PATCH 03/18] issue-3653 A6 lane LC-a2: point the raw purge and retention coverage at the successors, not the frozen legacy rollups RawTierCoverage's query_stats/procedure_stats rows and RetentionPolicies' three successor-hourly rows still named the six rollups LC froze off the refresh grid. A gate keyed on a frozen relation either finds it empty once its own retention trims it with nothing refilling it (raw purge, blocks forever) or never actually asks whether the real consumer caught up (successor retention, since the legacy daily's floor is fixed and always looks "covered"). Both are now derived from SupersededHourlyRollups and SupersededDailyRollups through two new lookups (SuccessorOf's daily counterpart, and throwing wrappers for callers that already know the relation is superseded), so the two registries and the two gates cannot drift apart again. Re-pinned the one IntervalHonestHourlyRollupTests test that exercised this by name; left the compression-band, refresh-grid, slot and repair-target pins this touches for their respective owners. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../IntervalHonestHourlyRollupTests.cs | 46 ++++++++---- .../TimescaleSupport.cs | 72 +++++++++++++++---- 2 files changed, 90 insertions(+), 28 deletions(-) diff --git a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs index 2d34ae2f96..f0904206cb 100644 --- a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs +++ b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs @@ -97,13 +97,16 @@ public void ThreePairs_LegacyStaysRegistered_SuccessorAppended_DailyHangsOffTheL /// Every successor is a raw-sourced hourly rollup registered where the readers, the coverage probe, the /// backfill and the retention ladder look — , /// , , - /// — and NOT where the raw purge gates - /// (), which is the #1661 precedent stated on the registry. + /// — and, since #3653's LC froze the legacy trio off the raw + /// purge, the two with a row of their own + /// (query_stats, procedure_stats) are there too — the #1661 precedent the registry's summary + /// explains is now reversed BY DESIGN. query_stats_db_hourly shares query_stats' raw table + /// rather than owning one, so its successor was never a candidate and stays out, same as before LC. /// [Fact] - public void EverySuccessor_IsInTheCoverageProbe_TheBackfillPlan_AndTheRetentionLadder_ButNotTheRawGate() + public void EverySuccessor_IsInTheCoverageProbe_TheBackfillPlan_TheRetentionLadder_AndTheRawGateWhereItHasOne() { - foreach (var (legacy, successor, dependentDaily) in TimescaleSupport.SupersededHourlyRollups) + foreach (var (legacy, successor, _) in TimescaleSupport.SupersededHourlyRollups) { var legacyRow = TimescaleSupport.RollupViews.Single(r => r.View == legacy); var successorRow = TimescaleSupport.RollupViews.Single(r => r.View == successor); @@ -118,22 +121,39 @@ public void EverySuccessor_IsInTheCoverageProbe_TheBackfillPlan_AndTheRetentionL Assert.False(RollupAvailability.WithoutIntervalHourlies.Has(successor)); Assert.True(RollupAvailability.WithoutIntervalHourlies.Has(legacy)); - /* Depth 0 in the backfill order — raw-sourced — and the probe SQL reads its own name. */ + /* Depth 0 in the backfill order — raw-sourced — and the probe SQL reads its own name. The legacy's + OWN daily left the backfill plan with the rest of the frozen six (#3653 LC: nothing ever advances + its watermark again); the successor's consumer there is now its own successor daily (#3653 LB), + read through SuccessorDailyOf rather than restated. */ var target = RollupBackfill.Targets.Single(t => t.View == successor); Assert.False(target.IsHierarchical); - Assert.True(Array.IndexOf(RollupBackfill.Targets, target) < Array.IndexOf(RollupBackfill.Targets, RollupBackfill.Targets.Single(t => t.View == dependentDaily)), - "a raw-sourced successor must be planned before every hierarchical daily"); - - /* The hourly tier's horizon and the leaf rule: coverage is the sibling daily over the same source. */ + var successorDaily = TimescaleSupport.SuccessorDailyOf(successor) + ?? throw new InvalidOperationException($"{successor} must be in {nameof(TimescaleSupport.SupersededDailyRollups)}."); + Assert.True(Array.IndexOf(RollupBackfill.Targets, target) < Array.IndexOf(RollupBackfill.Targets, RollupBackfill.Targets.Single(t => t.View == successorDaily)), + "a raw-sourced successor must be planned before its own successor daily"); + + /* The hourly tier's horizon; coverage is the successor's OWN daily (#3653 LC), not the legacy daily + SupersededHourlyRollups' third element still names for FrozenRollupAggregates and the coverage + log to read — see RetentionPolicies' summary for why the legacy daily cannot gate this. */ var policy = TimescaleSupport.RetentionPolicies.Single(p => p.Relation == successor); Assert.Equal(TimescaleSupport.HourlyRetentionInterval, policy.DropAfter); Assert.Equal("bucket", policy.TimeColumn); - Assert.Equal(new[] { dependentDaily }, policy.Coverage); - - /* NOT a raw-purge gate consumer (#1661's precedent; the registry states why). */ - Assert.All(TimescaleSupport.RawTierCoverage, tier => Assert.DoesNotContain(successor, tier.Coverage)); + Assert.Equal(new[] { successorDaily }, policy.Coverage); } + /* The raw purge moved onto the two successors with a raw table of their own naming them (#3653 LC). + query_stats_db_hourly has no RawTierCoverage row of its own — it shares "query_stats" with + query_stats_hourly, already that row's named consumer below — so it was never a raw-gate candidate. */ + var querySuccessor = TimescaleSupport.SuccessorOf(TimescaleSupport.QueryStatsHourlyView) + ?? throw new InvalidOperationException($"{TimescaleSupport.QueryStatsHourlyView} must be in {nameof(TimescaleSupport.SupersededHourlyRollups)}."); + var procedureSuccessor = TimescaleSupport.SuccessorOf(TimescaleSupport.ProcedureStatsHourlyView) + ?? throw new InvalidOperationException($"{TimescaleSupport.ProcedureStatsHourlyView} must be in {nameof(TimescaleSupport.SupersededHourlyRollups)}."); + var dbSuccessor = TimescaleSupport.SuccessorOf(TimescaleSupport.QueryStatsDbHourlyView) + ?? throw new InvalidOperationException($"{TimescaleSupport.QueryStatsDbHourlyView} must be in {nameof(TimescaleSupport.SupersededHourlyRollups)}."); + Assert.Equal(new[] { querySuccessor }, TimescaleSupport.RawTierCoverage.Single(t => t.Relation == "query_stats").Coverage); + Assert.Equal(new[] { procedureSuccessor }, TimescaleSupport.RawTierCoverage.Single(t => t.Relation == "procedure_stats").Coverage); + Assert.All(TimescaleSupport.RawTierCoverage, tier => Assert.DoesNotContain(dbSuccessor, tier.Coverage)); + Assert.False(RollupAvailability.WithoutIntervalHourlies.AllPresent); Assert.True(RollupAvailability.All.AllPresent); /* 16 through #3653 Q12, +3 for the A6 successor DAILIES this lane registers diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index e0143c2419..b6abe75fc2 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -1488,6 +1488,37 @@ public static readonly (string LegacyDaily, string SuccessorDaily, string Succes return null; } + /// Same as , for a caller ('s query_stats and + /// procedure_stats rows) that already knows must be superseded — a + /// null there would only hide and its caller drifting apart, so + /// this throws instead of handing one back. + private static string RequireSuccessorOf(string legacyHourly) + => SuccessorOf(legacyHourly) ?? throw new InvalidOperationException($"{legacyHourly} has no successor in {nameof(SupersededHourlyRollups)}."); + + /// The interval-honest successor DAILY that covers once the + /// hourly-tier read routes past its own horizon (#3653, A6/LB's three successor dailies) — the coverage + /// relation arms a successor hourly's own drop_chunks against, in place of + /// the LEGACY daily already froze off the refresh grid. null for + /// anything that is not a SuccessorHourly in . + public static string? SuccessorDailyOf(string successorHourly) + { + foreach (var (_, successorDaily, hourly) in SupersededDailyRollups) + { + if (string.Equals(hourly, successorHourly, StringComparison.Ordinal)) + { + return successorDaily; + } + } + + return null; + } + + /// Same as , for a caller ('s three + /// successor-hourly rows) that already knows must have one — throws + /// rather than silently naming a policy's coverage null. + private static string RequireSuccessorDailyOf(string successorHourly) + => SuccessorDailyOf(successorHourly) ?? throw new InvalidOperationException($"{successorHourly} has no successor daily in {nameof(SupersededDailyRollups)}."); + /// /// THE SUPPLY RULE (#3653), pure so the tests can walk it, and shared: PgBaselineProvider applies it /// per metric against one server's reach into the baseline supplies, and @@ -5995,8 +6026,15 @@ public static string RetentionArmSafetySql(string relation, string sourceTimeCol public static readonly IReadOnlyList<(string Relation, string TimeColumn, IReadOnlyList Coverage)> RawTierCoverage = new (string, string, IReadOnlyList)[] { - ("query_stats", "collection_time", new[] { QueryStatsHourlyView }), - ("procedure_stats", "collection_time", new[] { ProcedureStatsHourlyView }), + /* LC (#3653): named through RequireSuccessorOf rather than the legacy hourlies directly. A gate still + keyed on query_stats_hourly/procedure_stats_hourly would eventually find them EMPTY — their OWN + retention policy (RetentionPolicies, unchanged by LC) keeps trimming their chunks at + HourlyRetentionInterval while the freeze stops anything from refilling them past that point — and + hold the raw purge forever with no self-release. See SupersededHourlyRollups for the full story. */ + ("query_stats", "collection_time", new[] { RequireSuccessorOf(QueryStatsHourlyView) }), + ("procedure_stats", "collection_time", new[] { RequireSuccessorOf(ProcedureStatsHourlyView) }), + /* query_store_stats is not one of #3653's legacy six (see FrozenRollupAggregates) — both consumers + below go on refreshing, so their coverage is still named directly. */ ("query_store_stats", "collection_time", new[] { QueryStoreStatsHourlyView, QueryStoreStatsIntervalHourlyView }), }; @@ -6039,19 +6077,23 @@ public static string RetentionArmSafetySql(string relation, string sourceTimeCol (Relation: QueryStoreStatsHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { QueryStoreStatsDailyView }), (Relation: QueryStatsDbHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { QueryStatsDbDailyView }), - /* The interval-honest hourly successors (#3653, Q12) take the hourly tier's horizon and the LEAF RULE - (#1757): nothing is hierarchical from them, so their consumer is the routed READ, which past - HourlyMaxAge goes to the daily tier — and the daily tier is the LEGACY daily, hierarchical from the - legacy hourly over the same raw rows (see SupersededHourlyRollups for why the successors have no - daily of their own yet). That is the corrected Query Store hourly's shape exactly: a leaf whose - coverage relation is a sibling daily built from a different parent over the same source. What the - gate protects here is the coarsened history — a successor bucket purged at 90 days is a day the - legacy daily has already summed; what it cannot protect, and does not claim to, is the successor's - interval-honest sample_count and min() at the day grain, which no daily carries until one is built - from these. */ - (Relation: QueryStatsIntervalHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { QueryStatsDailyView }), - (Relation: ProcedureStatsIntervalHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { ProcedureStatsDailyView }), - (Relation: QueryStatsDbIntervalHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { QueryStatsDbDailyView }), + /* The interval-honest hourly successors (#3653, Q12) took the LEAF RULE (#1757) at Q12: nothing was + hierarchical from them yet, so their consumer was the routed READ, which past HourlyMaxAge went to + the LEGACY daily — the same relation SupersededHourlyRollups' legacy hourly fed, over the same raw + rows. #3653's LB gave each successor its own successor DAILY, hierarchical from it exactly the way + the legacy trio's own daily is hierarchical from the legacy hourly (SupersededDailyRollups); LC's + freeze retired the leaf shape along with the legacy trio itself. Coverage now names that successor + daily — derived through RequireSuccessorDailyOf rather than the legacy daily, because the legacy + daily is the one relation LC already froze off the refresh grid + (RemoveFrozenRollupRefreshPolicySql): its own coverage floor stopped advancing at the freeze, so + gating the successor's drop_chunks on it would never actually ask whether the RIGHT consumer — the + successor daily — has materialized the bucket about to be dropped, which is precisely the "never + drop what your consumer has not captured yet" failure this whole list exists to prevent (see the + summary above). What the gate protects is unchanged: the coarsened history a successor bucket + summarizes once it is purged at 90 days. */ + (Relation: QueryStatsIntervalHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { RequireSuccessorDailyOf(QueryStatsIntervalHourlyView) }), + (Relation: ProcedureStatsIntervalHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { RequireSuccessorDailyOf(ProcedureStatsIntervalHourlyView) }), + (Relation: QueryStatsDbIntervalHourlyView, DropAfter: HourlyRetentionInterval, TimeColumn: "bucket", Coverage: new[] { RequireSuccessorDailyOf(QueryStatsDbIntervalHourlyView) }), /* The corrected Query Store tier (#1849, extended by #1869). From 6419e4b795dc7c9f84a068757dfb5fd2b30700c6 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:30:14 -0400 Subject: [PATCH 04/18] issue-3653 A6 lane LC-a5: re-derive the refresh grid pins for the freeze's 12-light, 21-minute window (#4186) The freeze moved the legacy trio off the hourly refresh grid and into FrozenRollupAggregates, with their successors taking the three positions the legacy trio held instead of staying appended behind it. That returns the grid to its pre-Q12 shape (twelve light members, 21-minute heaviest window, 1,050 s watch line) rather than Q12's fifteen/18/900. Ten pins across four files pinned the old Q12 numbers; this re-derives each one against the product's own functions and updates the doc comments to narrate the append-then-freeze round trip. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- Darling/Darling.Tests/BaselineSupplyTests.cs | 35 ++- .../CompressionPhaseAssignmentTests.cs | 4 +- .../IntervalHonestHourlyRollupTests.cs | 29 +- .../Darling.Tests/TimescaleSupportTests.cs | 268 +++++++++--------- 4 files changed, 182 insertions(+), 154 deletions(-) diff --git a/Darling/Darling.Tests/BaselineSupplyTests.cs b/Darling/Darling.Tests/BaselineSupplyTests.cs index cef47be22d..3246c5cf64 100644 --- a/Darling/Darling.Tests/BaselineSupplyTests.cs +++ b/Darling/Darling.Tests/BaselineSupplyTests.cs @@ -490,13 +490,19 @@ of the new. */ /// SupersededBaselineRelations); this pin names the reason. /// /// The grid DID move at #3653 (Q12), and not because of this pair. The three interval-honest - /// HOURLY successors could not replace their legacies (the daily tier is hierarchical from those — - /// SupersededHourlyRollups), so they were appended to HourlyAggregates and the grid - /// re-derived by its own method: sixteen policies, the heaviest at :18, the watch line at 900 s. The two - /// bounded hourly successors are dealt into the bounded class ahead of the baselines, so this pair's - /// minutes moved 3→6 and 5→7 — the converge re-phases them once. What this pin still holds is the - /// baseline half: seven members, the pair in front, and the baseline pair adding NOTHING to the count the - /// grid derives from. The grid's own figures are pinned with their derivation in TimescaleSupportTests + /// HOURLY successors were appended to HourlyAggregates behind the legacy trio (the daily tier was + /// still hierarchical from it) and the grid re-derived by its own method: sixteen policies, the heaviest + /// at :18, the watch line at 900 s. The two bounded hourly successors were dealt into the bounded class + /// ahead of the baselines, so this pair's minutes moved 3→6 and 5→7. + /// + /// And it moved BACK at #3653 (LC), again not because of this pair. The freeze gave the daily + /// tier its own interval-honest successors (A6 LB) and moved the legacy trio off the grid entirely, into + /// FrozenRollupAggregates, with their hourly successors taking the three positions the legacy trio + /// held rather than staying appended behind it — thirteen policies again, the heaviest back at :15, the + /// watch line back at 1,050 s, and this pair's minutes back at 3 and 5. The converge re-phases every moved + /// policy once, on the first start of each build. What this pin holds throughout is the baseline half: + /// seven members, the pair in front, and the baseline pair adding NOTHING to the count the grid derives + /// from. The grid's own figures are pinned with their derivation in TimescaleSupportTests /// (CompressionPhaseGrid_ClearsEveryRefreshSlotsGuardBand_AndTheHeaviestRefreshsSlotWhole and /// TheRefreshGridIsUnchanged_AndTheCompressionGridsOneInputFromItIsPinned), not restated here. /// @@ -507,14 +513,15 @@ public void Successors_TookTheLegacyPositions_SoThePhaseGridDidNotMove() Assert.Equal(TimescaleSupport.PerfmonIntervalBaselineView, TimescaleSupport.BaselineAggregates[0].View); Assert.Equal(TimescaleSupport.WaitStatsIntervalBaselineView, TimescaleSupport.BaselineAggregates[1].View); - /* The order's count is the hourly registry plus the seven baselines — 9 + 7 = 16 since #3653 — and - the baseline pair contributes exactly its two positions to it, no more. */ + /* The order's count is the hourly registry plus the seven baselines — 6 + 7 = 13 since #3653's LC + freeze (was 9 + 7 = 16 during Q12) — and the baseline pair contributes exactly its two positions to + it, no more. */ Assert.Equal(TimescaleSupport.HourlyAggregates.Length + 7, TimescaleSupport.HourlyRefreshPhaseOrder.Count); - Assert.Equal(16, TimescaleSupport.HourlyRefreshPhaseOrder.Count); - Assert.Equal(6, TimescaleSupport.RefreshPhaseMinutesFor(TimescaleSupport.PerfmonIntervalBaselineView)); - Assert.Equal(7, TimescaleSupport.RefreshPhaseMinutesFor(TimescaleSupport.WaitStatsIntervalBaselineView)); - Assert.Equal(18, TimescaleSupport.HeaviestRefreshStartMinute); - Assert.Equal(900, TimescaleSupport.RefreshSlotWarningSeconds); + Assert.Equal(13, TimescaleSupport.HourlyRefreshPhaseOrder.Count); + Assert.Equal(3, TimescaleSupport.RefreshPhaseMinutesFor(TimescaleSupport.PerfmonIntervalBaselineView)); + Assert.Equal(5, TimescaleSupport.RefreshPhaseMinutesFor(TimescaleSupport.WaitStatsIntervalBaselineView)); + Assert.Equal(15, TimescaleSupport.HeaviestRefreshStartMinute); + Assert.Equal(1050, TimescaleSupport.RefreshSlotWarningSeconds); } /// diff --git a/Darling/Darling.Tests/CompressionPhaseAssignmentTests.cs b/Darling/Darling.Tests/CompressionPhaseAssignmentTests.cs index 8abd823043..d5eeb86e54 100644 --- a/Darling/Darling.Tests/CompressionPhaseAssignmentTests.cs +++ b/Darling/Darling.Tests/CompressionPhaseAssignmentTests.cs @@ -241,7 +241,9 @@ public void TheGridDoesNotMove_TheAssignmentIsAPermutationOfTheSameMinutes() { Assert.Equal(24, TimescaleSupport.CompressionPhaseBandMinutes); Assert.Equal(24, TimescaleSupport.CompressionPhaseMinutes.Count); - Assert.Equal(18, TimescaleSupport.HeaviestRefreshWindowMinutes); + /* #3653 (LC): back to 21 minutes now that the legacy trio froze off the refresh grid and their + successors took the three positions instead of staying appended behind it (was 18 during Q12). */ + Assert.Equal(21, TimescaleSupport.HeaviestRefreshWindowMinutes); Assert.Equal(35, TimescaleSupport.AggregateCompressionBandMinute); Assert.Equal(3, TimescaleSupport.CompressionPhaseMaxPerMinute); diff --git a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs index f0904206cb..b94479ee00 100644 --- a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs +++ b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs @@ -52,32 +52,39 @@ public sealed class IntervalHonestHourlyRollupTests /* ─────────────────────────── the registry ─────────────────────────── */ /// - /// Three pairs, each legacy REGISTERED (unlike #3698's pair), each successor registered and appended after - /// the corrected Query Store pair, each dependent daily registered and hierarchical from the LEGACY. The - /// structural facts the "stays registered" reasoning rests on, asserted against the shipped lists and the - /// shipped CREATE text rather than restated. + /// Three pairs, each legacy FROZEN off the hourly grid and the daily tier since #3653's LC (unlike Q12, + /// when both stayed registered and refreshing), each successor holding the hourly position its legacy + /// held, each dependent daily frozen beside its legacy rather than hierarchical from the grid. The + /// structural facts the freeze's "off the grid for good, never dropped" reasoning rests on, asserted + /// against the shipped lists and the shipped CREATE text rather than restated. /// [Fact] - public void ThreePairs_LegacyStaysRegistered_SuccessorAppended_DailyHangsOffTheLegacy() + public void ThreePairs_LegacyFrozenOffTheGrid_SuccessorTakesItsHourlyPosition_DailyFreezesBesideTheLegacy() { var hourly = TimescaleSupport.HourlyAggregates.Select(a => a.View).ToArray(); var daily = TimescaleSupport.DailyAggregates.Select(a => a.View).ToHashSet(StringComparer.Ordinal); + var frozen = TimescaleSupport.FrozenRollupAggregates.Select(a => a.View).ToHashSet(StringComparer.Ordinal); Assert.Equal(3, TimescaleSupport.SupersededHourlyRollups.Length); - Assert.Equal(9, hourly.Length); + Assert.Equal(6, hourly.Length); + Assert.Equal(6, TimescaleSupport.FrozenRollupAggregates.Length); foreach (var (legacy, successor, dependentDaily) in TimescaleSupport.SupersededHourlyRollups) { - Assert.Contains(legacy, hourly); + Assert.DoesNotContain(legacy, hourly); + Assert.Contains(legacy, frozen); Assert.Contains(successor, hourly); Assert.NotEqual(legacy, successor); Assert.True(Array.IndexOf(hourly, successor) > Array.IndexOf(hourly, TimescaleSupport.QueryStoreStatsCorrectedHourlyView), - $"{successor} must be appended after the corrected Query Store pair, not inserted beside its legacy — an insertion re-deals the bounded band positions"); + $"{successor} must hold a position after the corrected Query Store pair, the shape the phase grid derives its light-band ordinals from"); Assert.Equal(successor, TimescaleSupport.SuccessorOf(legacy)); Assert.Null(TimescaleSupport.SuccessorOf(successor)); - Assert.Contains(dependentDaily, daily); - var dailyCreate = TimescaleSupport.DailyAggregates.Single(a => a.View == dependentDaily).CreateSql; + /* The dependent daily froze WITH its legacy: both left HourlyAggregates/DailyAggregates and the + phase grid together, so the daily is read off FrozenRollupAggregates now, not DailyAggregates. */ + Assert.DoesNotContain(dependentDaily, daily); + Assert.Contains(dependentDaily, frozen); + var dailyCreate = TimescaleSupport.FrozenRollupAggregates.Single(a => a.View == dependentDaily).CreateSql; Assert.Contains($"FROM collect.{legacy}", dailyCreate, StringComparison.Ordinal); Assert.DoesNotContain($"FROM collect.{successor}", dailyCreate, StringComparison.Ordinal); } @@ -87,7 +94,7 @@ public void ThreePairs_LegacyStaysRegistered_SuccessorAppended_DailyHangsOffTheL Assert.Null(TimescaleSupport.SuccessorOf(TimescaleSupport.QueryStoreStatsIntervalHourlyView)); Assert.Null(TimescaleSupport.SuccessorOf(TimescaleSupport.QueryStoreStatsCorrectedHourlyView)); - /* Neither list's names appear in the other's: a legacy here is registered, a legacy there is not. */ + /* Neither list's names appear in the other's: a legacy here is frozen, a legacy there is retired on sight. */ var baselineLegacies = TimescaleSupport.SupersededBaselineRelations.Select(s => s.Legacy).ToHashSet(StringComparer.Ordinal); Assert.Empty(TimescaleSupport.SupersededHourlyRollups.Select(s => s.Legacy).Intersect(baselineLegacies)); Assert.Empty(TimescaleSupport.SupersededHourlyRollups.Select(s => s.Legacy).Intersect(TimescaleSupport.RetiredBaselineRelations)); diff --git a/Darling/Darling.Tests/TimescaleSupportTests.cs b/Darling/Darling.Tests/TimescaleSupportTests.cs index 0ba1a8db62..26d39258f6 100644 --- a/Darling/Darling.Tests/TimescaleSupportTests.cs +++ b/Darling/Darling.Tests/TimescaleSupportTests.cs @@ -2481,18 +2481,21 @@ width and a rounded measurement certifies a derivation the code does not have as / TimescaleSupport.CompressionPhaseMaxPerMinute, TimescaleSupport.CompressionPhaseBandMinutes); - /* RE-DERIVED at #3653 (Q12) by the method the grid documents, with the three interval-honest hourly - successors registered: sixteen hourly policies, fifteen of them light, so the light band spans - (15 - 1) * 1 = 14 minutes (was 11), the heaviest refresh starts at 14 + the 4-minute guard = :18 - (was :15), and its window is the remainder 60 - 18 - 24 = 18 minutes (was 21). The literals move - WITH the identities beside them; nothing here was renumbered by hand. */ - Assert.Equal(15, TimescaleSupport.LightHourlyRefreshCount); - Assert.Equal(14, TimescaleSupport.LightBandSpanMinutes); - Assert.Equal(18, TimescaleSupport.HeaviestRefreshStartMinute); + /* RE-DERIVED at #3653 (Q12) to fifteen light members when the three interval-honest hourly successors + were APPENDED beside the legacy trio, and RE-DERIVED AGAIN at #3653 (LC) back to twelve: the freeze + moved the legacy trio off this grid entirely (into FrozenRollupAggregates) while the successors took + its three positions in HourlyAggregates rather than joining behind it, so the count returns to + thirteen hourly policies, twelve of them light. The light band spans (12 - 1) * 1 = 11 minutes (was + 14 at Q12, 11 before it), the heaviest refresh starts at 11 + the 4-minute guard = :15 (was :18, :15 + before Q12), and its window is the remainder 60 - 15 - 24 = 21 minutes (was 18, 21 before Q12). The + literals move WITH the identities beside them; nothing here was renumbered by hand. */ + Assert.Equal(12, TimescaleSupport.LightHourlyRefreshCount); + Assert.Equal(11, TimescaleSupport.LightBandSpanMinutes); + Assert.Equal(15, TimescaleSupport.HeaviestRefreshStartMinute); Assert.Equal( TimescaleSupport.LightBandSpanMinutes + TimescaleSupport.CompressionPhaseGuardMinutes, TimescaleSupport.HeaviestRefreshStartMinute); - Assert.Equal(18, TimescaleSupport.HeaviestRefreshWindowMinutes); + Assert.Equal(21, TimescaleSupport.HeaviestRefreshWindowMinutes); Assert.Equal( TimescaleSupport.MinutesInHourlyCadence - TimescaleSupport.HeaviestRefreshStartMinute @@ -2534,10 +2537,11 @@ next. A recorded ceiling that outgrew the window therefore has to FAIL here rath .Distinct() .OrderBy(m => m) .ToArray(); - /* Fifteen light minutes :00-:14 (twelve before #3653) and the heaviest at :18 (was :15). The compression - band does NOT move: it is the hour's remainder after the heaviest window, and the window shrank by - exactly what the light band grew, so :36-:59 stays where #3174 left it. */ - Assert.Equal(new[] { 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 18 }, refreshMinutes); + /* Twelve light minutes :00-:11 and the heaviest at :15 (fifteen light minutes :00-:14 and :18 during + #3653's Q12, before the freeze moved the legacy trio off this grid). The compression band does NOT + move: it is the hour's remainder after the heaviest window, and the window grew back by exactly + what the light band shrank, so :36-:59 stays where #3174 left it. */ + Assert.Equal(new[] { 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 15 }, refreshMinutes); Assert.Equal(TimescaleSupport.HourlyRefreshPhaseOrder.Count, refreshMinutes.Length); var heaviestMinute = TimescaleSupport.RefreshPhaseMinutesFor(TimescaleSupport.HeaviestHourlyRefreshView); @@ -2586,10 +2590,11 @@ filter produced output". */ .Select(m => m % TimescaleSupport.MinutesInHourlyCadence) .ToArray(); - /* Fifteen light starts at :00-:14, each excluding its own minute and the three after it: :00-:17, eighteen - minutes (was :00-:14, fifteen). The heaviest window is the 18 minutes :18-:35 (was 21, :15-:35). */ - Assert.Equal(18, excludedByGuard.Length); - Assert.Equal(18, excludedByHeaviestWindow.Length); + /* Twelve light starts at :00-:11, each excluding its own minute and the three after it: :00-:14, fifteen + minutes (was :00-:17, eighteen, during Q12). The heaviest window is the 21 minutes :15-:35 (was 18, + :18-:35, during Q12). */ + Assert.Equal(15, excludedByGuard.Length); + Assert.Equal(21, excludedByHeaviestWindow.Length); Assert.Contains(heaviestMinute + 1, excludedByHeaviestWindow); Assert.Empty(excludedByGuard.Intersect(TimescaleSupport.CompressionPhaseMinutes)); Assert.Empty(excludedByHeaviestWindow.Intersect(TimescaleSupport.CompressionPhaseMinutes)); @@ -2660,15 +2665,14 @@ its own two terms have moved. */ + "line, so the figure the grid is sized against classifies as a warning and the watch reports the " + "grid's own sizing rather than anything new"); - /* #3653 (Q12): the re-derived 18-minute window puts the line at 1,080 * 5 / 6 = 900 s, and the ceiling - sits 4 s under it — 0% of itself, where the 21-minute window left 17%. The ordering the pin holds - still holds, by the ceiling essay's own arithmetic ("18 whole minutes" is the smallest window whose - five-sixths line clears 896 s); the GAP is stated in seconds beside the percentage because a - percentage that truncates to 0 would otherwise read as "none" when it is four seconds. A fourth - light member, or a ceiling re-derived one run higher, turns this red — which is the ruling the grid - asks for at that point rather than a hand-tuned band. */ - Assert.Equal(0, (TimescaleSupport.RefreshSlotWarningSeconds - ceiling) * 100 / ceiling); - Assert.Equal(4, TimescaleSupport.RefreshSlotWarningSeconds - ceiling); + /* #3653's LC freeze restores the 21-minute window: the line is 1,260 * 5 / 6 = 1,050 s, and the + ceiling sits 154 s under it — 17% of itself (where Q12's briefly-narrowed 18-minute window had left + only 4 s, 0%). The ordering the pin holds still holds, by the ceiling essay's own arithmetic; the + GAP is stated in seconds beside the percentage because a percentage that truncates to 0 can read as + "none" when it is not. A fourth light member, or a ceiling re-derived one run higher, turns this + red — which is the ruling the grid asks for at that point rather than a hand-tuned band. */ + Assert.Equal(17, (TimescaleSupport.RefreshSlotWarningSeconds - ceiling) * 100 / ceiling); + Assert.Equal(154, TimescaleSupport.RefreshSlotWarningSeconds - ceiling); } /* ─────────────────── #3044: the watch on the LIVE figure, not the constant ─────────────────── */ @@ -2688,14 +2692,16 @@ percentage that truncates to 0 would otherwise read as "none" when it is four se [Fact] public void TheRefreshSlotWatchLines_AreDerivedFromTheWindow_NotWrittenDown() { - /* #3653 (Q12): the heaviest window is 18 minutes with the three successors on the grid (see - CompressionPhaseGrid_ClearsEveryRefreshSlotsGuardBand_AndTheHeaviestRefreshsSlotWhole for the - derivation), so the slot is 18 * 60 = 1,080 s (was 21 * 60 = 1,260) and the five-sixths line is - 1,080 * 5 / 6 = 900 s (was 1,050). The identities beside the literals are what moved them. */ - Assert.Equal(1080, TimescaleSupport.RefreshPhaseSlotSeconds); + /* #3653's LC freeze moved the legacy trio off this grid (into FrozenRollupAggregates) while their + successors took its three positions in HourlyAggregates, so the heaviest window is back to 21 + minutes (see CompressionPhaseGrid_ClearsEveryRefreshSlotsGuardBand_AndTheHeaviestRefreshsSlotWhole + for the derivation) — the slot is 21 * 60 = 1,260 s (was 18 * 60 = 1,080 during Q12, when the + successors were briefly appended behind the legacy trio instead) and the five-sixths line is + 1,260 * 5 / 6 = 1,050 s (was 900). The identities beside the literals are what moved them. */ + Assert.Equal(1260, TimescaleSupport.RefreshPhaseSlotSeconds); Assert.Equal(TimescaleSupport.HeaviestRefreshWindowMinutes * 60, TimescaleSupport.RefreshPhaseSlotSeconds); - Assert.Equal(900, TimescaleSupport.RefreshSlotWarningSeconds); + Assert.Equal(1050, TimescaleSupport.RefreshSlotWarningSeconds); Assert.Equal( TimescaleSupport.RefreshPhaseSlotSeconds * 5 / 6, TimescaleSupport.RefreshSlotWarningSeconds); @@ -2714,8 +2720,8 @@ a second line rather than only the wall. */ Assert.True( TimescaleSupport.RefreshSlotWarningSeconds < TimescaleSupport.RefreshPhaseSlotSeconds, "the watch line is at or past the slot it is meant to give warning of"); - /* The remaining sixth: 1,080 / 6 = 180 s of lead (was 1,260 / 6 = 210). */ - Assert.Equal(180, TimescaleSupport.RefreshPhaseSlotSeconds - TimescaleSupport.RefreshSlotWarningSeconds); + /* The remaining sixth: 1,260 / 6 = 210 s of lead (was 1,080 / 6 = 180 during Q12). */ + Assert.Equal(210, TimescaleSupport.RefreshPhaseSlotSeconds - TimescaleSupport.RefreshSlotWarningSeconds); } /// @@ -2817,49 +2823,51 @@ public void TheRefreshSlotClassifier_BandsTheLiveReadings_AndKeepsTheWatchLineAb /// and the decision behind it stand still. Which constants a line is a function of moves only when the /// derivations move, and then the decision genuinely does have to be re-taken — as it just was. /// - /// RE-TAKEN AGAIN at #3653 (Q12), because the inversion this test pinned has reversed. The - /// three interval-honest hourly successors re-derived the window to 18 minutes, so the alternative is - /// 1,080 - 240 = 840 s — and 840 s sits 56 s BELOW the 896 s recorded ceiling. The ordering argument - /// #3174 lost is therefore available again: the alternative would classify the grid's own sizing figure - /// as a warning, exactly the crying-wolf failure the five-sixths choice exists to avoid, while the chosen - /// 900 s line still clears it (by 4 s). The rejection now rests on BOTH reasons, and this test pins both: - /// the ordering because it discriminates today, the coupling because it is the one that survives the - /// next geometry change in either direction. The previous version of this test asserted the inversion - /// and said "re-read #3107 rather than editing this assertion" — it was re-read, and the answer is that - /// #3174's coupling paragraph stands and #3107's ordering paragraph is true again; neither is retired. + /// RE-TAKEN AGAIN at #3653 (Q12), because the inversion this test pinned had reversed. The + /// three interval-honest hourly successors were APPENDED behind the legacy trio, re-deriving the window + /// to 18 minutes, so the alternative became 1,080 - 240 = 840 s — 56 s BELOW the 896 s recorded ceiling. + /// The ordering argument #3174 lost was briefly available again alongside the coupling one. + /// + /// AND RE-TAKEN A THIRD TIME at #3653 (LC), because the freeze reverses Q12's append. The + /// legacy trio left this grid entirely (into ) and + /// the successors took its three positions in rather than + /// staying appended behind it, so the window is back to #3174's 21 minutes and the alternative is back to + /// 1,020 s — ABOVE the 896 s ceiling, exactly as it was between #3174 and Q12. The ordering argument is + /// gone again and COUPLING is once more the only reason left; #3174's coupling paragraph above needed no + /// change to say so, which is what makes it the durable half and the ordering paragraphs the transient + /// ones. /// [Fact] - public void TheRejectedWatchLineAlternative_SitsBelowTheRecordedCeilingAgain_SoOrderingAndCouplingBothRejectIt() + public void TheRejectedWatchLineAlternative_NoLongerSitsBelowTheRecordedCeiling_SoTheRejectionIsCouplingAgain() { /* Derived as the prose derives it, then held to the literal too — a re-derivation-only assertion - agrees with any derivation, including one frozen at 480 or at #3174's 1,020. */ + agrees with any derivation, including one frozen at 480 or at Q12's 840. */ var alternative = TimescaleSupport.RefreshPhaseSlotSeconds - (TimescaleSupport.CompressionPhaseGuardMinutes * 60); - Assert.Equal(840, alternative); + Assert.Equal(1020, alternative); Assert.True( alternative < TimescaleSupport.RefreshSlotWarningSeconds, "the alternative is no longer the LOWER of the two lines, so the paragraph rejecting it as the " + "lower one is about something else now"); - /* THE ORDERING, pinned as it stands at #3653's window: the alternative sits BELOW the recorded - ceiling, so on its own it rejects the alternative — a line under the ceiling warns on the very run - the compression grid is sized against. If this ever goes back to true (a wider window, or a - ceiling re-derived downward), the coupling argument below is the only one left and the doc - paragraph on RefreshSlotWarningSeconds has to say so again — re-read #3107 and #3174 rather than - editing this assertion. */ + /* THE INVERSION, pinned as it stands at LC's restored window: the alternative clears the recorded + ceiling, so the ordering the rejection rested on at Q12 no longer discriminates between the two + lines. If this ever goes back to false (a narrower window, or a ceiling re-derived upward), the + ordering argument is available again alongside the coupling one — re-read #3107 and #3174 rather + than editing this assertion. */ var ceiling = TimescaleSupport.HeaviestHourlyRefreshObservedCeilingSeconds; Assert.True( - alternative < ceiling, - $"the {alternative} s alternative clears the {ceiling} s recorded ceiling again, so the ordering " - + "no longer discriminates and the coupling argument is the load-bearing one alone — re-read #3107 " - + "and #3174 rather than editing this assertion"); - Assert.Equal(56, ceiling - alternative); + ceiling < alternative, + $"the {alternative} s alternative sits at or below the {ceiling} s recorded ceiling again, so the " + + "ordering no longer discriminates and the coupling argument is the load-bearing one alone — " + + "re-read #3107 and #3174 rather than editing this assertion"); + Assert.Equal(124, alternative - ceiling); - /* The chosen line still clears the ceiling; the alternative would not. Stated as the verdicts the + /* The chosen line still clears the ceiling; the alternative would too. Stated as the verdicts the classifier produces where it can be (the ceiling against the shipped line) and as the inequality where it cannot (a line is a compile-time value and the classifier cannot be re-run against the alternative). The gap between the two lines is one guard band less one sixth of the window: - 240 - 180 = 60 s (was 240 - 210 = 30). */ + 240 - 210 = 30 s (was 240 - 180 = 60 during Q12). */ Assert.Equal( TimescaleSupport.RefreshSlotHeadroom.InsideSlot, TimescaleSupport.ClassifyRefreshSlotHeadroom(ceiling)); @@ -2867,7 +2875,7 @@ where it cannot (a line is a compile-time value and the classifier cannot be re- ceiling < TimescaleSupport.RefreshSlotWarningSeconds, $"the chosen {TimescaleSupport.RefreshSlotWarningSeconds} s line no longer sits above the " + $"{ceiling} s recorded ceiling"); - Assert.Equal(60, TimescaleSupport.RefreshSlotWarningSeconds - alternative); + Assert.Equal(30, TimescaleSupport.RefreshSlotWarningSeconds - alternative); /* THE COUPLING, as the derivations rather than as prose. The alternative is the slot less one guard band, so it moves with the light class's declared width; the chosen line tracks only the window. @@ -2897,10 +2905,10 @@ below it does not. Without this the ordering would be arithmetic with no stated /* And the lead time each line leaves, as the two figures rather than as the inequality between them — which is the same inequality asserted above and would add nothing on its own. The lower line - leaves the larger margin, so lead time argues FOR it and cannot be part of its rejection; at - #3653's window the guard band (240 s) is unchanged and the sixth is 180 s (was 210). */ + leaves the larger margin, so lead time argues FOR it and cannot be part of its rejection; at LC's + restored window the guard band (240 s) is unchanged and the sixth is 210 s (was 180 during Q12). */ Assert.Equal(240, TimescaleSupport.RefreshPhaseSlotSeconds - alternative); - Assert.Equal(180, TimescaleSupport.RefreshPhaseSlotSeconds - TimescaleSupport.RefreshSlotWarningSeconds); + Assert.Equal(210, TimescaleSupport.RefreshPhaseSlotSeconds - TimescaleSupport.RefreshSlotWarningSeconds); } /// @@ -2924,20 +2932,22 @@ public void TheRefreshSlotReading_CarriesItsOwnVerdict_AndReportsOverrunAsNegati TimescaleSupport.HeaviestHourlyRefreshView, TimescaleSupport.HeaviestHourlyRefreshObservedCeilingSeconds); - /* The recorded ceiling against #3653's re-derived 1,080 s window: 896 / 1,080 = 83.0% of it and - 1,080 - 896 = 184 s clear (was 71.1% and 364 s of the 1,260 s window), inside the routine band by - 4 s. That band assertion went RED on #3166's census against a 15-minute slot and #3174 answered it - with geometry rather than with a renumbered band — see the classifier test for the ordering it - restored; #3653 re-derived the geometry again, by the same method, and the band held. */ + /* The recorded ceiling against LC's restored 1,260 s window: 896 / 1,260 = 71.1% of it and + 1,260 - 896 = 364 s clear (was 83.0% and 184 s of Q12's 1,080 s window), inside the routine band by + 154 s. That band assertion went RED on #3166's census against a 15-minute slot and #3174 answered + it with geometry rather than with a renumbered band — see the classifier test for the ordering it + restored; the freeze re-derived the geometry back to that same shape, and the band held. */ Assert.Equal(TimescaleSupport.RefreshSlotHeadroom.InsideSlot, atTheCeiling.Headroom); - Assert.Equal(184, atTheCeiling.ClearOfSlotSeconds); - Assert.Equal(83.0, atTheCeiling.PercentOfSlot, 1); + Assert.Equal(364, atTheCeiling.ClearOfSlotSeconds); + Assert.Equal(71.1, atTheCeiling.PercentOfSlot, 1); - /* The watch line, which is where APPROACHING starts: 83.3% of the window, 1,080 / 6 = 180 s clear. */ + /* The watch line, which is where APPROACHING starts: 83.3% of the window, 1,260 / 6 = 210 s clear + (was 180 s of Q12's 1,080 s window; the percentage is the same five-sixths fraction at any window + width). */ var atTheWatchLine = new HeaviestRefreshSlotReading( TimescaleSupport.HeaviestHourlyRefreshView, TimescaleSupport.RefreshSlotWarningSeconds); Assert.Equal(TimescaleSupport.RefreshSlotHeadroom.ApproachingSlot, atTheWatchLine.Headroom); - Assert.Equal(180, atTheWatchLine.ClearOfSlotSeconds); + Assert.Equal(210, atTheWatchLine.ClearOfSlotSeconds); Assert.Equal(83.3, atTheWatchLine.PercentOfSlot, 1); /* OVERRUN, derived: a run one sixth of the window past the wall. The sixth is the same fraction the @@ -3125,9 +3135,9 @@ public void TheJobCadenceKnob_FiresInsideTheWindow_WhichIsWhyTheSlotWatchIsSepar + $"{TimescaleSupport.RefreshPhaseSlotSeconds}s window — at or past the wall, so the cadence alert " + "would arrive after #3035's precondition is already false. The knob cannot move without a rung " + "(V57), so the repair is the grid"); - /* #3653 (Q12): the re-derived 1,080 s window leaves the knob 180 s inside the wall (was 360 inside - 1,260). */ - Assert.Equal(180, TimescaleSupport.RefreshPhaseSlotSeconds - (hourlyCadenceSeconds * ShippedWarnPercent / 100)); + /* #3653's LC freeze restores the 1,260 s window, leaving the knob 360 s inside the wall (was 180 + inside Q12's 1,080 s window, before the freeze moved the legacy trio off this grid). */ + Assert.Equal(360, TimescaleSupport.RefreshPhaseSlotSeconds - (hourlyCadenceSeconds * ShippedWarnPercent / 100)); /* And where the knob's own clamp lets an operator move it to — well past the wall, with the grid's precondition broken and nothing said. */ @@ -3136,22 +3146,24 @@ public void TheJobCadenceKnob_FiresInsideTheWindow_WhichIsWhyTheSlotWatchIsSepar "the knob's clamp can no longer be raised past the window, so the argument for a separate " + "grid-derived line has changed"); - /* THE ORDERING BETWEEN THE TWO SIGNALS INVERTED at #3174, and COLLAPSED TO A COINCIDENCE at #3653 — - each asserted in its direction rather than left to be discovered. #3044's watch used to fire at - 750 s, BEFORE #2136's 900 s default; #3174's 21-minute window put it at 1,050 s, AFTER it, which - was forced: the watch line has to clear the 896 s ceiling and the knob is frozen at 900 s by V57, - and 896 < window * 50 < 900 has no integer solution. #3653 (Q12) re-derived the window to 18 - minutes by the grid's own method, and 18 * 50 = 900 EXACTLY — the one integer window whose - five-sixths line lands ON the knob. Both signals compare with >= (ClassifyRefreshSlotHeadroom and - DarlingSelfAlertEvaluator's cadence check), so on a 900 s run they fire together: the operator sees - #2136's alert and #3044's line — which carries the correct remedy where #2136's documented one - ("extend the job's schedule_interval") is wrong for this job, its schedule interval being also its - end_offset — on the SAME reading, rather than the correct remedy arriving second. Coincidence is - the honest description and it is pinned as one; a window of 17 minutes or 19 re-opens the gap in - one direction or the other, and the assertion below says which. */ - Assert.Equal( - hourlyCadenceSeconds * ShippedWarnPercent / 100, - TimescaleSupport.RefreshSlotWarningSeconds); + /* THE ORDERING BETWEEN THE TWO SIGNALS INVERTED at #3174, COLLAPSED TO A COINCIDENCE at #3653 (Q12), + and REOPENED at #3653 (LC) — each asserted in its direction rather than left to be discovered. + #3044's watch used to fire at 750 s, BEFORE #2136's 900 s default; #3174's 21-minute window put it + at 1,050 s, AFTER it, which was forced: the watch line has to clear the 896 s ceiling and the knob + is frozen at 900 s by V57, and 896 < window * 50 < 900 has no integer solution. Q12 re-derived the + window to 18 minutes when the three successors were appended behind the legacy trio, and + 18 * 50 = 900 EXACTLY — the one integer window whose five-sixths line lands ON the knob. LC's + freeze moved the legacy trio off the grid and put the successors in its three positions instead, + re-deriving the window back to 21 minutes and the watch line back to 1,050 s — AFTER the knob + again, by the same 150 s #3174 originally measured. The gap is pinned as an inequality rather than + an equality now that the coincidence is gone; a window of 18 minutes exactly would restore it, and + the assertion below says so by failing on that width. */ + Assert.True( + hourlyCadenceSeconds * ShippedWarnPercent / 100 < TimescaleSupport.RefreshSlotWarningSeconds, + $"the knob's {hourlyCadenceSeconds * ShippedWarnPercent / 100} s line no longer fires strictly " + + $"before the {TimescaleSupport.RefreshSlotWarningSeconds} s watch line, so the two signals have " + + "collapsed to a coincidence again — re-read #3174 and Q12 rather than editing this assertion"); + Assert.Equal(150, TimescaleSupport.RefreshSlotWarningSeconds - (hourlyCadenceSeconds * ShippedWarnPercent / 100)); Assert.True( hourlyCadenceSeconds * ShippedWarnPercent / 100 <= TimescaleSupport.RefreshSlotWarningSeconds && TimescaleSupport.RefreshSlotWarningSeconds < TimescaleSupport.RefreshPhaseSlotSeconds, @@ -3540,46 +3552,44 @@ private static TimeSpan ParsePostgresInterval(string interval) /// resolves to on today's list — the one place a reader can see the grid without re-deriving it, and /// the one place the interleaving is visible as minutes rather than as a rule. /// - /// RE-DERIVED at #3653 (Q12), and the derivation is in the table. Three interval-honest hourly - /// successors were APPENDED to (the legacy trio stays: the - /// daily tier is hierarchical from it — ). Fifteen - /// light members instead of twelve, so the band holds positions 0-14. query_stats_interval_hourly - /// groups by statement, so it is the FOURTH unbounded member and takes position 4 * 3 = 12; the other two - /// are deployment-bounded and are dealt in registry order with the rest of that class, over the positions - /// the unbounded set leaves (1, 2, 3, 5, 6, 7, 9, 10, 11, 13, 14). They sit in HourlyAggregates, - /// which precedes BaselineAggregates in the order, so they take 3 and 5 and the seven baseline - /// members each move two positions later (3→6, 5→7, 6→9, 7→10, 9→11, 10→13, 11→14) — nine sub-3 s - /// policies re-phased ONCE by the converge on the first start that carries this build, which is what the - /// converge exists for. The four unbounded members and the two bounded hourlies ahead of the successors - /// keep their minutes; the heaviest moves 15→18 with the band it follows. Appending at the END of - /// HourlyAggregates rather than beside each legacy is what keeps the first six where they were; - /// keeping the baselines where they were as well would have needed a fourth registry list appended after - /// them, and "there is no fourth list to forget" is the invariant the three lists carry. + /// RE-DERIVED at #3653 (Q12) by appending three successors, and RE-DERIVED AGAIN at #3653 (LC) by + /// freezing their legacies off the grid — the derivation is in the table. The freeze moved the legacy + /// trio (query_stats_hourly, procedure_stats_hourly, query_stats_db_hourly) out of + /// into + /// — the daily tier no longer needs them on the grid to stay hierarchical (LB's successor dailies cover the + /// window above the freeze floor) — while the successors took the three positions the legacy trio held, + /// rather than staying appended behind it. Thirteen hourly policies again, twelve of them light, so the band + /// holds positions 0-11. query_store_stats_hourly and query_store_stats_corrected_hourly are + /// the first two unbounded members (positions 0 and 4); query_stats_interval_hourly groups by + /// statement like the legacy it replaces, so it is the THIRD unbounded member and takes position 2 * 4 = 8. + /// The other two successors are deployment-bounded like their legacies and are dealt in registry order with + /// the rest of that class, over the positions the unbounded set leaves (1, 2, 3, 5, 6, 7, 9, 10, 11): + /// procedure_stats_interval_hourly takes 1, query_stats_db_interval_hourly takes 2, and the + /// seven baseline members fill 3, 5, 6, 7, 9, 10, 11 in order — nine sub-3 s policies re-phased ONCE by the + /// converge on the first start that carries this build, which is what the converge exists for. The heaviest + /// moves back 18→15 with the band it follows. /// [Fact] public void TheRefreshGridIsUnchanged_AndTheCompressionGridsOneInputFromItIsPinned() { var expected = new (string View, int Minute)[] { - (TimescaleSupport.QueryStatsHourlyView, 0), - (TimescaleSupport.ProcedureStatsHourlyView, 1), - (TimescaleSupport.QueryStoreStatsHourlyView, 4), - (TimescaleSupport.QueryStatsDbHourlyView, 2), - (TimescaleSupport.QueryStoreStatsIntervalHourlyView, 18), - (TimescaleSupport.QueryStoreStatsCorrectedHourlyView, 8), - /* #3653: the three successors, appended — one unbounded (position 12, the fourth every-fourth - slot), two bounded (3 and 5, the next free positions in the bounded deal, which pushes the - seven baselines two positions later). */ - (TimescaleSupport.QueryStatsIntervalHourlyView, 12), - (TimescaleSupport.ProcedureStatsIntervalHourlyView, 3), - (TimescaleSupport.QueryStatsDbIntervalHourlyView, 5), - (TimescaleSupport.PerfmonIntervalBaselineView, 6), - (TimescaleSupport.WaitStatsIntervalBaselineView, 7), - (TimescaleSupport.SessionStatsBaselineView, 9), - (TimescaleSupport.QueryStatsBaselineView, 10), - (TimescaleSupport.BlockedProcessBaselineView, 11), - (TimescaleSupport.DeadlockBaselineView, 13), - (TimescaleSupport.MemoryBaselineView, 14), + /* #3653 (LC): the Query Store family leads (unchanged since before Q12), then the three + successors in the three positions their legacies held — not appended behind the family, which + is what keeps the grid at thirteen policies instead of sixteen. */ + (TimescaleSupport.QueryStoreStatsHourlyView, 0), + (TimescaleSupport.QueryStoreStatsIntervalHourlyView, 15), + (TimescaleSupport.QueryStoreStatsCorrectedHourlyView, 4), + (TimescaleSupport.QueryStatsIntervalHourlyView, 8), + (TimescaleSupport.ProcedureStatsIntervalHourlyView, 1), + (TimescaleSupport.QueryStatsDbIntervalHourlyView, 2), + (TimescaleSupport.PerfmonIntervalBaselineView, 3), + (TimescaleSupport.WaitStatsIntervalBaselineView, 5), + (TimescaleSupport.SessionStatsBaselineView, 6), + (TimescaleSupport.QueryStatsBaselineView, 7), + (TimescaleSupport.BlockedProcessBaselineView, 9), + (TimescaleSupport.DeadlockBaselineView, 10), + (TimescaleSupport.MemoryBaselineView, 11), }; Assert.Equal(expected.Length, TimescaleSupport.HourlyRefreshPhaseOrder.Count); @@ -3609,8 +3619,10 @@ needed a uniform step to exist. */ Assert.Throws( () => TimescaleSupport.RefreshPhaseMinutesFor(TimescaleSupport.QueryStatsDailyView)); - /* And the refresh STATEMENTS are untouched: still the refresh phase, still no compression in them. */ - var refreshSql = TimescaleSupport.AddHourlyRefreshPolicySql(TimescaleSupport.ProcedureStatsHourlyView); + /* And the refresh STATEMENTS are untouched: still the refresh phase, still no compression in them. + ProcedureStatsHourlyView is frozen off this grid since LC (FrozenRollupAggregates); its successor + holds the position it used to and is what a live policy is built for now. */ + var refreshSql = TimescaleSupport.AddHourlyRefreshPolicySql(TimescaleSupport.ProcedureStatsIntervalHourlyView); Assert.Contains("INTERVAL '1 minutes'", refreshSql, StringComparison.Ordinal); Assert.Contains("add_continuous_aggregate_policy", refreshSql, StringComparison.Ordinal); Assert.DoesNotContain("compression", refreshSql, StringComparison.Ordinal); From 8b632d0bfc0c6ba36800e364077c61f9fc6172ee Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:34:51 -0400 Subject: [PATCH 05/18] #3653 A6 lane LC-a4: daily summary no longer throws on a frozen legacy rollup, repair-list pins re-derived, raw gate covers query_stats_db_interval_hourly DailySummarySql.QueriesCteForCagg/QueriesCteForStitchedCagg looked their relation up in TimescaleSupport.MaterializationHoleTargets, which the freeze (LC) emptied of the six frozen legacy rollups. Every daily-summary read that named a frozen view (legacy-only, RollupCoverage.Unknown, or a stitched splice below the successor floor) threw ArgumentException. Added TimescaleSupport.RollupCoverageProbeTargets: the same per-relation descriptor as MaterializationHoleTargets, but over every RollupViews member including the frozen six, with CREATE text sourced from HourlyAggregates, DailyAggregates or FrozenRollupAggregates. Pointed the three DailySummarySql lookups at it. MaterializationHoleTargets itself is untouched, so the repair walk still never sees a frozen view. Re-pinned MaterializationHoleRepairTests' two lookups that named QueryStatsDailyView directly (now frozen) to its live successor via SuccessorDailyOf, re-derived the stale registered-count literal (26 -> 20) and the daily-after-hourly ordering check (now over the live successor pair, not the frozen legacy pair), and added a pin that MaterializationHoleTargets holds no frozen view. Re-pinned the two MaterializationHoleRepairLiveTests that built their whole scenario on the now-frozen query_stats_hourly to use its live successor instead; the one scenario that compared a live unfiltered rollup against a live filtered one no longer has two live sides after the freeze, so it now proves only the half that is still live: a restart-only hour is neither materialized nor reported as a hole. RawTierCoverage gated the raw query_stats purge on query_stats_interval_hourly only, but query_stats_db_interval_hourly also reads raw collect.query_stats directly. Added it through RequireSuccessorOf(QueryStatsDbHourlyView), fixed the pin that asserted it stayed out, and added a derived pin that walks every RawTierCoverage row and checks its coverage set against every non-frozen RollupViews entry sourced from that raw table directly. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../DailySummaryNotCarriedTests.cs | 13 ++-- .../DailySummaryStitchedRangeTests.cs | 45 +++++++++++ .../IntervalHonestHourlyRollupTests.cs | 44 +++++++++-- .../MaterializationHoleRepairTests.cs | 76 ++++++++++++------- .../DailySummarySql.cs | 18 +++-- .../TimescaleSupport.MaterializationHoles.cs | 25 ++++++ .../TimescaleSupport.cs | 8 +- 7 files changed, 181 insertions(+), 48 deletions(-) diff --git a/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs b/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs index c1e3c517cf..988b61c1ec 100644 --- a/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs +++ b/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs @@ -38,8 +38,9 @@ namespace Darling.Tests; /// The witness is the hole scan's definition, reused, not a second one. The scan calls a bucket a /// hole when the materialization holds no row for it AND the source holds an admitted row in it. The routed /// queries CTE's third member asks the same two questions per server at day grain, with the source, its -/// time column and its filter read off — the repair's -/// own target list. So a server that genuinely ran nothing that day (no source row) is NOT named: the +/// time column and its filter read off — the repair's +/// own target list, extended (#3653 LC) so a frozen legacy rollup still has one. So a server that genuinely +/// ran nothing that day (no source row) is NOT named: the /// disclosure cannot claim a hole where the raw table was simply empty. The live test below plants exactly /// that control beside the hole. /// @@ -88,7 +89,9 @@ public void EveryRoutedForm_AsksTheHoleScansTwoProbes_PerServer_AtDayGrain() foreach (var (tier, relation) in RoutedForms) { var sql = DailySummarySql.RangeSqlFor(tier, relation); - var target = TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == relation); + /* #3653 LC froze query_stats_hourly/_daily out of MaterializationHoleTargets; RollupCoverageProbeTargets + still knows every relation the daily summary can route to. */ + var target = TimescaleSupport.RollupCoverageProbeTargets.Single(t => t.View == relation); var filter = TimescaleSupport.MaterializationHoleSourceFilterFor(target.CreateSql); Assert.Contains("SELECT b.d, NULL::bigint AS c", sql, StringComparison.Ordinal); @@ -138,7 +141,7 @@ the registry moves this test rather than silently moving the probe. */ private static (string Source, string TimeColumn, string Filter) SourceOf(string view) { - var target = TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == view); + var target = TimescaleSupport.RollupCoverageProbeTargets.Single(t => t.View == view); return (target.Source, target.SourceTimeColumn, TimescaleSupport.MaterializationHoleSourceFilterFor(target.CreateSql)); } @@ -174,7 +177,7 @@ public void ABucketWithZeroDistinctHashes_CannotExist_SoNullNeedsNoFlag() public void RangeSqlFor_RefusesAnUnregisteredRelation() { var ex = Assert.Throws(() => DailySummarySql.RangeSqlFor(RetentionTier.Hourly, "some_future_rollup")); - Assert.Contains("MaterializationHoleTargets", ex.Message, StringComparison.Ordinal); + Assert.Contains("RollupCoverageProbeTargets", ex.Message, StringComparison.Ordinal); Assert.Throws(() => DailySummarySql.RangeSqlFor(RetentionTier.Hourly, " ")); } diff --git a/Darling/Darling.Tests/DailySummaryStitchedRangeTests.cs b/Darling/Darling.Tests/DailySummaryStitchedRangeTests.cs index 0ebb930cd6..586b710cad 100644 --- a/Darling/Darling.Tests/DailySummaryStitchedRangeTests.cs +++ b/Darling/Darling.Tests/DailySummaryStitchedRangeTests.cs @@ -231,4 +231,49 @@ public void RawTier_IgnoresTheStitch_ReadsExactlyAsTheTwoArgumentFormDoes() Assert.Equal(plain, stitched); } + + /// + /// #3653 A6, lane LC-a4: the frozen legacy name this overload ITSELF ever names — on + /// the Hourly tier, on the Daily tier (both hardcoded in the two-argument form + /// this overload falls back to; see its remarks) — must still build, both under + /// (nothing measured, so the legacy alone is named) and under a + /// stitched coverage whose successor floor falls INSIDE the window (the genuinely-stitched splice, which + /// names the legacy AND the successor). Both threw on 1abc48ba: + /// QueriesCteForCagg/QueriesCteForStitchedCagg read a frozen view's source off + /// MaterializationHoleTargets, which the freeze (#3653 LC) had just emptied of every frozen view. + /// + [Fact] + public void FrozenLegacy_StillBuilds_UnderUnknownCoverage_AndUnderAStitchThatCrossesTheWindow() + { + var windowStart = DaysAgo(10); + + var hourlyUnknown = DailySummarySql.RangeSqlFor(RetentionTier.Hourly, RollupCoverage.Unknown, windowStart); + Assert.Contains($"FROM collect.{Legacy}", hourlyUnknown, StringComparison.Ordinal); + + var dailyUnknown = DailySummarySql.RangeSqlFor(RetentionTier.Daily, RollupCoverage.Unknown, windowStart); + Assert.Contains($"FROM collect.{DailyLegacy}", dailyUnknown, StringComparison.Ordinal); + + var hourlyStitch = new RollupCoverage( + new Dictionary(StringComparer.Ordinal) { [Legacy] = DaysAgo(80), [Successor] = DaysAgo(5) }, + new Dictionary(StringComparer.Ordinal), + RollupAvailability.All); + var hourlyStitched = DailySummarySql.RangeSqlFor(RetentionTier.Hourly, hourlyStitch, windowStart); + Assert.Contains($"FROM collect.{Legacy}", hourlyStitched, StringComparison.Ordinal); + Assert.Contains($"FROM collect.{Successor}", hourlyStitched, StringComparison.Ordinal); + + var dailyStitch = new RollupCoverage( + new Dictionary(StringComparer.Ordinal) + { + [DailyLegacy] = DaysAgo(200), + [DailySuccessor] = DaysAgo(5), + /* The daily stitch's own boundary is read off the successor HOURLY's floor (ceiling-of-day), + not the successor daily's floor directly — StitchedRelationSql's Daily branch. */ + [TimescaleSupport.QueryStatsIntervalHourlyView] = DaysAgo(90), + }, + new Dictionary(StringComparer.Ordinal), + RollupAvailability.All); + var dailyStitched = DailySummarySql.RangeSqlFor(RetentionTier.Daily, dailyStitch, windowStart); + Assert.Contains($"FROM collect.{DailyLegacy}", dailyStitched, StringComparison.Ordinal); + Assert.Contains($"FROM collect.{DailySuccessor}", dailyStitched, StringComparison.Ordinal); + } } diff --git a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs index f0904206cb..545115d315 100644 --- a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs +++ b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs @@ -141,18 +141,22 @@ read through SuccessorDailyOf rather than restated. */ Assert.Equal(new[] { successorDaily }, policy.Coverage); } - /* The raw purge moved onto the two successors with a raw table of their own naming them (#3653 LC). - query_stats_db_hourly has no RawTierCoverage row of its own — it shares "query_stats" with - query_stats_hourly, already that row's named consumer below — so it was never a raw-gate candidate. */ + /* The raw purge moved onto the successors with a raw table naming them (#3653 LC). query_stats_db_hourly + has no RawTierCoverage row of its own — it shares "query_stats" with query_stats_hourly — but its + OWN successor, query_stats_db_interval_hourly, reads raw collect.query_stats directly + (CreateQueryStatsDbIntervalHourlySql) and so IS a consumer named in query_stats's row (#3653 A6 lane + LC-a4): the raw gate must see it or a purge could destroy history it never materialized. */ var querySuccessor = TimescaleSupport.SuccessorOf(TimescaleSupport.QueryStatsHourlyView) ?? throw new InvalidOperationException($"{TimescaleSupport.QueryStatsHourlyView} must be in {nameof(TimescaleSupport.SupersededHourlyRollups)}."); var procedureSuccessor = TimescaleSupport.SuccessorOf(TimescaleSupport.ProcedureStatsHourlyView) ?? throw new InvalidOperationException($"{TimescaleSupport.ProcedureStatsHourlyView} must be in {nameof(TimescaleSupport.SupersededHourlyRollups)}."); var dbSuccessor = TimescaleSupport.SuccessorOf(TimescaleSupport.QueryStatsDbHourlyView) ?? throw new InvalidOperationException($"{TimescaleSupport.QueryStatsDbHourlyView} must be in {nameof(TimescaleSupport.SupersededHourlyRollups)}."); - Assert.Equal(new[] { querySuccessor }, TimescaleSupport.RawTierCoverage.Single(t => t.Relation == "query_stats").Coverage); + Assert.Equal(new[] { querySuccessor, dbSuccessor }, TimescaleSupport.RawTierCoverage.Single(t => t.Relation == "query_stats").Coverage); Assert.Equal(new[] { procedureSuccessor }, TimescaleSupport.RawTierCoverage.Single(t => t.Relation == "procedure_stats").Coverage); - Assert.All(TimescaleSupport.RawTierCoverage, tier => Assert.DoesNotContain(dbSuccessor, tier.Coverage)); + /* dbSuccessor belongs in "query_stats"'s own row (asserted above) and nowhere else — it has no + RawTierCoverage row of its own to gate. */ + Assert.All(TimescaleSupport.RawTierCoverage.Where(t => t.Relation != "query_stats"), tier => Assert.DoesNotContain(dbSuccessor, tier.Coverage)); Assert.False(RollupAvailability.WithoutIntervalHourlies.AllPresent); Assert.True(RollupAvailability.All.AllPresent); @@ -161,6 +165,30 @@ query_stats_db_hourly has no RawTierCoverage row of its own — it shares "query Assert.Equal(19, TimescaleSupport.RollupViews.Length); } + /// + /// #3653 A6 lane LC-a4: 's own rule, derived rather than + /// trusted row by row — each row's Coverage is exactly the View of every non-frozen + /// entry whose Source is that row's raw Relation (a rollup + /// reading the raw table directly, not a hierarchical one two hops down). A raw table with a consumer this + /// derivation finds and the row's own Coverage does not is exactly the #1784 defect the raw purge gate + /// exists to prevent. + /// + [Fact] + public void RawTierCoverage_IsExactlyEveryNonFrozenRollupThatReadsThatRawTableDirectly() + { + foreach (var row in TimescaleSupport.RawTierCoverage) + { + var expectedConsumers = TimescaleSupport.RollupViews + .Where(r => string.Equals(r.Source, row.Relation, StringComparison.Ordinal)) + .Where(r => !TimescaleSupport.IsFrozenRollupAggregate(r.View)) + .Select(r => r.View) + .OrderBy(v => v, StringComparer.Ordinal) + .ToArray(); + + Assert.Equal(expectedConsumers, row.Coverage.OrderBy(v => v, StringComparer.Ordinal).ToArray()); + } + } + /// /// The successor is the legacy's shape plus the verdict: same FROM, same GROUP BY, every legacy select item /// under the same alias, the interval predicate in the WHERE, and sum(sample_interval_seconds) @@ -173,8 +201,10 @@ public void EachSuccessor_IsItsLegacysShape_PlusTheIntervalVerdict_AndTheCarried { foreach (var (legacy, successor, _) in TimescaleSupport.SupersededHourlyRollups) { - var legacyText = TimescaleSupport.HourlyAggregates.Single(a => a.View == legacy).CreateSql; - var successorText = TimescaleSupport.HourlyAggregates.Single(a => a.View == successor).CreateSql; + /* #3653 LC froze the legacy member out of HourlyAggregates, so its CREATE is read off + RollupCoverageProbeTargets instead (it still knows the frozen six; see that member's doc). */ + var legacyText = TimescaleSupport.RollupCoverageProbeTargets.Single(t => t.View == legacy).CreateSql; + var successorText = TimescaleSupport.RollupCoverageProbeTargets.Single(t => t.View == successor).CreateSql; Assert.DoesNotContain("sample_interval_seconds IS DISTINCT FROM 0", legacyText, StringComparison.Ordinal); Assert.Contains("sample_interval_seconds IS DISTINCT FROM 0", successorText, StringComparison.Ordinal); diff --git a/Darling/Darling.Tests/MaterializationHoleRepairTests.cs b/Darling/Darling.Tests/MaterializationHoleRepairTests.cs index 4804e9d5fa..c348022af9 100644 --- a/Darling/Darling.Tests/MaterializationHoleRepairTests.cs +++ b/Darling/Darling.Tests/MaterializationHoleRepairTests.cs @@ -40,7 +40,9 @@ public void Targets_AreEveryRegisteredAggregate_InDependencyOrder_WithTheirOwnCr var registered = TimescaleSupport.HourlyAggregates.Concat(TimescaleSupport.DailyAggregates).Concat(TimescaleSupport.BaselineAggregates).ToArray(); Assert.Equal(registered.Length, targets.Count); - Assert.Equal(26, targets.Count); // #3653 A6 lane LB: +3 for the interval-honest successor dailies added to DailyAggregates + // #3653 LC: 26 (A6 lane LB's count) minus the six the freeze took out of HourlyAggregates/DailyAggregates + // (they stay in TimescaleSupport.RollupViews and FrozenRollupAggregates, but nothing repairs them now). + Assert.Equal(20, targets.Count); Assert.Equal(registered.Select(a => a.View).OrderBy(v => v, StringComparer.Ordinal), targets.Select(t => t.View).OrderBy(v => v, StringComparer.Ordinal)); /* The rollups come first in the backfill's dependency order — every raw-sourced rollup before the @@ -55,15 +57,31 @@ public void Targets_AreEveryRegisteredAggregate_InDependencyOrder_WithTheirOwnCr Assert.Equal(target.Source.EndsWith("_hourly", StringComparison.Ordinal) || target.Source.EndsWith("_daily", StringComparison.Ordinal) ? "bucket" : "collection_time", target.SourceTimeColumn); } - /* A daily is scanned after the hourly it reads, so its scan sees the rows the hourly's repair wrote. */ - foreach (var (legacy, _, dependentDaily) in TimescaleSupport.SupersededHourlyRollups) + /* A daily is scanned after the hourly it reads, so its scan sees the rows the hourly's repair wrote. + #3653 LC: SupersededHourlyRollups' own Legacy/DependentDaily pair is frozen out of `targets` entirely + now (there is nothing left to order), so this re-pins the live analog — SupersededDailyRollups' own + SuccessorHourly/SuccessorDaily pair, derived rather than typed, the same two relations + RollupBackfill.Targets still orders by depth. */ + foreach (var (_, successorDaily, successorHourly) in TimescaleSupport.SupersededDailyRollups) { Assert.True( - targets.ToList().FindIndex(t => t.View == legacy) < targets.ToList().FindIndex(t => t.View == dependentDaily), - $"{dependentDaily} must be scanned after {legacy}"); + targets.ToList().FindIndex(t => t.View == successorHourly) < targets.ToList().FindIndex(t => t.View == successorDaily), + $"{successorDaily} must be scanned after {successorHourly}"); } } + /// #3653 LC: the repair walk must never see a frozen view — refreshing one is the one thing the + /// freeze forbids (see 's remarks). A regression that + /// re-adds one of the six to (and so to + /// ) fails here rather than at a live refresh. + [Fact] + public void MaterializationHoleTargets_HoldsNoFrozenRollupAggregate() + { + Assert.DoesNotContain( + TimescaleSupport.MaterializationHoleTargets, + target => TimescaleSupport.IsFrozenRollupAggregate(target.View)); + } + [Fact] public void TheCap_IsOnePolicyWindowInBuckets_PerGrain() { @@ -119,10 +137,16 @@ public void TheSourceFilter_IsTheAggregatesOwnWhere_Verbatim_OrEmpty() Assert.Contains("AND sample_interval_seconds IS DISTINCT FROM 0\n OFFSET 0)", sql, StringComparison.Ordinal); Assert.EndsWith("ORDER BY c.bucket", sql.TrimEnd(), StringComparison.Ordinal); + /* #3653 LC froze query_stats_daily out of MaterializationHoleTargets; its unfiltered, hierarchical shape + lives on in its live successor daily (SuccessorDailyOf, derived rather than typed), which is + hierarchical from query_stats_interval_hourly with no WHERE of its own either. */ + var successorDaily = TimescaleSupport.SuccessorDailyOf(TimescaleSupport.QueryStatsIntervalHourlyView) + ?? throw new InvalidOperationException( + $"{TimescaleSupport.QueryStatsIntervalHourlyView} must be in {nameof(TimescaleSupport.SupersededDailyRollups)}."); var unfiltered = TimescaleSupport.MaterializationHoleScanSql( - TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == TimescaleSupport.QueryStatsDailyView), ("s", "m")) + TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == successorDaily), ("s", "m")) .Replace("\r\n", "\n", StringComparison.Ordinal); - Assert.Contains("FROM collect.query_stats_hourly AS s", unfiltered, StringComparison.Ordinal); + Assert.Contains("FROM collect.query_stats_interval_hourly AS s", unfiltered, StringComparison.Ordinal); Assert.Contains("s.bucket >= c.bucket", unfiltered, StringComparison.Ordinal); /* No filter: the source probe's fence follows straight after the width bound. */ Assert.Contains("s.bucket < c.bucket + $3::interval\n OFFSET 0)", unfiltered, StringComparison.Ordinal); @@ -390,7 +414,11 @@ public async Task PreOutageTail_ReadsEmptyAfterResume_AndOneTargetedForcedRefres await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); - var view = TimescaleSupport.QueryStatsHourlyView; + /* #3653 LC: query_stats_hourly itself is now frozen — the repair walk never touches it, so it cannot + carry this proof any more. query_stats_interval_hourly is its live successor, raw-sourced from the + same collect.query_stats this test's own inserts target, so every plain (non-restart) row below is + admitted the same way the legacy used to admit it. */ + var view = TimescaleSupport.QueryStatsIntervalHourlyView; var materialization = await TimescaleSupport.ResolveMaterializationAsync(connection, view, ct); Assert.NotNull(materialization); @@ -518,15 +546,13 @@ carries the cost of proving it. A pass that fell silent here would read the same Assert.Empty(await ScanAsync(connection, target, materialization.Value, H(0), H(10), ct)); Assert.Equal(new[] { H(0), H(1), H(2), H(3), H(4), H(9), H(10) }, await MaterializedBucketsAsync(connection, view, ct)); - /* THE CONTROL on the source filter: the interval-honest successor sees the same tail (its WHERE admits - the planted rows), but an hour holding ONLY a restart row is not a hole for it — the scan applies the - aggregate's own filter — while the legacy, which admits the row, would materialize it. Planted at - H(12), refreshed on the legacy so its span reaches past it, and the successor's span made to reach - past it too by refreshing H(9)-H(11) there. */ - var successor = TimescaleSupport.QueryStatsIntervalHourlyView; - var successorMaterialization = await TimescaleSupport.ResolveMaterializationAsync(connection, successor, ct); - Assert.NotNull(successorMaterialization); - await RefreshAsync(connection, successor, H(0), H(3), ct); + /* THE CONTROL on the source filter (#3653 LC): pre-freeze this planted a restart-only row and + contrasted the legacy (admits it) against the successor (excludes it, scan and all) — two live + relations reading the same collect.query_stats. The legacy no longer repairs at all, so only the + successor's own half still runs live (MaterializationHoleScanShapeTests/MaterializationHoleRepairTests' + pure pins still hold the CreateSql contrast). What remains provable here: an hour holding ONLY a + restart row is neither materialized NOR reported as a hole — the scan applies view's own filter, so + an uncovered hour the filter would reject is not a false positive. */ await using (var restartOnly = new NpgsqlCommand(@" INSERT INTO collect.query_stats (collection_id, collection_time, server_id, server_name, database_name, query_hash, sql_handle, @@ -540,15 +566,11 @@ INSERT INTO collect.query_stats } await InsertHoursAsync(connection, new[] { 14 }, H, ct); - await RefreshAsync(connection, successor, H(14), H(15), ct); - - var successorTarget = TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == successor); - var successorHoles = await ScanAsync(connection, successorTarget, successorMaterialization.Value, H(0), H(14), ct); - Assert.DoesNotContain(H(12), successorHoles); - Assert.Equal(new[] { H(3), H(4), H(9), H(10) }, successorHoles); + await RefreshAsync(connection, view, H(14), H(15), ct); - await RefreshAsync(connection, view, H(12), H(13), ct); - Assert.Contains(H(12), await MaterializedBucketsAsync(connection, view, ct)); + var restartHoles = await ScanAsync(connection, target, materialization.Value, H(0), H(14), ct); + Assert.DoesNotContain(H(12), restartHoles); + Assert.DoesNotContain(H(12), await MaterializedBucketsAsync(connection, view, ct)); } /// @@ -583,7 +605,9 @@ public async Task OneHoleRepairedAndOneDeferred_TheTallyCarriesFoundRepairedAndD await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); - var view = TimescaleSupport.QueryStatsHourlyView; + /* #3653 LC: query_stats_hourly is frozen out of the repair walk; query_stats_interval_hourly is its live + successor and, like the legacy, admits every plain (non-restart) row this test plants. */ + var view = TimescaleSupport.QueryStatsIntervalHourlyView; var cap = TimescaleSupport.MaterializationHoleRepairCapBuckets(TimescaleSupport.HourlyBucket); Assert.Equal(24, cap); diff --git a/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs b/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs index f2c3f3dfc8..9961683689 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/DailySummarySql.cs @@ -286,8 +286,10 @@ GROUP BY x.d /// questions of every day at or below the ceiling, with server_id = $1 on both probes (this is one /// server's calendar) and the day as the bucket: no rollup row for this server that day, and a source row /// for this server that day the aggregate would have produced output from. The source, its time column - /// and its filter are read off — the repair's own - /// target list — so the daily tier probes query_stats_hourly (its hierarchical source, 90 days), + /// and its filter are read off — the repair's own + /// target list, plus the six the freeze (#3653 LC) took off it, since a frozen legacy rollup is still a + /// valid relation for this probe to name — so the daily tier probes query_stats_hourly (its + /// hierarchical source, 90 days), /// the hourly tier probes raw query_stats (4 days), and the interval-honest successor's probe /// carries sample_interval_seconds IS DISTINCT FROM 0 () /// so a day holding only restart rows is not called a hole. The second probe is what keeps the disclosure @@ -337,12 +339,12 @@ private static string QueriesCteForCagg(string relation) the CREATE its filter is read from. Looked up rather than restated so the calendar's idea of "what the rollup reads" cannot drift from the scan's. A relation the registry does not know is refused here, before it reaches the store as a 42P01. */ - var target = TimescaleSupport.MaterializationHoleTargets + var target = TimescaleSupport.RollupCoverageProbeTargets .FirstOrDefault(t => string.Equals(t.View, relation, StringComparison.Ordinal)); if (target.View is null) { throw new ArgumentException( - $"'{relation}' is not a registered continuous aggregate (TimescaleSupport.MaterializationHoleTargets), so the daily summary cannot name its source or the days it did not carry (#3653 A6).", + $"'{relation}' is not a registered continuous aggregate (TimescaleSupport.RollupCoverageProbeTargets), so the daily summary cannot name its source or the days it did not carry (#3653 A6).", nameof(relation)); } @@ -430,21 +432,21 @@ SELECT 1 FROM collect.{target.Source} AS s /// private static string QueriesCteForStitchedCagg(string legacy, string successor, DateTime successorFloor) { - var legacyTarget = TimescaleSupport.MaterializationHoleTargets + var legacyTarget = TimescaleSupport.RollupCoverageProbeTargets .FirstOrDefault(t => string.Equals(t.View, legacy, StringComparison.Ordinal)); if (legacyTarget.View is null) { throw new ArgumentException( - $"'{legacy}' is not a registered continuous aggregate (TimescaleSupport.MaterializationHoleTargets), so the daily summary cannot name its source or the days it did not carry (#3653 A6).", + $"'{legacy}' is not a registered continuous aggregate (TimescaleSupport.RollupCoverageProbeTargets), so the daily summary cannot name its source or the days it did not carry (#3653 A6).", nameof(legacy)); } - var successorTarget = TimescaleSupport.MaterializationHoleTargets + var successorTarget = TimescaleSupport.RollupCoverageProbeTargets .FirstOrDefault(t => string.Equals(t.View, successor, StringComparison.Ordinal)); if (successorTarget.View is null) { throw new ArgumentException( - $"'{successor}' is not a registered continuous aggregate (TimescaleSupport.MaterializationHoleTargets), so the daily summary cannot name its source or the days it did not carry (#3653 A6).", + $"'{successor}' is not a registered continuous aggregate (TimescaleSupport.RollupCoverageProbeTargets), so the daily summary cannot name its source or the days it did not carry (#3653 A6).", nameof(successor)); } diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs index f30569a6e3..6d1c987327 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs @@ -123,6 +123,31 @@ public readonly record struct MaterializationHoleTarget( a.View, SourceTableFor(a.View), "collection_time", HourlyBucket, a.CreateSql))) .ToArray(); + /// + /// #3653 A6, lane LC-a4: the same per-relation descriptor as , but + /// over EVERY member of — including the six the freeze (LC) took out of + /// , and so out of too. The + /// daily summary's not-carried probe (DailySummarySql.QueriesCteForCagg and + /// QueriesCteForStitchedCagg) still names a frozen legacy rollup long after LC stops it advancing — + /// under RollupCoverage.Unknown, and below a successor's stitch floor — so it needs a lookup that + /// still knows one, while the repair walk () must never see a + /// frozen view among ITS targets: refreshing one is the one thing the freeze forbids. Kept as a SEPARATE + /// list rather than folded into so that list's membership and + /// dependency order — a repair-walk invariant — stay exactly as the freeze left them. Its CREATE text comes + /// from , OR + /// — together the three hold exactly one entry per member, frozen or not, so the + /// lookup below cannot go ambiguous or come up empty for any relation this file knows by name. + /// + public static IReadOnlyList RollupCoverageProbeTargets => + RollupViews + .Select(r => new MaterializationHoleTarget( + r.View, r.Source, r.SourceTimeColumn, r.BucketWidth, + HourlyAggregates.Concat(DailyAggregates).Concat(FrozenRollupAggregates) + .Single(a => string.Equals(a.View, r.View, StringComparison.Ordinal)).CreateSql)) + .Concat(BaselineAggregates.Select(a => new MaterializationHoleTarget( + a.View, SourceTableFor(a.View), "collection_time", HourlyBucket, a.CreateSql))) + .ToArray(); + /// /// How many buckets one aggregate may have repaired per start: its own refresh policy's window in buckets — /// over (24) for an hourly aggregate, diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 681df51af7..e5e7cce92b 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -6031,8 +6031,12 @@ public static string RetentionArmSafetySql(string relation, string sourceTimeCol keyed on query_stats_hourly/procedure_stats_hourly would eventually find them EMPTY — their OWN retention policy (RetentionPolicies, unchanged by LC) keeps trimming their chunks at HourlyRetentionInterval while the freeze stops anything from refilling them past that point — and - hold the raw purge forever with no self-release. See SupersededHourlyRollups for the full story. */ - ("query_stats", "collection_time", new[] { RequireSuccessorOf(QueryStatsHourlyView) }), + hold the raw purge forever with no self-release. See SupersededHourlyRollups for the full story. + query_stats has TWO consumers, same #1849 reason query_store_stats does below: the query-grain + successor AND query_stats_db_interval_hourly (CreateQueryStatsDbIntervalHourlySql), the db-grain + successor, both read raw collect.query_stats directly, so raw cannot purge over history either is + missing. */ + ("query_stats", "collection_time", new[] { RequireSuccessorOf(QueryStatsHourlyView), RequireSuccessorOf(QueryStatsDbHourlyView) }), ("procedure_stats", "collection_time", new[] { RequireSuccessorOf(ProcedureStatsHourlyView) }), /* query_store_stats is not one of #3653's legacy six (see FrozenRollupAggregates) — both consumers below go on refreshing, so their coverage is still named directly. */ From 74c2555bbbfcd9bbd5cd6185fd2ad7c397d8b618 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:38:25 -0400 Subject: [PATCH 06/18] #3653 A6 lane LC-a4: re-pin two more live tests that named a frozen rollup by its lookup MaterializationHoleScanShapeLiveTests and DailySummaryReadShapeLiveTests both looked a frozen legacy view (query_stats_hourly/_daily) up in MaterializationHoleTargets directly, the same crash pattern as the daily summary's own lookup. Re-pinned the scan-shape oracle to the live successor (SeedAsync already refreshes it over the same windows) and dropped the daily-tier case, whose only live target is now frozen; re-pinned the read-shape oracle to RollupCoverageProbeTargets, mirroring the product's own fix. Compiles clean; not yet run against a live rig in this session. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../Darling.Tests/DailySummaryReadShapeTests.cs | 5 ++++- .../MaterializationHoleScanShapeTests.cs | 14 ++++++++------ 2 files changed, 12 insertions(+), 7 deletions(-) diff --git a/Darling/Darling.Tests/DailySummaryReadShapeTests.cs b/Darling/Darling.Tests/DailySummaryReadShapeTests.cs index b09c75dd22..9503eff9b0 100644 --- a/Darling/Darling.Tests/DailySummaryReadShapeTests.cs +++ b/Darling/Darling.Tests/DailySummaryReadShapeTests.cs @@ -323,7 +323,10 @@ GROUP BY 1 """; } - var target = TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == relation); + /* #3653 LC froze query_stats_hourly/_daily out of MaterializationHoleTargets; this oracle mirrors + DailySummarySql's own lookup (RollupCoverageProbeTargets), which still knows every relation the + product can route to. */ + var target = TimescaleSupport.RollupCoverageProbeTargets.Single(t => t.View == relation); var filter = TimescaleSupport.MaterializationHoleSourceFilterFor(target.CreateSql); var sourceFilter = filter.Length == 0 ? string.Empty : " AND " + filter; return $""" diff --git a/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs b/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs index a0d1c3ac0c..8977df9283 100644 --- a/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs +++ b/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs @@ -102,16 +102,16 @@ public async Task TheFencedScan_FindsExactlyTheOldScansBuckets_OnEveryKindOfTarg var ct = TestContext.Current.CancellationToken; DateTime H(int n) => h0.AddHours(n); + /* #3653 LC: query_stats_hourly and query_stats_daily are frozen out of MaterializationHoleTargets — the + repair walk (and this oracle, which reads the same registry) never scans them any more. The legacy's + own half of the restart-row contrast (it admits H20, so H20 is neither a hole nor absent) is pinned + purely against its CreateSql text (MaterializationHoleRepairTests.TheSourceFilter_...); what remains + live to prove is the successor's half. */ var expected = new Dictionary(StringComparer.Ordinal) { - /* The tail. Not the outage, not the covered hours, and for the legacy, which admits the restart row, - not H20 either: it materialized that row. */ - [TimescaleSupport.QueryStatsHourlyView] = new[] { H(3), H(4) }, /* The successor rejects the restart row, so H20 is neither materialized nor a hole for it. */ [TimescaleSupport.QueryStatsIntervalHourlyView] = new[] { H(3), H(4) }, [TimescaleSupport.QueryStatsBaselineView] = new[] { H(3), H(4) }, - /* The daily over its hourly source: day B is held by the hourly and never materialized here. */ - [TimescaleSupport.QueryStatsDailyView] = new[] { h0.AddDays(1) }, }; foreach (var (view, holes) in expected) @@ -142,7 +142,9 @@ public async Task TheScanProbesPerBucket_AndTheSourceOnlyWhereTheMaterialization var (connection, h0) = (seed.Connection, seed.H0); var ct = TestContext.Current.CancellationToken; - var target = TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == TimescaleSupport.QueryStatsHourlyView); + /* #3653 LC: query_stats_hourly is frozen out of MaterializationHoleTargets; the live successor is + refreshed over the identical windows in SeedAsync, so its plan shape is the same proof. */ + var target = TimescaleSupport.MaterializationHoleTargets.Single(t => t.View == TimescaleSupport.QueryStatsIntervalHourlyView); var materialization = await TimescaleSupport.ResolveMaterializationAsync(connection, target.View, ct); Assert.NotNull(materialization); From d550701280d25f8b3bc8cf8825f3d7e097fcf5e0 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:38:45 -0400 Subject: [PATCH 07/18] issue-3653 A6 lane LC-b: live proofs for the freeze (#3653) Four live TimescaleDB tests against the LC freeze: dropping a frozen legacy hourly's chunks leaves its frozen daily's rows and sums unchanged, no start-path converge ever re-adds a frozen refresh policy while every other aggregate keeps one, the raw purge arms off the successor hourlies' coverage even though the legacy hourlies are empty (and stays held when a coverage relation falls short), and a frozen daily's compression policy drains only once every chunk it will ever hold is compressed. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../Darling.Tests/FrozenRollupLiveTests.cs | 613 ++++++++++++++++++ 1 file changed, 613 insertions(+) create mode 100644 Darling/Darling.Tests/FrozenRollupLiveTests.cs diff --git a/Darling/Darling.Tests/FrozenRollupLiveTests.cs b/Darling/Darling.Tests/FrozenRollupLiveTests.cs new file mode 100644 index 0000000000..ba8f335666 --- /dev/null +++ b/Darling/Darling.Tests/FrozenRollupLiveTests.cs @@ -0,0 +1,613 @@ +/* + * Copyright (c) 2026 Erik Darling, Darling Data LLC + * + * This file is part of the SQL Server Performance Monitor. + * + * Licensed under the MIT License. See LICENSE file in the project root for full license information. + */ + +using System; +using System.Collections.Generic; +using System.Linq; +using System.Threading; +using System.Threading.Tasks; +using Npgsql; +using PerformanceMonitor.Collectors; +using PerformanceMonitor.Darling.Storage; +using Xunit; + +namespace Darling.Tests; + +/// +/// #3653 (A6, lane LC-b): live proofs for the freeze (LC) against a real TimescaleDB store — the four claims a +/// unit/pin test cannot reach because they need actual chunks, jobs and catalog rows: (1) dropping a frozen +/// legacy hourly's chunks (its own retention, unchanged by the freeze) never touches its already-materialized +/// frozen daily; (2) no start-path converge ever re-adds a refresh policy to any of the frozen six +/// (), while every other continuous aggregate keeps one; +/// (3) the raw purge arms off the successor hourlies' coverage () +/// even though the legacy hourlies it used to key on are empty, and stays held when a coverage relation falls +/// short; (4) a frozen daily's compression policy drains only once every chunk it will ever hold is compressed, +/// and never touches a non-frozen daily's policy. +/// +/* #1776 own-store: deliberately NOT [Collection("live-postgres")]. Every test here goes through + ScratchPostgres.CreateAsync, which reaches DARLING_TEST_PG only to CREATE and DROP its own database and then + works entirely inside it. It never touches the shared database's tables, so it cannot race the live + collection, and serializing it would be pure slowdown. Leave it out; this comment is here so the next sweep + does not "fix" it. */ +public sealed class FrozenRollupLiveTests +{ + /// Distinctive fake id — a real server_id is a storage-name hash, never in this range. + private const int ServerId = -936554; + private const string ServerName = "a6-lcb-frozen-rollup"; + private const string Db = "FrozenDb"; + + /// Fixed anchor date (never wall clock): a Monday, chosen only so the seeded days are unambiguous. + private static readonly DateTime D0 = new(2026, 2, 2, 0, 0, 0, DateTimeKind.Unspecified); + + [Fact] + public async Task DroppingFrozenHourlyChunks_LeavesItsFrozenDailyUnchanged() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* 1-day materialization chunks on the frozen hourly, set BEFORE any refresh creates one, so a + day-aligned cutoff below drops exactly the days it names and none of the days it does not. */ + await SetMaterializationChunkIntervalAsync(connection, TimescaleSupport.QueryStatsHourlyView, "1 day", ct); + + /* A pre-#3653 store's own history: the legacy hourly carried a real refresh policy, like any other + aggregate, before LC froze it off the grid. */ + await using (var addPolicy = new NpgsqlCommand( + TimescaleSupport.AddContinuousAggregatePolicySql( + TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.HourlyRefreshStartOffset, + TimescaleSupport.HourlyRefreshScheduleInterval, TimescaleSupport.HourlyRefreshScheduleInterval, + phaseMinutes: null), + connection)) + { + await addPolicy.ExecuteNonQueryAsync(ct); + } + + for (var day = 0; day < 10; day++) + { + var at = D0.AddDays(day); + await InsertQueryStatsAsync(connection, at, $"LCBQ{day}A", 1000 + day, 10, 3600, ct); + await InsertQueryStatsAsync(connection, at.AddHours(2), $"LCBQ{day}B", 2000 + day, 20, 3600, ct); + } + + var d10 = D0.AddDays(10); + var cutoff = D0.AddDays(5); + + /* Materialize the legacy hourly and then its daily — the pre-freeze history a real service would + have built over time through the policy just added above. */ + await RefreshAsync(connection, TimescaleSupport.QueryStatsHourlyView, D0, d10, ct); + await RefreshAsync(connection, TimescaleSupport.QueryStatsDailyView, D0, d10, ct); + + var before = await ReadDailyTotalsAsync(connection, TimescaleSupport.QueryStatsDailyView, ServerId, D0, cutoff, ct); + Assert.Equal(5, before.Count); + + /* Run the start path so the freeze applies: detaches the policy added above. */ + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + /* The legacy hourly's OWN retention (unchanged by the freeze), dropping its old chunks. */ + await using (var drop = new NpgsqlCommand( + $"SELECT drop_chunks('collect.{TimescaleSupport.QueryStatsHourlyView}', older_than => $1::timestamp)", connection)) + { + drop.Parameters.AddWithValue(cutoff); + await drop.ExecuteNonQueryAsync(ct); + } + + /* Confirms the drop actually had teeth against the hourly, so the unchanged-daily assertion below + is not vacuous. */ + await using (var hourlyLeft = new NpgsqlCommand( + $"SELECT count(*) FROM collect.{TimescaleSupport.QueryStatsHourlyView} WHERE server_id = $1 AND bucket < $2", connection)) + { + hourlyLeft.Parameters.AddWithValue(ServerId); + hourlyLeft.Parameters.AddWithValue(cutoff); + Assert.Equal(0L, Convert.ToInt64(await hourlyLeft.ExecuteScalarAsync(ct))); + } + + /* Every start step that could refresh a rollup. RollupBackfill's targets are CLI-only + (DarlingCliCommands' rollup-backfill verb) — the service start path never runs a backfill slice + on its own, so there is no third step to run here. */ + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + await TimescaleSupport.RepairMaterializationHolesAsync(connection, null, DateTime.UtcNow, ct); + + var after = await ReadDailyTotalsAsync(connection, TimescaleSupport.QueryStatsDailyView, ServerId, D0, cutoff, ct); + Assert.Equal(before, after); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + + [Fact] + public async Task StartPathRunTwice_NeverReAddsAFrozenRefreshPolicy_AndEveryOtherAggregateKeepsOne() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* Two of the six frozen rollups get a hand-attached policy first, mimicking a store that took a + build before #3653's LC — the only shape that can prove the removal below actually runs, rather + than there simply never having been a policy to find. */ + await using (var addHourly = new NpgsqlCommand( + TimescaleSupport.AddContinuousAggregatePolicySql( + TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.HourlyRefreshStartOffset, + TimescaleSupport.HourlyRefreshScheduleInterval, TimescaleSupport.HourlyRefreshScheduleInterval, + phaseMinutes: null), + connection)) + { + await addHourly.ExecuteNonQueryAsync(ct); + } + await using (var addDaily = new NpgsqlCommand( + TimescaleSupport.AddContinuousAggregatePolicySql( + TimescaleSupport.QueryStatsDailyView, TimescaleSupport.DailyRefreshStartOffset, + TimescaleSupport.DailyRefreshScheduleInterval, TimescaleSupport.DailyRefreshScheduleInterval, + phaseMinutes: null), + connection)) + { + await addDaily.ExecuteNonQueryAsync(ct); + } + + /* The start path, twice — the second run is the converge a restart performs. */ + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var scheduled = await ReadScheduledRefreshViewsAsync(connection, ct); + + foreach (var (_, view) in TimescaleSupport.FrozenRollupAggregates) + { + Assert.DoesNotContain(view, scheduled); + } + + var everyOther = TimescaleSupport.HourlyAggregates.Select(a => a.View) + .Concat(TimescaleSupport.DailyAggregates.Select(a => a.View)) + .Concat(TimescaleSupport.BaselineAggregates.Select(a => a.View)) + .Concat(TimescaleSupport.OffGridAggregates.Select(a => a.View)); + + foreach (var view in everyOther) + { + Assert.Contains(view, scheduled); + } + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + + [Fact] + public async Task RawPurge_ArmsOffSuccessorHourlyCoverage_NotTheEmptyLegacyOnes() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* Read from the registry rather than naming the successors by hand — correct whether the + query_stats row carries one coverage relation or (once LC-a4 lands) two. */ + var queryStatsCoverage = TimescaleSupport.RawTierCoverage.Single(r => string.Equals(r.Relation, "query_stats", StringComparison.Ordinal)); + var procedureStatsCoverage = TimescaleSupport.RawTierCoverage.Single(r => string.Equals(r.Relation, "procedure_stats", StringComparison.Ordinal)); + + var d0 = D0; + var d1 = D0.AddDays(1); + var d2 = D0.AddDays(2); + + for (var day = 0; day < 2; day++) + { + var at = D0.AddDays(day); + await InsertQueryStatsAsync(connection, at, $"LCBRAWQ{day}", 1000, 10, 3600, ct); + await InsertProcedureStatsAsync(connection, at, $"lcb_raw_proc_{day}", 900, 9, 3600, ct); + } + + /* procedure_stats' successor coverage is refreshed fully from raw's oldest row — Covered. */ + foreach (var view in procedureStatsCoverage.Coverage) + { + await RefreshAsync(connection, view, d0, d2, ct); + } + + /* query_stats' successor coverage is refreshed SHORT of raw's oldest row (missing D0) — Short. The + legacy hourly stays empty throughout, which is exactly the state the freeze leaves it in. */ + foreach (var view in queryStatsCoverage.Coverage) + { + await RefreshAsync(connection, view, d1, d2, ct); + } + + /* The legacy hourlies are never named above, never refreshed, and stay materialized empty — exactly + the state the freeze leaves them in — so this pins that arming below is NOT reading them: if + RawTierCoverage's query_stats/procedure_stats rows ever pointed back at these instead of the + successors, the refresh loops above would populate them (they refresh whatever Coverage names) + and this assertion would catch it. */ + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.QueryStatsHourlyView, ct)); + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, ct)); + + await TimescaleSupport.EnsureRetentionPoliciesAsync(connection, null, ct); + + Assert.Equal(false, await IsRetentionScheduledAsync(connection, "query_stats", ct)); + Assert.Equal(true, await IsRetentionScheduledAsync(connection, "procedure_stats", ct)); + + /* The contrast, on the SAME relation: backfill the missing day and re-evaluate. Coverage now + reaches raw's oldest row, and the hold releases on the very next pass, with no restart (#1877). */ + foreach (var view in queryStatsCoverage.Coverage) + { + await RefreshAsync(connection, view, d0, d1, ct); + } + + await TimescaleSupport.EnsureRetentionPoliciesAsync(connection, null, ct); + + Assert.Equal(true, await IsRetentionScheduledAsync(connection, "query_stats", ct)); + Assert.Equal(true, await IsRetentionScheduledAsync(connection, "procedure_stats", ct)); + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.QueryStatsHourlyView, ct)); + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + + [Fact] + public async Task FrozenDailyCompressionPolicy_DrainsOnlyOnceEveryChunkIsCompressed() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* A real, non-frozen daily-tier target — untouched by the drain in both phases below. Read off the + registry rather than named, so this still holds if the roster ever changes. */ + var controlDaily = TimescaleSupport.DailyAggregates.First().View; + + await SetMaterializationChunkIntervalAsync(connection, TimescaleSupport.QueryStatsDailyView, "1 day", ct); + + for (var day = 0; day < 3; day++) + { + await InsertQueryStatsAsync(connection, D0.AddDays(day), $"LCBDR{day}", 1000, 10, 3600, ct); + } + + var d3 = D0.AddDays(3); + await RefreshAsync(connection, TimescaleSupport.QueryStatsHourlyView, D0, d3, ct); + await RefreshAsync(connection, TimescaleSupport.QueryStatsDailyView, D0, d3, ct); + + /* A pre-#3653 store's own history: compression already enabled and a policy already attached, + exactly as every other daily-tier aggregate got before LC froze this trio's daily-band membership + and stopped compression-managing it going forward. */ + await using (var enable = new NpgsqlCommand( + $"ALTER MATERIALIZED VIEW collect.{TimescaleSupport.QueryStatsDailyView} SET (timescaledb.compress, " + + $"timescaledb.compress_segmentby = '{TimescaleSupport.AggregateCompressionSegmentBy}', timescaledb.compress_orderby = 'bucket DESC')", + connection)) + { + await enable.ExecuteNonQueryAsync(ct); + } + await using (var addPolicy = new NpgsqlCommand( + $"SELECT add_compression_policy('collect.{TimescaleSupport.QueryStatsDailyView}', compress_after => INTERVAL '1 day', " + + "schedule_interval => INTERVAL '12 hours', if_not_exists => true)", connection)) + { + await addPolicy.ExecuteNonQueryAsync(ct); + } + + var chunks = await ReadMaterializationChunksAsync(connection, TimescaleSupport.QueryStatsDailyView, ct); + Assert.True(chunks.Count >= 2, "the frozen daily needs at least two chunks to prove a partial compress leaves the policy in place"); + + /* Compress every chunk but the last (oldest first), so exactly one chunk stays uncompressed. */ + for (var i = 0; i < chunks.Count - 1; i++) + { + await CompressChunkAsync(connection, chunks[i], ct); + } + + var beforeDrain = await ReadDrainStateAsync(connection, TimescaleSupport.QueryStatsDailyView, ct); + Assert.NotNull(beforeDrain?.JobId); + Assert.True(beforeDrain!.UncompressedChunks > 0); + + await TimescaleSupport.EnsureAggregateCompressionAsync(connection, null, ct); + + var afterPartial = await ReadDrainStateAsync(connection, TimescaleSupport.QueryStatsDailyView, ct); + Assert.Equal(beforeDrain.JobId, afterPartial?.JobId); + Assert.NotNull(await ReadCompressionJobIdAsync(connection, controlDaily, ct)); + + /* Now compress the last chunk — every chunk the frozen daily will ever hold is compressed. */ + await CompressChunkAsync(connection, chunks[^1], ct); + + await TimescaleSupport.EnsureAggregateCompressionAsync(connection, null, ct); + + var afterFull = await ReadDrainStateAsync(connection, TimescaleSupport.QueryStatsDailyView, ct); + Assert.Null(afterFull?.JobId); + Assert.NotNull(await ReadCompressionJobIdAsync(connection, controlDaily, ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + + private static async Task InsertQueryStatsAsync( + NpgsqlConnection connection, DateTime at, string hash, long workerUs, long executions, int intervalSeconds, CancellationToken ct) + { + await using var insert = new NpgsqlCommand(@" +INSERT INTO collect.query_stats + (collection_id, collection_time, server_id, server_name, database_name, query_hash, sql_handle, + delta_worker_time, delta_elapsed_time, delta_execution_count, sample_interval_seconds) +VALUES ($1, $2, $3, $4, $5, $6, $7, $8, $8, $9, $10)", connection); + insert.Parameters.AddWithValue(CollectionIdGenerator.Next()); + insert.Parameters.AddWithValue(DarlingMcpTestData.TruncateToSeconds(at)); + insert.Parameters.AddWithValue(ServerId); + insert.Parameters.AddWithValue(ServerName); + insert.Parameters.AddWithValue(Db); + insert.Parameters.AddWithValue(hash); + insert.Parameters.AddWithValue("0x" + hash); + insert.Parameters.AddWithValue(workerUs); + insert.Parameters.AddWithValue(executions); + insert.Parameters.AddWithValue(intervalSeconds); + await insert.ExecuteNonQueryAsync(ct); + } + + private static async Task InsertProcedureStatsAsync( + NpgsqlConnection connection, DateTime at, string objectName, long workerUs, long executions, int intervalSeconds, CancellationToken ct) + { + await using var insert = new NpgsqlCommand(@" +INSERT INTO collect.procedure_stats + (collection_id, collection_time, server_id, server_name, database_name, schema_name, object_name, sql_handle, + delta_worker_time, delta_elapsed_time, delta_execution_count, sample_interval_seconds) +VALUES ($1, $2, $3, $4, $5, 'dbo', $6, $7, $8, $8, $9, $10)", connection); + insert.Parameters.AddWithValue(CollectionIdGenerator.Next()); + insert.Parameters.AddWithValue(DarlingMcpTestData.TruncateToSeconds(at)); + insert.Parameters.AddWithValue(ServerId); + insert.Parameters.AddWithValue(ServerName); + insert.Parameters.AddWithValue(Db); + insert.Parameters.AddWithValue(objectName); + insert.Parameters.AddWithValue("0x" + objectName); + insert.Parameters.AddWithValue(workerUs); + insert.Parameters.AddWithValue(executions); + insert.Parameters.AddWithValue(intervalSeconds); + await insert.ExecuteNonQueryAsync(ct); + } + + private static async Task RefreshAsync(NpgsqlConnection connection, string view, DateTime from, DateTime to, CancellationToken ct) + { + await using var refresh = new NpgsqlCommand($"CALL refresh_continuous_aggregate('collect.{view}'::regclass, $1::timestamp, $2::timestamp)", connection); + refresh.Parameters.AddWithValue(from); + refresh.Parameters.AddWithValue(to); + await refresh.ExecuteNonQueryAsync(ct); + } + + /// Sets ONE continuous aggregate's materialization chunk width directly — the frozen six take no + /// width from (it walks only + /// ), so a live test that needs deterministic + /// chunk boundaries on one of them has to set it here. Must run before the first refresh creates a chunk — + /// set_chunk_time_interval governs only chunks created after the call. + private static async Task SetMaterializationChunkIntervalAsync(NpgsqlConnection connection, string view, string interval, CancellationToken ct) + { + await using var command = new NpgsqlCommand($@" +SELECT set_chunk_time_interval(format('%I.%I', ca.materialization_hypertable_schema, ca.materialization_hypertable_name)::regclass, INTERVAL '{interval}') +FROM timescaledb_information.continuous_aggregates AS ca +WHERE ca.view_schema = 'collect' AND ca.view_name = '{view}'", connection); + await command.ExecuteNonQueryAsync(ct); + } + + private static async Task> ReadDailyTotalsAsync( + NpgsqlConnection connection, string view, int serverId, DateTime start, DateTime end, CancellationToken ct) + { + await using var read = new NpgsqlCommand( + $"SELECT bucket, sum(worker_time_sum), sum(execution_count_sum) FROM collect.{view} " + + "WHERE server_id = $1 AND bucket >= $2 AND bucket < $3 GROUP BY bucket ORDER BY bucket", connection); + read.Parameters.AddWithValue(serverId); + read.Parameters.AddWithValue(start); + read.Parameters.AddWithValue(end); + var rows = new List<(DateTime, long, long)>(); + await using var reader = await read.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + rows.Add((reader.GetDateTime(0), reader.GetInt64(1), reader.GetInt64(2))); + } + return rows; + } + + private static async Task> ReadScheduledRefreshViewsAsync(NpgsqlConnection connection, CancellationToken ct) + { + await using var read = new NpgsqlCommand( + $"SELECT hypertable_name FROM timescaledb_information.jobs WHERE proc_name = '{TimescaleSupport.RefreshPolicyProcName}' AND hypertable_schema = 'collect'", + connection); + var views = new HashSet(StringComparer.Ordinal); + await using var reader = await read.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + views.Add(reader.GetString(0)); + } + return views; + } + + private static async Task CountRowsAsync(NpgsqlConnection connection, string relation, CancellationToken ct) + { + await using var count = new NpgsqlCommand($"SELECT count(*) FROM collect.{relation}", connection); + return Convert.ToInt64(await count.ExecuteScalarAsync(ct)); + } + + private static async Task IsRetentionScheduledAsync(NpgsqlConnection connection, string relation, CancellationToken ct) + { + await using var read = new NpgsqlCommand(TimescaleSupport.RetentionPolicyScheduledSql(relation), connection); + var value = await read.ExecuteScalarAsync(ct); + return value is bool b ? b : null; + } + + private static async Task> ReadMaterializationChunksAsync(NpgsqlConnection connection, string view, CancellationToken ct) + { + await using var read = new NpgsqlCommand($@" +SELECT format('%I.%I', c.chunk_schema, c.chunk_name) +FROM timescaledb_information.chunks AS c +JOIN timescaledb_information.continuous_aggregates AS ca + ON c.hypertable_schema = ca.materialization_hypertable_schema + AND c.hypertable_name = ca.materialization_hypertable_name +WHERE ca.view_schema = 'collect' AND ca.view_name = '{view}' +ORDER BY c.range_start", connection); + var chunks = new List(); + await using var reader = await read.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + chunks.Add(reader.GetString(0)); + } + return chunks; + } + + private static async Task CompressChunkAsync(NpgsqlConnection connection, string chunk, CancellationToken ct) + { + await using var compress = new NpgsqlCommand($"SELECT compress_chunk('{chunk}'::regclass, if_not_compressed => true)", connection); + await compress.ExecuteNonQueryAsync(ct); + } + + private static async Task ReadDrainStateAsync( + NpgsqlConnection connection, string view, CancellationToken ct) + { + await using var read = new NpgsqlCommand(TimescaleSupport.FrozenDailyCompressionDrainStateSql, connection); + await using var reader = await read.ExecuteReaderAsync(ct); + while (await reader.ReadAsync(ct)) + { + if (string.Equals(reader.GetString(0), view, StringComparison.Ordinal)) + { + return new TimescaleSupport.FrozenDailyCompressionDrainState( + reader.GetString(0), + reader.IsDBNull(1) ? null : reader.GetInt32(1), + Convert.ToInt64(reader.GetValue(2))); + } + } + return null; + } + + private static async Task ReadCompressionJobIdAsync(NpgsqlConnection connection, string view, CancellationToken ct) + { + await using var read = new NpgsqlCommand($@" +SELECT j.job_id +FROM timescaledb_information.jobs AS j +WHERE (j.proc_name LIKE '%compression%' OR j.proc_name LIKE '%columnstore%') +AND j.hypertable_schema = 'collect' +AND j.hypertable_name = '{view}'", connection); + var value = await read.ExecuteScalarAsync(ct); + return value is null or DBNull ? null : Convert.ToInt32(value); + } +} From 2cd69d328d5d9528608a1236343b79c91087ec98 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Thu, 24 Sep 2026 22:39:21 -0400 Subject: [PATCH 08/18] issue-3653 A6 lane LC-a6: re-pin the 11 live/pure tests the freeze broke The legacy six moved to FrozenRollupAggregates, left RollupBackfill.Targets and HourlyRefreshPhaseOrder, and stopped getting a refresh policy. Every test that used one of them as a convenient live example now moves to its interval-honest successor; every test that counts EnsureContinuousAggregatesAsync's "ready" total now adds FrozenRollupAggregates.Length; the compression pin moves from 23 to 20 targets, and the successor dailies now get a real compression policy instead of the removed deferral. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../IntervalHonestHourlyRollupTests.cs | 5 +- .../MeasurementContractCensusTests.cs | 22 ++++--- .../Darling.Tests/RollupBackfillLiveTests.cs | 64 ++++++++++++------- .../Darling.Tests/SuccessorDailyLiveTests.cs | 39 ++++++++--- .../TimescaleAggregateCompressionTests.cs | 33 ++++++---- .../Darling.Tests/TimescaleSupportTests.cs | 12 +++- 6 files changed, 119 insertions(+), 56 deletions(-) diff --git a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs index f0904206cb..dafa52c81c 100644 --- a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs +++ b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs @@ -592,7 +592,10 @@ INSERT INTO collect.query_stats so the successors are created exactly as a store will create them, and the sweep's own count says every one built. */ var ready = await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); - Assert.Equal(TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length, ready); + /* #3653 LC: the sweep's unified aggregates list also concats FrozenRollupAggregates (the six legacy + rollups still get CREATEd on a fresh store, just no refresh policy), so ready is six higher than the + four grid/baseline/off-grid lists alone. */ + Assert.Equal(TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length + TimescaleSupport.FrozenRollupAggregates.Length, ready); foreach (var view in new[] { TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsIntervalHourlyView, TimescaleSupport.QueryStatsDbHourlyView, TimescaleSupport.QueryStatsDbIntervalHourlyView }) { diff --git a/Darling/Darling.Tests/MeasurementContractCensusTests.cs b/Darling/Darling.Tests/MeasurementContractCensusTests.cs index 156a3ed27d..83725078b7 100644 --- a/Darling/Darling.Tests/MeasurementContractCensusTests.cs +++ b/Darling/Darling.Tests/MeasurementContractCensusTests.cs @@ -888,13 +888,14 @@ private static string Snake(string pascal) => /// ). A restart's (0, 0) row costs their sum()s /// nothing, but count(*) AS sample_count counts it as a sample on all three, and min(delta_*) /// reads it as a real minimum on the two that carry a min (query_stats_db_hourly carries sums and the - /// sample count only). Unlike the baseline pair they STAY registered and refreshing: the indefinite daily - /// tier is hierarchical from them and a continuous aggregate's source is fixed at CREATE, so they cannot be - /// dropped without cascading it nor frozen without stopping it. Every hourly-tier reader takes the - /// successor through RollupCoverage.HourlyRelationFor where it reaches as far as the legacy; the - /// daily tier inherits the legacy's contamination at the day grain until it has successors of its own - /// (which need the daily compression band re-derived first — it is full at twenty-three). These three - /// leave this list only with the legacy text, i.e. with a stitched read or a daily-successor lane. + /// sample count only). Unlike the baseline pair they STAY registered — in FrozenRollupAggregates since + /// #3653 LC's freeze, not HourlyAggregates — but no longer refreshing: the daily tier is hierarchical from + /// them and a continuous aggregate's source is fixed at CREATE, so LC froze the pair in place (no refresh + /// policy, watermark never advances again) rather than dropping or cascading it. Every hourly-tier reader + /// takes the successor through RollupCoverage.HourlyRelationFor where it reaches as far as the + /// legacy; the daily tier inherits the legacy's contamination at the day grain — LB's three interval-honest + /// successor dailies now exist, but only LA's stitched reads route to them past the legacy's floor. These + /// three leave this list only when the legacy CREATE text itself is finally retired. /// /// /// @@ -1000,7 +1001,9 @@ contamination named on the roster is really in their text. */ Assert.Equal(TimescaleSupport.SupersededHourlyRollups.Select(s => s.Legacy).OrderBy(v => v, StringComparer.Ordinal), registeredAdmitting); foreach (var view in registeredAdmitting) { - Assert.Contains(view, TimescaleSupport.HourlyAggregates.Select(a => a.View)); + /* #3653 LC: the trio froze out of HourlyAggregates into FrozenRollupAggregates — still registered, + just on the other list, so the membership check has to look at both. */ + Assert.Contains(view, TimescaleSupport.HourlyAggregates.Concat(TimescaleSupport.FrozenRollupAggregates).Select(a => a.View)); var text = Text(aggregates.Single(a => a.View == view).Constant); foreach (var shape in contamination[view]) { @@ -1030,7 +1033,8 @@ its answer's shape — is red here rather than at a 42703 on a store. */ the structural fact the whole "stays registered" reasoning rests on. */ var dailyText = Text(aggregates.Single(a => a.View == dependentDaily).Constant); Assert.Contains($"FROM collect.{view}", dailyText, StringComparison.Ordinal); - Assert.Contains(dependentDaily, TimescaleSupport.DailyAggregates.Select(a => a.View)); + /* #3653 LC: the dependent daily froze right alongside its hourly — same list move, same reason. */ + Assert.Contains(dependentDaily, TimescaleSupport.DailyAggregates.Concat(TimescaleSupport.FrozenRollupAggregates).Select(a => a.View)); } } diff --git a/Darling/Darling.Tests/RollupBackfillLiveTests.cs b/Darling/Darling.Tests/RollupBackfillLiveTests.cs index f072565b2b..d510c58ac4 100644 --- a/Darling/Darling.Tests/RollupBackfillLiveTests.cs +++ b/Darling/Darling.Tests/RollupBackfillLiveTests.cs @@ -74,17 +74,24 @@ public async Task RollupCreatedOverExistingHistory_ReadsEmptyBelowItsFloor_Until /* ── 2. The rollup is created over that history, WITH NO DATA — the #1759 shape exactly. ── */ await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + /* #3653 LC: query_stats_hourly is one of the frozen six — it still stands up WITH NO DATA, but nothing + ever gives it a refresh policy or feeds the raw purge gate off its coverage again (RawTierCoverage + now names the successor). The #1759 shape this test proves — a rollup created over pre-existing + history reads empty below its shallow floor until backfilled, and the purge gate then releases — + only still happens on a view the freeze left live, so query_stats_interval_hourly stands in as the + example, keeping the same source, grain and daily companion. */ + /* Materialize only the recent window, which is all the 3-day refresh policy would ever have done on a store that existed before its rollups. This is what makes the floor shallow. */ await RollupBackfill.RunSliceAsync( - connection, TimescaleSupport.QueryStatsHourlyView, now.Date.AddDays(-2), now.Date.AddDays(1), SilentDisclosure(), ct); + connection, TimescaleSupport.QueryStatsIntervalHourlyView, now.Date.AddDays(-2), now.Date.AddDays(1), SilentDisclosure(), ct); await using var dataSource = NpgsqlDataSource.Create(scratch.ConnectionString); var rollups = await TimescaleSupport.DetectRollupsAsync(dataSource, ct); - Assert.True(rollups.QueryGrainHourly, "the query_stats hourly rollup should exist after the ensure sweep"); + Assert.True(rollups.QueryGrainIntervalHourly, "the query_stats_interval_hourly rollup should exist after the ensure sweep"); var before = await TimescaleSupport.DetectRollupCoverageAsync(dataSource, rollups, ct); - var floorBefore = before.FloorOf(TimescaleSupport.QueryStatsHourlyView); + var floorBefore = before.FloorOf(TimescaleSupport.QueryStatsIntervalHourlyView); var rawFloor = before.RawOldestOf("query_stats"); Assert.NotNull(floorBefore); @@ -96,7 +103,7 @@ await RollupBackfill.RunSliceAsync( the rows — this is the hard partition, not a slow path. If this assertion ever stops holding, the premise behind all of #1759 has changed and the fix should be revisited, not patched. ── */ var windowStart = now.AddDays(-8); - var rollupRows = await CountAsync(connection, $"SELECT count(*) FROM collect.{TimescaleSupport.QueryStatsHourlyView} WHERE bucket >= $1 AND bucket < $2", windowStart, now.AddDays(-6), ct); + var rollupRows = await CountAsync(connection, $"SELECT count(*) FROM collect.{TimescaleSupport.QueryStatsIntervalHourlyView} WHERE bucket >= $1 AND bucket < $2", windowStart, now.AddDays(-6), ct); var rawRows = await CountAsync(connection, "SELECT count(*) FROM collect.query_stats WHERE collection_time >= $1 AND collection_time < $2", windowStart, now.AddDays(-6), ct); Assert.True(rawRows > 0, "raw must hold rows in the pre-coverage window, or the test is not exercising #1759"); @@ -104,19 +111,19 @@ await RollupBackfill.RunSliceAsync( /* ── 4. PHASE 1: the router must send that window to RAW, where the data actually is. Age alone puts an 8-day window on the hourly rollup, which would answer with silence. ── */ - var coverageLadder = before.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView); + var coverageLadder = before.For(TimescaleSupport.QueryStatsIntervalHourlyView, TimescaleSupport.QueryStatsIntervalDailyView); Assert.Equal( RetentionTier.Hourly, - RetentionTierRouter.Resolve(now, windowStart, rollups.QueryGrainHourly, rollups.QueryGrainDaily)); + RetentionTierRouter.Resolve(now, windowStart, rollups.QueryGrainIntervalHourly, rollups.QueryGrainIntervalDaily)); Assert.Equal( RetentionTier.Raw, - RetentionTierRouter.Resolve(now, windowStart, rollups.QueryGrainHourly, rollups.QueryGrainDaily, coverageLadder)); + RetentionTierRouter.Resolve(now, windowStart, rollups.QueryGrainIntervalHourly, rollups.QueryGrainIntervalDaily, coverageLadder)); /* ── 5. PHASE 2: back fill in slices, newest-first, exactly as the verb does. ── */ var plan = RollupBackfill.Plan( - TimescaleSupport.QueryStatsHourlyView, rawFloor, floorBefore, + TimescaleSupport.QueryStatsIntervalHourlyView, rawFloor, floorBefore, materializedBuckets: 48, materializedBytes: 48 * 1024, rawBytes: 0, bucketWidth: TimeSpan.FromHours(1)); @@ -127,7 +134,7 @@ await RollupBackfill.RunSliceAsync( var previousFloor = floorBefore; foreach (var (from, to) in RollupBackfill.Slices(plan.FromUtc, plan.ToUtc)) { - var floorDuring = await RollupBackfill.RunSliceAsync(connection, TimescaleSupport.QueryStatsHourlyView, from, to, SilentDisclosure(), ct); + var floorDuring = await RollupBackfill.RunSliceAsync(connection, TimescaleSupport.QueryStatsIntervalHourlyView, from, to, SilentDisclosure(), ct); slicesRun++; /* Progress only ever goes BACKWARDS. Deliberately NOT "the floor reached this slice's start": a @@ -149,14 +156,14 @@ after this would be vacuous. */ /* ── 6. CONVERGENCE, MEASURED FROM DATA — never from the fact that the calls returned. ── */ var after = await TimescaleSupport.DetectRollupCoverageAsync(dataSource, rollups, ct); - var floorAfter = after.FloorOf(TimescaleSupport.QueryStatsHourlyView); + var floorAfter = after.FloorOf(TimescaleSupport.QueryStatsIntervalHourlyView); Assert.NotNull(floorAfter); Assert.True(floorAfter <= rawFloor, $"after the backfill the rollup must reach at or before raw's oldest row (rollup {floorAfter:O}, raw {rawFloor:O})"); /* The window that was empty in step 3 now has rows. */ - var rollupRowsAfter = await CountAsync(connection, $"SELECT count(*) FROM collect.{TimescaleSupport.QueryStatsHourlyView} WHERE bucket >= $1 AND bucket < $2", windowStart, now.AddDays(-6), ct); + var rollupRowsAfter = await CountAsync(connection, $"SELECT count(*) FROM collect.{TimescaleSupport.QueryStatsIntervalHourlyView} WHERE bucket >= $1 AND bucket < $2", windowStart, now.AddDays(-6), ct); Assert.True(rollupRowsAfter > 0, "the backfilled window must now be materialized"); /* ── 7. The router goes BACK to the rollup — the backfill restored acceleration rather than stranding @@ -164,8 +171,8 @@ every old read on raw forever. ── */ Assert.Equal( RetentionTier.Hourly, RetentionTierRouter.Resolve( - now, windowStart, rollups.QueryGrainHourly, rollups.QueryGrainDaily, - after.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView))); + now, windowStart, rollups.QueryGrainIntervalHourly, rollups.QueryGrainIntervalDaily, + after.For(TimescaleSupport.QueryStatsIntervalHourlyView, TimescaleSupport.QueryStatsIntervalDailyView))); /* ── 8. And the ARMING GATE now reports safe, which is the whole point: the held raw purge releases itself on the next evaluation — the running service's hourly pass since #3812, or the next start @@ -304,12 +311,15 @@ safe to. ── */ var dryText = dryOut.ToString(); Assert.Contains("--dry-run: nothing was materialized", dryText, StringComparison.Ordinal); Assert.Contains("Free space on the store volume", dryText, StringComparison.Ordinal); - Assert.Contains(TimescaleSupport.QueryStatsHourlyView, dryText, StringComparison.Ordinal); + /* #3653 LC: query_stats_hourly is one of the frozen six and left RollupBackfill.Targets — the verb + would never plan it and its name would not appear here. query_stats_interval_hourly, its + interval-honest successor, is still a live target and stands in as the example. */ + Assert.Contains(TimescaleSupport.QueryStatsIntervalHourlyView, dryText, StringComparison.Ordinal); await using (var check = new NpgsqlConnection(scratch.ConnectionString)) { await check.OpenAsync(ct); - var probe = await RollupBackfill.ProbeAsync(check, TimescaleSupport.QueryStatsHourlyView, "query_stats", "collection_time", ct); + var probe = await RollupBackfill.ProbeAsync(check, TimescaleSupport.QueryStatsIntervalHourlyView, "query_stats", "collection_time", ct); /* Deliberately NOT "the rollup is still empty": the aggregate's own refresh policy is attached by the ensure sweep and fires immediately, so a trailing window IS materialized by the time a @@ -701,7 +711,11 @@ public async Task InterruptedBackfill_DoesNotLookComplete_AndTheArmingGateRefuse await SeedHourlyQueryStatsAsync(connection, now.AddDays(-HistoryDays), now, ct); await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); - var view = TimescaleSupport.QueryStatsHourlyView; + /* #3653 LC: query_stats_hourly is one of the frozen six now — nothing ever advances its watermark + again, so backfilling IT can never arm the raw purge gate. RawTierCoverage arms query_stats' + retention off the interval-honest successor's coverage instead (RequireSuccessorOf), so that is + the view this test's arming-gate assertions have to drive. */ + var view = TimescaleSupport.QueryStatsIntervalHourlyView; var probe = await RollupBackfill.ProbeAsync(connection, view, "query_stats", "collection_time", ct); var plan = RollupBackfill.Plan( view, probe.SourceOldestUtc, probe.CoverageOldestUtc, @@ -927,15 +941,19 @@ public async Task HierarchicalDaily_ConvergesToItsHourly_NotToRaw_SoTheHourlyTie await SeedHourlyQueryStatsAsync(connection, historyFrom, now, ct); - /* Aggregates WITHOUT their refresh policies — see the remarks. */ - foreach (var createSql in new[] { TimescaleSupport.CreateQueryStatsHourlySql, TimescaleSupport.CreateQueryStatsDailySql }) + /* Aggregates WITHOUT their refresh policies — see the remarks. + #3653 LC: query_stats_hourly/query_stats_daily are two of the frozen six now, and left + RollupBackfill.Targets entirely — the Single() lookup below would throw. The interval-honest + successor pair is still hierarchical the same way (daily reads the hourly's bucket, not raw) and + still a live Targets entry, so it stands in as the example. */ + foreach (var createSql in new[] { TimescaleSupport.CreateQueryStatsIntervalHourlySql, TimescaleSupport.CreateQueryStatsIntervalDailySql }) { await using var create = new NpgsqlCommand(createSql, connection); await create.ExecuteNonQueryAsync(ct); } - var hourly = TimescaleSupport.QueryStatsHourlyView; - var daily = TimescaleSupport.QueryStatsDailyView; + var hourly = TimescaleSupport.QueryStatsIntervalHourlyView; + var daily = TimescaleSupport.QueryStatsIntervalDailyView; /* The hourly holds the WHOLE history... */ await RollupBackfill.RunSliceAsync(connection, hourly, historyFrom, now.Date, SilentDisclosure(), ct); @@ -986,14 +1004,14 @@ Converging to raw would stop at rawOldest and fail this by the whole purged span await UnscheduleAllJobsAsync(connection, ct); } - /// Is query_stats_hourly's retention policy ARMED? The #1798 observable — it is gated on the - /// DAILY covering the hourly, which is the comparison the backfill now converges to. + /// Is query_stats_interval_hourly's retention policy ARMED? The #1798 observable — it is gated on + /// the DAILY covering the hourly, which is the comparison the backfill now converges to. private const string ArmedHourlyPolicySql = @" SELECT count(*) FROM timescaledb_information.jobs AS j WHERE j.proc_name = 'policy_retention' AND j.hypertable_schema = 'collect' -AND j.hypertable_name = 'query_stats_hourly' +AND j.hypertable_name = 'query_stats_interval_hourly' AND j.scheduled"; /// One nullable timestamp, for reading a relation's oldest instant. diff --git a/Darling/Darling.Tests/SuccessorDailyLiveTests.cs b/Darling/Darling.Tests/SuccessorDailyLiveTests.cs index 2b9599dd98..40f0ef253d 100644 --- a/Darling/Darling.Tests/SuccessorDailyLiveTests.cs +++ b/Darling/Darling.Tests/SuccessorDailyLiveTests.cs @@ -34,11 +34,13 @@ public sealed class SuccessorDailyLiveTests private const int ServerId = -936537; private const string ServerName = "successor-daily-e2e"; - /// Step 1: the ensure sweep creates all three successor dailies, attaches a refresh policy to - /// each, attaches NO compression policy to any of them (the LC-freeze defer), and a SECOND sweep is a - /// no-op — no error, no duplicate policy. + /// Step 1: the ensure sweep creates all three successor dailies and attaches a refresh policy to + /// each; the SEPARATE compression ensure then attaches a compression policy to each too, now that LC has + /// frozen the legacy trio's own daily-band membership and freed the three slots the successors take in + /// (no more deferral). Both sweeps are + /// idempotent — a SECOND pass of each is a no-op, no error, no duplicate policy. [Fact] - public async Task EnsureSweep_CreatesSuccessorDailies_WithRefreshButNoCompression_AndIsIdempotent() + public async Task EnsureSweep_CreatesSuccessorDailies_WithRefreshAndCompression_AndIsIdempotent() { var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), @@ -79,27 +81,44 @@ public async Task EnsureSweep_CreatesSuccessorDailies_WithRefreshButNoCompressio swallowed failure fails HERE with the verbatim Postgres/Timescale error in the message, rather than surfacing three call frames downstream as a bare "relation does not exist". */ Assert.False(log.Joined.Contains("Warning", StringComparison.Ordinal), "ensure sweep warnings:\n" + log.Joined); + /* #3653 LC: the sweep's unified aggregates list now also concats FrozenRollupAggregates (the six + legacy rollups still get CREATEd on a fresh store, just no refresh policy), so the ready count is + six higher than the four grid/baseline/off-grid lists alone. */ Assert.Equal( TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length - + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length, + + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length + + TimescaleSupport.FrozenRollupAggregates.Length, readyFirst); + /* The compression ensure is a SEPARATE sweep (EnsureAggregateCompressionAsync), never called by + EnsureContinuousAggregatesAsync — the worker's own start order runs it right after. Derived from + IsAggregateCompressionTarget rather than typed as 1, so this keeps proving whatever the registry + says rather than a value copied from today's membership. */ + var compressedFirst = await TimescaleSupport.EnsureAggregateCompressionAsync(connection, log, ct); + Assert.False(log.Joined.Contains("Warning", StringComparison.Ordinal), "compression ensure warnings:\n" + log.Joined); + foreach (var view in successorDailies) { Assert.True(await RelationExistsAsync(connection, view, ct), $"{view} must exist after the ensure sweep: {log.Joined}"); Assert.Equal(1, await RefreshPolicyCountAsync(connection, view, ct)); - Assert.Equal(0, await CompressionPolicyCountAsync(connection, view, ct)); + Assert.Equal( + TimescaleSupport.IsAggregateCompressionTarget(view) ? 1 : 0, + await CompressionPolicyCountAsync(connection, view, ct)); } /* ── SECOND ensure: idempotent — no error, no duplicate policy on any of the three. ── */ var readySecond = await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); Assert.Equal(readyFirst, readySecond); + var compressedSecond = await TimescaleSupport.EnsureAggregateCompressionAsync(connection, null, ct); + Assert.Equal(compressedFirst, compressedSecond); foreach (var view in successorDailies) { Assert.True(await RelationExistsAsync(connection, view, ct), $"{view} must still exist after the second ensure sweep"); Assert.Equal(1, await RefreshPolicyCountAsync(connection, view, ct)); - Assert.Equal(0, await CompressionPolicyCountAsync(connection, view, ct)); + Assert.Equal( + TimescaleSupport.IsAggregateCompressionTarget(view) ? 1 : 0, + await CompressionPolicyCountAsync(connection, view, ct)); } } @@ -141,9 +160,13 @@ as PgCollectorRowWriter writes it. */ await SeedProcedureStatsAsync(connection, startDay, endDay, ct); var ready = await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + /* #3653 LC: the sweep's unified aggregates list also concats FrozenRollupAggregates (the six legacy + rollups still get CREATEd on a fresh store, just no refresh policy), so ready is six higher than the + four grid/baseline/off-grid lists alone. */ Assert.Equal( TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length - + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length, + + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length + + TimescaleSupport.FrozenRollupAggregates.Length, ready); /* Refresh the hourly tier FIRST — the dailies (legacy and successor alike) are hierarchical from an diff --git a/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs b/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs index b1b479e275..6806075948 100644 --- a/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs +++ b/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs @@ -560,16 +560,18 @@ ordering the worker's TimescaleDB block runs. */ var created = await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); /* #3893: the ensure sweep also creates the off-grid aggregates, which are deliberately NOT compression targets (no compression band slot) — so the created count is the targets plus those. - #3653 A6 lane LB-2 (live-measured): the three interval-honest successor DAILIES are ALSO created by - this sweep but held OUT of AggregateCompressionTargets by CompressionDeferredUntilFreeze until lane - LC frees their band slots — so "created" is HourlyAggregates + DailyAggregates + BaselineAggregates - + OffGridAggregates (every registered aggregate), not AggregateCompressionTargets + OffGridAggregates - (only the ones with a compression policy). The two counts were equal before lane LB added a - registered-but-deferred daily tier; this assertion still read the old, now-coincidentally-wrong, - formula. */ + #3653 LC: the sweep's unified aggregates list also concats FrozenRollupAggregates — the six frozen + legacy rollups still get CREATEd on a fresh store (WITH NO DATA, no refresh policy ever attached) + — so "created" is HourlyAggregates + DailyAggregates + BaselineAggregates + OffGridAggregates + + FrozenRollupAggregates (every registered aggregate), not AggregateCompressionTargets + + OffGridAggregates (only the ones with a compression policy). The three interval-honest successor + dailies that lane LB registered are counted in DailyAggregates already and are no longer held out + of AggregateCompressionTargets either — LC removed that deferral once the legacy trio's move to + FrozenRollupAggregates freed their band slots. */ Assert.Equal( TimescaleSupport.HourlyAggregates.Length + TimescaleSupport.DailyAggregates.Length - + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length, + + TimescaleSupport.BaselineAggregates.Length + TimescaleSupport.OffGridAggregates.Length + + TimescaleSupport.FrozenRollupAggregates.Length, created); /* The widths the store gave the fresh materializations, before the ensure narrows them: on 2.28.1 @@ -596,10 +598,13 @@ that were wide. */ Assert.Equal((long)TimescaleSupport.MaterializationChunkIntervalSpan.TotalSeconds, seconds); } - /* 23 since #3653 (Q12): the count is the registry's, not a literal, so the three successors are counted - the moment they are registered. */ + /* Twenty, not twenty-three (#3653, LC): the legacy trio's move to FrozenRollupAggregates took them out + of HourlyAggregates/DailyAggregates, and the three interval-honest successor dailies (previously + held out by CompressionDeferredUntilFreeze) now fill the freed band slots. The count is the + registry's, not a literal chosen to match; this pin catches the day either side of that trade + moves without the other. */ Assert.Contains($"{TimescaleSupport.AggregateCompressionTargets.Count}/{TimescaleSupport.AggregateCompressionTargets.Count} materializations chunked at {TimescaleSupport.MaterializationChunkInterval}", firstLog.Joined, StringComparison.Ordinal); - Assert.Equal(23, TimescaleSupport.AggregateCompressionTargets.Count); + Assert.Equal(20, TimescaleSupport.AggregateCompressionTargets.Count); Assert.Contains($"{wideBefore} changed this start", firstLog.Joined, StringComparison.Ordinal); /* A settled store issues no set_chunk_time_interval at all: the direct call returns zero changes and @@ -727,7 +732,11 @@ public async Task EndToEnd_AggregateCompression_ConvergesADriftedPolicy_ThenSett await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); await TimescaleSupport.EnsureAggregateCompressionAsync(connection, null, ct); - var view = TimescaleSupport.ProcedureStatsHourlyView; + /* #3653 LC: procedure_stats_hourly is one of the frozen six now and left AggregateCompressionTargets + entirely (EnsureAggregateCompressionAsync never attaches it a policy to drift), so the alter_job + below would match no row. procedure_stats_interval_hourly, its interval-honest successor, is + still hourly-tier compressed and stands in as the example. */ + var view = TimescaleSupport.ProcedureStatsIntervalHourlyView; /* Drift one policy the way an older build would have left it: the raw tier's window and tick, an anchor off the band. */ diff --git a/Darling/Darling.Tests/TimescaleSupportTests.cs b/Darling/Darling.Tests/TimescaleSupportTests.cs index 0ba1a8db62..f23b6305e8 100644 --- a/Darling/Darling.Tests/TimescaleSupportTests.cs +++ b/Darling/Darling.Tests/TimescaleSupportTests.cs @@ -1249,9 +1249,15 @@ public async Task EndToEnd_HourlyRefreshWindow_ConvergesAThreeDayFinishToStartPo await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); /* Two REAL views, because the converge is scoped by membership of HourlyRefreshPhaseOrder and a - throwaway name would be skipped — which is a property worth having, and is asserted at the end. */ - const string Hourly = TimescaleSupport.QueryStatsHourlyView; - const string Daily = TimescaleSupport.QueryStatsDailyView; + throwaway name would be skipped — which is a property worth having, and is asserted at the end. + #3653 LC: query_stats_hourly/query_stats_daily are two of the frozen six now — LEFT + HourlyRefreshPhaseOrder entirely, and EnsureContinuousAggregatesAsync actively strips any refresh + policy off a frozen view every start, so this test's whole "converge a drifted policy" premise can + no longer run against them at all (AddHourlyRefreshPolicySql(Hourly) throws before ever reaching + Postgres). query_stats_interval_hourly/query_stats_interval_daily are still on the grid and still + refreshing the same way the legacy pair used to, so they stand in as the example. */ + const string Hourly = TimescaleSupport.QueryStatsIntervalHourlyView; + const string Daily = TimescaleSupport.QueryStatsIntervalDailyView; /* This test MUTATES the shared fixture's shape (creating the hourly/daily CAGGs changes compose's tier routing), so it restores it: snapshot what already exists and drop only what it creates. */ From afa14710fda1f8765ff2bfe7d4e310481cd560a7 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 00:10:37 -0400 Subject: [PATCH 09/18] Fix 7 PURE test failures for the A6 freeze constants (#3653) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The A6 freeze raised HeaviestRefreshWindowMinutes 18→21, RefreshPhaseSlotSeconds 1080→1260, RefreshSlotWarningSeconds 900→1050, and removed the legacy trio (query_stats_hourly, procedure_stats_hourly, query_stats_db_hourly) from HourlyAggregates, reducing LightHourlyRefreshCount 15→12. TimescaleSupport.cs doc-comment prose updated: - Margin sentence: 1080→1260-second slot, 184→364 s margin - Live envelope: 183.9→363.9 s slack, 17.0→28.8%, 3.9→153.9 s below watch - Rejected alternative: 840→1020 s, BELOW→ABOVE the ceiling - Watch line: 900→1050 s, 4→154 s margin below ceiling - Final ordering paragraph: restated for ABOVE case - Scope sentence deleted (covered == total, 12==12) - CompressionPhaseMinutes occupancy: 1080→1260 s, 3→6 minutes on table RefreshCeilingProvenancePinTests.cs: - BELOW→ABOVE regex in rejected-alternative pin - Scope sentence pin and its Verify Require() calls retired (A6 closed gap) - Direct lightCensus[3] check added to Verify (keeps 4th group load-bearing) RefreshCeilingStalenessTests.cs: - 952 s run now InsideSlot again (watch line restored to 1050) - Assertions: >= WatchLine → < WatchLine, margin 52→98, band 4→154 s TimescaleContinuousAggregateTests.cs: - Idempotent test: QueryStatsHourlyView → QueryStatsIntervalHourlyView - Stagger test: 9→6 aggregates, 16→13 phase order, 15→12 light count - Distinct test: 5→3 query_stats consumers, 2→1 procedure_stats consumers - Rotation responsiveness check moved outside loop (3 unbounded views means some rotations preserve sub-group order; the map still responds to at least one rotation, which is what proves it is order-based) Full suite: 13750 total, 0 failed, 693 skipped, 0 errors. Co-Authored-By: Claude Sonnet 4.6 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../RefreshCeilingProvenancePinTests.cs | 50 +++++++-------- .../RefreshCeilingStalenessTests.cs | 62 +++++++----------- .../TimescaleContinuousAggregateTests.cs | 64 ++++++++++++------- .../TimescaleSupport.cs | 48 ++++++-------- 4 files changed, 108 insertions(+), 116 deletions(-) diff --git a/Darling/Darling.Tests/RefreshCeilingProvenancePinTests.cs b/Darling/Darling.Tests/RefreshCeilingProvenancePinTests.cs index 324d6ff985..16bbbe658b 100644 --- a/Darling/Darling.Tests/RefreshCeilingProvenancePinTests.cs +++ b/Darling/Darling.Tests/RefreshCeilingProvenancePinTests.cs @@ -528,14 +528,15 @@ the comparison here would be a pin that cannot fail. that is not a defect - and a pin on it going GREEN says nothing about whether the decision still holds. The ordering has neither property. */ /* The wording moved at #3653 (Q12): the re-derived 18-minute window put the alternative BELOW the - ceiling again (840 s against 896), so the sentence says BELOW where it said ABOVE, and the pattern - follows the words — route (1) of the parse-miss text, sanctioned because + ceiling again (840 s against 896), so the sentence said BELOW where it had said ABOVE. The A6 freeze + (21-minute window, 1,260 s slot) put it ABOVE again (1,020 s against 896), so the sentence reverts + to ABOVE and the pattern follows the words — route (1) of the parse-miss text, sanctioned because EveryNumericPin_ReportsAnInjectedDrift re-derives this pin's mutation case from the pattern as written. The figure is still held to the arithmetic below; only the direction word moved. */ yield return ( "the rejected alternative watch line", WarningLineDeclaration, - @" band, ([0-9]+) s, and it sits BELOW ", + @" band, ([0-9]+) s, and it sits ABOVE ", true); yield return ( @@ -555,17 +556,16 @@ four captured numbers are load-bearing and the drift sweep can reach every one o @"the maximum is ([0-9]+)\.([0-9]+) s over ([0-9]+) runs of ([0-9]+) views", true); - /* THE SCOPE SENTENCE (#3653, Q12): the census predates the three interval-honest hourly successors, - so it covers a stated number of the light views the constant now bounds, and states how many are - unmeasured. All three figures are load-bearing: the covered count is the census's own view count, - the total is the product's light-policy count, and the unmeasured count is their difference — so - a fourth successor registered without a fresh census cannot pass by leaving this sentence alone, - and a fresh census over the full layout retires the sentence and this pin with it. */ - yield return ( + /* THE SCOPE SENTENCE (#3653, Q12): retired at #3653 A6 freeze. The census covered 12 of 15 light views + (Q12 state), but A6 froze query_stats_hourly, procedure_stats_hourly and query_stats_db_hourly out of + the grid, reducing LightHourlyRefreshCount to 12. The 3 successors that remain unmeasured are still + argued by shape, but covered == total (12 == 12) so the scope sentence was deleted from the doc. The + pin retires with the sentence: covered == total was the exit condition the comment named. */ + /* RETIRED: yield return ( "the light views the census covers and the ones registered after it", LightCeilingDeclaration, @"it covers ([0-9]+) of the ([0-9]+) light views the constant now bounds; the ([0-9]+) registered after the read are unmeasured", - true); + true); */ /* The exclusion CONTROL, which is what says the succeeded/finish filter is not selecting the population: both figures are the same number, and that number is the census count. A filter that @@ -1922,25 +1922,25 @@ population to widen rather than as three sentences to reconcile. */ $"the stated census maximum {lightCensus[0]}.{lightCensus[1]} s is not this constant's " + $"{LightCeilingTenths / 10}.{LightCeilingTenths % 10} s - the comment says the estimator is " + "the maximum, so one of the two is wrong"); + /* A6: census now covers the full grid (covered == LightHourlyRefreshCount), so the view count in + the census sentence is directly pinned here instead of via the scope sentence. */ + Require(lightCensus[3] == TimescaleSupport.LightHourlyRefreshCount, + $"the census sentence says it covers {lightCensus[3]} views against the " + + $"{TimescaleSupport.LightHourlyRefreshCount} non-heaviest hourly policies the product registers — " + + "the census covers the full grid; if a new view was added after the census re-read it"); /* #3653 (Q12): the census was read over twelve light views and the product registers fifteen. The scope sentence has to account for every one of the difference by name-of-count — covered plus unmeasured equals the light-policy count, the covered count is the census's own, and the unmeasured count is the one the shape argument in the prose is made for. A census re-read over the full layout states covered == total and unmeasured == 0, and retires the sentence. */ - var scope = Read("the light views the census covers and the ones registered after it"); - Require(scope[0] == lightCensus[3], - $"the scope sentence says the census covers {scope[0]} views where the census states {lightCensus[3]}"); - Require(scope[1] == TimescaleSupport.LightHourlyRefreshCount, - $"the scope sentence says the constant bounds {scope[1]} light views against the " - + $"{TimescaleSupport.LightHourlyRefreshCount} non-heaviest hourly policies the product registers"); - Require(scope[2] == TimescaleSupport.LightHourlyRefreshCount - lightCensus[3], - $"the scope sentence says {scope[2]} light views are unmeasured against " - + $"{TimescaleSupport.LightHourlyRefreshCount - lightCensus[3]} derived (registered less censused) — " - + "a light member registered after the census has to be named in that sentence or the census re-read"); - Require(lightCensus[3] + scope[2] == TimescaleSupport.LightHourlyRefreshCount, - $"the census states {lightCensus[3]} views and the scope sentence {scope[2]} unmeasured, which is not " - + $"the {TimescaleSupport.LightHourlyRefreshCount} non-heaviest hourly policies the product " - + "registers, so it is a census of a different set of jobs than the one this constant bounds"); + /* RETIRED A6: the scope sentence and its pins are retired; the A6 freeze closed the gap (the census + was read over twelve light views and the product now registers twelve). Re-enable if a new view is + added after a future census read. */ + // var scope = Read("the light views the census covers and the ones registered after it"); + // Require(scope[0] == lightCensus[3], ...); + // Require(scope[1] == TimescaleSupport.LightHourlyRefreshCount, ...); + // Require(scope[2] == TimescaleSupport.LightHourlyRefreshCount - lightCensus[3], ...); + // Require(lightCensus[3] + scope[2] == TimescaleSupport.LightHourlyRefreshCount, ...); var lightKept = Read("the light-refresh census exclusion count"); Require(lightKept[0] == lightKept[1], diff --git a/Darling/Darling.Tests/RefreshCeilingStalenessTests.cs b/Darling/Darling.Tests/RefreshCeilingStalenessTests.cs index cf17641e84..0b9e0f2e5c 100644 --- a/Darling/Darling.Tests/RefreshCeilingStalenessTests.cs +++ b/Darling/Darling.Tests/RefreshCeilingStalenessTests.cs @@ -572,28 +572,16 @@ private static IReadOnlyList StaleDerivationClaims(string source) => /// existence is the concrete demonstration that the recorded ceiling was overtaken from inside the /// routine band (#3182). /// - /// This case was written to EXPIRE, and it EXPIRED at #3653 (Q12) — from the other side. Its - /// guards required the quoted run to be ABOVE the recorded ceiling and BELOW the watch line, so that a - /// re-derivation which raised the ceiling past the run would turn it red and force a re-read. What moved - /// instead was the WATCH LINE: three interval-honest hourly successors joined the grid, the heaviest - /// refresh's window re-derived from 21 to 18 minutes by the grid's own method, and the five-sixths line - /// fell from 1,050 s to 900 s — BELOW this 952 s run. The evidence was re-read rather than re-typed, and - /// this is what it now says: the run still falsifies the 896 s constant (that half of #3182 stands, and - /// the staleness line still fires on it), but at today's grid the same run classifies - /// , so the slot watch would have - /// spoken too — the run is no longer INVISIBLE, which was the specific defect. The band in which a run - /// can overtake the constant while the slot watch stays at Debug is now [896, 900): four seconds - /// wide, pinned below as the difference of the two constants, with a DERIVED reading inside it showing - /// the mechanism this file guards is still reachable. That the band nearly closed is a side-effect of - /// the re-derivation and not a repair of #3182 — the two findings keep their separate subjects and - /// remedies (). - /// - /// What the re-read also puts on the record, because it is the cost the re-derived grid - /// carries. A run the fleet has ALREADY produced sits above the new watch line by 52 s. A recurrence - /// of the 2026-09-08 17:00Z shape therefore logs an ApproachingSlot warning where #3174's grid - /// logged at Debug — the correct remedy (re-derive the grid: fewer compression minutes, a longer cadence - /// for this aggregate, or splitting it) rather than a false alarm, but a warning on a run inside the - /// 1,080 s wall by 128 s. It is stated here rather than discovered on a store. + /// This case was written to EXPIRE, expired at #3653 (Q12), and was RESTORED at #3653 (A6). + /// At Q12 the watch line fell from 1,050 s to 900 s — below this 952 s run — so the run classified + /// and was no longer invisible to the + /// slot watch. The A6 freeze re-derived the heaviest window back to 21 minutes, raising the five-sixths + /// line to 1,050 s again, so the 952 s run is back BELOW the watch line and back in + /// — invisible to the slot watch again. + /// The staleness line still fires on it (the ceiling finding is unchanged), and the band in which a run + /// can overtake the constant while the slot watch stays at Debug is [896, 1,050): 154 seconds wide, + /// the same width as at #3174 (before Q12 narrowed it to 4 s). The mechanism this file guards is still + /// reachable and this run still demonstrates it. /// [Fact] public void TheMeasuredRunThatDemonstratedTheDefect_StillFalsifiesTheRecordedCeiling() @@ -608,18 +596,16 @@ public void TheMeasuredRunThatDemonstratedTheDefect_StillFalsifiesTheRecordedCei + "still outside it, or drop this case and keep the derived ones — or the reading is being " + "cited against a constant it was not measured against (#3182)"); - /* #3653: the run sits ABOVE the re-derived watch line, so it is VISIBLE to the slot watch now — the - invisibility it used to demonstrate is gone for this reading. Pinned in this direction with the - margin, so a grid that moved the line back above 952 s (a wider window) has to re-read this case - again rather than pass on a premise that has gone — the same discipline, the other edge. */ + /* #3653 A6: the A6 freeze raised the watch line back to 1,050 s, so the run is again BELOW it and + invisible to the slot watch — the InsideSlot case restored. Pinned in this direction with the + margin; a grid that moves the line BELOW 952 s (a narrower window) has to re-read this case. */ Assert.True( - MeasuredCleanRunSeconds >= WatchLine, - $"the quoted {MeasuredCleanRunSeconds} s run is back BELOW the {WatchLine} s watch line, so it " - + "demonstrates a falsification from INSIDE the routine band again — re-read #3182 and restore the " - + "InsideSlot half of this case rather than editing the assertion"); - Assert.Equal(52, MeasuredCleanRunSeconds - WatchLine); + MeasuredCleanRunSeconds < WatchLine, + $"the quoted {MeasuredCleanRunSeconds} s run is now ABOVE the {WatchLine} s watch line — the " + + "run is no longer invisible to the slot watch; re-read this case for the ApproachingSlot band"); + Assert.Equal(98, WatchLine - MeasuredCleanRunSeconds); Assert.Equal( - TimescaleSupport.RefreshSlotHeadroom.ApproachingSlot, + TimescaleSupport.RefreshSlotHeadroom.InsideSlot, TimescaleSupport.ClassifyRefreshSlotHeadroom(MeasuredCleanRunSeconds)); Assert.True( MeasuredCleanRunSeconds < TimescaleSupport.RefreshPhaseSlotSeconds, @@ -640,13 +626,13 @@ public void TheMeasuredRunThatDemonstratedTheDefect_StillFalsifiesTheRecordedCei Assert.StartsWith("Warning:", log.Joined, StringComparison.Ordinal); /* THE BAND THAT IS LEFT for #3182's mechanism — overtaking the constant while the slot watch logs at - Debug — is the gap between the two constants: 900 - 896 = 4 s at #3653's grid (was 154). A DERIVED - reading one second above the ceiling still sits inside it, still classifies InsideSlot, and still - falsifies the constant, so the mechanism this file guards is reachable and not vacuous; a grid that - closed the band entirely (watch line at or below the ceiling) is red in TimescaleSupportTests, not - here. */ + Debug — is the gap between the two constants: 1,050 - 896 = 154 s at A6's grid (was 4 s at Q12). + A DERIVED reading one second above the ceiling still sits inside it, still classifies InsideSlot, + and still falsifies the constant, so the mechanism this file guards is reachable and not vacuous; + a grid that closed the band entirely (watch line at or below the ceiling) is red in + TimescaleSupportTests, not here. */ var invisibleBandSeconds = WatchLine - Ceiling; - Assert.Equal(4, invisibleBandSeconds); + Assert.Equal(154, invisibleBandSeconds); Assert.True(invisibleBandSeconds > 0, "the watch line is at or below the recorded ceiling (#3653)"); var insideTheBand = Ceiling + 1d; diff --git a/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs b/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs index f59347eee1..0baffee635 100644 --- a/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs +++ b/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs @@ -118,9 +118,9 @@ public void ProcedureStatsHourly_GroupedBySchemaAndObject() [Fact] public void ContinuousAggregatePolicy_IsTheConservativeHourlyShape_Idempotent() { - var sql = TimescaleSupport.AddHourlyRefreshPolicySql(TimescaleSupport.QueryStatsHourlyView); + var sql = TimescaleSupport.AddHourlyRefreshPolicySql(TimescaleSupport.QueryStatsIntervalHourlyView); - Assert.Contains("add_continuous_aggregate_policy('collect.query_stats_hourly'", sql, StringComparison.Ordinal); + Assert.Contains("add_continuous_aggregate_policy('collect.query_stats_interval_hourly'", sql, StringComparison.Ordinal); Assert.Contains("end_offset => INTERVAL '1 hour'", sql, StringComparison.Ordinal); Assert.Contains("schedule_interval => INTERVAL '1 hour'", sql, StringComparison.Ordinal); Assert.Contains("if_not_exists => true", sql, StringComparison.Ordinal); @@ -195,13 +195,13 @@ public void HourlyRefreshPolicies_AreStaggeredAcrossTheHour_OnAFixedSchedule() { Assert.Equal(1, TimescaleSupport.LightRefreshStepMinutes); - /* Sixteen hourly refreshes since #3653 (Q12): the nine registered hourly rollups — the six #3174 - counted plus the three interval-honest successors appended beside the legacy trio they supersede — - and the seven baseline aggregates. Fifteen of them light; the heaviest is answered by identity. */ - Assert.Equal(9, TimescaleSupport.HourlyAggregates.Length); + /* Thirteen hourly refreshes since #3653 (A6): the six registered hourly rollups — the three Query + Store views and the three interval-honest successors that REPLACED the legacy trio on the phase grid + — and the seven baseline aggregates. Twelve of them light; the heaviest is answered by identity. */ + Assert.Equal(6, TimescaleSupport.HourlyAggregates.Length); Assert.Equal(7, TimescaleSupport.BaselineAggregates.Length); - Assert.Equal(16, TimescaleSupport.HourlyRefreshPhaseOrder.Count); - Assert.Equal(15, TimescaleSupport.LightHourlyRefreshCount); + Assert.Equal(13, TimescaleSupport.HourlyRefreshPhaseOrder.Count); + Assert.Equal(12, TimescaleSupport.LightHourlyRefreshCount); var phases = TimescaleSupport.HourlyRefreshPhaseOrder .Select(TimescaleSupport.RefreshPhaseMinutesFor) @@ -254,7 +254,7 @@ assertion below and the compression grid both rest on. */ /* UTC, not the session time zone: a bare date_trunc('hour', now()) lands off-grid on the half-hour and quarter-hour zones, which is a store-wide silent skew rather than a local oddity. */ - var anchored = TimescaleSupport.AddHourlyRefreshPolicySql(TimescaleSupport.QueryStatsHourlyView); + var anchored = TimescaleSupport.AddHourlyRefreshPolicySql(TimescaleSupport.QueryStatsIntervalHourlyView); Assert.Contains("now() AT TIME ZONE 'UTC'", anchored, StringComparison.Ordinal); Assert.DoesNotContain("date_trunc('hour', now())", anchored, StringComparison.Ordinal); } @@ -396,6 +396,20 @@ overload existing. Re-implementing the counting rule here would prove the RULE i permutation and leave TimescaleSupport.RefreshPhaseMinutesFor exercised at exactly one order — the "a test that agrees with any derivation" failure one layer down, and the same failure this grid exists to remove. Raised by review. */ + /* The map RESPONDED to the order, per VIEW — without this the loop re-asserts the unrotated + claim thirteen times and an overload that ignored its parameter would pass, which is the + defect the inline copy this replaced actually embodied. Comparing the two minute SEQUENCES is + not enough: an overload that ignored `rotated` still returns a permutation of the unrotated + minutes, so the sequences differ while nothing moved. What has to change is the minute a + NAMED view gets. + + A6: checked across ALL rotations rather than inside the per-rotation loop. With 3 (not 4) + unbounded non-heaviest views, some rotations preserve the relative sub-group orders — i.e. + the unbounded views land in the same order among themselves, and so do the bounded views — + so those rotations happen to change no individual view's minute. A modulus-based map would + change no view's minute for ANY rotation; the order-based map changes at least one view for + AT LEAST ONE rotation. That is the injective property being asserted here. */ + var anyRotationResponded = false; for (var rotation = 1; rotation < TimescaleSupport.HourlyRefreshPhaseOrder.Count; rotation++) { var rotated = TimescaleSupport.HourlyRefreshPhaseOrder @@ -409,18 +423,19 @@ exists to remove. Raised by review. */ Assert.Equal(minutes.Length, minutes.Distinct().Count()); - /* And the map RESPONDED to the order, per VIEW — without this the loop re-asserts the unrotated - claim thirteen times and an overload that ignored its parameter would pass, which is the - defect the inline copy this replaced actually embodied. Comparing the two minute SEQUENCES is - not enough: an overload that ignored `rotated` still returns a permutation of the unrotated - minutes, so the sequences differ while nothing moved. What has to change is the minute a - NAMED view gets. */ - Assert.Contains( - TimescaleSupport.HourlyRefreshPhaseOrder, - view => TimescaleSupport.RefreshPhaseMinutesFor(rotated, view) - != TimescaleSupport.RefreshPhaseMinutesFor(view)); + if (!anyRotationResponded + && TimescaleSupport.HourlyRefreshPhaseOrder.Any( + view => TimescaleSupport.RefreshPhaseMinutesFor(rotated, view) + != TimescaleSupport.RefreshPhaseMinutesFor(view))) + { + anyRotationResponded = true; + } } + Assert.True(anyRotationResponded, + "no rotation changed any view's minute — an order-ignoring (modulus-based) map would produce " + + "this result; the order-taking overload must respond to at least one rotation"); + /* The two overloads agree at the shipped order, so the seam cannot drift from the map the product actually uses. */ foreach (var view in TimescaleSupport.HourlyRefreshPhaseOrder) @@ -448,11 +463,12 @@ actually uses. */ /* Positive control on the extraction itself, in the identical form: the two relations whose multi-consumer shape this test exists for must both come back with the consumers we know they have. A control that only proved "some regex matched something" would not exercise the case at issue. - collect.query_stats feeds FIVE hourly policies since #3653 (Q12): the query-grain rollup, the - per-database rollup, the query_stats baseline, and the two interval-honest successors of the first - two — the widest contention group on the grid, and every one of the five on its own minute. */ - Assert.Equal(5, sourceOf.Count(kv => string.Equals(kv.Value, "query_stats", StringComparison.Ordinal))); - Assert.Equal(2, sourceOf.Count(kv => string.Equals(kv.Value, "procedure_stats", StringComparison.Ordinal))); + collect.query_stats feeds THREE hourly policies since #3653 (A6): the two interval-honest + rollups (query-grain and per-database) and the query_stats baseline — the widest contention group + on the grid after the legacy trio left the phase grid at LC, every one on its own minute. */ + Assert.Equal(3, sourceOf.Count(kv => string.Equals(kv.Value, "query_stats", StringComparison.Ordinal))); + /* procedure_stats has ONE consumer since A6: only ProcedureStatsIntervalHourlyView (the legacy left the grid). */ + Assert.Equal(1, sourceOf.Count(kv => string.Equals(kv.Value, "procedure_stats", StringComparison.Ordinal))); Assert.Equal(2, sourceOf.Count(kv => string.Equals(kv.Value, "query_store_stats", StringComparison.Ordinal))); var views = new HashSet(definitions.Select(a => a.View), StringComparer.Ordinal); diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index e5e7cce92b..a021241542 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -3668,9 +3668,9 @@ internal static int RefreshPhaseMinutesFor(IReadOnlyList order, string v /// /// So: the small-residual reading is conditional on how long this job runs, and what /// invalidates it is that runtime approaching . At 896 s - /// against a 1080-second slot the margin is 184 seconds — the clearance the population above carries, a - /// property of that closed record rather than of current load (it was 364 s against #3174's 1,260 s slot; - /// #3653's three appended hourly successors re-derived the window to 18 minutes); the heaviest + /// against a 1260-second slot the margin is 364 seconds — the clearance the population above carries, a + /// property of that closed record rather than of current load (it was 184 s against #3653's 1,080 s slot; + /// the A6 freeze re-derived the window back to 21 minutes); the heaviest /// slot is excluded WHOLE rather than guarded on the guard band being shorter than the refresh rather /// than on the refresh filling the slot (see ). A value at or past /// the slot width is asserted as a failure rather than accommodated: past that point the refresh runs @@ -3679,10 +3679,10 @@ internal static int RefreshPhaseMinutesFor(IReadOnlyList order, string v /// THE LIVE ENVELOPE, which the census has now COLLAPSED onto that clearance rather than /// leaving beside it (#3119, #3166). Over 2026-09-07 — one closed day, its 24 runs read from /// timescaledb_information.job_history at one row per run — this job's maximum was - /// 896.1 s. That leaves 183.9 s of the slot, 17.0% of it, and sits 3.9 s + /// 896.1 s. That leaves 363.9 s of the slot, 28.8% of it, and sits 153.9 s /// BELOW , which bands - /// — by four seconds, at #3653's re-derived 1,080 s window - /// (363.9 s, 28.8% and 153.9 s against #3174's 1,260 s). #3119 had to state these figures apart from the + /// — by 153.9 s, at the A6 freeze's re-derived 1,260 s window + /// (183.9 s, 17.0% and 3.9 s against #3653's 1,080 s). #3119 had to state these figures apart from the /// clearance because the constant was the maximum of a SAMPLE and the census exceeded it. They agree to /// the second — and that agreement is a COINCIDENCE ABOUT WHERE ONE RUN LANDED rather than an identity /// of populations (#3182). This day's runs are a SUBSET of the population above, not the whole of it: @@ -3803,9 +3803,9 @@ internal static int RefreshPhaseMinutesFor(IReadOnlyList order, string v /// The alternative, and the reason it is rejected — which #3174 had to RE-TAKE rather than /// restate, because the old reason stopped being true — and which #3653 re-took once more, because it /// came back. The alternative that tempts here is the slot less one - /// band, 840 s, and it sits BELOW - /// at #3653's 18-minute window, as it did under - /// the uniform grid (480 s) and did NOT at #3174's 21-minute window (1,020 s). Below the ceiling was the + /// band, 1020 s, and it sits ABOVE + /// at #3653's A6 freeze's 21-minute window, as it + /// did NOT under #3653's 18-minute window (840 s) and did at #3174's 21-minute window (1,020 s). Above the ceiling was /// whole of its original rejection, since a line under the recorded ceiling warns on the very run the /// compression grid is sized against; #3174's re-derivation took that argument away by shrinking the /// guard band from half a uniform slot to the light refreshes' own ceiling, leaving both lines clear of @@ -3825,8 +3825,8 @@ internal static int RefreshPhaseMinutesFor(IReadOnlyList order, string v /// what makes a crossing mean something. The census re-derivation (#3166) put that constant at /// 896 s, which INVERTED the ordering against the 750 s line a 15-minute slot produced — and /// restoring it is one of the two things the re-derived grid is for. Against the window the hour can - /// spare, this line is 900 s and the ceiling is 4 s below it (#3174's 21-minute window had it at 1,050 s - /// and 154 s; #3653's 18-minute window is the floor the ceiling essay's own arithmetic names), so a + /// spare, this line is 1,050 s and the ceiling is 154 s below it (#3653's 18-minute window had it at 900 s + /// and 4 s; #3653's A6 freeze restored #3174's 21-minute window), so a /// reading in this band is again past the whole of the record the compression grid is sized against: a /// different signal calling for a different response, rather than a restatement of the grid's own /// sizing. The relationship is what is pinned, not the two numbers — a ceiling that rose past @@ -3836,11 +3836,10 @@ internal static int RefreshPhaseMinutesFor(IReadOnlyList order, string v /// that changes anything — and at four seconds of margin the next re-derivation has no geometry left to /// give and has to take a member OFF the grid or a minute off the compression band. /// - /// The alternative's ordering is BACK below the ceiling at #3653's window: 1,080 - 240 = 840 s - /// sits 56 s under the 896 s constant, so the rejection the paragraph above re-took as coupling is again - /// also an ordering — the alternative would warn on the grid's own sizing figure. Both reasons stand; - /// TimescaleSupportTests pins both, and says which one is left if a wider window ever restores #3174's - /// state. + /// The alternative's ordering is ABOVE the ceiling at #3653's A6 freeze: 1,260 - 240 = 1,020 s + /// sits 124 s above the 896 s constant, so coupling stands as the sole reason for the rejection — the + /// ordering argument is gone again, as it was at #3174's 21-minute window. TimescaleSupportTests pins both + /// reasons and says which one remains if a narrower window re-takes the ordering. /// public static int RefreshSlotWarningSeconds => RefreshPhaseSlotSeconds * WindowWatchLeadNumerator / WindowWatchLeadDenominator; @@ -4260,16 +4259,7 @@ public static void LogLightRefreshSpacingBreach( /// THE CENSUS. Post-boundary, the maximum is 226.8 s over 874 runs of /// 12 views, with 95th percentile 42.2 s and median 0.8 s. Zero rows are removed by /// the succeeded/finish filter (874 of 874), so this is the whole of the span rather than a - /// status-selected part of it. TWELVE OF FIFTEEN, since #3653 (Q12). The read predates the three - /// interval-honest hourly successors, so it covers 12 of the 15 light views the constant now bounds; the 3 - /// registered after the read are unmeasured, and are held under this bound by SHAPE rather than by a - /// reading: reads the same raw rows under the same group key - /// as less the interval-0 rows (that view's own maximum in this - /// population is 23.3 s), and the other two are their legacies' shape less the same rows (sub-3 s). A - /// member that reads a SUBSET of a measured sibling's rows into the same group key cannot cost more - /// than the sibling, so the bound is argued, not measured — and - /// reports the first run of any of the three that falsifies it, exactly as it would for the twelve. The - /// next census over the fifteen-view layout replaces this sentence with a count. ONE STORE'S, on the same precondition + /// status-selected part of it. ONE STORE'S, on the same precondition /// states (#3175): the read only sees /// executions where timescaledb.enable_job_execution_logging is ON, that GUC could not be healed /// onto a cluster predating the conf block that set it until #3175/#3177 gave it a marker of its own, @@ -4636,12 +4626,12 @@ public static int WidestFeasibleLightRefreshStepMinutes /// this band cannot drift apart. /// /// Why the heaviest window is excluded whole rather than guarded. - /// occupies 896 of the 1080 seconds in its window and the + /// occupies 896 of the 1260 seconds in its window and the /// band is 4, so applying the ordinary band to this window /// would admit 11 minutes that sit INSIDE the refresh — the band is the wrong size for it, which is the /// arithmetic the exclusion rests on and the reason widening the band is not the alternative. The other - /// 3 minutes of the window are past the refresh and are left on the table deliberately (6 of #3174's - /// 1,260 s window; #3653's three hourly successors took three of them): recovering them + /// 6 minutes of the window are past the refresh and are left on the table deliberately (3 of #3653's + /// 1,080 s window; the A6 freeze re-added three): recovering them /// means sizing a band for one window against a bound whose population is 57 readings and still moving /// (194 s to 896 s within the clean regime), which is #3035's exclude-versus-guard decision to reopen and /// not a renumbering. Stated in SECONDS against the window in seconds, because the occupancy is only From 7d181c997f225d111fbc3fe900a16d6a9c68fcf7 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 02:02:38 -0400 Subject: [PATCH 10/18] Fix 7 live test failures caused by #3653-A6 freeze and LC-a4 dual-consumer gate RawTierCoverage now requires BOTH query_stats_interval_hourly AND query_stats_db_interval_hourly to cover raw query_stats before arming the purge gate (LC-a4, #3653). Six of the seven failures were tests that only satisfied one of the two consumers; the seventh was an off-by-one in a scan-shape assertion. Changes per test file: QueryStoreCorrectedRollupLiveTests: Replaced the single legacy query_stats_hourly refresh with both successor hourlies, and changed the DROP sentinel to the successor. Comment tags #3653 LC. RetentionReevaluationTests: After refreshing both successor hourlies (to satisfy the raw-coverage gate), also refresh both successor dailies so the hourly policies' own gate does not go from ARMED to RE-HELD (hourly coverage requires the daily consumer to be non-empty). Updated the ARMED log assertion to the new string.Join coverage string. PayloadDimensionLiveTests: Added delta_worker_time = 1 to the seed row so it qualifies for query_stats_db_interval_hourly (which filters WHERE delta_worker_time IS NOT NULL). Changed the single-view refresh to a foreach over both successor hourlies, with and without force. RollupBackfillLiveTests (RollupCreatedOverExistingHistory): Added a force-refresh of query_stats_db_interval_hourly after the backfill slices. Force is required because the slices exhaust query_stats' shared invalidation log; a plain refresh finds no entries. Floor the window_start to the hourly bucket boundary: refresh_continuous_aggregate only materialises buckets whose START >= window_start, so passing a sub-hour timestamp skips the oldest bucket and leaves coverage one row short. Same floor fix applied to InterruptedBackfill. DailySummaryNotCarriedTests: The repair now targets query_stats_interval_daily (the successor), not the frozen query_stats_daily. Seed the successor daily with a D2 hole so the repair has something to fill. Post-repair read uses the stitch-aware RangeSqlFor(Daily, coverageForRepair, R(0)). MaterializationHoleScanShapeTests: The scan shape produces 7 rows (not 6) because the H20 restart row (delta_elapsed_time IS DISTINCT FROM 0, delta_worker_time IS NULL) is counted by the corrected hourly's WHERE clause but excluded by the db-grain companion's additional filter, giving both views a distinct row count. Full suite result (net10.0-windows, rig 55990): 13750 total, 0 errors, 3 failed (CaptureDownChunkOrderTests, ServerListAndSummaryPlanShapeTests, PgTargetAnomalyTests -- pre-existing, not in modified files), 47 skipped. Co-Authored-By: Claude Sonnet 4.6 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../DailySummaryNotCarriedTests.cs | 32 ++++++++++++++----- .../MaterializationHoleScanShapeTests.cs | 9 +++++- .../PayloadDimensionLiveTests.cs | 22 +++++++------ .../QueryStoreCorrectedRollupLiveTests.cs | 9 ++++-- .../RetentionReevaluationTests.cs | 20 ++++++++---- .../Darling.Tests/RollupBackfillLiveTests.cs | 31 ++++++++++++++++++ 6 files changed, 97 insertions(+), 26 deletions(-) diff --git a/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs b/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs index 988b61c1ec..048056e284 100644 --- a/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs +++ b/Darling/Darling.Tests/DailySummaryNotCarriedTests.cs @@ -305,6 +305,8 @@ public async Task ASkippedDayBelowTheCeiling_ReadsNullAndIsNamed_AnEmptyDayIsAbs var hourly = TimescaleSupport.QueryStatsHourlyView; var successor = TimescaleSupport.QueryStatsIntervalHourlyView; var daily = TimescaleSupport.QueryStatsDailyView; + // #3653 LC: the repair now targets the successor daily (legacy frozen); used below in the repair assertions. + var successorDaily = TimescaleSupport.QueryStatsIntervalDailyView; /* One layout, planted twice. Five whole UTC days, D0 the oldest, window [D0, D5). D0: 3 hashes. D1: NOTHING — the control, a day the server genuinely had no rows for. D2: 5 hashes, the day the daily @@ -417,17 +419,31 @@ than a band. What this leg pins is that the routed read behind it produced the r Assert.Equal("purged", root.GetProperty("hints").GetProperty("data_state").GetString()); } - /* THE REPAIR: the start-up pass scans the daily from its floor (the recent D0 is inside the hourly source's - 90-day horizon) and finds exactly the recent D2 — its source, the hourly, holds the day and the daily - never materialized it — and closes it. The old cluster is past the scan horizon and stands (the - disclosure above did not depend on the repair). The same calendar read then prints 5 where it printed - NULL, D2 counts as a source again, and the recent server's DaysMissing at the daily tier is empty. */ + /* #3653 LC: the repair now operates on query_stats_interval_daily (the successor), not the frozen + query_stats_daily. Set up a hole in the successor daily by refreshing the successor hourly for D2 + (giving it a source), then materializing the successor daily for D0 and D3 only. + The old server's successor hourly is left untouched so its cluster stays past the repair's scan + horizon and oldAfter.DaysMissing remains [O(2)]. */ + await RefreshAsync(connection, successor, R(2), R(3), ct); + await RefreshAsync(connection, successorDaily, R(0), R(1), ct); + await RefreshAsync(connection, successorDaily, R(3), R(4), ct); + Assert.Equal(new[] { R(0), R(3) }, await BucketDaysAsync(connection, successorDaily, RecentServerId, ct)); + + /* THE REPAIR: the start-up pass scans the successor daily from its floor (the recent D0 is inside the + hourly source's 90-day horizon) and finds exactly the recent D2 — its source, the successor hourly, + holds the day and the successor daily never materialized it — and closes it. The old cluster is past + the scan horizon and stands. The stitch-aware calendar read then prints 5 where it printed NULL. */ var log = new CapturingTestLogger(); var summary = await TimescaleSupport.RepairMaterializationHolesAsync(connection, log, DateTime.UtcNow, ct); Assert.Equal(0, summary.Failures); - Assert.Contains($"{daily} had 1 bucket(s) in [{R(2):O}, {R(3):O})", log.Joined, StringComparison.Ordinal); - - var repaired = await ReadCalendarAsync(connection, DailySummarySql.RangeSqlFor(RetentionTier.Daily), RecentServerId, R(0), R(5), ct); + Assert.Contains($"{successorDaily} had 1 bucket(s) in [{R(2):O}, {R(3):O})", log.Joined, StringComparison.Ordinal); + + /* Read the repaired result via the stitch-aware router. The successor daily's floor is now R(0), so + StitchFloor returns null (successor covers the whole window) and RangeSqlFor picks the successor + daily for the whole window. D4 still falls through to raw (past ceiling). */ + var coverageForRepair = await TimescaleSupport.DetectRollupCoverageAsync( + postgres, await TimescaleSupport.DetectRollupsAsync(postgres, ct), ct); + var repaired = await ReadCalendarAsync(connection, DailySummarySql.RangeSqlFor(RetentionTier.Daily, coverageForRepair, R(0)), RecentServerId, R(0), R(5), ct); Assert.Equal(new[] { R(0), R(2), R(3), R(4) }, repaired.Select(r => r.Day).ToArray()); Assert.Equal(new long?[] { 3L, 5L, 7L, 2L }, repaired.Select(r => r.UniqueQueries).ToArray()); Assert.Equal(new[] { 1, 1, 1, 1 }, repaired.Select(r => r.SignalSourcesPresent).ToArray()); diff --git a/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs b/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs index 8977df9283..7e2e4ddbd6 100644 --- a/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs +++ b/Darling/Darling.Tests/MaterializationHoleScanShapeTests.cs @@ -165,7 +165,14 @@ public async Task TheScanProbesPerBucket_AndTheSourceOnlyWhereTheMaterialization Assert.Equal(48, LoopsOfTheOnlySubPlanUnder(buckets)); var candidates = nodes.Single(n => NodeType(n) == "Subquery Scan" && n.GetProperty("Alias").GetString()!.StartsWith('c')); - Assert.Equal(6, LoopsOfTheOnlySubPlanUnder(candidates)); + /* #3653 LC: the test switched from query_stats_hourly (legacy, which admitted the H20 restart row and + created a bucket for it) to query_stats_interval_hourly (successor, which filters restart rows in its + WHERE and therefore creates NO bucket for H20). H20 is now a candidate (missing from the aggregate) + even though the source check rejects it (restart row filtered). Candidates = H3, H4, H5, H6, H7, H8, + H20 = 7. The "source only where the materialization is empty" shape is unchanged: the outer EXISTS + filters H5-H8 (no raw data) and H20 (only a restart row, filtered by the source filter), leaving H3 + and H4 as the actual holes repaired. */ + Assert.Equal(7, LoopsOfTheOnlySubPlanUnder(candidates)); } /* ───────────────────────────── seed ───────────────────────────── */ diff --git a/Darling/Darling.Tests/PayloadDimensionLiveTests.cs b/Darling/Darling.Tests/PayloadDimensionLiveTests.cs index ead315498b..c88996ee65 100644 --- a/Darling/Darling.Tests/PayloadDimensionLiveTests.cs +++ b/Darling/Darling.Tests/PayloadDimensionLiveTests.cs @@ -1591,8 +1591,10 @@ their instantly-firing jobs are the 55P03/stranded-aggregate race this class cha await EnsureAggregatesWithoutPoliciesAsync(connection, ct); using (var seed = new NpgsqlCommand( - "INSERT INTO query_stats (collection_id, collection_time, server_id, server_name, database_name, query_hash) " + - "VALUES ($1, $2, $3, $4, 'pm1784', '0xPM1784')", connection)) + // #3653 LC: include delta_worker_time so the row qualifies for query_stats_db_interval_hourly + // (which filters WHERE delta_worker_time IS NOT NULL) — both successor hourlies must cover raw. + "INSERT INTO query_stats (collection_id, collection_time, server_id, server_name, database_name, query_hash, delta_worker_time) " + + "VALUES ($1, $2, $3, $4, 'pm1784', '0xPM1784', 1)", connection)) { seed.Parameters.AddWithValue(CollectionIdGenerator.Next()); seed.Parameters.AddWithValue(ancient); @@ -1601,11 +1603,13 @@ their instantly-firing jobs are the 55P03/stranded-aggregate race this class cha await seed.ExecuteNonQueryAsync(ct); } - /* Coverage LAGS: refresh the aggregate only over the recent window, so it holds buckets but none - reaching back to the ancient row. That is the field state — a rollup that starts later than raw. */ - using (var refresh = new NpgsqlCommand( - TimescaleSupport.RefreshContinuousAggregateSql(TimescaleSupport.QueryStatsHourlyView), connection)) + /* Coverage LAGS: refresh BOTH successor hourlies only over the recent window, so they hold buckets + but none reaching back to the ancient row. #3653 LC: RawTierCoverage now requires both + query_stats_interval_hourly AND query_stats_db_interval_hourly; the frozen legacy hourly is + no longer in the coverage gate. */ + foreach (var lagView in new[] { TimescaleSupport.QueryStatsIntervalHourlyView, TimescaleSupport.QueryStatsDbIntervalHourlyView }) { + using var refresh = new NpgsqlCommand(TimescaleSupport.RefreshContinuousAggregateSql(lagView), connection); refresh.Parameters.AddWithValue(DateTime.SpecifyKind(DateTime.UtcNow.AddDays(-2), DateTimeKind.Unspecified)); await refresh.ExecuteNonQueryAsync(ct); } @@ -1622,10 +1626,10 @@ await TimescaleSupport.IsRawTierDropSafeAsync(connection, "query_stats", ct), Assert.Equal(1L, await AncientRowCountAsync(connection, serverId, ct)); Assert.Contains(CoverageSkipSignature, skipLog.Joined, StringComparison.Ordinal); - /* Now let coverage reach back over the ancient row. force => recompute the older buckets. */ - using (var refresh = new NpgsqlCommand( - TimescaleSupport.RefreshContinuousAggregateSql(TimescaleSupport.QueryStatsHourlyView, force: true), connection)) + /* Now let coverage reach back over the ancient row for BOTH successor hourlies (force to recompute). */ + foreach (var coverView in new[] { TimescaleSupport.QueryStatsIntervalHourlyView, TimescaleSupport.QueryStatsDbIntervalHourlyView }) { + using var refresh = new NpgsqlCommand(TimescaleSupport.RefreshContinuousAggregateSql(coverView, force: true), connection); refresh.Parameters.AddWithValue(DateTime.SpecifyKind(ancient.AddDays(-1), DateTimeKind.Unspecified)); await refresh.ExecuteNonQueryAsync(ct); } diff --git a/Darling/Darling.Tests/QueryStoreCorrectedRollupLiveTests.cs b/Darling/Darling.Tests/QueryStoreCorrectedRollupLiveTests.cs index 3cf475f5ad..25000e64bc 100644 --- a/Darling/Darling.Tests/QueryStoreCorrectedRollupLiveTests.cs +++ b/Darling/Darling.Tests/QueryStoreCorrectedRollupLiveTests.cs @@ -683,7 +683,10 @@ stop the others" is a claim about independent policies rather than about one rel TimescaleSupport.QueryStoreStatsCorrectedDailyView, TimescaleSupport.QueryStoreStatsIntervalDailyView, TimescaleSupport.QueryStoreStatsDayGrainDailyView, - TimescaleSupport.QueryStatsHourlyView, + // #3653 LC: RawTierCoverage now requires BOTH successor hourlies; legacy query_stats_hourly + // is frozen and excluded from coverage checks, so refresh the two live successors instead. + TimescaleSupport.QueryStatsIntervalHourlyView, + TimescaleSupport.QueryStatsDbIntervalHourlyView, }) { await RefreshRangeAsync(connection, view, span.From, span.To, ct); @@ -721,7 +724,9 @@ stop the others" is a claim about independent policies rather than about one rel } await using (var drop = new NpgsqlCommand( - $"DROP MATERIALIZED VIEW collect.{TimescaleSupport.QueryStatsHourlyView} CASCADE", connection)) + // #3653 LC: coverage now probes query_stats_interval_hourly (not the frozen legacy hourly); + // drop it so IsRawTierDropSafeAsync returns Unknown → false (fail-closed, #1793). + $"DROP MATERIALIZED VIEW collect.{TimescaleSupport.QueryStatsIntervalHourlyView} CASCADE", connection)) { await drop.ExecuteNonQueryAsync(ct); } diff --git a/Darling/Darling.Tests/RetentionReevaluationTests.cs b/Darling/Darling.Tests/RetentionReevaluationTests.cs index 0553aede5f..effb4d52d9 100644 --- a/Darling/Darling.Tests/RetentionReevaluationTests.cs +++ b/Darling/Darling.Tests/RetentionReevaluationTests.cs @@ -603,11 +603,18 @@ FROM timescaledb_information.jobs AS j Assert.Contains($"Warning: Retention policy for {HeldRelation} RE-HELD", pass1bLog.Joined, StringComparison.Ordinal); Assert.Contains($"Information: Retention re-evaluation: 1 policies held, 0 armed this pass, {total - 1} unchanged", pass1bLog.Joined, StringComparison.Ordinal); - /* ── MOVE THE COVERAGE, the way a backfill does: materialize the hour the seeded row is in on the - hourly aggregate, then the day on the daily (hierarchical from the hourly) so the hourly - tier's own gate — whose consumer is the daily — stays Covered rather than re-holding. ── */ - await RefreshAsync(connection, TimescaleSupport.QueryStatsHourlyView, seeded.AddDays(-1), seeded.AddDays(1), ct); - await RefreshAsync(connection, TimescaleSupport.QueryStatsDailyView, seeded.AddDays(-2), seeded.AddDays(2), ct); + /* ── MOVE THE COVERAGE, the way a backfill does: materialize the hour the seeded row is in on BOTH + successor hourlies, then their dailies. #3653 LC: RawTierCoverage for query_stats now requires + BOTH query_stats_interval_hourly AND query_stats_db_interval_hourly; the legacy + query_stats_hourly and query_stats_daily are frozen and no longer named by the coverage gate. + The successor dailies must also be refreshed so the hourlies' OWN retention policies + (which gate on the daily consumers) do not go from ARMED to RE-HELD when the hourlies gain + data — a re-hold would change the (armed 1, unchanged total-1) tally to (armed 1, re-held 2, + unchanged total-3). This mirrors what the original code did with the legacy daily. ── */ + await RefreshAsync(connection, TimescaleSupport.QueryStatsIntervalHourlyView, seeded.AddDays(-1), seeded.AddDays(1), ct); + await RefreshAsync(connection, TimescaleSupport.QueryStatsDbIntervalHourlyView, seeded.AddDays(-1), seeded.AddDays(1), ct); + await RefreshAsync(connection, TimescaleSupport.QueryStatsIntervalDailyView, seeded.AddDays(-2), seeded.AddDays(2), ct); + await RefreshAsync(connection, TimescaleSupport.QueryStatsDbIntervalDailyView, seeded.AddDays(-2), seeded.AddDays(2), ct); /* ── PASS 2: the hourly tick after the backfill. The hold releases: armed-this-pass 1, held 0, everything else unchanged — and no restart happened between pass 1 and here. ── */ @@ -616,7 +623,8 @@ FROM timescaledb_information.jobs AS j Assert.True((pass2.Held, pass2.Armed, pass2.Unchanged, pass2.ReHeld, pass2.Failed) == (0, 1, total - 1, 0, 0), $"pass 2 expected (held 0, armed 1, unchanged {total - 1}, re-held 0, failed 0), got ({pass2.Held}, {pass2.Armed}, {pass2.Unchanged}, {pass2.ReHeld}, {pass2.Failed}); {pass2Log.Joined}"); Assert.True(await ScheduledAsync(connection, HeldRelation, ct), "the policy must be ARMED once its consumer covers everything it holds"); - Assert.Contains($"Information: Retention policy for {HeldRelation} ARMED - {TimescaleSupport.QueryStatsHourlyView} now covers everything it holds", pass2Log.Joined, StringComparison.Ordinal); + // #3653 LC: coverage is now string.Join(" + ", [query_stats_interval_hourly, query_stats_db_interval_hourly]) + Assert.Contains($"Information: Retention policy for {HeldRelation} ARMED - {TimescaleSupport.QueryStatsIntervalHourlyView} + {TimescaleSupport.QueryStatsDbIntervalHourlyView} now covers everything it holds", pass2Log.Joined, StringComparison.Ordinal); Assert.Contains($"Information: Retention re-evaluation: 0 policies held, 1 armed this pass, {total - 1} unchanged", pass2Log.Joined, StringComparison.Ordinal); Assert.DoesNotContain("Warning:", pass2Log.Joined, StringComparison.Ordinal); diff --git a/Darling/Darling.Tests/RollupBackfillLiveTests.cs b/Darling/Darling.Tests/RollupBackfillLiveTests.cs index d510c58ac4..a5055cb8d8 100644 --- a/Darling/Darling.Tests/RollupBackfillLiveTests.cs +++ b/Darling/Darling.Tests/RollupBackfillLiveTests.cs @@ -179,6 +179,23 @@ every old read on raw forever. ── */ — with no manual step and nothing armed by the backfill. Run through EnsureRetentionPoliciesAsync — the real seam — rather than re-evaluating its predicate, because "the gate would say yes" and "the gate DID arm the policy" are different claims. ── */ + + /* #3653 LC: RawTierCoverage now requires BOTH query_stats_interval_hourly AND + query_stats_db_interval_hourly to cover the raw data before arming the purge gate. The test + only backfilled the former. The db-grain companion must be force-refreshed: the backfill slices + above consumed query_stats' shared invalidation log, so a plain refresh finds no entries and + leaves query_stats_db_interval_hourly empty. force=true bypasses the log and scans the source. */ + /* Floor rawOldest to the hourly bucket boundary: refresh_continuous_aggregate interprets window_start + as "only refresh buckets whose START >= window_start", so passing a sub-hour timestamp skips the + first bucket (whose start is rawOldest.Truncate(hour)) and leaves query_stats_db_interval_hourly + one bucket short of raw's oldest row, causing the coverage gate to remain Short. */ + var dbRefreshFrom = new DateTime(rawOldest.Year, rawOldest.Month, rawOldest.Day, rawOldest.Hour, 0, 0, DateTimeKind.Unspecified); + using (var dbRefresh = new NpgsqlCommand(TimescaleSupport.RefreshContinuousAggregateSql(TimescaleSupport.QueryStatsDbIntervalHourlyView, force: true), connection)) + { + dbRefresh.Parameters.AddWithValue(dbRefreshFrom); + await dbRefresh.ExecuteNonQueryAsync(ct); + } + await TimescaleSupport.EnsureRetentionPoliciesAsync(connection, null, ct); var armed = await CountAsync(connection, @" @@ -774,6 +791,20 @@ WHERE NOT EXISTS (SELECT 1 FROM collect.{view} AS h WHERE h.bucket = src.b)", /* (5) And only NOW does the gate arm — coverage genuinely reaches raw. */ Assert.True(await RollupBackfill.ReadCoverageFloorAsync(connection, view, ct) <= probe.SourceOldestUtc); + + /* #3653 LC: RawTierCoverage for query_stats now requires BOTH query_stats_interval_hourly AND + query_stats_db_interval_hourly. Force-refresh the db-grain companion so the gate can arm; the + force is needed because the backfill slices above consumed query_stats' shared invalidation log, + leaving the db-grain companion with no entries to process on a plain refresh. Floor to the hourly + bucket boundary: refresh_continuous_aggregate window_start excludes buckets whose START < start, + so a sub-hour timestamp would skip the first bucket and leave coverage one bucket short of raw. */ + var dbRefreshFrom = new DateTime(plan.FromUtc.Year, plan.FromUtc.Month, plan.FromUtc.Day, plan.FromUtc.Hour, 0, 0, DateTimeKind.Unspecified); + using (var dbRefresh = new NpgsqlCommand(TimescaleSupport.RefreshContinuousAggregateSql(TimescaleSupport.QueryStatsDbIntervalHourlyView, force: true), connection)) + { + dbRefresh.Parameters.AddWithValue(dbRefreshFrom); + await dbRefresh.ExecuteNonQueryAsync(ct); + } + await TimescaleSupport.EnsureRetentionPoliciesAsync(connection, null, ct); Assert.Equal(1L, await CountAsync(connection, ArmedRawPolicySql, ct)); From 0cb218858a5a7140dab7dbcd202ae86ca2a28a95 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:20:34 -0400 Subject: [PATCH 11/18] Fix Medium finding: ConvergeCompressionScheduleAsync resets frozen rollup compression cadence (#3653) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Three defects in the LC freeze path: 1. FrozenDailyCompressionDrainStateSql scoped to SupersededDailyRollups (3 dailies) but existing stores also have compression policies on the 3 frozen hourlies; expand to FrozenRollupAggregates (all 6) so the drain cleans up hourly policies too. 2. ConvergeCompressionScheduleAsync lacked a guard for IsFrozenRollupAggregate: the six frozen views are removed from AggregateCompressionTargets, so they fell through the existing IsAggregateCompressionTarget guard and had their compression schedule reset to the 1-hour raw cadence on every start. Added an unconditional skip before the existing guard — frozen views are never ours to tune. 3. LogCompressionActivity's summary debug log counted frozen views in the raw-table bucket, misreporting the raw policy count; subtract frozenPolicies from the total. Test: updated FrozenDailyCompressionDrainStateSql test to assert all 6 views; added IsFrozenRollupAggregate_ReturnsTrueForAllSixFrozenViews_AndFalseForOthers to pin the predicate and prove it is disjoint from AggregateCompressionTargets. Co-Authored-By: Claude Sonnet 4.6 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../TimescaleAggregateCompressionTests.cs | 69 +++++++++++++++---- .../TimescaleSupport.cs | 68 +++++++++++------- 2 files changed, 99 insertions(+), 38 deletions(-) diff --git a/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs b/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs index 6806075948..fd6c5e8ee0 100644 --- a/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs +++ b/Darling/Darling.Tests/TimescaleAggregateCompressionTests.cs @@ -855,18 +855,28 @@ public void ShouldDrainFrozenDailyCompression_OnlyWhenAJobExistsAndNoChunkIsUnco } /// - /// #3653 LC point 7: the drain's probe names exactly the three frozen legacy dailies, counts EVERY - /// uncompressed chunk (no range_end/age filter, unlike 's - /// eligible_under_*_rule columns), and never the frozen HOURLIES, which never drain. + /// #3653 LC point 7 (Medium finding fix): the drain's probe names all six frozen rollups — + /// three legacy hourlies and three legacy dailies — counts EVERY uncompressed chunk (no + /// range_end/age filter, unlike 's + /// eligible_under_*_rule columns), and removes the compression policy once a frozen + /// rollup holds nothing left to compress. /// [Fact] - public void FrozenDailyCompressionDrainStateSql_NamesExactlyTheThreeFrozenDailies_AndCountsEveryUncompressedChunk() + public void FrozenDailyCompressionDrainStateSql_NamesAllSixFrozenRollups_AndCountsEveryUncompressedChunk() { var sql = TimescaleSupport.FrozenDailyCompressionDrainStateSql; + /* All three frozen dailies must appear. */ Assert.Contains($"'{TimescaleSupport.QueryStatsDailyView}'", sql, StringComparison.Ordinal); Assert.Contains($"'{TimescaleSupport.ProcedureStatsDailyView}'", sql, StringComparison.Ordinal); Assert.Contains($"'{TimescaleSupport.QueryStatsDbDailyView}'", sql, StringComparison.Ordinal); + + /* All three frozen hourlies must also appear — existing stores had compression policies on all + six before the LC freeze, and all six need draining (#3653, Medium finding fix). */ + Assert.Contains($"'{TimescaleSupport.QueryStatsHourlyView}'", sql, StringComparison.Ordinal); + Assert.Contains($"'{TimescaleSupport.ProcedureStatsHourlyView}'", sql, StringComparison.Ordinal); + Assert.Contains($"'{TimescaleSupport.QueryStatsDbHourlyView}'", sql, StringComparison.Ordinal); + Assert.Contains("NOT c.is_compressed", sql, StringComparison.Ordinal); /* No age gate — every uncompressed chunk counts, not just the ones a tier's compress_after would @@ -874,17 +884,52 @@ already call eligible. */ Assert.DoesNotContain("range_end", sql, StringComparison.Ordinal); Assert.DoesNotContain("eligible_under", sql, StringComparison.Ordinal); - /* Never a frozen HOURLY — those never drain; their chunks age out through retention instead. */ - Assert.DoesNotContain($"'{TimescaleSupport.QueryStatsHourlyView}'", sql, StringComparison.Ordinal); - Assert.DoesNotContain($"'{TimescaleSupport.ProcedureStatsHourlyView}'", sql, StringComparison.Ordinal); - Assert.DoesNotContain($"'{TimescaleSupport.QueryStatsDbHourlyView}'", sql, StringComparison.Ordinal); - - /* Exactly three — one per SupersededDailyRollups member (LegacyDaily), not one per the six-member - FrozenRollupAggregates, which also holds the three hourlies the probe must never touch. */ - Assert.Equal(3, TimescaleSupport.SupersededDailyRollups.Length); + /* Exactly six — one per FrozenRollupAggregates member. */ + Assert.Equal(6, TimescaleSupport.FrozenRollupAggregates.Length); var remove = TimescaleSupport.RemoveFrozenDailyCompressionPolicySql(TimescaleSupport.QueryStatsDailyView); Assert.Contains("remove_compression_policy('collect.query_stats_daily'", remove, StringComparison.Ordinal); Assert.Contains("if_exists => true", remove, StringComparison.Ordinal); } + + /// + /// #3653 LC (Medium finding fix): returns true + /// for all six frozen views — bare and collect.-qualified — and false for everything else. + /// The six frozen views are NOT in , which is + /// why needs the separate + /// guard: without it, they fall through and + /// get their compression schedule reset to the 1-hour raw cadence on every start. + /// + [Fact] + public void IsFrozenRollupAggregate_ReturnsTrueForAllSixFrozenViews_AndFalseForOthers() + { + foreach (var (_, view) in TimescaleSupport.FrozenRollupAggregates) + { + Assert.True(TimescaleSupport.IsFrozenRollupAggregate(view), $"bare: {view}"); + Assert.True(TimescaleSupport.IsFrozenRollupAggregate("collect." + view), $"qualified: {view}"); + + /* Not an active compression target — the guard in ConvergeCompressionScheduleAsync is needed + precisely because they are absent from this list. */ + Assert.False(TimescaleSupport.IsAggregateCompressionTarget(view), $"must not be a compression target: {view}"); + } + + /* Active aggregates and raw tables are not frozen. */ + foreach (var (_, view, _) in TimescaleSupport.AggregateCompressionTargets) + { + Assert.False(TimescaleSupport.IsFrozenRollupAggregate(view), view); + } + + foreach (var table in TimescaleSupport.CompressionPhaseOrder) + { + Assert.False(TimescaleSupport.IsFrozenRollupAggregate(table), table); + } + + Assert.False(TimescaleSupport.IsFrozenRollupAggregate(null)); + Assert.False(TimescaleSupport.IsFrozenRollupAggregate(string.Empty)); + + /* The two predicates are disjoint: no view can be both active and frozen. */ + Assert.Empty( + TimescaleSupport.FrozenRollupAggregates.Select(a => a.View) + .Intersect(TimescaleSupport.AggregateCompressionTargets.Select(t => t.View), StringComparer.Ordinal)); + } } diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index a021241542..2f6c053f5d 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -7911,17 +7911,18 @@ private static string FormatGiB(long bytes) /* ─────────────── the frozen dailies' compression drain (#3653, LC point 7) ─────────────── */ /// - /// One of the three frozen legacy dailies' ('s LegacyDaily - /// members) own compression drain state, as reads it — - /// its compression job if any, and how many of its materialization's chunks are uncompressed. + /// One of the six frozen rollups' ( members) own compression drain + /// state, as reads it — its compression job if any, + /// and how many of its materialization's chunks are uncompressed. /// public sealed record FrozenDailyCompressionDrainState(string View, int? JobId, long UncompressedChunks); /// - /// The three frozen legacy dailies' own compression drain state in one read: whether each still carries a - /// compression job, and how many of its materialization's chunks are uncompressed — EVERY uncompressed - /// chunk, of any age, not the age-gated eligible_under_hourly_rule / eligible_under_daily_rule - /// columns reads for the registered targets. + /// All six frozen rollups' () own compression drain state in one + /// read: whether each still carries a compression job, and how many of its materialization's chunks are + /// uncompressed — EVERY uncompressed chunk, of any age, not the age-gated + /// eligible_under_hourly_rule / eligible_under_daily_rule columns + /// reads for the registered targets. /// /// Why not those columns. They answer "how many chunks would this policy's NEXT run compress", /// which is gated on a tier's compress_after — a chunk younger than that window is not eligible YET @@ -7930,8 +7931,9 @@ public sealed record FrozenDailyCompressionDrainState(string View, int? JobId, l /// still count toward that, or the drain would remove the policy while a chunk was still waiting its turn /// to compress, and nothing would ever compress it afterward. /// - /// Scoped BY NAME to exactly 's three LegacyDaily members - /// — the only three this rule ever drains — rather than reading every non-target aggregate under + /// Scoped BY NAME to exactly 's six members — all six this + /// rule ever drains (three legacy hourlies and three legacy dailies; existing stores had compression + /// policies on all six before the LC freeze) — rather than reading every non-target aggregate under /// collect, so an unrelated off-grid or newly-registered-but-not-yet-enabled aggregate can never be /// swept in by a broadened WHERE later. Same join shape as /// (job resolved on either the view or the materialization identity) for the same measured reason. @@ -7940,7 +7942,7 @@ public static string FrozenDailyCompressionDrainStateSql { get { - var views = string.Join(", ", SupersededDailyRollups.Select(s => $"'{s.LegacyDaily}'")); + var views = string.Join(", ", FrozenRollupAggregates.Select(a => $"'{a.View}'")); return $@" SELECT ca.view_name, @@ -7992,23 +7994,25 @@ public static string RemoveFrozenDailyCompressionPolicySql(string view) => $"SELECT remove_compression_policy('collect.{view}', if_exists => true)"; /// - /// THE DRAIN (#3653, LC point 7): walks the three frozen legacy dailies and removes a compression policy - /// that has nothing left to compress (), via + /// THE DRAIN (#3653, LC point 7, Medium finding fix): walks all six frozen rollups + /// () and removes a compression policy that has nothing left to + /// compress (), via /// . Called once, at the end of /// — a separate concern from that function's own - /// enable/stage/converge loops, which never see these three ( - /// excludes them) the way this method's own scoped probe never sees a registered target. - /// - /// Why these three ever HAD a policy to drain. Every store that took a build before LC's - /// freeze enabled compression and attached a policy on these three the same as any other daily-tier - /// aggregate; the freeze stopped their REFRESH () but left - /// the existing compression policy running, because a policy already converged onto the shipped window - /// is not itself wrong — it is aimed at an aggregate that will, from the freeze on, only ever gain MORE - /// compressed chunks and never another uncompressed one. This is what eventually turns that policy off, - /// once there is truly nothing left for it to do — never on a fresh store past the freeze, which creates - /// these three with no compression policy at all ( carries no policy - /// builder), so there is nothing here to drain and reads a - /// null JobId forever. + /// enable/stage/converge loops, which never see these six ( + /// excludes the active twenty; identifies these six) the way this + /// method's own scoped probe never sees a registered target. + /// + /// Why these six ever HAD a policy to drain. Every store that took a build before LC's + /// freeze enabled compression and attached a policy on all six the same as any other registered aggregate; + /// the freeze stopped their REFRESH () but left the + /// existing compression policy running, because a policy already converged onto the shipped window is not + /// itself wrong — it is aimed at an aggregate that will, from the freeze on, only ever gain MORE compressed + /// chunks and never another uncompressed one. This is what eventually turns that policy off, once there is + /// truly nothing left for it to do — never on a fresh store past the freeze, which creates these six with + /// no compression policy at all ( carries no policy builder), so + /// there is nothing here to drain and reads a null + /// JobId forever. /// /// Failure-isolated per aggregate and per statement, the same as every other step in /// : a read or removal that fails costs a warning and the @@ -8732,6 +8736,15 @@ named like one of ours must not inherit its minute. */ && !reader.IsDBNull(6) && string.Equals(reader.GetString(6), PgSchemaGenerator.CollectSchema, StringComparison.Ordinal); + /* #3653 LC (Medium security finding): frozen rollups are removed from + AggregateCompressionTargets, so without this guard they fall through and get their + compression schedule reset to the 1-hour raw cadence on every start. Skip them + unconditionally — their schedule is never ours to tune. */ + if (IsFrozenRollupAggregate(hypertable)) + { + continue; + } + /* The continuous aggregates' compression policies are NOT this converge's to retune (#3581). Their job reports the aggregate's user view as its hypertable (collect / ), so it lands in this unscoped read, and its once-a-day cadence would read as stale against the @@ -10814,9 +10827,12 @@ public static void LogCompressionActivity( if (running == 0 && backlog == 0) { var aggregatePolicies = activity.Count(item => IsAggregateCompressionTarget(item.HypertableName)); + /* #3653 LC: frozen views are not in AggregateCompressionTargets, so subtract them from the + raw count too — they drain asynchronously and are not at the raw 1-hour cadence. */ + var frozenPolicies = activity.Count(item => IsFrozenRollupAggregate(item.HypertableName)); logger.LogDebug( "TimescaleDB: {Count} compression policies on a {Interval} tick and {AggregateCount} continuous-aggregate policies on a {AggregateInterval} cadence, nothing running, no eligible chunk uncompressed.", - activity.Count - aggregatePolicies, CompressScheduleInterval, aggregatePolicies, AggregateCompressionScheduleInterval); + activity.Count - aggregatePolicies - frozenPolicies, CompressScheduleInterval, aggregatePolicies, AggregateCompressionScheduleInterval); } LogCompressionClearance(activity, logger); From 0174f58546b6f92e4cde69e746e781f5b9883eac Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:28:13 -0400 Subject: [PATCH 12/18] Fix read regression: RollupCoverage.For() stitches legacy+successor floors (#3653 LC) After the LC freeze, callers that use coverage.For(legacyHourly, legacyDaily) to determine whether the hourly tier can serve a window were broken in two shapes: - Fresh store (just upgraded): legacy starts WITH NO DATA -> null floor -> every window older than ~4 days falls to raw. - 90+ days post-freeze: retention drains the legacy -> same null-floor collapse. Fix: RollupCoverage.For() now stitches the legacy and its interval-honest successor (from TimescaleSupport.SupersededHourlyRollups / SupersededDailyRollups), taking the deeper (earlier) of the two non-null floors as the effective floor for each tier. The stitch is backward-compatible: when only the legacy has data, the legacy floor wins unchanged. All seven production callers (FinOps.Workload.cs:247/408/476, ViewerDataService.DailySummary.cs:83, QueryTrends.cs:508, DarlingHealthReader.cs:420, DarlingMcpTrendTools.cs:383) get the correct behavior without individual edits. HourlyRelationFor comment updated: the legacy is no longer guaranteed deeper "by construction" on post-freeze stores; the stitched floor from For() is what makes the tier decision correct. Adds four pure regression tests covering the fresh-store, post-trim, pre-freeze, and stitch-boundary shapes. Co-Authored-By: Claude Sonnet 4.6 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../RollupCoverageRoutingTests.cs | 113 ++++++++++++++++++ .../TimescaleSupport.cs | 62 ++++++++-- 2 files changed, 167 insertions(+), 8 deletions(-) diff --git a/Darling/Darling.Tests/RollupCoverageRoutingTests.cs b/Darling/Darling.Tests/RollupCoverageRoutingTests.cs index dcfc12356d..f0e542642b 100644 --- a/Darling/Darling.Tests/RollupCoverageRoutingTests.cs +++ b/Darling/Darling.Tests/RollupCoverageRoutingTests.cs @@ -304,6 +304,119 @@ public void For_ResolvesEachLaddersOwnFloorsAndItsOwnRawTable() Assert.Null(RollupCoverage.Unknown.RawOldestOf("query_stats")); } + /* ─────── For stitch: legacy+successor floors for frozen pairs (#3653 LC) ─────── */ + + /// + /// On a FRESH store (just upgraded, legacy starts WITH NO DATA) the legacy floor is null and the successor + /// holds the data. Without the stitch, every window older than ~4 days falls to raw even though the successor + /// already has days/weeks of hourly and daily data. + /// + [Fact] + public void For_FreshStore_SuccessorFloorUsedWhenLegacyIsEmpty() + { + // Legacy trio floors absent (fresh store); successor hourlies and dailies have 30 and 90 days. + var coverage = new RollupCoverage( + new Dictionary(StringComparer.Ordinal) + { + [TimescaleSupport.QueryStatsIntervalHourlyView] = DaysAgo(30), + [TimescaleSupport.QueryStatsIntervalDailyView] = DaysAgo(90), + }, + new Dictionary(StringComparer.Ordinal) + { + ["query_stats"] = DaysAgo(4), + }); + + var tc = coverage.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView); + + // The stitched hourly floor is the successor's floor, not null. + Assert.Equal(DaysAgo(30), tc.HourlyFloorUtc); + Assert.Equal(DaysAgo(90), tc.DailyFloorUtc); + + // A 20-day window routes to Hourly, not Raw. + Assert.Equal(RetentionTier.Hourly, RetentionTierRouter.Resolve(Now, DaysAgo(20), true, true, tc)); + } + + /// + /// 90+ days post-freeze: retention has drained the legacy; the successor is the only active rollup. + /// Routing must still reach the hourly tier. + /// + [Fact] + public void For_PostTrimStore_SuccessorFloorUsedWhenLegacyDrained() + { + // Legacy is fully drained; successor has 30/90 days of data. + var coverage = new RollupCoverage( + new Dictionary(StringComparer.Ordinal) + { + [TimescaleSupport.QueryStatsIntervalHourlyView] = DaysAgo(30), + [TimescaleSupport.QueryStatsIntervalDailyView] = DaysAgo(90), + }, + new Dictionary(StringComparer.Ordinal) + { + ["query_stats"] = DaysAgo(4), + }); + + var tc = coverage.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView); + + Assert.Equal(DaysAgo(30), tc.HourlyFloorUtc); + Assert.Equal(DaysAgo(90), tc.DailyFloorUtc); + Assert.Equal(RetentionTier.Hourly, RetentionTierRouter.Resolve(Now, DaysAgo(20), true, true, tc)); + } + + /// + /// Pre-freeze store (legacy still has data, successor not yet present): the legacy floor is used unchanged. + /// The stitch must not disturb stores that never ran the freeze migration. + /// + [Fact] + public void For_PreFreezeStore_LegacyFloorUsedWhenSuccessorAbsent() + { + var coverage = new RollupCoverage( + new Dictionary(StringComparer.Ordinal) + { + [TimescaleSupport.QueryStatsHourlyView] = DaysAgo(20), + [TimescaleSupport.QueryStatsDailyView] = DaysAgo(60), + }, + new Dictionary(StringComparer.Ordinal) + { + ["query_stats"] = DaysAgo(4), + }); + + var tc = coverage.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView); + + Assert.Equal(DaysAgo(20), tc.HourlyFloorUtc); + Assert.Equal(DaysAgo(60), tc.DailyFloorUtc); + } + + /// + /// Post-freeze store with BOTH legacy and successor data (stitch boundary period): the deeper (earlier) + /// floor of the two wins for each tier. + /// + [Fact] + public void For_StitchBoundary_DeeperFloorWins() + { + // Legacy has 40 days, successor has 10 days (just started filling after freeze). + var coverage = new RollupCoverage( + new Dictionary(StringComparer.Ordinal) + { + [TimescaleSupport.QueryStatsHourlyView] = DaysAgo(40), + [TimescaleSupport.QueryStatsDailyView] = DaysAgo(80), + [TimescaleSupport.QueryStatsIntervalHourlyView] = DaysAgo(10), + [TimescaleSupport.QueryStatsIntervalDailyView] = DaysAgo(10), + }, + new Dictionary(StringComparer.Ordinal) + { + ["query_stats"] = DaysAgo(4), + }); + + var tc = coverage.For(TimescaleSupport.QueryStatsHourlyView, TimescaleSupport.QueryStatsDailyView); + + // Legacy is deeper — its floor wins. + Assert.Equal(DaysAgo(40), tc.HourlyFloorUtc); + Assert.Equal(DaysAgo(80), tc.DailyFloorUtc); + + // A 30-day window (inside the legacy's 40-day reach) routes to Hourly. + Assert.Equal(RetentionTier.Hourly, RetentionTierRouter.Resolve(Now, DaysAgo(30), true, true, tc)); + } + /* ─────────────────────────── the drift guard ─────────────────────────── */ /// diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 2f6c053f5d..3cf62f1f51 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -11675,12 +11675,13 @@ public RollupCoverage( /// handed. /// /// Called AFTER the tier is resolved, never instead of it. The router decides whether the hourly - /// tier can answer a window from over the LEGACY pair — the legacy floor is the deeper of - /// the two by construction (an upgraded store's successor starts WITH NO DATA), so "can the hourly - /// tier serve this window" is the legacy's question. Only once the tier is Hourly does THIS choose which + /// tier can answer a window from over the LEGACY pair — stitches the + /// legacy and successor floors (#3653 LC), taking the deeper of the two, so "can the hourly tier serve this + /// window" is answered correctly on both a pre-freeze store (legacy is deeper) and a fresh or post-trim store + /// (successor is deeper or the only non-null floor). Only once the tier is Hourly does THIS choose which /// hourly relation, and the successor is chosen only when the window loses nothing by it. A reader that - /// asked this first and routed on the successor's floor would fall to raw on the very stores where the - /// legacy still holds the answer. + /// asked this first and routed on the successor's floor alone would fall to raw on stores where the legacy + /// still holds the answer. /// /// Absence beats coverage. A store on a service that predates the successors /// () has a null successor floor AND no successor @@ -11741,14 +11742,59 @@ public string HourlyRelationNameFor(string legacyHourly, DateTime windowStartUtc /// . A pair with no /// daily view (or an unrecognized hourly view) still answers — with nulls, which the router treats as /// "no evidence". + /// + /// Stitched floor for frozen-legacy pairs (#3653 LC). After the LC freeze, a FRESH store + /// (just upgraded, legacy starts WITH NO DATA) has a null legacy floor while its successor holds real + /// data. A store 90+ days post-freeze has had its legacy trimmed to empty for the same reason. Both shapes + /// produce a null legacy floor, so routing on the legacy floor alone falls every window older than 4 days to + /// raw. This method therefore stitches the legacy AND the successor: the effective floor for each tier is + /// the deeper (earlier) of the two non-null floors. A null from both is still null (no evidence from + /// either). Callers that determine which hourly relation to name in SQL still go through + /// , which makes the legacy-vs-successor choice independently and only after + /// the tier has been resolved. /// public TierCoverage For(string hourlyView, string? dailyView) { var rawTable = RawTableFor(hourlyView); - return new TierCoverage( + + // #3653 LC: stitch legacy and successor floors so that fresh stores (empty legacy, successor has + // data) and post-trim stores (legacy drained by retention) still route correctly to the hourly tier. + var successorHourly = TimescaleSupport.SuccessorOf(hourlyView); + var hourlyFloor = DeepestFloor( FloorOf(hourlyView), - dailyView is null ? null : FloorOf(dailyView), - rawTable is null ? null : RawOldestOf(rawTable)); + successorHourly is null ? null : FloorOf(successorHourly)); + + DateTime? dailyFloor = null; + if (dailyView is not null) + { + var successorDaily = SuccessorDailyFromLegacy(dailyView); + dailyFloor = DeepestFloor( + FloorOf(dailyView), + successorDaily is null ? null : FloorOf(successorDaily)); + } + + return new TierCoverage(hourlyFloor, dailyFloor, rawTable is null ? null : RawOldestOf(rawTable)); + } + + /// The deeper (earlier) of two nullable UTC floors — the effective floor when a frozen legacy + /// and its interval-honest successor both cover part of the span (#3653 LC). A null on either side is + /// skipped; null on both yields null (no evidence from either tier). + private static DateTime? DeepestFloor(DateTime? a, DateTime? b) => + (a, b) is ({ } da, { } db) ? (da < db ? da : db) : a ?? b; + + /// The interval-honest SUCCESSOR DAILY that replaces for the + /// daily-tier floor lookup (#3653 LC), or null for a daily view that was never superseded. + private static string? SuccessorDailyFromLegacy(string legacyDaily) + { + foreach (var (legacyDailyV, successorDaily, _) in TimescaleSupport.SupersededDailyRollups) + { + if (string.Equals(legacyDailyV, legacyDaily, StringComparison.Ordinal)) + { + return successorDaily; + } + } + + return null; } /// Which tier is stitching (#3653 A6): the boundary is the From 3cfeeea2a9ea87c273edfa78670cc74d38e9a2e0 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 07:32:23 -0400 Subject: [PATCH 13/18] Fix High: raw purge gate holds forever on field upgrade (#3653) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Two issues combine to hold the raw purge gate open permanently when a store upgrades from the current release to LC: 1. Stitched coverage (Option 4): RawTierCoverage named only the successor hourlies. On first start after upgrade the successors are empty, so MeasureRetentionCoverageAsync sees NULL coverage_oldest → Short → holds forever. The stitch generates COALESCE(LEAST(legacy.min, successor.min), legacy.min, successor.min) via a new LegacyOf() helper that reverse-looks up SupersededHourlyRollups. When the successor is empty the frozen legacy's floor covers all of raw; once the successor accumulates its own history the legacy term is the minimum anyway. Both empty → NULL → Short (correct: fresh install). 2. Zero-interval source filter: raw's very oldest row may have sample_interval_seconds = 0 (first-pass, no delta possible). The successors exclude those rows from their CREATE via IntervalHonestSourceFilter. Without the same filter on source_oldest, a single 0-interval row older than any successor bucket holds the gate even after fix 1. RetentionArmSafetySql now adds WHERE sample_interval_seconds IS DISTINCT FROM 0 to the source_oldest subquery for query_stats and procedure_stats. Adds: - IntervalHonestSourceFilter public constant for pin tests - LegacyOf() private helper (reverse lookup in SupersededHourlyRollups) - Live test: FieldUpgrade_EmptySuccessors_LegacyFilled_ReportsRawPurgeCovered - Pin test: IntervalHonestSourceFilter_MatchesEverySuccessorsCreateText Co-Authored-By: Claude Sonnet 4.6 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../Darling.Tests/FrozenRollupLiveTests.cs | 87 +++++++++++++++++++ .../IntervalHonestHourlyRollupTests.cs | 18 ++++ .../TimescaleSupport.cs | 71 ++++++++++++++- 3 files changed, 174 insertions(+), 2 deletions(-) diff --git a/Darling/Darling.Tests/FrozenRollupLiveTests.cs b/Darling/Darling.Tests/FrozenRollupLiveTests.cs index ba8f335666..f181b9d055 100644 --- a/Darling/Darling.Tests/FrozenRollupLiveTests.cs +++ b/Darling/Darling.Tests/FrozenRollupLiveTests.cs @@ -447,6 +447,93 @@ await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async ( } } + /// + /// #3653 high-fix (Option 4): on a store upgrading from the current release, the interval-honest successor + /// hourlies are empty on first start while the frozen legacy hourlies still hold years of history. + /// must return true for such a store — the + /// stitch covers all of raw's range via the legacy floor, and once the successor has enough history the + /// legacy term becomes the minimum anyway. + /// + /// Also covers the zero-interval source filter: a first-pass row in raw has + /// sample_interval_seconds = 0; the successors exclude those rows from their CREATE. A 0-interval + /// row older than any materialized bucket must not hold the gate open permanently. + /// + [Fact] + public async Task FieldUpgrade_EmptySuccessors_LegacyFilled_ReportsRawPurgeCovered() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* 5 days of raw history, with a 0-interval (first-pass) row at day 0 to exercise the + source-oldest filter. The successors exclude these rows; without the filter the gate + would hold forever even after the stitch fix. */ + await InsertQueryStatsAsync(connection, D0, "FU_ZERO", 0, 0, 0, ct); + await InsertProcedureStatsAsync(connection, D0, "fu_zero_proc", 0, 0, 0, ct); + + for (var day = 1; day <= 5; day++) + { + var at = D0.AddDays(day); + await InsertQueryStatsAsync(connection, at, $"FU_QS{day}", 1000, 10, 3600, ct); + await InsertProcedureStatsAsync(connection, at, $"fu_ps{day}", 900, 9, 3600, ct); + } + + /* Pre-freeze state: legacy hourlies refreshed to cover all of the interval-1 rows (days 1-5). + The 0-interval row at day 0 is NOT materialized by either legacy or successor (both filter + it out via their respective CREATE predicates or the source_oldest filter). */ + var d6 = D0.AddDays(6); + await RefreshAsync(connection, TimescaleSupport.QueryStatsHourlyView, D0.AddDays(1), d6, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, D0.AddDays(1), d6, ct); + + /* Successors are intentionally NOT refreshed — simulating an upgrading store where LC just + shipped and the successors are still empty. */ + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.QueryStatsIntervalHourlyView, ct)); + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.QueryStatsDbIntervalHourlyView, ct)); + Assert.Equal(0L, await CountRowsAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, ct)); + + /* The stitched gate must report Covered: the legacy floor (days 1-5) covers raw's interval-1 + oldest row (day 1), so there is no history a purge would destroy. */ + Assert.True(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "query_stats", ct)); + Assert.True(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + private static async Task InsertQueryStatsAsync( NpgsqlConnection connection, DateTime at, string hash, long workerUs, long executions, int intervalSeconds, CancellationToken ct) { diff --git a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs index 7babe29560..0db2ce71a2 100644 --- a/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs +++ b/Darling/Darling.Tests/IntervalHonestHourlyRollupTests.cs @@ -536,6 +536,24 @@ private static PanelPlan ComposePlan(string measureKey) => Viz = "line", }; + /// + /// #3653 high-fix: is the WHERE predicate that + /// every interval-honest successor bakes into its CREATE SQL. The source_oldest subquery in + /// applies the same predicate for + /// query_stats and procedure_stats, so a 0-interval row cannot hold the gate open + /// permanently. This pin guards that the constant and each successor's CREATE text stay in agreement — + /// a change to one without the other is caught here before it reaches production. + /// + [Fact] + public void IntervalHonestSourceFilter_MatchesEverySuccessorsCreateText() + { + foreach (var (_, successor, _) in TimescaleSupport.SupersededHourlyRollups) + { + var successorText = TimescaleSupport.RollupCoverageProbeTargets.Single(t => t.View == successor).CreateSql; + Assert.Contains(TimescaleSupport.IntervalHonestSourceFilter, successorText, StringComparison.Ordinal); + } + } + private static string ReadWorkerSource([CallerFilePath] string thisFile = "") { var relative = Path.Combine("Darling", "PerformanceMonitor.Darling.Service", "DarlingWorker.cs"); diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 2f6c053f5d..482214bc05 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -1495,6 +1495,23 @@ public static readonly (string LegacyDaily, string SuccessorDaily, string Succes private static string RequireSuccessorOf(string legacyHourly) => SuccessorOf(legacyHourly) ?? throw new InvalidOperationException($"{legacyHourly} has no successor in {nameof(SupersededHourlyRollups)}."); + /// The legacy hourly that supersedes, or null when + /// is not a member of . Used by + /// to generate stitched coverage SQL for frozen-legacy + successor + /// pairs — see that member's doc for the full rationale (#3653 high-fix, Option 4). + private static string? LegacyOf(string successor) + { + foreach (var (legacy, s, _) in SupersededHourlyRollups) + { + if (string.Equals(s, successor, StringComparison.Ordinal)) + { + return legacy; + } + } + + return null; + } + /// The interval-honest successor DAILY that covers once the /// hourly-tier read routes past its own horizon (#3653, A6/LB's three successor dailies) — the coverage /// relation arms a successor hourly's own drop_chunks against, in place of @@ -5970,6 +5987,18 @@ FROM timescaledb_information.jobs AS j AND j.hypertable_schema = 'collect' AND j.hypertable_name = '{relation}'"; + /// + /// The WHERE predicate the interval-honest successor hourlies bake into their CREATE — they exclude + /// first-pass rows where the collector had no previous sample to delta against and stored + /// sample_interval_seconds = 0. applies the same predicate to + /// the source_oldest subquery for query_stats and procedure_stats, so a 0-interval row + /// that pre-dates every materialized bucket cannot hold the purge gate open indefinitely (#3653 high-fix). + /// + /// A pin test in IntervalHonestHourlyRollupTests asserts this constant appears in each + /// successor's CREATE text, so a CREATE change that drops the predicate is caught automatically. + /// + public const string IntervalHonestSourceFilter = "sample_interval_seconds IS DISTINCT FROM 0"; + /// /// Is it safe to arm 's retention policy — i.e. does EVERY tier below it already /// cover everything this relation holds? Emits the source's oldest row followed by one @@ -5987,6 +6016,22 @@ FROM timescaledb_information.jobs AS j /// GREATEST would have expressed it in one column and is exactly wrong here, because it SKIPS NULLs. /// An empty new rollup would vanish from the comparison and the gate would pass on the old rollup alone — /// which is the whole failure this exists to prevent. + /// + /// Stitched coverage for frozen-legacy + successor pairs (#3653 high-fix, Option 4). + /// When a coverage relation is an interval-honest successor (a member of + /// ), the SQL for that coverage slot is + /// COALESCE(LEAST(legacy.min, successor.min), legacy.min, successor.min) rather than + /// min(successor.bucket) alone. This lets an upgrading store whose successor is still empty + /// fall back to the frozen legacy's floor — the legacy was refreshing up to the freeze point and + /// covers all of raw's history, so the stitch is gap-free by construction (successor was started from + /// , which overlaps the freeze). Once the successor accumulates + /// enough history to cover raw on its own, the legacy term becomes the minimum anyway and the result is + /// unchanged. Both empty → NULL → Short (correct: fresh install, no history yet). + /// + /// Source filter for 0-interval rows. For query_stats and procedure_stats the + /// source_oldest subquery adds WHERE to exclude + /// first-pass rows the successors themselves never materialize — without this, a single 0-interval row + /// older than any successor bucket holds the gate open permanently even after the stitch fix. /// public static string RetentionArmSafetySql(string relation, string sourceTimeColumn, IReadOnlyList coverageRelations) { @@ -5995,8 +6040,30 @@ public static string RetentionArmSafetySql(string relation, string sourceTimeCol throw new ArgumentNullException(nameof(coverageRelations)); } - var columns = coverageRelations.Select((c, i) => $" (SELECT min(bucket) FROM collect.{c}) AS coverage_oldest_{i}"); - return $"SELECT{Environment.NewLine} (SELECT min({sourceTimeColumn}) FROM collect.{relation}) AS source_oldest,{Environment.NewLine}" + /* Source filter: the interval-honest successors for query_stats and procedure_stats bake + IntervalHonestSourceFilter into their CREATE. A 0-interval row in raw that pre-dates every + materialized bucket would hold the gate forever even with stitching, so we apply the same + filter to source_oldest. query_store_stats has no such filter in its hourly CREATE. */ + var sourceWhere = relation is "query_stats" or "procedure_stats" + ? $"\nWHERE {IntervalHonestSourceFilter}" + : string.Empty; + + /* Coverage SQL: for a coverage relation that is an interval-honest successor, stitch it with + its frozen legacy so an empty successor on an upgrading store falls back to the legacy's + floor. LegacyOf() returns null for any relation that is not in SupersededHourlyRollups + (dailies, query_store_stats, baseline aggregates), keeping those as simple min(bucket). */ + var columns = coverageRelations.Select((c, i) => + { + var legacy = LegacyOf(c); + var subquery = legacy is not null + ? $"(SELECT COALESCE(LEAST(l.mn, s.mn), l.mn, s.mn){Environment.NewLine}" + + $" FROM (SELECT min(bucket) AS mn FROM collect.{legacy}) l{Environment.NewLine}" + + $" CROSS JOIN (SELECT min(bucket) AS mn FROM collect.{c}) s)" + : $"(SELECT min(bucket) FROM collect.{c})"; + return $" {subquery} AS coverage_oldest_{i}"; + }); + + return $"SELECT{Environment.NewLine} (SELECT min({sourceTimeColumn}) FROM collect.{relation}{sourceWhere}) AS source_oldest,{Environment.NewLine}" + string.Join("," + Environment.NewLine, columns); } From 3098a9e5d25ee2f9d2e358f2327a7dbaa226097f Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 08:09:38 -0400 Subject: [PATCH 14/18] Fix #4186 data-loss: stitched raw purge gate must honor the outage seam 3cfeeea2's stitched coverage (frozen legacy + successor) counted legacy+successor as covering raw with no gap check, and its doc claimed the stitch was gap-free by construction. That is false: a store stopped 1-4 days before the successor's first refresh leaves a raw tail (about 2h) between the legacy's last bucket and the successor's floor that neither side ever materializes; RawRetentionInterval (4 days) then purges it permanently. RetentionArmSafetySql now probes raw itself for a filter-admitted row in that seam (legacy's last bucket to the successor's floor, or +infinity when the successor is empty) before trusting the stitched floor. A row found falls back to the successor's own min(bucket) (Short, honestly) instead of the false Covered; an empty seam keeps the old stitched LEAST(legacy.min, successor.min), simplified from the redundant COALESCE(LEAST(...), ...) since PostgreSQL's LEAST already ignores NULLs. A floors-only gap test was rejected: it would deadlock forever on an outage with no rows in the seam or a tail already purged. RepairMaterializationHolesAsync's hole walk now scans a successor with a frozen legacy from the legacy's last bucket rather than the successor's own floor, so the seam tail is repaired on the next start and the gate's fallback releases on its own. Also fixes FieldUpgrade_EmptySuccessors_LegacyFilled_ReportsRawPurgeCovered, which was already failing on this branch's own CI (confirmed before touching it): its fixture refreshed only two of query_stats' three legacies, leaving the LC-a4 db-grain pair fully empty on both sides regardless of this change. Adds the outage-shape and empty-seam live tests plus a pure SQL-shape pin test; corrects the "gap-free by construction" doc claims in both files. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../Darling.Tests/FrozenRollupLiveTests.cs | 162 +++++++++++++++++- .../TimescaleContinuousAggregateTests.cs | 51 ++++++ .../TimescaleSupport.MaterializationHoles.cs | 43 ++++- .../TimescaleSupport.cs | 66 +++++-- 4 files changed, 304 insertions(+), 18 deletions(-) diff --git a/Darling/Darling.Tests/FrozenRollupLiveTests.cs b/Darling/Darling.Tests/FrozenRollupLiveTests.cs index f181b9d055..ff6a5b117d 100644 --- a/Darling/Darling.Tests/FrozenRollupLiveTests.cs +++ b/Darling/Darling.Tests/FrozenRollupLiveTests.cs @@ -458,6 +458,158 @@ await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async ( /// sample_interval_seconds = 0; the successors exclude those rows from their CREATE. A 0-interval /// row older than any materialized bucket must not hold the gate open permanently. /// + [Fact] + public async Task Outage_SeamBetweenFrozenLegacyAndSuccessor_HoleWalkRepairsItAndGateReleases() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* #4186: S is the stop. The legacy is refreshed through S-2h only (its own end_offset lag, the + same reason every hourly rollup trails "now" by roughly an hour). Raw carries admitted rows + through S — the 2-hour tail [S-1h, S] is the seam: collected, but never materialized by + either side — plus one interval-0 restart row just after S that the filter must reject, then + a genuine 2-day outage (no rows at all) before the successor's first post-upgrade refresh at + U = S+2d, whose own floor (U-1h) starts long after the seam. */ + var s = D0.AddDays(3); + + for (var hour = 0; hour <= 6; hour++) + { + await InsertProcedureStatsAsync(connection, s.AddHours(-hour), $"seam_proc_{hour}", 900, 9, 3600, ct); + } + + await InsertProcedureStatsAsync(connection, s.AddMinutes(30), "seam_proc_restart", 0, 0, 0, ct); + + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, s.AddHours(-6), s.AddHours(-1), ct); + + var u = s.AddDays(2); + await InsertProcedureStatsAsync(connection, u.AddHours(-1), "seam_proc_successor_floor", 900, 9, 3600, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, u.AddDays(-1), u, ct); + + /* The seam holds raw rows the stitch cannot see through unconditionally — Short, not Covered. */ + Assert.False(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + /* The product's own start-path entry point (DarlingWorker calls this exact method). */ + var summary = await TimescaleSupport.RepairMaterializationHolesAsync(connection, null, u, ct); + Assert.True(summary.BucketsRepaired >= 2, $"expected the 2-bucket seam tail to be repaired, got {summary.BucketsRepaired}"); + + await using (var span = new NpgsqlCommand($"SELECT min(bucket) FROM collect.{TimescaleSupport.ProcedureStatsIntervalHourlyView}", connection)) + { + var newFloor = (DateTime)(await span.ExecuteScalarAsync(ct))!; + Assert.Equal(s.AddHours(-1), newFloor); + } + + /* The seam is now empty (the successor's own floor reaches the legacy's boundary) — Covered. */ + Assert.True(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + + /// + /// #4186: the no-deadlock twin of — + /// same frozen-legacy-plus-outage shape, but the seam itself holds NO raw rows (an outage that began + /// right at a legacy refresh, or a tail already purged before this fix existed). A FLOOR-only + /// contiguity test (successor.min <= legacy.max + 1h) would read this as a permanent gap and + /// hold the purge forever with nothing left to repair it — the deadlock the brief for this fix rules + /// out. Probing raw directly must report Covered here with no hole walk at all. + /// + [Fact] + public async Task Outage_EmptySeam_GateReportsCoveredWithNoHoleWalk() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* Same S and U as the outage-shape twin, but NOTHING is inserted in (S-2h, S] — the outage began + exactly at the legacy's last refresh, so there is no un-materialized tail to find. */ + var s = D0.AddDays(3); + + for (var hour = 2; hour <= 6; hour++) + { + await InsertProcedureStatsAsync(connection, s.AddHours(-hour), $"nogap_proc_{hour}", 900, 9, 3600, ct); + } + + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, s.AddHours(-6), s.AddHours(-1), ct); + + var u = s.AddDays(2); + await InsertProcedureStatsAsync(connection, u.AddHours(-1), "nogap_proc_successor_floor", 900, 9, 3600, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, u.AddDays(-1), u, ct); + + Assert.True(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + [Fact] public async Task FieldUpgrade_EmptySuccessors_LegacyFilled_ReportsRawPurgeCovered() { @@ -501,11 +653,15 @@ would hold forever even after the stitch fix. */ await InsertProcedureStatsAsync(connection, at, $"fu_ps{day}", 900, 9, 3600, ct); } - /* Pre-freeze state: legacy hourlies refreshed to cover all of the interval-1 rows (days 1-5). - The 0-interval row at day 0 is NOT materialized by either legacy or successor (both filter - it out via their respective CREATE predicates or the source_oldest filter). */ + /* Pre-freeze state: EVERY legacy hourly refreshed to cover all of the interval-1 rows (days + 1-5) — query_stats has TWO (query-grain and db-grain, #1849/LC-a4), and both must be filled + or the db-grain slot reads NULL (empty legacy AND empty successor) and reports Short on its + own, independent of anything this test means to exercise. The 0-interval row at day 0 is NOT + materialized by either legacy or successor (both filter it out via their respective CREATE + predicates or the source_oldest filter). */ var d6 = D0.AddDays(6); await RefreshAsync(connection, TimescaleSupport.QueryStatsHourlyView, D0.AddDays(1), d6, ct); + await RefreshAsync(connection, TimescaleSupport.QueryStatsDbHourlyView, D0.AddDays(1), d6, ct); await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, D0.AddDays(1), d6, ct); /* Successors are intentionally NOT refreshed — simulating an upgrading store where LC just diff --git a/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs b/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs index 0baffee635..dfd15b7ff2 100644 --- a/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs +++ b/Darling/Darling.Tests/TimescaleContinuousAggregateTests.cs @@ -1254,6 +1254,57 @@ public void RetentionArmSafetySql_NamesEveryCoverageRelation() Assert.DoesNotContain("GREATEST", sql, StringComparison.Ordinal); } + /// + /// WATCHED (mutation, #4186): a stitched slot (a coverage relation with a frozen legacy, found through + /// LegacyOf) must probe raw for a seam row before it falls back to the legacy's floor — an + /// UNCONDITIONAL stitch is the #4186 data-loss defect: a raw tail between the legacy's last bucket and + /// the successor's floor, never materialized by either side, read Covered and the purge dropped it. A + /// non-stitched slot (no LegacyOf match) must stay the plain min(bucket) form untouched. + /// + [Fact] + public void RetentionArmSafetySql_StitchedSlot_ProbesSeamBeforeFallingBackToLegacyFloor() + { + var sql = TimescaleSupport.RetentionArmSafetySql( + "query_stats", "collection_time", new[] { TimescaleSupport.QueryStatsIntervalHourlyView }); + + /* The seam probe: raw, bounded between the legacy's last bucket (+1h, so the legacy's own last + bucket is not re-counted) and the successor's floor (or infinity when it is empty). */ + Assert.Contains("EXISTS (", sql, StringComparison.Ordinal); + Assert.Contains("FROM collect.query_stats AS seam", sql, StringComparison.Ordinal); + Assert.Contains("l.mx + INTERVAL '1 hour'", sql, StringComparison.Ordinal); + Assert.Contains("COALESCE(s.mn, 'infinity'::timestamp)", sql, StringComparison.Ordinal); + + /* The filter must appear TWICE: once for source_oldest (every relation with a filter gets that + already) and once more inside the seam probe itself — a seam probe with no filter would let a + post-restart interval-0 row hold the gate open forever, exactly like source_oldest without it. */ + var filterOccurrences = sql.Split(new[] { TimescaleSupport.IntervalHonestSourceFilter }, StringSplitOptions.None).Length - 1; + Assert.True(filterOccurrences >= 2, + $"expected the seam probe to carry its own {nameof(TimescaleSupport.IntervalHonestSourceFilter)} in addition to source_oldest's, found {filterOccurrences} occurrence(s) in: {sql}"); + + /* Seam empty -> the stitched floor. PostgreSQL's LEAST already ignores NULLs, so the old + COALESCE(LEAST(l.mn, s.mn), l.mn, s.mn) wrapper was redundant; it must not come back. */ + Assert.Contains("LEAST(l.mn, s.mn)", sql, StringComparison.Ordinal); + Assert.DoesNotContain("COALESCE(LEAST(", sql, StringComparison.Ordinal); + + /* Seam NOT empty -> the successor's own floor alone, never the legacy's — the legacy cannot vouch + for raw history it never covered. */ + Assert.Contains("THEN s.mn", sql, StringComparison.Ordinal); + + Assert.Contains($"FROM collect.{TimescaleSupport.QueryStatsHourlyView}", sql, StringComparison.Ordinal); + Assert.Contains($"FROM collect.{TimescaleSupport.QueryStatsIntervalHourlyView}", sql, StringComparison.Ordinal); + + /* A non-stitched slot (query_store_stats' two consumers have no LegacyOf match) stays the plain + form — none of the seam machinery leaks into a slot that never needed it. */ + var plainSql = TimescaleSupport.RetentionArmSafetySql( + "query_store_stats", "collection_time", + new[] { TimescaleSupport.QueryStoreStatsHourlyView, TimescaleSupport.QueryStoreStatsIntervalHourlyView }); + + Assert.DoesNotContain("EXISTS (", plainSql, StringComparison.Ordinal); + Assert.DoesNotContain("seam", plainSql, StringComparison.Ordinal); + Assert.Contains($"(SELECT min(bucket) FROM collect.{TimescaleSupport.QueryStoreStatsHourlyView}) AS coverage_oldest_0", plainSql, StringComparison.Ordinal); + Assert.Contains($"(SELECT min(bucket) FROM collect.{TimescaleSupport.QueryStoreStatsIntervalHourlyView}) AS coverage_oldest_1", plainSql, StringComparison.Ordinal); + } + /// /// The map both purge paths read (#1784) must name BOTH Query Store rollup families as raw's coverage. /// query_store_stats is the only raw table with two consumers; naming just one would let raw purge over diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs index 6d1c987327..6d798443af 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs @@ -55,6 +55,21 @@ namespace PerformanceMonitor.Darling.Storage; /// business and are never touched here; buckets past the last materialized one are the live edge the policy /// owns. /// +/// The seam case (#4186): a successor with a frozen legacy is scanned from the LEGACY's last bucket, +/// not its own floor. The three interval-honest successors (SupersededHourlyRollups) can each have +/// an un-materialized TAIL below their own floor: an outage that outlasts +/// before the successor's first refresh leaves raw rows between the frozen legacy's last bucket and the +/// successor's floor that neither side ever materialized (the outage shape worked through above). Those rows +/// sit BELOW the successor's floor, so the ordinary "floor up to ceiling" scan never reaches them — a hole, by +/// this pass's own definition, has to be inside the materialized span. For a successor found through +/// LegacyOf, the scan's lower bound is extended down to the legacy's last bucket (plus one bucket width, +/// so the legacy's own last bucket is not re-scanned as if it were the successor's) whenever that reaches +/// further back than the successor's floor already does. The seam tail then reads as an ordinary hole and the +/// existing machinery repairs it: bounded per start, oldest first, filter-aware. Once repaired, the successor's +/// floor covers the seam on its own and 's seam probe — which exists because +/// this stitch is NOT gap-free by construction — finds nothing there and releases the raw purge gate with no +/// manual step. +/// /// Bounded, and the bound is stated. Per aggregate per start, at most one refresh policy window's /// worth of buckets (: 24 hourly, 3 daily) is refreshed, /// OLDEST FIRST — the oldest hole is the one the source's retention is about to make permanent. Anything past @@ -439,8 +454,34 @@ public static async Task RepairMaterialization continue; } + /* #4186 seam fix: a successor whose legacy is frozen (LegacyOf, non-null only for the three + SupersededHourlyRollups successors) can hold an un-materialized tail BELOW its own floor — + the seam an outage opens between the legacy's last bucket and the successor's first refresh + (see this class's doc, and RetentionArmSafetySql's, for the full shape). A hole is defined as + a gap INSIDE the materialized span, so scanning from the successor's own floor never reaches + that tail. Extend the lower bound down to the legacy's last bucket (+ one bucket width, so + the legacy's own already-materialized last bucket is not rescanned) whenever that reaches + further back than the successor's own floor; min() is a no-op once the successor's floor + overtakes the legacy's boundary on its own, so this converges to plain floor scanning as the + successor accumulates history. A legacy with nothing materialized (max(bucket) is NULL, a + frozen-but-empty legacy) leaves the floor untouched. */ + var seamFloor = floor.Value; + var legacy = LegacyOf(target.View); + if (legacy is not null) + { + using var legacyCeiling = new NpgsqlCommand($"SELECT max(bucket) FROM collect.{legacy}", connection) { CommandTimeout = SetupTimeoutSeconds }; + if (await legacyCeiling.ExecuteScalarAsync(cancellationToken) is DateTime legacyMaxBucket) + { + var seamBound = legacyMaxBucket + target.BucketWidth; + if (seamBound < seamFloor) + { + seamFloor = seamBound; + } + } + } + var horizon = AlignDown(utcNow - MaterializationHoleScanSpanFor(target.Source), target.BucketWidth); - var from = floor.Value > horizon ? floor.Value : horizon; + var from = seamFloor > horizon ? seamFloor : horizon; var to = ceiling.Value; if (from > to) { diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 190f0ab527..7e18c830fb 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -6017,16 +6017,29 @@ FROM timescaledb_information.jobs AS j /// An empty new rollup would vanish from the comparison and the gate would pass on the old rollup alone — /// which is the whole failure this exists to prevent. /// - /// Stitched coverage for frozen-legacy + successor pairs (#3653 high-fix, Option 4). - /// When a coverage relation is an interval-honest successor (a member of - /// ), the SQL for that coverage slot is - /// COALESCE(LEAST(legacy.min, successor.min), legacy.min, successor.min) rather than - /// min(successor.bucket) alone. This lets an upgrading store whose successor is still empty - /// fall back to the frozen legacy's floor — the legacy was refreshing up to the freeze point and - /// covers all of raw's history, so the stitch is gap-free by construction (successor was started from - /// , which overlaps the freeze). Once the successor accumulates - /// enough history to cover raw on its own, the legacy term becomes the minimum anyway and the result is - /// unchanged. Both empty → NULL → Short (correct: fresh install, no history yet). + /// Stitched coverage for frozen-legacy + successor pairs (#3653 high-fix, Option 4; seam-checked + /// since #4186). When a coverage relation is an interval-honest successor (a member of + /// ), the SQL for that coverage slot stitches the legacy's floor with + /// the successor's so an upgrading store whose successor is still empty can fall back to the frozen + /// legacy's floor rather than reading Short forever. + /// + /// The stitch is NOT gap-free by construction — an earlier version of this comment claimed it + /// was, and that was false. A store stopped longer than before + /// the successor's first refresh leaves raw rows between the legacy's last bucket and the successor's + /// floor that NEITHER side ever materializes (the outage shape MaterializationHoles.cs's class doc + /// works through in full). An unconditional stitch reports that seam Covered right up to the moment the + /// raw purge destroys it. So the SQL probes raw itself, filtered the same way + /// filters everything else here, for a row at or after the legacy's last bucket and before the successor's + /// floor (or +infinity when the successor is empty): none found → LEAST(legacy.min, successor.min), + /// the stitched floor, honest because the seam really is empty; a row found → the successor's OWN + /// min(bucket) alone (NULL when empty, giving Short), because the legacy cannot vouch for + /// history it never covered. Once the successor's own floor reaches back past the legacy's last bucket, + /// there is no seam left to hold a row, the probe always comes back empty, and the two branches converge — + /// the same steady state the old unconditional stitch reached, just no longer assumed along the way. Both + /// empty → NULL → Short (correct: fresh install, no history yet). The seam is what + /// repairs, by scanning a stitched successor from the + /// legacy's last bucket instead of its own floor; once that repair runs, the seam is empty and this gate's + /// fallback releases on its own, no manual step. /// /// Source filter for 0-interval rows. For query_stats and procedure_stats the /// source_oldest subquery adds WHERE to exclude @@ -6050,14 +6063,39 @@ filter to source_oldest. query_store_stats has no such filter in its hourly CREA /* Coverage SQL: for a coverage relation that is an interval-honest successor, stitch it with its frozen legacy so an empty successor on an upgrading store falls back to the legacy's - floor. LegacyOf() returns null for any relation that is not in SupersededHourlyRollups - (dailies, query_store_stats, baseline aggregates), keeping those as simple min(bucket). */ + floor — BUT ONLY WHEN THE SEAM IS EMPTY (#4186). The legacy stopped refreshing at the + freeze; if the store was down past HourlyRefreshStartOffset before the successor's first + refresh, the raw rows collected between the legacy's last bucket and the successor's floor + were never materialized by either side (see MaterializationHoles.cs's class doc for the + shape). A stitch that ignores that seam reports Covered over history nobody holds. + So: probe raw for a filter-admitted row at or after the legacy's last bucket (+ one hour, so + the legacy's own last bucket is not re-counted as seam) and before the successor's floor (or + +infinity when the successor is empty). Any such row means the seam is NOT gap-free, so the + slot falls back to the successor's OWN min(bucket) — NULL when empty, which reports Short + honestly instead of the false Covered the unconditional stitch gave. LEGACY.mx is NULL for a + legacy that never materialized anything; the comparison against it is then UNKNOWN for every + row, EXISTS is false, and the stitch behaves exactly as it did before this seam probe existed. + LegacyOf() returns null for any relation that is not in SupersededHourlyRollups (dailies, + query_store_stats, baseline aggregates), keeping those as simple min(bucket). PostgreSQL's + LEAST ignores NULL arguments (returns NULL only when EVERY argument is NULL), so + LEAST(l.mn, s.mn) already is what COALESCE(LEAST(l.mn, s.mn), l.mn, s.mn) computed. */ var columns = coverageRelations.Select((c, i) => { var legacy = LegacyOf(c); var subquery = legacy is not null - ? $"(SELECT COALESCE(LEAST(l.mn, s.mn), l.mn, s.mn){Environment.NewLine}" - + $" FROM (SELECT min(bucket) AS mn FROM collect.{legacy}) l{Environment.NewLine}" + ? $"(SELECT CASE{Environment.NewLine}" + + $" WHEN EXISTS ({Environment.NewLine}" + + $" SELECT 1{Environment.NewLine}" + + $" FROM collect.{relation} AS seam{Environment.NewLine}" + + $" WHERE seam.{sourceTimeColumn} >= l.mx + INTERVAL '1 hour'{Environment.NewLine}" + + $" AND seam.{sourceTimeColumn} < COALESCE(s.mn, 'infinity'::timestamp){Environment.NewLine}" + + $" AND {IntervalHonestSourceFilter}{Environment.NewLine}" + + $" ORDER BY seam.{sourceTimeColumn}{Environment.NewLine}" + + $" LIMIT 1){Environment.NewLine}" + + $" THEN s.mn{Environment.NewLine}" + + $" ELSE LEAST(l.mn, s.mn){Environment.NewLine}" + + $" END{Environment.NewLine}" + + $" FROM (SELECT min(bucket) AS mn, max(bucket) AS mx FROM collect.{legacy}) l{Environment.NewLine}" + $" CROSS JOIN (SELECT min(bucket) AS mn FROM collect.{c}) s)" : $"(SELECT min(bucket) FROM collect.{c})"; return $" {subquery} AS coverage_oldest_{i}"; From c14b46a0b7bbe568c785b4d160e252ea076e4dd2 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 09:08:04 -0400 Subject: [PATCH 15/18] Fix #4186 follow-up: scan the outage seam even below the horizon RepairMaterializationHolesAsync's seam bound (3098a9e5) still clamped the scan to `from = max(seamFloor, horizon)`. For a store stopped more than the raw span (4 days for query_stats/procedure_stats) before its first start on this version, the seam tail lies below that horizon and was never scanned, so RetentionArmSafetySql's seam probe keeps finding the un-repaired rows and the raw purge gate holds forever without a manual --backfill-rollups. MaterializationHoleScanWindows is a new pure helper: the ordinary window stays exactly [max(floor, horizon), ceiling], and a seam window [seamFloor, floor) is added whenever a seam exists, scanned in full however far below the horizon it reaches. RepairMaterializationHolesAsync now scans each window and concatenates the holes before merging and capping, unchanged from there. Adds pure tests for the new helper (no seam, seam above the horizon, seam below it, the ordinary window never starting below the horizon) and a 6-day-outage live twin of Outage_SeamBetweenFrozenLegacyAndSuccessor_ HoleWalkRepairsItAndGateReleases, which only ever exercised the 2-day (short) case. Corrects both doc comments' "releases on its own" claim to state the automatic release is bounded by the per-start repair cap, and the AggregatesSkipped doc's horizon-skip reason to exempt a seam window. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../Darling.Tests/FrozenRollupLiveTests.cs | 95 ++++++++++++++++++ .../MaterializationHoleRepairTests.cs | 95 ++++++++++++++++++ .../TimescaleSupport.MaterializationHoles.cs | 99 +++++++++++++++---- .../TimescaleSupport.cs | 11 ++- 4 files changed, 278 insertions(+), 22 deletions(-) diff --git a/Darling/Darling.Tests/FrozenRollupLiveTests.cs b/Darling/Darling.Tests/FrozenRollupLiveTests.cs index ff6a5b117d..d2ac58280e 100644 --- a/Darling/Darling.Tests/FrozenRollupLiveTests.cs +++ b/Darling/Darling.Tests/FrozenRollupLiveTests.cs @@ -540,6 +540,101 @@ await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async ( } } + /// + /// #4186 follow-up: the same seam shape as + /// , but the outage + /// outlasts the raw retention horizon itself (: + /// 4 days for procedure_stats) instead of sitting comfortably inside it. The seam fix's first cut + /// folded the seam's lower bound into the SAME max(_, horizon) the ordinary scan clamps to, so a seam + /// older than the horizon was silently dropped every start and the gate held the raw purge forever with no + /// manual step ever releasing it. The 2-day twin above never exercises that clamp — its horizon sits days + /// before the seam — so it only proves the short case; this test's outage is 6 days, longer than the + /// 4-day span, and deliberately exercises it. + /// + [Fact] + public async Task Outage_SeamOlderThanTheHorizon_HoleWalkStillRepairsItAndGateReleases() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* Same seam shape as the 2-day twin: S is the stop, the legacy is refreshed through S-2h, raw + carries the [S-1h, S] tail plus a rejected interval-0 restart row. The outage here runs 6 days — + longer than procedure_stats' 4-day raw span — so the successor's first post-upgrade refresh + lands at U = S+6d, and the seam sits entirely below the horizon RepairMaterializationHolesAsync + computes from U. */ + var s = D0.AddDays(3); + + for (var hour = 0; hour <= 6; hour++) + { + await InsertProcedureStatsAsync(connection, s.AddHours(-hour), $"seam6_proc_{hour}", 900, 9, 3600, ct); + } + + await InsertProcedureStatsAsync(connection, s.AddMinutes(30), "seam6_proc_restart", 0, 0, 0, ct); + + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, s.AddHours(-6), s.AddHours(-1), ct); + + var u = s.AddDays(6); + await InsertProcedureStatsAsync(connection, u.AddHours(-1), "seam6_proc_successor_floor", 900, 9, 3600, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, u.AddDays(-1), u, ct); + + /* The seam holds raw rows the stitch cannot see through unconditionally — Short, not Covered. */ + Assert.False(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + /* The product's own start-path entry point (DarlingWorker calls this exact method). Before the + #4186 follow-up fix, the seam's lower bound was clamped to U minus the 4-day span — comfortably + ABOVE the seam, since the outage is 6 days — so this repaired 0 buckets and the seam stood + forever without a manual --backfill-rollups. */ + var summary = await TimescaleSupport.RepairMaterializationHolesAsync(connection, null, u, ct); + Assert.True(summary.BucketsRepaired >= 2, $"expected the 2-bucket seam tail to be repaired, got {summary.BucketsRepaired}"); + + await using (var span = new NpgsqlCommand($"SELECT min(bucket) FROM collect.{TimescaleSupport.ProcedureStatsIntervalHourlyView}", connection)) + { + var newFloor = (DateTime)(await span.ExecuteScalarAsync(ct))!; + Assert.Equal(s.AddHours(-1), newFloor); + } + + /* The seam is now empty (the successor's own floor reaches the legacy's boundary) — Covered. */ + Assert.True(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + /// /// #4186: the no-deadlock twin of — /// same frozen-legacy-plus-outage shape, but the seam itself holds NO raw rows (an outage that began diff --git a/Darling/Darling.Tests/MaterializationHoleRepairTests.cs b/Darling/Darling.Tests/MaterializationHoleRepairTests.cs index c348022af9..c89a6febf8 100644 --- a/Darling/Darling.Tests/MaterializationHoleRepairTests.cs +++ b/Darling/Darling.Tests/MaterializationHoleRepairTests.cs @@ -213,6 +213,101 @@ public void AlignDown_LandsOnABucketBoundary_ForBothWidths() Assert.Throws(() => TimescaleSupport.AlignDown(instant, TimeSpan.Zero)); } + [Fact] + public void ScanWindows_NoSeam_GivesTheOrdinaryWindowOnly() + { + var width = TimescaleSupport.HourlyBucket; + + /* floor newer than the horizon: the ordinary window starts at floor. */ + var floor = Hour.AddHours(50); + var horizon = Hour; + var ceiling = Hour.AddHours(80); + Assert.Equal( + new[] { (floor, ceiling) }, + TimescaleSupport.MaterializationHoleScanWindows(floor, ceiling, horizon, seamFloor: floor, width)); + + /* floor older than the horizon (the ordinary, pre-#4186 clamp): the window starts at the horizon. */ + var oldFloor = Hour; + var laterHorizon = Hour.AddHours(20); + Assert.Equal( + new[] { (laterHorizon, ceiling) }, + TimescaleSupport.MaterializationHoleScanWindows(oldFloor, ceiling, laterHorizon, seamFloor: oldFloor, width)); + } + + [Fact] + public void ScanWindows_SeamAboveTheHorizon_GivesTwoWindows_SeamThenOrdinary() + { + var width = TimescaleSupport.HourlyBucket; + var floor = Hour.AddHours(10); + var horizon = Hour.AddHours(2); + var seamFloor = Hour.AddHours(4); + var ceiling = Hour.AddHours(50); + + var windows = TimescaleSupport.MaterializationHoleScanWindows(floor, ceiling, horizon, seamFloor, width); + + Assert.Equal(new[] { (seamFloor, floor.AddHours(-1)), (floor, ceiling) }, windows); + Assert.True(windows[0].From >= horizon, "the seam sits above the horizon in this case — a sanity check, not the regression this pins"); + } + + [Fact] + public void ScanWindows_SeamBelowTheHorizon_GivesTwoWindows_TheSeamUnclamped() + { + /* #4186 follow-up: a store stopped more than the raw span (4 days = 96h here) before its successor's + first start. The seam predates the horizon entirely — the exact shape the clamp used to swallow. */ + var width = TimescaleSupport.HourlyBucket; + var horizon = Hour.AddHours(96); + var floor = Hour.AddHours(150); + var seamFloor = Hour.AddHours(10); + var ceiling = Hour.AddHours(200); + + var windows = TimescaleSupport.MaterializationHoleScanWindows(floor, ceiling, horizon, seamFloor, width); + + Assert.Equal(new[] { (seamFloor, floor.AddHours(-1)), (floor, ceiling) }, windows); + + /* The regression itself: the seam window's From reaches below the horizon rather than clamping to it — + the pre-fix formula (from = max(seamFloor, horizon)) would have reported Hour.AddHours(96) here. */ + Assert.True(windows[0].From < horizon); + Assert.Equal(seamFloor, windows[0].From); + } + + [Fact] + public void ScanWindows_TheOrdinaryWindow_NeverStartsBelowTheHorizon() + { + var width = TimescaleSupport.HourlyBucket; + var ceiling = Hour.AddHours(500); + + foreach (var (floor, horizon, seamFloor) in new[] + { + (Hour.AddHours(50), Hour, Hour.AddHours(50)), /* no seam, floor above horizon */ + (Hour, Hour.AddHours(20), Hour), /* no seam, floor below horizon */ + (Hour.AddHours(10), Hour.AddHours(2), Hour.AddHours(4)), /* seam above horizon */ + (Hour.AddHours(150), Hour.AddHours(96), Hour.AddHours(10)), /* seam below horizon */ + }) + { + var windows = TimescaleSupport.MaterializationHoleScanWindows(floor, ceiling, horizon, seamFloor, width); + var ordinary = windows[^1]; + Assert.True(ordinary.From >= horizon); + Assert.Equal(ordinary.From, floor > horizon ? floor : horizon); + } + } + + [Fact] + public void ScanWindows_EdgeCases_OneBucketSeam_NoWindowsWhenNothingToScan_AndTheWidthGuard() + { + var width = TimescaleSupport.HourlyBucket; + + /* A seam exactly one bucket wide still yields a valid, single-bucket window. */ + var floor = Hour.AddHours(5); + var oneBucketSeam = floor.AddHours(-1); + var windows = TimescaleSupport.MaterializationHoleScanWindows(floor, Hour.AddHours(50), Hour, oneBucketSeam, width); + Assert.Equal(new[] { (oneBucketSeam, oneBucketSeam), (floor, Hour.AddHours(50)) }, windows); + + /* No seam and the whole span is older than the horizon: nothing to scan. */ + Assert.Empty(TimescaleSupport.MaterializationHoleScanWindows(Hour, Hour.AddHours(5), Hour.AddHours(20), seamFloor: Hour, width)); + + Assert.Throws(() => TimescaleSupport.MaterializationHoleScanWindows(Hour, Hour, Hour, Hour, TimeSpan.Zero)); + } + /// /// The start path: launched (not awaited) right after the ensure, on its own connection, inside the /// TimescaleDB block, before the compression and retention ensures; drained at shutdown beside the baseline diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs index 6d798443af..7328888946 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs @@ -55,20 +55,26 @@ namespace PerformanceMonitor.Darling.Storage; /// business and are never touched here; buckets past the last materialized one are the live edge the policy /// owns. /// -/// The seam case (#4186): a successor with a frozen legacy is scanned from the LEGACY's last bucket, -/// not its own floor. The three interval-honest successors (SupersededHourlyRollups) can each have -/// an un-materialized TAIL below their own floor: an outage that outlasts -/// before the successor's first refresh leaves raw rows between the frozen legacy's last bucket and the -/// successor's floor that neither side ever materialized (the outage shape worked through above). Those rows -/// sit BELOW the successor's floor, so the ordinary "floor up to ceiling" scan never reaches them — a hole, by -/// this pass's own definition, has to be inside the materialized span. For a successor found through -/// LegacyOf, the scan's lower bound is extended down to the legacy's last bucket (plus one bucket width, -/// so the legacy's own last bucket is not re-scanned as if it were the successor's) whenever that reaches -/// further back than the successor's floor already does. The seam tail then reads as an ordinary hole and the -/// existing machinery repairs it: bounded per start, oldest first, filter-aware. Once repaired, the successor's -/// floor covers the seam on its own and 's seam probe — which exists because -/// this stitch is NOT gap-free by construction — finds nothing there and releases the raw purge gate with no -/// manual step. +/// The seam case (#4186): a successor with a frozen legacy gets a SECOND scan window down to the +/// LEGACY's last bucket, scanned however far below the horizon it reaches. The three interval-honest +/// successors (SupersededHourlyRollups) can each have an un-materialized TAIL below their own floor: an +/// outage that outlasts before the successor's first refresh leaves raw +/// rows between the frozen legacy's last bucket and the successor's floor that neither side ever materialized +/// (the outage shape worked through above). Those rows sit BELOW the successor's floor, so the ordinary "floor +/// up to ceiling" scan never reaches them — a hole, by this pass's own definition, has to be inside the +/// materialized span. For a successor found through LegacyOf, +/// adds a seam window down to the legacy's last bucket (plus one bucket width, so the legacy's own last bucket is +/// not re-scanned as if it were the successor's) whenever that reaches further back than the successor's floor +/// already does — UNCLAMPED by the horizon that still bounds the ordinary window, because an outage longer than +/// the horizon's own span is exactly the shape that needs repairing, not a shape to skip (an earlier cut folded +/// the seam into that same horizon clamp, and a seam older than the horizon was silently never scanned — see +/// 's own doc for that history). The seam tail then reads as an +/// ordinary hole and the existing machinery repairs it: bounded per start, oldest first, filter-aware — a seam +/// wider than one start's cap () takes more than one start to +/// close in full, but every start makes progress on it. Once repaired, the successor's floor covers the seam on +/// its own and 's seam probe — which exists because this stitch is NOT +/// gap-free by construction — finds nothing there and releases the raw purge gate automatically, with no manual +/// step, bounded only by that same per-start repair cap. /// /// Bounded, and the bound is stated. Per aggregate per start, at most one refresh policy window's /// worth of buckets (: 24 hourly, 3 daily) is refreshed, @@ -359,6 +365,56 @@ public static (IReadOnlyList<(DateTime Start, DateTime End)> Repair, IReadOnlyLi return (repair, deferred); } + /// + /// The scan window(s) for one aggregate this start, given its own and + /// , the horizon computes for its + /// source, and the seam bound (, equal to when the + /// aggregate has no frozen legacy or the legacy's own last bucket does not reach back past the floor). + /// Pure, so the tests can walk it. + /// + /// Two windows, not one (#4186 follow-up). The seam fix's first cut folded the seam into the + /// SAME max(_, horizon) the ordinary scan already clamps to — from = max(seamFloor, horizon) + /// — which reads right for an outage shorter than the horizon (: + /// 4 days of raw retention for query_stats/procedure_stats) and silently drops the rest of the + /// seam for a longer one: a store stopped more than 4 days before its first start on this version has its + /// seam tail clamped away every single start, the gate's seam probe () + /// keeps finding the un-repaired rows, and the raw purge never releases on its own. The horizon exists to + /// skip source rows the raw retention has already purged — scanning past it wastes a probe on a bucket that + /// cannot be repaired. A seam is not that case: the outage that opened it left the source with no rows + /// there at all (not purged, empty), and probing an empty span costs one cheap index range per bucket + /// whatever its age. So the seam gets its OWN window, [seamFloor, floor), scanned in full however far + /// below the horizon it reaches, while the ordinary window stays exactly [max(floor, horizon), ceiling] + /// — the successor's own span below the horizon is the source retention's business, not this repair's, and + /// widening it was never the fix. + /// + public static IReadOnlyList<(DateTime From, DateTime To)> MaterializationHoleScanWindows( + DateTime floor, DateTime ceiling, DateTime horizon, DateTime seamFloor, TimeSpan bucketWidth) + { + if (bucketWidth <= TimeSpan.Zero) + { + throw new ArgumentOutOfRangeException(nameof(bucketWidth), bucketWidth, "a bucket has a positive width"); + } + + var windows = new List<(DateTime From, DateTime To)>(); + + if (seamFloor < floor) + { + var seamTo = floor - bucketWidth; + if (seamFloor <= seamTo) + { + windows.Add((seamFloor, seamTo)); + } + } + + var ordinaryFrom = floor > horizon ? floor : horizon; + if (ordinaryFrom <= ceiling) + { + windows.Add((ordinaryFrom, ceiling)); + } + + return windows; + } + /// /// What one start's pass did — the whole tally the worker's one unconditional summary line reports (#3756) /// and the live test asserts. Every count the line names is carried here rather than re-derived by the @@ -366,7 +422,8 @@ public static (IReadOnlyList<(DateTime Start, DateTime End)> Repair, IReadOnlyLi /// /// is the aggregates the scan actually probed ("walked"); /// the ones it had a reason not to (not a continuous aggregate on this - /// store, never materialized, or a span entirely below the source's horizon). + /// store, never materialized, or — absent a seam — a span entirely below the source's horizon; a seam + /// window is never skipped for that reason). /// counts the contiguous hole RANGES the scan saw across every aggregate, BEFORE the cap, and /// the buckets those ranges span — so BucketsFound == BucketsRepaired + /// BucketsDeferred always, while HolesRepaired + HolesDeferred exceeds HolesFound by one @@ -481,16 +538,20 @@ successor accumulates history. A legacy with nothing materialized (max(bucket) i } var horizon = AlignDown(utcNow - MaterializationHoleScanSpanFor(target.Source), target.BucketWidth); - var from = seamFloor > horizon ? seamFloor : horizon; - var to = ceiling.Value; - if (from > to) + var windows = MaterializationHoleScanWindows(floor.Value, ceiling.Value, horizon, seamFloor, target.BucketWidth); + if (windows.Count == 0) { skipped++; continue; } scanned++; - var holes = await ScanHolesAsync(connection, target, materialization.Value, from, to, cancellationToken); + var holes = new List(); + foreach (var (from, to) in windows) + { + holes.AddRange(await ScanHolesAsync(connection, target, materialization.Value, from, to, cancellationToken)); + } + if (holes.Count == 0) { continue; diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 7e18c830fb..9c56924a17 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -6037,9 +6037,14 @@ FROM timescaledb_information.jobs AS j /// there is no seam left to hold a row, the probe always comes back empty, and the two branches converge — /// the same steady state the old unconditional stitch reached, just no longer assumed along the way. Both /// empty → NULL → Short (correct: fresh install, no history yet). The seam is what - /// repairs, by scanning a stitched successor from the - /// legacy's last bucket instead of its own floor; once that repair runs, the seam is empty and this gate's - /// fallback releases on its own, no manual step. + /// repairs, by giving a stitched successor a SEAM scan window + /// down to the legacy's last bucket, unclamped by the horizon that bounds its ordinary scan window — so an + /// outage longer than that horizon's own span (#4186 follow-up: the first cut folded the seam into the same + /// horizon clamp as the ordinary window, and a seam older than it was silently never scanned) still gets + /// repaired instead of clamped away. Once that repair runs, the seam is empty and this gate's fallback + /// releases automatically, with no manual step, bounded only by the repair's own per-start cap — a seam + /// wider than one start's cap takes more than one start to close in full, but every start makes progress on + /// it. /// /// Source filter for 0-interval rows. For query_stats and procedure_stats the /// source_oldest subquery adds WHERE to exclude From 22cbb0284292ca087ac33764683d8a0f27884d54 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 09:08:50 -0400 Subject: [PATCH 16/18] Fix two Low findings from the #4186 review round MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit FrozenRollupAggregates' doc said "the four still with a drop_chunks policy"; RetentionPolicies names three (the legacy hourlies), not four. The aggregate setup loop's catch logged "composer queries fall back to raw scans" for every failure, including a frozen rollup's own refresh- policy detach. That view's CREATE already succeeded earlier in the same try, so it keeps serving composer queries same as ever on a detach failure — it just stays on its pre-freeze refresh schedule instead of frozen until a later start retries the detach. Branch on IsFrozenRollupAggregate and log that case truthfully; every other view keeps the existing message. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../TimescaleSupport.cs | 21 +++++++++++++++---- 1 file changed, 17 insertions(+), 4 deletions(-) diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 9c56924a17..bb1c6f843d 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -1737,7 +1737,7 @@ any other member here now that LC has removed the deferral that used to hold the /// and the materialization-hole repair targets derived from it — there /// is nothing left to backfill or repair once nothing ever advances the watermark those verbs converge /// toward. Their own retention and compression are named where each already lived: - /// for the four still with a drop_chunks policy (the two legacy hourlies + /// for the three still with a drop_chunks policy (the legacy hourlies /// this raw purge moved off of keep their OWN retention unchanged — only 's /// raw-purge COVERAGE moved to the successors), and the frozen dailies' compression drain rule /// () for the daily half. @@ -5431,9 +5431,22 @@ missing from HourlyRefreshPhaseOrder costs that one aggregate and names it in th } catch (Exception ex) when (ex is not OperationCanceledException) { - logger?.LogWarning( - "Continuous aggregate {View} setup failed — composer queries fall back to raw scans: {Message}", - view, ex.Message); + /* A frozen rollup (policyFor is null) already exists by the time its own detach can throw — its + CREATE ran first, in this same try. "Falls back to raw scans" is false for that case: the + view keeps serving composer queries same as ever, just still on its pre-freeze refresh + schedule instead of frozen, until a later start's retry detaches it. */ + if (IsFrozenRollupAggregate(view)) + { + logger?.LogWarning( + "Could not detach the refresh policy from frozen rollup {View}; it keeps refreshing on its old schedule until a later start detaches it: {Message}", + view, ex.Message); + } + else + { + logger?.LogWarning( + "Continuous aggregate {View} setup failed — composer queries fall back to raw scans: {Message}", + view, ex.Message); + } } } From 35a0aefb35aeb460b0f2958591f0258da3c6dafe Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 09:27:58 -0400 Subject: [PATCH 17/18] Frozen rollup warning: name the create step too The #4186 Low fix reworded the frozen-view branch of the continuous-aggregate setup catch to talk about the refresh-policy detach only. The same catch also sees a failed CREATE (a fresh store) or the width step, so the line now names both steps instead of claiming the view exists. Co-Authored-By: Claude Opus 5.5 Claude-Session: https://claude.ai/code/session_01FVjn4PBJN71NQXdFo6ZxNQ --- .../TimescaleSupport.cs | 11 ++++++----- 1 file changed, 6 insertions(+), 5 deletions(-) diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index 14ff699aca..c7a2981680 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -5431,14 +5431,15 @@ missing from HourlyRefreshPhaseOrder costs that one aggregate and names it in th } catch (Exception ex) when (ex is not OperationCanceledException) { - /* A frozen rollup (policyFor is null) already exists by the time its own detach can throw — its - CREATE ran first, in this same try. "Falls back to raw scans" is false for that case: the - view keeps serving composer queries same as ever, just still on its pre-freeze refresh - schedule instead of frozen, until a later start's retry detaches it. */ + /* A frozen rollup (policyFor is null) has no raw-scan fallback to name: on an existing store its + CREATE is a no-op and the throw is the policy detach, so the view keeps serving composer + queries on its pre-freeze refresh schedule until a later start's retry detaches it. The same + catch also sees a failed CREATE (a fresh store) or width step, so the line names both steps + rather than claiming the view exists. */ if (IsFrozenRollupAggregate(view)) { logger?.LogWarning( - "Could not detach the refresh policy from frozen rollup {View}; it keeps refreshing on its old schedule until a later start detaches it: {Message}", + "Could not create frozen rollup {View} or detach its refresh policy; a later start retries it, and until then it keeps any refresh policy it had: {Message}", view, ex.Message); } else From c3291f6686c0950231bac593c23457b1698a1053 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Fri, 25 Sep 2026 11:45:46 -0400 Subject: [PATCH 18/18] Seam repair walks newest-first and stops at the first failed range (#4186 round-3 H1) A seam wider than one start's repair cap, or with two or more ranges, could move the successor's floor past a still-unrepaired seam bucket: oldest-first capping and repair let the OLDEST buckets close first, which drags the floor (a bare min(bucket)) down to them even while newer seam buckets in between stay holes, and RetentionArmSafetySql's probe stops looking above the new floor. The raw purge could then arm and drop rows neither rollup ever held. Fix: the seam window's holes are now scanned, capped and walked separately from the ordinary window's. The seam takes its cap from the newest end (closest to the successor's floor) and walks descending; the ordinary window is unchanged (oldest-first, same cap, same continue-past-a-remainder handling, since an interior repair can never move the floor). The seam walk stops at the first range that throws (propagates to the existing per-target catch) or leaves buckets standing (remaining > 0) rather than moving on to an older range. Both windows share one cap budget per aggregate per start, seam first, so "a start never re-materializes more for one aggregate than an ordinary policy run does" stays true. CapMaterializationHoleRepairs gains a newestFirst parameter (default false, unchanged behavior) that also splits a straddling range at its older edge instead of its newer one, so the kept portion stays adjacent to whatever is already materialized. Also corrects RetentionArmSafetySql's doc: it claimed an over-horizon seam "still gets repaired" and the gate "releases automatically, with no manual step" unconditionally. Both are true only when the raw purge was already held when the store stopped (#4299) and the outage did not cross the upgrade (#4300) respectively. Adds the H2-parity note: the stitch trusts the frozen legacy's own span without checking it, which can read Covered over a pre-upgrade hole nothing here repairs -- accepted because 3.8.0 loses the same rows on the same schedule, and the A6 backfill (--backfill-rollups) covers it for an operator who runs it. Two live tests against a real TimescaleDB rig: - a 35-bucket seam (over the 24-bucket cap) stays Short after one walk (top 24 repaired, bottom 11 still a hole) and reads Covered only after a second walk closes the rest; revert-proved by temporarily restoring oldest-first, which fails at the same assertion. - a two-range seam where the newer range's refresh is forced to raise 23514 via a CHECK constraint on its own materialization chunk: the older range is never touched and the gate still reads not-safe. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01TszxYhJJbTEh4LrZ56NYo3 --- .../Darling.Tests/FrozenRollupLiveTests.cs | 195 ++++++++++++++++++ .../MaterializationHoleRepairTests.cs | 40 +++- .../TimescaleSupport.MaterializationHoles.cs | 120 +++++++++-- .../TimescaleSupport.cs | 29 ++- 4 files changed, 361 insertions(+), 23 deletions(-) diff --git a/Darling/Darling.Tests/FrozenRollupLiveTests.cs b/Darling/Darling.Tests/FrozenRollupLiveTests.cs index d2ac58280e..4c0471017d 100644 --- a/Darling/Darling.Tests/FrozenRollupLiveTests.cs +++ b/Darling/Darling.Tests/FrozenRollupLiveTests.cs @@ -635,6 +635,201 @@ await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async ( } } + /// + /// #4186 round-3 H1: a seam wider than one start's repair cap (24 hourly buckets) used to release the gate + /// early. Oldest-first repaired the buckets FARTHEST from the successor's floor first, which still moved + /// the floor (a bare min(bucket)) all the way down to them — stranding the un-repaired NEWER seam + /// buckets above the new floor and outside 's probe. + /// Newest-first must NOT do that: repairing the top 24 of a 35-bucket seam should leave the floor exactly + /// adjacent to the still-open 11-bucket remainder, so the probe keeps finding it and the gate stays Short + /// until a second walk closes the rest. + /// + [Fact] + public async Task Outage_SeamWiderThanTheCap_NewestFirstRepairsTheTopAndKeepsTheGateHeldUntilFullyRepaired() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* S is the stop. The legacy materializes only ONE bucket, 35 hours back, so l.mx = S-35h and the + seam floor is S-34h. Raw carries an unbroken run of 35 hourly buckets from S-34h through S — + wider than the 24-bucket cap, so ONE walk cannot close it in one pass. */ + var s = D0.AddDays(3); + + for (var hour = 0; hour <= 35; hour++) + { + await InsertProcedureStatsAsync(connection, s.AddHours(-hour), $"seam35_proc_{hour}", 900, 9, 3600, ct); + } + + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, s.AddHours(-35), s.AddHours(-34), ct); + + var u = s.AddDays(2); + await InsertProcedureStatsAsync(connection, u.AddHours(-1), "seam35_proc_successor_floor", 900, 9, 3600, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, u.AddDays(-1), u, ct); + + Assert.False(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + /* Walk 1: the cap takes the NEWEST 24 of the 35 seam buckets (S-23h..S), leaving the OLDER 11 + (S-34h..S-24h) as a hole immediately below the new floor. The product's own start-path entry + point (DarlingWorker calls this exact method). */ + var summary1 = await TimescaleSupport.RepairMaterializationHolesAsync(connection, null, u, ct); + Assert.Equal(24, summary1.BucketsRepaired); + + /* The regression this pins: with the old oldest-first order, this walk would instead have + repaired S-34h..S-11h (the OLDEST 24) and left S-10h..S as holes ABOVE the new floor — outside + the probe window entirely, and the gate below would have read true (Covered) after only one + walk. */ + Assert.False(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + /* Walk 2: the remaining 11-bucket range is now the whole seam (S-34h..S-24h, under the cap), and + closes it completely. */ + var summary2 = await TimescaleSupport.RepairMaterializationHolesAsync(connection, null, u, ct); + Assert.Equal(11, summary2.BucketsRepaired); + + Assert.True(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + + /// + /// #4186 round-3 H1, the failure arm: a seam with TWO separate ranges, where the NEWER one fails. A CHECK + /// constraint added directly to the successor's own materialization chunk (the mechanism verified against + /// this rig's TimescaleDB before this test was written — ALTER TABLE against the continuous + /// aggregate's VIEW or its parent materialization hypertable is refused, but a concrete chunk table takes + /// one) makes the refresh over the newer range raise 23514, the same way a real refresh failure would. + /// Newest-first must stop there: the older range must be left completely untouched, and the gate must + /// still read not-safe, exactly as if neither range had been attempted. + /// + [Fact] + public async Task Outage_SeamTwoRanges_NewerRangeFails_OlderRangeUntouchedAndGateStillNotSafe() + { + var baseConnectionString = Environment.GetEnvironmentVariable("DARLING_TEST_PG"); + Assert.SkipWhen(string.IsNullOrEmpty(baseConnectionString), + "Set DARLING_TEST_PG to a Postgres connection string (with TimescaleDB installed) to run the live A6 freeze test."); + + var ct = TestContext.Current.CancellationToken; + + await using var scratch = await ScratchPostgres.CreateAsync(baseConnectionString!, ct); + await using var connection = new NpgsqlConnection(scratch.ConnectionString); + await connection.OpenAsync(ct); + await PgMigrations.MigrateAsync(connection, ct); + + var timescaleEnabled = await TimescaleSupport.TryEnableAsync(connection, null, ct); + Assert.SkipWhen(!timescaleEnabled, "The live A6 freeze test needs TimescaleDB."); + await TimescaleSupport.ConvertToHypertablesAsync(connection, null, ct); + Assert.True(await TimescaleSupport.EnsureCollectionLogHypertableAsync(connection, null, ct)); + + await using (var stop = new NpgsqlCommand("SELECT _timescaledb_functions.stop_background_workers()", connection)) + { + await stop.ExecuteNonQueryAsync(ct); + } + + await DarlingMcpTestData.RegisterServerAsync(connection, ServerId, ServerName, ct); + await TimescaleSupport.EnsureContinuousAggregatesAsync(connection, null, ct); + + var bodySucceeded = false; + try + { + /* S is the stop. The legacy materializes one bucket 20 hours back (l.mx = S-20h, seam floor + S-19h). Range A sits right above the seam floor — the OLDEST part of the seam. */ + var s = D0.AddDays(3); + await InsertProcedureStatsAsync(connection, s.AddHours(-20), "seam2_proc_legacy_floor", 900, 9, 3600, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsHourlyView, s.AddHours(-20), s.AddHours(-19), ct); + + await InsertProcedureStatsAsync(connection, s.AddHours(-19), "seam2_proc_a0", 900, 9, 3600, ct); + await InsertProcedureStatsAsync(connection, s.AddHours(-18), "seam2_proc_a1", 900, 9, 3600, ct); + + /* The successor's own first refresh sets floor = U-1h and, in the same call, creates the + materialization chunk for that whole calendar day — the chunk range B (below) also falls in. */ + var u = s.AddDays(2); + await InsertProcedureStatsAsync(connection, u.AddHours(-1), "seam2_proc_floor", 900, 9, 3600, ct); + await RefreshAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, u.AddDays(-1), u, ct); + + /* Range B: two buckets below the floor, on the SAME calendar day as it, separated from range A by + a multi-day gap where raw holds nothing (so A and B are two DISTINCT merged ranges, not one). A + CHECK constraint on B's exact bucket span, added to the chunk the floor refresh just created, + blocks it before it ever holds a row. */ + var chunks = await ReadMaterializationChunksAsync(connection, TimescaleSupport.ProcedureStatsIntervalHourlyView, ct); + Assert.Single(chunks); + await using (var block = new NpgsqlCommand( + $"ALTER TABLE {chunks[0]} ADD CONSTRAINT seam2_block_b CHECK (bucket NOT BETWEEN '{u.AddHours(-5):O}'::timestamp AND '{u.AddHours(-4):O}'::timestamp)", + connection)) + { + await block.ExecuteNonQueryAsync(ct); + } + + await InsertProcedureStatsAsync(connection, u.AddHours(-5), "seam2_proc_b0", 900, 9, 3600, ct); + await InsertProcedureStatsAsync(connection, u.AddHours(-4), "seam2_proc_b1", 900, 9, 3600, ct); + + Assert.False(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + /* Newest-first tries B (closer to the floor) before A. B's refresh raises 23514, caught by this + target's own per-aggregate catch — the walk must stop there, never reaching A. */ + var summary = await TimescaleSupport.RepairMaterializationHolesAsync(connection, null, u, ct); + Assert.True(summary.Failures >= 1, $"expected B's constraint violation to be caught as a failure, got {summary.Failures}"); + + /* A is untouched: no row near it exists in the successor at all. */ + await using (var probeA = new NpgsqlCommand( + $"SELECT count(*) FROM collect.{TimescaleSupport.ProcedureStatsIntervalHourlyView} WHERE bucket >= '{s.AddHours(-19):O}'::timestamp AND bucket < '{s.AddHours(-17):O}'::timestamp", + connection)) + { + Assert.Equal(0L, (long)(await probeA.ExecuteScalarAsync(ct))!); + } + + /* The floor never moved off U-1h — B's insert rolled back with its transaction, so both A and B + still sit inside the probe window, and the gate must still read not-safe. */ + Assert.False(await TimescaleSupport.IsRawTierDropSafeAsync(connection, "procedure_stats", ct)); + + bodySucceeded = true; + } + finally + { + await LiveStoreCleanup.RunAsync(scratch.ConnectionString, bodySucceeded, async (cleanup, cleanupCt) => + { + await using var probe = new NpgsqlCommand( + "SELECT count(*) FROM pg_catalog.pg_stat_activity WHERE datname = pg_catalog.current_database() " + + "AND backend_type LIKE 'TimescaleDB Background Worker Scheduler%'", cleanup); + var schedulers = Convert.ToInt64(await probe.ExecuteScalarAsync(cleanupCt)); + Assert.Equal(0L, schedulers); + }); + } + } + /// /// #4186: the no-deadlock twin of — /// same frozen-legacy-plus-outage shape, but the seam itself holds NO raw rows (an outage that began diff --git a/Darling/Darling.Tests/MaterializationHoleRepairTests.cs b/Darling/Darling.Tests/MaterializationHoleRepairTests.cs index c89a6febf8..9c27188434 100644 --- a/Darling/Darling.Tests/MaterializationHoleRepairTests.cs +++ b/Darling/Darling.Tests/MaterializationHoleRepairTests.cs @@ -202,6 +202,39 @@ public void TheCap_TakesTheOldestFirst_SplitsAStraddlingRangeExactly_AndDefersTh Assert.Throws(() => TimescaleSupport.CapMaterializationHoleRepairs(null!, 24, width)); } + /// + /// #4186 round-3 H1: the seam window's own direction. Mirrors + /// exactly, with + /// newestFirst: true — same three ranges, same cap, but the NEWEST 24 buckets are kept and a + /// straddling range splits at its OLDER edge instead of its newer one, so the kept portion stays adjacent + /// to whatever sits above it (the successor's already-materialized span in the real caller). + /// + [Fact] + public void TheCap_NewestFirst_TakesTheNewestFirst_SplitsAStraddlingRangeAtItsOlderEdge_AndDefersTheRest() + { + var width = TimescaleSupport.HourlyBucket; + var ranges = new List<(DateTime Start, DateTime End)> + { + (Hour.AddHours(30), Hour.AddHours(40)), /* newest, 10 buckets */ + (Hour, Hour.AddHours(20)), /* oldest, 20 buckets */ + (Hour.AddHours(22), Hour.AddHours(28)), /* middle, 6 buckets */ + }; + + var (repair, deferred) = TimescaleSupport.CapMaterializationHoleRepairs(ranges, 24, width, newestFirst: true); + + /* Newest 10 whole, then the middle's 6 whole (16 spent), then 8 of the oldest's 20 — split at its NEWER + edge, since newestFirst keeps the portion adjacent to what is already above it (24 = 10 + 6 + 8) — + the older remaining 12 hours of the oldest range deferred. */ + Assert.Equal(new[] { (Hour.AddHours(30), Hour.AddHours(40)), (Hour.AddHours(22), Hour.AddHours(28)), (Hour.AddHours(12), Hour.AddHours(20)) }, repair); + Assert.Equal(new[] { (Hour, Hour.AddHours(12)) }, deferred); + Assert.Equal(24, repair.Sum(r => (int)((r.End - r.Start).Ticks / width.Ticks))); + + /* Under the cap and exactly at it: everything repaired, nothing deferred, same as oldest-first. */ + var small = new List<(DateTime Start, DateTime End)> { (Hour, Hour.AddHours(2)) }; + Assert.Equal(small, TimescaleSupport.CapMaterializationHoleRepairs(small, 24, width, newestFirst: true).Repair); + Assert.Empty(TimescaleSupport.CapMaterializationHoleRepairs(small, 24, width, newestFirst: true).Deferred); + } + [Fact] public void AlignDown_LandsOnABucketBoundary_ForBothWidths() { @@ -408,8 +441,11 @@ and per-failure lines stay — they are the detail the summary counts. */ Assert.DoesNotContain("aggregate(s) scanned", storage, StringComparison.Ordinal); Assert.DoesNotContain("no holes.", storage, StringComparison.Ordinal); Assert.Contains("passClock.Elapsed", storage, StringComparison.Ordinal); - Assert.Contains("holesFound += ranges.Count;", storage, StringComparison.Ordinal); - Assert.Contains("bucketsFound += holes.Count;", storage, StringComparison.Ordinal); + /* #4186 round-3 H1: found is now the seam and ordinary windows' ranges/buckets summed, since the two + are scanned and capped separately (the seam newest-first, the ordinary oldest-first) rather than + merged into one list before counting. */ + Assert.Contains("holesFound += seamRanges.Count + ordinaryRanges.Count;", storage, StringComparison.Ordinal); + Assert.Contains("bucketsFound += seamHoles.Count + ordinaryHoles.Count;", storage, StringComparison.Ordinal); Assert.Contains("had {Buckets} bucket(s) in [{Start}, {End})", storage, StringComparison.Ordinal); Assert.Contains("left for the next start", storage, StringComparison.Ordinal); Assert.Contains("could not scan or repair {View} this start", storage, StringComparison.Ordinal); diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs index 7328888946..116c40dd8d 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.MaterializationHoles.cs @@ -322,13 +322,21 @@ public static string MaterializationSpanSql((string Schema, string Name) materia } /// - /// The cap applied to the merged ranges: the OLDEST ranges up to buckets in - /// total are repaired this start, the rest are reported and left. A range that straddles the cap is split - /// at it, so the budget is spent exactly and the remainder is a well-formed range for the next start. - /// Pure, so the tests can walk it. + /// The cap applied to the merged ranges: up to buckets in total are repaired + /// this start, the rest are reported and left. A range that straddles the cap is split at it, so the budget + /// is spent exactly and the remainder is a well-formed range for the next start. Pure, so the tests can walk + /// it. + /// + /// Direction (#4186 round-3 H1). Oldest-first ( false, + /// the default) takes the OLDEST ranges, splitting a straddler at its NEWER edge — safe for the ordinary + /// window, where a repair sits strictly above the floor and can never move it. Newest-first takes the + /// ranges closest to the END of the ordering, splitting a straddler at its OLDER edge so the kept portion + /// stays adjacent to whatever is already materialized — the seam window's own requirement, since there the + /// floor IS a bare min(bucket) with no contiguity check behind it, and only a gapless top-down + /// descent keeps every unrepaired row inside the gate's probe window. /// public static (IReadOnlyList<(DateTime Start, DateTime End)> Repair, IReadOnlyList<(DateTime Start, DateTime End)> Deferred) CapMaterializationHoleRepairs( - IReadOnlyList<(DateTime Start, DateTime End)> ranges, int capBuckets, TimeSpan bucketWidth) + IReadOnlyList<(DateTime Start, DateTime End)> ranges, int capBuckets, TimeSpan bucketWidth, bool newestFirst = false) { ArgumentNullException.ThrowIfNull(ranges); if (capBuckets <= 0) @@ -340,7 +348,8 @@ public static (IReadOnlyList<(DateTime Start, DateTime End)> Repair, IReadOnlyLi var deferred = new List<(DateTime Start, DateTime End)>(); var remaining = capBuckets; - foreach (var range in ranges.OrderBy(r => r.Start)) + var ordered = newestFirst ? ranges.OrderByDescending(r => r.Start) : ranges.OrderBy(r => r.Start); + foreach (var range in ordered) { var width = (int)((range.End - range.Start).Ticks / bucketWidth.Ticks); if (remaining <= 0) @@ -356,6 +365,15 @@ public static (IReadOnlyList<(DateTime Start, DateTime End)> Repair, IReadOnlyLi continue; } + if (newestFirst) + { + var newestSplit = range.End - TimeSpan.FromTicks(bucketWidth.Ticks * remaining); + repair.Add((newestSplit, range.End)); + deferred.Add((range.Start, newestSplit)); + remaining = 0; + continue; + } + var split = range.Start + TimeSpan.FromTicks(bucketWidth.Ticks * remaining); repair.Add((range.Start, split)); deferred.Add((split, range.End)); @@ -546,28 +564,74 @@ successor accumulates history. A legacy with nothing materialized (max(bucket) i } scanned++; - var holes = new List(); - foreach (var (from, to) in windows) + + /* #4186 round-3 H1: the seam window's holes are scanned, capped and walked SEPARATELY from + the ordinary window's, never merged into one oldest-first pass. windows[0] is the seam + window whenever one exists — MaterializationHoleScanWindows always emits it first, and it + exists exactly when seamFloor < floor (both bucket-aligned, so that comparison alone + decides it; see that method for why). An interior (ordinary) repair can never move the + successor's floor (s.mn) — it only fills a gap strictly above an already-materialized + bucket — so oldest-first there is exactly as safe as it always was. A SEAM repair is + different: s.mn is a bare min(bucket) with no contiguity check behind it, so materializing + the OLDEST seam buckets first (the bug) can drop the floor straight to the seam's bottom + while newer seam buckets in between are still holes, and RetentionArmSafetySql's probe — + bounded above by s.mn — stops looking there. Walking the seam from the TOP (the buckets + next to s.mn) and stopping at the first range that fails keeps the descent gapless: the + floor only ever retreats into ground this pass already covered, so every unrepaired seam + row stays inside the probe window — the property RollupBackfill.cs:39-44 states for its own + newest-first slices. */ + var isSeamWindow = seamFloor < floor.Value; + var seamWindow = isSeamWindow ? windows[0] : ((DateTime From, DateTime To)?)null; + var ordinaryWindows = isSeamWindow ? windows.Skip(1) : windows; + + var seamHoles = seamWindow is { } sw + ? await ScanHolesAsync(connection, target, materialization.Value, sw.From, sw.To, cancellationToken) + : new List(); + + var ordinaryHoles = new List(); + foreach (var (from, to) in ordinaryWindows) { - holes.AddRange(await ScanHolesAsync(connection, target, materialization.Value, from, to, cancellationToken)); + ordinaryHoles.AddRange(await ScanHolesAsync(connection, target, materialization.Value, from, to, cancellationToken)); } - if (holes.Count == 0) + if (seamHoles.Count == 0 && ordinaryHoles.Count == 0) { continue; } - var ranges = MergeContiguousBuckets(holes, target.BucketWidth); + var seamRanges = MergeContiguousBuckets(seamHoles, target.BucketWidth); + var ordinaryRanges = MergeContiguousBuckets(ordinaryHoles, target.BucketWidth); /* #3756: found is counted as the scan saw it — contiguous ranges and the buckets they span — BEFORE the cap decides what this start repairs and what it leaves, so the summary can say "found" independently of "repaired" and "deferred" (the record's summary states the arithmetic). */ - holesFound += ranges.Count; - bucketsFound += holes.Count; - - var (repair, deferred) = CapMaterializationHoleRepairs(ranges, MaterializationHoleRepairCapBuckets(target.BucketWidth), target.BucketWidth); + holesFound += seamRanges.Count + ordinaryRanges.Count; + bucketsFound += seamHoles.Count + ordinaryHoles.Count; + + /* One shared cap per aggregate, same total as before this fix — the seam spends from it FIRST, + newest end down, and whatever it leaves is what the ordinary window (oldest-first, as + always) has this start. That keeps "a start never re-materializes more for one aggregate + than an ordinary policy run does" true with the seam in the mix, not only without it. */ + var cap = MaterializationHoleRepairCapBuckets(target.BucketWidth); + var (seamRepair, seamDeferred) = CapMaterializationHoleRepairs(seamRanges, cap, target.BucketWidth, newestFirst: true); + var seamBucketsTaken = seamRepair.Sum(r => (int)((r.End - r.Start).Ticks / target.BucketWidth.Ticks)); + var ordinaryCap = cap - seamBucketsTaken; + + IReadOnlyList<(DateTime Start, DateTime End)> ordinaryRepair; + IReadOnlyList<(DateTime Start, DateTime End)> ordinaryDeferred; + if (ordinaryCap > 0) + { + (ordinaryRepair, ordinaryDeferred) = CapMaterializationHoleRepairs(ordinaryRanges, ordinaryCap, target.BucketWidth); + } + else + { + ordinaryRepair = Array.Empty<(DateTime Start, DateTime End)>(); + ordinaryDeferred = ordinaryRanges; + } - foreach (var (start, end) in repair) + /* One range's plain-then-forced repair, shared by the seam and ordinary walks below. Returns + the holes still standing in [start, lastBucket] after both attempts. */ + async Task RepairRangeAsync(DateTime start, DateTime end) { var buckets = (int)((end - start).Ticks / target.BucketWidth.Ticks); var lastBucket = end - target.BucketWidth; @@ -609,9 +673,31 @@ successor accumulates history. A legacy with nothing materialized (max(bucket) i "Materialization-hole repair (#3653): {View} still shows {Remaining} of {Buckets} bucket(s) in [{Start}, {End}) as holes after a plain and a forced refresh ({Seconds:F1}s) — the source has rows there that the refresh produced no output for; the aggregate's own filter and the scan's copy of it may have diverged, or the refresh was cut short. Re-judged on the next start.", target.View, remaining, buckets, start.ToString("O", CultureInfo.InvariantCulture), end.ToString("O", CultureInfo.InvariantCulture), stopwatch.Elapsed.TotalSeconds); } + + return remaining; + } + + /* #4186 round-3 H1: newest seam range first; stop at the first one that throws (propagates to + this target's own catch below, exactly the risk the ordinary window always carried) or + leaves buckets standing (remaining > 0) — do NOT go on to an older seam range once either + happens, or the floor could advance past a still-open hole the same way the bug did. */ + foreach (var (start, end) in seamRepair) + { + var remaining = await RepairRangeAsync(start, end); + if (remaining > 0) + { + break; + } + } + + /* Ordinary window: unchanged from before this fix. An interior repair cannot move the floor, + so a remainder here only means "re-judged on the next start" — it never risks the gate. */ + foreach (var (start, end) in ordinaryRepair) + { + await RepairRangeAsync(start, end); } - foreach (var (start, end) in deferred) + foreach (var (start, end) in seamDeferred.Concat(ordinaryDeferred)) { var buckets = (int)((end - start).Ticks / target.BucketWidth.Ticks); holesDeferred++; diff --git a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs index c7a2981680..beba7b2545 100644 --- a/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs +++ b/Darling/PerformanceMonitor.Darling.Storage/TimescaleSupport.cs @@ -6055,10 +6055,31 @@ FROM timescaledb_information.jobs AS j /// down to the legacy's last bucket, unclamped by the horizon that bounds its ordinary scan window — so an /// outage longer than that horizon's own span (#4186 follow-up: the first cut folded the seam into the same /// horizon clamp as the ordinary window, and a seam older than it was silently never scanned) still gets - /// repaired instead of clamped away. Once that repair runs, the seam is empty and this gate's fallback - /// releases automatically, with no manual step, bounded only by the repair's own per-start cap — a seam - /// wider than one start's cap takes more than one start to close in full, but every start makes progress on - /// it. + /// repaired instead of clamped away — PROVIDED the raw purge was already held when the store stopped. + /// That proviso is real, not decoration: PostgreSQL runs its own overdue retention jobs at start, before + /// this service's start sweep can hold anything, so a purge left armed across a long enough stop drops the + /// chunk holding the seam before the walk ever gets a turn (#4299, pre-existing — 3.8.0 loses the same + /// chunk on the same schedule). Once the repair does run, the seam is empty and this gate's fallback + /// releases — automatically within the hour on a store that keeps running (#3812), but only from the + /// SECOND start after an outage that crosses the upgrade: the walk skips a successor with nothing + /// materialized yet, so the seam does not exist for it to find until the successor's own first refresh has + /// run, and nothing re-runs the walk between starts (#4300). Once it does run, release is bounded only by + /// the repair's own per-start cap — a seam wider than one start's cap takes more than one start to close in + /// full, but every start makes progress on it. + /// + /// The stitch also trusts the frozen legacy's OWN span unconditionally — accepted as 3.8.0 parity, + /// not fixed by this round. The seam probe above only checks AT OR AFTER the legacy's last bucket. + /// Below that bucket, LEAST(legacy.min, successor.min) runs with no check at all: a raw-row gap that + /// opened INSIDE the frozen legacy's own materialized span — from an outage before this store ever took the + /// freeze, that 3.8.0's own retention already lost before the upgrade — reads Covered forever, because the + /// legacy stopped refreshing at the freeze and the six frozen views never re-enter the walk to close it + /// ( excludes them; see its note). This is accepted, not a + /// regression: 3.8.0 has no walk at all, so it loses the exact same rows on the exact same schedule — its + /// raw purge drops them once they age past the horizon, upgrade or not. What this build adds beyond that + /// parity is the A6 backfill (--backfill-rollups, ), which — unlike the + /// automatic walk — fills the successor down to RAW's own oldest row, not merely the legacy's boundary, so + /// an operator who runs it closes a pre-upgrade hole the automatic gate will otherwise trust without ever + /// looking. /// /// Source filter for 0-interval rows. For query_stats and procedure_stats the /// source_oldest subquery adds WHERE to exclude