From f4c86ffa39ba45ccd16113cc557f34474668645a Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Mon, 27 Jul 2026 23:31:18 -0400 Subject: [PATCH 1/3] Raise managed store maintenance_work_mem to the measured compression floor (#1777) Field measurement on a production field instance (16 GB RAM class) showed TimescaleDB compression throughput rising ~70% when maintenance_work_mem went from the old formula's ~800 MB landing point to 1536 MB, and gaining nothing measurable at 4096 MB. The formula becomes min(max(5% RAM, 1536MB), 25% RAM, 2048MB): the 1536 MB floor is the measured capture point, the 25%-of-RAM term keeps the floor from overcommitting a small host, and the 2 GB cap bounds the big-RAM case where the data showed nothing further to gain. A formula change alone would only ever reach a fresh initdb, and the stores that need this are already collecting -- so a v7 conf block (ConfMarkerV7) re-states maintenance_work_mem the same way v5 re-states shared_buffers. postgresql.conf takes the LAST assignment, so an existing store adopts the raised value on its next service-owned start without the v3 block ever being rewritten. Also corrects three stale claims in the Darling README's memory-sizing paragraph that predate #1559 (shared_buffers cap and its 8 GB example, "all three appends"). --- .../DarlingManagedPostgresTests.cs | 246 +++++++++++++++++- .../Darling.Tests/DarlingStoreUpgradeTests.cs | 1 + .../DarlingManagedPostgres.cs | 73 +++++- Darling/README.md | 2 +- 4 files changed, 304 insertions(+), 18 deletions(-) diff --git a/Darling/Darling.Tests/DarlingManagedPostgresTests.cs b/Darling/Darling.Tests/DarlingManagedPostgresTests.cs index 05371007d6..d528e32a01 100644 --- a/Darling/Darling.Tests/DarlingManagedPostgresTests.cs +++ b/Darling/Darling.Tests/DarlingManagedPostgresTests.cs @@ -123,8 +123,9 @@ public void WorkerSizingConfAppend_PinsV2MarkerAndSizing() /// /// SCALE-READINESS memory tuning (mirrors the worker sizing): the v3 block derives shared_buffers / /// effective_cache_size / maintenance_work_mem / work_mem from the host's physical RAM, injected here so - /// the derivation is deterministic and unit-testable. 8 GB is the current DARLING01 box — none of the - /// caps engage, so it exercises the raw percentages (and pins work_mem at its 16 MB floor). + /// the derivation is deterministic and unit-testable. 8 GB is the current DARLING01 box — it pins + /// work_mem at its 16 MB floor, shared_buffers at the 1 GB co-located cap, and maintenance_work_mem at + /// the #1777 compression floor (5% of 8 GB = 409 MB is well under it, and 25% = 2 GB does not bite). /// [Fact] public void MemorySizingConfAppend_PinsV3Marker_AndDerivesFrom8GbRam() @@ -135,7 +136,7 @@ public void MemorySizingConfAppend_PinsV3Marker_AndDerivesFrom8GbRam() Assert.Contains(DarlingManagedPostgres.ConfMarkerV3, block, StringComparison.Ordinal); Assert.Contains("shared_buffers = 1024MB", block, StringComparison.Ordinal); /* 25% of 8 GB, capped at the 1 GB co-located ceiling (#1559) */ Assert.Contains("effective_cache_size = 6144MB", block, StringComparison.Ordinal); /* 75% of 8 GB */ - Assert.Contains("maintenance_work_mem = 409MB", block, StringComparison.Ordinal); /* 5% of 8 GB */ + Assert.Contains("maintenance_work_mem = 1536MB", block, StringComparison.Ordinal); /* the #1777 measured floor; 5% of 8 GB = 409 MB is far under it */ Assert.Contains("work_mem = 16MB", block, StringComparison.Ordinal); /* RAM/512, at the 16 MB floor */ /* The blocks compose, they don't compete — v3 must not restate v1/v2 settings. */ @@ -212,6 +213,73 @@ public void LogRotationConfAppend_PinsV6Marker_AndTheSelfCappingWeekdayRing() Assert.DoesNotContain("max_connections", block, StringComparison.Ordinal); } + /// + /// The v7 compression-memory override (#1777) — the PROPAGATION half of the raised floor, and the only + /// reason an EXISTING store adopts it. A store provisioned before #1777 carries a v3 block whose + /// maintenance_work_mem was written under the old min(5% RAM, 1 GB) rule: on a 16 GB host that is the + /// 819 MB line simulated here. Appending v7 must make the LAST occurrence the new 1536 MB value, which + /// is the one PostgreSQL honors — the v3 block is never rewritten in place. + /// + [Fact] + public void CompressionMemoryConfAppend_PinsV7Marker_AndOverridesAnOlderV3Line() + { + const long sixteenGb = 16L * 1024 * 1024 * 1024; + var block = DarlingManagedPostgres.BuildCompressionMemoryConfAppend(sixteenGb); + + Assert.Contains(DarlingManagedPostgres.ConfMarkerV7, block, StringComparison.Ordinal); + Assert.Contains("maintenance_work_mem = 1536MB", block, StringComparison.Ordinal); + + /* The pre-#1777 conf shape: a v3 block carrying the OLD landing value. Appending v7 is what an + existing store's next service-owned start does, and last-occurrence-wins is what makes it real. */ + var legacyConf = + DarlingManagedPostgres.ConfMarkerV3 + "\n" + + "shared_buffers = 1024MB\n" + + "effective_cache_size = 12288MB\n" + + "maintenance_work_mem = 819MB\n" + + "work_mem = 32MB\n"; + Assert.Equal("819MB", LastSettingValue(legacyConf, "maintenance_work_mem")); + Assert.Equal("1536MB", LastSettingValue(legacyConf + block, "maintenance_work_mem")); + + /* The older block is preserved, not edited — the heal path only ever appends. */ + Assert.Contains("maintenance_work_mem = 819MB", legacyConf + block, StringComparison.Ordinal); + + /* The blocks compose, they don't compete — v7 restates ONLY maintenance_work_mem. (The work_mem + probe is anchored to a line start: "maintenance_work_mem = " trivially contains "work_mem = ".) */ + Assert.DoesNotContain("shared_buffers", block, StringComparison.Ordinal); + Assert.DoesNotContain("\nwork_mem = ", block, StringComparison.Ordinal); + Assert.DoesNotContain("effective_cache_size", block, StringComparison.Ordinal); + } + + /// + /// The value PostgreSQL would honor for : the LAST assignment in the file, + /// which is the whole mechanism behind the versioned override blocks (v5 shared_buffers, v7 + /// maintenance_work_mem). Ignores comment lines so a marker can never be read as an assignment. + /// + private static string? LastSettingValue(string conf, string setting) + { + string? value = null; + foreach (var raw in conf.Split('\n')) + { + var line = raw.Trim(); + if (line.StartsWith('#')) + { + continue; + } + + var separator = line.IndexOf('=', StringComparison.Ordinal); + if (separator > 0 && line[..separator].Trim().Equals(setting, StringComparison.Ordinal)) + { + /* The stock conf trails inline comments after the value ("100 # (change requires + restart)"); ours never do, but the parser must not depend on that. */ + var assignment = line[(separator + 1)..]; + var comment = assignment.IndexOf('#', StringComparison.Ordinal); + value = (comment >= 0 ? assignment[..comment] : assignment).Trim(); + } + } + + return value; + } + /// /// #1652: the diagnostics tail must follow the log wherever Postgres last wrote it — pg.log for /// pre-collector/startup failures, the v6 ring for a server that came up and then complained. @@ -292,9 +360,9 @@ public void CapLegacyServerLog_RollsOnlyOversizedFiles() } /// - /// On a big box every cap/ceiling engages: shared_buffers pins at 8 GB (not 25% = 16 GB), - /// maintenance_work_mem at 1 GB (not 5% = 3.2 GB), work_mem at 64 MB (not RAM/512 = 128 MB); - /// effective_cache_size stays the uncapped 75% planner hint. + /// On a big box every cap/ceiling engages: shared_buffers pins at the 1 GB co-located cap (not 25% = + /// 16 GB), maintenance_work_mem at the #1777 2 GB cap (not 5% = 3.2 GB), work_mem at 64 MB (not + /// RAM/512 = 128 MB); effective_cache_size stays the uncapped 75% planner hint. /// [Fact] public void MemorySizingConfAppend_EngagesCapsOnLargeRam() @@ -304,7 +372,7 @@ public void MemorySizingConfAppend_EngagesCapsOnLargeRam() Assert.Contains("shared_buffers = 1024MB", block, StringComparison.Ordinal); /* capped at the 1 GB co-located ceiling (#1559) */ Assert.Contains("effective_cache_size = 49152MB", block, StringComparison.Ordinal); /* 75% of 64 GB, uncapped */ - Assert.Contains("maintenance_work_mem = 1024MB", block, StringComparison.Ordinal); /* capped at 1 GB */ + Assert.Contains("maintenance_work_mem = 2048MB", block, StringComparison.Ordinal); /* capped at 2 GB (#1777) */ Assert.Contains("work_mem = 64MB", block, StringComparison.Ordinal); /* capped at 64 MB */ } @@ -312,13 +380,22 @@ public void MemorySizingConfAppend_EngagesCapsOnLargeRam() /// The pure derivation across RAM tiers, pinning each formula and its cap/clamp. work_mem is the /// flagged per-connection setting: it scales RAM/512 and reaches the 64 MB ceiling at 32 GB, so the /// pathological max_connections × sorts × work_mem never grows past the ceiling on a bigger box. + /// + /// The maintenance_work_mem column is the #1777 LANDING TABLE, and each row exercises a different + /// one of the three terms so the interaction cannot silently change: at 2 GB and 4 GB the 25%-of-RAM + /// SMALL-HOST GUARD wins (512 / 1024 MB — the floor is held back rather than overcommitting the box); + /// at 8 GB and 16 GB the measured 1536 MB FLOOR wins (16 GB is the RAM class the field measurement came + /// from, and it must land exactly on the 1536 MB capture point); at 32 GB the raw 5%-of-RAM term has + /// finally overtaken the floor and wins on its own (1638 MB); at 64 GB the 2 GB CAP wins (5% would be + /// 3276 MB, and the field data showed nothing to gain past 1536). /// [Theory] - [InlineData(2, 512, 1536, 102, 16)] /* 2 GB: under every cap; work_mem at the 16 MB floor */ - [InlineData(8, 1024, 6144, 409, 16)] /* 8 GB: shared_buffers hits the 1 GB co-located cap (#1559); work_mem at the floor */ - [InlineData(16, 1024, 12288, 819, 32)] /* 16 GB: shared_buffers capped (the field box); work_mem RAM/512 = 32 MB */ - [InlineData(32, 1024, 24576, 1024, 64)] /* 32 GB: shared_buffers at the 1 GB cap, maintenance hits 1 GB, work_mem hits the 64 MB ceiling */ - [InlineData(64, 1024, 49152, 1024, 64)] /* 64 GB: everything but effective_cache_size capped; the planner hint keeps scaling */ + [InlineData(2, 512, 1536, 512, 16)] /* 2 GB: maintenance held to 25% of RAM by the small-host guard; work_mem at the 16 MB floor */ + [InlineData(4, 1024, 3072, 1024, 16)] /* 4 GB: the smallest host — the 25% guard holds the 1536 floor down to 1 GB (#1777) */ + [InlineData(8, 1024, 6144, 1536, 16)] /* 8 GB: shared_buffers hits the 1 GB co-located cap (#1559); maintenance at the measured floor; work_mem at the floor */ + [InlineData(16, 1024, 12288, 1536, 32)] /* 16 GB: the field-measured class — maintenance lands exactly on the 1536 MB capture point; work_mem RAM/512 = 32 MB */ + [InlineData(32, 1024, 24576, 1638, 64)] /* 32 GB: 5% of RAM has overtaken the floor and wins outright; work_mem hits the 64 MB ceiling */ + [InlineData(64, 1024, 49152, 2048, 64)] /* 64 GB: maintenance at the 2 GB cap; everything but effective_cache_size capped */ public void DeriveMemorySettings_PerTier(long ramGb, int sharedBuffersMb, int effectiveCacheMb, int maintenanceMb, int workMemMb) { var settings = DarlingManagedPostgres.DeriveMemorySettings(ramGb * 1024 * 1024 * 1024); @@ -338,7 +415,7 @@ public void DeriveMemorySettings_NonPositiveRam_FallsBackToConservative4Gb() Assert.Equal(1024, settings.SharedBuffersMb); /* 25% of the 4 GB fallback */ Assert.Equal(3072, settings.EffectiveCacheSizeMb); /* 75% of 4 GB */ - Assert.Equal(204, settings.MaintenanceWorkMemMb); /* 5% of 4 GB */ + Assert.Equal(1024, settings.MaintenanceWorkMemMb); /* the #1777 1536 MB floor, held to 25% of the 4 GB fallback */ Assert.Equal(16, settings.WorkMemMb); /* RAM/512 = 8 MB, lifted to the 16 MB floor */ } @@ -568,6 +645,13 @@ hypertable count (catalog + collection_log, the V23 non-catalog hypertable). */ var ringFiles = Directory.GetFiles(Path.Combine(dataDirectory, "log"), "postgresql-*.log"); Assert.NotEmpty(ringFiles); + /* v7 compression memory (#1777) rode the same first-run append. Its value derives from THIS + host's RAM, so pin the marker and capture the conf's EFFECTIVE value (the last assignment, + which is the one the server honors) to compare against the live setting below. */ + Assert.Contains(DarlingManagedPostgres.ConfMarkerV7, conf, StringComparison.Ordinal); + var confMaintenanceWorkMem = LastSettingValue(conf, "maintenance_work_mem"); + Assert.NotNull(confMaintenanceWorkMem); + /* The derived credential really authenticates (scram, not trust) into the darling database — and the server started with our appended conf, so the timescaledb preload line was accepted; the v2 worker sizing was accepted too (the setting is @@ -576,8 +660,10 @@ hypertable count (catalog + collection_log, the V23 non-catalog hypertable). */ { await connection.OpenAsync(timeout.Token); using var current = new NpgsqlCommand( - "SELECT current_database(), current_user, current_setting('max_worker_processes'), current_setting('work_mem'), current_setting('shared_buffers')", + "SELECT current_database(), current_user, current_setting('max_worker_processes'), current_setting('work_mem'), current_setting('shared_buffers'), " + + "pg_size_bytes(current_setting('maintenance_work_mem')), pg_size_bytes(@confMaintenance)", connection); + current.Parameters.AddWithValue("confMaintenance", confMaintenanceWorkMem); using var reader = await current.ExecuteReaderAsync(timeout.Token); Assert.True(await reader.ReadAsync(timeout.Token)); Assert.Equal("darling", reader.GetString(0)); @@ -591,6 +677,13 @@ derived values (>= the 16 MB work_mem floor / 25%-of-RAM shared_buffers on any r never the stock 4 MB / 128 MB defaults. */ Assert.NotEqual("4MB", reader.GetString(3)); Assert.NotEqual("128MB", reader.GetString(4)); + + /* #1777: the v7 override is LIVE, not merely written — the server holds exactly the conf's + effective (last) assignment, never the stock 64 MB default. Compared in BYTES because + PostgreSQL normalizes units on the way out: a conf line of "2048MB" reads back as "2GB", + the same setting and a failed string compare (seen live on a large-RAM runner). */ + Assert.Equal(reader.GetInt64(6), reader.GetInt64(5)); + Assert.NotEqual(64L * 1024 * 1024, reader.GetInt64(5)); } /* Second EnsureRunning against the live server: idempotent — no re-init (credential @@ -610,6 +703,7 @@ never the stock 4 MB / 128 MB defaults. */ Assert.Equal(1, CountOccurrences(confAfterSecond, DarlingManagedPostgres.ConfMarkerV4)); Assert.Equal(1, CountOccurrences(confAfterSecond, DarlingManagedPostgres.ConfMarkerV5)); Assert.Equal(1, CountOccurrences(confAfterSecond, DarlingManagedPostgres.ConfMarkerV6)); + Assert.Equal(1, CountOccurrences(confAfterSecond, DarlingManagedPostgres.ConfMarkerV7)); /* Both up/down probes below must bypass Npgsql's pool: OpenAsync on a pooled string can hand back an idle socket with no I/O at all, which "succeeds" against a stopped @@ -641,6 +735,130 @@ threw mid-flight. */ } } + /// + /// #1777 PROPAGATION, proven against a real server: an EXISTING store — one whose conf carries the v3 + /// block written under the old min(5% RAM, 1 GB) rule and no v7 marker — must adopt the raised + /// maintenance_work_mem on its next service-owned start. This is the half that actually reaches the + /// field; a formula change alone would only ever have applied to a fresh initdb, and the boxes that + /// need it are already collecting. + /// + /// The pre-#1777 conf is reconstructed exactly, not approximated: the v7 block is removed (it is + /// the last thing appended, so truncating at its marker restores the old file byte-for-byte) and the v3 + /// block's value is rewritten to 819 MB, which is what the old formula produced on a 16 GB host — the + /// RAM class the field measurement came from. The BEFORE reading is taken from the live server, so the + /// old value is proven in effect before the new one is proven to replace it. + /// + [Fact] + public async Task ExistingStore_AdoptsRaisedMaintenanceWorkMem_OnNextStart_Gated() + { + var runtimeRoot = Environment.GetEnvironmentVariable("DARLING_TEST_PGRUNTIME"); + Assert.SkipWhen(string.IsNullOrWhiteSpace(runtimeRoot), + "Set DARLING_TEST_PGRUNTIME to an assembled pg-runtime directory (the folder containing pgsql\\bin\\pg_ctl.exe; " + + "Darling\\tools\\fetch-pg-runtime.ps1 -KeepWork leaves one under artifacts\\pg-runtime-work\\assemble\\pg-runtime) " + + "to run the #1777 conf-propagation E2E."); + Assert.SkipUnless(OperatingSystem.IsWindows(), "The bundled runtime is Windows-only."); + Assert.SkipUnless(File.Exists(Path.Combine(runtimeRoot!, "pgsql", "bin", "pg_ctl.exe")), + $"DARLING_TEST_PGRUNTIME={runtimeRoot} does not contain pgsql\\bin\\pg_ctl.exe."); + + var root = Directory.CreateTempSubdirectory("darling-pgv7-"); + var dataDirectory = Path.Combine(root.FullName, "pg"); + var config = new PostgresConfig + { + Managed = true, + Port = FindFreeTcpPort(), + DataDirectory = dataDirectory, + }; + var confPath = Path.Combine(dataDirectory, "postgresql.conf"); + const string legacyValue = "819MB"; + + var owner = new DarlingManagedPostgres(config, NullLogger.Instance, runtimeRoot); + try + { + using var timeout = new CancellationTokenSource(TimeSpan.FromMinutes(8)); + + /* A real store, provisioned the normal way. */ + await owner.EnsureRunningAsync(timeout.Token); + await owner.StopIfStartedByThisProcessAsync(); + + /* Rewind the conf to its pre-#1777 shape: drop the v7 block (appended last, so the marker is a + clean truncation point) and put the OLD formula's 16 GB landing value in the v3 block. */ + var fresh = await File.ReadAllTextAsync(confPath, timeout.Token); + var v7Index = fresh.IndexOf(DarlingManagedPostgres.ConfMarkerV7, StringComparison.Ordinal); + Assert.True(v7Index > 0, "The fresh conf should carry the v7 block before it is rewound."); + var derivedValue = LastSettingValue(fresh, "maintenance_work_mem"); + Assert.NotNull(derivedValue); + /* The whole test turns on before != after. A host with ~3.2 GB RAM would derive exactly 819MB + through the 25% guard and make the comparison vacuous — that is a property of the RUNNER, + not a product failure, so skip rather than pass emptily. */ + Assert.SkipWhen(string.Equals(derivedValue, legacyValue, StringComparison.Ordinal), + $"This host derives maintenance_work_mem = {derivedValue}, the same value the test uses as the legacy reading."); + + var legacyConf = fresh[..v7Index] + .Replace($"maintenance_work_mem = {derivedValue}", $"maintenance_work_mem = {legacyValue}", StringComparison.Ordinal); + await File.WriteAllTextAsync(confPath, legacyConf, timeout.Token); + + Assert.DoesNotContain(DarlingManagedPostgres.ConfMarkerV7, legacyConf, StringComparison.Ordinal); + Assert.Equal(legacyValue, LastSettingValue(legacyConf, "maintenance_work_mem")); + + /* The service-owned start: EnsureConfAppended heals BEFORE pg_ctl start, so the raised value is + live on this very start rather than one restart later. That ordering is what makes the live + assertion below decisive — without the v7 append this server would have come up on the + rewound conf and reported the legacy 819MB. */ + var healedOwner = new DarlingManagedPostgres(config, NullLogger.Instance, runtimeRoot); + var healedConnectionString = await healedOwner.EnsureRunningAsync(timeout.Token); + try + { + var healedConf = await File.ReadAllTextAsync(confPath, timeout.Token); + Assert.Equal(1, CountOccurrences(healedConf, DarlingManagedPostgres.ConfMarkerV7)); + Assert.Equal(derivedValue, LastSettingValue(healedConf, "maintenance_work_mem")); + /* Appended, never rewritten in place — the legacy line is still there, just outvoted. */ + Assert.Contains($"maintenance_work_mem = {legacyValue}", healedConf, StringComparison.Ordinal); + + var (live, expected) = await ReadSettingAndLiteralBytesAsync( + healedConnectionString, "maintenance_work_mem", derivedValue, timeout.Token); + Assert.Equal(expected, live); + Assert.NotEqual(819L * 1024 * 1024, live); + } + finally + { + await healedOwner.StopIfStartedByThisProcessAsync(); + } + + /* A third start must not append a second v7 block. */ + Assert.Equal(1, CountOccurrences(await File.ReadAllTextAsync(confPath, timeout.Token), DarlingManagedPostgres.ConfMarkerV7)); + } + finally + { + await owner.StopIfStartedByThisProcessAsync(); + TryDeleteRecursive(root.FullName); + } + } + + /// + /// Reads one live GUC and one postgresql.conf size literal, both as BYTES, through an UNPOOLED + /// connection — so the reading always costs real I/O against the server running right now rather than a + /// recycled idle socket. + /// + /// Bytes rather than the raw strings, and measured BY THE SERVER rather than by reimplementing + /// PostgreSQL's unit parsing here: the server normalizes memory units on the way out, so a conf line of + /// 2048MB reads back as 2GB — the same setting, and a string compare that fails. That is + /// not hypothetical; it is what a large-RAM runner did to the first version of this test. + /// + private static async Task<(long Live, long Expected)> ReadSettingAndLiteralBytesAsync( + string connectionString, string setting, string literal, CancellationToken cancellationToken) + { + var unpooled = new NpgsqlConnectionStringBuilder(connectionString) { Pooling = false }.ConnectionString; + await using var connection = new NpgsqlConnection(unpooled); + await connection.OpenAsync(cancellationToken); + using var command = new NpgsqlCommand( + "SELECT pg_size_bytes(current_setting(@setting)), pg_size_bytes(@literal)", connection); + command.Parameters.AddWithValue("setting", setting); + command.Parameters.AddWithValue("literal", literal); + using var reader = await command.ExecuteReaderAsync(cancellationToken); + Assert.True(await reader.ReadAsync(cancellationToken)); + return (reader.GetInt64(0), reader.GetInt64(1)); + } + private static int CountOccurrences(string text, string value) { var count = 0; diff --git a/Darling/Darling.Tests/DarlingStoreUpgradeTests.cs b/Darling/Darling.Tests/DarlingStoreUpgradeTests.cs index 86ba5b253e..5653adc0c9 100644 --- a/Darling/Darling.Tests/DarlingStoreUpgradeTests.cs +++ b/Darling/Darling.Tests/DarlingStoreUpgradeTests.cs @@ -705,6 +705,7 @@ the normal heal path did not duplicate them. */ Assert.Contains("shared_preload_libraries = 'timescaledb'", conf, StringComparison.Ordinal); Assert.Equal(1, CountOccurrences(conf, DarlingManagedPostgres.ConfMarker)); Assert.Equal(1, CountOccurrences(conf, DarlingManagedPostgres.ConfMarkerV6)); + Assert.Equal(1, CountOccurrences(conf, DarlingManagedPostgres.ConfMarkerV7)); } finally { diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs index daf5a9ebe1..de2c8cf2b8 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs @@ -169,6 +169,17 @@ public sealed class DarlingManagedPostgres /// public const string ConfMarkerV6 = "# Managed by PerformanceMonitor Darling (v6 log rotation) -- do not remove this block"; + /// + /// Marker for the v7 compression-memory override (#1777). A SEVENTH independently versioned block, + /// same heal discipline as v5: it re-states ONLY maintenance_work_mem, so a store already + /// carrying a v3 block written under the old min(5% RAM, 1 GB) rule adopts the raised floor on + /// its next service-owned start — postgresql.conf takes the LAST occurrence of a setting, so appending + /// wins without rewriting v3. This is the half that reaches EXISTING stores: a formula change alone + /// would only ever have applied to a fresh initdb, and the boxes that need it are the ones already + /// collecting. + /// + public const string ConfMarkerV7 = "# Managed by PerformanceMonitor Darling (v7 compression memory) -- do not remove this block"; + /// /// Markers delimiting the Darling-managed network access block in pg_hba.conf /// (darling-network-endpoints, D5). replaces exactly the lines @@ -452,8 +463,25 @@ internal readonly record struct MemorySettings( /// RESTART-ONLY, like max_worker_processes; it applies on the next server start. /// effective_cache_size = 75% RAM — a PLANNER HINT (no allocation) telling the planner how /// much data is likely cached (PG + OS cache), biasing it toward index scans. - /// maintenance_work_mem = min(5% RAM, 1 GB) — headroom for VACUUM / CREATE INDEX; capped at - /// 1 GB because several autovacuum workers can each take up to this. + /// maintenance_work_mem = min(max(5% RAM, 1.5 GB), 25% RAM, 2 GB) — headroom for VACUUM / + /// CREATE INDEX, and the setting TimescaleDB's compression sort runs on, which is what drove the + /// shape (#1777). MEASURED on a production field instance (16 GB RAM class), three points during a + /// one-time catch-up of large backlog chunks: at the old formula's landing point (~800 MB) compression + /// moved ~9.1 MB/s of uncompressed input; at 1536 MB it moved 15.5 MB/s (a pure-linear null hypothesis + /// predicted 2833s, actual was 1657s — a real effect, not noise); at 4096 MB it moved 16.1 MB/s, so + /// going 2.7x further past 1536 bought nothing measurable. Hence the terms: the 1.5 GB floor is + /// the measured capture point and is the fix itself (the old 5%-of-RAM term landed UNDER the old 1 GB + /// cap on a 16 GB host, so raising the cap alone would have changed nothing); the 25%-of-RAM + /// term keeps the floor from overcommitting a small host; the 2 GB cap concedes nothing + /// measurable and bounds the big-RAM case. Honest limits, both recorded in #1777: the three points are + /// different tables at different sizes rather than a controlled experiment, so the exact threshold + /// between ~800 MB and 1536 MB is unknown; and the floor raises SMALL hosts more than the 16 GB host + /// it was measured on (a 4 GB host goes 204 -> 1024 MB, an 8 GB host 409 -> 1536 MB). That is + /// deliberate and bounded: this is a per-operation CEILING, not a reservation — PostgreSQL grows the + /// sort/TidStore allocation to fit the work, and a small host's chunks are small, so the ceiling is + /// simply never reached there. PG 17+ (the bundle pins 18.4) also made vacuum's dead-TID store grow + /// incrementally rather than allocating the full limit up front, which is what made the old comment's + /// "several autovacuum workers can each take up to this" the binding worry it no longer is. /// work_mem = clamp(RAM/512, 16 MB, 64 MB) — the ONE with real downside (per-sort, /// per-connection: worst-case ≈ max_connections × sorts × work_mem), so it is deliberately modest. /// At the PG default max_connections = 100 with ~3 concurrent sort/hash nodes, the pathological @@ -472,7 +500,12 @@ internal static MemorySettings DeriveMemorySettings(long totalPhysicalMemoryByte var sharedBuffers = Math.Min(ram / 4, oneGb); /* 25% RAM, capped at 1 GB (co-located store + Windows 487 mitigation, #1559) — restart-only */ var effectiveCache = ram / 4 * 3; /* 75% RAM — planner hint, not an allocation */ - var maintenanceWorkMem = Math.Min(ram / 20, oneGb); /* 5% RAM, capped at 1 GB */ + /* #1777: 5% RAM with a MEASURED 1.5 GB floor (compression throughput rose ~70% reaching it and + plateaued there), guarded by 25% of RAM so the floor cannot overcommit a small host, and capped + at 2 GB where the field data showed nothing further to gain. */ + var maintenanceWorkMem = Math.Min( + Math.Min(Math.Max(ram / 20, 1536 * oneMb), ram / 4), + 2048 * oneMb); var workMem = Math.Clamp(ram / 512, 16 * oneMb, 64 * oneMb); return new MemorySettings( @@ -504,6 +537,27 @@ public static string BuildMemorySizingConfAppend(long totalPhysicalMemoryBytes) return builder.ToString(); } + /// + /// The v7 compression-memory override (#1777) — the propagation half of the raised + /// maintenance_work_mem floor. Built exactly like the v5 shared_buffers override: it re-states + /// ONE setting so a store whose v3 block was written under the old min(5% RAM, 1 GB) rule heals + /// UP by conf last-occurrence-wins, without ever rewriting the v3 block (which + /// never does). Without this block the new formula would reach fresh + /// initdbs only, and the stores that measurably need it are the ones already running. + /// + /// maintenance_work_mem is plain SIGHUP-reloadable rather than restart-only, but the append + /// happens before pg_ctl start, so an existing store picks it up on that very start. + /// + public static string BuildCompressionMemoryConfAppend(long totalPhysicalMemoryBytes) + { + var settings = DeriveMemorySettings(totalPhysicalMemoryBytes); + var builder = new StringBuilder(); + builder.Append('\n'); + builder.Append(ConfMarkerV7).Append('\n'); + builder.Append("maintenance_work_mem = ").Append(settings.MaintenanceWorkMemMb).Append("MB\n"); + return builder.ToString(); + } + /// /// The derived managed-mode connection string: 127.0.0.1 + port + darling/darling + the /// generated password, carrying the collect/config so every pooled connection @@ -919,6 +973,19 @@ effective immediately on this very start. */ "Appended v6 log rotation to postgresql.conf (logging collector, 7-file weekday ring under {LogDir}; pg.log keeps only pg_ctl and pre-collector startup lines)", Path.Combine(dataDirectory, "log")); } + + /* Checked independently of v1-v6: a store whose v3 block wrote maintenance_work_mem under the old + min(5% RAM, 1 GB) rule heals UP to the measured 1.5 GB compression floor (last-occurrence-wins + override, #1777). This is what carries the fix to EXISTING stores — the formula alone would only + ever have reached a fresh initdb. */ + if (!conf.Contains(ConfMarkerV7, StringComparison.Ordinal)) + { + var v7RamBytes = GetTotalPhysicalMemoryBytes(); + File.AppendAllText(confPath, BuildCompressionMemoryConfAppend(v7RamBytes)); + _logger.LogInformation( + "Appended v7 compression memory to postgresql.conf (maintenance_work_mem = {Maintenance}MB from min(max(5% RAM, 1536MB), 25% RAM, 2048MB); TimescaleDB compression sorts on this setting)", + DeriveMemorySettings(v7RamBytes).MaintenanceWorkMemMb); + } } /// diff --git a/Darling/README.md b/Darling/README.md index f485a56cdf..e33a62398e 100644 --- a/Darling/README.md +++ b/Darling/README.md @@ -572,7 +572,7 @@ With `postgres.managed = true` (the sample's default), the service runs its own } ``` -**What first run does.** The service looks for `pg-runtime\pgsql\` beside its binary, extracting it from `pg-runtime.zip` when only the zip is present (deleting the extracted directory is therefore always safe — it self-heals). If the data directory has no cluster, it generates a 32-character random password, protects it with DPAPI LocalMachine into `pg-credential.dpapi` beside the data directory (credential first, so a crash mid-initdb never strands a cluster nobody can log into), then runs `initdb` with `scram-sha-256` auth, data checksums, and UTF8/C locale. A marker-guarded block appended to `postgresql.conf` preloads TimescaleDB, sets the port, and restricts listening to `127.0.0.1`; a second versioned block sizes background workers up for the per-hypertable compression jobs (`timescaledb.max_background_workers = 28`, `max_worker_processes = 40` — PostgreSQL's default of 8 workers cannot launch them); a third versioned block sizes memory from the host's physical RAM for the up-to-500-servers case (`shared_buffers = min(25% RAM, 8GB)`, `effective_cache_size = 75% RAM`, `maintenance_work_mem = min(5% RAM, 1GB)`, and a deliberately-modest per-connection `work_mem = clamp(RAM/512, 16MB, 64MB)` — on an 8 GB box that is `shared_buffers 2048MB` / `work_mem 16MB`; the stock 128 MB / 4 MB defaults are fine at small scale but bottleneck at fleet scale). All three appends are re-checked on every start, so a crash between initdb and the append heals itself instead of silently degrading — and clusters initialized before the worker or memory sizing existed gain the missing block on their next start (effective at the next PostgreSQL restart). Then `pg_ctl start`, `CREATE DATABASE darling`, and the normal startup path (migrations, TimescaleDB adoption — you should see `32/32 collector table(s) are hypertables`) continues exactly as in bring-your-own mode. The connection string is derived from the stored credential; the Viewer and the MCP host on the same machine derive it the same way, so nothing needs configuring there either. +**What first run does.** The service looks for `pg-runtime\pgsql\` beside its binary, extracting it from `pg-runtime.zip` when only the zip is present (deleting the extracted directory is therefore always safe — it self-heals). If the data directory has no cluster, it generates a 32-character random password, protects it with DPAPI LocalMachine into `pg-credential.dpapi` beside the data directory (credential first, so a crash mid-initdb never strands a cluster nobody can log into), then runs `initdb` with `scram-sha-256` auth, data checksums, and UTF8/C locale. A marker-guarded block appended to `postgresql.conf` preloads TimescaleDB, sets the port, and restricts listening to `127.0.0.1`; a second versioned block sizes background workers up for the per-hypertable compression jobs (`timescaledb.max_background_workers = 28`, `max_worker_processes = 40` — PostgreSQL's default of 8 workers cannot launch them); a third versioned block sizes memory from the host's physical RAM for the up-to-500-servers case (`shared_buffers = min(25% RAM, 1GB)`, `effective_cache_size = 75% RAM`, `maintenance_work_mem = min(max(5% RAM, 1536MB), 25% RAM, 2048MB)`, and a deliberately-modest per-connection `work_mem = clamp(RAM/512, 16MB, 64MB)` — on an 8 GB box that is `shared_buffers 1024MB` / `work_mem 16MB`; the stock 128 MB / 4 MB defaults are fine at small scale but bottleneck at fleet scale). Later blocks re-state single settings that field measurement moved: a fifth caps `shared_buffers` for the co-located store, a sixth turns on the log-rotation ring, and a seventh carries the `maintenance_work_mem` floor that TimescaleDB's compression sort runs on (measured at ~+70% compression throughput on a 16 GB-class host, plateauing by 1536 MB). `postgresql.conf` takes the LAST assignment of a setting, so these override without rewriting anything. Every append is re-checked on every start, so a crash between initdb and the append heals itself instead of silently degrading — and clusters initialized before a given block existed gain it on their next start (effective at the next PostgreSQL restart). Then `pg_ctl start`, `CREATE DATABASE darling`, and the normal startup path (migrations, TimescaleDB adoption — you should see `32/32 collector table(s) are hypertables`) continues exactly as in bring-your-own mode. The connection string is derived from the stored credential; the Viewer and the MCP host on the same machine derive it the same way, so nothing needs configuring there either. **Why scram and not trust, even loopback-only.** Trust auth would hand superuser to any local code that can open a loopback socket — every other local user, and network-capable-but-not-filesystem-capable attack primitives like SSRF from a co-hosted app. With scram the credential travels on the wire, failed attempts are auditable, and access is confined to what can read the DPAPI-protected credential file. `listen_addresses = '127.0.0.1'` keeps the server unreachable off the machine on top — unless you deliberately opt into a LAN endpoint (see [Opt-in Network Endpoints (LAN)](#opt-in-network-endpoints-lan)), which reconciles `listen_addresses`, a `hostssl` pg_hba rule, and TLS on every start and is otherwise off. From b351952185ea48b48a7367b4b1a3a84ee1ede7ec Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Mon, 27 Jul 2026 23:33:58 -0400 Subject: [PATCH 2/3] CHANGELOG entry for #1780 (#1777 maintenance_work_mem floor) --- CHANGELOG.md | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 375046afd9..fcc71a0e0c 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -63,6 +63,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ### Changed +- **Darling store: `maintenance_work_mem` raised to the measured compression floor, and existing stores actually get it** ([#1780], closes [#1777]) - TimescaleDB's compression sort runs on `maintenance_work_mem`, not `work_mem`, so that one setting gates how fast the background compression job moves. Measured on a production field instance (16 GB RAM class), three real points during a one-time catch-up of large backlog chunks: at the old formula's landing point (~800 MB) compression moved **9.1 MB/s** of uncompressed input; at 1536 MB it moved **15.5 MB/s**, with a pure-linear null hypothesis predicting 2833s against an actual 1657s - a real effect, not noise; at 4096 MB it moved 16.1 MB/s, so going 2.7x further past 1536 bought nothing measurable. The formula becomes `min(max(5% RAM, 1536MB), 25% RAM, 2048MB)`. The **1536 MB floor is the fix itself**: the old `min(5% RAM, 1 GB)` landed *under* its own cap on a 16 GB host (819 MB), so raising the cap alone would have changed nothing there. The **25%-of-RAM** term is the small-host guard and the **2 GB cap** bounds the big-RAM case at the point the data stopped improving. Landing values are 1024MB / 1536MB / 1536MB / 1638MB / 2048MB on 4 / 8 / 16 / 32 / 64 GB hosts, each row pinned by a test because each one is a different term winning. + + **A formula change alone would have reached only a fresh `initdb`, and every store that needs this is already collecting.** So the change ships with its propagation half: a **v7 marker block** on the same versioned-append pattern v2 through v6 already use, re-stating `maintenance_work_mem` alone the way v5 re-states `shared_buffers`. `postgresql.conf` takes the LAST assignment of a setting, so an existing store adopts the raised value with the v3 block never rewritten - and because the append runs before `pg_ctl start`, it applies on that very start rather than one restart later. Proven end to end against a real PostgreSQL 18.4 + TimescaleDB 2.28.1 store, not asserted: a store was provisioned normally, rewound to its pre-change shape (v7 block removed, v3's value set back to `819MB`), read live at `819MB`, then restarted through the production bootstrap and read live at the new value, with the healed conf showing the legacy line still present and simply outvoted. Every guard was verified RED by mutation: reverting the formula took 10 tests red including every landing row; disabling the heal branch took the propagation E2E red with the existing store keeping 819MB; and dropping the 25% guard took **only** the small-host rows red, leaving 8/16/32/64 GB green, which is the guard proving its own scope. **Honest limits, both carried from the issue:** the three field points are different tables at different sizes rather than a controlled experiment, so the exact threshold between ~800 MB and 1536 MB is unknown and the controlled repro stays open; and the floor raises small hosts proportionally more than the 16 GB class it was measured on (4 GB goes 204 -> 1024 MB). That second one is bounded rather than hand-waved - `maintenance_work_mem` is a per-operation ceiling and not a reservation, PostgreSQL grows the allocation to fit the work, a small host's chunks are small, and PG 17+ (the bundle pins 18.4) made vacuum's dead-TID store grow incrementally instead of allocating the limit up front. One thing operators should not misread: PostgreSQL normalizes units on the way out, so `SHOW maintenance_work_mem` reports `1GB` on a 4 GB host and `2GB` on a 64 GB host. Same setting, different string. That is also why the tests compare bytes via `pg_size_bytes` - the first version compared strings and went red with `Expected: "2048MB", Actual: "2GB"` against a real server + - **CI: a dev/main push now cancels that branch's superseded in-flight build — newest SHA wins** ([#1729]) - [#1715] scoped cancel-in-progress to pull requests and deliberately left every push run uncancellable; 2026-07-26's ~20-merge evening measured what that costs at train speed: of the day's 30 dev-push `build.yml` runs, **13 were superseded mid-flight** (a newer merge landed before the run finished) and **~60 runner-minutes went to SHAs that were already stale** — capacity the shared Windows runner pool bills against every queued PR (#1697's original complaint). Push events in `build.yml` and `sql-validation.yml` now share a per-branch concurrency group with cancel-in-progress, so only the newest head keeps building. "Safe" was verified from the workflow graph, not asserted: a push run produces nothing any other run consumes — every `upload-artifact` in `build.yml` is gated to the release event (the SignPath path) or `failure()` (darling-pg diagnostics), `sql-validation.yml` uploads nothing at all, no `download-artifact` / `gh run download` / `workflow_run` consumer exists anywhere in the repo, nightly builds its own tree from its own checkout, and a release compiles fresh on the `release` event. Release and merge-queue runs keep their unique per-run groups and remain uncancellable: a release waits on SignPath's manual approval gate, and a queue validation is the last check before its result lands on dev. The accepted trade, stated rather than hidden: push builds are diff-scoped, so a cancelled run's areas are not re-verified until the next change touches them — the nightly and the all-areas dev→main release PR are the backstops. The merge queue that would eliminate the train itself stays unavailable on this personal-account repo ([#1716]: organization-owned repositories only, `422 Invalid rule 'merge_queue'`, re-verified against GitHub's GA announcement); this is the slice of that win that IS available here. - **CI: build.yml handles the merge_group event** ([#1716]) - lands what [#1715] named as the merge-queue prerequisite: without a `merge_group` trigger, the required `build` and `Darling PostgreSQL tests` checks would never report inside a queue and every queued PR would stall. Queue runs take the same always-restore path as dev/main pushes - a queue run is the last validation before its result lands on dev - and the per-run concurrency group, so they are never cancelled or replaced. Path classification works unchanged in a queue: dorny/paths-filter v4.0.1+ resolves merge_group diffs from the payload's base_sha/head_sha whenever the `base` input is empty, which is exactly what the filter steps pass for non-push events. **Post-merge correction to this entry's original "safe one-click" claim: the click does not exist here.** Merge queues are available only on ORGANIZATION-owned repositories (public on any plan, private on Enterprise Cloud - per the GA announcement, and confirmed empirically: creating the ruleset on this personal-account repo returns `422 Invalid rule 'merge_queue'`). The `.gitattributes merge=union` alternative for the CHANGELOG conflict trains is also out: GitHub's server-side PR merging ignores merge attributes (community discussion #9288), so it would only automate local resolution, not the DIRTY state or the re-push. The wiring stays - inert, zero cost, live the day the repo ever moves to an organization - and until then the train-tax relief is [#1715]'s classification fix: single-area re-pushes re-run ~3m40s instead of ~6m30s, docs-only re-pushes 13s. - **CI: the change classification the v4 pin bump silently broke is restored - and this time it is probe-validated** ([#1715]) - a timing baseline over the 30 most recent build.yml runs plus a throwaway probe PR (#1714) turned the planned build-time audit into a regression find. The 2026-07-26 08:02 pin bump moved dorny/paths-filter v3 -> v4, and v4 evaluates every filter pattern as an INDEPENDENT predicate under its default predicate-quantifier 'some' (a filter is true when any changed file matches at least one rule), so a bare `!**/*.md` exclusion stopped being a subtraction and became its own rule: "any file that is not markdown". Every area filter carrying that line went true for ANY non-markdown change anywhere in the repo - a single root .gitignore edit built and tested all four products (probe run 30219202642), darling-pg ran the full TimescaleDB suite on every PR since the bump including md-only ones (run 30218459544 matched CHANGELOG.md against the darling filter), and the [#1712] docs fast path shipped unable to engage, because every file matches '**' so its code: gate was never false - its measured md-only 1m43s runs were real, but they were the area filters at work (markdown matches no include), not the fast path. The regression was invisible by construction: the bump's own PR touched build.yml, so root=true forced a full build that looks identical to a correct run, and so does every over-built run after it. Fixed keeping v4 (v3 is on the deprecated-runtime track): area filters carry the markdown carve-out INSIDE each include as an extglob (`Darling/**/!(*.md)`) where quantifier semantics cannot detach it; the uninvertible code: filter is replaced by an `all:` counter with docs-only decided by all_count == docs_count; the classify step additionally refuses to engage while any area filter is lit, so the two classifications can never disagree into a `dotnet build --no-restore` with no restore behind it; and check-version-bump.yml, which had the identical '**'-plus-exclusions shape and an equally dead md-only skip, takes the same counter fix. Validated with a per-file truth table on the probe PR (run 30219765613: a Darling .txt probe, a Darling .md probe, .gitignore and the workflow file in one diff - each filter's matched-file list recorded in the PR body). **Also in this pass:** a PR re-push now cancels that PR's superseded in-flight build.yml/sql-validation.yml runs, while push and release runs keep unique per-run concurrency groups - never queued behind or cancelled by anything, every dev/main commit keeps its own check result; the [#1712] allowlist's flagged judgment calls are ratified (CITATION.cff, Screenshots/) but its directory-wide grants become extension-explicit (`docs/**/*.{md,svg,png,jpg,jpeg,gif}`) so a .sql dropped into docs/ tomorrow defaults to code; the scheduled nightly re-dispatches itself onto the dev ref instead of doing real work from main's copy of the workflow file - scheduled workflows execute the DEFAULT branch's copy against dev's checked-out tree, which is exactly how the 2026-07-26 06:00 nightly failed (run 30194606068: main's stale copy read Dashboard/Dashboard.csproj, moved to deprecated/ by #1612 - the #1550 trap again) - so after a ONE-TIME sync of nightly.yml to main (command in the PR body; the schedule stays red each morning until it happens) nightly logic changes take effect the night they merge to dev, with manual dispatches still always building and the artifact job still pinned to dev; and every job that never carries the release/signing path gets a timeout-minutes ceiling at ~3x its worst cold path (darling-pg 30, nightly build 90 / pg 60 / check 10, sql-validation 30 per leg, claude-review 30) so hung-not-slow failures stop holding a shared-pool runner for the 6h default - build.yml's build job stays unbounded on purpose, because the release path waits on SignPath's manual approval gate. **Measured and deliberately not done** (numbers in the PR body): per-area restore splitting (warm restore is 19-33s; four condition-mirrored restore steps buy seconds at the price of the drift risk [#1701] just retired) and cross-job test splitting (Run Lite tests ~2m25s dominates the full build, but a second Windows job costs ~2m45s of checkout/setup/restore/build before its first test - a net loss on a shared serialized pool). Merge queue remains a recommendation with exact settings in the PR body: it is a repo setting, and build.yml needs a merge_group trigger first or queued PRs stall on never-reporting required checks. @@ -1805,3 +1809,5 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 [#1773]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1773 [#1774]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1774 [#1775]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1775 +[#1777]: https://github.com/erikdarlingdata/PerformanceMonitor/issues/1777 +[#1780]: https://github.com/erikdarlingdata/PerformanceMonitor/pull/1780 From 922caa68dc6d288c13085299c125a049f1393b90 Mon Sep 17 00:00:00 2001 From: Erik Darling <2136037+erikdarlingdata@users.noreply.github.com> Date: Mon, 27 Jul 2026 23:48:21 -0400 Subject: [PATCH 3/3] Record the incremental-allocation constraint the small-host landings rest on (#1777) Comment only. Both of today's maintenance_work_mem consumers grow their allocation to fit the work (tuplesort spills past the ceiling; PG 17+ TidStore grows), so the setting bounds what an operation MAY use rather than what it WILL use -- which is what makes the 4 GB host's 1024MB landing safe. A future consumer that PRE-ALLOCATES would break that reasoning, so the note lives at the formula, where it would be violated. --- .../DarlingManagedPostgres.cs | 9 ++++++++- 1 file changed, 8 insertions(+), 1 deletion(-) diff --git a/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs b/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs index de2c8cf2b8..4bfa3d6954 100644 --- a/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs +++ b/Darling/PerformanceMonitor.Darling.Service/DarlingManagedPostgres.cs @@ -502,7 +502,14 @@ internal static MemorySettings DeriveMemorySettings(long totalPhysicalMemoryByte var effectiveCache = ram / 4 * 3; /* 75% RAM — planner hint, not an allocation */ /* #1777: 5% RAM with a MEASURED 1.5 GB floor (compression throughput rose ~70% reaching it and plateaued there), guarded by 25% of RAM so the floor cannot overcommit a small host, and capped - at 2 GB where the field data showed nothing further to gain. */ + at 2 GB where the field data showed nothing further to gain. + + THE CONSTRAINT THE SMALL-HOST LANDINGS REST ON: both of today's consumers allocate + INCREMENTALLY — a tuplesort grows to fit its input and SPILLS past the ceiling rather than + reserving it, and PG 17+ builds vacuum's dead-TID store (TidStore) the same way. So this number + bounds what an operation MAY use, not what it WILL use, which is what makes a 4 GB host's + 1024 MB landing safe despite being 5x its old 204 MB. If a future consumer ever PRE-ALLOCATES + maintenance_work_mem, that reasoning breaks and the small-host landings need revisiting here. */ var maintenanceWorkMem = Math.Min( Math.Min(Math.Max(ram / 20, 1536 * oneMb), ram / 4), 2048 * oneMb);