From 7056ae22cce3c073e70ac26cbae50937527fb463 Mon Sep 17 00:00:00 2001 From: sh <37271604+shumvgolove@users.noreply.github.com> Date: Sat, 19 Sep 2026 12:02:04 +0400 Subject: [PATCH] smp-server: reduce Prometheus and expiration DB load (#1872) * smp server: add prometheus scan indexes * smp server: estimate queue count in metrics * docs: add smp server db load report * smp server: reduce msg_queues write load * docs: document batch-2 fixes and operator actions --- docs/smp-server-db-load.md | 200 ++++++++++++++++++ .../Messaging/Server/MsgStore/Postgres.hs | 3 +- .../Messaging/Server/QueueStore/Postgres.hs | 3 +- .../Server/QueueStore/Postgres/Migrations.hs | 148 ++++++++++++- .../QueueStore/Postgres/server_schema.sql | 26 ++- 5 files changed, 367 insertions(+), 13 deletions(-) create mode 100644 docs/smp-server-db-load.md diff --git a/docs/smp-server-db-load.md b/docs/smp-server-db-load.md new file mode 100644 index 000000000..cd94d7d12 --- /dev/null +++ b/docs/smp-server-db-load.md @@ -0,0 +1,200 @@ +# SMP server database load: root causes and fixes + +## Summary + +The recurring `pMsgFwdsOwn_pErrorsOther` spikes are a symptom of chronic disk saturation of the +`smp_server` PostgreSQL database, not a proxy or forwarding bug. The baseline read load is the +Prometheus scrape running whole-table `COUNT` scans of `msg_queues` (48.5M rows) every 30 seconds, +compounded by the message-expiration sweep. The write load comes from indexing `updated_at`: +`updateQueueTime` stamps it on the first send to each active queue per UTC day, and these non-HOT +updates bloat the indexes to 104 GB and amplify WAL to about 425 GB per day (37% full-page images, +~31% from `updateQueueTime`). Because the stamp is day-granular, this write burst clusters at 00:00 UTC, +tipping the saturated disk into the observed spike and loading the pgBackRest backup repository. + +Code fixes replace the largest metric scan with an estimate and index the sweep and notifier counts. +Reclaiming the index bloat, which also clears the remaining metric scan, and keeping it down are +operator actions, because PostgreSQL never shrinks indexes automatically. + +## Environment + +The affected deployment runs the pure PostgreSQL backend (`store_queues: database`, +`store_messages: database`), schema `smp_server`, `db_pool_size 20`, `prometheus_interval 30`, +`[NAMES]` enabled, own-server proxying enabled, and forwards routed over a SOCKS proxy. + +`msg_queues`: 48.5M live rows, 12.3% dead tuples, 14 GB heap, 104 GB indexes, 118 GB total. + +## Evidence + +`pg_stat_statements` over a 5 minute window, ordered by disk reads (`track_io_timing` was off, so +`shared_blks_read` is the disk proxy; the two `EXPLAIN ANALYZE` rows are manual and excluded): + +| query | calls | mean ms | reads | share of all reads | +| --- | --- | --- | --- | --- | +| `getEntityCounts` (the six-`COUNT` metric query) | 30 | 111,456 | 990 GB | 70.7% | +| queue-record load (`SELECT recipient_id, …`) | 47,027 | 121 | 89 GB | 6.3% | +| expiration sweep batch (`array_agg` in `expire_old_messages`) | 13 | 233,143 | 62 GB | 4.5% | +| message peek (`DISTINCT ON (recipient_id)`) | 46,932 | 76 | 43 GB | 3.0% | +| `updateQueueTime` (`UPDATE msg_queues SET updated_at`) | 172,885 | 4 | 7.7 GB | 0.5% | +| `write_message` | 110,911 | 5 | 7.0 GB | 0.5% | + +`EXPLAIN (ANALYZE, BUFFERS)` of each `msg_queues` count inside `getEntityCounts`: + +| count | plan | time | +| --- | --- | --- | +| `queue_count` (`deleted_at IS NULL`) | Parallel Seq Scan | 30 s | +| `notifier_count` (`deleted_at IS NULL AND notifier_id IS NOT NULL`) | Parallel Seq Scan | 32.6 s | +| `ntf_service_queues_count` (`ntf_service_id IS NOT NULL AND deleted_at IS NULL`) | Parallel Seq Scan | 30.2 s | +| `rcv_service_queues_count` (`rcv_service_id IS NOT NULL AND deleted_at IS NULL`) | Index Only Scan | 1.4 ms | + +The expiration sweep batch: 229 s for one 10,000-row batch, walking `msg_queues_pkey` and discarding +809,729 rows by filter to find 10,000 expirable ones. + +`pg_stat_wal` over the measurement window: 27 TB of WAL accumulated, 37.2% of records full-page +images, averaging about 425 GB per day; recent days reach ~470 GB (from the `archive-push` log) as +the queue count grows. + +Per-statement WAL from `pg_stat_statements` (`wal_bytes`), leaf statements only (this instance runs +`pg_stat_statements.track = all`, so the `write_message` wrapper double-counts the `INSERT` and +`UPDATE` it runs, and is excluded): + +| category | share of WAL | main statements | +| --- | --- | --- | +| `msg_queues` updates | ~42% | `updateQueueTime` ~31%, `msg_can_write` flags ~9%, insert/delete ~2% | +| `messages` insert and delete | ~34% | message `INSERT`, `DELETE` by `message_id` and `recipient_id` | +| read path (hint bits, FPIs on SELECT) | ~24% | message peek ~15%, `msg_queue_size`, `getEntityCounts` | + +## Root causes + +1. Prometheus scrape. `getEntityCounts` (`QueueStore/Postgres.hs:154`) runs six `COUNT(1)` subqueries + every `prometheus_interval` (30 s), called from the metrics path (`Server.hs:698`). Four count + `msg_queues`; three of those seq-scan the 48.5M-row heap. At 70.7% of all reads it is the dominant + disk consumer, and it runs continuously (mean 111 s per scrape exceeds the 30 s interval). + +2. Message-expiration sweep. `expireMessagesThread` (`Server.hs:477`) calls the `expire_old_messages` + procedure (`server_schema.sql`), whose inner batch query selects expirable queues ordered by + `recipient_id`. With `oldQueue = 0` the `updated_at` filter matches everything, so the planner + walks the primary key and filters `msg_queue_expire`, discarding ~99% of rows per batch. + +3. Index bloat. 104 GB of indexes on a 14 GB table. `updated_at` is part of + `idx_msg_queues_updated_at_recipient_id` (`server_schema.sql:526`) and `updateQueueTime` + (`QueueStore/Postgres.hs:443`) updates it on the day's first send per queue (172,885 times in the + window above). Updating an indexed column prevents HOT updates, so every update writes a new row + version plus new entries in every index on the table, and the old index entries accumulate as bloat. + +4. WAL amplification. Only 15.6% of `msg_queues` updates are HOT, because its composite index covers + `updated_at` and `msg_queue_expire`, both changed on the hot path (cause 3); the rest rewrite every + index on the table. With the 104 GB of bloated indexes and frequent checkpoints (`max_wal_size 4GB`, + `checkpoint_timeout 5min`, `wal_compression off`), 37% of WAL records are full-page images, and the + server generates about 425 GB of WAL per day (27 TB accumulated; recent days near 470 GB). + `updateQueueTime`, stamping `updated_at` on the ~5.3M daily-active queues, is the single largest + statement at ~31% of WAL (see Evidence). pgBackRest `archive-push` ships all of it, loading the + primary disk and the backup repository. The load peaks + at 00:00 UTC because `getSystemDate` rounds `updated_at` to the UTC day, so the first send to each + active queue after midnight runs `updateQueueTime`, clustering these writes and the `archive-push` + volume in the 00h hour. + +## Index bloat and autovacuum + +Autovacuum is running (60 runs on this table) and is not misconfigured, but two limits apply. + +Autovacuum reclaims heap dead tuples and refreshes planner statistics; it does not shrink indexes. +Only `REINDEX` or `pg_repack` rebuilds a bloated B-tree. Index bloat therefore has no automatic +remedy. + +Default autovacuum pacing assumes moderate churn. `autovacuum_vacuum_scale_factor = 0.2` waits until +20% of the table is dead (about 9.7M rows here) before vacuuming, and cost-based throttling caps its +I/O, so a hot table stays around 12% dead. Per-table tuning makes it run sooner and faster. + +The amplifier is indexing `updated_at`, a column updated on every send, which the fixes and operator +actions below address. + +## Implemented fixes (batch 1) + +Branch `sh/leaks-batch-1`, two commits. + +| commit | change | effect | +| --- | --- | --- | +| `88673eee` | migration `20260916_prometheus_indexes` adds partial indexes `idx_msg_queues_expire (recipient_id) WHERE deleted_at IS NULL AND msg_queue_expire` and `idx_msg_queues_notifier_active (notifier_id) WHERE deleted_at IS NULL AND notifier_id IS NOT NULL` | the sweep batch and `notifier_count` become index scans | +| `2912be1e` | `getEntityCounts.queue_count` uses `pg_class.reltuples` (`QueueStore/Postgres.hs:154`) | removes the 30 s, 990 GB whole-table `COUNT` on every scrape | + +Verified: both indexes are created and used as index-only scans, the schema-dump test passes, and +`smp-server` builds. `queueCount` is consumed only by metrics, logs, and display +(`Server.hs:528,698,815,2369`), so an estimate is safe. + +Not fixed by new indexes: + +- `ntf_service_queues_count` seq-scans only because of index bloat; `idx_msg_queues_ntf_service_id` + already covers it. `REINDEX` restores the index-only scan, as proven by `rcv_service_queues_count`. +- `rcv_service_queues_count` and the two `services` counts are already cheap. + +## Implemented fixes (batch 2) + +Migration `20260917_msg_queues_hot` and `MsgStore/Postgres.hs`. + +| change | effect | +| --- | --- | +| drop `idx_msg_queues_updated_at_recipient_id`, set `msg_queues` `fillfactor = 80` | `updateQueueTime` updates become HOT, so they rewrite no index entries; removes its index bloat and cuts its WAL | +| remove the dead `updated_at > p_old_queue` filter and its parameter from `expire_old_messages`, matching the `CALL` (`MsgStore/Postgres.hs:113`) | the sweep query runs as an index-only scan on `idx_msg_queues_expire`; `updated_at` is no longer read on the hot path | +| autovacuum reloptions on `msg_queues`, `messages` (and its TOAST), and `services` | keeps dead tuples and bloat low per table (see the table below) | + +Verified on PostgreSQL 16 with the daily-active pattern (~11% of queues updated per day): `updateQueueTime` +goes from 0% HOT (indexed `updated_at`) to 100% HOT at `fillfactor = 80` (96.7% at 90), and per-update +WAL drops about 4x. The HOT ratio depends on how many rows per page change before autovacuum reclaims +space, so it is workload-dependent; `fillfactor = 80` reached 100% for this fraction, and updating a much +larger share of queues at once would need a lower value. `fillfactor` applies to pages rewritten after +the change, so the ratio ramps up as the heap turns over, or immediately after `pg_repack`. + +Dropping the index regresses the deprecated journal message store's expiration (`foldRecentQueueRecs`, +`Journal.hs:434`) to a sequential scan; this is accepted because that store is being retired. The +Postgres message store does not read `updated_at`. + +Autovacuum reloptions applied by the migration: + +| table | reloptions | reason | +| --- | --- | --- | +| `msg_queues` | `fillfactor 80`, `autovacuum_vacuum_scale_factor 0.02`, `autovacuum_analyze_scale_factor 0.01`, `autovacuum_vacuum_cost_limit 1000` | HOT headroom; vacuum/analyze at ~2%/1% churn instead of 20%/10%; faster vacuum | +| `messages` (and TOAST) | `autovacuum_vacuum_scale_factor 0.02`, `autovacuum_analyze_scale_factor 0.01`, `toast.autovacuum_vacuum_scale_factor 0.02` | high insert and delete churn; 152 GB, mostly TOASTed message bodies | +| `services` | `fillfactor 70`, `autovacuum_vacuum_threshold 1000`, `autovacuum_vacuum_scale_factor 0` | tiny, very hot table was autovacuumed tens of thousands of times; make updates HOT and vacuum it far less | + +## Operator actions + +1. Reclaim index bloat (online, no lock; needs free disk near the index size; run largest first): + ``` + REINDEX INDEX CONCURRENTLY smp_server.idx_msg_queues_updated_at_recipient_id; + REINDEX INDEX CONCURRENTLY smp_server.idx_msg_queues_notifier_id; + REINDEX INDEX CONCURRENTLY smp_server.idx_msg_queues_ntf_service_id; + REINDEX INDEX CONCURRENTLY smp_server.idx_msg_queues_rcv_service_id; + REINDEX INDEX CONCURRENTLY smp_server.idx_msg_queues_sender_id; + REINDEX INDEX CONCURRENTLY smp_server.idx_msg_queues_link_id; + ANALYZE smp_server.msg_queues; + ``` + Expect `pg_indexes_size('smp_server.msg_queues')` to drop from 104 GB to roughly 12 GB (measured + after reindexing). `ANALYZE` restores the index-only scan for `ntf_service_queues_count`. With the + batch-2 `fillfactor` the index bloat no longer rebuilds, so this is a one-time reclaim. + +2. Autovacuum reloptions are applied by the batch-2 migration (`msg_queues`, `messages` and its TOAST, + `services`), so no manual `ALTER TABLE` is needed. Reclaim the current `messages` bloat once with + `pg_repack smp_server.messages` (152 GB heap and TOAST, roughly 30-50 GB reclaimable); running + `pg_repack` on `msg_queues` also makes `fillfactor = 80` effective immediately rather than as pages + turn over. With `fillfactor = 80` the `msg_queues` index bloat no longer rebuilds, so scheduled + reindexing is rarely needed. To automate it anyway, run `reindexdb --concurrently` or `pg_repack` + from a scheduler; `pg_cron` runs each job inside a transaction and cannot execute + `REINDEX ... CONCURRENTLY`, so use `pg_cron` only for `VACUUM`/`ANALYZE` or a non-concurrent + maintenance-window `REINDEX`. + +3. Cut WAL and checkpoint pressure (all reloadable, no restart): + ``` + ALTER SYSTEM SET wal_compression = 'lz4'; + ALTER SYSTEM SET max_wal_size = '32GB'; + ALTER SYSTEM SET checkpoint_timeout = '30min'; + SELECT pg_reload_conf(); + ``` + Fewer checkpoints produce fewer full-page images; `wal_compression` shrinks those that remain. + Together these reduce the WAL written on the primary and shipped by `archive-push`. The message + insert and delete volume itself is fixed by traffic, but most of its WAL cost is amplification + (full-page images and hint-bit writes on bloated pages), which these settings and `REINDEX` reduce. + This is separate from pgBackRest repo-side compression, which does not reduce WAL on the primary. + Verify with `pg_stat_reset_shared('wal')`, wait, then re-check `wal_fpi` and `wal_bytes` in + `pg_stat_wal`. + +4. Enable `track_io_timing = on` so disk wait time is measurable in `pg_stat_statements`. diff --git a/src/Simplex/Messaging/Server/MsgStore/Postgres.hs b/src/Simplex/Messaging/Server/MsgStore/Postgres.hs index b855346d4..a5b05b8e5 100644 --- a/src/Simplex/Messaging/Server/MsgStore/Postgres.hs +++ b/src/Simplex/Messaging/Server/MsgStore/Postgres.hs @@ -110,10 +110,9 @@ instance MsgStoreClass PostgresMsgStore where expireOldMessages :: Bool -> PostgresMsgStore -> Int64 -> Int64 -> IO MessageStats expireOldMessages _tty ms now ttl = maybeFirstRow' newMessageStats toMessageStats $ withConnection st $ \db -> - DB.query db "CALL expire_old_messages(?,?,?,0,0,0)" (oldQueue, oldMsg, batchSize) + DB.query db "CALL expire_old_messages(?,?,0,0,0)" (oldMsg, batchSize) where st = dbStore $ queueStore_ ms - oldQueue = 0 :: Int64 -- expire all queues oldMsg = now - ttl batchSize = 10000 :: Int toMessageStats (expiredMsgsCount, storedMsgsCount, storedQueues) = diff --git a/src/Simplex/Messaging/Server/QueueStore/Postgres.hs b/src/Simplex/Messaging/Server/QueueStore/Postgres.hs index 8738bee99..39f02131c 100644 --- a/src/Simplex/Messaging/Server/QueueStore/Postgres.hs +++ b/src/Simplex/Messaging/Server/QueueStore/Postgres.hs @@ -159,7 +159,8 @@ instance StoreQueueClass q => QueueStoreClass q (PostgresQueueStore q) where db [sql| SELECT - (SELECT COUNT(1) FROM msg_queues WHERE deleted_at IS NULL) AS queue_count, + -- estimate via reltuples to avoid a full heap scan on every scrape + (SELECT GREATEST(reltuples, 0)::bigint FROM pg_class WHERE oid = 'msg_queues'::regclass) AS queue_count, (SELECT COUNT(1) FROM msg_queues WHERE deleted_at IS NULL AND notifier_id IS NOT NULL) AS notifier_count, (SELECT COUNT(1) FROM services WHERE service_role = ?) AS rcv_service_count, (SELECT COUNT(1) FROM services WHERE service_role = ?) AS ntf_service_count, diff --git a/src/Simplex/Messaging/Server/QueueStore/Postgres/Migrations.hs b/src/Simplex/Messaging/Server/QueueStore/Postgres/Migrations.hs index fdcdeba0a..a42561430 100644 --- a/src/Simplex/Messaging/Server/QueueStore/Postgres/Migrations.hs +++ b/src/Simplex/Messaging/Server/QueueStore/Postgres/Migrations.hs @@ -20,7 +20,9 @@ serverSchemaMigrations = ("20250320_short_links", m20250320_short_links, Just down_m20250320_short_links), ("20250514_service_certs", m20250514_service_certs, Just down_m20250514_service_certs), ("20250903_store_messages", m20250903_store_messages, Just down_m20250903_store_messages), - ("20250915_queue_ids_hash", m20250915_queue_ids_hash, Just down_m20250915_queue_ids_hash) + ("20250915_queue_ids_hash", m20250915_queue_ids_hash, Just down_m20250915_queue_ids_hash), + ("20260916_prometheus_indexes", m20260916_prometheus_indexes, Just down_m20260916_prometheus_indexes), + ("20260917_msg_queues_hot", m20260917_msg_queues_hot, Just down_m20260917_msg_queues_hot) ] -- | The list of migrations in ascending order by date @@ -588,3 +590,147 @@ ALTER TABLE services DROP COLUMN queue_ids_hash; |] <> dropXorHashFuncs + +m20260916_prometheus_indexes :: Text +m20260916_prometheus_indexes = + [r| +CREATE INDEX idx_msg_queues_expire ON msg_queues (recipient_id) WHERE deleted_at IS NULL AND msg_queue_expire; +CREATE INDEX idx_msg_queues_notifier_active ON msg_queues (notifier_id) WHERE deleted_at IS NULL AND notifier_id IS NOT NULL; + |] + +down_m20260916_prometheus_indexes :: Text +down_m20260916_prometheus_indexes = + [r| +DROP INDEX idx_msg_queues_expire; +DROP INDEX idx_msg_queues_notifier_active; + |] + +m20260917_msg_queues_hot :: Text +m20260917_msg_queues_hot = + [r| +DROP INDEX idx_msg_queues_updated_at_recipient_id; + +ALTER TABLE msg_queues SET ( + fillfactor = 80, + autovacuum_vacuum_scale_factor = 0.02, + autovacuum_analyze_scale_factor = 0.01, + autovacuum_vacuum_cost_limit = 1000 +); + +ALTER TABLE messages SET ( + autovacuum_vacuum_scale_factor = 0.02, + autovacuum_analyze_scale_factor = 0.01, + toast.autovacuum_vacuum_scale_factor = 0.02 +); + +ALTER TABLE services SET ( + fillfactor = 70, + autovacuum_vacuum_threshold = 1000, + autovacuum_vacuum_scale_factor = 0 +); + +DROP PROCEDURE expire_old_messages(bigint, bigint, integer); + +CREATE PROCEDURE expire_old_messages(IN p_old_ts bigint, IN batch_size integer, OUT r_expired_msgs_count bigint, OUT r_stored_msgs_count bigint, OUT r_stored_queues bigint) + LANGUAGE plpgsql + AS $$ +DECLARE + rids BYTEA[]; + rid BYTEA; + last_rid BYTEA := '\x'; + del_count BIGINT; + total_deleted BIGINT := 0; +BEGIN + LOOP + SELECT array_agg(recipient_id) + INTO rids + FROM ( + SELECT recipient_id + FROM msg_queues + WHERE deleted_at IS NULL + AND msg_queue_expire = TRUE + AND recipient_id > last_rid + ORDER BY recipient_id ASC + LIMIT batch_size + ) qs; + + EXIT WHEN rids IS NULL OR cardinality(rids) = 0; + + FOREACH rid IN ARRAY rids + LOOP + BEGIN + del_count := delete_expired_msgs(rid, p_old_ts); + total_deleted := total_deleted + del_count; + EXCEPTION WHEN OTHERS THEN + RAISE WARNING 'STORE, expire_old_messages, error expiring queue %: %', encode(rid, 'base64'), SQLERRM; + CONTINUE; + END; + COMMIT; + END LOOP; + last_rid := rids[cardinality(rids)]; + END LOOP; + + r_expired_msgs_count := total_deleted; + r_stored_msgs_count := (SELECT COUNT(1) FROM messages); + r_stored_queues := (SELECT COUNT(1) FROM msg_queues WHERE deleted_at IS NULL); +END; +$$; + |] + +down_m20260917_msg_queues_hot :: Text +down_m20260917_msg_queues_hot = + [r| +DROP PROCEDURE expire_old_messages(bigint, integer); + +CREATE PROCEDURE expire_old_messages(IN p_old_queue bigint, IN p_old_ts bigint, IN batch_size integer, OUT r_expired_msgs_count bigint, OUT r_stored_msgs_count bigint, OUT r_stored_queues bigint) + LANGUAGE plpgsql + AS $$ +DECLARE + rids BYTEA[]; + rid BYTEA; + last_rid BYTEA := '\x'; + del_count BIGINT; + total_deleted BIGINT := 0; +BEGIN + LOOP + SELECT array_agg(recipient_id) + INTO rids + FROM ( + SELECT recipient_id + FROM msg_queues + WHERE deleted_at IS NULL + AND updated_at > p_old_queue + AND msg_queue_expire = TRUE + AND recipient_id > last_rid + ORDER BY recipient_id ASC + LIMIT batch_size + ) qs; + + EXIT WHEN rids IS NULL OR cardinality(rids) = 0; + + FOREACH rid IN ARRAY rids + LOOP + BEGIN + del_count := delete_expired_msgs(rid, p_old_ts); + total_deleted := total_deleted + del_count; + EXCEPTION WHEN OTHERS THEN + RAISE WARNING 'STORE, expire_old_messages, error expiring queue %: %', encode(rid, 'base64'), SQLERRM; + CONTINUE; + END; + COMMIT; + END LOOP; + last_rid := rids[cardinality(rids)]; + END LOOP; + + r_expired_msgs_count := total_deleted; + r_stored_msgs_count := (SELECT COUNT(1) FROM messages); + r_stored_queues := (SELECT COUNT(1) FROM msg_queues WHERE deleted_at IS NULL); +END; +$$; + +CREATE INDEX idx_msg_queues_updated_at_recipient_id ON msg_queues (deleted_at, updated_at, msg_queue_expire, recipient_id); + +ALTER TABLE msg_queues RESET (fillfactor, autovacuum_vacuum_scale_factor, autovacuum_analyze_scale_factor, autovacuum_vacuum_cost_limit); +ALTER TABLE messages RESET (autovacuum_vacuum_scale_factor, autovacuum_analyze_scale_factor, toast.autovacuum_vacuum_scale_factor); +ALTER TABLE services RESET (fillfactor, autovacuum_vacuum_threshold, autovacuum_vacuum_scale_factor); + |] diff --git a/src/Simplex/Messaging/Server/QueueStore/Postgres/server_schema.sql b/src/Simplex/Messaging/Server/QueueStore/Postgres/server_schema.sql index f0da5272d..6c6cd27b0 100644 --- a/src/Simplex/Messaging/Server/QueueStore/Postgres/server_schema.sql +++ b/src/Simplex/Messaging/Server/QueueStore/Postgres/server_schema.sql @@ -1,5 +1,6 @@ + SET statement_timeout = 0; SET lock_timeout = 0; SET idle_in_transaction_session_timeout = 0; @@ -56,7 +57,7 @@ $$; -CREATE PROCEDURE smp_server.expire_old_messages(IN p_old_queue bigint, IN p_old_ts bigint, IN batch_size integer, OUT r_expired_msgs_count bigint, OUT r_stored_msgs_count bigint, OUT r_stored_queues bigint) +CREATE PROCEDURE smp_server.expire_old_messages(IN p_old_ts bigint, IN batch_size integer, OUT r_expired_msgs_count bigint, OUT r_stored_msgs_count bigint, OUT r_stored_queues bigint) LANGUAGE plpgsql AS $$ DECLARE @@ -73,7 +74,6 @@ BEGIN SELECT recipient_id FROM msg_queues WHERE deleted_at IS NULL - AND updated_at > p_old_queue AND msg_queue_expire = TRUE AND recipient_id > last_rid ORDER BY recipient_id ASC @@ -397,7 +397,8 @@ CREATE TABLE smp_server.messages ( msg_quota boolean NOT NULL, msg_ntf_flag boolean NOT NULL, msg_body bytea NOT NULL -); +) +WITH (autovacuum_vacuum_scale_factor='0.02', autovacuum_analyze_scale_factor='0.01', toast.autovacuum_vacuum_scale_factor='0.02'); @@ -441,7 +442,8 @@ CREATE TABLE smp_server.msg_queues ( msg_can_write boolean DEFAULT true NOT NULL, msg_queue_expire boolean DEFAULT false NOT NULL, msg_queue_size bigint DEFAULT 0 NOT NULL -); +) +WITH (fillfactor='80', autovacuum_vacuum_scale_factor='0.02', autovacuum_analyze_scale_factor='0.01', autovacuum_vacuum_cost_limit='1000'); @@ -453,7 +455,8 @@ CREATE TABLE smp_server.services ( created_at bigint NOT NULL, queue_count bigint DEFAULT 0 NOT NULL, queue_ids_hash bytea DEFAULT '\x00000000000000000000000000000000'::bytea NOT NULL -); +) +WITH (fillfactor='70', autovacuum_vacuum_threshold='1000', autovacuum_vacuum_scale_factor='0'); @@ -494,10 +497,18 @@ CREATE INDEX idx_messages_recipient_id_msg_ts ON smp_server.messages USING btree +CREATE INDEX idx_msg_queues_expire ON smp_server.msg_queues USING btree (recipient_id) WHERE ((deleted_at IS NULL) AND msg_queue_expire); + + + CREATE UNIQUE INDEX idx_msg_queues_link_id ON smp_server.msg_queues USING btree (link_id); +CREATE INDEX idx_msg_queues_notifier_active ON smp_server.msg_queues USING btree (notifier_id) WHERE ((deleted_at IS NULL) AND (notifier_id IS NOT NULL)); + + + CREATE UNIQUE INDEX idx_msg_queues_notifier_id ON smp_server.msg_queues USING btree (notifier_id); @@ -514,10 +525,6 @@ CREATE UNIQUE INDEX idx_msg_queues_sender_id ON smp_server.msg_queues USING btre -CREATE INDEX idx_msg_queues_updated_at_recipient_id ON smp_server.msg_queues USING btree (deleted_at, updated_at, msg_queue_expire, recipient_id); - - - CREATE INDEX idx_services_service_role ON smp_server.services USING btree (service_role); @@ -549,3 +556,4 @@ ALTER TABLE ONLY smp_server.msg_queues +