diff --git a/changelog.d/20285.misc b/changelog.d/20285.misc new file mode 100644 index 0000000000..3eb3f49f99 --- /dev/null +++ b/changelog.d/20285.misc @@ -0,0 +1 @@ +Prune `user_ips` less frequently (less stress on the database). diff --git a/synapse/storage/databases/main/client_ips.py b/synapse/storage/databases/main/client_ips.py index 7cd3667a2b..f9e36ea8be 100644 --- a/synapse/storage/databases/main/client_ips.py +++ b/synapse/storage/databases/main/client_ips.py @@ -438,7 +438,23 @@ class ClientIpWorkerStore(ClientIpBackgroundUpdateStore, MonthlyActiveUsersWorke ) if hs.config.worker.run_background_tasks and self.user_ips_max_age: - self.clock.looping_call(self._prune_old_user_ips, Duration(seconds=5)) + self.clock.looping_call( + self._prune_old_user_ips, + # Based on a measured value of ~9.17 average deletes/second on the + # `user_ips` table on `matrix.org`. If we prune every 60 seconds, this + # query will pick-up ~550 rows which is under the query `LIMIT` set + # (5000) with about an order of magnitude head-room to catch-up from + # downtime or peak/heavy traffic. On `matrix.org`, the peak delete + # activity on the `user_ips` table is ~2x the daily low. + # + # Running this more often seems good as that means we pick up a smaller + # number of rows each time (less work), therefore less disruptive the + # queries will be. But in practice, the query still has a fixed cost as + # it has to scan through dead tuples left behind by all of the previous + # prunes until the table is vacuumed which isn't free so just run this a + # reasonable amount. + Duration(seconds=60), + ) if self._update_on_this_worker: # This is the designated worker that can write to the client IP