Checkpoint SQLite WAL off the request thread safely

Disable wal_autocheckpoint and run PASSIVE checkpoints from one elected
writer process. Contenders start in console/rake (initializer), Puma
workers (including WEB_CONCURRENCY=auto), and Resque children, with
failover, capped error backoff, and tests for lock takeover.

Co-authored-by: Thomas Klemm <github@tklemm.eu>
This commit is contained in:
Cursor Agent
2026-10-07 18:53:19 +00:00
parent 6e312c6028
commit 88acc4e026
7 changed files with 389 additions and 1 deletions
+4
View File
@@ -9,6 +9,10 @@ default: &default
pool: <%= ENV.fetch("RAILS_MAX_THREADS") { 10 } %>
timeout: 5000
default_transaction_mode: immediate
# Checkpoint on a background connection instead (SqliteWalCheckpoint). Every
# non-test writer process runs a contender so this is never left without one.
pragmas:
wal_autocheckpoint: 0
development:
primary:
@@ -0,0 +1,10 @@
# Start a checkpoint contender in console, runner, rake and other non-Puma writers.
# Puma skips this path: config/puma.rb and config/puma_dev.rb start after the
# correct process is chosen (worker boot vs single-process), so the master that
# only forks workers never holds the lock alone. Resque starts after_prefork.
Rails.application.config.after_initialize do
next if Rails.env.test?
next if defined?(Puma::CLI)
SqliteWalCheckpoint.start
end
+13 -1
View File
@@ -33,7 +33,10 @@ pidfile ENV.fetch("PIDFILE") { "tmp/pids/server.pid" }
# processes).
#
worker_count = (Concurrent.processor_count * 0.666).ceil
workers ENV.fetch("WEB_CONCURRENCY") { worker_count }
# Keep the raw setting: WEB_CONCURRENCY=auto must not be coerced with to_i (that
# is 0 and would look like single-process mode while Puma still forks workers).
configured_workers = ENV.fetch("WEB_CONCURRENCY") { worker_count }
workers configured_workers
ENV["JOB_CONCURRENCY"] ||= worker_count.to_s
@@ -50,6 +53,15 @@ plugin :tmp_restart
# Reset all membership connections
Membership.disconnect_all
# Only the literal 0 is single-process. "auto" and positive counts fork workers;
# those must start the checkpointer after boot so a surviving worker can take
# the lock if the previous leader exits.
if SqliteWalCheckpoint.single_puma_process?(configured_workers)
SqliteWalCheckpoint.start
else
on_worker_boot { SqliteWalCheckpoint.start }
end
Signal.trap :SIGPROF do
Thread.list.each do |t|
puts t
+2
View File
@@ -1,5 +1,7 @@
require File.expand_path("../config/environment", File.dirname(__FILE__))
SqliteWalCheckpoint.start
Signal.trap :SIGPROF do
Thread.list.each do |t|
puts t