Checkpoint SQLite WAL off the request thread safely

Disable wal_autocheckpoint and run PASSIVE checkpoints from one elected
writer process. Contenders start in console/rake (initializer), Puma
workers (including WEB_CONCURRENCY=auto), and Resque children, with
failover, capped error backoff, and tests for lock takeover.

Co-authored-by: Thomas Klemm <github@tklemm.eu>
This commit is contained in:
Cursor Agent
2026-10-07 18:53:19 +00:00
parent 6e312c6028
commit 88acc4e026
7 changed files with 389 additions and 1 deletions
+4
View File
@@ -9,6 +9,10 @@ default: &default
pool: <%= ENV.fetch("RAILS_MAX_THREADS") { 10 } %>
timeout: 5000
default_transaction_mode: immediate
# Checkpoint on a background connection instead (SqliteWalCheckpoint). Every
# non-test writer process runs a contender so this is never left without one.
pragmas:
wal_autocheckpoint: 0
development:
primary:
@@ -0,0 +1,10 @@
# Start a checkpoint contender in console, runner, rake and other non-Puma writers.
# Puma skips this path: config/puma.rb and config/puma_dev.rb start after the
# correct process is chosen (worker boot vs single-process), so the master that
# only forks workers never holds the lock alone. Resque starts after_prefork.
Rails.application.config.after_initialize do
next if Rails.env.test?
next if defined?(Puma::CLI)
SqliteWalCheckpoint.start
end
+13 -1
View File
@@ -33,7 +33,10 @@ pidfile ENV.fetch("PIDFILE") { "tmp/pids/server.pid" }
# processes).
#
worker_count = (Concurrent.processor_count * 0.666).ceil
workers ENV.fetch("WEB_CONCURRENCY") { worker_count }
# Keep the raw setting: WEB_CONCURRENCY=auto must not be coerced with to_i (that
# is 0 and would look like single-process mode while Puma still forks workers).
configured_workers = ENV.fetch("WEB_CONCURRENCY") { worker_count }
workers configured_workers
ENV["JOB_CONCURRENCY"] ||= worker_count.to_s
@@ -50,6 +53,15 @@ plugin :tmp_restart
# Reset all membership connections
Membership.disconnect_all
# Only the literal 0 is single-process. "auto" and positive counts fork workers;
# those must start the checkpointer after boot so a surviving worker can take
# the lock if the previous leader exits.
if SqliteWalCheckpoint.single_puma_process?(configured_workers)
SqliteWalCheckpoint.start
else
on_worker_boot { SqliteWalCheckpoint.start }
end
Signal.trap :SIGPROF do
Thread.list.each do |t|
puts t
+2
View File
@@ -1,5 +1,7 @@
require File.expand_path("../config/environment", File.dirname(__FILE__))
SqliteWalCheckpoint.start
Signal.trap :SIGPROF do
Thread.list.each do |t|
puts t
+181
View File
@@ -0,0 +1,181 @@
# SQLite's default WAL auto-checkpoint (~1,000 pages) runs inside the committing
# writer and fsyncs during the request. database.yml sets wal_autocheckpoint=0;
# this module copies pages off the request thread with PASSIVE checkpoints.
#
# Every non-test writer process starts a contender thread (initializer, Puma,
# Resque). A file lock elects one leader; if that process exits, another takes
# over on the next interval so auto-checkpoint stays off safely.
module SqliteWalCheckpoint
INTERVAL = 0.25
MAX_BACKOFF = 30.0
LOCK_PATH = Rails.root.join("tmp/pids/sqlite_wal_checkpoint.lock")
class << self
attr_writer :lock_path, :database_path_override
def start(interval: INTERVAL, enabled: !Rails.env.test?)
return unless enabled
@mutex ||= Mutex.new
@mutex.synchronize do
return if @thread&.alive?
@stop = false
install_exit_handler
@thread = Thread.new { run(interval) }
@thread.report_on_exception = false
end
@thread
end
def stop
@stop = true
thread = @mutex&.synchronize { @thread }
thread&.join(2)
ensure
@mutex&.synchronize { @thread = nil if @thread && !@thread.alive? }
release_lock
end
def checkpoint
with_database { |database| checkpoint_on(database) }
end
# One leadership attempt for tests: acquire the lock, checkpoint once, release.
def tick
return :no_database unless database_path
return :busy unless acquire_lock
begin
result = checkpoint
result.nil? ? :no_database : :checkpointed
ensure
release_lock
end
end
def lock_path
@lock_path || LOCK_PATH
end
def reset!
stop
@lock_path = nil
@database_path_override = nil
@stop = false
@exit_handler_installed = false
end
# Puma's WEB_CONCURRENCY=auto must not use to_i (that is 0). Only the literal
# 0 means single-process mode; everything else forks workers.
def single_puma_process?(configured_workers)
configured_workers.to_s == "0"
end
private
def run(interval)
Thread.current.name = "sqlite-wal-checkpoint"
backoff = interval
until @stop
begin
unless database_path
interruptible_sleep interval
next
end
unless acquire_lock
interruptible_sleep interval
next
end
begin
with_database do |database|
until @stop
checkpoint_on(database)
backoff = interval
interruptible_sleep interval
end
end
ensure
release_lock
end
rescue => error
Rails.logger.warn "SQLite WAL checkpoint failed: #{error.class}: #{error.message}"
interruptible_sleep backoff
backoff = [ backoff * 2, MAX_BACKOFF ].min
end
end
ensure
release_lock
end
def checkpoint_on(database)
database.execute("PRAGMA wal_checkpoint(PASSIVE)").first
end
def with_database
path = database_path
return unless path && File.exist?(path)
result = nil
SQLite3::Database.new(path) do |database|
database.busy_handler_timeout = 5_000
result = yield database
end
result
end
def database_path
return @database_path_override if @database_path_override
config = ActiveRecord::Base.connection_db_config
return unless config.adapter.to_s == "sqlite3"
ActiveRecord::ConnectionAdapters::SQLite3Adapter.resolve_path(config.database)
end
def acquire_lock
FileUtils.mkdir_p(File.dirname(lock_path))
file = File.open(lock_path, File::RDWR | File::CREAT, 0644)
if file.flock(File::LOCK_EX | File::LOCK_NB)
@lock_file = file
true
else
file.close
false
end
end
def release_lock
return unless @lock_file
@lock_file.flock(File::LOCK_UN)
@lock_file.close
rescue Errno::EBADF, IOError
# Already closed by a racing ensure / exit handler.
ensure
@lock_file = nil
end
def interruptible_sleep(seconds)
deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds
while !@stop && (remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)) > 0
sleep [ remaining, 0.05 ].min
end
end
def install_exit_handler
return if @exit_handler_installed
@exit_handler_installed = true
at_exit do
if @lock_file
checkpoint rescue nil
end
stop
end
end
end
end
+1
View File
@@ -8,5 +8,6 @@ task "resque:pool:setup" do
Resque::Pool.after_prefork do |job|
ActiveRecord::Base.establish_connection
Resque.redis.client.close
SqliteWalCheckpoint.start
end
end
+178
View File
@@ -0,0 +1,178 @@
require "test_helper"
class SqliteWalCheckpointTest < ActiveSupport::TestCase
setup do
@tmpdir = Dir.mktmpdir("sqlite-wal-checkpoint")
SqliteWalCheckpoint.reset!
SqliteWalCheckpoint.lock_path = File.join(@tmpdir, "checkpoint.lock")
end
teardown do
SqliteWalCheckpoint.reset!
FileUtils.remove_entry(@tmpdir) if @tmpdir && File.directory?(@tmpdir)
end
test "connections disable WAL auto-checkpoint so commits do not fsync on the writer" do
assert_equal 0, ActiveRecord::Base.connection.select_value("PRAGMA wal_autocheckpoint").to_i
end
test "a passive checkpoint against the primary database does not raise" do
assert_nothing_raised { SqliteWalCheckpoint.checkpoint }
end
test "does not start a background thread in test by default" do
assert_nil SqliteWalCheckpoint.start
assert_empty checkpoint_threads
end
test "start with enabled: true runs a named contender thread that stop joins" do
thread = SqliteWalCheckpoint.start(interval: 0.05, enabled: true)
assert thread.alive?
wait_until { thread.name == "sqlite-wal-checkpoint" }
SqliteWalCheckpoint.stop
assert_not thread.alive?
assert_empty checkpoint_threads
end
test "single_puma_process? treats only literal zero as single-process" do
assert SqliteWalCheckpoint.single_puma_process?(0)
assert SqliteWalCheckpoint.single_puma_process?("0")
assert_not SqliteWalCheckpoint.single_puma_process?("auto")
assert_not SqliteWalCheckpoint.single_puma_process?(2)
assert_not SqliteWalCheckpoint.single_puma_process?("2")
assert_not SqliteWalCheckpoint.single_puma_process?(8)
end
test "tick checkpoints through the elected lock holder" do
db_path = build_wal_database(rows: 50)
SqliteWalCheckpoint.database_path_override = db_path
assert_equal :checkpointed, SqliteWalCheckpoint.tick
end
test "tick reports busy while another process holds the lock and succeeds after it exits" do
db_path = build_wal_database(rows: 20)
SqliteWalCheckpoint.database_path_override = db_path
lock = SqliteWalCheckpoint.lock_path
ready = File.join(@tmpdir, "holder-ready")
holder = spawn_lock_holder(lock, ready_path: ready, hold_for: 0.4)
begin
wait_until { File.exist?(ready) }
assert_equal :busy, SqliteWalCheckpoint.tick
ensure
Process.wait(holder)
end
assert_equal :checkpointed, SqliteWalCheckpoint.tick
end
test "passive checkpoint moves WAL pages after writes with autocheckpoint disabled" do
db_path = File.join(@tmpdir, "writer-#{SecureRandom.hex(4)}.sqlite3")
SQLite3::Database.new(db_path) do |writer|
writer.execute("PRAGMA journal_mode=WAL")
writer.execute("PRAGMA wal_autocheckpoint=0")
writer.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, body TEXT)")
200.times do |i|
writer.execute("INSERT INTO items (body) VALUES (?)", "row-#{i}-#{"x" * 200}")
end
wal_path = "#{db_path}-wal"
assert File.exist?(wal_path), "expected a WAL file after inserts"
assert File.size(wal_path) > 0
SqliteWalCheckpoint.database_path_override = db_path
assert SqliteWalCheckpoint.send(:acquire_lock)
begin
_busy, _log, checkpointed = SqliteWalCheckpoint.checkpoint
assert checkpointed.to_i > 0, "expected PASSIVE checkpoint to copy WAL pages, got #{checkpointed.inspect}"
ensure
SqliteWalCheckpoint.send(:release_lock)
end
end
end
test "a contender thread takes over checkpointing after the lock holder stops" do
db_path = build_wal_database(rows: 30)
SqliteWalCheckpoint.database_path_override = db_path
lock = SqliteWalCheckpoint.lock_path
ready = File.join(@tmpdir, "holder-ready")
holder = spawn_lock_holder(lock, ready_path: ready, hold_for: 0.25)
wait_until { File.exist?(ready) }
contender = SqliteWalCheckpoint.start(interval: 0.05, enabled: true)
begin
assert_equal :busy, SqliteWalCheckpoint.tick
Process.wait(holder)
holder = nil
wait_until(timeout: 2) { lock_held_by_other_process?(lock) }
ensure
Process.wait(holder) if holder
SqliteWalCheckpoint.stop
assert_not contender.alive?
end
end
private
def checkpoint_threads
Thread.list.select { |thread| thread.name == "sqlite-wal-checkpoint" }
end
def build_wal_database(rows:)
path = File.join(@tmpdir, "writer-#{SecureRandom.hex(4)}.sqlite3")
SQLite3::Database.new(path) do |database|
database.execute("PRAGMA journal_mode=WAL")
database.execute("PRAGMA wal_autocheckpoint=0")
database.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, body TEXT)")
rows.times do |i|
database.execute("INSERT INTO items (body) VALUES (?)", "row-#{i}-#{"x" * 200}")
end
end
path
end
def spawn_lock_holder(lock, ready_path:, hold_for:)
Process.spawn(
RbConfig.ruby, "-e", <<~RUBY
require "fileutils"
lock = #{lock.inspect}
ready = #{ready_path.inspect}
FileUtils.mkdir_p(File.dirname(lock))
file = File.open(lock, File::RDWR | File::CREAT, 0644)
abort "lock failed" unless file.flock(File::LOCK_EX | File::LOCK_NB)
File.write(ready, "1")
sleep #{hold_for}
RUBY
)
end
def lock_held_by_other_process?(lock)
return false unless File.exist?(lock)
probe = File.open(lock, File::RDWR | File::CREAT, 0644)
if probe.flock(File::LOCK_EX | File::LOCK_NB)
probe.flock(File::LOCK_UN)
probe.close
false
else
probe.close
true
end
end
def wait_until(timeout: 1)
deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout
until yield
raise "condition not met within #{timeout}s" if Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline
sleep 0.02
end
end
end