From 88acc4e026025dcfe40acb22fbabd5507e88529e Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 18:53:19 +0000 Subject: [PATCH 1/4] Checkpoint SQLite WAL off the request thread safely Disable wal_autocheckpoint and run PASSIVE checkpoints from one elected writer process. Contenders start in console/rake (initializer), Puma workers (including WEB_CONCURRENCY=auto), and Resque children, with failover, capped error backoff, and tests for lock takeover. Co-authored-by: Thomas Klemm --- config/database.yml | 4 + config/initializers/sqlite_wal_checkpoint.rb | 10 + config/puma.rb | 14 +- config/puma_dev.rb | 2 + lib/sqlite_wal_checkpoint.rb | 181 +++++++++++++++++++ lib/tasks/resque.rake | 1 + test/lib/sqlite_wal_checkpoint_test.rb | 178 ++++++++++++++++++ 7 files changed, 389 insertions(+), 1 deletion(-) create mode 100644 config/initializers/sqlite_wal_checkpoint.rb create mode 100644 lib/sqlite_wal_checkpoint.rb create mode 100644 test/lib/sqlite_wal_checkpoint_test.rb diff --git a/config/database.yml b/config/database.yml index ec220a4..2f8472a 100644 --- a/config/database.yml +++ b/config/database.yml @@ -9,6 +9,10 @@ default: &default pool: <%= ENV.fetch("RAILS_MAX_THREADS") { 10 } %> timeout: 5000 default_transaction_mode: immediate + # Checkpoint on a background connection instead (SqliteWalCheckpoint). Every + # non-test writer process runs a contender so this is never left without one. + pragmas: + wal_autocheckpoint: 0 development: primary: diff --git a/config/initializers/sqlite_wal_checkpoint.rb b/config/initializers/sqlite_wal_checkpoint.rb new file mode 100644 index 0000000..ccf901a --- /dev/null +++ b/config/initializers/sqlite_wal_checkpoint.rb @@ -0,0 +1,10 @@ +# Start a checkpoint contender in console, runner, rake and other non-Puma writers. +# Puma skips this path: config/puma.rb and config/puma_dev.rb start after the +# correct process is chosen (worker boot vs single-process), so the master that +# only forks workers never holds the lock alone. Resque starts after_prefork. +Rails.application.config.after_initialize do + next if Rails.env.test? + next if defined?(Puma::CLI) + + SqliteWalCheckpoint.start +end diff --git a/config/puma.rb b/config/puma.rb index 960ab6a..6e1c7fc 100644 --- a/config/puma.rb +++ b/config/puma.rb @@ -33,7 +33,10 @@ pidfile ENV.fetch("PIDFILE") { "tmp/pids/server.pid" } # processes). # worker_count = (Concurrent.processor_count * 0.666).ceil -workers ENV.fetch("WEB_CONCURRENCY") { worker_count } +# Keep the raw setting: WEB_CONCURRENCY=auto must not be coerced with to_i (that +# is 0 and would look like single-process mode while Puma still forks workers). +configured_workers = ENV.fetch("WEB_CONCURRENCY") { worker_count } +workers configured_workers ENV["JOB_CONCURRENCY"] ||= worker_count.to_s @@ -50,6 +53,15 @@ plugin :tmp_restart # Reset all membership connections Membership.disconnect_all +# Only the literal 0 is single-process. "auto" and positive counts fork workers; +# those must start the checkpointer after boot so a surviving worker can take +# the lock if the previous leader exits. +if SqliteWalCheckpoint.single_puma_process?(configured_workers) + SqliteWalCheckpoint.start +else + on_worker_boot { SqliteWalCheckpoint.start } +end + Signal.trap :SIGPROF do Thread.list.each do |t| puts t diff --git a/config/puma_dev.rb b/config/puma_dev.rb index 371759a..2c7e512 100644 --- a/config/puma_dev.rb +++ b/config/puma_dev.rb @@ -1,5 +1,7 @@ require File.expand_path("../config/environment", File.dirname(__FILE__)) +SqliteWalCheckpoint.start + Signal.trap :SIGPROF do Thread.list.each do |t| puts t diff --git a/lib/sqlite_wal_checkpoint.rb b/lib/sqlite_wal_checkpoint.rb new file mode 100644 index 0000000..1060b3c --- /dev/null +++ b/lib/sqlite_wal_checkpoint.rb @@ -0,0 +1,181 @@ +# SQLite's default WAL auto-checkpoint (~1,000 pages) runs inside the committing +# writer and fsyncs during the request. database.yml sets wal_autocheckpoint=0; +# this module copies pages off the request thread with PASSIVE checkpoints. +# +# Every non-test writer process starts a contender thread (initializer, Puma, +# Resque). A file lock elects one leader; if that process exits, another takes +# over on the next interval so auto-checkpoint stays off safely. +module SqliteWalCheckpoint + INTERVAL = 0.25 + MAX_BACKOFF = 30.0 + LOCK_PATH = Rails.root.join("tmp/pids/sqlite_wal_checkpoint.lock") + + class << self + attr_writer :lock_path, :database_path_override + + def start(interval: INTERVAL, enabled: !Rails.env.test?) + return unless enabled + + @mutex ||= Mutex.new + @mutex.synchronize do + return if @thread&.alive? + + @stop = false + install_exit_handler + @thread = Thread.new { run(interval) } + @thread.report_on_exception = false + end + + @thread + end + + def stop + @stop = true + thread = @mutex&.synchronize { @thread } + thread&.join(2) + ensure + @mutex&.synchronize { @thread = nil if @thread && !@thread.alive? } + release_lock + end + + def checkpoint + with_database { |database| checkpoint_on(database) } + end + + # One leadership attempt for tests: acquire the lock, checkpoint once, release. + def tick + return :no_database unless database_path + return :busy unless acquire_lock + + begin + result = checkpoint + result.nil? ? :no_database : :checkpointed + ensure + release_lock + end + end + + def lock_path + @lock_path || LOCK_PATH + end + + def reset! + stop + @lock_path = nil + @database_path_override = nil + @stop = false + @exit_handler_installed = false + end + + # Puma's WEB_CONCURRENCY=auto must not use to_i (that is 0). Only the literal + # 0 means single-process mode; everything else forks workers. + def single_puma_process?(configured_workers) + configured_workers.to_s == "0" + end + + private + def run(interval) + Thread.current.name = "sqlite-wal-checkpoint" + backoff = interval + + until @stop + begin + unless database_path + interruptible_sleep interval + next + end + + unless acquire_lock + interruptible_sleep interval + next + end + + begin + with_database do |database| + until @stop + checkpoint_on(database) + backoff = interval + interruptible_sleep interval + end + end + ensure + release_lock + end + rescue => error + Rails.logger.warn "SQLite WAL checkpoint failed: #{error.class}: #{error.message}" + interruptible_sleep backoff + backoff = [ backoff * 2, MAX_BACKOFF ].min + end + end + ensure + release_lock + end + + def checkpoint_on(database) + database.execute("PRAGMA wal_checkpoint(PASSIVE)").first + end + + def with_database + path = database_path + return unless path && File.exist?(path) + + result = nil + SQLite3::Database.new(path) do |database| + database.busy_handler_timeout = 5_000 + result = yield database + end + result + end + + def database_path + return @database_path_override if @database_path_override + + config = ActiveRecord::Base.connection_db_config + return unless config.adapter.to_s == "sqlite3" + + ActiveRecord::ConnectionAdapters::SQLite3Adapter.resolve_path(config.database) + end + + def acquire_lock + FileUtils.mkdir_p(File.dirname(lock_path)) + file = File.open(lock_path, File::RDWR | File::CREAT, 0644) + if file.flock(File::LOCK_EX | File::LOCK_NB) + @lock_file = file + true + else + file.close + false + end + end + + def release_lock + return unless @lock_file + + @lock_file.flock(File::LOCK_UN) + @lock_file.close + rescue Errno::EBADF, IOError + # Already closed by a racing ensure / exit handler. + ensure + @lock_file = nil + end + + def interruptible_sleep(seconds) + deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds + while !@stop && (remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)) > 0 + sleep [ remaining, 0.05 ].min + end + end + + def install_exit_handler + return if @exit_handler_installed + + @exit_handler_installed = true + at_exit do + if @lock_file + checkpoint rescue nil + end + stop + end + end + end +end diff --git a/lib/tasks/resque.rake b/lib/tasks/resque.rake index 63ecc9d..5b8bf1a 100644 --- a/lib/tasks/resque.rake +++ b/lib/tasks/resque.rake @@ -8,5 +8,6 @@ task "resque:pool:setup" do Resque::Pool.after_prefork do |job| ActiveRecord::Base.establish_connection Resque.redis.client.close + SqliteWalCheckpoint.start end end diff --git a/test/lib/sqlite_wal_checkpoint_test.rb b/test/lib/sqlite_wal_checkpoint_test.rb new file mode 100644 index 0000000..e38a82c --- /dev/null +++ b/test/lib/sqlite_wal_checkpoint_test.rb @@ -0,0 +1,178 @@ +require "test_helper" + +class SqliteWalCheckpointTest < ActiveSupport::TestCase + setup do + @tmpdir = Dir.mktmpdir("sqlite-wal-checkpoint") + SqliteWalCheckpoint.reset! + SqliteWalCheckpoint.lock_path = File.join(@tmpdir, "checkpoint.lock") + end + + teardown do + SqliteWalCheckpoint.reset! + FileUtils.remove_entry(@tmpdir) if @tmpdir && File.directory?(@tmpdir) + end + + test "connections disable WAL auto-checkpoint so commits do not fsync on the writer" do + assert_equal 0, ActiveRecord::Base.connection.select_value("PRAGMA wal_autocheckpoint").to_i + end + + test "a passive checkpoint against the primary database does not raise" do + assert_nothing_raised { SqliteWalCheckpoint.checkpoint } + end + + test "does not start a background thread in test by default" do + assert_nil SqliteWalCheckpoint.start + assert_empty checkpoint_threads + end + + test "start with enabled: true runs a named contender thread that stop joins" do + thread = SqliteWalCheckpoint.start(interval: 0.05, enabled: true) + assert thread.alive? + wait_until { thread.name == "sqlite-wal-checkpoint" } + + SqliteWalCheckpoint.stop + assert_not thread.alive? + assert_empty checkpoint_threads + end + + test "single_puma_process? treats only literal zero as single-process" do + assert SqliteWalCheckpoint.single_puma_process?(0) + assert SqliteWalCheckpoint.single_puma_process?("0") + assert_not SqliteWalCheckpoint.single_puma_process?("auto") + assert_not SqliteWalCheckpoint.single_puma_process?(2) + assert_not SqliteWalCheckpoint.single_puma_process?("2") + assert_not SqliteWalCheckpoint.single_puma_process?(8) + end + + test "tick checkpoints through the elected lock holder" do + db_path = build_wal_database(rows: 50) + SqliteWalCheckpoint.database_path_override = db_path + + assert_equal :checkpointed, SqliteWalCheckpoint.tick + end + + test "tick reports busy while another process holds the lock and succeeds after it exits" do + db_path = build_wal_database(rows: 20) + SqliteWalCheckpoint.database_path_override = db_path + lock = SqliteWalCheckpoint.lock_path + ready = File.join(@tmpdir, "holder-ready") + + holder = spawn_lock_holder(lock, ready_path: ready, hold_for: 0.4) + + begin + wait_until { File.exist?(ready) } + assert_equal :busy, SqliteWalCheckpoint.tick + ensure + Process.wait(holder) + end + + assert_equal :checkpointed, SqliteWalCheckpoint.tick + end + + test "passive checkpoint moves WAL pages after writes with autocheckpoint disabled" do + db_path = File.join(@tmpdir, "writer-#{SecureRandom.hex(4)}.sqlite3") + + SQLite3::Database.new(db_path) do |writer| + writer.execute("PRAGMA journal_mode=WAL") + writer.execute("PRAGMA wal_autocheckpoint=0") + writer.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, body TEXT)") + 200.times do |i| + writer.execute("INSERT INTO items (body) VALUES (?)", "row-#{i}-#{"x" * 200}") + end + + wal_path = "#{db_path}-wal" + assert File.exist?(wal_path), "expected a WAL file after inserts" + assert File.size(wal_path) > 0 + + SqliteWalCheckpoint.database_path_override = db_path + assert SqliteWalCheckpoint.send(:acquire_lock) + begin + _busy, _log, checkpointed = SqliteWalCheckpoint.checkpoint + assert checkpointed.to_i > 0, "expected PASSIVE checkpoint to copy WAL pages, got #{checkpointed.inspect}" + ensure + SqliteWalCheckpoint.send(:release_lock) + end + end + end + + test "a contender thread takes over checkpointing after the lock holder stops" do + db_path = build_wal_database(rows: 30) + SqliteWalCheckpoint.database_path_override = db_path + lock = SqliteWalCheckpoint.lock_path + ready = File.join(@tmpdir, "holder-ready") + + holder = spawn_lock_holder(lock, ready_path: ready, hold_for: 0.25) + wait_until { File.exist?(ready) } + + contender = SqliteWalCheckpoint.start(interval: 0.05, enabled: true) + + begin + assert_equal :busy, SqliteWalCheckpoint.tick + Process.wait(holder) + holder = nil + + wait_until(timeout: 2) { lock_held_by_other_process?(lock) } + ensure + Process.wait(holder) if holder + SqliteWalCheckpoint.stop + assert_not contender.alive? + end + end + + private + def checkpoint_threads + Thread.list.select { |thread| thread.name == "sqlite-wal-checkpoint" } + end + + def build_wal_database(rows:) + path = File.join(@tmpdir, "writer-#{SecureRandom.hex(4)}.sqlite3") + + SQLite3::Database.new(path) do |database| + database.execute("PRAGMA journal_mode=WAL") + database.execute("PRAGMA wal_autocheckpoint=0") + database.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, body TEXT)") + rows.times do |i| + database.execute("INSERT INTO items (body) VALUES (?)", "row-#{i}-#{"x" * 200}") + end + end + + path + end + + def spawn_lock_holder(lock, ready_path:, hold_for:) + Process.spawn( + RbConfig.ruby, "-e", <<~RUBY + require "fileutils" + lock = #{lock.inspect} + ready = #{ready_path.inspect} + FileUtils.mkdir_p(File.dirname(lock)) + file = File.open(lock, File::RDWR | File::CREAT, 0644) + abort "lock failed" unless file.flock(File::LOCK_EX | File::LOCK_NB) + File.write(ready, "1") + sleep #{hold_for} + RUBY + ) + end + + def lock_held_by_other_process?(lock) + return false unless File.exist?(lock) + + probe = File.open(lock, File::RDWR | File::CREAT, 0644) + if probe.flock(File::LOCK_EX | File::LOCK_NB) + probe.flock(File::LOCK_UN) + probe.close + false + else + probe.close + true + end + end + + def wait_until(timeout: 1) + deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout + until yield + raise "condition not met within #{timeout}s" if Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline + sleep 0.02 + end + end +end From 61631ea5f9bccb5625bba1a7bf6edfd8a6be7a86 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 19:00:40 +0000 Subject: [PATCH 2/4] Stop WAL checkpointer before fork; one thread owns the flock Unify startup: initializer starts every non-test process, Puma and Resque stop before fork and start again in the child. Only the contender thread releases the lock. Drop Puma::CLI / single_puma_process? special cases. Co-authored-by: Thomas Klemm --- config/database.yml | 5 ++- config/initializers/sqlite_wal_checkpoint.rb | 11 ++--- config/puma.rb | 18 +++----- config/puma_dev.rb | 2 - lib/sqlite_wal_checkpoint.rb | 47 ++++---------------- lib/tasks/resque.rake | 3 ++ test/lib/sqlite_wal_checkpoint_test.rb | 9 ---- 7 files changed, 24 insertions(+), 71 deletions(-) diff --git a/config/database.yml b/config/database.yml index 2f8472a..845eb70 100644 --- a/config/database.yml +++ b/config/database.yml @@ -9,8 +9,9 @@ default: &default pool: <%= ENV.fetch("RAILS_MAX_THREADS") { 10 } %> timeout: 5000 default_transaction_mode: immediate - # Checkpoint on a background connection instead (SqliteWalCheckpoint). Every - # non-test writer process runs a contender so this is never left without one. + # Checkpoint on a background connection instead (SqliteWalCheckpoint). + # Non-test processes start a contender; forking servers stop before fork and + # start again in the child so the flock is never inherited. pragmas: wal_autocheckpoint: 0 diff --git a/config/initializers/sqlite_wal_checkpoint.rb b/config/initializers/sqlite_wal_checkpoint.rb index ccf901a..3e72802 100644 --- a/config/initializers/sqlite_wal_checkpoint.rb +++ b/config/initializers/sqlite_wal_checkpoint.rb @@ -1,10 +1,5 @@ -# Start a checkpoint contender in console, runner, rake and other non-Puma writers. -# Puma skips this path: config/puma.rb and config/puma_dev.rb start after the -# correct process is chosen (worker boot vs single-process), so the master that -# only forks workers never holds the lock alone. Resque starts after_prefork. +# Start a checkpoint contender in every non-test process. Puma and Resque pool +# stop before fork and start again in the child so the flock is never inherited. Rails.application.config.after_initialize do - next if Rails.env.test? - next if defined?(Puma::CLI) - - SqliteWalCheckpoint.start + SqliteWalCheckpoint.start unless Rails.env.test? end diff --git a/config/puma.rb b/config/puma.rb index 6e1c7fc..e626470 100644 --- a/config/puma.rb +++ b/config/puma.rb @@ -33,10 +33,7 @@ pidfile ENV.fetch("PIDFILE") { "tmp/pids/server.pid" } # processes). # worker_count = (Concurrent.processor_count * 0.666).ceil -# Keep the raw setting: WEB_CONCURRENCY=auto must not be coerced with to_i (that -# is 0 and would look like single-process mode while Puma still forks workers). -configured_workers = ENV.fetch("WEB_CONCURRENCY") { worker_count } -workers configured_workers +workers ENV.fetch("WEB_CONCURRENCY") { worker_count } ENV["JOB_CONCURRENCY"] ||= worker_count.to_s @@ -53,14 +50,11 @@ plugin :tmp_restart # Reset all membership connections Membership.disconnect_all -# Only the literal 0 is single-process. "auto" and positive counts fork workers; -# those must start the checkpointer after boot so a surviving worker can take -# the lock if the previous leader exits. -if SqliteWalCheckpoint.single_puma_process?(configured_workers) - SqliteWalCheckpoint.start -else - on_worker_boot { SqliteWalCheckpoint.start } -end +# Initializer starts a contender in this process. Stop before fork so workers +# do not inherit the flock; each worker starts its own contender. Single-process +# mode never forks, so the initializer's contender keeps running. +before_fork { SqliteWalCheckpoint.stop } +on_worker_boot { SqliteWalCheckpoint.start } Signal.trap :SIGPROF do Thread.list.each do |t| diff --git a/config/puma_dev.rb b/config/puma_dev.rb index 2c7e512..371759a 100644 --- a/config/puma_dev.rb +++ b/config/puma_dev.rb @@ -1,7 +1,5 @@ require File.expand_path("../config/environment", File.dirname(__FILE__)) -SqliteWalCheckpoint.start - Signal.trap :SIGPROF do Thread.list.each do |t| puts t diff --git a/lib/sqlite_wal_checkpoint.rb b/lib/sqlite_wal_checkpoint.rb index 1060b3c..4fc3e94 100644 --- a/lib/sqlite_wal_checkpoint.rb +++ b/lib/sqlite_wal_checkpoint.rb @@ -2,9 +2,9 @@ # writer and fsyncs during the request. database.yml sets wal_autocheckpoint=0; # this module copies pages off the request thread with PASSIVE checkpoints. # -# Every non-test writer process starts a contender thread (initializer, Puma, -# Resque). A file lock elects one leader; if that process exits, another takes -# over on the next interval so auto-checkpoint stays off safely. +# Every non-test process starts a contender. Callers that fork (Puma, Resque pool) +# must stop before fork and start again in the child so the flock is never shared +# across an inherited file descriptor. module SqliteWalCheckpoint INTERVAL = 0.25 MAX_BACKOFF = 30.0 @@ -21,7 +21,6 @@ module SqliteWalCheckpoint return if @thread&.alive? @stop = false - install_exit_handler @thread = Thread.new { run(interval) } @thread.report_on_exception = false end @@ -29,13 +28,13 @@ module SqliteWalCheckpoint @thread end + # Signal the contender to exit and wait for it. Only the contender thread + # acquires or releases the flock (see run's ensure). def stop @stop = true thread = @mutex&.synchronize { @thread } thread&.join(2) - ensure @mutex&.synchronize { @thread = nil if @thread && !@thread.alive? } - release_lock end def checkpoint @@ -64,13 +63,6 @@ module SqliteWalCheckpoint @lock_path = nil @database_path_override = nil @stop = false - @exit_handler_installed = false - end - - # Puma's WEB_CONCURRENCY=auto must not use to_i (that is 0). Only the literal - # 0 means single-process mode; everything else forks workers. - def single_puma_process?(configured_workers) - configured_workers.to_s == "0" end private @@ -81,12 +73,12 @@ module SqliteWalCheckpoint until @stop begin unless database_path - interruptible_sleep interval + sleep interval next end unless acquire_lock - interruptible_sleep interval + sleep interval next end @@ -95,7 +87,7 @@ module SqliteWalCheckpoint until @stop checkpoint_on(database) backoff = interval - interruptible_sleep interval + sleep interval end end ensure @@ -103,7 +95,7 @@ module SqliteWalCheckpoint end rescue => error Rails.logger.warn "SQLite WAL checkpoint failed: #{error.class}: #{error.message}" - interruptible_sleep backoff + sleep backoff backoff = [ backoff * 2, MAX_BACKOFF ].min end end @@ -153,29 +145,8 @@ module SqliteWalCheckpoint @lock_file.flock(File::LOCK_UN) @lock_file.close - rescue Errno::EBADF, IOError - # Already closed by a racing ensure / exit handler. ensure @lock_file = nil end - - def interruptible_sleep(seconds) - deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds - while !@stop && (remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)) > 0 - sleep [ remaining, 0.05 ].min - end - end - - def install_exit_handler - return if @exit_handler_installed - - @exit_handler_installed = true - at_exit do - if @lock_file - checkpoint rescue nil - end - stop - end - end end end diff --git a/lib/tasks/resque.rake b/lib/tasks/resque.rake index 5b8bf1a..6e04c74 100644 --- a/lib/tasks/resque.rake +++ b/lib/tasks/resque.rake @@ -4,6 +4,9 @@ end task "resque:pool:setup" do ActiveRecord::Base.connection.disconnect! + # Initializer may have started a contender in the pool master; drop it before + # workers fork so they do not inherit the flock. + SqliteWalCheckpoint.stop Resque::Pool.after_prefork do |job| ActiveRecord::Base.establish_connection diff --git a/test/lib/sqlite_wal_checkpoint_test.rb b/test/lib/sqlite_wal_checkpoint_test.rb index e38a82c..2e973a3 100644 --- a/test/lib/sqlite_wal_checkpoint_test.rb +++ b/test/lib/sqlite_wal_checkpoint_test.rb @@ -35,15 +35,6 @@ class SqliteWalCheckpointTest < ActiveSupport::TestCase assert_empty checkpoint_threads end - test "single_puma_process? treats only literal zero as single-process" do - assert SqliteWalCheckpoint.single_puma_process?(0) - assert SqliteWalCheckpoint.single_puma_process?("0") - assert_not SqliteWalCheckpoint.single_puma_process?("auto") - assert_not SqliteWalCheckpoint.single_puma_process?(2) - assert_not SqliteWalCheckpoint.single_puma_process?("2") - assert_not SqliteWalCheckpoint.single_puma_process?(8) - end - test "tick checkpoints through the elected lock holder" do db_path = build_wal_database(rows: 50) SqliteWalCheckpoint.database_path_override = db_path From 8704baa8603516785645b14f2b3e2072ebfc4e28 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 19:05:44 +0000 Subject: [PATCH 3/4] Load WAL checkpointer outside Zeitwerk reload; harden fork stop Keep the module under lib/rails_ext, use before_worker_boot, shorten the SQLite busy timeout under stop's join, and PASSIVE-checkpoint on exit for short-lived writers. Co-authored-by: Thomas Klemm --- config/initializers/sqlite_wal_checkpoint.rb | 7 +++++-- config/puma.rb | 2 +- lib/{ => rails_ext}/sqlite_wal_checkpoint.rb | 15 ++++++++++++++- test/lib/sqlite_wal_checkpoint_test.rb | 18 ++++++++++++++++++ 4 files changed, 38 insertions(+), 4 deletions(-) rename lib/{ => rails_ext}/sqlite_wal_checkpoint.rb (86%) diff --git a/config/initializers/sqlite_wal_checkpoint.rb b/config/initializers/sqlite_wal_checkpoint.rb index 3e72802..aa14b80 100644 --- a/config/initializers/sqlite_wal_checkpoint.rb +++ b/config/initializers/sqlite_wal_checkpoint.rb @@ -1,5 +1,8 @@ -# Start a checkpoint contender in every non-test process. Puma and Resque pool -# stop before fork and start again in the child so the flock is never inherited. +# Loaded from lib/rails_ext (autoload_lib ignore list) so reloads do not orphan +# the contender thread. Non-test processes start here; Puma/Resque stop before +# fork and start again in the child. +require Rails.root.join("lib/rails_ext/sqlite_wal_checkpoint") + Rails.application.config.after_initialize do SqliteWalCheckpoint.start unless Rails.env.test? end diff --git a/config/puma.rb b/config/puma.rb index e626470..06220b8 100644 --- a/config/puma.rb +++ b/config/puma.rb @@ -54,7 +54,7 @@ Membership.disconnect_all # do not inherit the flock; each worker starts its own contender. Single-process # mode never forks, so the initializer's contender keeps running. before_fork { SqliteWalCheckpoint.stop } -on_worker_boot { SqliteWalCheckpoint.start } +before_worker_boot { SqliteWalCheckpoint.start } Signal.trap :SIGPROF do Thread.list.each do |t| diff --git a/lib/sqlite_wal_checkpoint.rb b/lib/rails_ext/sqlite_wal_checkpoint.rb similarity index 86% rename from lib/sqlite_wal_checkpoint.rb rename to lib/rails_ext/sqlite_wal_checkpoint.rb index 4fc3e94..75103bc 100644 --- a/lib/sqlite_wal_checkpoint.rb +++ b/lib/rails_ext/sqlite_wal_checkpoint.rb @@ -21,6 +21,7 @@ module SqliteWalCheckpoint return if @thread&.alive? @stop = false + install_exit_checkpoint @thread = Thread.new { run(interval) } @thread.report_on_exception = false end @@ -33,6 +34,7 @@ module SqliteWalCheckpoint def stop @stop = true thread = @mutex&.synchronize { @thread } + # Join longer than busy_handler_timeout so before_fork does not race a PRAGMA. thread&.join(2) @mutex&.synchronize { @thread = nil if @thread && !@thread.alive? } end @@ -63,6 +65,7 @@ module SqliteWalCheckpoint @lock_path = nil @database_path_override = nil @stop = false + @exit_checkpoint_installed = false end private @@ -113,7 +116,8 @@ module SqliteWalCheckpoint result = nil SQLite3::Database.new(path) do |database| - database.busy_handler_timeout = 5_000 + # Keep below stop's join timeout so before_fork can finish cleanly. + database.busy_handler_timeout = 1_000 result = yield database end result @@ -148,5 +152,14 @@ module SqliteWalCheckpoint ensure @lock_file = nil end + + # Best-effort PASSIVE for short-lived console/rake writers that exit before + # the contender acquires the flock. Does not touch lock ownership. + def install_exit_checkpoint + return if @exit_checkpoint_installed + + @exit_checkpoint_installed = true + at_exit { checkpoint rescue nil } + end end end diff --git a/test/lib/sqlite_wal_checkpoint_test.rb b/test/lib/sqlite_wal_checkpoint_test.rb index 2e973a3..508fddd 100644 --- a/test/lib/sqlite_wal_checkpoint_test.rb +++ b/test/lib/sqlite_wal_checkpoint_test.rb @@ -35,6 +35,24 @@ class SqliteWalCheckpointTest < ActiveSupport::TestCase assert_empty checkpoint_threads end + test "stop before start again mimics a fork-safe worker boot" do + db_path = build_wal_database(rows: 10) + SqliteWalCheckpoint.database_path_override = db_path + + first = SqliteWalCheckpoint.start(interval: 0.05, enabled: true) + wait_until { first.name == "sqlite-wal-checkpoint" } + SqliteWalCheckpoint.stop + assert_not first.alive? + + second = SqliteWalCheckpoint.start(interval: 0.05, enabled: true) + wait_until { second.name == "sqlite-wal-checkpoint" } + assert second.alive? + + SqliteWalCheckpoint.stop + assert_not second.alive? + assert_equal :checkpointed, SqliteWalCheckpoint.tick + end + test "tick checkpoints through the elected lock holder" do db_path = build_wal_database(rows: 50) SqliteWalCheckpoint.database_path_override = db_path From 85ad590632151d7e44df91ff2a99b899b3292b96 Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 19:23:20 +0000 Subject: [PATCH 4/4] Wait for checkpoint thread exit; avoid lock spin without a DB stop joins until the contender finishes so before_fork never inherits an open flock or SQLite connection. Skip acquire when the database file is missing, and sleep if with_database no-ops after a race. Co-authored-by: Thomas Klemm --- lib/rails_ext/sqlite_wal_checkpoint.rb | 20 ++++++++++++++------ 1 file changed, 14 insertions(+), 6 deletions(-) diff --git a/lib/rails_ext/sqlite_wal_checkpoint.rb b/lib/rails_ext/sqlite_wal_checkpoint.rb index 75103bc..1ccda79 100644 --- a/lib/rails_ext/sqlite_wal_checkpoint.rb +++ b/lib/rails_ext/sqlite_wal_checkpoint.rb @@ -29,14 +29,15 @@ module SqliteWalCheckpoint @thread end - # Signal the contender to exit and wait for it. Only the contender thread - # acquires or releases the flock (see run's ensure). + # Signal the contender to exit and wait until it has released the flock and + # closed its SQLite connection. before_fork must not return while those are open. def stop @stop = true thread = @mutex&.synchronize { @thread } - # Join longer than busy_handler_timeout so before_fork does not race a PRAGMA. - thread&.join(2) - @mutex&.synchronize { @thread = nil if @thread && !@thread.alive? } + return unless thread + + thread.join + @mutex&.synchronize { @thread = nil if @thread.equal?(thread) } end def checkpoint @@ -75,7 +76,8 @@ module SqliteWalCheckpoint until @stop begin - unless database_path + path = database_path + unless path && File.exist?(path) sleep interval next end @@ -86,7 +88,9 @@ module SqliteWalCheckpoint end begin + ran = false with_database do |database| + ran = true until @stop checkpoint_on(database) backoff = interval @@ -96,6 +100,10 @@ module SqliteWalCheckpoint ensure release_lock end + + # with_database no-ops if the file vanished between the exist? check + # and open; sleep so we do not spin on the lock file. + sleep interval unless ran || @stop rescue => error Rails.logger.warn "SQLite WAL checkpoint failed: #{error.class}: #{error.message}" sleep backoff