From 88acc4e026025dcfe40acb22fbabd5507e88529e Mon Sep 17 00:00:00 2001 From: Cursor Agent Date: Wed, 7 Oct 2026 18:53:19 +0000 Subject: [PATCH] Checkpoint SQLite WAL off the request thread safely Disable wal_autocheckpoint and run PASSIVE checkpoints from one elected writer process. Contenders start in console/rake (initializer), Puma workers (including WEB_CONCURRENCY=auto), and Resque children, with failover, capped error backoff, and tests for lock takeover. Co-authored-by: Thomas Klemm --- config/database.yml | 4 + config/initializers/sqlite_wal_checkpoint.rb | 10 + config/puma.rb | 14 +- config/puma_dev.rb | 2 + lib/sqlite_wal_checkpoint.rb | 181 +++++++++++++++++++ lib/tasks/resque.rake | 1 + test/lib/sqlite_wal_checkpoint_test.rb | 178 ++++++++++++++++++ 7 files changed, 389 insertions(+), 1 deletion(-) create mode 100644 config/initializers/sqlite_wal_checkpoint.rb create mode 100644 lib/sqlite_wal_checkpoint.rb create mode 100644 test/lib/sqlite_wal_checkpoint_test.rb diff --git a/config/database.yml b/config/database.yml index ec220a4..2f8472a 100644 --- a/config/database.yml +++ b/config/database.yml @@ -9,6 +9,10 @@ default: &default pool: <%= ENV.fetch("RAILS_MAX_THREADS") { 10 } %> timeout: 5000 default_transaction_mode: immediate + # Checkpoint on a background connection instead (SqliteWalCheckpoint). Every + # non-test writer process runs a contender so this is never left without one. + pragmas: + wal_autocheckpoint: 0 development: primary: diff --git a/config/initializers/sqlite_wal_checkpoint.rb b/config/initializers/sqlite_wal_checkpoint.rb new file mode 100644 index 0000000..ccf901a --- /dev/null +++ b/config/initializers/sqlite_wal_checkpoint.rb @@ -0,0 +1,10 @@ +# Start a checkpoint contender in console, runner, rake and other non-Puma writers. +# Puma skips this path: config/puma.rb and config/puma_dev.rb start after the +# correct process is chosen (worker boot vs single-process), so the master that +# only forks workers never holds the lock alone. Resque starts after_prefork. +Rails.application.config.after_initialize do + next if Rails.env.test? + next if defined?(Puma::CLI) + + SqliteWalCheckpoint.start +end diff --git a/config/puma.rb b/config/puma.rb index 960ab6a..6e1c7fc 100644 --- a/config/puma.rb +++ b/config/puma.rb @@ -33,7 +33,10 @@ pidfile ENV.fetch("PIDFILE") { "tmp/pids/server.pid" } # processes). # worker_count = (Concurrent.processor_count * 0.666).ceil -workers ENV.fetch("WEB_CONCURRENCY") { worker_count } +# Keep the raw setting: WEB_CONCURRENCY=auto must not be coerced with to_i (that +# is 0 and would look like single-process mode while Puma still forks workers). +configured_workers = ENV.fetch("WEB_CONCURRENCY") { worker_count } +workers configured_workers ENV["JOB_CONCURRENCY"] ||= worker_count.to_s @@ -50,6 +53,15 @@ plugin :tmp_restart # Reset all membership connections Membership.disconnect_all +# Only the literal 0 is single-process. "auto" and positive counts fork workers; +# those must start the checkpointer after boot so a surviving worker can take +# the lock if the previous leader exits. +if SqliteWalCheckpoint.single_puma_process?(configured_workers) + SqliteWalCheckpoint.start +else + on_worker_boot { SqliteWalCheckpoint.start } +end + Signal.trap :SIGPROF do Thread.list.each do |t| puts t diff --git a/config/puma_dev.rb b/config/puma_dev.rb index 371759a..2c7e512 100644 --- a/config/puma_dev.rb +++ b/config/puma_dev.rb @@ -1,5 +1,7 @@ require File.expand_path("../config/environment", File.dirname(__FILE__)) +SqliteWalCheckpoint.start + Signal.trap :SIGPROF do Thread.list.each do |t| puts t diff --git a/lib/sqlite_wal_checkpoint.rb b/lib/sqlite_wal_checkpoint.rb new file mode 100644 index 0000000..1060b3c --- /dev/null +++ b/lib/sqlite_wal_checkpoint.rb @@ -0,0 +1,181 @@ +# SQLite's default WAL auto-checkpoint (~1,000 pages) runs inside the committing +# writer and fsyncs during the request. database.yml sets wal_autocheckpoint=0; +# this module copies pages off the request thread with PASSIVE checkpoints. +# +# Every non-test writer process starts a contender thread (initializer, Puma, +# Resque). A file lock elects one leader; if that process exits, another takes +# over on the next interval so auto-checkpoint stays off safely. +module SqliteWalCheckpoint + INTERVAL = 0.25 + MAX_BACKOFF = 30.0 + LOCK_PATH = Rails.root.join("tmp/pids/sqlite_wal_checkpoint.lock") + + class << self + attr_writer :lock_path, :database_path_override + + def start(interval: INTERVAL, enabled: !Rails.env.test?) + return unless enabled + + @mutex ||= Mutex.new + @mutex.synchronize do + return if @thread&.alive? + + @stop = false + install_exit_handler + @thread = Thread.new { run(interval) } + @thread.report_on_exception = false + end + + @thread + end + + def stop + @stop = true + thread = @mutex&.synchronize { @thread } + thread&.join(2) + ensure + @mutex&.synchronize { @thread = nil if @thread && !@thread.alive? } + release_lock + end + + def checkpoint + with_database { |database| checkpoint_on(database) } + end + + # One leadership attempt for tests: acquire the lock, checkpoint once, release. + def tick + return :no_database unless database_path + return :busy unless acquire_lock + + begin + result = checkpoint + result.nil? ? :no_database : :checkpointed + ensure + release_lock + end + end + + def lock_path + @lock_path || LOCK_PATH + end + + def reset! + stop + @lock_path = nil + @database_path_override = nil + @stop = false + @exit_handler_installed = false + end + + # Puma's WEB_CONCURRENCY=auto must not use to_i (that is 0). Only the literal + # 0 means single-process mode; everything else forks workers. + def single_puma_process?(configured_workers) + configured_workers.to_s == "0" + end + + private + def run(interval) + Thread.current.name = "sqlite-wal-checkpoint" + backoff = interval + + until @stop + begin + unless database_path + interruptible_sleep interval + next + end + + unless acquire_lock + interruptible_sleep interval + next + end + + begin + with_database do |database| + until @stop + checkpoint_on(database) + backoff = interval + interruptible_sleep interval + end + end + ensure + release_lock + end + rescue => error + Rails.logger.warn "SQLite WAL checkpoint failed: #{error.class}: #{error.message}" + interruptible_sleep backoff + backoff = [ backoff * 2, MAX_BACKOFF ].min + end + end + ensure + release_lock + end + + def checkpoint_on(database) + database.execute("PRAGMA wal_checkpoint(PASSIVE)").first + end + + def with_database + path = database_path + return unless path && File.exist?(path) + + result = nil + SQLite3::Database.new(path) do |database| + database.busy_handler_timeout = 5_000 + result = yield database + end + result + end + + def database_path + return @database_path_override if @database_path_override + + config = ActiveRecord::Base.connection_db_config + return unless config.adapter.to_s == "sqlite3" + + ActiveRecord::ConnectionAdapters::SQLite3Adapter.resolve_path(config.database) + end + + def acquire_lock + FileUtils.mkdir_p(File.dirname(lock_path)) + file = File.open(lock_path, File::RDWR | File::CREAT, 0644) + if file.flock(File::LOCK_EX | File::LOCK_NB) + @lock_file = file + true + else + file.close + false + end + end + + def release_lock + return unless @lock_file + + @lock_file.flock(File::LOCK_UN) + @lock_file.close + rescue Errno::EBADF, IOError + # Already closed by a racing ensure / exit handler. + ensure + @lock_file = nil + end + + def interruptible_sleep(seconds) + deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + seconds + while !@stop && (remaining = deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC)) > 0 + sleep [ remaining, 0.05 ].min + end + end + + def install_exit_handler + return if @exit_handler_installed + + @exit_handler_installed = true + at_exit do + if @lock_file + checkpoint rescue nil + end + stop + end + end + end +end diff --git a/lib/tasks/resque.rake b/lib/tasks/resque.rake index 63ecc9d..5b8bf1a 100644 --- a/lib/tasks/resque.rake +++ b/lib/tasks/resque.rake @@ -8,5 +8,6 @@ task "resque:pool:setup" do Resque::Pool.after_prefork do |job| ActiveRecord::Base.establish_connection Resque.redis.client.close + SqliteWalCheckpoint.start end end diff --git a/test/lib/sqlite_wal_checkpoint_test.rb b/test/lib/sqlite_wal_checkpoint_test.rb new file mode 100644 index 0000000..e38a82c --- /dev/null +++ b/test/lib/sqlite_wal_checkpoint_test.rb @@ -0,0 +1,178 @@ +require "test_helper" + +class SqliteWalCheckpointTest < ActiveSupport::TestCase + setup do + @tmpdir = Dir.mktmpdir("sqlite-wal-checkpoint") + SqliteWalCheckpoint.reset! + SqliteWalCheckpoint.lock_path = File.join(@tmpdir, "checkpoint.lock") + end + + teardown do + SqliteWalCheckpoint.reset! + FileUtils.remove_entry(@tmpdir) if @tmpdir && File.directory?(@tmpdir) + end + + test "connections disable WAL auto-checkpoint so commits do not fsync on the writer" do + assert_equal 0, ActiveRecord::Base.connection.select_value("PRAGMA wal_autocheckpoint").to_i + end + + test "a passive checkpoint against the primary database does not raise" do + assert_nothing_raised { SqliteWalCheckpoint.checkpoint } + end + + test "does not start a background thread in test by default" do + assert_nil SqliteWalCheckpoint.start + assert_empty checkpoint_threads + end + + test "start with enabled: true runs a named contender thread that stop joins" do + thread = SqliteWalCheckpoint.start(interval: 0.05, enabled: true) + assert thread.alive? + wait_until { thread.name == "sqlite-wal-checkpoint" } + + SqliteWalCheckpoint.stop + assert_not thread.alive? + assert_empty checkpoint_threads + end + + test "single_puma_process? treats only literal zero as single-process" do + assert SqliteWalCheckpoint.single_puma_process?(0) + assert SqliteWalCheckpoint.single_puma_process?("0") + assert_not SqliteWalCheckpoint.single_puma_process?("auto") + assert_not SqliteWalCheckpoint.single_puma_process?(2) + assert_not SqliteWalCheckpoint.single_puma_process?("2") + assert_not SqliteWalCheckpoint.single_puma_process?(8) + end + + test "tick checkpoints through the elected lock holder" do + db_path = build_wal_database(rows: 50) + SqliteWalCheckpoint.database_path_override = db_path + + assert_equal :checkpointed, SqliteWalCheckpoint.tick + end + + test "tick reports busy while another process holds the lock and succeeds after it exits" do + db_path = build_wal_database(rows: 20) + SqliteWalCheckpoint.database_path_override = db_path + lock = SqliteWalCheckpoint.lock_path + ready = File.join(@tmpdir, "holder-ready") + + holder = spawn_lock_holder(lock, ready_path: ready, hold_for: 0.4) + + begin + wait_until { File.exist?(ready) } + assert_equal :busy, SqliteWalCheckpoint.tick + ensure + Process.wait(holder) + end + + assert_equal :checkpointed, SqliteWalCheckpoint.tick + end + + test "passive checkpoint moves WAL pages after writes with autocheckpoint disabled" do + db_path = File.join(@tmpdir, "writer-#{SecureRandom.hex(4)}.sqlite3") + + SQLite3::Database.new(db_path) do |writer| + writer.execute("PRAGMA journal_mode=WAL") + writer.execute("PRAGMA wal_autocheckpoint=0") + writer.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, body TEXT)") + 200.times do |i| + writer.execute("INSERT INTO items (body) VALUES (?)", "row-#{i}-#{"x" * 200}") + end + + wal_path = "#{db_path}-wal" + assert File.exist?(wal_path), "expected a WAL file after inserts" + assert File.size(wal_path) > 0 + + SqliteWalCheckpoint.database_path_override = db_path + assert SqliteWalCheckpoint.send(:acquire_lock) + begin + _busy, _log, checkpointed = SqliteWalCheckpoint.checkpoint + assert checkpointed.to_i > 0, "expected PASSIVE checkpoint to copy WAL pages, got #{checkpointed.inspect}" + ensure + SqliteWalCheckpoint.send(:release_lock) + end + end + end + + test "a contender thread takes over checkpointing after the lock holder stops" do + db_path = build_wal_database(rows: 30) + SqliteWalCheckpoint.database_path_override = db_path + lock = SqliteWalCheckpoint.lock_path + ready = File.join(@tmpdir, "holder-ready") + + holder = spawn_lock_holder(lock, ready_path: ready, hold_for: 0.25) + wait_until { File.exist?(ready) } + + contender = SqliteWalCheckpoint.start(interval: 0.05, enabled: true) + + begin + assert_equal :busy, SqliteWalCheckpoint.tick + Process.wait(holder) + holder = nil + + wait_until(timeout: 2) { lock_held_by_other_process?(lock) } + ensure + Process.wait(holder) if holder + SqliteWalCheckpoint.stop + assert_not contender.alive? + end + end + + private + def checkpoint_threads + Thread.list.select { |thread| thread.name == "sqlite-wal-checkpoint" } + end + + def build_wal_database(rows:) + path = File.join(@tmpdir, "writer-#{SecureRandom.hex(4)}.sqlite3") + + SQLite3::Database.new(path) do |database| + database.execute("PRAGMA journal_mode=WAL") + database.execute("PRAGMA wal_autocheckpoint=0") + database.execute("CREATE TABLE items (id INTEGER PRIMARY KEY, body TEXT)") + rows.times do |i| + database.execute("INSERT INTO items (body) VALUES (?)", "row-#{i}-#{"x" * 200}") + end + end + + path + end + + def spawn_lock_holder(lock, ready_path:, hold_for:) + Process.spawn( + RbConfig.ruby, "-e", <<~RUBY + require "fileutils" + lock = #{lock.inspect} + ready = #{ready_path.inspect} + FileUtils.mkdir_p(File.dirname(lock)) + file = File.open(lock, File::RDWR | File::CREAT, 0644) + abort "lock failed" unless file.flock(File::LOCK_EX | File::LOCK_NB) + File.write(ready, "1") + sleep #{hold_for} + RUBY + ) + end + + def lock_held_by_other_process?(lock) + return false unless File.exist?(lock) + + probe = File.open(lock, File::RDWR | File::CREAT, 0644) + if probe.flock(File::LOCK_EX | File::LOCK_NB) + probe.flock(File::LOCK_UN) + probe.close + false + else + probe.close + true + end + end + + def wait_until(timeout: 1) + deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout + until yield + raise "condition not met within #{timeout}s" if Process.clock_gettime(Process::CLOCK_MONOTONIC) > deadline + sleep 0.02 + end + end +end