mirror of
https://github.com/basecamp/once-campfire.git
synced 2026-10-09 00:00:12 +09:00
a196a93ed2
Opengraph::Fetch runs inside POST /unfurl_link, against a host the member picked, with Net::HTTP's defaults: 60 s to connect, 60 s for each read, and one retry of the GET. A per-read timeout doesn't bound a host that sends a byte at a time, in its headers or its body, so nothing limited how long a fetch could hold the request thread. Give each operation 7 s, as Webhook does, turn off the retry, and put the whole fetch, redirects included, under one 10 s deadline. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01Bj8KnxpTf9sj2Ysa8aLAVa
92 lines
3.1 KiB
Ruby
92 lines
3.1 KiB
Ruby
require "net/http"
|
|
require "restricted_http/private_network_guard"
|
|
|
|
class Opengraph::Fetch
|
|
ALLOWED_DOCUMENT_CONTENT_TYPE = "text/html"
|
|
MAX_BODY_SIZE = 5.megabytes
|
|
MAX_REDIRECTS = 10
|
|
TIMEOUT = 7.seconds # to connect (TLS included), and for each read and write, as Webhook does
|
|
DEADLINE = 10.seconds # for the whole fetch: redirects, headers and body
|
|
|
|
class TooManyRedirectsError < StandardError; end
|
|
class RedirectDeniedError < StandardError; end
|
|
|
|
def fetch_document(url, ip: RestrictedHTTP::PrivateNetworkGuard.resolve(url.host))
|
|
request(url, Net::HTTP::Get, ip: ip) do |response|
|
|
return body_if_acceptable(response)
|
|
end
|
|
end
|
|
|
|
def fetch_content_type(url, ip: RestrictedHTTP::PrivateNetworkGuard.resolve(url.host))
|
|
request(url, Net::HTTP::Head, ip: ip) do |response|
|
|
return response["Content-Type"]
|
|
end
|
|
end
|
|
|
|
private
|
|
# A member triggers this from a web request, and the host on the other end decides how fast it answers.
|
|
# The timeouts bound each operation, but Net::HTTP reads headers and body in as many reads as the host
|
|
# likes, so only a deadline over the whole fetch stops one that sends a byte at a time.
|
|
def request(url, request_class, ip:)
|
|
Timeout.timeout(DEADLINE) do
|
|
MAX_REDIRECTS.times do
|
|
Net::HTTP.start(url.host, url.port, ipaddr: ip, use_ssl: url.scheme == "https", **timeouts) do |http|
|
|
http.request request_class.new(url) do |response|
|
|
if response.is_a?(Net::HTTPRedirection)
|
|
url, ip = resolve_redirect(response["location"])
|
|
else
|
|
yield response
|
|
end
|
|
end
|
|
end
|
|
end
|
|
|
|
raise TooManyRedirectsError
|
|
end
|
|
end
|
|
|
|
# Without max_retries: 0, Net::HTTP sends a GET or HEAD that timed out a second time.
|
|
def timeouts
|
|
{ open_timeout: TIMEOUT, read_timeout: TIMEOUT, write_timeout: TIMEOUT, max_retries: 0 }
|
|
end
|
|
|
|
def resolve_redirect(location)
|
|
url = URI.parse(location)
|
|
raise RedirectDeniedError unless url.is_a?(URI::HTTP)
|
|
[ url, RestrictedHTTP::PrivateNetworkGuard.resolve(url.host) ]
|
|
end
|
|
|
|
def body_if_acceptable(response)
|
|
size_restricted_body(response) if response_valid?(response)
|
|
end
|
|
|
|
def size_restricted_body(response)
|
|
# We've already checked the Content-Length header, to try to avoid reading
|
|
# the body of any large responses. But that header could be wrong or
|
|
# missing. To be on the safe side, we'll read the body in chunks, and bail
|
|
# if it runs over our size limit.
|
|
StringIO.new.tap do |body|
|
|
response.read_body do |chunk|
|
|
return nil if body.string.bytesize + chunk.bytesize > MAX_BODY_SIZE
|
|
body << chunk
|
|
end
|
|
end.string
|
|
end
|
|
|
|
def response_valid?(response)
|
|
status_valid?(response) && content_type_valid?(response) && content_length_valid?(response)
|
|
end
|
|
|
|
def status_valid?(response)
|
|
response.is_a?(Net::HTTPOK)
|
|
end
|
|
|
|
def content_type_valid?(response)
|
|
response.content_type == ALLOWED_DOCUMENT_CONTENT_TYPE
|
|
end
|
|
|
|
def content_length_valid?(response)
|
|
response.content_length.to_i <= MAX_BODY_SIZE
|
|
end
|
|
end
|