Store: per-host backoff after 403/429 honouring Retry-After

A host that answers 403/429 is left alone until its Retry-After (or GitHub's
rate-limit reset; default 10 minutes). Meanwhile cached data is served, or the
source reports 'limited' with its own name, e.g. 'GitHub is limiting requests;
try again in 10 minutes'. Covers _web reads/downloads and F-Droid fetches.

Co-Authored-By: Claude Opus 5.5 (1M context) <noreply@anthropic.com>
This commit is contained in:
saphidandClaude Opus 5.5 committed 2026-09-28 22:09:55 +10:00
1 parent c68afa6c5d
commit 7c439fdf28
7 files changed
+159 -28

No files matched your search

+40 -2
View File
@@ -18,6 +18,8 @@ class PublisherSources(unittest.TestCase):
self.network = patch('urllib.request.OpenerDirector.open', side_effect=AssertionError('network in test')) self.network = patch('urllib.request.OpenerDirector.open', side_effect=AssertionError('network in test'))
self.network.start() self.network.start()
self.addCleanup(self.network.stop) self.addCleanup(self.network.stop)
_web._limited.clear()
self.addCleanup(_web._limited.clear)
def test_curated_search_is_offline(self): def test_curated_search_is_offline(self):
self.assertEqual(github.search(github.sources()[0], 'hello')[0]['id'], 'KhronosGroup/OpenXR-SDK-Source') self.assertEqual(github.search(github.sources()[0], 'hello')[0]['id'], 'KhronosGroup/OpenXR-SDK-Source')
@@ -138,13 +140,49 @@ class PublisherSources(unittest.TestCase):
self.assertEqual(_web.read('https://itch.io/test', ('itch.io',)), b'index') self.assertEqual(_web.read('https://itch.io/test', ('itch.io',)), b'index')
self.assertEqual(op.call_count, 1) self.assertEqual(op.call_count, 1)
error = urllib.error.HTTPError('https://api.github.com/x', 403, 'limited', {}, None) error = urllib.error.HTTPError('https://api.github.com/x', 403, 'limited', {}, None)
with patch.object(_web, 'open_url', side_effect=error), self.assertRaisesRegex(SourceError, 'FRAME_GITHUB_TOKEN'): with patch.object(_web, 'open_url', side_effect=error), patch.dict(os.environ, {'FRAME_GITHUB_TOKEN': ''}), \
_web.read('https://api.github.com/x', ('api.github.com',)) self.assertRaisesRegex(SourceError, 'FRAME_GITHUB_TOKEN'):
github._api('/x')
with patch.object(_web, 'open_url', side_effect=error), patch.object(_web.time, 'time', return_value=1e12): with patch.object(_web, 'open_url', side_effect=error), patch.object(_web.time, 'time', return_value=1e12):
self.assertEqual(_web.read('https://itch.io/test', ('itch.io',)), b'index') # throttled: stale copy self.assertEqual(_web.read('https://itch.io/test', ('itch.io',)), b'index') # throttled: stale copy
with patch.object(_web, 'read', return_value=b'<html>'), self.assertRaises(SourceError): with patch.object(_web, 'read', return_value=b'<html>'), self.assertRaises(SourceError):
github._api('/x') github._api('/x')
def test_backoff_honours_retry_after_per_host(self):
from apk_sources import SourceLimited
from email.utils import formatdate
now = [1e9]
clock = patch.object(_web.time, 'time', side_effect=lambda: now[0])
clock.start()
self.addCleanup(clock.stop)
error = urllib.error.HTTPError('https://itch.io/a', 429, 'slow down', {'Retry-After': '120'}, None)
with patch.object(_web, 'open_url', side_effect=error) as op:
with self.assertRaisesRegex(SourceLimited, '^itch.io is limiting requests; try again in 2 minutes$'):
itch.search(itch.sources()[0], '')
with self.assertRaises(SourceLimited) as caught: # other URLs on the host wait too
_web.read('https://itch.io/b', ('itch.io',), name='itch.io')
self.assertEqual(op.call_count, 1)
self.assertAlmostEqual(caught.exception.retry_after, 120)
with self.assertRaises(SourceLimited):
_web.apk('https://itch.io/c.apk', ('itch.io',))
self.assertEqual(op.call_count, 1)
with patch.object(_web, 'open_url', return_value=io.BytesIO(b'fresh')):
self.assertEqual(_web.read('https://api.github.com/x', ('api.github.com',)), b'fresh') # other hosts unaffected
now[0] += 121
with patch.object(_web, 'open_url', return_value=io.BytesIO(b'feed')):
self.assertEqual(_web.read('https://itch.io/b', ('itch.io',)), b'feed')
cases = [({}, 600), ({'Retry-After': formatdate(now[0] + 300, usegmt=True)}, 300),
({'X-RateLimit-Remaining': '0', 'X-RateLimit-Reset': str(int(now[0]) + 60)}, 60)]
for headers, expected in cases:
with self.subTest(headers=headers):
_web._limited.clear()
self.assertAlmostEqual(_web.throttle('https://h.test/x', headers), expected, delta=1)
self.assertAlmostEqual(_web.wait_time('https://h.test/y'), expected, delta=1)
error = urllib.error.HTTPError('https://api.github.com/x', 403, 'limited', {}, None)
with patch.object(_web, 'open_url', side_effect=error), patch.dict(os.environ, {'FRAME_GITHUB_TOKEN': ''}), \
self.assertRaisesRegex(SourceLimited, '^GitHub is limiting requests; try again in 10 minutes .*TOKEN'):
github._api('/y')
if __name__ == '__main__': if __name__ == '__main__':
unittest.main() unittest.main()
+17 -2
View File
@@ -10,7 +10,7 @@ import urllib.error
import zipfile import zipfile
sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'ui')) sys.path.insert(0, str(Path(__file__).resolve().parents[1] / 'ui'))
from apk_sources import SourceError, fdroid from apk_sources import SourceError, SourceLimited, _web, fdroid
FIXTURES = Path(__file__).parent / 'fixtures' / 'fdroid' FIXTURES = Path(__file__).parent / 'fixtures' / 'fdroid'
PIN = (FIXTURES / 'fingerprint.txt').read_text().strip() PIN = (FIXTURES / 'fingerprint.txt').read_text().strip()
@@ -69,12 +69,14 @@ class Repositories(unittest.TestCase):
net = patch.object(fdroid.urllib.request, 'build_opener', side_effect=AssertionError('network forbidden')) net = patch.object(fdroid.urllib.request, 'build_opener', side_effect=AssertionError('network forbidden'))
net.start() net.start()
self.addCleanup(net.stop) self.addCleanup(net.stop)
mock = patch.object(fdroid, '_fetch', side_effect=self.fetch) mock = self.fetch_patch = patch.object(fdroid, '_fetch', side_effect=self.fetch)
self.fetch_mock = mock.start() self.fetch_mock = mock.start()
self.addCleanup(mock.stop) self.addCleanup(mock.stop)
self.v1 = False self.v1 = False
self.corrupt = None self.corrupt = None
self.files = {} self.files = {}
_web._limited.clear()
self.addCleanup(_web._limited.clear)
def fetch(self, url, path, maximum): def fetch(self, url, path, maximum):
name = url.rsplit('/', 1)[-1] name = url.rsplit('/', 1)[-1]
@@ -332,6 +334,19 @@ class Repositories(unittest.TestCase):
stream = io.BytesIO(self.files['entry.jar']) stream = io.BytesIO(self.files['entry.jar'])
self.assertIn(b'index', fdroid._jar(stream, 'entry.json', None)[0]) # index-v1.jar may still use SHA-1 self.assertIn(b'index', fdroid._jar(stream, 'entry.json', None)[0]) # index-v1.jar may still use SHA-1
def test_rate_limited_host_backs_off(self):
source = dict(self.add(), name='My repo')
self.fetch_patch.stop()
error = urllib.error.HTTPError(URL, 429, 'slow down', {'Retry-After': '300'}, None)
with patch.object(fdroid.urllib.request, 'build_opener') as opener:
opener.return_value.open.side_effect = error
with self.assertRaisesRegex(SourceLimited, '^My repo is limiting requests; try again in 5 minutes$'):
fdroid._load(source, force=True)
with self.assertRaisesRegex(SourceLimited, 'My repo'):
fdroid.download(source, 'org.example.app', 1)
self.assertEqual(opener.return_value.open.call_count, 1)
self.fetch_mock = self.fetch_patch.start()
def test_cached_index_does_not_cross_pins(self): def test_cached_index_does_not_cross_pins(self):
source = self.add() source = self.add()
source['fingerprint'] = '0' * 64 source['fingerprint'] = '0' * 64
+69 -10
View File
@@ -1,10 +1,50 @@
"""Small HTTPS cache and APK downloader for public publisher sources.""" """Small HTTPS cache and APK downloader for public publisher sources."""
import hashlib, os, tempfile, time, urllib.error, urllib.parse, urllib.request, zipfile import hashlib, os, tempfile, threading, time, urllib.error, urllib.parse, urllib.request, zipfile
from email.utils import parsedate_to_datetime
import frame_host import frame_host
from . import SourceError from . import SourceError, SourceLimited
UA = 'FrameControl/0.1' UA = 'FrameControl/0.1'
BACKOFF = 600 # seconds to leave a host alone after 403/429 without Retry-After
_limited = {} # host -> time.time() before which we don't contact it
_limited_lock = threading.Lock()
def _host(url):
return (urllib.parse.urlsplit(url).hostname or '').lower()
def throttle(url, headers=None):
"""Remember that url's host asked us to back off; return the delay in seconds."""
headers = headers or {}
now, delay = time.time(), None
value = (headers.get('Retry-After') or '').strip()
if value.isdigit():
delay = int(value)
elif value:
try:
delay = parsedate_to_datetime(value).timestamp() - now
except (TypeError, ValueError, OverflowError):
pass
reset = headers.get('X-RateLimit-Reset') or ''
if delay is None and headers.get('X-RateLimit-Remaining') == '0' and reset.isdigit():
delay = int(reset) - now # GitHub
delay = min(max(delay if delay is not None else BACKOFF, 1), 6 * 3600)
with _limited_lock:
_limited[_host(url)] = max(_limited.get(_host(url), 0), now + delay)
return delay
def wait_time(url):
with _limited_lock:
return max(0, _limited.get(_host(url), 0) - time.time())
def limited_error(name, seconds, hint=''):
minutes = max(1, int(round(seconds / 60)))
return SourceLimited('%s is limiting requests; try again in %d minute%s%s'
% (name, minutes, '' if minutes == 1 else 's', hint), seconds)
def cache(): def cache():
@@ -42,12 +82,24 @@ def open_url(url, hosts, headers=None):
urllib.request.Request(url, headers={'User-Agent': UA, **(headers or {})}), timeout=60) urllib.request.Request(url, headers={'User-Agent': UA, **(headers or {})}), timeout=60)
def read(url, hosts, headers=None, ttl=3600): def read(url, hosts, headers=None, ttl=3600, name=None, hint=''):
path = os.path.join(cache(), hashlib.sha256(url.encode()).hexdigest() + '.data') path = os.path.join(cache(), hashlib.sha256(url.encode()).hexdigest() + '.data')
name = name or _host(url) or 'The source'
def cached():
if os.path.isfile(path): # throttled: an older copy beats no results
with open(path, 'rb') as f:
return f.read()
try: try:
if os.path.isfile(path) and time.time() - os.path.getmtime(path) < ttl: if os.path.isfile(path) and time.time() - os.path.getmtime(path) < ttl:
with open(path, 'rb') as f: with open(path, 'rb') as f:
return f.read() return f.read()
wait = wait_time(url)
if wait:
data = cached()
if data is None:
raise limited_error(name, wait, hint)
return data
with open_url(url, hosts, headers) as r: with open_url(url, hosts, headers) as r:
data = r.read(8 * 1024 * 1024 + 1) data = r.read(8 * 1024 * 1024 + 1)
if len(data) > 8 * 1024 * 1024: if len(data) > 8 * 1024 * 1024:
@@ -63,19 +115,22 @@ def read(url, hosts, headers=None, ttl=3600):
return data return data
except urllib.error.HTTPError as e: except urllib.error.HTTPError as e:
if e.code in (403, 429): if e.code in (403, 429):
if os.path.isfile(path): # throttled: an older copy beats no results delay = throttle(url, e.headers)
with open(path, 'rb') as f: data = cached()
return f.read() if data is None:
host = urllib.parse.urlsplit(url).hostname or 'The source' raise limited_error(name, delay, hint) from e
hint = ' (set FRAME_GITHUB_TOKEN to raise the limit)' if host.endswith('github.com') else '' return data
raise SourceError(host + ' is limiting requests right now; try again later' + hint) from e
raise SourceError('Source HTTP error: ' + str(e.code)) from e raise SourceError('Source HTTP error: ' + str(e.code)) from e
except (OSError, ValueError) as e: except (OSError, ValueError) as e:
raise SourceError('Could not read source: ' + str(e)) from e raise SourceError('Could not read source: ' + str(e)) from e
def apk(url, hosts, digest=None): def apk(url, hosts, digest=None, name=None):
tmp = None tmp = None
name = name or _host(url)
wait = wait_time(url)
if wait:
raise limited_error(name, wait)
try: try:
fd, tmp = tempfile.mkstemp(dir=cache(), suffix='.part') fd, tmp = tempfile.mkstemp(dir=cache(), suffix='.part')
h = hashlib.sha256() h = hashlib.sha256()
@@ -99,6 +154,10 @@ def apk(url, hosts, digest=None):
path = os.path.join(cache(), actual + '.apk') path = os.path.join(cache(), actual + '.apk')
os.replace(tmp, path) os.replace(tmp, path)
return {'apk': path, 'obb': [], 'sha256': actual, 'verified': bool(digest)} return {'apk': path, 'obb': [], 'sha256': actual, 'verified': bool(digest)}
except urllib.error.HTTPError as e:
if e.code in (403, 429):
raise limited_error(name, throttle(url, e.headers)) from e
raise SourceError('Could not download APK: HTTP error ' + str(e.code)) from e
except (OSError, ValueError, zipfile.BadZipFile) as e: except (OSError, ValueError, zipfile.BadZipFile) as e:
raise SourceError('Could not download APK: ' + str(e)) from e raise SourceError('Could not download APK: ' + str(e)) from e
finally: finally:
+28 -11
View File
@@ -18,7 +18,7 @@ import zipfile
if __package__ in (None, ''): if __package__ in (None, ''):
sys.path.insert(0, str(Path(__file__).resolve().parents[1])) sys.path.insert(0, str(Path(__file__).resolve().parents[1]))
from apk_sources import SourceError from apk_sources import SourceError, SourceLimited, _web
import frame_host import frame_host
from frame_apk_sign import _der_parts, _cert_key, der from frame_apk_sign import _der_parts, _cert_key, der
from frame_catalog import _IndexReader, _reduce_index, _sha256 from frame_catalog import _IndexReader, _reduce_index, _sha256
@@ -85,17 +85,30 @@ class _HTTPSRedirect(urllib.request.HTTPRedirectHandler):
def _fetch(url, path, maximum): def _fetch(url, path, maximum):
host = urllib.parse.urlsplit(url).hostname
wait = _web.wait_time(url)
if wait:
raise _web.limited_error(host, wait)
request = urllib.request.Request(url, headers={'User-Agent': 'FrameControl/1.0'}) request = urllib.request.Request(url, headers={'User-Agent': 'FrameControl/1.0'})
with urllib.request.build_opener(_HTTPSRedirect()).open(request, timeout=60) as r, open(path, 'wb') as f: try:
total = 0 with urllib.request.build_opener(_HTTPSRedirect()).open(request, timeout=60) as r, open(path, 'wb') as f:
while True: total = 0
chunk = r.read(1 << 20) while True:
if not chunk: chunk = r.read(1 << 20)
break if not chunk:
total += len(chunk) break
if total > maximum: total += len(chunk)
raise SourceError('repository file exceeds size limit') if total > maximum:
f.write(chunk) raise SourceError('repository file exceeds size limit')
f.write(chunk)
except urllib.error.HTTPError as e:
if e.code in (403, 429):
raise _web.limited_error(host, _web.throttle(url, e.headers)) from e
raise
def _limited(source, error):
return _web.limited_error(source['name'], error.retry_after or _web.BACKOFF)
def _children(item): def _children(item):
@@ -474,6 +487,8 @@ def _load(source, force=False):
_write(cache, {'version': CACHE_VERSION, 'url': source['url'], 'fingerprint': pin, 'apps': apps}) _write(cache, {'version': CACHE_VERSION, 'url': source['url'], 'fingerprint': pin, 'apps': apps})
_accept(source, timestamp, v2) _accept(source, timestamp, v2)
return apps, pin return apps, pin
except SourceLimited as e:
raise _limited(source, e) from e
except SourceError: except SourceError:
raise raise
except (OSError, ValueError, KeyError, TypeError, IndexError) as e: except (OSError, ValueError, KeyError, TypeError, IndexError) as e:
@@ -572,6 +587,8 @@ def download(source, entry_id, version_code=None):
if os.path.exists(tmp): if os.path.exists(tmp):
os.unlink(tmp) os.unlink(tmp)
return {'apk': str(path), 'obb': [], 'sha256': sha, 'verified': True} return {'apk': str(path), 'obb': [], 'sha256': sha, 'verified': True}
except SourceLimited as e:
raise _limited(source, e) from e
except OSError as e: except OSError as e:
raise SourceError('cannot download APK: ' + str(e)) from e raise SourceError('cannot download APK: ' + str(e)) from e
+3 -2
View File
@@ -25,7 +25,8 @@ def _api(path):
if token: if token:
headers['Authorization'] = 'Bearer ' + token headers['Authorization'] = 'Bearer ' + token
try: try:
return json.loads(_web.read(API + path, ('api.github.com',), headers)) return json.loads(_web.read(API + path, ('api.github.com',), headers, name='GitHub',
hint='' if token else ' (set FRAME_GITHUB_TOKEN to raise the limit)'))
except (ValueError, TypeError) as e: except (ValueError, TypeError) as e:
raise SourceError('Invalid GitHub response') from e raise SourceError('Invalid GitHub response') from e
@@ -112,4 +113,4 @@ def download(source, entry_id, version_code=None):
expected = 'https://github.com/' + entry['id'] + '/releases/download/' expected = 'https://github.com/' + entry['id'] + '/releases/download/'
if not v['url'].startswith(expected): if not v['url'].startswith(expected):
raise SourceError('APK URL does not belong to the curated publisher') raise SourceError('APK URL does not belong to the curated publisher')
return _web.apk(v['url'], HOSTS, v['sha256']) return _web.apk(v['url'], HOSTS, v['sha256'], name='GitHub')
+1 -1
View File
@@ -54,7 +54,7 @@ def search(source, query, limit=50):
if tag not in FEEDS: if tag not in FEEDS:
raise SourceError('Unsupported itch.io feed') raise SourceError('Unsupported itch.io feed')
url = 'https://itch.io/games/free/platform-android/tag-' + tag + '.xml' url = 'https://itch.io/games/free/platform-android/tag-' + tag + '.xml'
for entry in _parse(source, _web.read(url, ('itch.io',))): for entry in _parse(source, _web.read(url, ('itch.io',), name='itch.io')):
entries.setdefault(entry['id'], entry) entries.setdefault(entry['id'], entry)
words = query.lower().split() words = query.lower().split()
return [e for e in entries.values() if all(w in (e['name'] + ' ' + e['summary']).lower() return [e for e in entries.values() if all(w in (e['name'] + ' ' + e['summary']).lower()
+1
View File
@@ -2009,6 +2009,7 @@ async function searchSources(quiet) {
const notes=[]; const notes=[];
if(loading.length)notes.push(`Still fetching ${loading.join(" and ")}; more results will appear in a moment.`); if(loading.length)notes.push(`Still fetching ${loading.join(" and ")}; more results will appear in a moment.`);
if(unavailable.length)notes.push(`${unavailable.join(" and ")} ${unavailable.length===1 ? "isn't" : "aren't"} responding right now. You can still explore the other sources.`); if(unavailable.length)notes.push(`${unavailable.join(" and ")} ${unavailable.length===1 ? "isn't" : "aren't"} responding right now. You can still explore the other sources.`);
result.sources.filter(s=>s.status==='limited').forEach(s=>notes.push(`${s.error}.`));
$("sourceStatus").innerHTML=notes.map(esc).join(" ")+(result.elsewhere||[]).map(x=>` <a href="${esc(x.url)}" target="_blank" rel="noopener">Browse ${esc(storeName(x.name))} ↗</a>`).join(""); $("sourceStatus").innerHTML=notes.map(esc).join(" ")+(result.elsewhere||[]).map(x=>` <a href="${esc(x.url)}" target="_blank" rel="noopener">Browse ${esc(storeName(x.name))} ↗</a>`).join("");
$("sourceStatus").hidden=!notes.length && !(result.elsewhere||[]).length; $("sourceStatus").hidden=!notes.length && !(result.elsewhere||[]).length;
clearTimeout(sourceState.retry); clearTimeout(sourceState.retry);