Skip to content

Commit 691fae3

Browse files
fix: use bounded read for integration catalog HTTP responses
The integration catalog fetch used unbounded resp.read() to read HTTP responses into memory. A malicious or misconfigured catalog server could return an arbitrarily large response causing OOM. Replace with read_response_limited() capped at MAX_JSON_METADATA_BYTES (1 MiB), consistent with how other JSON fetch paths in the codebase (_version.py, _github_http.py, authentication/azure_devops.py) already enforce bounded reads. Pass error_type=IntegrationCatalogError so oversized catalogs are caught by the existing per-entry recovery path in _get_merged_integrations() rather than aborting the entire merge. Add regression test verifying oversized responses are rejected as IntegrationCatalogError and that healthy catalogs remain usable.
1 parent 795d963 commit 691fae3

2 files changed

Lines changed: 138 additions & 5 deletions

File tree

‎src/specify_cli/integrations/catalog.py‎

Lines changed: 9 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -21,6 +21,7 @@
2121
import yaml
2222
from packaging import version as pkg_version
2323

24+
from .._download_security import MAX_JSON_METADATA_BYTES, read_response_limited
2425
from ..catalogs import CatalogEntry, CatalogStackBase
2526

2627

@@ -200,7 +201,14 @@ def _fetch_single_catalog(
200201
final_url = resp.geturl()
201202
if final_url != entry.url:
202203
self._validate_catalog_url(final_url)
203-
catalog_data = json.loads(resp.read())
204+
catalog_data = json.loads(
205+
read_response_limited(
206+
resp,
207+
max_bytes=MAX_JSON_METADATA_BYTES,
208+
error_type=IntegrationCatalogError,
209+
label=f"catalog from {entry.url}",
210+
)
211+
)
204212

205213
shape_error = _catalog_shape_error(catalog_data)
206214
if shape_error is not None:

‎tests/integrations/test_integration_catalog.py‎

Lines changed: 129 additions & 4 deletions
Original file line numberDiff line numberDiff line change
@@ -220,6 +220,33 @@ def test_load_catalog_config_rejects_falsy_non_mapping_roots(
220220
# ---------------------------------------------------------------------------
221221

222222

223+
class _OversizedResponse:
224+
"""Response stub that supports bounded streaming reads for oversized-catalog tests."""
225+
226+
def __init__(self, data, url=""):
227+
self._data = json.dumps(data).encode()
228+
self._url = url if isinstance(url, str) else url.full_url
229+
self._pos = 0
230+
231+
def read(self, n=-1):
232+
if n < 0:
233+
chunk = self._data[self._pos:]
234+
self._pos = len(self._data)
235+
return chunk
236+
chunk = self._data[self._pos : self._pos + n]
237+
self._pos += len(chunk)
238+
return chunk
239+
240+
def geturl(self):
241+
return self._url
242+
243+
def __enter__(self):
244+
return self
245+
246+
def __exit__(self, *a):
247+
pass
248+
249+
223250
class TestCatalogFetch:
224251
"""Tests that use a local HTTP server stub via monkeypatch."""
225252

@@ -230,9 +257,16 @@ class FakeResponse:
230257
def __init__(self, data, url=""):
231258
self._data = json.dumps(data).encode()
232259
self._url = url if isinstance(url, str) else url.full_url
260+
self._pos = 0
233261

234-
def read(self):
235-
return self._data
262+
def read(self, n=-1):
263+
if n < 0:
264+
chunk = self._data[self._pos:]
265+
self._pos = len(self._data)
266+
return chunk
267+
chunk = self._data[self._pos:self._pos + n]
268+
self._pos += len(chunk)
269+
return chunk
236270

237271
def geturl(self):
238272
return self._url
@@ -395,6 +429,90 @@ def test_invalid_catalog_format(self, tmp_path, monkeypatch):
395429
with pytest.raises(IntegrationCatalogError, match="Failed to fetch any integration catalog"):
396430
cat.search()
397431

432+
def test_oversized_catalog_response_rejected(self, tmp_path, monkeypatch):
433+
"""Response exceeding MAX_JSON_METADATA_BYTES is caught as IntegrationCatalogError.
434+
435+
The per-entry error is logged as a warning and skipped (not fatal).
436+
When ALL catalogs are oversized, search() raises the aggregate error.
437+
"""
438+
from specify_cli._download_security import MAX_JSON_METADATA_BYTES
439+
440+
monkeypatch.setenv("HOME", str(tmp_path))
441+
monkeypatch.setenv("USERPROFILE", str(tmp_path))
442+
monkeypatch.delenv("SPECKIT_INTEGRATION_CATALOG_URL", raising=False)
443+
(tmp_path / ".specify").mkdir()
444+
cat = IntegrationCatalog(tmp_path)
445+
446+
# Build a valid catalog dict whose JSON encoding exceeds the limit.
447+
oversized = {
448+
"schema_version": "1.0",
449+
"integrations": {},
450+
"padding": "x" * (MAX_JSON_METADATA_BYTES + 1),
451+
}
452+
453+
import specify_cli.authentication.http as _auth_http
454+
455+
def _oversized_urlopen(req, timeout=10):
456+
url = req if isinstance(req, str) else req.full_url
457+
return _OversizedResponse(oversized, url)
458+
459+
monkeypatch.setattr(_auth_http.urllib.request, "urlopen", _oversized_urlopen)
460+
461+
# Both default + community catalogs are oversized → all fail → aggregate error.
462+
# The per-entry IntegrationCatalogError (with "exceeds maximum size") is
463+
# logged as a warning; the aggregate raise has a different message.
464+
with pytest.raises(IntegrationCatalogError, match="Failed to fetch any integration catalog"):
465+
cat.search()
466+
467+
def test_oversized_catalog_does_not_block_healthy_one(self, tmp_path, monkeypatch):
468+
"""When one catalog is oversized, the healthy catalog still returns results."""
469+
from specify_cli._download_security import MAX_JSON_METADATA_BYTES
470+
471+
monkeypatch.setenv("HOME", str(tmp_path))
472+
monkeypatch.setenv("USERPROFILE", str(tmp_path))
473+
monkeypatch.delenv("SPECKIT_INTEGRATION_CATALOG_URL", raising=False)
474+
specify = tmp_path / ".specify"
475+
specify.mkdir()
476+
477+
healthy_catalog = {
478+
"schema_version": "1.0",
479+
"integrations": {
480+
"good-agent": {
481+
"id": "good-agent",
482+
"name": "Good Agent",
483+
"version": "1.0.0",
484+
"description": "A healthy integration",
485+
"author": "test-org",
486+
},
487+
},
488+
}
489+
oversized_catalog = {
490+
"schema_version": "1.0",
491+
"integrations": {},
492+
"padding": "x" * (MAX_JSON_METADATA_BYTES + 1),
493+
}
494+
cfg = specify / "integration-catalogs.yml"
495+
cfg.write_text(yaml.dump({"catalogs": [
496+
{"url": "https://healthy.example.com/catalog.json", "name": "healthy", "priority": 1, "install_allowed": True},
497+
{"url": "https://oversized.example.com/catalog.json", "name": "oversized", "priority": 2, "install_allowed": True},
498+
]}))
499+
cat = IntegrationCatalog(tmp_path)
500+
501+
import specify_cli.authentication.http as _auth_http
502+
503+
def _multi_catalog_urlopen(req, timeout=10):
504+
url = req if isinstance(req, str) else req.full_url
505+
if "oversized" in url:
506+
return _OversizedResponse(oversized_catalog, url)
507+
return _OversizedResponse(healthy_catalog, url)
508+
509+
monkeypatch.setattr(_auth_http.urllib.request, "urlopen", _multi_catalog_urlopen)
510+
511+
# The oversized catalog is skipped; the healthy catalog's integrations are returned.
512+
results = cat.search()
513+
ids = [r["id"] for r in results]
514+
assert "good-agent" in ids
515+
398516
def test_clear_cache(self, tmp_path):
399517
(tmp_path / ".specify").mkdir()
400518
cat = IntegrationCatalog(tmp_path)
@@ -592,8 +710,15 @@ class FakeResponse:
592710
def __init__(self, data, url=""):
593711
self._data = json.dumps(data).encode()
594712
self._url = url if isinstance(url, str) else url.full_url
595-
def read(self):
596-
return self._data
713+
self._pos = 0
714+
def read(self, n=-1):
715+
if n < 0:
716+
chunk = self._data[self._pos:]
717+
self._pos = len(self._data)
718+
return chunk
719+
chunk = self._data[self._pos:self._pos + n]
720+
self._pos += len(chunk)
721+
return chunk
597722
def geturl(self):
598723
return self._url
599724
def __enter__(self):

0 commit comments

Comments
 (0)