diff --git a/src/specify_cli/extensions/__init__.py b/src/specify_cli/extensions/__init__.py index 6d78354809..3a64c74092 100644 --- a/src/specify_cli/extensions/__init__.py +++ b/src/specify_cli/extensions/__init__.py @@ -641,8 +641,12 @@ def _load(self) -> dict: if not isinstance(data.get("extensions"), dict): data["extensions"] = {} return data - except (json.JSONDecodeError, FileNotFoundError): - # Corrupted or missing registry, start fresh + except (json.JSONDecodeError, UnicodeDecodeError, FileNotFoundError): + # Corrupted or missing registry, start fresh. A registry whose + # bytes cannot be decoded as UTF-8 is the same corruption class + # as malformed JSON — only the exception type differs. OSError is + # deliberately not caught: the data may be intact on disk, and + # starting fresh would let a later _save() wipe it. return {"schema_version": self.SCHEMA_VERSION, "extensions": {}} def _save(self): @@ -4255,11 +4259,11 @@ def _sibling_extension_ids(self) -> list[str]: Returns an empty list if the registry is missing or corrupted (fresh project, ad-hoc test harness) so ``_get_env_config`` degrades to its pre-fix behaviour rather than crashing. ``UnicodeError`` is - caught alongside ``OSError`` because ``ExtensionRegistry._load()`` - opens the file in text mode and only handles ``JSONDecodeError`` / - ``FileNotFoundError``, so a registry file with non-UTF-8 bytes would - otherwise surface a ``UnicodeDecodeError`` here and break *every* - config read instead of degrading gracefully. + caught alongside ``OSError`` as defense in depth: + ``ExtensionRegistry._load()`` now starts fresh on a non-UTF-8 + registry itself, but this scan must degrade gracefully even if that + contract regresses, because a failure here would break *every* + config read. Used by ``_get_env_config`` to detect env vars whose remainder claims a longer, sibling-owned prefix (e.g. ``SPECKIT_GIT_HOOKS_URL`` is diff --git a/tests/test_extensions.py b/tests/test_extensions.py index 3d9146d52b..07cc3a6759 100644 --- a/tests/test_extensions.py +++ b/tests/test_extensions.py @@ -1241,6 +1241,28 @@ def test_list_returns_empty_dict_for_corrupted_registry(self, temp_dir): result = registry.list() assert result == {} + def test_load_starts_fresh_for_non_utf8_registry(self, temp_dir): + """A registry file with undecodable bytes must start fresh, not raise. + + ``_load()`` documents "Corrupted or missing registry, start fresh" and + already treats malformed JSON that way, but a registry whose *bytes* + cannot be decoded as UTF-8 raised a raw ``UnicodeDecodeError`` from the + same boundary — the same corruption class reaching a different + exception type. + """ + extensions_dir = temp_dir / "extensions" + extensions_dir.mkdir() + (extensions_dir / ExtensionRegistry.REGISTRY_FILE).write_bytes( + b"\xff\xfe not utf-8 \xc3\x28" + ) + + registry = ExtensionRegistry(extensions_dir) + + assert registry.data == { + "schema_version": ExtensionRegistry.SCHEMA_VERSION, + "extensions": {}, + } + # ===== ExtensionManager Tests ===== @@ -10450,10 +10472,10 @@ def test_config_only_leftover_not_treated_as_sibling(self, tmp_path, monkeypatch def test_non_utf8_registry_does_not_crash(self, tmp_path, monkeypatch): """A registry file with invalid text encoding must NOT propagate ``UnicodeDecodeError`` out of the sibling scan and abort every - config read. ``ExtensionRegistry._load()`` catches ``JSONDecodeError`` - / ``FileNotFoundError`` only, so ``_sibling_extension_ids`` must - additionally swallow ``UnicodeError`` and degrade to the documented - pre-fix behaviour. + config read. ``ExtensionRegistry._load()`` now starts fresh on a + non-UTF-8 registry itself, but ``_sibling_extension_ids`` keeps + swallowing ``UnicodeError`` as defense in depth so a regression of + that contract still degrades to the documented pre-fix behaviour. """ extensions_dir = tmp_path / ".specify" / "extensions" extensions_dir.mkdir(parents=True)