mirror of
https://github.com/github/spec-kit.git
synced 2026-08-03 06:26:30 +08:00
yamlio.py is the single chokepoint for every bundler read, and its module
docstring states the contract: "All reads/writes go through these functions so
that IO failures degrade into actionable BundlerError rather than raw
tracebacks."
Both readers catch only OSError, but `Path.read_text(encoding="utf-8")` and
`json.load()` raise UnicodeDecodeError on a non-UTF-8 file --
`issubclass(UnicodeDecodeError, OSError)` is False (its MRO is UnicodeError ->
ValueError). So the decode error escaped uncaught:
load_yaml: LEAKED UnicodeDecodeError -> 'utf-8' codec can't decode byte 0xff
load_json: LEAKED UnicodeDecodeError -> 'utf-8' codec can't decode byte 0xff
In load_json, json.JSONDecodeError does not help: it is a *sibling* of
UnicodeDecodeError, not a parent.
This is realistic rather than theoretical -- on Windows, PowerShell 5.1's
`Out-File` and `>` default to UTF-16, so a hand-edited
`.specify/bundle-catalogs.yml` or records file hits it.
Widen both read clauses to `(OSError, UnicodeError)`, matching the sibling
catalog readers (catalogs.py:101, workflows/catalog.py:336). JSONDecodeError
deliberately stays FIRST so malformed-but-decodable JSON keeps its more
specific "Invalid JSON" message; a regression test locks that ordering.
Write paths are unaffected -- verified that dump_yaml/dump_json do not leak
UnicodeEncodeError (both escape unencodable input), so this stays scoped to the
two read paths.
🤖 Generated with [Claude Code](https://claude.com/claude-code)
Co-authored-by: Claude Opus 5 (1M context) <noreply@anthropic.com>
59 lines
2.3 KiB
Python
59 lines
2.3 KiB
Python
"""Unit tests for the bundler YAML I/O helpers."""
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import pytest
|
|
|
|
from specify_cli.bundler import BundlerError
|
|
from specify_cli.bundler.lib.yamlio import dump_yaml, load_json, load_yaml
|
|
|
|
|
|
def test_dump_yaml_preserves_unicode(tmp_path: Path):
|
|
# dump_yaml must write literal UTF-8, not \xNN / \uXXXX escapes, so bundle
|
|
# config stays human-readable — matching _utils.dump_frontmatter and the
|
|
# extensions/presets config writers (all allow_unicode=True).
|
|
path = tmp_path / "f.yml"
|
|
data = {"note": "café-münchen", "url": "https://例え.example"}
|
|
dump_yaml(path, data)
|
|
raw = path.read_text(encoding="utf-8")
|
|
assert "café-münchen" in raw
|
|
assert "例え" in raw
|
|
assert "\\x" not in raw and "\\u" not in raw
|
|
|
|
|
|
def test_dump_yaml_round_trips_unicode(tmp_path: Path):
|
|
path = tmp_path / "f.yml"
|
|
data = {"note": "café", "city": "münchen"}
|
|
dump_yaml(path, data)
|
|
assert load_yaml(path) == data
|
|
|
|
|
|
def test_load_yaml_non_utf8_raises_bundler_error(tmp_path: Path):
|
|
"""A non-UTF-8 file must degrade into BundlerError, per this module's
|
|
documented contract. UnicodeDecodeError is a ValueError, not an OSError, so
|
|
it previously escaped as a raw traceback. UTF-16 is the realistic case:
|
|
PowerShell 5.1's `Out-File`/`>` default to it."""
|
|
path = tmp_path / "bundle-catalogs.yml"
|
|
path.write_bytes('catalogs: []\n'.encode("utf-16"))
|
|
with pytest.raises(BundlerError, match="Could not read"):
|
|
load_yaml(path)
|
|
|
|
|
|
def test_load_json_non_utf8_raises_bundler_error(tmp_path: Path):
|
|
"""Same for the JSON reader: json.JSONDecodeError is a *sibling* of
|
|
UnicodeDecodeError, so it does not cover a decode failure."""
|
|
path = tmp_path / "records.json"
|
|
path.write_bytes('{"bundles": []}'.encode("utf-16"))
|
|
with pytest.raises(BundlerError, match="Could not read"):
|
|
load_json(path)
|
|
|
|
|
|
def test_load_json_malformed_still_reports_invalid_json(tmp_path: Path):
|
|
"""Clause order regression guard: decodable-but-malformed JSON must keep the
|
|
more specific 'Invalid JSON' message rather than the read-error one."""
|
|
path = tmp_path / "records.json"
|
|
path.write_text('{"bundles": [', encoding="utf-8")
|
|
with pytest.raises(BundlerError, match="Invalid JSON"):
|
|
load_json(path)
|