Skip to content
Closed
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1,123 changes: 1,123 additions & 0 deletions scripts/check-encoding-safety.py

Large diffs are not rendered by default.

11 changes: 11 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f1_env_loader_primary.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,11 @@
"""Fixed form of H1 — primary path uses utf-8-sig (#65124)."""
from pathlib import Path
from dotenv import load_dotenv


def _load_dotenv_with_fallback(path: Path, *, override: bool) -> None:
try:
# utf-8-sig strips a leading UTF-8 BOM if present.
load_dotenv(dotenv_path=path, override=override, encoding="utf-8-sig")
except UnicodeDecodeError:
pass
15 changes: 15 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f2_env_loader_fallback.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
"""Fixed form of H2 — latin-1 fallback strips BOM_UTF8 first (#65124)."""
import codecs
import io
from pathlib import Path
from dotenv import load_dotenv


def _load_dotenv_with_fallback(path: Path, *, override: bool) -> None:
try:
load_dotenv(dotenv_path=path, override=override, encoding="utf-8-sig")
except UnicodeDecodeError:
raw = path.read_bytes()
if raw.startswith(codecs.BOM_UTF8):
raw = raw[len(codecs.BOM_UTF8) :]
load_dotenv(stream=io.StringIO(raw.decode("latin-1")), override=override)
29 changes: 29 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f3_send_cmd.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,29 @@
"""Fixed form of H3 — send_cmd private loader BOM-safe (#65124)."""
import codecs
import io
from pathlib import Path


def _load_hermes_env() -> None:
try:
from dotenv import load_dotenv
except Exception:
load_dotenv = None # type: ignore[assignment]

home = Path.home() / ".hermes"
env_path = home / ".env"
if load_dotenv and env_path.exists():
try:
load_dotenv(str(env_path), override=True, encoding="utf-8-sig")
except UnicodeDecodeError:
try:
raw = Path(env_path).read_bytes()
if raw.startswith(codecs.BOM_UTF8):
raw = raw[len(codecs.BOM_UTF8) :]
load_dotenv(
stream=io.StringIO(raw.decode("latin-1")), override=True
)
except Exception:
pass
except Exception:
pass
66 changes: 66 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f4_sanitize.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,66 @@
"""Fixed form of H4 — BOM-sniff before decode; refuse-to-mangle (#66475)."""
import codecs
import io
import os
import tempfile
from pathlib import Path


def atomic_replace(src, dst):
os.replace(src, dst)


def _sanitize_env_lines(lines):
return lines


def _sanitize_env_file_if_needed(path: Path) -> None:
if not path.exists():
return
try:
raw = path.read_bytes()
except Exception:
return

force_utf8_rewrite = False
if raw.startswith(codecs.BOM_UTF32_LE) or raw.startswith(codecs.BOM_UTF32_BE):
return # refuse-to-mangle
if raw.startswith(codecs.BOM_UTF16_LE) or raw.startswith(codecs.BOM_UTF16_BE):
try:
with io.TextIOWrapper(
io.BytesIO(raw), encoding="utf-16", newline=None
) as f:
original = f.readlines()
except UnicodeDecodeError:
return
force_utf8_rewrite = True
else:
# utf-8-sig path WITHOUT persisting errors=replace corruption:
# read with replace for NUL stripping, but abort rewrite if the
# first line starts with U+FFFD (unknown binary / mis-decoded).
try:
with open(path, encoding="utf-8-sig", errors="replace") as f: # encoding-safety: ok — guarded: abort rewrite when first line starts with U+FFFD; UTF-16 sniffed above
original = f.readlines()
except Exception:
return
if original and original[0].startswith("\ufffd"):
return

stripped = [line.replace("\x00", "") for line in original]
sanitized = _sanitize_env_lines(stripped)
if sanitized != original or force_utf8_rewrite:
fd, tmp = tempfile.mkstemp(
dir=str(path.parent), suffix=".tmp", prefix=".env_"
)
try:
with os.fdopen(fd, "w", encoding="utf-8") as f:
f.writelines(sanitized)
f.flush()
os.fsync(f.fileno())
atomic_replace(tmp, path)
except BaseException:
try:
os.unlink(tmp)
except OSError:
pass
raise
41 changes: 41 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f5_quote_env_read.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,41 @@
"""Fixed form of H5 — save_env_value reads with utf-8-sig."""
from pathlib import Path
import os
import tempfile


def _quote_env_value(value: str) -> str:
if value == "" or value == value.strip() and "#" not in value:
return value
return f'"{value}"'


def save_env_value(key: str, value: str, env_path: Path) -> None:
# utf-8-sig strips BOM; errors=replace is for embedded NULs only and
# the rewrite path is the intentional writer for this key — not a
# sanitize-on-unknown-encoding path (see R3 / #66474).
read_kw = {"encoding": "utf-8-sig", "errors": "replace"}
write_kw = {"encoding": "utf-8"}

lines = []
if env_path.exists():
with open(env_path, **read_kw) as f: # encoding-safety: ok — intentional writer; utf-8-sig primary; not sanitize-unknown-encoding
lines = f.readlines()

lines.append(f"{key}={_quote_env_value(value)}\n")

fd, tmp_path = tempfile.mkstemp(
dir=str(env_path.parent), suffix=".tmp", prefix=".env_"
)
try:
with os.fdopen(fd, "w", **write_kw) as f:
f.writelines(lines)
f.flush()
os.fsync(f.fileno())
os.replace(tmp_path, env_path)
except BaseException:
try:
os.unlink(tmp_path)
except OSError:
pass
raise
37 changes: 37 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f6_managed_scope.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,37 @@
"""Fixed form of H6 — managed_scope reads with utf-8-sig."""
from pathlib import Path
from typing import Dict
import copy


_ENV_CACHE: Dict[str, tuple] = {}
_CACHE_LOCK = __import__("threading").Lock()


def _parse_env(f):
return {}


def _cached_read(path: Path, cache: Dict[str, tuple], parse):
try:
st = path.stat()
except OSError:
return None
key = (st.st_mtime_ns, st.st_size)
path_key = str(path)
with _CACHE_LOCK:
hit = cache.get(path_key)
if hit is not None and hit[:2] == key:
return copy.deepcopy(hit[2])
try:
with open(path, encoding="utf-8-sig") as f:
parsed = parse(f)
except Exception:
return None
with _CACHE_LOCK:
cache[path_key] = (key[0], key[1], copy.deepcopy(parsed))
return parsed


def load_managed_env(managed_dir: Path) -> dict:
return _cached_read(managed_dir / ".env", _ENV_CACHE, _parse_env) or {}
7 changes: 7 additions & 0 deletions tests/fixtures/encoding_safety/fixed/f_benign_internal.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,7 @@
"""Benign: plain utf-8 on an internal package resource (not user-writable)."""
from pathlib import Path


def load_bundled_schema(root: Path) -> str:
schema = root / "schemas" / "internal_v1.json"
return schema.read_text(encoding="utf-8")
15 changes: 15 additions & 0 deletions tests/fixtures/encoding_safety/historical/h1_env_loader_primary.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
"""Historical instance #1 — env_loader primary path (pre-#65124).

load_dotenv(..., encoding="utf-8") on user .env: a UTF-8 BOM sticks to the
first key as U+FEFF and silently drops it from os.environ under its
canonical name. See #65123.
"""
from pathlib import Path
from dotenv import load_dotenv


def _load_dotenv_with_fallback(path: Path, *, override: bool) -> None:
try:
load_dotenv(dotenv_path=path, override=override, encoding="utf-8")
except UnicodeDecodeError:
pass
Original file line number Diff line number Diff line change
@@ -0,0 +1,15 @@
"""Historical instance #2 — env_loader latin-1 fallback (pre-#65124).

UnicodeDecodeError → load_dotenv(..., encoding="latin-1") without stripping
codecs.BOM_UTF8. A UTF-8 BOM survives as U+FEFF on the first key.
"""
from pathlib import Path
from dotenv import load_dotenv


def _load_dotenv_with_fallback(path: Path, *, override: bool) -> None:
try:
# Primary intentionally uses utf-8-sig so this fixture isolates R2.
load_dotenv(dotenv_path=path, override=override, encoding="utf-8-sig")
except UnicodeDecodeError:
load_dotenv(dotenv_path=path, override=override, encoding="latin-1")
27 changes: 27 additions & 0 deletions tests/fixtures/encoding_safety/historical/h3_send_cmd.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,27 @@
"""Historical instance #3 — send_cmd._load_hermes_env (pre-#65124 send_cmd fix).

Private dotenv loader: plain utf-8 primary + latin-1 fallback, same BOM drop
as env_loader. Intentionally non-equivalent to the shared loader (profile
path resolution; no sanitize/secret-source side effects).
"""
from pathlib import Path


def _load_hermes_env() -> None:
try:
from dotenv import load_dotenv
except Exception:
load_dotenv = None # type: ignore[assignment]

home = Path.home() / ".hermes"
env_path = home / ".env"
if load_dotenv and env_path.exists():
try:
load_dotenv(str(env_path), override=True, encoding="utf-8")
except UnicodeDecodeError:
try:
load_dotenv(str(env_path), override=True, encoding="latin-1")
except Exception:
pass
except Exception:
pass
47 changes: 47 additions & 0 deletions tests/fixtures/encoding_safety/historical/h4_sanitize.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,47 @@
"""Historical instance #4 — _sanitize_env_file_if_needed (pre-#66475).

utf-8-sig + errors=replace on a user .env, then rewrite when sanitize changes
lines. UTF-16 BOM (Notepad "Unicode") becomes U+FFFD U+FFFD KEY and is
permanently written back. See #66474.
"""
from pathlib import Path
import os
import tempfile


def atomic_replace(src, dst):
os.replace(src, dst)


def _sanitize_env_lines(lines):
return lines


def _sanitize_env_file_if_needed(path: Path) -> None:
if not path.exists():
return

read_kw = {"encoding": "utf-8-sig", "errors": "replace"}
try:
with open(path, **read_kw) as f:
original = f.readlines()
stripped = [line.replace("\x00", "") for line in original]
sanitized = _sanitize_env_lines(stripped)
if sanitized != original:
fd, tmp = tempfile.mkstemp(
dir=str(path.parent), suffix=".tmp", prefix=".env_"
)
try:
with os.fdopen(fd, "w", encoding="utf-8") as f:
f.writelines(sanitized)
f.flush()
os.fsync(f.fileno())
atomic_replace(tmp, path)
except BaseException:
try:
os.unlink(tmp)
except OSError:
pass
raise
except Exception:
pass
45 changes: 45 additions & 0 deletions tests/fixtures/encoding_safety/historical/h5_quote_env_read.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,45 @@
"""Historical instance #5 — quote_env_value / save_env_value read context.

Pre-utf-8-sig: save_env_value read ~/.hermes/.env with plain utf-8 before
rewriting a quoted KEY=value line. A BOM stuck to the first key; the
rewrite then persisted the mangled name. (Current main uses utf-8-sig;
this fixture freezes the pre-fix shape.)
"""
from pathlib import Path
import os
import tempfile


def _quote_env_value(value: str) -> str:
if value == "" or value == value.strip() and "#" not in value:
return value
return f'"{value.replace(chr(34), chr(92) + chr(34))}"'


def save_env_value(key: str, value: str, env_path: Path) -> None:
read_kw = {"encoding": "utf-8", "errors": "replace"}
write_kw = {"encoding": "utf-8"}

lines = []
if env_path.exists():
with open(env_path, **read_kw) as f:
lines = f.readlines()

serialized_value = _quote_env_value(value)
lines.append(f"{key}={serialized_value}\n")

fd, tmp_path = tempfile.mkstemp(
dir=str(env_path.parent), suffix=".tmp", prefix=".env_"
)
try:
with os.fdopen(fd, "w", **write_kw) as f:
f.writelines(lines)
f.flush()
os.fsync(f.fileno())
os.replace(tmp_path, env_path)
except BaseException:
try:
os.unlink(tmp_path)
except OSError:
pass
raise
Loading
Loading