Fix UTF-8 config reads on Windows (#318)

* Fix UTF-8 config reads on Windows

* test: cover utf8 production loaders

* fix: read and write settings as utf8

* Apply ruff formatting
This commit is contained in:
renaissancefieldlite
2026-07-04 22:10:24 -07:00
committed by GitHub
parent 5dacab7e4c
commit f086d77756
6 changed files with 143 additions and 17 deletions
+9 -3
View File
@@ -527,7 +527,7 @@ def load_config() -> EvoScientistConfig:
return EvoScientistConfig()
try:
with open(config_path) as f:
with open(config_path, encoding="utf-8") as f:
data = yaml.safe_load(f) or {}
# Filter to only valid fields
@@ -552,8 +552,14 @@ def save_config(config: EvoScientistConfig) -> None:
data = _config_to_dict(config)
# Save all fields including empty API keys (users can set them via env vars instead)
with open(config_path, "w") as f:
yaml.safe_dump(data, f, default_flow_style=False, sort_keys=False)
with open(config_path, "w", encoding="utf-8") as f:
yaml.safe_dump(
data,
f,
default_flow_style=False,
sort_keys=False,
allow_unicode=True,
)
def reset_config() -> None:
+11 -5
View File
@@ -194,7 +194,9 @@ def _write_workspace_sidecar(workspace_dir: Path, pid: int) -> None:
try:
RUNTIME.pid_dir.mkdir(parents=True, exist_ok=True)
tmp = RUNTIME.workspace_sidecar.with_suffix(".json.tmp")
tmp.write_text(json.dumps({"workspace": str(workspace_dir), "pid": pid}))
tmp.write_text(
json.dumps({"workspace": str(workspace_dir), "pid": pid}), encoding="utf-8"
)
os.replace(tmp, RUNTIME.workspace_sidecar)
except OSError as exc:
logger.warning(
@@ -218,7 +220,7 @@ def _read_workspace_sidecar() -> dict | None:
if not RUNTIME.workspace_sidecar.exists():
return None
try:
data = json.loads(RUNTIME.workspace_sidecar.read_text())
data = json.loads(RUNTIME.workspace_sidecar.read_text(encoding="utf-8"))
except (OSError, ValueError):
return None
if not isinstance(data, dict):
@@ -450,7 +452,7 @@ def _kill_owned_stale_process(port: int) -> bool:
if not RUNTIME.pid_file.exists():
return False
try:
owned_pid = int(RUNTIME.pid_file.read_text().strip())
owned_pid = int(RUNTIME.pid_file.read_text(encoding="utf-8").strip())
except (OSError, ValueError):
return False
@@ -672,6 +674,8 @@ def start_langgraph_dev(
# what the parent had inherited from its own environment.
sub_env = os.environ.copy()
sub_env["EVOSCIENTIST_WORKSPACE_DIR"] = str(workspace_dir)
sub_env["PYTHONIOENCODING"] = "utf-8"
sub_env["PYTHONUTF8"] = "1"
# By default, let langgraph dev write its full ``.langgraph_api/`` cache
# so future use cases — cross-session async tasks, Store API persistence,
@@ -732,7 +736,7 @@ def start_langgraph_dev(
log_handle.close()
except Exception:
pass
RUNTIME.pid_file.write_text(str(proc.pid))
RUNTIME.pid_file.write_text(str(proc.pid), encoding="utf-8")
_write_workspace_sidecar(workspace_dir=workspace_dir, pid=proc.pid)
global _PROCESS_WORKSPACE
_PROCESS = proc
@@ -746,7 +750,9 @@ def start_langgraph_dev(
if proc.poll() is not None:
tail = ""
try:
tail = RUNTIME.log_file.read_text()[-2000:]
tail = RUNTIME.log_file.read_text(encoding="utf-8", errors="replace")[
-2000:
]
except Exception:
pass
# Subprocess died on its own — clear our module-level bookkeeping
+4 -3
View File
@@ -190,7 +190,7 @@ def _load_user_config() -> dict[str, Any]:
"""Load the user-level MCP config, returning an empty dict if absent."""
if USER_MCP_CONFIG.is_file():
try:
data = yaml.safe_load(USER_MCP_CONFIG.read_text()) or {}
data = yaml.safe_load(USER_MCP_CONFIG.read_text(encoding="utf-8")) or {}
return data if isinstance(data, dict) else {}
except Exception:
return {}
@@ -201,7 +201,8 @@ def _save_user_config(config: dict[str, Any]) -> None:
"""Write *config* to the user-level MCP config file."""
USER_CONFIG_DIR.mkdir(parents=True, exist_ok=True)
USER_MCP_CONFIG.write_text(
yaml.dump(config, default_flow_style=False, sort_keys=False)
yaml.dump(config, default_flow_style=False, sort_keys=False),
encoding="utf-8",
)
@@ -584,7 +585,7 @@ def load_mcp_config() -> dict[str, Any]:
return {}
try:
data = yaml.safe_load(USER_MCP_CONFIG.read_text()) or {}
data = yaml.safe_load(USER_MCP_CONFIG.read_text(encoding="utf-8")) or {}
if not isinstance(data, dict):
return {}
except Exception as exc:
+2 -2
View File
@@ -138,7 +138,7 @@ def _uv_tool_existing_requirements() -> dict[str, str]:
except ModuleNotFoundError:
return {}
try:
data = tomllib.loads(receipt.read_text())
data = tomllib.loads(receipt.read_text(encoding="utf-8"))
except Exception:
return {}
tool_name = _uv_tool_name() or ""
@@ -410,7 +410,7 @@ def _clone_repo(repo: str, ref: str | None, dest: str) -> None:
def parse_marketplace_yaml(path: Path) -> MCPServerEntry:
"""Parse a single marketplace YAML file into an MCPServerEntry."""
data = yaml.safe_load(path.read_text()) or {}
data = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
if not isinstance(data, dict):
raise ValueError(f"Expected a YAML mapping in {path}")
+24
View File
@@ -254,6 +254,30 @@ class TestLoadSaveReset:
assert loaded.anthropic_api_key == "test-key"
assert loaded.provider == "openai"
def test_load_reads_manually_edited_utf8_config(self, temp_config_dir, clean_env):
"""Config loading handles localized content from manually edited YAML."""
config_path = get_config_path()
config_path.parent.mkdir(parents=True, exist_ok=True)
config_path.write_text(
"provider: openai\nmodel: gpt-4o\ndefault_workdir: /tmp/café\n",
encoding="utf-8",
)
loaded = load_config()
assert loaded.provider == "openai"
assert loaded.model == "gpt-4o"
assert loaded.default_workdir == "/tmp/café"
def test_save_writes_utf8_config(self, temp_config_dir, clean_env):
"""Config saving writes localized content with UTF-8 encoding."""
save_config(EvoScientistConfig(default_workdir="/tmp/résumé"))
config_path = get_config_path()
raw = config_path.read_text(encoding="utf-8")
assert "default_workdir: /tmp/résumé" in raw
assert load_config().default_workdir == "/tmp/résumé"
def test_reset_deletes_config_file(self, temp_config_dir, clean_env):
"""Test that reset deletes the config file."""
config = EvoScientistConfig(provider="openai")
+93 -4
View File
@@ -92,7 +92,7 @@ def test_scheduler_yaml_loads_as_async():
subagents_dir = Path(EvoScientist.__file__).parent / "subagents"
yaml_path = subagents_dir / "scheduler.yaml"
assert yaml_path.exists(), f"Missing {yaml_path}"
data = yaml.safe_load(yaml_path.read_text())
data = yaml.safe_load(yaml_path.read_text(encoding="utf-8"))
assert "scheduler" in data, "Top-level key must be 'scheduler'"
spec = data["scheduler"]
assert spec.get("async") is True, "scheduler must have 'async: true'"
@@ -109,7 +109,9 @@ def test_langgraph_json_registers_scheduler():
import EvoScientist
root = Path(EvoScientist.__file__).parent
manifest = json.loads((root / "langgraph_dev" / "langgraph.json").read_text())
manifest = json.loads(
(root / "langgraph_dev" / "langgraph.json").read_text(encoding="utf-8")
)
assert manifest["graphs"]["scheduler"] == (
"EvoScientist.langgraph_dev.graphs:scheduler"
)
@@ -126,11 +128,98 @@ def test_scheduler_graph_id_matches_registration():
from EvoScientist.cron import schedule as crons
root = Path(EvoScientist.__file__).parent
manifest = json.loads((root / "langgraph_dev" / "langgraph.json").read_text())
manifest = json.loads(
(root / "langgraph_dev" / "langgraph.json").read_text(encoding="utf-8")
)
assert crons.SCHEDULER_GRAPH_ID in manifest["graphs"], (
f"SCHEDULER_GRAPH_ID={crons.SCHEDULER_GRAPH_ID!r} not found in langgraph.json graphs"
)
spec = yaml.safe_load((root / "subagents" / "scheduler.yaml").read_text())
spec = yaml.safe_load(
(root / "subagents" / "scheduler.yaml").read_text(encoding="utf-8")
)
assert crons.SCHEDULER_GRAPH_ID in spec, (
f"SCHEDULER_GRAPH_ID={crons.SCHEDULER_GRAPH_ID!r} is not the top-level key in scheduler.yaml"
)
def test_production_loaders_accept_utf8_content(tmp_path, monkeypatch):
"""Production config readers must handle localized UTF-8 content."""
import os
from EvoScientist.langgraph_dev import manager
from EvoScientist.mcp import client, registry
mcp_config = tmp_path / "mcp.yaml"
mcp_config.write_text(
"""
écho:
transport: stdio
command: python
args: ["-m", "démo"]
env:
MESSAGE: "bonjour café"
""".lstrip(),
encoding="utf-8",
)
monkeypatch.setattr(client, "USER_MCP_CONFIG", mcp_config)
assert client._load_user_config()["écho"]["env"]["MESSAGE"] == "bonjour café"
assert client.load_mcp_config()["écho"]["args"] == ["-m", "démo"]
marketplace_yaml = tmp_path / "marketplace-écho.yaml"
marketplace_yaml.write_text(
"""
name: écho
label: "Écho café"
description: "localized marketplace entry"
tags: "démo, marché"
transport: stdio
command: python
args: ["-m", "écho"]
""".lstrip(),
encoding="utf-8",
)
entry = registry.parse_marketplace_yaml(marketplace_yaml)
assert entry.name == "écho"
assert entry.tags == ["démo", "marché"]
venv = tmp_path / "uv" / "tools" / "evoscientist"
venv.mkdir(parents=True)
(venv / "uv-receipt.toml").write_text(
"""
[tool]
requirements = [
{ name = "evoscientist" },
{ name = "mcp-écho", specifier = ">=1.0" },
]
""".lstrip(),
encoding="utf-8",
)
monkeypatch.setenv("VIRTUAL_ENV", str(venv))
assert registry._uv_tool_existing_requirements()["mcp-écho"] == "mcp-écho>=1.0"
runtime = manager.LanggraphRuntimePaths.for_directory(tmp_path / "runtime")
monkeypatch.setattr(manager, "RUNTIME", runtime)
workspace = tmp_path / "workspace-écho"
manager._write_workspace_sidecar(workspace, 12345)
assert manager._read_workspace_sidecar() == {
"workspace": str(workspace),
"pid": 12345,
}
runtime.pid_file.write_text(str(os.getpid()), encoding="utf-8")
class _NotLanggraphProcess:
def __init__(self, pid):
self.pid = pid
def cmdline(self):
return ["python", "localized-test"]
def kill(self):
raise AssertionError("non-langgraph pid must not be killed")
monkeypatch.setattr(manager, "_list_pids_on_port", lambda port: [os.getpid()])
monkeypatch.setattr(manager.psutil, "Process", _NotLanggraphProcess)
assert manager._kill_owned_stale_process(6174) is False
assert not runtime.pid_file.exists()
assert not runtime.workspace_sidecar.exists()