diff --git a/evals/compaction/fixtures.py b/evals/compaction/fixtures.py index 7c5b4e5a59..7b5b1f018b 100644 --- a/evals/compaction/fixtures.py +++ b/evals/compaction/fixtures.py @@ -30,7 +30,7 @@ def load_transcript(path: str, cap_tokens: int | None = None) -> List[Dict[str, The cap takes the chronological prefix, then drops trailing assistant tool_calls whose results were cut off so the input is well-formed. """ - data = json.load(open(path)) + data = json.load(open(path, encoding="utf-8")) msgs = data["messages"] if isinstance(data, dict) else data if cap_tokens is None: return msgs diff --git a/evals/compaction/report.py b/evals/compaction/report.py index c80efb788a..7dfb999803 100644 --- a/evals/compaction/report.py +++ b/evals/compaction/report.py @@ -11,7 +11,7 @@ from pathlib import Path def main(): out_dir = Path(sys.argv[1]) - card = json.loads((out_dir / "scorecard.json").read_text()) + card = json.loads((out_dir / "scorecard.json").read_text(encoding="utf-8")) card.sort(key=lambda s: -s["recall_pct"]) rows = [] @@ -35,7 +35,7 @@ def main(): md = ["| " + " | ".join(headers) + " |", "|" + "|".join("---" for _ in headers) + "|"] for r in rows: md.append("| " + " | ".join(r) + " |") - (out_dir / "scorecard.md").write_text("\n".join(md) + "\n") + (out_dir / "scorecard.md").write_text("\n".join(md) + "\n", encoding="utf-8") print(f"\nmarkdown -> {out_dir}/scorecard.md") diff --git a/evals/compaction/runner.py b/evals/compaction/runner.py index 63d48d2238..920bbd017c 100644 --- a/evals/compaction/runner.py +++ b/evals/compaction/runner.py @@ -204,7 +204,7 @@ def summarized_region(compressor_module, messages): def generate_questions(messages, n: int, cache_path: Path) -> list: if cache_path.exists(): - return json.loads(cache_path.read_text()) + return json.loads(cache_path.read_text(encoding="utf-8")) import agent.context_compressor as cc region = summarized_region(cc, messages) @@ -212,7 +212,7 @@ def generate_questions(messages, n: int, cache_path: Path) -> list: raw = _call(QUESTION_PROMPT.format(n=n, transcript=text), max_tokens=4000) questions = _extract_json(raw)[:n] cache_path.parent.mkdir(parents=True, exist_ok=True) - cache_path.write_text(json.dumps(questions, indent=1)) + cache_path.write_text(json.dumps(questions, indent=1), encoding="utf-8") return questions @@ -289,7 +289,7 @@ def run_policy(name: str, spec: dict, messages, questions, out_dir: Path, "summary_error": getattr(comp, "_last_summary_error", None), } out_dir.mkdir(parents=True, exist_ok=True) - (out_dir / f"{label.replace('+', '_')}.json").write_text(json.dumps({"summary": summary, "results": results}, indent=1)) + (out_dir / f"{label.replace('+', '_')}.json").write_text(json.dumps({"summary": summary, "results": results}, indent=1), encoding="utf-8") return summary @@ -333,7 +333,7 @@ def main(): "scores": scored, } out_dir.mkdir(parents=True, exist_ok=True) - (out_dir / "uncompacted_control.json").write_text(json.dumps({"summary": ctl, "results": results}, indent=1)) + (out_dir / "uncompacted_control.json").write_text(json.dumps({"summary": ctl, "results": results}, indent=1), encoding="utf-8") summaries.append(ctl) print(json.dumps(ctl, indent=1)) @@ -348,7 +348,7 @@ def main(): summaries.append(s) print(json.dumps(s, indent=1)) - (out_dir / "scorecard.json").write_text(json.dumps(summaries, indent=1)) + (out_dir / "scorecard.json").write_text(json.dumps(summaries, indent=1), encoding="utf-8") print(f"\nscorecard -> {out_dir}/scorecard.json")