From 5332232be7f8f4fd1e20d0bc0ebd67ebdade5a58 Mon Sep 17 00:00:00 2001 From: NishilRathod Date: Thu, 8 Oct 2026 12:25:16 +0530 Subject: [PATCH] fix(workflows): keep non-ASCII text readable in run artifacts The run records under .specify/workflows/runs// are meant to be human-auditable, but every writer used the serializer's ASCII-only default, so non-English text in the definition snapshot (workflow.yml), state.json, inputs.json and log.jsonl came out as \uXXXX escapes. Write them as authored: allow_unicode=True for the YAML snapshot and ensure_ascii=False for the JSON/JSONL writers, as #4148 and #4773 did for overlay files and merged settings. The JSON writers also use errors="backslashreplace", so a lone surrogate (an undecodable byte in a CLI argument) is still written as its JSON \u escape and loads back unchanged instead of failing the save. Fixes #4875 Assisted-by: Claude Code (model: Claude Opus 5.5, autonomous) Co-Authored-By: Claude Opus 5.5 --- src/specify_cli/workflows/engine.py | 19 +++++-- tests/test_workflows.py | 81 +++++++++++++++++++++++++++++ 2 files changed, 95 insertions(+), 5 deletions(-) diff --git a/src/specify_cli/workflows/engine.py b/src/specify_cli/workflows/engine.py index d7ad0fb857..cb0f3db790 100644 --- a/src/specify_cli/workflows/engine.py +++ b/src/specify_cli/workflows/engine.py @@ -784,8 +784,12 @@ def _atomic_write_json(path: Path, data: dict[str, Any]) -> None: dir=str(path.parent), prefix=f".{path.name}.", suffix=".tmp" ) try: - with os.fdopen(fd, "w", encoding="utf-8") as f: - json.dump(data, f, indent=2) + # Keep non-ASCII text readable in these human-auditable records; + # backslashreplace writes a lone surrogate as its JSON \u escape. + with os.fdopen( + fd, "w", encoding="utf-8", errors="backslashreplace" + ) as f: + json.dump(data, f, indent=2, ensure_ascii=False) os.replace(tmp, path) except BaseException: try: @@ -929,8 +933,13 @@ def append_log(self, entry: dict[str, Any]) -> None: runs_dir.mkdir(parents=True, exist_ok=True) with self._log_lock: self.log_entries.append(entry) - with open(runs_dir / "log.jsonl", "a", encoding="utf-8") as f: - f.write(json.dumps(entry) + "\n") + with open( + runs_dir / "log.jsonl", + "a", + encoding="utf-8", + errors="backslashreplace", + ) as f: + f.write(json.dumps(entry, ensure_ascii=False) + "\n") # -- Workflow Engine ------------------------------------------------------ @@ -1064,7 +1073,7 @@ def execute( workflow_copy = run_dir / "workflow.yml" import yaml with open(workflow_copy, "w", encoding="utf-8") as f: - yaml.safe_dump(definition.data, f, sort_keys=False) + yaml.safe_dump(definition.data, f, sort_keys=False, allow_unicode=True) # Resolve inputs resolved_inputs = self._resolve_inputs(definition, inputs or {}) diff --git a/tests/test_workflows.py b/tests/test_workflows.py index 7529f4595c..9d0e87d6a1 100644 --- a/tests/test_workflows.py +++ b/tests/test_workflows.py @@ -8436,6 +8436,87 @@ def test_save_and_load(self, project_dir): assert loaded.inputs == {"name": "login"} assert loaded.step_results == state.step_results + def test_run_artifacts_keep_non_ascii_text_readable(self, project_dir): + """state.json, inputs.json and log.jsonl are human-auditable run + records, so non-ASCII text must be written as authored rather than + as ``\\uXXXX`` escapes (#4875).""" + from specify_cli.workflows.engine import RunState + + text = "演示:完整闭环 — ¿aprobar? 日本語" + state = RunState( + run_id="non-ascii-run", + workflow_id="test-workflow", + project_root=project_dir, + ) + state.inputs = {"spec": text} + state.step_results = {"gate": {"output": {"message": text}}} + state.save() + state.append_log({"event": "gate_message", "message": text}) + + for name in ("state.json", "inputs.json", "log.jsonl"): + raw = (state.runs_dir / name).read_text(encoding="utf-8") + assert text in raw, name + assert "\\u" not in raw, name + + loaded = RunState.load("non-ascii-run", project_dir) + assert loaded.inputs == {"spec": text} + assert loaded.step_results == state.step_results + log_line = (state.runs_dir / "log.jsonl").read_text(encoding="utf-8") + assert json.loads(log_line)["message"] == text + + def test_run_artifacts_round_trip_a_lone_surrogate(self, project_dir): + """A lone surrogate (e.g. an undecodable byte in a CLI argument) cannot + be encoded as UTF-8. It must still save, escaped, and load back intact + instead of crashing the run.""" + from specify_cli.workflows.engine import RunState + + value = "bad byte: \udc80" + state = RunState( + run_id="surrogate-run", + workflow_id="test-workflow", + project_root=project_dir, + ) + state.inputs = {"spec": value} + state.save() + state.append_log({"event": "input", "value": value}) + + loaded = RunState.load("surrogate-run", project_dir) + assert loaded.inputs == {"spec": value} + log_line = (state.runs_dir / "log.jsonl").read_text(encoding="utf-8") + assert json.loads(log_line)["value"] == value + + def test_workflow_snapshot_keeps_non_ascii_text_readable(self, project_dir): + """The workflow.yml copied into the run directory is the definition as + authored, so its non-ASCII text must not be escaped (#4875).""" + from specify_cli.workflows.engine import WorkflowDefinition, WorkflowEngine + + name = "Demo Hello Pipeline (教学演示版)" + wf_dir = project_dir / "non-ascii-snapshot" + wf_dir.mkdir() + wf_file = wf_dir / "workflow.yml" + wf_file.write_text( + f""" +schema_version: "1.0" +workflow: + id: "non-ascii-snapshot" + name: "{name}" + version: "1.0.0" +steps: + - id: noop + type: shell + run: "echo ok" +""", + encoding="utf-8", + ) + definition = WorkflowDefinition.from_yaml(wf_file) + state = WorkflowEngine(project_dir).execute(definition) + + snapshot = state.runs_dir / "workflow.yml" + raw = snapshot.read_text(encoding="utf-8") + assert name in raw + assert "\\u" not in raw and "\\x" not in raw + assert yaml.safe_load(raw)["workflow"]["name"] == name + @pytest.mark.parametrize("invalid_step_results", [None, [], "invalid", 1, True]) def test_load_rejects_non_object_step_results( self, project_dir, invalid_step_results