feat(job_writer): write PySpark code to disk with env-var override

spark_executor/core/job_writer.py provides write_job_file(code, jobs_dir=None)
that resolves the output directory in this order:
  1. explicit jobs_dir argument (for tests + programmatic override)
  2. SPARK_EXECUTOR_JOBS_DIR environment variable (operator override)
  3. DEFAULT_JOBS_DIR = './data/jobs' (gitignored, persists via volume mount)

Filenames are job_<UTC-stamp>_<random-hex>.py — same-second writes still
get distinct names via the random suffix, so concurrent agents won't
clobber each other.

8 tests cover: arg>env>default precedence, dir auto-creation, env-var
override, default fallback in a tmp cwd, and uniqueness under concurrent
calls.
This commit is contained in:
Claude
2026-06-25 10:17:12 +08:00
parent 516541a4df
commit 9340061b30
2 changed files with 131 additions and 0 deletions
+53
View File
@@ -0,0 +1,53 @@
# coding=utf-8
"""
@Time :2026/6/24
@Author :tao.chen
Writes LLM-generated PySpark code to disk so the agent can then call
prepare_submit_job(script_path=...) on the returned path.
Resolution order for the output directory:
1. explicit jobs_dir argument (used in tests)
2. SPARK_EXECUTOR_JOBS_DIR environment variable (operator override)
3. DEFAULT_JOBS_DIR constant (./data/jobs/ — gitignored, persists across
container restarts when ./data is mounted)
"""
import os
import secrets
from datetime import datetime
from common.logging import logger
DEFAULT_JOBS_DIR = "./data/jobs"
ENV_JOBS_DIR = "SPARK_EXECUTOR_JOBS_DIR"
def resolve_jobs_dir(jobs_dir: str | None = None) -> str:
"""Pick the effective jobs directory in priority order: arg > env > default."""
if jobs_dir is not None:
return jobs_dir
env = os.environ.get(ENV_JOBS_DIR)
if env:
return env
return DEFAULT_JOBS_DIR
def write_job_file(code: str, jobs_dir: str | None = None) -> str:
"""Write `code` to <jobs_dir>/job_<timestamp>_<rand>.py; return abs path.
Creates the directory if missing. Filename is unique per call (timestamp
down to the second + 3 random bytes) so concurrent agents cannot collide.
"""
effective_dir = resolve_jobs_dir(jobs_dir)
os.makedirs(effective_dir, exist_ok=True)
stamp = datetime.utcnow().strftime("%Y%m%d%H%M%S")
name = f"job_{stamp}_{secrets.token_hex(3)}.py"
path = os.path.join(effective_dir, name)
abs_path = os.path.abspath(path)
logger.debug(
f"write_job_file enter jobs_dir={effective_dir} code_bytes={len(code)}"
)
with open(abs_path, "w", encoding="utf-8") as f:
f.write(code)
logger.info(f"write_job_file ok script_path={abs_path} bytes={len(code)}")
return abs_path
+78
View File
@@ -0,0 +1,78 @@
# coding=utf-8
import os
from pathlib import Path
from spark_executor.core.job_writer import (
DEFAULT_JOBS_DIR,
ENV_JOBS_DIR,
resolve_jobs_dir,
write_job_file,
)
# --- resolve_jobs_dir (priority: arg > env > default) ---
def test_resolve_explicit_arg_wins(tmp_path: Path, monkeypatch):
monkeypatch.setenv(ENV_JOBS_DIR, "/from/env")
assert resolve_jobs_dir(str(tmp_path)) == str(tmp_path)
def test_resolve_env_var_used_when_no_arg(monkeypatch, tmp_path: Path):
monkeypatch.setenv(ENV_JOBS_DIR, str(tmp_path))
assert resolve_jobs_dir() == str(tmp_path)
def test_resolve_default_when_neither_set(monkeypatch):
monkeypatch.delenv(ENV_JOBS_DIR, raising=False)
assert resolve_jobs_dir() == DEFAULT_JOBS_DIR
assert resolve_jobs_dir() == "./data/jobs"
# --- write_job_file ---
def test_write_creates_file_with_code_and_returns_abs_path(tmp_path: Path):
out = write_job_file("print('hi')\n", jobs_dir=str(tmp_path))
assert os.path.isfile(out)
assert os.path.isabs(out)
assert out.startswith(str(tmp_path))
assert out.endswith(".py")
with open(out) as f:
assert f.read() == "print('hi')\n"
def test_write_creates_jobs_dir_if_missing(tmp_path: Path):
target = tmp_path / "newdir"
assert not target.exists()
out = write_job_file("x = 1\n", jobs_dir=str(target))
assert target.is_dir()
assert os.path.isfile(out)
def test_write_uses_env_var_when_no_arg(monkeypatch, tmp_path: Path):
monkeypatch.setenv(ENV_JOBS_DIR, str(tmp_path))
out = write_job_file("env-driven\n")
assert out.startswith(str(tmp_path))
with open(out) as f:
assert f.read() == "env-driven\n"
def test_write_uses_default_when_neither_set(monkeypatch, tmp_path: Path):
"""Default is ./data/jobs relative to cwd. Run in a tmp dir so the test
doesn't pollute the real project and stays self-contained."""
monkeypatch.delenv(ENV_JOBS_DIR, raising=False)
monkeypatch.chdir(tmp_path) # pytest built-in: chdir for this test only
out = write_job_file("default\n")
assert os.path.isabs(out)
# Should resolve to <tmp_path>/data/jobs/job_*.py
expected_dir = tmp_path / "data" / "jobs"
assert expected_dir.is_dir()
assert str(out).startswith(str(expected_dir))
with open(out) as f:
assert f.read() == "default\n"
def test_write_returns_unique_paths_for_concurrent_calls(tmp_path: Path):
"""Two writes in the same second must still get distinct filenames (via random suffix)."""
out1 = write_job_file("a", jobs_dir=str(tmp_path))
out2 = write_job_file("b", jobs_dir=str(tmp_path))
assert out1 != out2