# coding=utf-8 """ @Time :2026/6/24 @Author :tao.chen Writes LLM-generated PySpark code to disk so the agent can then call prepare_submit_job(script_path=...) on the returned path. Resolution order for the output directory: 1. explicit jobs_dir argument (used in tests) 2. SPARK_EXECUTOR_JOBS_DIR environment variable (operator override) 3. DEFAULT_JOBS_DIR constant (./data/jobs/ — gitignored, persists across container restarts when ./data is mounted) """ import os import secrets from datetime import datetime from common.config import settings from common.logging import logger ENV_JOBS_DIR = "SPARK_EXECUTOR_JOBS_DIR" def resolve_jobs_dir(jobs_dir: str | None = None) -> str: """Pick the effective jobs directory in priority order: arg > settings.jobs_dir. `settings.jobs_dir` itself is computed in `common/config.py` as: SPARK_EXECUTOR_JOBS_DIR env var (if set) > /jobs """ if jobs_dir is not None: return jobs_dir return settings.jobs_dir def write_job_file(code: str, jobs_dir: str | None = None) -> str: """Write `code` to /job__.py; return abs path. Creates the directory if missing. Filename is unique per call (timestamp down to the second + 3 random bytes) so concurrent agents cannot collide. """ effective_dir = resolve_jobs_dir(jobs_dir) os.makedirs(effective_dir, exist_ok=True) stamp = datetime.utcnow().strftime("%Y%m%d%H%M%S") name = f"job_{stamp}_{secrets.token_hex(3)}.py" path = os.path.join(effective_dir, name) abs_path = os.path.abspath(path) logger.debug( f"write_job_file enter jobs_dir={effective_dir} code_bytes={len(code)}" ) with open(abs_path, "w", encoding="utf-8") as f: f.write(code) logger.info(f"write_job_file ok script_path={abs_path} bytes={len(code)}") return abs_path