fix: validate script_path exists + tell agent to call generate_job_file
Two problems with the prior prepare_submit_job flow:
1. The agent could pass a script_path that only existed in its own
context (LLM-generated code not yet on disk) or a path on the host
filesystem that's invisible inside the container. The new check
surfaces this as a 400 with a clear remediation hint instead of
letting spark-submit fail later with an opaque FileNotFoundError -> 500.
2. The container-isolation issue: any path the agent gives is interpreted
inside the container. The two ways a file can legitimately exist there
are (a) generate_job_file(code=...) just wrote it to
SPARK_EXECUTOR_JOBS_DIR (the default ./data/jobs/ is the only
gitignored dir that survives restarts), or (b) a host dir was
mounted via -v. The error message spells both out so an agent can
self-correct.
Implementation:
- submit.py: new _check_script_path() that raises ValueError (-> 400)
when the path is missing, empty, or a directory. Called in both
prepare_submit_job and confirm_submit_job (defense in depth).
- prepare_submit_job checks the connection FIRST (KeyError -> 404)
before the script (ValueError -> 400), so an agent with both problems
sees the more fundamental 'unknown connection' error first.
- server.py / requests.py: route description and Pydantic field
description spell out the generate_job_file pattern so an LLM
reading the tool schema learns the right next step.
8 new tests; 8 existing tests adjusted to create real files (they used
synthetic /tmp/*.py paths that don't exist).
This commit is contained in:
@@ -73,7 +73,13 @@ def health_check():
|
||||
"Snapshot the named Connection's master / deploy_mode / spark_conf / "
|
||||
"yarn_rm_url into a PendingSubmission record and persist it. "
|
||||
"Does NOT invoke spark-submit. Returns pending_id for use with "
|
||||
"confirm_submit_job (the user-second-confirmation step)."
|
||||
"confirm_submit_job (the user-second-confirmation step).\n\n"
|
||||
"REQUIRED PATTERN for LLM-generated code: call generate_job_file(code=...) "
|
||||
"first, then pass the returned script_path here. Direct submission with a "
|
||||
"synthetic path (one that only exists in the agent's context) will be "
|
||||
"rejected with HTTP 400 — the script must exist inside the container's "
|
||||
"filesystem. For pre-existing files, mount the host directory into the "
|
||||
"container and pass the in-container path."
|
||||
),
|
||||
)
|
||||
def _prepare_submit_job(req: PrepareSubmitJobRequest):
|
||||
|
||||
@@ -26,7 +26,18 @@ class SaveConnectionRequest(BaseModel):
|
||||
|
||||
class PrepareSubmitJobRequest(BaseModel):
|
||||
connection: str
|
||||
script_path: str
|
||||
script_path: str = Field(
|
||||
...,
|
||||
description=(
|
||||
"Absolute path to the PySpark script inside the container's "
|
||||
"filesystem. Must point at an existing regular file. For "
|
||||
"LLM-generated code, call generate_job_file(code=...) first "
|
||||
"and pass the returned script_path here. For pre-existing "
|
||||
"files on the host, mount them via a docker volume and pass "
|
||||
"the in-container path. Returns 400 with a remediation hint "
|
||||
"if the path is missing or not a file."
|
||||
),
|
||||
)
|
||||
queue: str = "default"
|
||||
executor_memory: str = "4G"
|
||||
executor_cores: int = 2
|
||||
|
||||
@@ -3,6 +3,7 @@
|
||||
@Time :2026/6/24
|
||||
@Author :tao.chen
|
||||
"""
|
||||
import os
|
||||
import secrets
|
||||
import uuid
|
||||
from datetime import datetime
|
||||
@@ -20,6 +21,27 @@ from spark_executor.core.spark_submit import (
|
||||
from spark_executor.models import Job, PendingSubmission, SubmitResult
|
||||
|
||||
|
||||
def _script_path_error(path: str) -> ValueError:
|
||||
"""Instructive error for an invalid script_path. The MCP client receives
|
||||
this via the ValueError -> 400 exception handler in server.py."""
|
||||
return ValueError(
|
||||
f"script_path does not exist or is not a file: {path!r}. "
|
||||
f"Two ways to fix this:\n"
|
||||
f" 1. (Recommended) Call generate_job_file(code=...) first to write "
|
||||
f"the PySpark source to disk, then pass the returned script_path.\n"
|
||||
f" 2. If the file already exists on the host, mount it into the "
|
||||
f"container (e.g. -v /host/path:/app/scripts:ro in docker run) and "
|
||||
f"pass the in-container path here."
|
||||
)
|
||||
|
||||
|
||||
def _check_script_path(script_path: str) -> None:
|
||||
"""Verify script_path points at an existing regular file. Raises ValueError
|
||||
(-> 400 via the FastAPI exception handler) with an instructive message."""
|
||||
if not script_path or not os.path.isfile(script_path):
|
||||
raise _script_path_error(script_path)
|
||||
|
||||
|
||||
def _new_pending_id() -> str:
|
||||
return "p_" + secrets.token_hex(6)
|
||||
|
||||
@@ -39,9 +61,23 @@ def prepare_submit_job(
|
||||
f"queue={queue} executor_memory={executor_memory} executor_cores={executor_cores} "
|
||||
f"num_executors={num_executors}"
|
||||
)
|
||||
# Order of checks matters for the error the agent sees:
|
||||
# 1. Unknown connection -> 404 (KeyError -> 404 in server.py)
|
||||
# 2. Missing script file -> 400 (ValueError -> 400)
|
||||
# Connection is checked first because it is a more fundamental problem
|
||||
# (the agent is asking about a cluster that doesn't exist), and the
|
||||
# agent shouldn't have to fix the script path only to learn the
|
||||
# connection name is wrong.
|
||||
conn = conn_store.get(connection)
|
||||
if conn is None:
|
||||
raise KeyError(f"Unknown connection: {connection}")
|
||||
# Fail fast: a non-existent path is the most common agent mistake (it
|
||||
# generated the code in its own context but forgot to call
|
||||
# generate_job_file first, or its path refers to the host filesystem
|
||||
# which is invisible inside the container). Better to surface this with
|
||||
# a 400 + clear remediation than to let spark-submit fail later with
|
||||
# an opaque FileNotFoundError -> 500.
|
||||
_check_script_path(script_path)
|
||||
|
||||
pending_id = _new_pending_id()
|
||||
pending = PendingSubmission(
|
||||
@@ -85,6 +121,10 @@ def confirm_submit_job(*, pending_id: str) -> SubmitResult:
|
||||
raise ValueError(
|
||||
f"pending_id {pending_id} is in status {pending.status!r}, not PENDING"
|
||||
)
|
||||
# Defense in depth: re-verify the script still exists. A user could
|
||||
# delete the file between prepare and confirm (or an external cleanup
|
||||
# job could remove it). 400 via the ValueError -> 400 handler.
|
||||
_check_script_path(pending.script_path)
|
||||
|
||||
cmd = build_spark_submit_command(
|
||||
master=pending.master,
|
||||
|
||||
Reference in New Issue
Block a user