Add per-Connection ssl_verify / ssl_ca_bundle plus global defaults so CDH 5 / on-prem clusters with self-signed certs or custom CA bundles can be queried without patching code. - Connection gets ssl_verify (bool|None) and ssl_ca_bundle (str|None) - Settings gets ssl_verify_default and ssl_ca_bundle_default - New YarnClientConfig dataclass carries the resolved verify= value - _request passes verify= through to httpx.request - All public yarn_client functions now take YarnClientConfig instead of a bare yarn_rm_url string; tool call sites resolve the Connection - SaveConnectionRequest exposes the two new fields Tests cover per-connection CA bundle, per-connection verify=False, global default fallback, and connection-not-found error.
104 lines
2.9 KiB
Python
104 lines
2.9 KiB
Python
# coding=utf-8
|
|
"""
|
|
@Time :2026/6/24
|
|
@Author :tao.chen
|
|
|
|
Pydantic request models for the FastAPI route layer. The underlying tool
|
|
functions in tools/*.py still take keyword arguments; these models exist only
|
|
so fastapi-mcp can call the routes via tools/call (which sends args as a
|
|
JSON body) without 422-ing on dict-typed parameters like spark_conf.
|
|
"""
|
|
from pydantic import BaseModel, Field
|
|
|
|
|
|
class EmptyRequest(BaseModel):
|
|
"""Used for tools that take no arguments (list_connections, list_pending_jobs)."""
|
|
pass
|
|
|
|
|
|
class SaveConnectionRequest(BaseModel):
|
|
name: str
|
|
master: str
|
|
deploy_mode: str = "cluster"
|
|
yarn_rm_url: str | None = None
|
|
spark_conf: dict[str, str] | None = None
|
|
ssl_verify: bool | None = None
|
|
ssl_ca_bundle: str | None = None
|
|
|
|
|
|
class PrepareSubmitJobRequest(BaseModel):
|
|
connection: str
|
|
script_path: str = Field(
|
|
...,
|
|
description=(
|
|
"Absolute path to the PySpark script inside the container's "
|
|
"filesystem. Must point at an existing regular file. For "
|
|
"LLM-generated code, call generate_job_file(code=...) first "
|
|
"and pass the returned script_path here. For pre-existing "
|
|
"files on the host, mount them via a docker volume and pass "
|
|
"the in-container path. Returns 400 with a remediation hint "
|
|
"if the path is missing or not a file."
|
|
),
|
|
)
|
|
queue: str = "default"
|
|
executor_memory: str = "4G"
|
|
executor_cores: int = 2
|
|
num_executors: int = 2
|
|
|
|
|
|
class PendingIdRequest(BaseModel):
|
|
pending_id: str
|
|
|
|
|
|
class JobIdRequest(BaseModel):
|
|
job_id: str
|
|
|
|
|
|
class GetJobLogsRequest(BaseModel):
|
|
job_id: str
|
|
tail_chars: int = 5000
|
|
|
|
|
|
class ConnectionNameRequest(BaseModel):
|
|
name: str
|
|
|
|
|
|
class GenerateJobFileRequest(BaseModel):
|
|
code: str = Field(
|
|
...,
|
|
description=(
|
|
"Full PySpark source code to write to disk. Will be passed verbatim "
|
|
"to spark-submit after the agent calls prepare_submit_job on the "
|
|
"returned path."
|
|
),
|
|
)
|
|
|
|
|
|
class ReadJobFileRequest(BaseModel):
|
|
script_path: str = Field(
|
|
...,
|
|
description=(
|
|
"Absolute path to a PySpark script inside the container's "
|
|
"filesystem. Must point at an existing regular file."
|
|
),
|
|
)
|
|
|
|
|
|
class UpdateJobFileRequest(BaseModel):
|
|
script_path: str = Field(
|
|
...,
|
|
description=(
|
|
"Absolute path to an existing PySpark script inside the "
|
|
"container's filesystem. Must be under SPARK_EXECUTOR_JOBS_DIR "
|
|
"(the same dir generate_job_file writes to) — protects against "
|
|
"overwriting host-mounted configs or other critical files."
|
|
),
|
|
)
|
|
content: str = Field(
|
|
...,
|
|
description=(
|
|
"New file content (replaces the file in full; no merge/diff). "
|
|
"Maximum 1 MB to keep the MCP response bounded."
|
|
),
|
|
)
|