Add per-Connection authentication so CDH 5 / Kerberos / HTTP Basic clusters can be queried. - auth_type: none / simple / basic / kerberos - auth_user / auth_password for HTTP Basic - auth_principal / auth_keytab stored for audit/display; actual SPNEGO handled by httpx-kerberos using the system Kerberos credential cache - YarnClientConfig.auth_for_httpx() returns the right httpx.Auth object - _request passes auth= through to httpx.request alongside verify= New dependency: httpx-kerberos. Tests cover none/simple (no auth object), Basic auth header, Kerberos auth object, missing basic user, invalid auth_type, and tool-layer config propagation.
109 lines
3.1 KiB
Python
109 lines
3.1 KiB
Python
# coding=utf-8
|
|
"""
|
|
@Time :2026/6/24
|
|
@Author :tao.chen
|
|
|
|
Pydantic request models for the FastAPI route layer. The underlying tool
|
|
functions in tools/*.py still take keyword arguments; these models exist only
|
|
so fastapi-mcp can call the routes via tools/call (which sends args as a
|
|
JSON body) without 422-ing on dict-typed parameters like spark_conf.
|
|
"""
|
|
from pydantic import BaseModel, Field
|
|
|
|
|
|
class EmptyRequest(BaseModel):
|
|
"""Used for tools that take no arguments (list_connections, list_pending_jobs)."""
|
|
pass
|
|
|
|
|
|
class SaveConnectionRequest(BaseModel):
|
|
name: str
|
|
master: str
|
|
deploy_mode: str = "cluster"
|
|
yarn_rm_url: str | None = None
|
|
spark_conf: dict[str, str] | None = None
|
|
ssl_verify: bool | None = None
|
|
ssl_ca_bundle: str | None = None
|
|
auth_type: str = "none"
|
|
auth_user: str | None = None
|
|
auth_password: str | None = None
|
|
auth_principal: str | None = None
|
|
auth_keytab: str | None = None
|
|
|
|
|
|
class PrepareSubmitJobRequest(BaseModel):
|
|
connection: str
|
|
script_path: str = Field(
|
|
...,
|
|
description=(
|
|
"Absolute path to the PySpark script inside the container's "
|
|
"filesystem. Must point at an existing regular file. For "
|
|
"LLM-generated code, call generate_job_file(code=...) first "
|
|
"and pass the returned script_path here. For pre-existing "
|
|
"files on the host, mount them via a docker volume and pass "
|
|
"the in-container path. Returns 400 with a remediation hint "
|
|
"if the path is missing or not a file."
|
|
),
|
|
)
|
|
queue: str = "default"
|
|
executor_memory: str = "4G"
|
|
executor_cores: int = 2
|
|
num_executors: int = 2
|
|
|
|
|
|
class PendingIdRequest(BaseModel):
|
|
pending_id: str
|
|
|
|
|
|
class JobIdRequest(BaseModel):
|
|
job_id: str
|
|
|
|
|
|
class GetJobLogsRequest(BaseModel):
|
|
job_id: str
|
|
tail_chars: int = 5000
|
|
|
|
|
|
class ConnectionNameRequest(BaseModel):
|
|
name: str
|
|
|
|
|
|
class GenerateJobFileRequest(BaseModel):
|
|
code: str = Field(
|
|
...,
|
|
description=(
|
|
"Full PySpark source code to write to disk. Will be passed verbatim "
|
|
"to spark-submit after the agent calls prepare_submit_job on the "
|
|
"returned path."
|
|
),
|
|
)
|
|
|
|
|
|
class ReadJobFileRequest(BaseModel):
|
|
script_path: str = Field(
|
|
...,
|
|
description=(
|
|
"Absolute path to a PySpark script inside the container's "
|
|
"filesystem. Must point at an existing regular file."
|
|
),
|
|
)
|
|
|
|
|
|
class UpdateJobFileRequest(BaseModel):
|
|
script_path: str = Field(
|
|
...,
|
|
description=(
|
|
"Absolute path to an existing PySpark script inside the "
|
|
"container's filesystem. Must be under SPARK_EXECUTOR_JOBS_DIR "
|
|
"(the same dir generate_job_file writes to) — protects against "
|
|
"overwriting host-mounted configs or other critical files."
|
|
),
|
|
)
|
|
content: str = Field(
|
|
...,
|
|
description=(
|
|
"New file content (replaces the file in full; no merge/diff). "
|
|
"Maximum 1 MB to keep the MCP response bounded."
|
|
),
|
|
)
|