Files
mcp-server/spark_executor/models.py
T
ClaudeandClaude Fable 5 0291f36b01 feat(fetch_url): per-Connection allowed_url_hosts glob allowlist
The default 'URL host shares >= 2 labels of suffix with yarn_rm_url
host' rule is too strict for clusters whose hostnames are single-label
(e.g. 'ccam1' through 'ccam99'). The user's cluster is reachable at
http://ccam1:8088, and they want fetch_url to work for any ccamN
host — but ccam1 and ccam50 share 0 suffix labels, so the existing
rule rejects everything.

Add an opt-in allowlist field on Connection:

  allowed_url_hosts: list[str] | None

Each entry is an fnmatch glob pattern. The URL host is allowed if it
matches ANY pattern, regardless of the suffix rule. Save a Connection
with ['ccam*'] to allow ccam1, ccam2, ..., ccam99.

SSRF safety: the implementation does NOT use vanilla fnmatch on the
flat host string (because '*' in fnmatch crosses '.', so 'ccam*' would
match 'ccam50.evil.com' — a security hole). Instead, both the pattern
and the host are split on '.' and matched LABEL-BY-LABEL with the
label counts required to match exactly. So 'ccam*' matches 'ccam50'
but NOT 'ccam50.evil.com' (different label counts).

Test coverage:
- 8 new tests in tests/unit/test_fetch_url.py (glob allow, glob deny,
  dot-boundary SSRF test, multiple globs, empty list, None fallback,
  error message hint)
- All 21 fetch_url tests pass; 377 total.

- spark_executor/models.py: Connection.allowed_url_hosts (with description)
- spark_executor/tools/requests.py: SaveConnectionRequest.allowed_url_hosts
- spark_executor/tools/fetch_url.py: new _host_matches_any_glob helper;
  _validate_url_host now takes allowed_hosts and checks globs BEFORE
  the suffix rule
- tests/unit/test_fetch_url.py: 8 new tests
- README.md: fetch_url row mentions the allowlist

Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-07-09 11:22:55 +08:00

131 lines
4.0 KiB
Python

# coding=utf-8
"""
@Time :2026/6/24
@Author :tao.chen
"""
from datetime import datetime
from pydantic import BaseModel, Field, field_validator
class Job(BaseModel):
job_id: str
application_id: str
script_path: str
queue: str
submit_time: datetime
connection: str
yarn_rm_url: str | None = None
class JobStatus(BaseModel):
application_id: str
state: str
raw: str = Field(default="")
class JobResult(BaseModel):
application_id: str
state: str
final_status: str | None = None
diagnostics: str | None = None
tracking_url: str | None = None
started_time: int | None = None
finished_time: int | None = None
class SubmitResult(BaseModel):
job_id: str
application_id: str
tracking_url: str | None = None
class FetchUrlResult(BaseModel):
url: str
status_code: int
content_type: str
body: str
truncated: bool = False
class Connection(BaseModel):
name: str
# Defaults to "yarn" because that's the literal string spark-submit wants
# for --master when targeting YARN. Override for Standalone (spark://...),
# Kubernetes (k8s://...), or local mode.
master: str = "yarn"
deploy_mode: str = "cluster"
yarn_rm_url: str | None = None
spark_conf: dict[str, str] = Field(default_factory=dict)
# None means "fall back to Settings.ssl_verify_default". Explicit True/False
# overrides the global default for this connection.
ssl_verify: bool | None = None
ssl_ca_bundle: str | None = None
# Authentication for YARN REST calls.
auth_type: str = "none" # "none" | "simple" | "basic" | "kerberos"
auth_user: str | None = None
auth_password: str | None = None
# Display/audit only for kerberos; actual SPNEGO uses the system cache.
auth_principal: str | None = None
auth_keytab: str | None = None
allowed_url_hosts: list[str] | None = Field(
default=None,
description=(
"Optional list of fnmatch glob patterns for hosts the fetch_url tool "
"may access, in addition to the default 'shares >= 2 labels of suffix "
"with yarn_rm_url host' rule. Useful for clusters whose hostnames do "
"NOT share a 2+ label suffix — e.g. single-label hosts like 'ccam1'-"
"'ccam99' (configure ['ccam*']) or HDFS namenode on a different "
"subdomain ('*.hadoop.internal'). Patterns are matched against the "
"URL host only (no port, no path). fnmatch rules apply: '*' does NOT "
"match '.', so 'ccam*' matches 'ccam50' but not 'ccam50.evil.com'. "
"Default None means the suffix rule alone applies."
),
)
@field_validator("master")
@classmethod
def _check_master(cls, v: str) -> str:
"""Catch common typos like 'yarn-cluster' or 'http://...'. """
if v == "yarn":
return v
if v.startswith(("spark://", "k8s://", "mesos://", "local")):
return v
raise ValueError(
f"master must be 'yarn', 'spark://...', 'k8s://...', 'mesos://...', "
f"or 'local[/N]'; got {v!r}"
)
@field_validator("auth_type")
@classmethod
def _check_auth_type(cls, v: str) -> str:
if v not in {"none", "simple", "basic", "kerberos"}:
raise ValueError(
f"auth_type must be one of none/simple/basic/kerberos; got {v!r}"
)
return v
class PendingSubmission(BaseModel):
pending_id: str
app_name: str | None = None
connection: str
master: str
deploy_mode: str
yarn_rm_url: str | None = None
script_path: str
queue: str
executor_memory: str
executor_cores: int
num_executors: int
spark_conf: dict[str, str] = Field(default_factory=dict)
extra_args: dict[str, str] = Field(default_factory=dict)
created_at: datetime
status: str = "PENDING" # PENDING | SUBMITTED | CANCELLED | FAILED
error: str | None = None
job_id: str | None = None
application_id: str | None = None
tracking_url: str | None = None