Add a new MCP tool that lets the agent fetch URLs on the cluster's
network (YARN tracking UI, Spark History Server, NodeManager web UIs)
when the agent is on a different network and cannot reach those hosts
directly.
The MCP service runs on the YARN RM node, so it can reach every host
the cluster knows about — the agent just needs a way to ask.
Security: SSRF guard via host suffix overlap
- URL host must share >= 2 labels of suffix with the named
Connection's yarn_rm_url host (e.g. yarn_rm_url='rm.prod.internal'
allows 'http://nm01.prod.internal/...')
- IP literals (10.0.0.1, ::1) rejected
- Non-http(s) schemes (file://, gopher://, ftp://) rejected
- Connection with no yarn_rm_url cannot use this tool
- Reuses Connection.auth_for_httpx() and verify_for_httpx() so the
agent does not need cluster credentials
- Response body capped at 1 MB (truncated=true if larger)
- 30s timeout, follows redirects, loguru INFO audit log on every call
- spark_executor/tools/fetch_url.py: new tool + 2 helpers
(_host_suffix_overlap, _validate_url_host)
- spark_executor/models.py: FetchUrlResult Pydantic model
- spark_executor/tools/requests.py: FetchUrlRequest with descriptions
- spark_executor/server.py: /fetch_url route, operation_id='fetch_url'
- tests/unit/test_fetch_url.py: 13 unit tests covering all guards,
truncation, auth/SSL pass-through, redirect follow
- tests/integration/test_mcp_routes.py: assert 21 tool routes
- README.md: 1 row in Spark Executor 工具 table
Tests: 369 passed (up from 356).
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
109 lines
3.6 KiB
Python
109 lines
3.6 KiB
Python
# coding=utf-8
|
|
"""
|
|
@Time :2026/7/9
|
|
@Author :tao.chen
|
|
|
|
Generic HTTP GET proxy for the agent. Lets the agent fetch URLs on the
|
|
cluster's network when it cannot reach those hosts directly. Security:
|
|
URL host must share >= 2 labels of suffix with the connection's yarn_rm_url
|
|
host; IP literals and non-HTTP schemes are rejected. Reuses the connection's
|
|
saved auth/SSL config so the agent doesn't need cluster credentials.
|
|
"""
|
|
import ipaddress
|
|
from urllib.parse import urlparse
|
|
|
|
import httpx
|
|
|
|
from common.logging import logger
|
|
from spark_executor.core.yarn_client import YarnClientConfig
|
|
from spark_executor.models import FetchUrlResult
|
|
from spark_executor.tools.connections import store as conn_store
|
|
|
|
_MAX_BODY_BYTES = 1_000_000 # 1 MB cap on response body
|
|
_REQUEST_TIMEOUT_SECONDS = 30
|
|
|
|
|
|
def _host_suffix_overlap(host_a: str, host_b: str, min_labels: int = 2) -> bool:
|
|
"""Return True if host_a and host_b share at least min_labels suffix labels."""
|
|
labels_a = host_a.lower().split(".")
|
|
labels_b = host_b.lower().split(".")
|
|
n = 0
|
|
i, j = len(labels_a) - 1, len(labels_b) - 1
|
|
while i >= 0 and j >= 0 and labels_a[i] == labels_b[j]:
|
|
n += 1
|
|
i -= 1
|
|
j -= 1
|
|
return n >= min_labels
|
|
|
|
|
|
def _validate_url_host(url: str, yarn_rm_url: str | None) -> None:
|
|
"""Validate that url is an allowed http(s) hostname on the cluster network."""
|
|
parsed = urlparse(url)
|
|
if parsed.scheme not in ("http", "https"):
|
|
raise ValueError(f"URL scheme must be http or https, got {parsed.scheme!r}")
|
|
|
|
host = parsed.hostname
|
|
if not host:
|
|
raise ValueError(f"URL has no host: {url!r}")
|
|
|
|
try:
|
|
ipaddress.ip_address(host)
|
|
except ValueError:
|
|
pass
|
|
else:
|
|
raise ValueError(
|
|
f"URL host {host!r} is an IP literal — IP targets are not allowed. "
|
|
"Use a hostname on the cluster network."
|
|
)
|
|
|
|
if not yarn_rm_url:
|
|
raise ValueError(
|
|
"Connection has no yarn_rm_url set, cannot validate URL host. "
|
|
"Save a Connection with yarn_rm_url first."
|
|
)
|
|
|
|
anchor = urlparse(yarn_rm_url).hostname or ""
|
|
if not _host_suffix_overlap(host, anchor, min_labels=2):
|
|
raise ValueError(
|
|
f"URL host {host!r} does not share enough suffix with the connection's "
|
|
f"yarn_rm_url host {anchor!r} (need >= 2 labels of common suffix). "
|
|
"Reject this fetch to prevent SSRF."
|
|
)
|
|
|
|
|
|
def fetch_url(url: str, connection_name: str) -> FetchUrlResult:
|
|
"""Proxy an HTTP GET to url using the auth/SSL settings of connection_name."""
|
|
logger.debug(f"fetch_url enter url={url} connection_name={connection_name}")
|
|
conn = conn_store.get(connection_name)
|
|
if conn is None:
|
|
raise KeyError(f"Connection not found: {connection_name}")
|
|
|
|
_validate_url_host(url, conn.yarn_rm_url)
|
|
|
|
config = YarnClientConfig.from_connection(conn)
|
|
resp = httpx.get(
|
|
url,
|
|
auth=config.auth_for_httpx(),
|
|
verify=config.verify_for_httpx(),
|
|
timeout=_REQUEST_TIMEOUT_SECONDS,
|
|
follow_redirects=True,
|
|
)
|
|
|
|
body = resp.text[:_MAX_BODY_BYTES]
|
|
truncated = len(resp.text) > _MAX_BODY_BYTES
|
|
|
|
logger.info(
|
|
f"fetch_url ok url={url} connection_name={connection_name} "
|
|
f"status_code={resp.status_code} "
|
|
f"content_type={resp.headers.get('content-type', '')} "
|
|
f"body_bytes={len(body)} truncated={truncated}"
|
|
)
|
|
|
|
return FetchUrlResult(
|
|
url=url,
|
|
status_code=resp.status_code,
|
|
content_type=resp.headers.get("content-type", ""),
|
|
body=body,
|
|
truncated=truncated,
|
|
)
|