Add a new MCP tool that lets the agent fetch URLs on the cluster's
network (YARN tracking UI, Spark History Server, NodeManager web UIs)
when the agent is on a different network and cannot reach those hosts
directly.
The MCP service runs on the YARN RM node, so it can reach every host
the cluster knows about — the agent just needs a way to ask.
Security: SSRF guard via host suffix overlap
- URL host must share >= 2 labels of suffix with the named
Connection's yarn_rm_url host (e.g. yarn_rm_url='rm.prod.internal'
allows 'http://nm01.prod.internal/...')
- IP literals (10.0.0.1, ::1) rejected
- Non-http(s) schemes (file://, gopher://, ftp://) rejected
- Connection with no yarn_rm_url cannot use this tool
- Reuses Connection.auth_for_httpx() and verify_for_httpx() so the
agent does not need cluster credentials
- Response body capped at 1 MB (truncated=true if larger)
- 30s timeout, follows redirects, loguru INFO audit log on every call
- spark_executor/tools/fetch_url.py: new tool + 2 helpers
(_host_suffix_overlap, _validate_url_host)
- spark_executor/models.py: FetchUrlResult Pydantic model
- spark_executor/tools/requests.py: FetchUrlRequest with descriptions
- spark_executor/server.py: /fetch_url route, operation_id='fetch_url'
- tests/unit/test_fetch_url.py: 13 unit tests covering all guards,
truncation, auth/SSL pass-through, redirect follow
- tests/integration/test_mcp_routes.py: assert 21 tool routes
- README.md: 1 row in Spark Executor 工具 table
Tests: 369 passed (up from 356).
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
165 lines
6.3 KiB
Python
165 lines
6.3 KiB
Python
# coding=utf-8
|
|
from unittest.mock import patch
|
|
|
|
import httpx
|
|
import pytest
|
|
|
|
from spark_executor.core import connection_store
|
|
from spark_executor.core.connection_store import ConnectionStore
|
|
from spark_executor.models import Connection
|
|
from spark_executor.tools import connections, fetch_url
|
|
|
|
|
|
def _fresh_stores(tmp_path, monkeypatch):
|
|
"""Reset connection store singletons for a single test."""
|
|
monkeypatch.setattr(connection_store, "DEFAULT_DATA_DIR", str(tmp_path))
|
|
store = ConnectionStore()
|
|
monkeypatch.setattr(connection_store, "store", store)
|
|
connections.store = store
|
|
fetch_url.conn_store = store
|
|
|
|
|
|
def test_fetch_url_returns_body_and_status(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
resp = httpx.Response(200, text="hello", headers={"content-type": "text/html"})
|
|
with patch("spark_executor.tools.fetch_url.httpx.get", return_value=resp) as m:
|
|
out = fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "prod")
|
|
assert out.url == "http://nm01.prod.internal:8042/node"
|
|
assert out.status_code == 200
|
|
assert out.content_type == "text/html"
|
|
assert out.body == "hello"
|
|
assert out.truncated is False
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_raises_for_missing_connection(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
with pytest.raises(KeyError, match="Connection not found"):
|
|
fetch_url.fetch_url("http://rm.prod.internal:8088/", "missing")
|
|
|
|
|
|
def test_fetch_url_rejects_non_http_scheme(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
with pytest.raises(ValueError, match="scheme"):
|
|
fetch_url.fetch_url("file:///etc/passwd", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_ftp_scheme(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
with pytest.raises(ValueError, match="scheme"):
|
|
fetch_url.fetch_url("ftp://rm.prod.internal/foo", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_ip_literal(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
with pytest.raises(ValueError, match="IP literal"):
|
|
fetch_url.fetch_url("http://10.0.0.1/secrets", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_ipv6_literal(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
with pytest.raises(ValueError, match="IP literal"):
|
|
fetch_url.fetch_url("http://[::1]:8080/", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_external_host(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
with pytest.raises(ValueError, match="does not share enough suffix"):
|
|
fetch_url.fetch_url("http://evil.com/foo", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_too_short_suffix_overlap(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm/8088")
|
|
)
|
|
with pytest.raises(ValueError, match="does not share enough suffix"):
|
|
fetch_url.fetch_url("http://other-rm/", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_when_connection_has_no_yarn_rm_url(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(Connection(name="prod", master="yarn", yarn_rm_url=None))
|
|
with pytest.raises(ValueError, match="no yarn_rm_url set"):
|
|
fetch_url.fetch_url("http://anything.com/", "prod")
|
|
|
|
|
|
def test_fetch_url_truncates_body_over_1mb(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
big = "x" * (fetch_url._MAX_BODY_BYTES + 1)
|
|
resp = httpx.Response(200, text=big)
|
|
with patch("spark_executor.tools.fetch_url.httpx.get", return_value=resp):
|
|
out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod")
|
|
assert out.truncated is True
|
|
assert len(out.body) == fetch_url._MAX_BODY_BYTES
|
|
|
|
|
|
def test_fetch_url_passes_auth_from_connection(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="auth",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
auth_type="basic",
|
|
auth_user="u",
|
|
auth_password="p",
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="ok")
|
|
with patch("spark_executor.tools.fetch_url.httpx.get", return_value=resp) as m:
|
|
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "auth")
|
|
auth = m.call_args.kwargs["auth"]
|
|
assert isinstance(auth, httpx.BasicAuth)
|
|
import base64
|
|
creds = base64.b64decode(auth._auth_header.split()[1]).decode()
|
|
assert creds == "u:p"
|
|
|
|
|
|
def test_fetch_url_passes_ssl_verify_from_connection(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="insecure",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
ssl_verify=False,
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="ok")
|
|
with patch("spark_executor.tools.fetch_url.httpx.get", return_value=resp) as m:
|
|
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "insecure")
|
|
assert m.call_args.kwargs["verify"] is False
|
|
|
|
|
|
def test_fetch_url_follows_redirects(tmp_path, monkeypatch):
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088")
|
|
)
|
|
resp = httpx.Response(200, text="ok")
|
|
with patch("spark_executor.tools.fetch_url.httpx.get", return_value=resp) as m:
|
|
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "prod")
|
|
assert m.call_args.kwargs["follow_redirects"] is True
|