Files
mcp-server/tests/unit/test_fetch_url.py
ClaudeandClaude deafb5a26b refactor: drop fetch_url body cap and modernize datetime usage
- fetch_url: remove the 1MB body cap and the streaming helper. The
  manual redirect loop with allowlist re-check (the SSRF fix) is kept
  intact; the body is now read in full via resp.text. Drop the now-
  meaningless `truncated` field from FetchUrlResult and the tests that
  asserted on it. Switched from httpx.stream() back to httpx.get()
  for the redirect loop — cleaner without the body cap.

- datetime: replace deprecated datetime.utcnow() with
  datetime.now(timezone.utc) in submit.py (3 sites) and
  core/job_writer.py (1 site). Update the stale comment in
  core/pending_store.py that referenced the old call.

- Clean up an unused `from common.config import settings` import in
  submit.py that ruff flagged.

Co-Authored-By: Claude <noreply@anthropic.com>
2026-07-09 14:35:16 +08:00

330 lines
11 KiB
Python

# coding=utf-8
from pathlib import Path
from unittest.mock import patch
import httpx
import pytest
from pydantic import ValidationError
from spark_executor.core import connection_store
from spark_executor.core.connection_store import ConnectionStore
from spark_executor.models import Connection
from spark_executor.tools import connections, fetch_url
from spark_executor.tools.requests import ListApplicationsRequest
def _fresh_stores(tmp_path, monkeypatch):
"""Reset connection store singletons for a single test."""
monkeypatch.setattr(connection_store, "DEFAULT_DATA_DIR", str(tmp_path))
store = ConnectionStore()
monkeypatch.setattr(connection_store, "store", store)
connections.store = store
fetch_url.conn_store = store
@pytest.fixture
def fresh_stores(tmp_path: Path, monkeypatch):
"""Reset connection store singletons for a single test."""
_fresh_stores(tmp_path, monkeypatch)
yield
def test_fetch_url_returns_body_and_status(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
url_allowlist=["*.prod.internal"],
)
)
resp = httpx.Response(200, text="hello", headers={"content-type": "text/html"})
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "prod")
assert out.url == "http://nm01.prod.internal:8042/node"
assert out.status_code == 200
assert out.content_type == "text/html"
assert out.body == "hello"
assert "truncated" not in fetch_url.FetchUrlResult.model_fields
assert m.call_count == 1
def test_fetch_url_raises_for_missing_connection(fresh_stores):
with pytest.raises(KeyError, match="Connection not found"):
fetch_url.fetch_url("http://rm.prod.internal:8088/", "missing")
def test_fetch_url_passes_auth_from_connection(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="auth",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
auth_type="basic",
auth_user="u",
auth_password="p",
url_allowlist=["*.prod.internal"],
)
)
resp = httpx.Response(200, text="ok")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "auth")
auth = m.call_args.kwargs["auth"]
assert isinstance(auth, httpx.BasicAuth)
import base64
creds = base64.b64decode(auth._auth_header.split()[1]).decode()
assert creds == "u:p"
def test_fetch_url_passes_ssl_verify_from_connection(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="insecure",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
ssl_verify=False,
url_allowlist=["*.prod.internal"],
)
)
resp = httpx.Response(200, text="ok")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "insecure")
assert m.call_args.kwargs["verify"] is False
def test_fetch_url_allows_host_matching_glob_pattern(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="ccam",
master="yarn",
yarn_rm_url="http://ccam1:8088",
url_allowlist=["ccam*"],
)
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://ccam50:8088/foo", "ccam")
assert out.status_code == 200
assert out.body == "hello"
assert m.call_count == 1
def test_fetch_url_allows_host_matching_any_of_multiple_globs(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
url_allowlist=["ccam*", "*.prod.internal"],
)
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url(
"http://history.prod.internal:18080/api/v1/info", "prod"
)
assert out.status_code == 200
assert out.body == "hello"
assert m.call_count == 1
def test_fetch_url_glob_does_not_match_unrelated_host(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="ccam",
master="yarn",
yarn_rm_url="http://ccam1:8088",
url_allowlist=["ccam*"],
)
)
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://evil.com/foo", "ccam")
def test_fetch_url_glob_does_not_cross_dot_boundary(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="ccam",
master="yarn",
yarn_rm_url="http://ccam1:8088",
url_allowlist=["ccam*"],
)
)
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://ccam50.evil.com/", "ccam")
def test_fetch_url_glob_match_does_not_require_suffix_overlap(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm:8088",
url_allowlist=["ccam*"],
)
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://ccam50:8088/", "prod")
assert out.status_code == 200
assert out.body == "hello"
assert m.call_count == 1
def test_fetch_url_accepts_ip_literal_when_in_url_allowlist(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
url_allowlist=["*.*.*.*"],
)
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://10.0.0.1/secret", "prod")
assert out.status_code == 200
assert out.body == "hello"
assert m.call_count == 1
def test_fetch_url_accepts_https_when_in_url_allowlist(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
url_allowlist=["ccam*.example.com"],
)
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("https://ccam1.example.com/secure", "prod")
assert out.status_code == 200
assert out.body == "hello"
assert m.call_count == 1
def test_fetch_url_rejects_when_url_has_no_host(fresh_stores):
with pytest.raises(ValueError, match="URL has no host"):
fetch_url._validate_url_host("", ["*"])
with pytest.raises(ValueError, match="URL has no host"):
fetch_url._validate_url_host("not-a-url", ["*"])
def test_fetch_url_rejects_when_url_allowlist_empty(fresh_stores):
fetch_url.conn_store.save(
Connection(name="prod", master="yarn", url_allowlist=[])
)
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://anything.com/", "prod")
def test_fetch_url_error_message_mentions_url_allowlist(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
)
)
with pytest.raises(ValueError, match="url_allowlist"):
fetch_url.fetch_url("http://evil.com/foo", "prod")
def test_fetch_url_omitted_url_allowlist_defaults_to_empty_and_rejects(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm.prod.internal",
)
)
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://nm.prod.internal/", "prod")
def test_fetch_url_rejects_redirect_to_disallowed_host(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="ccam",
master="yarn",
yarn_rm_url="http://ccam1:8088",
url_allowlist=["ccam*"],
)
)
redirect = httpx.Response(302, headers={"Location": "http://evil.com/"})
with patch(
"spark_executor.tools.fetch_url.httpx.get",
return_value=redirect,
) as m:
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://ccam50/foo", "ccam")
assert m.call_count == 1
def test_fetch_url_follows_redirect_to_allowed_host(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
url_allowlist=["*.internal"],
)
)
redirect = httpx.Response(
302, headers={"Location": "http://other.internal/"}
)
final = httpx.Response(200, text="ok")
with patch(
"spark_executor.tools.fetch_url.httpx.get",
side_effect=[redirect, final],
) as m:
out = fetch_url.fetch_url("http://foo.internal/", "prod")
assert out.status_code == 200
assert out.body == "ok"
assert m.call_count == 2
assert m.call_args_list[1].args[0] == "http://other.internal/"
def test_fetch_url_rejects_redirect_to_ip_literal_not_in_allowlist(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="ccam",
master="yarn",
yarn_rm_url="http://ccam1:8088",
url_allowlist=["ccam*"],
)
)
redirect = httpx.Response(302, headers={"Location": "http://10.0.0.1/"})
with patch(
"spark_executor.tools.fetch_url.httpx.get",
return_value=redirect,
) as m:
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://ccam50/foo", "ccam")
assert m.call_count == 1
def test_list_applications_request_limit_bounds():
assert ListApplicationsRequest(connection_name="prod", limit=10000).limit == 10000
with pytest.raises(ValidationError):
ListApplicationsRequest(connection_name="prod", limit=0)
with pytest.raises(ValidationError):
ListApplicationsRequest(connection_name="prod", limit=10001)