- fetch_url: revalidate allowlist on every redirect hop (fixes SSRF where 302 to disallowed host / 169.254.169.254 / file:// bypassed the url_allowlist). Stream response body with iter_bytes and cap at 1MB so a multi-GB response from an allowlisted host cannot OOM the service. Reuses the manual-redirect-loop pattern from yarn_client. - list_applications: stop swallowing 404 (YARN returns 200+empty for "no match"; 404 means the RM doesn't support the endpoint — surface the YarnError instead of hiding it as an empty result). Add Field(ge=1, le=10000) to ListApplicationsRequest.limit so a runaway limit is rejected at the Pydantic layer with 422. - save_connection: PATCH semantics for existing records. Re-route to update_connection when the name already exists so partial updates (e.g. only master) no longer wipe url_allowlist back to []. Uses an _UNSET sentinel in the tool function to distinguish "omitted" from "None" without breaking the existing parameter list. - README: drop leading space on 5 new connection-tool table rows that was breaking GitHub Flavored Markdown table continuity. - Indentation: normalize connections.py and requests.py to 4-space indent (auth_password/auth_principal/auth_keytab were 3-space). Co-Authored-By: Claude <noreply@anthropic.com>
405 lines
13 KiB
Python
405 lines
13 KiB
Python
# coding=utf-8
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
import httpx
|
|
import pytest
|
|
from pydantic import ValidationError
|
|
|
|
from spark_executor.core import connection_store
|
|
from spark_executor.core.connection_store import ConnectionStore
|
|
from spark_executor.models import Connection
|
|
from spark_executor.tools import connections, fetch_url
|
|
from spark_executor.tools.requests import ListApplicationsRequest
|
|
|
|
|
|
def _stream_cm(resp):
|
|
"""Wrap a response (or fake response) in a context manager for httpx.stream."""
|
|
class _CM:
|
|
def __enter__(self):
|
|
return resp
|
|
|
|
def __exit__(self, exc_type, exc, tb):
|
|
return False
|
|
|
|
return _CM()
|
|
|
|
|
|
def _fresh_stores(tmp_path, monkeypatch):
|
|
"""Reset connection store singletons for a single test."""
|
|
monkeypatch.setattr(connection_store, "DEFAULT_DATA_DIR", str(tmp_path))
|
|
store = ConnectionStore()
|
|
monkeypatch.setattr(connection_store, "store", store)
|
|
connections.store = store
|
|
fetch_url.conn_store = store
|
|
|
|
|
|
@pytest.fixture
|
|
def fresh_stores(tmp_path: Path, monkeypatch):
|
|
"""Reset connection store singletons for a single test."""
|
|
_fresh_stores(tmp_path, monkeypatch)
|
|
yield
|
|
|
|
|
|
def test_fetch_url_returns_body_and_status(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
url_allowlist=["*.prod.internal"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="hello", headers={"content-type": "text/html"})
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
out = fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "prod")
|
|
assert out.url == "http://nm01.prod.internal:8042/node"
|
|
assert out.status_code == 200
|
|
assert out.content_type == "text/html"
|
|
assert out.body == "hello"
|
|
assert out.truncated is False
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_raises_for_missing_connection(fresh_stores):
|
|
with pytest.raises(KeyError, match="Connection not found"):
|
|
fetch_url.fetch_url("http://rm.prod.internal:8088/", "missing")
|
|
|
|
|
|
def test_fetch_url_truncates_body_over_1mb(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
url_allowlist=["*.prod.internal"],
|
|
)
|
|
)
|
|
big = "x" * (fetch_url._MAX_BODY_BYTES + 1)
|
|
resp = httpx.Response(200, text=big)
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
):
|
|
out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod")
|
|
assert out.truncated is True
|
|
assert len(out.body) == fetch_url._MAX_BODY_BYTES
|
|
|
|
|
|
def test_fetch_url_passes_auth_from_connection(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="auth",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
auth_type="basic",
|
|
auth_user="u",
|
|
auth_password="p",
|
|
url_allowlist=["*.prod.internal"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="ok")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "auth")
|
|
auth = m.call_args.kwargs["auth"]
|
|
assert isinstance(auth, httpx.BasicAuth)
|
|
import base64
|
|
|
|
creds = base64.b64decode(auth._auth_header.split()[1]).decode()
|
|
assert creds == "u:p"
|
|
|
|
|
|
def test_fetch_url_passes_ssl_verify_from_connection(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="insecure",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
ssl_verify=False,
|
|
url_allowlist=["*.prod.internal"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="ok")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "insecure")
|
|
assert m.call_args.kwargs["verify"] is False
|
|
|
|
|
|
def test_fetch_url_allows_host_matching_glob_pattern(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="ccam",
|
|
master="yarn",
|
|
yarn_rm_url="http://ccam1:8088",
|
|
url_allowlist=["ccam*"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="hello")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
out = fetch_url.fetch_url("http://ccam50:8088/foo", "ccam")
|
|
assert out.status_code == 200
|
|
assert out.body == "hello"
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_allows_host_matching_any_of_multiple_globs(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
url_allowlist=["ccam*", "*.prod.internal"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="hello")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
out = fetch_url.fetch_url(
|
|
"http://history.prod.internal:18080/api/v1/info", "prod"
|
|
)
|
|
assert out.status_code == 200
|
|
assert out.body == "hello"
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_glob_does_not_match_unrelated_host(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="ccam",
|
|
master="yarn",
|
|
yarn_rm_url="http://ccam1:8088",
|
|
url_allowlist=["ccam*"],
|
|
)
|
|
)
|
|
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
|
|
fetch_url.fetch_url("http://evil.com/foo", "ccam")
|
|
|
|
|
|
def test_fetch_url_glob_does_not_cross_dot_boundary(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="ccam",
|
|
master="yarn",
|
|
yarn_rm_url="http://ccam1:8088",
|
|
url_allowlist=["ccam*"],
|
|
)
|
|
)
|
|
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
|
|
fetch_url.fetch_url("http://ccam50.evil.com/", "ccam")
|
|
|
|
|
|
def test_fetch_url_glob_match_does_not_require_suffix_overlap(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm:8088",
|
|
url_allowlist=["ccam*"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="hello")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
out = fetch_url.fetch_url("http://ccam50:8088/", "prod")
|
|
assert out.status_code == 200
|
|
assert out.body == "hello"
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_accepts_ip_literal_when_in_url_allowlist(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
url_allowlist=["*.*.*.*"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="hello")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
out = fetch_url.fetch_url("http://10.0.0.1/secret", "prod")
|
|
assert out.status_code == 200
|
|
assert out.body == "hello"
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_accepts_https_when_in_url_allowlist(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
url_allowlist=["ccam*.example.com"],
|
|
)
|
|
)
|
|
resp = httpx.Response(200, text="hello")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
|
|
) as m:
|
|
out = fetch_url.fetch_url("https://ccam1.example.com/secure", "prod")
|
|
assert out.status_code == 200
|
|
assert out.body == "hello"
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_rejects_when_url_has_no_host(fresh_stores):
|
|
with pytest.raises(ValueError, match="URL has no host"):
|
|
fetch_url._validate_url_host("", ["*"])
|
|
with pytest.raises(ValueError, match="URL has no host"):
|
|
fetch_url._validate_url_host("not-a-url", ["*"])
|
|
|
|
|
|
def test_fetch_url_rejects_when_url_allowlist_empty(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(name="prod", master="yarn", url_allowlist=[])
|
|
)
|
|
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
|
|
fetch_url.fetch_url("http://anything.com/", "prod")
|
|
|
|
|
|
def test_fetch_url_error_message_mentions_url_allowlist(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
)
|
|
)
|
|
with pytest.raises(ValueError, match="url_allowlist"):
|
|
fetch_url.fetch_url("http://evil.com/foo", "prod")
|
|
|
|
|
|
def test_fetch_url_omitted_url_allowlist_defaults_to_empty_and_rejects(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal",
|
|
)
|
|
)
|
|
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
|
|
fetch_url.fetch_url("http://nm.prod.internal/", "prod")
|
|
|
|
|
|
def test_fetch_url_rejects_redirect_to_disallowed_host(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="ccam",
|
|
master="yarn",
|
|
yarn_rm_url="http://ccam1:8088",
|
|
url_allowlist=["ccam*"],
|
|
)
|
|
)
|
|
redirect = httpx.Response(302, headers={"Location": "http://evil.com/"})
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream",
|
|
return_value=_stream_cm(redirect),
|
|
) as m:
|
|
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
|
|
fetch_url.fetch_url("http://ccam50/foo", "ccam")
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_follows_redirect_to_allowed_host(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
yarn_rm_url="http://rm.prod.internal:8088",
|
|
url_allowlist=["*.internal"],
|
|
)
|
|
)
|
|
redirect = httpx.Response(
|
|
302, headers={"Location": "http://other.internal/"}
|
|
)
|
|
final = httpx.Response(200, text="ok")
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream",
|
|
side_effect=[_stream_cm(redirect), _stream_cm(final)],
|
|
) as m:
|
|
out = fetch_url.fetch_url("http://foo.internal/", "prod")
|
|
assert out.status_code == 200
|
|
assert out.body == "ok"
|
|
assert m.call_count == 2
|
|
assert m.call_args_list[1].args[1] == "http://other.internal/"
|
|
|
|
|
|
def test_fetch_url_rejects_redirect_to_ip_literal_not_in_allowlist(fresh_stores):
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="ccam",
|
|
master="yarn",
|
|
yarn_rm_url="http://ccam1:8088",
|
|
url_allowlist=["ccam*"],
|
|
)
|
|
)
|
|
redirect = httpx.Response(302, headers={"Location": "http://10.0.0.1/"})
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream",
|
|
return_value=_stream_cm(redirect),
|
|
) as m:
|
|
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
|
|
fetch_url.fetch_url("http://ccam50/foo", "ccam")
|
|
assert m.call_count == 1
|
|
|
|
|
|
def test_fetch_url_streams_body_and_truncates_without_buffering_full(fresh_stores):
|
|
"""The response body must be read incrementally; .text must not be accessed."""
|
|
|
|
class _FakeStreamResponse:
|
|
def __init__(self, chunks):
|
|
self.status_code = 200
|
|
self.headers = httpx.Headers({"content-type": "text/plain"})
|
|
self.encoding = "utf-8"
|
|
self._chunks = chunks
|
|
|
|
def iter_bytes(self):
|
|
yield from self._chunks
|
|
|
|
@property
|
|
def text(self):
|
|
raise AssertionError("response.text should not be accessed")
|
|
|
|
fetch_url.conn_store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
url_allowlist=["*.prod.internal"],
|
|
)
|
|
)
|
|
chunk_size = 400_000
|
|
chunks = [
|
|
b"a" * chunk_size,
|
|
b"b" * chunk_size,
|
|
b"c" * chunk_size,
|
|
]
|
|
fake = _FakeStreamResponse(chunks)
|
|
with patch(
|
|
"spark_executor.tools.fetch_url.httpx.stream",
|
|
return_value=_stream_cm(fake),
|
|
) as m:
|
|
out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod")
|
|
assert m.call_count == 1
|
|
assert out.truncated is True
|
|
assert len(out.body) == fetch_url._MAX_BODY_BYTES
|
|
assert out.body.startswith("a" * chunk_size)
|
|
# Only the first 200 KB of the third chunk were consumed before truncation.
|
|
assert out.body[-200_000:] == "c" * 200_000
|
|
|
|
|
|
def test_list_applications_request_limit_bounds():
|
|
assert ListApplicationsRequest(connection_name="prod", limit=10000).limit == 10000
|
|
with pytest.raises(ValidationError):
|
|
ListApplicationsRequest(connection_name="prod", limit=0)
|
|
with pytest.raises(ValidationError):
|
|
ListApplicationsRequest(connection_name="prod", limit=10001)
|