# coding=utf-8 from pathlib import Path from unittest.mock import patch import httpx import pytest from pydantic import ValidationError from spark_executor.core import connection_store from spark_executor.core.connection_store import ConnectionStore from spark_executor.models import Connection from spark_executor.tools import connections, fetch_url from spark_executor.tools.requests import ListApplicationsRequest def _stream_cm(resp): """Wrap a response (or fake response) in a context manager for httpx.stream.""" class _CM: def __enter__(self): return resp def __exit__(self, exc_type, exc, tb): return False return _CM() def _fresh_stores(tmp_path, monkeypatch): """Reset connection store singletons for a single test.""" monkeypatch.setattr(connection_store, "DEFAULT_DATA_DIR", str(tmp_path)) store = ConnectionStore() monkeypatch.setattr(connection_store, "store", store) connections.store = store fetch_url.conn_store = store @pytest.fixture def fresh_stores(tmp_path: Path, monkeypatch): """Reset connection store singletons for a single test.""" _fresh_stores(tmp_path, monkeypatch) yield def test_fetch_url_returns_body_and_status(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", url_allowlist=["*.prod.internal"], ) ) resp = httpx.Response(200, text="hello", headers={"content-type": "text/html"}) with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: out = fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "prod") assert out.url == "http://nm01.prod.internal:8042/node" assert out.status_code == 200 assert out.content_type == "text/html" assert out.body == "hello" assert out.truncated is False assert m.call_count == 1 def test_fetch_url_raises_for_missing_connection(fresh_stores): with pytest.raises(KeyError, match="Connection not found"): fetch_url.fetch_url("http://rm.prod.internal:8088/", "missing") def test_fetch_url_truncates_body_over_1mb(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", url_allowlist=["*.prod.internal"], ) ) big = "x" * (fetch_url._MAX_BODY_BYTES + 1) resp = httpx.Response(200, text=big) with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ): out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod") assert out.truncated is True assert len(out.body) == fetch_url._MAX_BODY_BYTES def test_fetch_url_passes_auth_from_connection(fresh_stores): fetch_url.conn_store.save( Connection( name="auth", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", auth_type="basic", auth_user="u", auth_password="p", url_allowlist=["*.prod.internal"], ) ) resp = httpx.Response(200, text="ok") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "auth") auth = m.call_args.kwargs["auth"] assert isinstance(auth, httpx.BasicAuth) import base64 creds = base64.b64decode(auth._auth_header.split()[1]).decode() assert creds == "u:p" def test_fetch_url_passes_ssl_verify_from_connection(fresh_stores): fetch_url.conn_store.save( Connection( name="insecure", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", ssl_verify=False, url_allowlist=["*.prod.internal"], ) ) resp = httpx.Response(200, text="ok") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "insecure") assert m.call_args.kwargs["verify"] is False def test_fetch_url_allows_host_matching_glob_pattern(fresh_stores): fetch_url.conn_store.save( Connection( name="ccam", master="yarn", yarn_rm_url="http://ccam1:8088", url_allowlist=["ccam*"], ) ) resp = httpx.Response(200, text="hello") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: out = fetch_url.fetch_url("http://ccam50:8088/foo", "ccam") assert out.status_code == 200 assert out.body == "hello" assert m.call_count == 1 def test_fetch_url_allows_host_matching_any_of_multiple_globs(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", url_allowlist=["ccam*", "*.prod.internal"], ) ) resp = httpx.Response(200, text="hello") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: out = fetch_url.fetch_url( "http://history.prod.internal:18080/api/v1/info", "prod" ) assert out.status_code == 200 assert out.body == "hello" assert m.call_count == 1 def test_fetch_url_glob_does_not_match_unrelated_host(fresh_stores): fetch_url.conn_store.save( Connection( name="ccam", master="yarn", yarn_rm_url="http://ccam1:8088", url_allowlist=["ccam*"], ) ) with pytest.raises(ValueError, match="is not in Connection.url_allowlist"): fetch_url.fetch_url("http://evil.com/foo", "ccam") def test_fetch_url_glob_does_not_cross_dot_boundary(fresh_stores): fetch_url.conn_store.save( Connection( name="ccam", master="yarn", yarn_rm_url="http://ccam1:8088", url_allowlist=["ccam*"], ) ) with pytest.raises(ValueError, match="is not in Connection.url_allowlist"): fetch_url.fetch_url("http://ccam50.evil.com/", "ccam") def test_fetch_url_glob_match_does_not_require_suffix_overlap(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm:8088", url_allowlist=["ccam*"], ) ) resp = httpx.Response(200, text="hello") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: out = fetch_url.fetch_url("http://ccam50:8088/", "prod") assert out.status_code == 200 assert out.body == "hello" assert m.call_count == 1 def test_fetch_url_accepts_ip_literal_when_in_url_allowlist(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", url_allowlist=["*.*.*.*"], ) ) resp = httpx.Response(200, text="hello") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: out = fetch_url.fetch_url("http://10.0.0.1/secret", "prod") assert out.status_code == 200 assert out.body == "hello" assert m.call_count == 1 def test_fetch_url_accepts_https_when_in_url_allowlist(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", url_allowlist=["ccam*.example.com"], ) ) resp = httpx.Response(200, text="hello") with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp) ) as m: out = fetch_url.fetch_url("https://ccam1.example.com/secure", "prod") assert out.status_code == 200 assert out.body == "hello" assert m.call_count == 1 def test_fetch_url_rejects_when_url_has_no_host(fresh_stores): with pytest.raises(ValueError, match="URL has no host"): fetch_url._validate_url_host("", ["*"]) with pytest.raises(ValueError, match="URL has no host"): fetch_url._validate_url_host("not-a-url", ["*"]) def test_fetch_url_rejects_when_url_allowlist_empty(fresh_stores): fetch_url.conn_store.save( Connection(name="prod", master="yarn", url_allowlist=[]) ) with pytest.raises(ValueError, match="is not in Connection.url_allowlist"): fetch_url.fetch_url("http://anything.com/", "prod") def test_fetch_url_error_message_mentions_url_allowlist(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", ) ) with pytest.raises(ValueError, match="url_allowlist"): fetch_url.fetch_url("http://evil.com/foo", "prod") def test_fetch_url_omitted_url_allowlist_defaults_to_empty_and_rejects(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal", ) ) with pytest.raises(ValueError, match="is not in Connection.url_allowlist"): fetch_url.fetch_url("http://nm.prod.internal/", "prod") def test_fetch_url_rejects_redirect_to_disallowed_host(fresh_stores): fetch_url.conn_store.save( Connection( name="ccam", master="yarn", yarn_rm_url="http://ccam1:8088", url_allowlist=["ccam*"], ) ) redirect = httpx.Response(302, headers={"Location": "http://evil.com/"}) with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(redirect), ) as m: with pytest.raises(ValueError, match="is not in Connection.url_allowlist"): fetch_url.fetch_url("http://ccam50/foo", "ccam") assert m.call_count == 1 def test_fetch_url_follows_redirect_to_allowed_host(fresh_stores): fetch_url.conn_store.save( Connection( name="prod", master="yarn", yarn_rm_url="http://rm.prod.internal:8088", url_allowlist=["*.internal"], ) ) redirect = httpx.Response( 302, headers={"Location": "http://other.internal/"} ) final = httpx.Response(200, text="ok") with patch( "spark_executor.tools.fetch_url.httpx.stream", side_effect=[_stream_cm(redirect), _stream_cm(final)], ) as m: out = fetch_url.fetch_url("http://foo.internal/", "prod") assert out.status_code == 200 assert out.body == "ok" assert m.call_count == 2 assert m.call_args_list[1].args[1] == "http://other.internal/" def test_fetch_url_rejects_redirect_to_ip_literal_not_in_allowlist(fresh_stores): fetch_url.conn_store.save( Connection( name="ccam", master="yarn", yarn_rm_url="http://ccam1:8088", url_allowlist=["ccam*"], ) ) redirect = httpx.Response(302, headers={"Location": "http://10.0.0.1/"}) with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(redirect), ) as m: with pytest.raises(ValueError, match="is not in Connection.url_allowlist"): fetch_url.fetch_url("http://ccam50/foo", "ccam") assert m.call_count == 1 def test_fetch_url_streams_body_and_truncates_without_buffering_full(fresh_stores): """The response body must be read incrementally; .text must not be accessed.""" class _FakeStreamResponse: def __init__(self, chunks): self.status_code = 200 self.headers = httpx.Headers({"content-type": "text/plain"}) self.encoding = "utf-8" self._chunks = chunks def iter_bytes(self): yield from self._chunks @property def text(self): raise AssertionError("response.text should not be accessed") fetch_url.conn_store.save( Connection( name="prod", master="yarn", url_allowlist=["*.prod.internal"], ) ) chunk_size = 400_000 chunks = [ b"a" * chunk_size, b"b" * chunk_size, b"c" * chunk_size, ] fake = _FakeStreamResponse(chunks) with patch( "spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(fake), ) as m: out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod") assert m.call_count == 1 assert out.truncated is True assert len(out.body) == fetch_url._MAX_BODY_BYTES assert out.body.startswith("a" * chunk_size) # Only the first 200 KB of the third chunk were consumed before truncation. assert out.body[-200_000:] == "c" * 200_000 def test_list_applications_request_limit_bounds(): assert ListApplicationsRequest(connection_name="prod", limit=10000).limit == 10000 with pytest.raises(ValidationError): ListApplicationsRequest(connection_name="prod", limit=0) with pytest.raises(ValidationError): ListApplicationsRequest(connection_name="prod", limit=10001)