refactor: drop fetch_url body cap and modernize datetime usage

- fetch_url: remove the 1MB body cap and the streaming helper. The
  manual redirect loop with allowlist re-check (the SSRF fix) is kept
  intact; the body is now read in full via resp.text. Drop the now-
  meaningless `truncated` field from FetchUrlResult and the tests that
  asserted on it. Switched from httpx.stream() back to httpx.get()
  for the redirect loop — cleaner without the body cap.

- datetime: replace deprecated datetime.utcnow() with
  datetime.now(timezone.utc) in submit.py (3 sites) and
  core/job_writer.py (1 site). Update the stale comment in
  core/pending_store.py that referenced the old call.

- Clean up an unused `from common.config import settings` import in
  submit.py that ruff flagged.

Co-Authored-By: Claude <noreply@anthropic.com>
This commit is contained in:
Claude
2026-07-09 14:35:16 +08:00
co-authored by Claude
parent 7d4e512cb0
commit deafb5a26b
7 changed files with 50 additions and 158 deletions
+16 -91
View File
@@ -13,18 +13,6 @@ from spark_executor.tools import connections, fetch_url
from spark_executor.tools.requests import ListApplicationsRequest
def _stream_cm(resp):
"""Wrap a response (or fake response) in a context manager for httpx.stream."""
class _CM:
def __enter__(self):
return resp
def __exit__(self, exc_type, exc, tb):
return False
return _CM()
def _fresh_stores(tmp_path, monkeypatch):
"""Reset connection store singletons for a single test."""
monkeypatch.setattr(connection_store, "DEFAULT_DATA_DIR", str(tmp_path))
@@ -52,14 +40,14 @@ def test_fetch_url_returns_body_and_status(fresh_stores):
)
resp = httpx.Response(200, text="hello", headers={"content-type": "text/html"})
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "prod")
assert out.url == "http://nm01.prod.internal:8042/node"
assert out.status_code == 200
assert out.content_type == "text/html"
assert out.body == "hello"
assert out.truncated is False
assert "truncated" not in fetch_url.FetchUrlResult.model_fields
assert m.call_count == 1
@@ -68,25 +56,6 @@ def test_fetch_url_raises_for_missing_connection(fresh_stores):
fetch_url.fetch_url("http://rm.prod.internal:8088/", "missing")
def test_fetch_url_truncates_body_over_1mb(fresh_stores):
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
yarn_rm_url="http://rm.prod.internal:8088",
url_allowlist=["*.prod.internal"],
)
)
big = "x" * (fetch_url._MAX_BODY_BYTES + 1)
resp = httpx.Response(200, text=big)
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
):
out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod")
assert out.truncated is True
assert len(out.body) == fetch_url._MAX_BODY_BYTES
def test_fetch_url_passes_auth_from_connection(fresh_stores):
fetch_url.conn_store.save(
Connection(
@@ -101,7 +70,7 @@ def test_fetch_url_passes_auth_from_connection(fresh_stores):
)
resp = httpx.Response(200, text="ok")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "auth")
auth = m.call_args.kwargs["auth"]
@@ -124,7 +93,7 @@ def test_fetch_url_passes_ssl_verify_from_connection(fresh_stores):
)
resp = httpx.Response(200, text="ok")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
fetch_url.fetch_url("http://nm01.prod.internal:8042/node", "insecure")
assert m.call_args.kwargs["verify"] is False
@@ -141,7 +110,7 @@ def test_fetch_url_allows_host_matching_glob_pattern(fresh_stores):
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://ccam50:8088/foo", "ccam")
assert out.status_code == 200
@@ -160,7 +129,7 @@ def test_fetch_url_allows_host_matching_any_of_multiple_globs(fresh_stores):
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url(
"http://history.prod.internal:18080/api/v1/info", "prod"
@@ -207,7 +176,7 @@ def test_fetch_url_glob_match_does_not_require_suffix_overlap(fresh_stores):
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://ccam50:8088/", "prod")
assert out.status_code == 200
@@ -225,7 +194,7 @@ def test_fetch_url_accepts_ip_literal_when_in_url_allowlist(fresh_stores):
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("http://10.0.0.1/secret", "prod")
assert out.status_code == 200
@@ -243,7 +212,7 @@ def test_fetch_url_accepts_https_when_in_url_allowlist(fresh_stores):
)
resp = httpx.Response(200, text="hello")
with patch(
"spark_executor.tools.fetch_url.httpx.stream", return_value=_stream_cm(resp)
"spark_executor.tools.fetch_url.httpx.get", return_value=resp
) as m:
out = fetch_url.fetch_url("https://ccam1.example.com/secure", "prod")
assert out.status_code == 200
@@ -301,8 +270,8 @@ def test_fetch_url_rejects_redirect_to_disallowed_host(fresh_stores):
)
redirect = httpx.Response(302, headers={"Location": "http://evil.com/"})
with patch(
"spark_executor.tools.fetch_url.httpx.stream",
return_value=_stream_cm(redirect),
"spark_executor.tools.fetch_url.httpx.get",
return_value=redirect,
) as m:
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://ccam50/foo", "ccam")
@@ -323,14 +292,14 @@ def test_fetch_url_follows_redirect_to_allowed_host(fresh_stores):
)
final = httpx.Response(200, text="ok")
with patch(
"spark_executor.tools.fetch_url.httpx.stream",
side_effect=[_stream_cm(redirect), _stream_cm(final)],
"spark_executor.tools.fetch_url.httpx.get",
side_effect=[redirect, final],
) as m:
out = fetch_url.fetch_url("http://foo.internal/", "prod")
assert out.status_code == 200
assert out.body == "ok"
assert m.call_count == 2
assert m.call_args_list[1].args[1] == "http://other.internal/"
assert m.call_args_list[1].args[0] == "http://other.internal/"
def test_fetch_url_rejects_redirect_to_ip_literal_not_in_allowlist(fresh_stores):
@@ -344,58 +313,14 @@ def test_fetch_url_rejects_redirect_to_ip_literal_not_in_allowlist(fresh_stores)
)
redirect = httpx.Response(302, headers={"Location": "http://10.0.0.1/"})
with patch(
"spark_executor.tools.fetch_url.httpx.stream",
return_value=_stream_cm(redirect),
"spark_executor.tools.fetch_url.httpx.get",
return_value=redirect,
) as m:
with pytest.raises(ValueError, match="is not in Connection.url_allowlist"):
fetch_url.fetch_url("http://ccam50/foo", "ccam")
assert m.call_count == 1
def test_fetch_url_streams_body_and_truncates_without_buffering_full(fresh_stores):
"""The response body must be read incrementally; .text must not be accessed."""
class _FakeStreamResponse:
def __init__(self, chunks):
self.status_code = 200
self.headers = httpx.Headers({"content-type": "text/plain"})
self.encoding = "utf-8"
self._chunks = chunks
def iter_bytes(self):
yield from self._chunks
@property
def text(self):
raise AssertionError("response.text should not be accessed")
fetch_url.conn_store.save(
Connection(
name="prod",
master="yarn",
url_allowlist=["*.prod.internal"],
)
)
chunk_size = 400_000
chunks = [
b"a" * chunk_size,
b"b" * chunk_size,
b"c" * chunk_size,
]
fake = _FakeStreamResponse(chunks)
with patch(
"spark_executor.tools.fetch_url.httpx.stream",
return_value=_stream_cm(fake),
) as m:
out = fetch_url.fetch_url("http://nm01.prod.internal/big", "prod")
assert m.call_count == 1
assert out.truncated is True
assert len(out.body) == fetch_url._MAX_BODY_BYTES
assert out.body.startswith("a" * chunk_size)
# Only the first 200 KB of the third chunk were consumed before truncation.
assert out.body[-200_000:] == "c" * 200_000
def test_list_applications_request_limit_bounds():
assert ListApplicationsRequest(connection_name="prod", limit=10000).limit == 10000
with pytest.raises(ValidationError):