feat(result): get_job_result tool for terminal view of Spark job
`get_job_status` returns YARN state + the raw response blob, so the terminal fields (finalStatus, diagnostics, trackingUrl, startedTime, finishedTime) are buried inside `raw` and not surfaced in a structured form. Add a new tool that parses them. `get_job_result(job_id)` reuses `yarn_client.get_application_status` and extracts: - finalStatus (SUCCEEDED / FAILED / KILLED / UNDEFINED) - diagnostics (YARN final message) - tracking_url (Spark Web UI) - started_time / finished_time (epoch ms) All five fields are optional: running jobs have no `finishedTime`, and older YARN versions (CDH 5 / H2) may omit some fields. Missing fields stay None — never raise. Coexistence with `get_job_status` is intentional: the latter is for polling the running YARN state, the former is the terminal view. Tests: 4 cases — happy path, running job (no finishedTime), bare-minimum raw (all optionals None), unknown job_id raises KeyError. uv run pytest -> 171 passed.
This commit is contained in:
@@ -0,0 +1,123 @@
|
||||
# coding=utf-8
|
||||
from datetime import datetime
|
||||
from unittest.mock import patch
|
||||
|
||||
import pytest
|
||||
|
||||
from spark_executor.core.job_store import JobStore
|
||||
from spark_executor.models import Job
|
||||
from spark_executor.tools import result
|
||||
|
||||
|
||||
def test_result_returns_parsed_fields():
|
||||
result.store = JobStore()
|
||||
result.store.put(
|
||||
Job(
|
||||
job_id="abc",
|
||||
application_id="application_1",
|
||||
script_path="/tmp/j.py",
|
||||
queue="default",
|
||||
submit_time=datetime(2026, 6, 24),
|
||||
connection="prod",
|
||||
yarn_rm_url="http://rm:8088",
|
||||
)
|
||||
)
|
||||
raw = {
|
||||
"app": {
|
||||
"id": "application_1",
|
||||
"state": "FINISHED",
|
||||
"finalStatus": "SUCCEEDED",
|
||||
"diagnostics": "Application completed successfully",
|
||||
"trackingUrl": "http://nm:8088/proxy/application_1",
|
||||
"startedTime": 1700000000000,
|
||||
"finishedTime": 1700000123000,
|
||||
}
|
||||
}
|
||||
import json
|
||||
|
||||
with patch(
|
||||
"spark_executor.tools.result.get_application_status",
|
||||
return_value=("FINISHED", json.dumps(raw)),
|
||||
) as m:
|
||||
out = result.get_job_result("abc")
|
||||
|
||||
assert out.application_id == "application_1"
|
||||
assert out.state == "FINISHED"
|
||||
assert out.final_status == "SUCCEEDED"
|
||||
assert out.diagnostics == "Application completed successfully"
|
||||
assert out.tracking_url == "http://nm:8088/proxy/application_1"
|
||||
assert out.started_time == 1700000000000
|
||||
assert out.finished_time == 1700000123000
|
||||
assert m.call_args.args == ("application_1", "http://rm:8088")
|
||||
|
||||
|
||||
def test_result_handles_running_job():
|
||||
result.store = JobStore()
|
||||
result.store.put(
|
||||
Job(
|
||||
job_id="running",
|
||||
application_id="application_2",
|
||||
script_path="/tmp/j.py",
|
||||
queue="default",
|
||||
submit_time=datetime(2026, 6, 24),
|
||||
connection="prod",
|
||||
yarn_rm_url="http://rm:8088",
|
||||
)
|
||||
)
|
||||
raw = {
|
||||
"app": {
|
||||
"id": "application_2",
|
||||
"state": "RUNNING",
|
||||
"finalStatus": "UNDEFINED",
|
||||
"trackingUrl": "http://rm:8088/proxy/application_2",
|
||||
"startedTime": 1700000000000,
|
||||
}
|
||||
}
|
||||
import json
|
||||
|
||||
with patch(
|
||||
"spark_executor.tools.result.get_application_status",
|
||||
return_value=("RUNNING", json.dumps(raw)),
|
||||
):
|
||||
out = result.get_job_result("running")
|
||||
|
||||
assert out.state == "RUNNING"
|
||||
assert out.final_status == "UNDEFINED"
|
||||
assert out.finished_time is None
|
||||
|
||||
|
||||
def test_result_handles_missing_optional_fields():
|
||||
result.store = JobStore()
|
||||
result.store.put(
|
||||
Job(
|
||||
job_id="accepted",
|
||||
application_id="application_3",
|
||||
script_path="/tmp/j.py",
|
||||
queue="default",
|
||||
submit_time=datetime(2026, 6, 24),
|
||||
connection="prod",
|
||||
yarn_rm_url="http://rm:8088",
|
||||
)
|
||||
)
|
||||
raw = {"app": {"id": "application_3", "state": "ACCEPTED"}}
|
||||
import json
|
||||
|
||||
with patch(
|
||||
"spark_executor.tools.result.get_application_status",
|
||||
return_value=("ACCEPTED", json.dumps(raw)),
|
||||
):
|
||||
out = result.get_job_result("accepted")
|
||||
|
||||
assert out.state == "ACCEPTED"
|
||||
assert out.application_id == "application_3"
|
||||
assert out.final_status is None
|
||||
assert out.diagnostics is None
|
||||
assert out.tracking_url is None
|
||||
assert out.started_time is None
|
||||
assert out.finished_time is None
|
||||
|
||||
|
||||
def test_result_raises_keyerror_for_unknown_job():
|
||||
result.store = JobStore()
|
||||
with pytest.raises(KeyError, match="Unknown job_id"):
|
||||
result.get_job_result("missing")
|
||||
Reference in New Issue
Block a user