80 lines
3.0 KiB
Python
80 lines
3.0 KiB
Python
# coding=utf-8
|
|
from pathlib import Path
|
|
from unittest.mock import patch
|
|
|
|
import pytest
|
|
|
|
from spark_executor.core import connection_store, pending_store
|
|
from spark_executor.core.pending_store import PendingStore
|
|
from spark_executor.models import Connection
|
|
from spark_executor.tools import submit
|
|
|
|
|
|
@pytest.fixture(autouse=True)
|
|
def _fresh(tmp_path: Path, monkeypatch):
|
|
monkeypatch.setattr(connection_store, "DEFAULT_DATA_DIR", str(tmp_path))
|
|
monkeypatch.setattr(connection_store, "store", connection_store.ConnectionStore())
|
|
monkeypatch.setattr(pending_store, "DEFAULT_DATA_DIR", str(tmp_path))
|
|
monkeypatch.setattr(pending_store, "store", PendingStore())
|
|
submit.conn_store = connection_store.store
|
|
submit.pending_store = pending_store.store
|
|
connection_store.store.save(Connection(name="prod", master="yarn", deploy_mode="cluster"))
|
|
|
|
|
|
def _last_pending_id() -> str:
|
|
pending = submit.pending_store.list_all()
|
|
assert pending, "no pending submission was created"
|
|
return pending[-1].pending_id
|
|
|
|
|
|
# --- prepare_submit_job ---
|
|
|
|
def test_prepare_does_not_invoke_spark_submit(monkeypatch):
|
|
called = []
|
|
monkeypatch.setattr(submit, "run_spark_submit", lambda cmd: called.append(cmd))
|
|
out = submit.prepare_submit_job(connection="prod", script_path="/tmp/j.py")
|
|
assert called == [] # never called
|
|
assert out["status"] == "PENDING"
|
|
assert out["pending_id"].startswith("p_")
|
|
|
|
|
|
def test_prepare_persists_pending_with_snapshot(monkeypatch):
|
|
monkeypatch.setattr(submit, "run_spark_submit", lambda cmd: None)
|
|
connection_store.store.save(
|
|
Connection(
|
|
name="prod",
|
|
master="yarn",
|
|
deploy_mode="cluster",
|
|
spark_conf={"spark.sql.shuffle.partitions": "200"},
|
|
)
|
|
)
|
|
submit.prepare_submit_job(connection="prod", script_path="/tmp/j.py", queue="research")
|
|
p = submit.pending_store.get(_last_pending_id())
|
|
assert p is not None
|
|
assert p.connection == "prod"
|
|
assert p.master == "yarn"
|
|
assert p.deploy_mode == "cluster"
|
|
assert p.spark_conf == {"spark.sql.shuffle.partitions": "200"}
|
|
assert p.queue == "research"
|
|
assert p.script_path == "/tmp/j.py"
|
|
|
|
|
|
def test_prepare_raises_for_unknown_connection():
|
|
with pytest.raises(KeyError, match="missing"):
|
|
submit.prepare_submit_job(connection="missing", script_path="/tmp/j.py")
|
|
|
|
|
|
def test_prepare_snapshots_connection_at_prepare_time(monkeypatch):
|
|
"""Editing the connection between prepare and confirm must NOT silently retarget."""
|
|
monkeypatch.setattr(submit, "run_spark_submit", lambda cmd: None)
|
|
submit.prepare_submit_job(connection="prod", script_path="/tmp/j.py")
|
|
p = submit.pending_store.get(_last_pending_id())
|
|
# Now mutate the connection
|
|
connection_store.store.save(
|
|
Connection(name="prod", master="spark://attacker:7077", deploy_mode="client")
|
|
)
|
|
# Snapshot is unchanged
|
|
p2 = submit.pending_store.get(p.pending_id)
|
|
assert p2.master == "yarn"
|
|
assert p2.deploy_mode == "cluster"
|