# spark-executor-mcp — production runtime # # Bring up with: # docker compose up -d --build # # Prerequisites (one-time): # 1. Put your YARN/Hadoop client configs in ./hadoop-conf/ (must contain # core-site.xml + yarn-site.xml + hdfs-site.xml matching the target # cluster). The Dockerfile's RUN mkdir -p already creates the dir, but # it will be empty until you populate it. # 2. Set YARN_RESOURCE_MANAGER_URL in .env (or export it in your shell). # This is the fallback when a Connection was saved without yarn_rm_url. # # After it starts, the MCP endpoint is at: # http://localhost:8000/spark-executor-mcp (initialize -> tools/list -> tools/call) services: mcp-server: build: context: . dockerfile: Dockerfile image: mcp-server:latest container_name: mcp-server restart: unless-stopped ports: - "8000:8000" environment: # Stream Python output to stdout/stderr line-by-line (live loguru output). PYTHONUNBUFFERED: "1" # --- common/config.py knobs (see .env.example for full docs) --- # Base dir for connections.json, pending_jobs.json, loguru logs/, jobs/. # Defaults to ./data inside the container (mounted from host via volumes below). SPARK_EXECUTOR_DATA_DIR: ${SPARK_EXECUTOR_DATA_DIR:-/app/data} # Where generate_job_file writes LLM-generated PySpark code. # Defaults to /jobs. SPARK_EXECUTOR_JOBS_DIR: ${SPARK_EXECUTOR_JOBS_DIR:-/app/data/jobs} # Base dir for loguru file output (debug/ + info/ subdirs are # auto-created). Defaults to /logs. SPARK_EXECUTOR_LOG_DIR: ${SPARK_EXECUTOR_LOG_DIR:-/app/data/logs} # Fallback YARN RM URL used by the REST client when a Connection's # yarn_rm_url is not set or a Job lacks a snapshot. Leave empty if # you always set yarn_rm_url per Connection via save_connection. YARN_RESOURCE_MANAGER_URL: ${YARN_RESOURCE_MANAGER_URL:-} # Loguru verbosity for stderr + info file. DEBUG | INFO. SPARK_EXECUTOR_LOG_LEVEL: ${SPARK_EXECUTOR_LOG_LEVEL:-DEBUG} # Spark CLI binary name. 'spark-submit' by default; override to # 'spark2-submit' on mixed-version hosts, or to a wrapper path. SPARK_EXECUTOR_SPARK_SUBMIT_BIN: ${SPARK_EXECUTOR_SPARK_SUBMIT_BIN:-spark-submit} # --- files_mcp sandbox --- # The only directory the files_mcp tools can read or write. Lives # under the data volume so it persists across restarts. Override # at deploy time to mount a separate host directory for stricter # isolation from connections / pending_jobs / logs. FILES_MCP_ROOT: ${FILES_MCP_ROOT:-/app/data/files} # --- Runtime paths (consumed by docker-entrypoint.sh, not common/config.py) --- # Override to point at a different JDK install or pre-mounted Spark # distribution. The entrypoint re-derives PATH from these at every # container start, so overrides actually take effect. # JAVA_HOME=/usr/lib/jvm/java-17-openjdk-amd64 # SPARK_HOME=/opt/spark JAVA_HOME: ${JAVA_HOME:-/usr/lib/jvm/java-17-openjdk-amd64} SPARK_HOME: ${SPARK_HOME:-/opt/spark} # --- gunicorn.conf.py knobs (NOT read by common/config.py) --- # 2 workers is a good default for a small MCP service; raise for # high-concurrency deploys. GUNICORN_WORKERS: ${GUNICORN_WORKERS:-2} GUNICORN_TIMEOUT: ${GUNICORN_TIMEOUT:-120} GUNICORN_BIND: ${GUNICORN_BIND:-0.0.0.0:8000} # Optional: pass JVM options to spark-submit (e.g. for proxies, memory). # SPARK_SUBMIT_OPTS: "-Dhttps.proxyHost=..." volumes: # Persist Connections, PendingSubmissions, and loguru logs across # container restarts. Gitignored. - ./data:/app/data # Real Hadoop/YARN client configs read by spark-submit at submit time. # Read-only so the running container cannot mutate cluster config. - ./hadoop-conf:/etc/hadoop/conf:ro # No resource limits — spark-submit talks to YARN, which does the actual # heavy lifting. The MCP server itself is lightweight (FastAPI + httpx). # Uncomment to cap if needed: # deploy: # resources: # limits: # cpus: "1.0" # memory: 1G