# syntax=docker/dockerfile:1 # # Spark Executor MCP — runtime image. # Build deps with uv (frozen, prod-only), then drop in the source on top of # python:3.12-slim with Spark + YARN configs mounted for the spark-submit / # yarn CLI calls inside the MCP tools. FROM python:3.12-slim # --- uv (official binary) --- COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /usr/local/bin/ # --- Spark + Hadoop config (matches the original Dockerfile) --- ARG SPARK_VERSION=4.1.2 RUN apt-get update && \ apt-get install -y --no-install-recommends \ curl \ ca-certificates \ tar && \ curl -L \ https://mirrors.tuna.tsinghua.edu.cn/apache/spark/spark-${SPARK_VERSION}/spark-${SPARK_VERSION}-bin-hadoop3.tgz \ -o /tmp/spark.tgz && \ mkdir -p /opt && \ tar -xzf /tmp/spark.tgz -C /opt && \ mv /opt/spark-${SPARK_VERSION}-bin-hadoop3 /opt/spark && \ rm -f /tmp/spark.tgz && \ apt-get clean && \ rm -rf /var/lib/apt/lists/* ENV SPARK_HOME=/opt/spark ENV PATH=${SPARK_HOME}/bin:${PATH} # Hadoop/Yarn 配置目录(运行时挂载) RUN mkdir -p /etc/hadoop/conf # 默认值,可在 docker run 时覆盖 ENV HADOOP_CONF_DIR=/etc/hadoop/conf ENV YARN_CONF_DIR=/etc/hadoop/conf # --- App --- WORKDIR /app # Install Python deps first so this layer caches independently of source. # --frozen pins to uv.lock exactly; --no-dev skips pytest etc. for a slim # production image; --no-install-project defers copying the source. COPY pyproject.toml uv.lock ./ RUN uv sync --index-url=https://pypi.tuna.tsinghua.edu.cn/simple/ --frozen --no-dev --no-install-project # Now copy the source and let uv wire it in. COPY main.py ./ COPY spark_executor ./spark_executor COPY common ./common RUN uv sync --index-url=https://pypi.tuna.tsinghua.edu.cn/simple/ --frozen --no-dev # Put the venv on PATH so `python` / `uvicorn` resolve to the project env. ENV PATH=/app/.venv/bin:$PATH ENV PYTHONUNBUFFERED=1 EXPOSE 8000 CMD ["python", "main.py"]