# syntax=docker/dockerfile:1 # # Spark Executor MCP — runtime image. # Build deps with uv (frozen, prod-only), then drop in the source on top of # ppython:3.12-slim-bookworm with Spark + YARN configs mounted for the spark-submit / # yarn CLI calls inside the MCP tools. FROM python:3.12-slim-bookworm # --- uv (official binary) --- COPY --from=ghcr.io/astral-sh/uv:latest /uv /uvx /usr/local/bin/ # --- Spark + Hadoop config (matches the original Dockerfile) --- ARG SPARK_VERSION=3.1.2 RUN sed -i 's|deb.debian.org|mirrors.tuna.tsinghua.edu.cn|g' /etc/apt/sources.list.d/debian.sources && \ apt-get update && \ apt-cache search openjdk && \ apt-get install -y --no-install-recommends \ curl \ ca-certificates \ tar \ openjdk-11-jre-headless && \ curl -L \ https://mirrors.tuna.tsinghua.edu.cn/apache/spark/spark-${SPARK_VERSION}/spark-${SPARK_VERSION}-bin-hadoop2.7.tgz \ -o /tmp/spark.tgz && \ mkdir -p /opt && \ tar -xzf /tmp/spark.tgz -C /opt && \ mv /opt/spark-${SPARK_VERSION}-bin-hadoop2.7 /opt/spark && \ rm -f /tmp/spark.tgz && \ apt-get clean && \ rm -rf /var/lib/apt/lists/* && \ # Defensive: rewrite the first line of every bin/* script to use a # known-good shebang. Some Spark distributions have shipped with # 'bach' (typo for 'bash') in the shebang, which makes the kernel # refuse to exec the script at all. Idempotent and cheap. find /opt/spark/bin -type f -exec sed -i '1s|^.*$|#!/usr/bin/env bash|' {} + ENV JAVA_HOME=/usr/lib/jvm/java-11-openjdk-amd64 ENV SPARK_HOME=/opt/spark ENV PATH=${JAVA_HOME}/bin:${SPARK_HOME}/bin:${PATH} # Hadoop/Yarn 配置目录(运行时挂载) RUN mkdir -p /etc/hadoop/conf # 默认值,可在 docker run 时覆盖 ENV HADOOP_CONF_DIR=/etc/hadoop/conf ENV YARN_CONF_DIR=/etc/hadoop/conf # --- App --- WORKDIR /app # Install Python deps first so this layer caches independently of source. # --frozen pins to uv.lock exactly; --no-dev skips pytest etc. for a slim # production image; --no-install-project defers copying the source. COPY pyproject.toml uv.lock ./ RUN uv sync --index-url=https://pypi.tuna.tsinghua.edu.cn/simple/ --frozen --no-dev --no-install-project # Now copy the source and let uv wire it in. COPY main.py ./ COPY spark_executor ./spark_executor COPY common ./common COPY gunicorn.conf.py ./ RUN uv sync --index-url=https://pypi.tuna.tsinghua.edu.cn/simple/ --frozen --no-dev # Put the venv on PATH so `python` / `gunicorn` / `uvicorn` resolve to the project env. ENV PATH=/app/.venv/bin:$PATH ENV PYTHONUNBUFFERED=1 # Entrypoint re-derives PATH from JAVA_HOME and SPARK_HOME at container # start, so overriding either via docker-compose / .env actually changes # which `java` and `spark-submit` binaries the gunicorn process picks up. COPY docker-entrypoint.sh /usr/local/bin/docker-entrypoint.sh RUN chmod +x /usr/local/bin/docker-entrypoint.sh ENTRYPOINT ["/usr/local/bin/docker-entrypoint.sh"] # gunicorn is the prod entrypoint — multiple ASGI workers, graceful # shutdown, stdout/stderr logs. Config knobs are env-var driven (see # gunicorn.conf.py). # # Common overrides via -e flags at `docker run`: # -e GUNICORN_WORKERS=4 # -e GUNICORN_TIMEOUT=180 # -e GUNICORN_BIND=0.0.0.0:9000 EXPOSE 8000 CMD ["gunicorn", "main:app"]