diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index ae1e3422..88937415 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -93,5 +93,6 @@ jobs: file: docker/Dockerfile push: false tags: ashim:ci + build-args: SKIP_MODEL_DOWNLOADS=true cache-from: type=gha,scope=unified cache-to: type=gha,mode=max,scope=unified diff --git a/docker/Dockerfile b/docker/Dockerfile index 49391572..ecc8eb69 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -63,6 +63,8 @@ ARG TARGETARCH FROM base-${TARGETOS}-${TARGETARCH} AS production ARG TARGETARCH +# Set to "true" to skip model downloads (for CI builds that just test the image structure) +ARG SKIP_MODEL_DOWNLOADS=false # Install Node.js on amd64 (CUDA base has no Node; arm64 base already has it) RUN if [ "$TARGETARCH" = "amd64" ]; then \ @@ -147,8 +149,14 @@ RUN /opt/venv/bin/pip install numpy==1.26.4 # Note: on amd64, paddlepaddle-gpu can't import without the CUDA driver (only # available at runtime). The download script gracefully skips PaddleOCR model # pre-download in this case; models download on first use at runtime instead. +# In CI (SKIP_MODEL_DOWNLOADS=true), skip downloads — the image structure is +# what matters for the build test; models are verified in integration tests. COPY docker/download_models.py /tmp/download_models.py -RUN /opt/venv/bin/python3 /tmp/download_models.py && rm -f /tmp/download_models.py +RUN if [ "$SKIP_MODEL_DOWNLOADS" = "true" ]; then \ + echo "Skipping model downloads (CI build)"; \ + else \ + /opt/venv/bin/python3 /tmp/download_models.py; \ + fi && rm -f /tmp/download_models.py WORKDIR /app