diff --git a/lending-poc/docker-compose.yml b/lending-poc/docker-compose.yml index d8132bc..3229fe6 100644 --- a/lending-poc/docker-compose.yml +++ b/lending-poc/docker-compose.yml @@ -2,13 +2,8 @@ # combined backend (app/translation/field_mapping/ocr/gateway), and the # frontend. # -# Most backend-facing services come in "cpu"/"gpu" pairs (e.g. ollama / -# ollama-gpu), selected at `docker compose up` time via COMPOSE_PROFILES in -# .env (defaults to "cpu" so a plain `docker compose up` always works with -# no GPU setup). Each gpu variant is written with YAML's `<<: *name` merge -# key to copy its cpu sibling's config and override only what differs (a -# GPU device reservation, a different build), rather than duplicating the -# whole block. +# Backend services that need GPU (ollama, surya-inference) come in "cpu"/"gpu" +# pairs, selected via COMPOSE_PROFILES in .env (defaults to "cpu"). services: db: image: postgres:16 @@ -27,10 +22,6 @@ services: timeout: 5s retries: 5 - # Ollama's own image already auto-detects CUDA at runtime and falls back - # to CPU on its own — no custom build needed here, just the same cpu/gpu - # profile split used for surya-inference/ocr below, so GPU access is only - # requested when COMPOSE_PROFILES=gpu. ollama: &ollama image: ollama/ollama:latest @@ -44,10 +35,6 @@ services: ollama-gpu: <<: *ollama profiles: ["gpu"] - # Only one of ollama/ollama-gpu is ever running (profile-gated), so this - # alias makes ollama-gpu also answer to the plain name "ollama" on the - # Docker network — backend's OLLAMA_HOST doesn't need to know which - # variant is active. networks: default: aliases: @@ -62,7 +49,10 @@ services: surya-inference: &surya-inference - build: ./surya-inference + build: + context: ./surya-inference + args: + LLAMA_IMAGE: ghcr.io/ggml-org/llama.cpp:server profiles: ["cpu"] environment: SURYA_INFERENCE_PARALLEL: ${SURYA_INFERENCE_PARALLEL:-4} @@ -76,14 +66,11 @@ services: build: context: ./surya-inference args: - BASE_IMAGE: nvidia/cuda:12.4.1-devel-ubuntu22.04 - RUNTIME_IMAGE: nvidia/cuda:12.4.1-runtime-ubuntu22.04 - GGML_CUDA: "ON" + LLAMA_IMAGE: ghcr.io/ggml-org/llama.cpp:server-cuda networks: default: aliases: - surya-inference - deploy: resources: reservations: @@ -93,12 +80,10 @@ services: capabilities: [gpu] backend: - &backend build: context: . args: GPU: "0" - profiles: ["cpu"] restart: unless-stopped ports: - "8080:8080" @@ -121,18 +106,7 @@ services: APP_REQUEST_TIMEOUT_SECONDS: ${APP_REQUEST_TIMEOUT_SECONDS:-1800} volumes: - .:/app - # OCR (surya-ocr) pulls its models from the HuggingFace hub on first - # use. Without this the download repeats on every container recreate, - # which takes minutes and needs network — same reasoning as - # ollama_models/surya_models above. - hf_cache:/root/.cache/huggingface - # Only one of ollama/ollama-gpu and one of surya-inference/surya-inference-gpu - # exists at runtime (profile-gated), so all four are listed here as - # required: false — backend can't hard-depend on whichever specific one - # isn't active for the current profile. Without required: false (the - # default), Compose would refuse to start backend at all: it would wait - # on a service that was never created because it belongs to the profile - # you're not running (e.g. waiting on ollama-gpu while running COMPOSE_PROFILES=cpu). depends_on: db: condition: service_healthy @@ -149,25 +123,6 @@ services: condition: service_started required: false - backend-gpu: - <<: *backend - profiles: ["gpu"] - build: - context: . - args: - GPU: "1" - networks: - default: - aliases: - - backend - deploy: - resources: - reservations: - devices: - - driver: nvidia - count: all - capabilities: [gpu] - frontend: build: ./frontend ports: @@ -186,3 +141,4 @@ volumes: ollama_models: surya_models: hf_cache: + diff --git a/lending-poc/surya-inference/Dockerfile b/lending-poc/surya-inference/Dockerfile index 100edbe..d376777 100644 --- a/lending-poc/surya-inference/Dockerfile +++ b/lending-poc/surya-inference/Dockerfile @@ -1,43 +1,15 @@ -# Builds llama.cpp's `llama-server` from source and serves it standalone, so -# ocr-api can point SURYA_INFERENCE_URL here instead of spawning its own copy -# in-process (see surya/inference/backends/llamacpp.py in the ocr-api image -# for the equivalent in-process spawn logic this mirrors). -# -# CPU by default. For a CUDA-capable build (paired with the "gpu" compose -# profile), pass: -# --build-arg BASE_IMAGE=nvidia/cuda:12.4.1-devel-ubuntu22.04 -# --build-arg RUNTIME_IMAGE=nvidia/cuda:12.4.1-runtime-ubuntu22.04 -# --build-arg GGML_CUDA=ON -# entrypoint.sh still probes for a visible GPU at container start and falls -# back to -ngl 0 (CPU) if none is found, so this image works either way. -ARG BASE_IMAGE=python:3.12-slim -ARG RUNTIME_IMAGE=python:3.12-slim - -FROM ${BASE_IMAGE} AS build - -ARG GGML_CUDA=OFF +ARG LLAMA_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda -RUN apt-get update && apt-get install -y --no-install-recommends \ - build-essential cmake git ca-certificates \ - && rm -rf /var/lib/apt/lists/* - -RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /llama.cpp -WORKDIR /llama.cpp -RUN cmake -B build -DGGML_NATIVE=OFF -DGGML_CUDA=${GGML_CUDA} -DLLAMA_CURL=OFF -DCMAKE_BUILD_TYPE=Release \ - && cmake --build build -j"$(nproc)" --target llama-server +FROM ${LLAMA_IMAGE} -FROM ${RUNTIME_IMAGE} +USER root RUN apt-get update && apt-get install -y --no-install-recommends \ - curl ca-certificates libgomp1 \ + curl ca-certificates \ && rm -rf /var/lib/apt/lists/* -# llama-server is dynamically linked against the other .so files built -# alongside it (libllama-server-impl.so, libllama-common.so, libmtmd.so, ...), -# so the whole bin/ directory needs to come along, not just the executable. -COPY --from=build /llama.cpp/build/bin /opt/llama.cpp/bin -ENV LD_LIBRARY_PATH=/opt/llama.cpp/bin -RUN ln -s /opt/llama.cpp/bin/llama-server /usr/local/bin/llama-server +# Ensure llama-server is in system PATH +RUN if [ -f /app/llama-server ]; then ln -sf /app/llama-server /usr/local/bin/llama-server; fi ENV SURYA_GGUF_REPO=datalab-to/surya-ocr-2-gguf \ SURYA_GGUF_MODEL_FILE=surya-2.gguf \ @@ -53,4 +25,4 @@ RUN chmod +x /entrypoint.sh EXPOSE 8000 -ENTRYPOINT ["/entrypoint.sh"] +ENTRYPOINT ["/entrypoint.sh"] \ No newline at end of file diff --git a/lending-poc/surya-inference/entrypoint.sh b/lending-poc/surya-inference/entrypoint.sh index 803a06c..ebb0ddb 100644 --- a/lending-poc/surya-inference/entrypoint.sh +++ b/lending-poc/surya-inference/entrypoint.sh @@ -11,14 +11,17 @@ mkdir -p "$MODEL_DIR" MODEL_PATH="${MODEL_DIR}/${SURYA_GGUF_MODEL_FILE}" MMPROJ_PATH="${MODEL_DIR}/${SURYA_GGUF_MMPROJ_FILE}" -if [ ! -f "$MODEL_PATH" ]; then +# Ensure files exist and are not corrupt/empty (must be > 10MB) +if [ ! -f "$MODEL_PATH" ] || [ $(wc -c < "$MODEL_PATH" 2>/dev/null || echo 0) -lt 10000000 ]; then + rm -f "$MODEL_PATH" echo "Downloading ${SURYA_GGUF_MODEL_FILE} from ${SURYA_GGUF_REPO}..." - curl -fL -o "$MODEL_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MODEL_FILE}" + curl -fL -H "User-Agent: Mozilla/5.0" -o "$MODEL_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MODEL_FILE}" fi -if [ ! -f "$MMPROJ_PATH" ]; then +if [ ! -f "$MMPROJ_PATH" ] || [ $(wc -c < "$MMPROJ_PATH" 2>/dev/null || echo 0) -lt 10000000 ]; then + rm -f "$MMPROJ_PATH" echo "Downloading ${SURYA_GGUF_MMPROJ_FILE} from ${SURYA_GGUF_REPO}..." - curl -fL -o "$MMPROJ_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MMPROJ_FILE}" + curl -fL -H "User-Agent: Mozilla/5.0" -o "$MMPROJ_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MMPROJ_FILE}" fi # nvidia-smi only shows up here if the container was actually started with