Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
60 changes: 8 additions & 52 deletions lending-poc/docker-compose.yml
Original file line number Diff line number Diff line change
Expand Up @@ -2,13 +2,8 @@
# combined backend (app/translation/field_mapping/ocr/gateway), and the
# frontend.
#
# Most backend-facing services come in "cpu"/"gpu" pairs (e.g. ollama /
# ollama-gpu), selected at `docker compose up` time via COMPOSE_PROFILES in
# .env (defaults to "cpu" so a plain `docker compose up` always works with
# no GPU setup). Each gpu variant is written with YAML's `<<: *name` merge
# key to copy its cpu sibling's config and override only what differs (a
# GPU device reservation, a different build), rather than duplicating the
# whole block.
# Backend services that need GPU (ollama, surya-inference) come in "cpu"/"gpu"
# pairs, selected via COMPOSE_PROFILES in .env (defaults to "cpu").
services:
db:
image: postgres:16
Expand All @@ -27,10 +22,6 @@ services:
timeout: 5s
retries: 5

# Ollama's own image already auto-detects CUDA at runtime and falls back
# to CPU on its own — no custom build needed here, just the same cpu/gpu
# profile split used for surya-inference/ocr below, so GPU access is only
# requested when COMPOSE_PROFILES=gpu.
ollama:
&ollama
image: ollama/ollama:latest
Expand All @@ -44,10 +35,6 @@ services:
ollama-gpu:
<<: *ollama
profiles: ["gpu"]
# Only one of ollama/ollama-gpu is ever running (profile-gated), so this
# alias makes ollama-gpu also answer to the plain name "ollama" on the
# Docker network — backend's OLLAMA_HOST doesn't need to know which
# variant is active.
networks:
default:
aliases:
Expand All @@ -62,7 +49,10 @@ services:

surya-inference:
&surya-inference
build: ./surya-inference
build:
context: ./surya-inference
args:
LLAMA_IMAGE: ghcr.io/ggml-org/llama.cpp:server
profiles: ["cpu"]
environment:
SURYA_INFERENCE_PARALLEL: ${SURYA_INFERENCE_PARALLEL:-4}
Expand All @@ -76,14 +66,11 @@ services:
build:
context: ./surya-inference
args:
BASE_IMAGE: nvidia/cuda:12.4.1-devel-ubuntu22.04
RUNTIME_IMAGE: nvidia/cuda:12.4.1-runtime-ubuntu22.04
GGML_CUDA: "ON"
LLAMA_IMAGE: ghcr.io/ggml-org/llama.cpp:server-cuda
networks:
default:
aliases:
- surya-inference

deploy:
resources:
reservations:
Expand All @@ -93,12 +80,10 @@ services:
capabilities: [gpu]

backend:
&backend
build:
context: .
args:
GPU: "0"
profiles: ["cpu"]
restart: unless-stopped
ports:
- "8080:8080"
Expand All @@ -121,18 +106,7 @@ services:
APP_REQUEST_TIMEOUT_SECONDS: ${APP_REQUEST_TIMEOUT_SECONDS:-1800}
volumes:
- .:/app
# OCR (surya-ocr) pulls its models from the HuggingFace hub on first
# use. Without this the download repeats on every container recreate,
# which takes minutes and needs network — same reasoning as
# ollama_models/surya_models above.
- hf_cache:/root/.cache/huggingface
# Only one of ollama/ollama-gpu and one of surya-inference/surya-inference-gpu
# exists at runtime (profile-gated), so all four are listed here as
# required: false — backend can't hard-depend on whichever specific one
# isn't active for the current profile. Without required: false (the
# default), Compose would refuse to start backend at all: it would wait
# on a service that was never created because it belongs to the profile
# you're not running (e.g. waiting on ollama-gpu while running COMPOSE_PROFILES=cpu).
depends_on:
db:
condition: service_healthy
Expand All @@ -149,25 +123,6 @@ services:
condition: service_started
required: false

backend-gpu:
<<: *backend
profiles: ["gpu"]
build:
context: .
args:
GPU: "1"
networks:
default:
aliases:
- backend
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]

frontend:
build: ./frontend
ports:
Expand All @@ -186,3 +141,4 @@ volumes:
ollama_models:
surya_models:
hf_cache:

42 changes: 7 additions & 35 deletions lending-poc/surya-inference/Dockerfile
Original file line number Diff line number Diff line change
@@ -1,43 +1,15 @@
# Builds llama.cpp's `llama-server` from source and serves it standalone, so
# ocr-api can point SURYA_INFERENCE_URL here instead of spawning its own copy
# in-process (see surya/inference/backends/llamacpp.py in the ocr-api image
# for the equivalent in-process spawn logic this mirrors).
#
# CPU by default. For a CUDA-capable build (paired with the "gpu" compose
# profile), pass:
# --build-arg BASE_IMAGE=nvidia/cuda:12.4.1-devel-ubuntu22.04
# --build-arg RUNTIME_IMAGE=nvidia/cuda:12.4.1-runtime-ubuntu22.04
# --build-arg GGML_CUDA=ON
# entrypoint.sh still probes for a visible GPU at container start and falls
# back to -ngl 0 (CPU) if none is found, so this image works either way.
ARG BASE_IMAGE=python:3.12-slim
ARG RUNTIME_IMAGE=python:3.12-slim

FROM ${BASE_IMAGE} AS build

ARG GGML_CUDA=OFF
ARG LLAMA_IMAGE=ghcr.io/ggml-org/llama.cpp:server-cuda

RUN apt-get update && apt-get install -y --no-install-recommends \
build-essential cmake git ca-certificates \
&& rm -rf /var/lib/apt/lists/*

RUN git clone --depth 1 https://github.com/ggml-org/llama.cpp /llama.cpp
WORKDIR /llama.cpp
RUN cmake -B build -DGGML_NATIVE=OFF -DGGML_CUDA=${GGML_CUDA} -DLLAMA_CURL=OFF -DCMAKE_BUILD_TYPE=Release \
&& cmake --build build -j"$(nproc)" --target llama-server
FROM ${LLAMA_IMAGE}

FROM ${RUNTIME_IMAGE}
USER root

RUN apt-get update && apt-get install -y --no-install-recommends \
curl ca-certificates libgomp1 \
curl ca-certificates \
&& rm -rf /var/lib/apt/lists/*

# llama-server is dynamically linked against the other .so files built
# alongside it (libllama-server-impl.so, libllama-common.so, libmtmd.so, ...),
# so the whole bin/ directory needs to come along, not just the executable.
COPY --from=build /llama.cpp/build/bin /opt/llama.cpp/bin
ENV LD_LIBRARY_PATH=/opt/llama.cpp/bin
RUN ln -s /opt/llama.cpp/bin/llama-server /usr/local/bin/llama-server
# Ensure llama-server is in system PATH
RUN if [ -f /app/llama-server ]; then ln -sf /app/llama-server /usr/local/bin/llama-server; fi

ENV SURYA_GGUF_REPO=datalab-to/surya-ocr-2-gguf \
SURYA_GGUF_MODEL_FILE=surya-2.gguf \
Expand All @@ -53,4 +25,4 @@ RUN chmod +x /entrypoint.sh

EXPOSE 8000

ENTRYPOINT ["/entrypoint.sh"]
ENTRYPOINT ["/entrypoint.sh"]
11 changes: 7 additions & 4 deletions lending-poc/surya-inference/entrypoint.sh
Original file line number Diff line number Diff line change
Expand Up @@ -11,14 +11,17 @@ mkdir -p "$MODEL_DIR"
MODEL_PATH="${MODEL_DIR}/${SURYA_GGUF_MODEL_FILE}"
MMPROJ_PATH="${MODEL_DIR}/${SURYA_GGUF_MMPROJ_FILE}"

if [ ! -f "$MODEL_PATH" ]; then
# Ensure files exist and are not corrupt/empty (must be > 10MB)
if [ ! -f "$MODEL_PATH" ] || [ $(wc -c < "$MODEL_PATH" 2>/dev/null || echo 0) -lt 10000000 ]; then
rm -f "$MODEL_PATH"
echo "Downloading ${SURYA_GGUF_MODEL_FILE} from ${SURYA_GGUF_REPO}..."
curl -fL -o "$MODEL_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MODEL_FILE}"
curl -fL -H "User-Agent: Mozilla/5.0" -o "$MODEL_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MODEL_FILE}"
Comment on lines +15 to +18
fi

if [ ! -f "$MMPROJ_PATH" ]; then
if [ ! -f "$MMPROJ_PATH" ] || [ $(wc -c < "$MMPROJ_PATH" 2>/dev/null || echo 0) -lt 10000000 ]; then
rm -f "$MMPROJ_PATH"
echo "Downloading ${SURYA_GGUF_MMPROJ_FILE} from ${SURYA_GGUF_REPO}..."
curl -fL -o "$MMPROJ_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MMPROJ_FILE}"
curl -fL -H "User-Agent: Mozilla/5.0" -o "$MMPROJ_PATH" "https://huggingface.co/${SURYA_GGUF_REPO}/resolve/main/${SURYA_GGUF_MMPROJ_FILE}"
fi

# nvidia-smi only shows up here if the container was actually started with
Expand Down
Loading