docling-serve in a docker with lazy loading model to clear GPU VRAM
529
Built with CUDA 12.8 support.
More information at https://github.com/bankjaneo/docling-serve
docker-compose.yml example
services:
docling-serve:
image: bankja/docling-serve:cuda
container_name: docling-serve
ports:
- "5001:5001"
environment:
- DOCLING_SERVE_ENABLE_UI=1
- DOCLING_SERVE_MAX_SYNC_WAIT=600
- DOCLING_SERVE_ENABLE_REMOTE_SERVICES=1
- DOCLING_NUM_THREADS=7
- DOCLING_SERVE_MAX_TASKS_BEFORE_RESTART=1
- DOCLING_SERVE_FREE_VRAM_ON_IDLE=True
- DOCLING_SERVE_UNLOAD_LLAMA_SWAP_BASE_URL=http://172.17.0.1:9292/v1
restart: unless-stopped
logging:
driver: "json-file"
options:
max-file: "1"
max-size: "10m"
deploy:
resources:
reservations:
devices:
- driver: nvidia
count: all
capabilities: [gpu]
Content type
Image
Digest
sha256:729c88827…
Size
5.4 GB
Last updated
6 months ago
docker pull bankja/docling-serve:cuda