feat(docling): add Chinese OCR support (simplified + traditional)

- Add entrypoint script to auto-download OCR models on first run
- Include ch_sim (Simplified) and ch_tra (Traditional) for HK law firm use
- Add docling_models volume for model persistence
- Bump start_period to 120s for first-run download time
This commit is contained in:
MangoPig
2026-02-21 23:04:39 +00:00
parent 4f367357fc
commit 430cdf32cf
+17 -1
View File
@@ -3,20 +3,34 @@ services:
docling:
# https://github.com/docling-project/docling-serve
# 5001: Web UI + API
# OCR: EasyOCR with English + Chinese Simplified + Traditional (auto-downloaded on first run)
image: ghcr.io/docling-project/docling-serve-cpu:v1.13.0
container_name: jt-docling
restart: unless-stopped
environment:
DOCLING_SERVE_ENABLE_UI: "true"
volumes:
- docling_models:/opt/app-root/src/.cache/docling/models
ports:
- "5001:5001"
entrypoint: ["/bin/sh", "-c"]
command:
- |
# Download Chinese OCR models if missing (first run only)
if [ ! -f /opt/app-root/src/.cache/docling/models/EasyOcr/zh_sim_g2.pth ] || [ ! -f /opt/app-root/src/.cache/docling/models/EasyOcr/zh_tra_g2.pth ]; then
echo "Downloading Chinese OCR models (Simplified + Traditional)..."
python3 -c "import easyocr; easyocr.Reader(['en','ch_sim','ch_tra'], gpu=False)"
# Copy downloaded models to correct location
cp -n /opt/app-root/src/.EasyOCR/model/*.pth /opt/app-root/src/.cache/docling/models/EasyOcr/ 2>/dev/null || true
fi
exec uvicorn docling_serve.app:app --host 0.0.0.0 --port 5001
healthcheck:
test: ["CMD", "curl", "-sf", "http://localhost:5001/health"]
interval: 30s
timeout: 10s
retries: 3
start_period: 60s
start_period: 120s
# Ollama
ollama:
@@ -55,5 +69,7 @@ services:
# retries: 3
volumes:
docling_models:
name: jt-docling-models
ollama_data:
name: jt-ollama-data