Test Excel endpoint

This commit is contained in:
MangoPig
2026-02-21 23:25:13 +00:00
parent 50ed6e48f5
commit 6dce685906
5 changed files with 242 additions and 15 deletions
+9 -1
View File
@@ -49,7 +49,15 @@ Rationale: De-risk unknowns (Docling OCR quality, Ollama CPU inference speed) be
**Tested Formats:**
- PDF (Filing Receipt) — extracted text, tables, CJK content ✅
- PNG (Email screenshot) — extracted subject, dates, recipient ✅ (CJK trademark garbled, may need `ocr_lang`)
- PNG (Email screenshot) — extracted subject, dates, recipient ✅
- PNG with `ocr_lang=ch_tra` — Traditional Chinese extracted correctly (官藥坊) ✅
**Portable Chinese OCR:**
- Auto-downloads `ch_sim` (Simplified) and `ch_tra` (Traditional) models on first run
- Models persist in `docling_models` volume
- First run takes ~1-2 min extra for model download; subsequent runs instant
- Use `ocr_lang=ch_tra` for HK/Taiwan, `ocr_lang=ch_sim` for Mainland China
### 1b. Ollama (DONE)
+12
View File
@@ -0,0 +1,12 @@
FROM python:3.12-slim
WORKDIR /app
COPY requirements.txt .
RUN pip install --no-cache-dir -r requirements.txt
COPY main.py .
EXPOSE 8000
CMD ["uvicorn", "main:app", "--host", "0.0.0.0", "--port", "8000"]
+200
View File
@@ -0,0 +1,200 @@
"""
Generic Excel API Service
A FastAPI service for reading Excel files (.xlsx).
All file paths are relative to /data/ mount.
Endpoints:
- GET /health - Health check
- GET /excel/sheets - List all sheet names
- GET /excel/schema - Get column headers
- GET /excel/read - Read all rows
- GET /excel/row/{row_num} - Get specific row (1-indexed)
"""
from fastapi import FastAPI, HTTPException, Query
from pydantic import BaseModel
from pathlib import Path
from typing import Optional
from openpyxl import load_workbook
import os
app = FastAPI(
title="JingTian Tools Service",
description="Generic Excel API for reading xlsx files",
version="0.2.0",
)
DATA_ROOT = Path(os.getenv("DATA_ROOT", "/data"))
def get_workbook(file_path: str):
"""Load workbook from file path relative to DATA_ROOT."""
full_path = DATA_ROOT / file_path
if not full_path.exists():
raise HTTPException(status_code=404, detail=f"File not found: {file_path}")
if not full_path.suffix.lower() == ".xlsx":
raise HTTPException(status_code=400, detail="Only .xlsx files supported")
try:
return load_workbook(full_path, read_only=True, data_only=True)
except Exception as e:
raise HTTPException(
status_code=500, detail=f"Failed to open workbook: {str(e)}"
)
def get_sheet(wb, sheet_name: str):
"""Get sheet by name."""
if sheet_name not in wb.sheetnames:
raise HTTPException(
status_code=404,
detail=f"Sheet '{sheet_name}' not found. Available: {wb.sheetnames}",
)
return wb[sheet_name]
def cell_to_str(cell_value) -> str:
"""Convert cell value to string, empty string for None."""
if cell_value is None:
return ""
return str(cell_value)
@app.get("/health")
async def health():
return {"status": "healthy", "service": "tools", "version": "0.2.0"}
@app.get("/excel/sheets")
async def list_sheets(
file_path: str = Query(..., description="Path to xlsx file relative to /data/"),
):
"""List all sheet names in the workbook."""
wb = get_workbook(file_path)
sheets = wb.sheetnames
wb.close()
return {"file_path": file_path, "sheets": sheets, "count": len(sheets)}
@app.get("/excel/schema")
async def get_schema(
file_path: str = Query(..., description="Path to xlsx file relative to /data/"),
sheet_name: str = Query(..., description="Sheet name to read"),
header_row: int = Query(1, description="Row number containing headers (1-indexed)"),
):
"""Get column headers from the specified sheet."""
wb = get_workbook(file_path)
ws = get_sheet(wb, sheet_name)
headers = []
for cell in ws[header_row]:
headers.append(cell_to_str(cell.value))
wb.close()
return {
"file_path": file_path,
"sheet_name": sheet_name,
"header_row": header_row,
"columns": headers,
"column_count": len(headers),
}
@app.get("/excel/read")
async def read_all(
file_path: str = Query(..., description="Path to xlsx file relative to /data/"),
sheet_name: str = Query(..., description="Sheet name to read"),
header_row: int = Query(1, description="Row number containing headers (1-indexed)"),
start_row: Optional[int] = Query(
None,
description="Start reading from this row (1-indexed, defaults to header_row + 1)",
),
limit: Optional[int] = Query(None, description="Max rows to return"),
):
"""Read all rows from the specified sheet."""
wb = get_workbook(file_path)
ws = get_sheet(wb, sheet_name)
# Get headers
headers = [cell_to_str(cell.value) for cell in ws[header_row]]
# Determine data start row
data_start = start_row if start_row else header_row + 1
# Read rows
rows = []
row_count = 0
for row_num, row in enumerate(ws.iter_rows(min_row=data_start), start=data_start):
# Stop if limit reached
if limit and row_count >= limit:
break
# Skip completely empty rows
values = [cell_to_str(cell.value) for cell in row]
if all(v == "" for v in values):
continue
# Build row dict with headers as keys
row_data = {"_row_num": row_num}
for i, header in enumerate(headers):
if i < len(values):
key = header if header else f"_col_{i + 1}"
row_data[key] = values[i]
rows.append(row_data)
row_count += 1
wb.close()
return {
"file_path": file_path,
"sheet_name": sheet_name,
"columns": headers,
"rows": rows,
"row_count": len(rows),
}
@app.get("/excel/row/{row_num}")
async def get_row(
row_num: int,
file_path: str = Query(..., description="Path to xlsx file relative to /data/"),
sheet_name: str = Query(..., description="Sheet name to read"),
header_row: int = Query(1, description="Row number containing headers (1-indexed)"),
):
"""Get a specific row by row number (1-indexed)."""
if row_num < 1:
raise HTTPException(status_code=400, detail="Row number must be >= 1")
wb = get_workbook(file_path)
ws = get_sheet(wb, sheet_name)
# Get headers
headers = [cell_to_str(cell.value) for cell in ws[header_row]]
# Get the specific row
try:
row = ws[row_num]
except Exception:
wb.close()
raise HTTPException(status_code=404, detail=f"Row {row_num} not found")
values = [cell_to_str(cell.value) for cell in row]
# Build row dict
row_data = {"_row_num": row_num}
for i, header in enumerate(headers):
if i < len(values):
key = header if header else f"_col_{i + 1}"
row_data[key] = values[i]
wb.close()
return {"file_path": file_path, "sheet_name": sheet_name, "row": row_data}
# Future endpoints (Phase 2 & 3)
# POST /excel/row - Add new row
# PUT /excel/row/{row_num} - Update row
# DELETE /excel/row/{row_num} - Delete row
# POST /excel/column - Add column
# POST /excel/format - Apply formatting
# POST /excel/create - Create new workbook
+4
View File
@@ -0,0 +1,4 @@
fastapi==0.115.0
uvicorn==0.30.6
openpyxl==3.1.5
pydantic==2.9.0
+17 -14
View File
@@ -58,20 +58,23 @@ services:
retries: 5
start_period: 120s
# Tools (Python FastAPI - Excel operations)
# tools:
# build: ./Tools/Service
# container_name: jt-tools
# restart: unless-stopped
# volumes:
# - /data/jingtian/BenjaminTeam:/data:rw
# ports:
# - "8080:8080"
# healthcheck:
# test: ["CMD", "curl", "-sf", "http://localhost:8080/health"]
# interval: 30s
# timeout: 10s
# retries: 3
# Tools (Python FastAPI - Generic Excel API)
tools:
build: ./Tools/Service
container_name: jt-tools
restart: unless-stopped
environment:
DATA_ROOT: /data
volumes:
- /data/jingtian/BenjaminTeam:/data:rw
ports:
- "8000:8000"
healthcheck:
test: ["CMD", "curl", "-sf", "http://localhost:8000/health"]
interval: 30s
timeout: 10s
retries: 3
start_period: 10s
volumes:
docling_models: