""" Generic Excel API Service A FastAPI service for reading and writing Excel files (.xlsx). All file paths are relative to /data/ mount. Endpoints: - GET /health - Health check - GET /excel/sheets - List all sheet names - GET /excel/schema - Get column headers - GET /excel/read - Read all rows - GET /excel/row/{row_num} - Get specific row (1-indexed) - GET /excel/find - Find row by column value - POST /excel/row - Add new row - PUT /excel/row/{row_num} - Update row - POST /excel/create - Create new workbook """ from fastapi import FastAPI, HTTPException, Query from pydantic import BaseModel from pathlib import Path from typing import Optional, List, Any from openpyxl import Workbook, load_workbook from openpyxl.styles import Font, Alignment import os app = FastAPI( title="JingTian Tools Service", description="Generic Excel API for reading and writing xlsx files", version="0.3.0", ) DATA_ROOT = Path(os.getenv("DATA_ROOT", "/data")) # Request models class AddRowRequest(BaseModel): file_path: str sheet_name: str header_row: int = 1 row: dict[str, Any] class UpdateRowRequest(BaseModel): file_path: str sheet_name: str header_row: int = 1 row: dict[str, Any] class CreateWorkbookRequest(BaseModel): file_path: str sheet_name: str columns: List[str] def get_workbook_readonly(file_path: str): """Load workbook from file path relative to DATA_ROOT (read-only).""" full_path = DATA_ROOT / file_path if not full_path.exists(): raise HTTPException(status_code=404, detail=f"File not found: {file_path}") if not full_path.suffix.lower() == ".xlsx": raise HTTPException(status_code=400, detail="Only .xlsx files supported") try: return load_workbook(full_path, read_only=True, data_only=True) except Exception as e: raise HTTPException( status_code=500, detail=f"Failed to open workbook: {str(e)}" ) def get_workbook_writable(file_path: str): """Load workbook from file path relative to DATA_ROOT (writable).""" full_path = DATA_ROOT / file_path if not full_path.exists(): raise HTTPException(status_code=404, detail=f"File not found: {file_path}") if not full_path.suffix.lower() == ".xlsx": raise HTTPException(status_code=400, detail="Only .xlsx files supported") try: return load_workbook(full_path), full_path except Exception as e: raise HTTPException( status_code=500, detail=f"Failed to open workbook: {str(e)}" ) def get_sheet(wb, sheet_name: str): """Get sheet by name.""" if sheet_name not in wb.sheetnames: raise HTTPException( status_code=404, detail=f"Sheet '{sheet_name}' not found. Available: {wb.sheetnames}", ) return wb[sheet_name] def cell_to_str(cell_value) -> str: """Convert cell value to string, empty string for None.""" if cell_value is None: return "" return str(cell_value) def get_headers(ws, header_row: int) -> list[str]: """Get headers from a worksheet.""" headers = [] for cell in ws[header_row]: headers.append(cell_to_str(cell.value)) return headers def get_column_index(headers: list[str], column_name: str) -> int: """Get 1-indexed column index for a header name.""" try: return headers.index(column_name) + 1 except ValueError: raise HTTPException( status_code=400, detail=f"Column '{column_name}' not found. Available: {headers}", ) @app.get("/health") async def health(): return {"status": "healthy", "service": "tools", "version": "0.3.0"} @app.get("/excel/sheets") async def list_sheets( file_path: str = Query(..., description="Path to xlsx file relative to /data/"), ): """List all sheet names in the workbook.""" wb = get_workbook_readonly(file_path) sheets = wb.sheetnames wb.close() return {"file_path": file_path, "sheets": sheets, "count": len(sheets)} @app.get("/excel/schema") async def get_schema( file_path: str = Query(..., description="Path to xlsx file relative to /data/"), sheet_name: str = Query(..., description="Sheet name to read"), header_row: int = Query(1, description="Row number containing headers (1-indexed)"), ): """Get column headers from the specified sheet.""" wb = get_workbook_readonly(file_path) ws = get_sheet(wb, sheet_name) headers = get_headers(ws, header_row) wb.close() return { "file_path": file_path, "sheet_name": sheet_name, "header_row": header_row, "columns": headers, "column_count": len(headers), } @app.get("/excel/read") async def read_all( file_path: str = Query(..., description="Path to xlsx file relative to /data/"), sheet_name: str = Query(..., description="Sheet name to read"), header_row: int = Query(1, description="Row number containing headers (1-indexed)"), start_row: Optional[int] = Query( None, description="Start reading from this row (1-indexed, defaults to header_row + 1)", ), limit: Optional[int] = Query(None, description="Max rows to return"), ): """Read all rows from the specified sheet.""" wb = get_workbook_readonly(file_path) ws = get_sheet(wb, sheet_name) headers = get_headers(ws, header_row) data_start = start_row if start_row else header_row + 1 rows = [] row_count = 0 for row_num, row in enumerate(ws.iter_rows(min_row=data_start), start=data_start): if limit and row_count >= limit: break values = [cell_to_str(cell.value) for cell in row] if all(v == "" for v in values): continue row_data = {"_row_num": row_num} for i, header in enumerate(headers): if i < len(values): key = header if header else f"_col_{i + 1}" row_data[key] = values[i] rows.append(row_data) row_count += 1 wb.close() return { "file_path": file_path, "sheet_name": sheet_name, "columns": headers, "rows": rows, "row_count": len(rows), } @app.get("/excel/row/{row_num}") async def get_row( row_num: int, file_path: str = Query(..., description="Path to xlsx file relative to /data/"), sheet_name: str = Query(..., description="Sheet name to read"), header_row: int = Query(1, description="Row number containing headers (1-indexed)"), ): """Get a specific row by row number (1-indexed).""" if row_num < 1: raise HTTPException(status_code=400, detail="Row number must be >= 1") wb = get_workbook_readonly(file_path) ws = get_sheet(wb, sheet_name) headers = get_headers(ws, header_row) try: row = ws[row_num] except Exception: wb.close() raise HTTPException(status_code=404, detail=f"Row {row_num} not found") values = [cell_to_str(cell.value) for cell in row] row_data = {"_row_num": row_num} for i, header in enumerate(headers): if i < len(values): key = header if header else f"_col_{i + 1}" row_data[key] = values[i] wb.close() return {"file_path": file_path, "sheet_name": sheet_name, "row": row_data} @app.get("/excel/find") async def find_row( file_path: str = Query(..., description="Path to xlsx file relative to /data/"), sheet_name: str = Query(..., description="Sheet name to search"), header_row: int = Query(1, description="Row number containing headers (1-indexed)"), column: str = Query(..., description="Column name to search"), value: str = Query(..., description="Value to find"), ): """Find a row by column value.""" wb = get_workbook_readonly(file_path) ws = get_sheet(wb, sheet_name) headers = get_headers(ws, header_row) col_idx = get_column_index(headers, column) data_start = header_row + 1 for row_num, row in enumerate(ws.iter_rows(min_row=data_start), start=data_start): values = [cell_to_str(cell.value) for cell in row] if col_idx - 1 < len(values) and values[col_idx - 1] == value: row_data = {"_row_num": row_num} for i, header in enumerate(headers): if i < len(values): key = header if header else f"_col_{i + 1}" row_data[key] = values[i] wb.close() return { "found": True, "row_num": row_num, "row": row_data, "message": f"Found at row {row_num}", } wb.close() raise HTTPException( status_code=404, detail=f"No row found with {column}='{value}'", ) @app.post("/excel/row") async def add_row(request: AddRowRequest): """Add a new row to the sheet.""" wb, full_path = get_workbook_writable(request.file_path) ws = get_sheet(wb, request.sheet_name) headers = get_headers(ws, request.header_row) # Find next empty row next_row = ws.max_row + 1 # Write values for col_name, value in request.row.items(): if col_name.startswith("_"): continue try: col_idx = headers.index(col_name) + 1 ws.cell(row=next_row, column=col_idx, value=value) except ValueError: pass # Skip unknown columns wb.save(full_path) wb.close() return { "success": True, "row_num": next_row, "message": f"Added row at {next_row}", } @app.put("/excel/row/{row_num}") async def update_row(row_num: int, request: UpdateRowRequest): """Update an existing row.""" if row_num < 1: raise HTTPException(status_code=400, detail="Row number must be >= 1") wb, full_path = get_workbook_writable(request.file_path) ws = get_sheet(wb, request.sheet_name) headers = get_headers(ws, request.header_row) # Check row exists if row_num > ws.max_row: wb.close() raise HTTPException(status_code=404, detail=f"Row {row_num} not found") # Update values for col_name, value in request.row.items(): if col_name.startswith("_"): continue try: col_idx = headers.index(col_name) + 1 ws.cell(row=row_num, column=col_idx, value=value) except ValueError: pass # Skip unknown columns wb.save(full_path) wb.close() return { "success": True, "row_num": row_num, "message": f"Updated row {row_num}", } @app.post("/excel/create") async def create_workbook(request: CreateWorkbookRequest): """Create a new workbook with specified columns.""" full_path = DATA_ROOT / request.file_path # Create parent directories if needed full_path.parent.mkdir(parents=True, exist_ok=True) # Check if file exists if full_path.exists(): raise HTTPException( status_code=409, detail=f"File already exists: {request.file_path}", ) # Create workbook wb = Workbook() ws = wb.active ws.title = request.sheet_name # Write headers with formatting header_font = Font(bold=True) for col_idx, col_name in enumerate(request.columns, start=1): cell = ws.cell(row=1, column=col_idx, value=col_name) cell.font = header_font cell.alignment = Alignment(horizontal="center") # Auto-adjust column widths for col_idx, col_name in enumerate(request.columns, start=1): ws.column_dimensions[ws.cell(row=1, column=col_idx).column_letter].width = max( len(col_name) + 2, 12 ) wb.save(full_path) wb.close() return { "success": True, "file_path": request.file_path, "sheet_name": request.sheet_name, "columns": request.columns, "message": f"Created workbook with {len(request.columns)} columns", }