aditya0103/structured-data-extractor
0
1[project]2name = "structured-data-extraction"3version = "0.1.0"4description = "Multi-domain document extraction service: invoices, receipts, and SEC filings to schema-validated JSON."5authors = [{ name = "ASP", email = "adityapatel1801@gmail.com" }]6readme = "README.md"7requires-python = ">=3.11"8license = { text = "MIT" }9 10[tool.ruff]11line-length = 10012target-version = "py311"13 14[tool.ruff.lint]15select = ["E", "F", "I", "N", "W", "UP", "B", "SIM"]16ignore = ["E501", "UP017"] # E501: line-length via formatter. UP017: datetime.UTC (3.11-only) — we still support timezone.utc for broader interpreter reach in tooling.17 18[tool.ruff.lint.per-file-ignores]19# HTTP-error taxonomy — these are our own class hierarchy, not surprises.20# Renaming to `*Error` would leak "Error" everywhere at call sites without21# adding meaning.22"src/api/errors.py" = ["N818"]23# FastAPI convention: File(...) / Depends() go in function default args.24# The whole framework is built around this pattern.25"src/api/routers/*.py" = ["B008"]26# Parser branches carry different explanatory comments — a ternary would27# drop them and hurt readability more than help.28"src/data_prep/parsers.py" = ["SIM108"]29# Test tuple-unpacking sometimes leaves parts unused for clarity.30"tests/**/*.py" = ["B007"]31 32[tool.black]33line-length = 10034target-version = ["py311"]35 36[tool.pytest.ini_options]37testpaths = ["tests"]38python_files = "test_*.py"39python_classes = "Test*"40python_functions = "test_*"41addopts = "-v"42# Coverage: run `pytest --cov=src --cov-report=term-missing --cov-report=html` locally.43# Kept out of default addopts because coverage's temp-file cleanup fails on some44# network/mounted filesystems (works fine on native Windows/Linux).45asyncio_mode = "auto"46 47[tool.coverage.run]48source = ["src"]49omit = ["*/tests/*", "*/__init__.py"]50 