feat(ingestion): add DOCX/CSV/XLSX parsing and fixed-size chunking (ADR-0018)
Adds src/application/ingestion/ -- Persian normalization, DOCX body walk with structural data/layout table classification, CSV/XLSX row rendering, and fixed-size token chunking (cl100k_base, 400/60/512) -- as pure functions per ADR-0015, tested against real production documents (asia_data_sample, kept out of the repo). ADR-0018 records where this diverges from ADR-0004 (fixed-size default, no invented headings/tree, structural table classification, header-provable labeling only). Plan 001's scope line is corrected from CSV-only to DOCX/XLSX/CSV, and CLAUDE.md's stale project-status paragraph is updated to match current implementation state.
This commit is contained in:
30
tests/support/documents.json
Normal file
30
tests/support/documents.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"prose_docx": {
|
||||
"filename": "bimeh_havades.docx",
|
||||
"why": "38 paragraphs, all Normal, no tables: the plain-prose path, and the proof that no heading is invented where the document declares none."
|
||||
},
|
||||
"list_docx": {
|
||||
"filename": "طرح ها-عمروحوادث-14050220.docx",
|
||||
"why": "109 paragraphs mixing Normal and List Paragraph: prose carrying list styling."
|
||||
},
|
||||
"table_docx": {
|
||||
"filename": "شماره تماس معاونت و مدیریتها14050309.docx",
|
||||
"why": "One 35x5 contact table, no body prose. A data table with a real header row and a vertically merged first column."
|
||||
},
|
||||
"mixed_docx": {
|
||||
"filename": "شرایط عمومی بیمه حوادث اشخاص14041205.docx",
|
||||
"why": "Prose interleaved with two tables: a 30x3 list of injury compensations with NO header row, and an 11x2 table that has one."
|
||||
},
|
||||
"layout_docx": {
|
||||
"filename": "چت بات-مهندسی.docx",
|
||||
"why": "A 3x2 layout table whose single merged cell holds an entire 77k-character sub-document across 738 paragraphs and 5 nested tables."
|
||||
},
|
||||
"qa_xlsx": {
|
||||
"filename": "fire14050319.xlsx",
|
||||
"why": "Two-column Q&A sheet (q/a) plus a dead second sheet."
|
||||
},
|
||||
"branches_xlsx": {
|
||||
"filename": "مشخصات شعب 28 اردیبهشت 1405.xlsx",
|
||||
"why": "Branch directory: a merged title row above a two-row-tall merged header, and a vertically merged province column spanning each province's branches."
|
||||
}
|
||||
}
|
||||
52
tests/support/documents.py
Normal file
52
tests/support/documents.py
Normal file
@@ -0,0 +1,52 @@
|
||||
"""Load real production documents as test fixtures.
|
||||
|
||||
Real files rather than generated ones: a fixture built with the same
|
||||
understanding of the format that produced the parser cannot catch a wrong
|
||||
mental model -- it is wrong in the same way, so the test passes and production
|
||||
breaks. Every table shape this parser handles was found by reading these
|
||||
files, not by reasoning about what DOCX can contain.
|
||||
|
||||
The documents live outside the repository so no customer-facing file enters
|
||||
git history. Point `TEST_DOCUMENTS_DIR` at the directory holding them; tests
|
||||
that need one skip when it is absent.
|
||||
|
||||
These skips cover a missing *external input*, never a failing assertion -- but
|
||||
a machine without the directory does run a smaller suite than CI should.
|
||||
|
||||
Filenames live in `documents.json` rather than in this module: they are mostly
|
||||
Persian, and Arabic letterforms in Python source are flagged as ambiguous with
|
||||
Latin lookalikes (RUF001). They are data, so they belong in a data file.
|
||||
"""
|
||||
|
||||
import json
|
||||
import os
|
||||
from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
DOCUMENTS_DIR = Path(
|
||||
os.environ.get("TEST_DOCUMENTS_DIR", "~/Documents/asia_data_sample")
|
||||
).expanduser()
|
||||
|
||||
_MANIFEST = json.loads((Path(__file__).parent / "documents.json").read_text("utf-8"))
|
||||
|
||||
# Each entry is the only sample exercising some structure; `why` in the
|
||||
# manifest records what would go untested without it.
|
||||
PROSE_DOCX = _MANIFEST["prose_docx"]["filename"]
|
||||
LIST_DOCX = _MANIFEST["list_docx"]["filename"]
|
||||
TABLE_DOCX = _MANIFEST["table_docx"]["filename"]
|
||||
MIXED_DOCX = _MANIFEST["mixed_docx"]["filename"]
|
||||
LAYOUT_DOCX = _MANIFEST["layout_docx"]["filename"]
|
||||
QA_XLSX = _MANIFEST["qa_xlsx"]["filename"]
|
||||
BRANCHES_XLSX = _MANIFEST["branches_xlsx"]["filename"]
|
||||
|
||||
|
||||
def load_document(filename: str) -> bytes:
|
||||
"""Return a real document's bytes, skipping the test when it is unavailable."""
|
||||
path = DOCUMENTS_DIR / filename
|
||||
if not path.is_file():
|
||||
pytest.skip(
|
||||
f"{filename} not found in {DOCUMENTS_DIR}; "
|
||||
f"set TEST_DOCUMENTS_DIR to the directory holding the sample documents"
|
||||
)
|
||||
return path.read_bytes()
|
||||
Reference in New Issue
Block a user