xlsongc commented on code in PR #73926:
URL: https://github.com/apache/airflow/pull/73926#discussion_r4227512254
##########
providers/common/ai/src/airflow/providers/common/ai/operators/document_loader.py:
##########
@@ -450,21 +453,68 @@ def _parse_pdf_stream(self, stream: BinaryIO) ->
list[dict[str, Any]]:
documents.append({"text": text, "metadata": {"page_number":
page_num + 1}})
return documents
- def _parse_docx_stream(self, stream: BinaryIO) -> list[dict[str, Any]]:
+ def _parse_docx_stream(self, stream: BinaryIO, *, source_hint: str) ->
list[dict[str, Any]]:
"""
- Parse a DOCX stream into documents.
+ Parse a DOCX stream into a single document.
- Extracts paragraph text only. Tables, headers, footers, and footnotes
- are not included. For richer DOCX parsing, plug in a dedicated
- extraction tool (``Unstructured``, ``docling``) as a custom parser
- backend.
+ Paragraphs and tables in the document body are extracted in document
+ order. Each table row becomes one "| cell | cell |" line, and a nested
+ table is flattened into its cell. Headers, footers, footnotes, content
+ controls, text boxes, and pending tracked insertions are not included.
"""
try:
from docx import Document
+ from docx.table import Table
except ImportError as e:
raise AirflowOptionalProviderFeatureException(e)
doc = Document(stream)
- paragraphs = [p.text for p in doc.paragraphs if p.text.strip()]
- text = "\n\n".join(paragraphs)
+ blocks = []
+ for block in doc.iter_inner_content():
+ if isinstance(block, Table):
+ try:
+ rows = self._get_docx_table_rows(block, Table)
+ except (ValueError, RecursionError) as e:
+ # python-docx walks vertical merges up recursively: a
malformed merge raises
+ # ValueError and a merge spanning about 1000 rows raises
RecursionError.
+ self.log.warning(
+ "Skipping a table in %s that python-docx could not
read: %r", source_hint, e
+ )
+ continue
+ text = "\n".join(f"| {' | '.join(cells)} |" for cells in rows)
+ else:
+ text = block.text
+ if text.strip():
+ blocks.append(text)
+ text = "\n\n".join(blocks)
return [{"text": text, "metadata": {}}]
+
+ def _get_docx_table_rows(self, table: Table, table_cls: type[Table]) ->
list[list[str]]:
+ rows = []
+ for row in table.rows:
+ cells: list[str] = [""] * row.grid_cols_before
Review Comment:
Done. `grid_cols_before` and `grid_cols_after` are now capped at
`len(table.columns)`. A new parametrized test sets `w:val="1000"` on
`gridBefore`, `gridAfter` and `gridSpan` in a two-column table, and it fails
without the cap.
--
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.
To unsubscribe, e-mail: [email protected]
For queries about this service, please contact Infrastructure at:
[email protected]