xlsongc commented on code in PR #73926:
URL: https://github.com/apache/airflow/pull/73926#discussion_r4165508103


##########
providers/common/ai/src/airflow/providers/common/ai/operators/document_loader.py:
##########
@@ -452,19 +454,56 @@ def _parse_pdf_stream(self, stream: BinaryIO) -> 
list[dict[str, Any]]:
 
     def _parse_docx_stream(self, stream: BinaryIO) -> list[dict[str, Any]]:
         """
-        Parse a DOCX stream into documents.
+        Parse a DOCX stream into a single document.
 
-        Extracts paragraph text only. Tables, headers, footers, and footnotes
-        are not included. For richer DOCX parsing, plug in a dedicated
-        extraction tool (``Unstructured``, ``docling``) as a custom parser
-        backend.
+        Paragraphs and tables in the document body are extracted in document
+        order. Each table row becomes one "| cell | cell |" line, and a nested
+        table is flattened into its cell. Headers, footers, footnotes, and
+        content controls are not included.
         """
         try:
             from docx import Document
+            from docx.table import Table
         except ImportError as e:
             raise AirflowOptionalProviderFeatureException(e)
 
         doc = Document(stream)
-        paragraphs = [p.text for p in doc.paragraphs if p.text.strip()]
-        text = "\n\n".join(paragraphs)
+        blocks = []
+        for block in doc.iter_inner_content():
+            if isinstance(block, Table):
+                text = "\n".join(f"| {' | '.join(cells)} |" for cells in 
self._get_docx_table_rows(block))
+            else:
+                text = block.text
+            if text.strip():
+                blocks.append(text)
+        text = "\n\n".join(blocks)
         return [{"text": text, "metadata": {}}]
+
+    def _get_docx_table_rows(self, table: Table) -> list[list[str]]:
+        rows = []
+        for row in table.rows:
+            cells: list[str] = []
+            previous_cell = None
+            for cell in row.cells:
+                # python-docx repeats the same _Cell object for every grid 
column a
+                # horizontal merge spans. A vertical merge repeats on each row 
it spans.
+                if cell is previous_cell:
+                    continue
+                previous_cell = cell
+                cells.append(self._get_docx_cell_text(cell))
+            if any(cells):
+                rows.append(cells)
+        return rows
+
+    def _get_docx_cell_text(self, cell: _Cell) -> str:
+        from docx.table import Table

Review Comment:
   Done, `Table` is now passed down as `table_cls`.



-- 
This is an automated message from the Apache Git Service.
To respond to the message, please log on to GitHub and use the
URL above to go to the specific comment.

To unsubscribe, e-mail: [email protected]

For queries about this service, please contact Infrastructure at:
[email protected]

Reply via email to