diff --git a/CHANGELOG.md b/CHANGELOG.md index 5f1d11f8f6..56928761c7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,3 +1,9 @@ +## 0.25.2 + +### Fixes + +- **Extract tables nested in DOCX text boxes**: `partition_docx()` now finds tables stored inside `w:txbxContent`, as produced when Pages exports a spreadsheet table to Word. Equivalent `mc:Choice` and `mc:Fallback` representations are treated as alternatives so the visible table is emitted once rather than duplicated; ordinary shape text remains ignored as before. + ## 0.25.1 ### Fixes diff --git a/test_unstructured/partition/test_docx.py b/test_unstructured/partition/test_docx.py index 6911fefc52..c7cf84fd30 100644 --- a/test_unstructured/partition/test_docx.py +++ b/test_unstructured/partition/test_docx.py @@ -9,12 +9,14 @@ import pathlib import re import tempfile +from copy import deepcopy from typing import Any, Iterator import docx import pytest from docx.document import Document from docx.text.paragraph import Paragraph +from lxml import etree from pytest_mock import MockFixture from test_unstructured.unit_utils import ( @@ -149,6 +151,92 @@ def test_partition_docx_processes_table(): assert elements[0].metadata.filename == "fake_table.docx" +def test_partition_docx_processes_table_in_textbox_once(): + """Tables in equivalent AlternateContent branches produce one Table element.""" + document = docx.Document() + table = document.add_table(rows=2, cols=2) + table.cell(0, 0).text = "Name" + table.cell(0, 1).text = "Role" + table.cell(1, 0).text = "Alice" + table.cell(1, 1).text = "Engineer" + + table_element = table._tbl + table_element.getparent().remove(table_element) + + mc_namespace = "http://schemas.openxmlformats.org/markup-compatibility/2006" + w_namespace = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + wps_namespace = "http://schemas.microsoft.com/office/word/2010/wordprocessingShape" + alternate_content = etree.Element( + f"{{{mc_namespace}}}AlternateContent", + nsmap={"mc": mc_namespace, "w": w_namespace, "wps": wps_namespace}, + ) + choice = etree.SubElement(alternate_content, f"{{{mc_namespace}}}Choice") + choice.set("Requires", "wps") + fallback = etree.SubElement(alternate_content, f"{{{mc_namespace}}}Fallback") + for branch in (choice, fallback): + textbox_content = etree.SubElement(branch, f"{{{w_namespace}}}txbxContent") + textbox_content.append(deepcopy(table_element)) + + document.add_paragraph().add_run()._r.append(alternate_content) + file = io.BytesIO() + document.save(file) + file.seek(0) + + elements = partition_docx(file=file, infer_table_structure=True) + + assert [type(element) for element in elements] == [Table] + assert elements[0].text == "Name Role Alice Engineer" + assert elements[0].metadata.text_as_html == ( + "
NameRole
AliceEngineer
" + ) + + +@pytest.mark.parametrize("choice_requires", ["unsupported", None, ""]) +def test_partition_docx_uses_textbox_table_fallback_for_unsupported_choice( + choice_requires: str | None, +): + """An unsupported or invalid AlternateContent Choice does not hide its Fallback.""" + document = docx.Document() + choice_table = document.add_table(rows=1, cols=1) + choice_table.cell(0, 0).text = "Unsupported choice" + fallback_table = document.add_table(rows=1, cols=1) + fallback_table.cell(0, 0).text = "Compatible fallback" + + choice_table_element = choice_table._tbl + choice_table_element.getparent().remove(choice_table_element) + fallback_table_element = fallback_table._tbl + fallback_table_element.getparent().remove(fallback_table_element) + + mc_namespace = "http://schemas.openxmlformats.org/markup-compatibility/2006" + w_namespace = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" + unsupported_namespace = "urn:example:unsupported-word-feature" + alternate_content = etree.Element( + f"{{{mc_namespace}}}AlternateContent", + nsmap={"mc": mc_namespace, "w": w_namespace, "unsupported": unsupported_namespace}, + ) + choice = etree.SubElement(alternate_content, f"{{{mc_namespace}}}Choice") + if choice_requires is not None: + choice.set("Requires", choice_requires) + choice_textbox = etree.SubElement(choice, f"{{{w_namespace}}}txbxContent") + choice_textbox.append(choice_table_element) + fallback = etree.SubElement(alternate_content, f"{{{mc_namespace}}}Fallback") + fallback_textbox = etree.SubElement(fallback, f"{{{w_namespace}}}txbxContent") + fallback_textbox.append(fallback_table_element) + + document.add_paragraph().add_run()._r.append(alternate_content) + file = io.BytesIO() + document.save(file) + file.seek(0) + + elements = partition_docx(file=file, infer_table_structure=True) + + assert [type(element) for element in elements] == [Table] + assert elements[0].text == "Compatible fallback" + assert elements[0].metadata.text_as_html == ( + "
Compatible fallback
" + ) + + def test_partition_docx_grabs_header_and_footer(): elements = partition_docx(example_doc_path("handbook-1p.docx")) diff --git a/unstructured/__version__.py b/unstructured/__version__.py index c7cc966b51..b20e30fc9b 100644 --- a/unstructured/__version__.py +++ b/unstructured/__version__.py @@ -1 +1 @@ -__version__ = "0.25.1" # pragma: no cover +__version__ = "0.25.2" # pragma: no cover diff --git a/unstructured/partition/docx.py b/unstructured/partition/docx.py index 1347d88134..46141b7773 100644 --- a/unstructured/partition/docx.py +++ b/unstructured/partition/docx.py @@ -16,6 +16,7 @@ from docx.enum.section import WD_SECTION_START from docx.oxml.table import CT_Tbl from docx.oxml.text.paragraph import CT_P +from docx.oxml.xmlchemy import BaseOxmlElement from docx.section import Section, _Footer, _Header from docx.table import Table as DocxTable from docx.table import _Cell, _Row @@ -91,7 +92,17 @@ DETECTION_ORIGIN: str = "docx" # -- CT_* stands for "complex-type", an XML element type in docx parlance -- BlockElement: TypeAlias = "CT_P | CT_Tbl" -BlockItem: TypeAlias = "Paragraph | DocxTable" + +_MC_NAMESPACE = "http://schemas.openxmlformats.org/markup-compatibility/2006" +_MC_CHOICE_TAG = f"{{{_MC_NAMESPACE}}}Choice" +_MC_FALLBACK_TAG = f"{{{_MC_NAMESPACE}}}Fallback" +_WORDPROCESSINGML_NAMESPACE = "http://schemas.openxmlformats.org/wordprocessingml/2006/main" +_WORDPROCESSING_SHAPE_NAMESPACE = ( + "http://schemas.microsoft.com/office/word/2010/wordprocessingShape" +) +_SUPPORTED_MC_CHOICE_NAMESPACES = frozenset( + {_WORDPROCESSINGML_NAMESPACE, _WORDPROCESSING_SHAPE_NAMESPACE} +) def register_picture_partitioner(picture_partitioner: PicturePartitionerT) -> None: @@ -432,6 +443,64 @@ def _iter_sectionless_document_elements(self) -> Iterator[Element]: elif isinstance(block_item, DocxTable): # pyright: ignore[reportUnnecessaryIsInstance] yield from self._iter_table_element(block_item) + @staticmethod + def _is_supported_alternate_content_choice(choice: BaseOxmlElement) -> bool: + """True when `choice` has a valid `Requires` list and all its namespaces are supported. + + ECMA-376 Part 3 section 7.6 requires one or more namespace prefixes. Treat a missing or + empty attribute as unsupported so a conforming Fallback can be selected instead. + """ + requires = choice.get("Requires") + if not requires: + return False + + return all( + choice.nsmap.get(prefix) in _SUPPORTED_MC_CHOICE_NAMESPACES + for prefix in requires.split() + ) + + @staticmethod + def _is_active_alternate_content_branch(block: BlockElement) -> bool: + """True when `block` is in the selected branch of any enclosing AlternateContent. + + Pages exports text boxes in both `mc:Choice` and `mc:Fallback`, representing the same + visible content twice. Select the first Choice whose required namespaces are supported, + otherwise select the Fallback, so equivalent representations do not produce duplicates. + """ + for ancestor in block.iterancestors(): + if ancestor.tag not in {_MC_CHOICE_TAG, _MC_FALLBACK_TAG}: + continue + + alternate_content = ancestor.getparent() + selected_branch = next( + ( + child + for child in alternate_content + if child.tag == _MC_CHOICE_TAG + and _DocxPartitioner._is_supported_alternate_content_choice(child) + ), + None, + ) + if selected_branch is None: + selected_branch = next( + (child for child in alternate_content if child.tag == _MC_FALLBACK_TAG), + None, + ) + + if ancestor is not selected_branch: + return False + + return True + + def _iter_textbox_tables(self, paragraph: Paragraph) -> Iterator[DocxTable]: + """Generate tables nested in text boxes in `paragraph`.""" + tables: list[CT_Tbl] = paragraph._p.xpath(".//w:txbxContent/w:tbl") + + for table in tables: + if not self._is_active_alternate_content_branch(table): + continue + yield DocxTable(table, paragraph) + def _classify_paragraph_to_element(self, paragraph: Paragraph) -> Iterator[Element]: """Generate zero-or-one document element for `paragraph`. @@ -640,6 +709,11 @@ def iter_paragraph_items(paragraph: Paragraph) -> Iterator[Paragraph | RenderedP else: yield from self._opts.increment_page_number() + # -- Text-box tables are nested in a paragraph's drawing XML and are not returned by + # -- python-docx's body/section iterators. Shape text remains intentionally ignored. + for table in self._iter_textbox_tables(paragraph): + yield from self._iter_table_element(table) + def _iter_paragraph_emphasis(self, paragraph: Paragraph) -> Iterator[dict[str, str]]: """Generate e.g. {"text": "MUST", "tag": "b"} for each emphasis in `paragraph`.""" for run in paragraph.runs: