Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
23 changes: 22 additions & 1 deletion haystack/components/converters/docx.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,10 @@
# Word writes a text box twice inside `mc:AlternateContent`: a `wps:txbx` under
# `mc:Choice` and the same text as a VML text box under `mc:Fallback`.
_FALLBACK_TAG = f"{{{_MARKUP_COMPATIBILITY_NS}}}Fallback"
# A block-level content control keeps its paragraphs and tables in `w:sdtContent`.
# Word uses them for cover pages, tables of contents and template fields.
_CONTENT_CONTROL_TAG = f"{{{_WORD_NS}}}sdt"
_CONTENT_CONTROL_CONTENT_TAG = f"{{{_WORD_NS}}}sdtContent"

# A Markdown table row ends at a line break and its columns are separated by pipes, so
# neither can survive inside a cell.
Expand Down Expand Up @@ -249,7 +253,7 @@ def _extract_elements(self, document: "DocxDocument") -> list[str]:
:returns: List of strings (paragraph texts and table representations) with page breaks added as '\f' characters.
"""
elements = []
for element in document.element.body:
for element in self._block_elements(document.element.body):
if isinstance(element, _Comment):
continue
if element.tag.endswith("p"):
Expand All @@ -271,6 +275,23 @@ def _extract_elements(self, document: "DocxDocument") -> list[str]:

return elements

def _block_elements(self, parent: Any) -> list[Any]:
"""
Lists the block-level children of an element, replacing each content control with the elements it holds.

:param parent: The element to list the children of, such as the document body.
:returns: List of child elements in reading order, with nested content controls expanded.
"""
elements = []
for element in parent:
if element.tag == _CONTENT_CONTROL_TAG:
content = element.find(_CONTENT_CONTROL_CONTENT_TAG)
if content is not None:
elements.extend(self._block_elements(content))
else:
elements.append(element)
return elements

def _extract_text_boxes(self, element: Any, document: "DocxDocument") -> list[str]:
"""
Extracts the text of any text box anchored to a paragraph.
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,6 @@
---
fixes:
- |
``DOCXToDocument`` now reads the paragraphs and tables inside block-level content controls (``w:sdt``),
including nested ones. Word uses them for cover pages, tables of contents and template fields, and their
content was dropped from the Document without any error.
48 changes: 42 additions & 6 deletions test/components/converters/test_docx_file_to_document.py
Original file line number Diff line number Diff line change
Expand Up @@ -65,9 +65,18 @@ def docx_converter():
"""


def _docx_with_text_box(template: str, content: str, link_url: str | None = None) -> bytes:
# A block-level content control, as Word writes for a cover page, a table of contents or a template field.
_BLOCK_CONTENT_CONTROL = """
<w:sdt {ns}>
<w:sdtPr><w:alias w:val="Summary"/></w:sdtPr>
<w:sdtContent>{content}</w:sdtContent>
</w:sdt>
"""


def _docx_with_block(template: str, content: str, link_url: str | None = None) -> bytes:
"""
Build a DOCX whose body is BEFORE, a text box holding `content`, then AFTER.
Build a DOCX whose body is BEFORE, the block from `template` holding `content`, then AFTER.

With `link_url`, `{rid}` in `content` becomes the id of an external hyperlink relationship to it.
"""
Expand Down Expand Up @@ -557,7 +566,7 @@ def test_no_link_extraction(self, test_files_path):
def test_run_reads_text_inside_a_text_box(self):
"""A text box keeps its paragraphs in `w:txbxContent`, which the anchoring
paragraph's own text never reaches."""
docx_bytes = _docx_with_text_box(_MODERN_TEXT_BOX, "<w:p><w:r><w:t>TEXT INSIDE A TEXT BOX</w:t></w:r></w:p>")
docx_bytes = _docx_with_block(_MODERN_TEXT_BOX, "<w:p><w:r><w:t>TEXT INSIDE A TEXT BOX</w:t></w:r></w:p>")

output = DOCXToDocument().run(sources=[ByteStream(data=docx_bytes)])

Expand All @@ -568,7 +577,7 @@ def test_run_reads_text_inside_a_text_box(self):

def test_run_does_not_repeat_a_text_box_that_has_a_vml_fallback(self):
"""Word writes the same text under `mc:Choice` and again under `mc:Fallback`."""
docx_bytes = _docx_with_text_box(_TEXT_BOX_WITH_VML_FALLBACK, "<w:p><w:r><w:t>CALLOUT TEXT</w:t></w:r></w:p>")
docx_bytes = _docx_with_block(_TEXT_BOX_WITH_VML_FALLBACK, "<w:p><w:r><w:t>CALLOUT TEXT</w:t></w:r></w:p>")

output = DOCXToDocument().run(sources=[ByteStream(data=docx_bytes)])

Expand All @@ -581,15 +590,15 @@ def test_run_reads_a_table_inside_a_text_box(self):
"<w:tc><w:p><w:r><w:t>CELL B</w:t></w:r></w:p></w:tc>"
"</w:tr></w:tbl>"
)
docx_bytes = _docx_with_text_box(_MODERN_TEXT_BOX, table_xml)
docx_bytes = _docx_with_block(_MODERN_TEXT_BOX, table_xml)

output = DOCXToDocument(table_format=DOCXTableFormat.CSV).run(sources=[ByteStream(data=docx_bytes)])

assert "CELL A,CELL B" in output["documents"][0].content

def test_run_formats_links_inside_a_text_box(self):
"""`link_format` has to reach a text box too."""
docx_bytes = _docx_with_text_box(
docx_bytes = _docx_with_block(
_MODERN_TEXT_BOX,
'<w:p><w:hyperlink r:id="{rid}"><w:r><w:t>LINK</w:t></w:r></w:hyperlink></w:p>',
link_url="https://example.com",
Expand All @@ -599,6 +608,33 @@ def test_run_formats_links_inside_a_text_box(self):

assert "[LINK](https://example.com)" in output["documents"][0].content

def test_run_reads_text_inside_a_content_control(self):
"""A block-level content control keeps its paragraphs in `w:sdtContent`, also when it is nested."""
docx_bytes = _docx_with_block(
_BLOCK_CONTENT_CONTROL,
"<w:p><w:r><w:t>COVER TITLE</w:t></w:r></w:p>"
"<w:sdt><w:sdtContent><w:p><w:r><w:t>NESTED FIELD</w:t></w:r></w:p></w:sdtContent></w:sdt>",
)

output = DOCXToDocument().run(sources=[ByteStream(data=docx_bytes)])

content = output["documents"][0].content
assert content.index("BEFORE") < content.index("COVER TITLE") < content.index("NESTED FIELD")
assert content.index("NESTED FIELD") < content.index("AFTER")

def test_run_reads_a_table_inside_a_content_control(self):
table_xml = (
"<w:tbl><w:tr>"
"<w:tc><w:p><w:r><w:t>CELL A</w:t></w:r></w:p></w:tc>"
"<w:tc><w:p><w:r><w:t>CELL B</w:t></w:r></w:p></w:tc>"
"</w:tr></w:tbl>"
)
docx_bytes = _docx_with_block(_BLOCK_CONTENT_CONTROL, table_xml)

output = DOCXToDocument(table_format=DOCXTableFormat.CSV).run(sources=[ByteStream(data=docx_bytes)])

assert "CELL A,CELL B" in output["documents"][0].content

@pytest.mark.parametrize("table_format", ["markdown", "csv"])
@pytest.mark.parametrize(
("link_format", "expected_link"),
Expand Down
Loading