Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
2 changes: 2 additions & 0 deletions docling/cli/export_utils.py
Original file line number Diff line number Diff line change
Expand Up @@ -17,6 +17,7 @@
OutputFormat.VTT,
OutputFormat.DOCLANG,
OutputFormat.CHUNKS,
OutputFormat.LATEX,
}
)

Expand Down Expand Up @@ -51,6 +52,7 @@ def _export_flags_from_formats(to_formats: list[OutputFormat]) -> dict[str, bool
"export_doclang": OutputFormat.DOCLANG in to_formats,
"export_dclx": OutputFormat.DCLX in to_formats,
"export_chunks": OutputFormat.CHUNKS in to_formats,
"export_latex": OutputFormat.LATEX in to_formats,
}


Expand Down
10 changes: 10 additions & 0 deletions docling/cli/main.py
Original file line number Diff line number Diff line change
Expand Up @@ -40,6 +40,7 @@
HTMLOutputStyle,
HTMLParams,
)
from docling_core.transforms.serializer.latex import LaTeXDocSerializer
from docling_core.transforms.visualizer.layout_visualizer import LayoutVisualizer
from docling_core.types.doc import ImageRefMode
from docling_core.utils.file import resolve_source_to_path
Expand Down Expand Up @@ -462,6 +463,7 @@ def export_documents(
image_export_mode: ImageRefMode,
export_dclx: bool = False,
export_chunks: bool = False,
export_latex: bool = False,
chunker_type: ChunkerType = ChunkerType.HYBRID,
chunk_max_tokens: int | None = None,
chunk_tokenizer: str = "sentence-transformers/all-MiniLM-L6-v2",
Expand Down Expand Up @@ -609,6 +611,14 @@ def export_documents(
_log.info(f"writing DCLX output to {fname}")
conv_res.document.save_as_doclang_archive(filename=fname)

# Export LaTeX format:
if export_latex:
fname = output_dir / f"{doc_filename}.tex"
_log.info(f"writing LaTeX output to {fname}")
ser_res = LaTeXDocSerializer(doc=conv_res.document).serialize()
with fname.open("w", encoding="utf-8") as fp:
fp.write(ser_res.text)

# Export Chunks format:
if export_chunks and chunker_obj is not None:
fname = output_dir / f"{doc_filename}.chunks.jsonl"
Expand Down
1 change: 1 addition & 0 deletions docling/datamodel/base_models.py
Original file line number Diff line number Diff line change
Expand Up @@ -142,6 +142,7 @@ class OutputFormat(str, Enum):
DOCLANG = "doclang"
DCLX = "dclx"
CHUNKS = "chunks"
LATEX = "latex"


FormatToExtensions: dict[InputFormat, list[str]] = {
Expand Down
1 change: 1 addition & 0 deletions docs/usage/supported_formats.md
Original file line number Diff line number Diff line change
Expand Up @@ -51,3 +51,4 @@ Schema-specific support:
| WebVTT | Web Video Text Tracks format for displaying timed text |
| DocLang archive | Zipped DocLang bundle including page images; CLI output format: `dclx` |
| Chunks (JSONL) | Chunked document output for RAG pipelines; configurable via `--chunks-type`, `--chunks-max-tokens`, `--chunks-tokenizer` |
| LaTeX | Standalone `.tex` document; images are emitted as placeholders |
33 changes: 33 additions & 0 deletions tests/test_cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -156,6 +156,39 @@ def test_cli_exports_dclx(tmp_path):
assert b"DCLX CLI" in payload


def test_cli_exports_latex(tmp_path):
source = tmp_path / "input.md"
source.write_text(
"# LaTeX CLI\n\nHello from Markdown with 100% special chars.",
encoding="utf-8",
)
output = tmp_path / "out"

result = runner.invoke(
app,
[
str(source),
"--from",
"md",
"--to",
"latex",
"--output",
str(output),
],
)

assert result.exit_code == 0
converted = output / "input.tex"
assert converted.exists()
content = converted.read_text(encoding="utf-8")
assert content.startswith("\\documentclass")
assert "\\begin{document}" in content
assert "\\end{document}" in content
assert "LaTeX CLI" in content
# LaTeX special characters must be escaped
assert "100\\% special chars" in content


def test_cli_from_odf_expands_to_open_document_formats(
tmp_path: Path, monkeypatch: pytest.MonkeyPatch
) -> None:
Expand Down
Loading