Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 9 additions & 0 deletions packages/markitdown/src/markitdown/_markitdown.py
Original file line number Diff line number Diff line change
Expand Up @@ -734,6 +734,15 @@ def _get_stream_info_guesses(
if charset_result is not None:
charset = self._normalize_charset(charset_result.encoding)

# Detection only saw this first page. If the stream continues
# past it, non-ASCII bytes may appear later and decoding the
# whole stream as ASCII would fail. ASCII is a strict subset of
# UTF-8, so widening the guess still decodes every byte ASCII
# would, plus any multi-byte runs further in. A stream that ends
# within the page was fully inspected, so ASCII stands.
if charset == "ascii" and len(stream_page) == 4096:
charset = "utf-8"

# Normalize the first extension listed
guessed_extension = None
if len(result.prediction.output.extensions) > 0:
Expand Down
25 changes: 25 additions & 0 deletions packages/markitdown/src/markitdown/converter_utils/charset.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,25 @@
from typing import Optional

from charset_normalizer import from_bytes


def decode_text(data: bytes, charset: Optional[str]) -> str:
"""
Decode bytes to text, re-detecting the charset when the supplied one does not fit.

Charset detection inspects only the first 4k of a stream, so a file whose non-ASCII
bytes appear later can be labeled with a charset that fails on the full content.
Re-running detection over every byte recovers those files instead of raising
UnicodeDecodeError.
"""
if charset:
try:
return data.decode(charset)
except UnicodeDecodeError:
pass

best_match = from_bytes(data).best()
if best_match is None:
# Detection found no viable encoding; keep the readable parts rather than failing.
return data.decode("utf-8", errors="replace")
return str(best_match)
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
import csv
import io
from typing import BinaryIO, Any
from charset_normalizer import from_bytes
from .._base_converter import DocumentConverter, DocumentConverterResult
from .._stream_info import StreamInfo
from ..converter_utils.charset import decode_text

ACCEPTED_MIME_TYPE_PREFIXES = [
"text/csv",
Expand Down Expand Up @@ -42,10 +42,7 @@ def convert(
**kwargs: Any, # Options to pass to the converter
) -> DocumentConverterResult:
# Read the file content
if stream_info.charset:
content = file_stream.read().decode(stream_info.charset)
else:
content = str(from_bytes(file_stream.read()).best())
content = decode_text(file_stream.read(), stream_info.charset)

# Parse CSV content
reader = csv.reader(io.StringIO(content))
Expand Down
Original file line number Diff line number Diff line change
@@ -1,9 +1,9 @@
import sys

from typing import BinaryIO, Any
from charset_normalizer import from_bytes
from .._base_converter import DocumentConverter, DocumentConverterResult
from .._stream_info import StreamInfo
from ..converter_utils.charset import decode_text

# Try loading optional (but in this case, required) dependencies
# Save reporting of any exceptions for later
Expand Down Expand Up @@ -63,9 +63,6 @@ def convert(
stream_info: StreamInfo,
**kwargs: Any, # Options to pass to the converter
) -> DocumentConverterResult:
if stream_info.charset:
text_content = file_stream.read().decode(stream_info.charset)
else:
text_content = str(from_bytes(file_stream.read()).best())
text_content = decode_text(file_stream.read(), stream_info.charset)

return DocumentConverterResult(markdown=text_content)
79 changes: 79 additions & 0 deletions packages/markitdown/tests/test_charset_detection.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,79 @@
#!/usr/bin/env python3 -m pytest
import io

from markitdown import MarkItDown, StreamInfo
from markitdown.converter_utils.charset import decode_text

# Charset detection only inspects the first 4k of a stream, so the non-ASCII bytes in
# these fixtures are placed well past that boundary to exercise the misdetection path.
PADDING = "a" * 5000
ACCENTED_TEXT = "Considerações de segurança: acentuação e cedilha."


def test_utf8_bytes_after_detection_window() -> None:
"""A UTF-8 file that looks like ASCII in its first 4k must still convert."""
content = (PADDING + "\n" + ACCENTED_TEXT + "\n").encode("utf-8")

markitdown = MarkItDown()
result = markitdown.convert_stream(
io.BytesIO(content), stream_info=StreamInfo(extension=".md")
)

assert ACCENTED_TEXT in result.markdown


def test_csv_utf8_bytes_after_detection_window() -> None:
"""The same misdetection must not break the CSV converter."""
content = ("header\n" + PADDING + "\n" + ACCENTED_TEXT + "\n").encode("utf-8")

markitdown = MarkItDown()
result = markitdown.convert_stream(
io.BytesIO(content), stream_info=StreamInfo(extension=".csv")
)

assert ACCENTED_TEXT in result.markdown


def test_decode_text_recovers_from_wrong_charset() -> None:
"""A charset that fails on the full content must not raise.

'ascii' fits the first 4k but not the accented tail. Re-detection then picks the
encoding it judges best. Which one that is depends on charset_normalizer's
heuristics, so the guarantee under test is that decoding yields usable text
instead of raising UnicodeDecodeError.
"""
content = (PADDING + "\n" + ACCENTED_TEXT + "\n").encode("cp1252")

decoded = decode_text(content, "ascii")

assert isinstance(decoded, str)
assert PADDING in decoded


def test_decode_text_honors_a_valid_charset() -> None:
"""An explicit charset that does decode the content is used as given."""
content = ACCENTED_TEXT.encode("cp1252")

assert decode_text(content, "cp1252") == ACCENTED_TEXT


def test_decode_text_without_charset() -> None:
"""With no charset supplied, detection runs over the whole content."""
content = (PADDING + "\n" + ACCENTED_TEXT + "\n").encode("utf-8")

assert ACCENTED_TEXT in decode_text(content, None)


if __name__ == "__main__":
"""Runs this file's tests from the command line."""
for test in [
test_utf8_bytes_after_detection_window,
test_csv_utf8_bytes_after_detection_window,
test_decode_text_recovers_from_wrong_charset,
test_decode_text_honors_a_valid_charset,
test_decode_text_without_charset,
]:
print(f"Running {test.__name__}...", end="")
test()
print("OK")
print("All tests passed!")