fix(ipynb): strip the UTF-8 BOM so a notebook is not emitted as raw JSON (#2425)

This commit is contained in:
kevin
2026-09-09 11:33:59 -07:00
committed by GitHub
parent 481f22a9a9
commit 2e71e117ea
2 changed files with 71 additions and 0 deletions
@@ -54,6 +54,11 @@ class IpynbConverter(DocumentConverter):
# Parse and convert the notebook
encoding = stream_info.charset or "utf-8"
notebook_content = file_stream.read().decode(encoding=encoding)
# Editors and Windows tooling prepend a UTF-8 BOM, which json rejects
# outright; strip it so the notebook is read as a notebook.
notebook_content = notebook_content.lstrip("\ufeff")
return self._convert(json.loads(notebook_content))
def _convert(self, notebook_content: dict) -> DocumentConverterResult:
@@ -1412,6 +1412,72 @@ def test_ipynb_heading_title_preserves_leading_hash() -> None:
assert result.title == "#hashtag campaign results"
_UTF8_BOM = b"\xef\xbb\xbf"
_BOM_NOTEBOOK = {
"nbformat": 4,
"nbformat_minor": 5,
"metadata": {},
"cells": [
{"cell_type": "markdown", "source": ["# Quarterly notes\n"], "metadata": {}},
{"cell_type": "code", "source": ["print('hello')\n"], "metadata": {}},
],
}
def test_ipynb_with_a_utf8_bom_is_still_converted() -> None:
"""A leading BOM must not stop the notebook from being parsed."""
from markitdown.converters._ipynb_converter import IpynbConverter
data = _UTF8_BOM + json.dumps(_BOM_NOTEBOOK).encode("utf-8")
result = IpynbConverter().convert(
io.BytesIO(data), StreamInfo(extension=".ipynb", charset="utf-8")
)
assert "# Quarterly notes" in result.markdown
assert "```python\nprint('hello')" in result.markdown
assert result.title == "Quarterly notes"
def test_ipynb_with_a_utf8_bom_is_not_emitted_as_raw_json() -> None:
"""The whole stack must not fall through to the plain-text converter."""
data = _UTF8_BOM + json.dumps(_BOM_NOTEBOOK).encode("utf-8")
result = MarkItDown().convert_stream(
io.BytesIO(data), stream_info=StreamInfo(extension=".ipynb")
)
assert result.markdown.startswith("# Quarterly notes")
assert "nbformat" not in result.markdown
def test_ipynb_without_a_bom_is_unchanged() -> None:
"""The case that already worked must produce exactly the same markdown."""
from markitdown.converters._ipynb_converter import IpynbConverter
data = json.dumps(_BOM_NOTEBOOK).encode("utf-8")
result = IpynbConverter().convert(
io.BytesIO(data), StreamInfo(extension=".ipynb", charset="utf-8")
)
assert result.markdown == "# Quarterly notes\n\n\n```python\nprint('hello')\n\n```"
def test_ipynb_utf8_sig_charset_still_works() -> None:
"""A charset that already consumes the BOM must not be double-stripped."""
from markitdown.converters._ipynb_converter import IpynbConverter
data = _UTF8_BOM + json.dumps(_BOM_NOTEBOOK).encode("utf-8")
result = IpynbConverter().convert(
io.BytesIO(data), StreamInfo(extension=".ipynb", charset="utf-8-sig")
)
assert "# Quarterly notes" in result.markdown
def test_ipynb_accepts_non_ascii() -> None:
"""IpynbConverter.accepts() must not raise on non-ASCII binary content."""
from markitdown.converters._ipynb_converter import IpynbConverter