diff --git a/CHANGELOG.md b/CHANGELOG.md index d40b54f..dfb023a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Index ePub documents for full-text search, with a new `get_epub_index_data` helper and automatic indexing of ePub items, like PDFs already are (#333) + ### Fixed - Carry a script's top-level `const`, `let` and `class` declarations across the wombat block, so other scripts on the page can still see them (#329) diff --git a/src/zimscraperlib/zim/indexing.py b/src/zimscraperlib/zim/indexing.py index b23cb5b..3413e15 100644 --- a/src/zimscraperlib/zim/indexing.py +++ b/src/zimscraperlib/zim/indexing.py @@ -62,8 +62,17 @@ def content(self, value: str): "lcms: not an ICC profile, invalid signature.", "format error: cmsOpenProfileFromMem failed", "ignoring broken ICC profile", + # MuPDF only knows EPUB 2 but reads EPUB 3 documents fine + "unknown epub version: 3.0", ] +# messages carrying a document-specific detail, ignored by prefix; images and +# styles do not matter for the text content +IGNORED_MUPDF_MESSAGE_PREFIXES = ( + "html: cannot load image", + "syntax error: css syntax error", +) + def get_pdf_index_data( *, @@ -75,6 +84,34 @@ def get_pdf_index_data( PDF can be passed either as content or fileobject or filepath """ + return _get_document_index_data( + filetype="pdf", content=content, fileobj=fileobj, filepath=filepath + ) + + +def get_epub_index_data( + *, + content: str | bytes | None = None, + fileobj: io.BytesIO | None = None, + filepath: pathlib.Path | None = None, +) -> IndexData: + """Returns the IndexData information for a given EPUB + + EPUB can be passed either as content or fileobject or filepath + """ + return _get_document_index_data( + filetype="epub", content=content, fileobj=fileobj, filepath=filepath + ) + + +def _get_document_index_data( + *, + filetype: str, + content: str | bytes | None = None, + fileobj: io.BytesIO | None = None, + filepath: pathlib.Path | None = None, +) -> IndexData: + """IndexData of any document PyMuPDF can read, title built from its metadata""" # do not display all pymupdf errors, we will filter them afterwards pymupdf.TOOLS.mupdf_display_errors( # pyright: ignore[reportUnknownMemberType] @@ -82,26 +119,26 @@ def get_pdf_index_data( ) if content: - doc = pymupdf.open(stream=content) + doc = pymupdf.open(stream=content, filetype=filetype) elif fileobj: - doc = pymupdf.open(stream=fileobj) + doc = pymupdf.open(stream=fileobj, filetype=filetype) else: - doc = pymupdf.open(filename=filepath) + doc = pymupdf.open(filename=filepath, filetype=filetype) metadata = ( # pyright: ignore[reportUnknownVariableType, reportUnknownMemberType] doc.metadata ) title = "" - if metadata: # pragma: no branch (always metadata in test PDFs) + if metadata: # pragma: no branch (always metadata in test documents) parts: list[str] = [] for key in ["title", "author", "subject"]: if metadata.get(key): # pyright: ignore[reportUnknownMemberType] parts.append( metadata[key] # pyright: ignore[reportUnknownArgumentType] ) - if parts: # pragma: no branch (always metadata in test PDFs) + if parts: # pragma: no branch (always metadata in test documents) title = " - ".join(parts) - def get_pdf_content(page: pymupdf.Page) -> str: + def get_page_content(page: pymupdf.Page) -> str: text = ( # pyright: ignore[reportUnknownVariableType] page.get_text() # pyright: ignore[reportUnknownMemberType] ) @@ -109,7 +146,7 @@ def get_pdf_content(page: pymupdf.Page) -> str: raise Exception("Unexpected text content") return text - content = "\n".join(get_pdf_content(page) for page in doc) + content = "\n".join(get_page_content(page) for page in doc) # build list of messages and filter messages which are known to not be relevant # in our use-case @@ -117,12 +154,13 @@ def get_pdf_content(page: pymupdf.Page) -> str: warning for warning in pymupdf.TOOLS.mupdf_warnings().splitlines() if warning not in IGNORED_MUPDF_MESSAGES + and not warning.startswith(IGNORED_MUPDF_MESSAGE_PREFIXES) ) if mupdf_messages: logger.warning( f"PyMuPDF issues:\n{mupdf_messages}" - ) # pragma: no cover (no known error in test PDFs) + ) # pragma: no cover (no known error in test documents) return IndexData( title=title, diff --git a/src/zimscraperlib/zim/items.py b/src/zimscraperlib/zim/items.py index f0f4650..418cc8a 100644 --- a/src/zimscraperlib/zim/items.py +++ b/src/zimscraperlib/zim/items.py @@ -12,7 +12,11 @@ from zimscraperlib.download import stream_file from zimscraperlib.filesystem import get_content_mimetype, get_file_mimetype -from zimscraperlib.zim.indexing import IndexData, get_pdf_index_data +from zimscraperlib.zim.indexing import ( + IndexData, + get_epub_index_data, + get_pdf_index_data, +) from zimscraperlib.zim.providers import ( FileLikeProvider, FileProvider, @@ -75,8 +79,8 @@ class StaticItem(Item): the Item and we can be notified that we're effectively through with our content By default, content is automatically indexed (either by the libzim itself for - supported documents - text or html for now or by the python-scraperlib - only PDF - supported for now). If you do not want this, set `auto_index` to False to disable + supported documents - text or html for now or by the python-scraperlib - PDF and + ePub for now). If you do not want this, set `auto_index` to False to disable both indexing (libzim and python-scraperlib). It is also possible to pass index_data to configure custom indexing of the item. @@ -160,6 +164,9 @@ def _get_auto_index(self): if mimetype == "application/pdf": index_data = get_pdf_index_data(content=content) self.get_indexdata = lambda: index_data + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(content=content) + self.get_indexdata = lambda: index_data else: return @@ -174,6 +181,9 @@ def _get_auto_index(self): if mimetype == "application/pdf": index_data = get_pdf_index_data(fileobj=fileobj) self.get_indexdata = lambda: index_data + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(fileobj=fileobj) + self.get_indexdata = lambda: index_data else: return @@ -190,6 +200,11 @@ def _get_auto_index(self): self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride] lambda: index_data ) + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(filepath=filepath) + self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride] + lambda: index_data + ) else: return diff --git a/tests/conftest.py b/tests/conftest.py index 817e004..af4a3c7 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,4 +1,5 @@ import pathlib +import zipfile import pytest import requests @@ -240,3 +241,67 @@ def undecodable_byte_stream() -> bytes: b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" ) + + +@pytest.fixture(scope="session") +def epub_file(tmpdir_factory: pytest.TempdirFactory) -> pathlib.Path: + """Minimal EPUB 3 with title/author metadata and two chapters""" + dst = pathlib.Path(tmpdir_factory.mktemp("data") / "minimal.epub") + # a missing image and a CSS selector MuPDF does not support, as found in + # real-world ePubs, make MuPDF warn without affecting the text + chapter = ( + '' + '{title}' + '' + "

{title}

{text}

" + '' + ) + with zipfile.ZipFile(dst, "w") as epub: + # mimetype must be the first entry, uncompressed + epub.writestr("mimetype", "application/epub+zip", zipfile.ZIP_STORED) + epub.writestr( + "META-INF/container.xml", + '' + '' + '', + ) + epub.writestr( + "OEBPS/content.opf", + '' + '' + '' + 'urn:uuid:7c2b1a52-1d4e-4b0e-9c55-3f7e1d2a9b10' + "" + "Lighthouse Almanac" + "Ada Quillfeather" + "en" + '2026-01-01T00:00:00Z' + "" + '' + '' + '' + '' + "", + ) + epub.writestr( + "OEBPS/nav.xhtml", + '' + 'Contents' + '', + ) + epub.writestr( + "OEBPS/c1.xhtml", + chapter.format(title="Storm", text="The keeper trimmed the wick."), + ) + epub.writestr( + "OEBPS/c2.xhtml", + chapter.format(title="Harbour", text="Gulls circled the breakwater."), + ) + return dst diff --git a/tests/zim/test_indexing.py b/tests/zim/test_indexing.py index d8bf48e..3b1bed2 100644 --- a/tests/zim/test_indexing.py +++ b/tests/zim/test_indexing.py @@ -3,9 +3,14 @@ import libzim.writer # pyright: ignore[reportMissingModuleSource] import pytest +from pytest_mock import MockerFixture -from zimscraperlib.zim import Archive, Creator -from zimscraperlib.zim.indexing import IndexData, get_pdf_index_data +from zimscraperlib.zim import Archive, Creator, indexing +from zimscraperlib.zim.indexing import ( + IndexData, + get_epub_index_data, + get_pdf_index_data, +) from zimscraperlib.zim.items import StaticItem @@ -368,3 +373,46 @@ def test_index_data_wordcount( IndexData(title="foo", content=content, wordcount=wordcount).get_wordcount() == expected_wordcount ) + + +@pytest.mark.parametrize("source", ["content", "fileobj", "filepath"]) +def test_get_epub_index_data( + source: str, epub_file: pathlib.Path, mocker: MockerFixture +): + warning = mocker.patch.object(indexing.logger, "warning") + if source == "content": + index_data = get_epub_index_data(content=epub_file.read_bytes()) + elif source == "fileobj": + index_data = get_epub_index_data(fileobj=io.BytesIO(epub_file.read_bytes())) + else: + index_data = get_epub_index_data(filepath=epub_file) + assert index_data.get_title() == "Lighthouse Almanac - Ada Quillfeather" + assert "The keeper trimmed the wick." in index_data.get_content() + assert "Gulls circled the breakwater." in index_data.get_content() + assert index_data.has_indexdata() + # MuPDF does not know EPUB 3 but reads it fine, this must not be reported + warning.assert_not_called() + + +@pytest.mark.parametrize("source", ["content", "fileobj", "filepath"]) +def test_indexing_item_epub( + tmp_path: pathlib.Path, epub_file: pathlib.Path, source: str +): + """An EPUB item can be automatically indexed""" + fpath = tmp_path / "test.zim" + hints: dict[libzim.writer.Hint, int] = {libzim.writer.Hint.FRONT_ARTICLE: True} + if source == "content": + item = StaticItem(content=epub_file.read_bytes(), path="welcome", hints=hints) + elif source == "fileobj": + fileobj = io.BytesIO(epub_file.read_bytes()) + item = StaticItem(fileobj=fileobj, path="welcome", hints=hints) + else: + item = StaticItem(filepath=epub_file, path="welcome", hints=hints) + with Creator(fpath, "welcome").config_dev_metadata() as creator: + creator.add_item(item) + + reader = Archive(fpath) + assert "welcome" in list(reader.get_suggestions("lighthouse")) + assert "welcome" in list(reader.get_suggestions("quillfeather")) + assert reader.get_search_results_count("breakwater") == 1 + assert reader.get_search_results_count("wick") == 1