From bcb11ae5d77521f94f482293ed82bac094f89232 Mon Sep 17 00:00:00 2001 From: Sriram-PR <160517347+Sriram-PR@users.noreply.github.com> Date: Wed, 7 Oct 2026 02:18:17 +1100 Subject: [PATCH] Index ePub documents for full-text search PyMuPDF, which is already used to index PDFs, also reads ePub. So the PDF indexing code becomes a shared helper taking the document type, with get_pdf_index_data and a new get_epub_index_data on top of it, and StaticItem auto-indexes application/epub+zip items the same way it does PDFs. MuPDF warns "unknown epub version: 3.0" on every EPUB 3 document while reading it fine, so that message joins the ignored ones. Warnings about images it cannot load and CSS it does not support are ignored by prefix, since they carry document-specific details and do not affect the text. Fix #333 --- CHANGELOG.md | 4 ++ src/zimscraperlib/zim/indexing.py | 54 +++++++++++++++++++++---- src/zimscraperlib/zim/items.py | 21 ++++++++-- tests/conftest.py | 65 +++++++++++++++++++++++++++++++ tests/zim/test_indexing.py | 52 ++++++++++++++++++++++++- 5 files changed, 183 insertions(+), 13 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index d40b54f..dfb023a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Index ePub documents for full-text search, with a new `get_epub_index_data` helper and automatic indexing of ePub items, like PDFs already are (#333) + ### Fixed - Carry a script's top-level `const`, `let` and `class` declarations across the wombat block, so other scripts on the page can still see them (#329) diff --git a/src/zimscraperlib/zim/indexing.py b/src/zimscraperlib/zim/indexing.py index b23cb5b..3413e15 100644 --- a/src/zimscraperlib/zim/indexing.py +++ b/src/zimscraperlib/zim/indexing.py @@ -62,8 +62,17 @@ def content(self, value: str): "lcms: not an ICC profile, invalid signature.", "format error: cmsOpenProfileFromMem failed", "ignoring broken ICC profile", + # MuPDF only knows EPUB 2 but reads EPUB 3 documents fine + "unknown epub version: 3.0", ] +# messages carrying a document-specific detail, ignored by prefix; images and +# styles do not matter for the text content +IGNORED_MUPDF_MESSAGE_PREFIXES = ( + "html: cannot load image", + "syntax error: css syntax error", +) + def get_pdf_index_data( *, @@ -75,6 +84,34 @@ def get_pdf_index_data( PDF can be passed either as content or fileobject or filepath """ + return _get_document_index_data( + filetype="pdf", content=content, fileobj=fileobj, filepath=filepath + ) + + +def get_epub_index_data( + *, + content: str | bytes | None = None, + fileobj: io.BytesIO | None = None, + filepath: pathlib.Path | None = None, +) -> IndexData: + """Returns the IndexData information for a given EPUB + + EPUB can be passed either as content or fileobject or filepath + """ + return _get_document_index_data( + filetype="epub", content=content, fileobj=fileobj, filepath=filepath + ) + + +def _get_document_index_data( + *, + filetype: str, + content: str | bytes | None = None, + fileobj: io.BytesIO | None = None, + filepath: pathlib.Path | None = None, +) -> IndexData: + """IndexData of any document PyMuPDF can read, title built from its metadata""" # do not display all pymupdf errors, we will filter them afterwards pymupdf.TOOLS.mupdf_display_errors( # pyright: ignore[reportUnknownMemberType] @@ -82,26 +119,26 @@ def get_pdf_index_data( ) if content: - doc = pymupdf.open(stream=content) + doc = pymupdf.open(stream=content, filetype=filetype) elif fileobj: - doc = pymupdf.open(stream=fileobj) + doc = pymupdf.open(stream=fileobj, filetype=filetype) else: - doc = pymupdf.open(filename=filepath) + doc = pymupdf.open(filename=filepath, filetype=filetype) metadata = ( # pyright: ignore[reportUnknownVariableType, reportUnknownMemberType] doc.metadata ) title = "" - if metadata: # pragma: no branch (always metadata in test PDFs) + if metadata: # pragma: no branch (always metadata in test documents) parts: list[str] = [] for key in ["title", "author", "subject"]: if metadata.get(key): # pyright: ignore[reportUnknownMemberType] parts.append( metadata[key] # pyright: ignore[reportUnknownArgumentType] ) - if parts: # pragma: no branch (always metadata in test PDFs) + if parts: # pragma: no branch (always metadata in test documents) title = " - ".join(parts) - def get_pdf_content(page: pymupdf.Page) -> str: + def get_page_content(page: pymupdf.Page) -> str: text = ( # pyright: ignore[reportUnknownVariableType] page.get_text() # pyright: ignore[reportUnknownMemberType] ) @@ -109,7 +146,7 @@ def get_pdf_content(page: pymupdf.Page) -> str: raise Exception("Unexpected text content") return text - content = "\n".join(get_pdf_content(page) for page in doc) + content = "\n".join(get_page_content(page) for page in doc) # build list of messages and filter messages which are known to not be relevant # in our use-case @@ -117,12 +154,13 @@ def get_pdf_content(page: pymupdf.Page) -> str: warning for warning in pymupdf.TOOLS.mupdf_warnings().splitlines() if warning not in IGNORED_MUPDF_MESSAGES + and not warning.startswith(IGNORED_MUPDF_MESSAGE_PREFIXES) ) if mupdf_messages: logger.warning( f"PyMuPDF issues:\n{mupdf_messages}" - ) # pragma: no cover (no known error in test PDFs) + ) # pragma: no cover (no known error in test documents) return IndexData( title=title, diff --git a/src/zimscraperlib/zim/items.py b/src/zimscraperlib/zim/items.py index f0f4650..418cc8a 100644 --- a/src/zimscraperlib/zim/items.py +++ b/src/zimscraperlib/zim/items.py @@ -12,7 +12,11 @@ from zimscraperlib.download import stream_file from zimscraperlib.filesystem import get_content_mimetype, get_file_mimetype -from zimscraperlib.zim.indexing import IndexData, get_pdf_index_data +from zimscraperlib.zim.indexing import ( + IndexData, + get_epub_index_data, + get_pdf_index_data, +) from zimscraperlib.zim.providers import ( FileLikeProvider, FileProvider, @@ -75,8 +79,8 @@ class StaticItem(Item): the Item and we can be notified that we're effectively through with our content By default, content is automatically indexed (either by the libzim itself for - supported documents - text or html for now or by the python-scraperlib - only PDF - supported for now). If you do not want this, set `auto_index` to False to disable + supported documents - text or html for now or by the python-scraperlib - PDF and + ePub for now). If you do not want this, set `auto_index` to False to disable both indexing (libzim and python-scraperlib). It is also possible to pass index_data to configure custom indexing of the item. @@ -160,6 +164,9 @@ def _get_auto_index(self): if mimetype == "application/pdf": index_data = get_pdf_index_data(content=content) self.get_indexdata = lambda: index_data + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(content=content) + self.get_indexdata = lambda: index_data else: return @@ -174,6 +181,9 @@ def _get_auto_index(self): if mimetype == "application/pdf": index_data = get_pdf_index_data(fileobj=fileobj) self.get_indexdata = lambda: index_data + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(fileobj=fileobj) + self.get_indexdata = lambda: index_data else: return @@ -190,6 +200,11 @@ def _get_auto_index(self): self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride] lambda: index_data ) + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(filepath=filepath) + self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride] + lambda: index_data + ) else: return diff --git a/tests/conftest.py b/tests/conftest.py index 817e004..af4a3c7 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,4 +1,5 @@ import pathlib +import zipfile import pytest import requests @@ -240,3 +241,67 @@ def undecodable_byte_stream() -> bytes: b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" ) + + +@pytest.fixture(scope="session") +def epub_file(tmpdir_factory: pytest.TempdirFactory) -> pathlib.Path: + """Minimal EPUB 3 with title/author metadata and two chapters""" + dst = pathlib.Path(tmpdir_factory.mktemp("data") / "minimal.epub") + # a missing image and a CSS selector MuPDF does not support, as found in + # real-world ePubs, make MuPDF warn without affecting the text + chapter = ( + '' + '
{text}
" + '
'
+ )
+ with zipfile.ZipFile(dst, "w") as epub:
+ # mimetype must be the first entry, uncompressed
+ epub.writestr("mimetype", "application/epub+zip", zipfile.ZIP_STORED)
+ epub.writestr(
+ "META-INF/container.xml",
+ ''
+ '