diff --git a/CHANGELOG.md b/CHANGELOG.md index d40b54f..dfb023a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Added + +- Index ePub documents for full-text search, with a new `get_epub_index_data` helper and automatic indexing of ePub items, like PDFs already are (#333) + ### Fixed - Carry a script's top-level `const`, `let` and `class` declarations across the wombat block, so other scripts on the page can still see them (#329) diff --git a/src/zimscraperlib/zim/indexing.py b/src/zimscraperlib/zim/indexing.py index b23cb5b..3413e15 100644 --- a/src/zimscraperlib/zim/indexing.py +++ b/src/zimscraperlib/zim/indexing.py @@ -62,8 +62,17 @@ def content(self, value: str): "lcms: not an ICC profile, invalid signature.", "format error: cmsOpenProfileFromMem failed", "ignoring broken ICC profile", + # MuPDF only knows EPUB 2 but reads EPUB 3 documents fine + "unknown epub version: 3.0", ] +# messages carrying a document-specific detail, ignored by prefix; images and +# styles do not matter for the text content +IGNORED_MUPDF_MESSAGE_PREFIXES = ( + "html: cannot load image", + "syntax error: css syntax error", +) + def get_pdf_index_data( *, @@ -75,6 +84,34 @@ def get_pdf_index_data( PDF can be passed either as content or fileobject or filepath """ + return _get_document_index_data( + filetype="pdf", content=content, fileobj=fileobj, filepath=filepath + ) + + +def get_epub_index_data( + *, + content: str | bytes | None = None, + fileobj: io.BytesIO | None = None, + filepath: pathlib.Path | None = None, +) -> IndexData: + """Returns the IndexData information for a given EPUB + + EPUB can be passed either as content or fileobject or filepath + """ + return _get_document_index_data( + filetype="epub", content=content, fileobj=fileobj, filepath=filepath + ) + + +def _get_document_index_data( + *, + filetype: str, + content: str | bytes | None = None, + fileobj: io.BytesIO | None = None, + filepath: pathlib.Path | None = None, +) -> IndexData: + """IndexData of any document PyMuPDF can read, title built from its metadata""" # do not display all pymupdf errors, we will filter them afterwards pymupdf.TOOLS.mupdf_display_errors( # pyright: ignore[reportUnknownMemberType] @@ -82,26 +119,26 @@ def get_pdf_index_data( ) if content: - doc = pymupdf.open(stream=content) + doc = pymupdf.open(stream=content, filetype=filetype) elif fileobj: - doc = pymupdf.open(stream=fileobj) + doc = pymupdf.open(stream=fileobj, filetype=filetype) else: - doc = pymupdf.open(filename=filepath) + doc = pymupdf.open(filename=filepath, filetype=filetype) metadata = ( # pyright: ignore[reportUnknownVariableType, reportUnknownMemberType] doc.metadata ) title = "" - if metadata: # pragma: no branch (always metadata in test PDFs) + if metadata: # pragma: no branch (always metadata in test documents) parts: list[str] = [] for key in ["title", "author", "subject"]: if metadata.get(key): # pyright: ignore[reportUnknownMemberType] parts.append( metadata[key] # pyright: ignore[reportUnknownArgumentType] ) - if parts: # pragma: no branch (always metadata in test PDFs) + if parts: # pragma: no branch (always metadata in test documents) title = " - ".join(parts) - def get_pdf_content(page: pymupdf.Page) -> str: + def get_page_content(page: pymupdf.Page) -> str: text = ( # pyright: ignore[reportUnknownVariableType] page.get_text() # pyright: ignore[reportUnknownMemberType] ) @@ -109,7 +146,7 @@ def get_pdf_content(page: pymupdf.Page) -> str: raise Exception("Unexpected text content") return text - content = "\n".join(get_pdf_content(page) for page in doc) + content = "\n".join(get_page_content(page) for page in doc) # build list of messages and filter messages which are known to not be relevant # in our use-case @@ -117,12 +154,13 @@ def get_pdf_content(page: pymupdf.Page) -> str: warning for warning in pymupdf.TOOLS.mupdf_warnings().splitlines() if warning not in IGNORED_MUPDF_MESSAGES + and not warning.startswith(IGNORED_MUPDF_MESSAGE_PREFIXES) ) if mupdf_messages: logger.warning( f"PyMuPDF issues:\n{mupdf_messages}" - ) # pragma: no cover (no known error in test PDFs) + ) # pragma: no cover (no known error in test documents) return IndexData( title=title, diff --git a/src/zimscraperlib/zim/items.py b/src/zimscraperlib/zim/items.py index f0f4650..418cc8a 100644 --- a/src/zimscraperlib/zim/items.py +++ b/src/zimscraperlib/zim/items.py @@ -12,7 +12,11 @@ from zimscraperlib.download import stream_file from zimscraperlib.filesystem import get_content_mimetype, get_file_mimetype -from zimscraperlib.zim.indexing import IndexData, get_pdf_index_data +from zimscraperlib.zim.indexing import ( + IndexData, + get_epub_index_data, + get_pdf_index_data, +) from zimscraperlib.zim.providers import ( FileLikeProvider, FileProvider, @@ -75,8 +79,8 @@ class StaticItem(Item): the Item and we can be notified that we're effectively through with our content By default, content is automatically indexed (either by the libzim itself for - supported documents - text or html for now or by the python-scraperlib - only PDF - supported for now). If you do not want this, set `auto_index` to False to disable + supported documents - text or html for now or by the python-scraperlib - PDF and + ePub for now). If you do not want this, set `auto_index` to False to disable both indexing (libzim and python-scraperlib). It is also possible to pass index_data to configure custom indexing of the item. @@ -160,6 +164,9 @@ def _get_auto_index(self): if mimetype == "application/pdf": index_data = get_pdf_index_data(content=content) self.get_indexdata = lambda: index_data + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(content=content) + self.get_indexdata = lambda: index_data else: return @@ -174,6 +181,9 @@ def _get_auto_index(self): if mimetype == "application/pdf": index_data = get_pdf_index_data(fileobj=fileobj) self.get_indexdata = lambda: index_data + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(fileobj=fileobj) + self.get_indexdata = lambda: index_data else: return @@ -190,6 +200,11 @@ def _get_auto_index(self): self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride] lambda: index_data ) + elif mimetype == "application/epub+zip": + index_data = get_epub_index_data(filepath=filepath) + self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride] + lambda: index_data + ) else: return diff --git a/tests/conftest.py b/tests/conftest.py index 817e004..af4a3c7 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -1,4 +1,5 @@ import pathlib +import zipfile import pytest import requests @@ -240,3 +241,67 @@ def undecodable_byte_stream() -> bytes: b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00" ) + + +@pytest.fixture(scope="session") +def epub_file(tmpdir_factory: pytest.TempdirFactory) -> pathlib.Path: + """Minimal EPUB 3 with title/author metadata and two chapters""" + dst = pathlib.Path(tmpdir_factory.mktemp("data") / "minimal.epub") + # a missing image and a CSS selector MuPDF does not support, as found in + # real-world ePubs, make MuPDF warn without affecting the text + chapter = ( + '' + '
{text}
" + '
'
+ )
+ with zipfile.ZipFile(dst, "w") as epub:
+ # mimetype must be the first entry, uncompressed
+ epub.writestr("mimetype", "application/epub+zip", zipfile.ZIP_STORED)
+ epub.writestr(
+ "META-INF/container.xml",
+ ''
+ '