Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 4 additions & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0

## [Unreleased]

### Added

- Index ePub documents for full-text search, with a new `get_epub_index_data` helper and automatic indexing of ePub items, like PDFs already are (#333)

### Fixed

- Carry a script's top-level `const`, `let` and `class` declarations across the wombat block, so other scripts on the page can still see them (#329)
Expand Down
54 changes: 46 additions & 8 deletions src/zimscraperlib/zim/indexing.py
Original file line number Diff line number Diff line change
Expand Up @@ -62,8 +62,17 @@ def content(self, value: str):
"lcms: not an ICC profile, invalid signature.",
"format error: cmsOpenProfileFromMem failed",
"ignoring broken ICC profile",
# MuPDF only knows EPUB 2 but reads EPUB 3 documents fine
"unknown epub version: 3.0",
]

# messages carrying a document-specific detail, ignored by prefix; images and
# styles do not matter for the text content
IGNORED_MUPDF_MESSAGE_PREFIXES = (
"html: cannot load image",
"syntax error: css syntax error",
)


def get_pdf_index_data(
*,
Expand All @@ -75,54 +84,83 @@ def get_pdf_index_data(

PDF can be passed either as content or fileobject or filepath
"""
return _get_document_index_data(
filetype="pdf", content=content, fileobj=fileobj, filepath=filepath
)


def get_epub_index_data(
*,
content: str | bytes | None = None,
fileobj: io.BytesIO | None = None,
filepath: pathlib.Path | None = None,
) -> IndexData:
"""Returns the IndexData information for a given EPUB

EPUB can be passed either as content or fileobject or filepath
"""
return _get_document_index_data(
filetype="epub", content=content, fileobj=fileobj, filepath=filepath
)


def _get_document_index_data(
*,
filetype: str,
content: str | bytes | None = None,
fileobj: io.BytesIO | None = None,
filepath: pathlib.Path | None = None,
) -> IndexData:
"""IndexData of any document PyMuPDF can read, title built from its metadata"""

# do not display all pymupdf errors, we will filter them afterwards
pymupdf.TOOLS.mupdf_display_errors( # pyright: ignore[reportUnknownMemberType]
False
)

if content:
doc = pymupdf.open(stream=content)
doc = pymupdf.open(stream=content, filetype=filetype)
elif fileobj:
doc = pymupdf.open(stream=fileobj)
doc = pymupdf.open(stream=fileobj, filetype=filetype)
else:
doc = pymupdf.open(filename=filepath)
doc = pymupdf.open(filename=filepath, filetype=filetype)
metadata = ( # pyright: ignore[reportUnknownVariableType, reportUnknownMemberType]
doc.metadata
)
title = ""
if metadata: # pragma: no branch (always metadata in test PDFs)
if metadata: # pragma: no branch (always metadata in test documents)
parts: list[str] = []
for key in ["title", "author", "subject"]:
if metadata.get(key): # pyright: ignore[reportUnknownMemberType]
parts.append(
metadata[key] # pyright: ignore[reportUnknownArgumentType]
)
if parts: # pragma: no branch (always metadata in test PDFs)
if parts: # pragma: no branch (always metadata in test documents)
title = " - ".join(parts)

def get_pdf_content(page: pymupdf.Page) -> str:
def get_page_content(page: pymupdf.Page) -> str:
text = ( # pyright: ignore[reportUnknownVariableType]
page.get_text() # pyright: ignore[reportUnknownMemberType]
)
if not isinstance(text, str): # pragma: no cover
raise Exception("Unexpected text content")
return text

content = "\n".join(get_pdf_content(page) for page in doc)
content = "\n".join(get_page_content(page) for page in doc)

# build list of messages and filter messages which are known to not be relevant
# in our use-case
mupdf_messages = "\n".join(
warning
for warning in pymupdf.TOOLS.mupdf_warnings().splitlines()
if warning not in IGNORED_MUPDF_MESSAGES
and not warning.startswith(IGNORED_MUPDF_MESSAGE_PREFIXES)
)

if mupdf_messages:
logger.warning(
f"PyMuPDF issues:\n{mupdf_messages}"
) # pragma: no cover (no known error in test PDFs)
) # pragma: no cover (no known error in test documents)

return IndexData(
title=title,
Expand Down
21 changes: 18 additions & 3 deletions src/zimscraperlib/zim/items.py
Original file line number Diff line number Diff line change
Expand Up @@ -12,7 +12,11 @@

from zimscraperlib.download import stream_file
from zimscraperlib.filesystem import get_content_mimetype, get_file_mimetype
from zimscraperlib.zim.indexing import IndexData, get_pdf_index_data
from zimscraperlib.zim.indexing import (
IndexData,
get_epub_index_data,
get_pdf_index_data,
)
from zimscraperlib.zim.providers import (
FileLikeProvider,
FileProvider,
Expand Down Expand Up @@ -75,8 +79,8 @@ class StaticItem(Item):
the Item and we can be notified that we're effectively through with our content

By default, content is automatically indexed (either by the libzim itself for
supported documents - text or html for now or by the python-scraperlib - only PDF
supported for now). If you do not want this, set `auto_index` to False to disable
supported documents - text or html for now or by the python-scraperlib - PDF and
ePub for now). If you do not want this, set `auto_index` to False to disable
both indexing (libzim and python-scraperlib).

It is also possible to pass index_data to configure custom indexing of the item.
Expand Down Expand Up @@ -160,6 +164,9 @@ def _get_auto_index(self):
if mimetype == "application/pdf":
index_data = get_pdf_index_data(content=content)
self.get_indexdata = lambda: index_data
elif mimetype == "application/epub+zip":
index_data = get_epub_index_data(content=content)
self.get_indexdata = lambda: index_data
else:
return

Expand All @@ -174,6 +181,9 @@ def _get_auto_index(self):
if mimetype == "application/pdf":
index_data = get_pdf_index_data(fileobj=fileobj)
self.get_indexdata = lambda: index_data
elif mimetype == "application/epub+zip":
index_data = get_epub_index_data(fileobj=fileobj)
self.get_indexdata = lambda: index_data
else:
return

Expand All @@ -190,6 +200,11 @@ def _get_auto_index(self):
self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride]
lambda: index_data
)
elif mimetype == "application/epub+zip":
index_data = get_epub_index_data(filepath=filepath)
self.get_indexdata = ( # pyright:ignore [reportIncompatibleVariableOverride]
lambda: index_data
)
else:
return

Expand Down
65 changes: 65 additions & 0 deletions tests/conftest.py
Original file line number Diff line number Diff line change
@@ -1,4 +1,5 @@
import pathlib
import zipfile

import pytest
import requests
Expand Down Expand Up @@ -240,3 +241,67 @@ def undecodable_byte_stream() -> bytes:
b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00"
b"\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00\x00"
)


@pytest.fixture(scope="session")
def epub_file(tmpdir_factory: pytest.TempdirFactory) -> pathlib.Path:
"""Minimal EPUB 3 with title/author metadata and two chapters"""
dst = pathlib.Path(tmpdir_factory.mktemp("data") / "minimal.epub")
# a missing image and a CSS selector MuPDF does not support, as found in
# real-world ePubs, make MuPDF warn without affecting the text
chapter = (
'<?xml version="1.0" encoding="utf-8"?>'
'<html xmlns="http://www.w3.org/1999/xhtml"><head><title>{title}</title>'
'<style>img[src*="wiki"] {{ border: 0 }}</style>'
"</head><body><h1>{title}</h1><p>{text}</p>"
'<img src="images/missing.png" alt=""/></body></html>'
)
with zipfile.ZipFile(dst, "w") as epub:
# mimetype must be the first entry, uncompressed
epub.writestr("mimetype", "application/epub+zip", zipfile.ZIP_STORED)
epub.writestr(
"META-INF/container.xml",
'<?xml version="1.0"?>'
'<container version="1.0" '
'xmlns="urn:oasis:names:tc:opendocument:xmlns:container">'
'<rootfiles><rootfile full-path="OEBPS/content.opf" '
'media-type="application/oebps-package+xml"/></rootfiles></container>',
)
epub.writestr(
"OEBPS/content.opf",
'<?xml version="1.0" encoding="utf-8"?>'
'<package xmlns="http://www.idpf.org/2007/opf" version="3.0" '
'unique-identifier="id">'
'<metadata xmlns:dc="http://purl.org/dc/elements/1.1/">'
'<dc:identifier id="id">urn:uuid:7c2b1a52-1d4e-4b0e-9c55-3f7e1d2a9b10'
"</dc:identifier>"
"<dc:title>Lighthouse Almanac</dc:title>"
"<dc:creator>Ada Quillfeather</dc:creator>"
"<dc:language>en</dc:language>"
'<meta property="dcterms:modified">2026-01-01T00:00:00Z</meta>'
"</metadata><manifest>"
'<item id="nav" href="nav.xhtml" media-type="application/xhtml+xml" '
'properties="nav"/>'
'<item id="c1" href="c1.xhtml" media-type="application/xhtml+xml"/>'
'<item id="c2" href="c2.xhtml" media-type="application/xhtml+xml"/>'
'</manifest><spine><itemref idref="c1"/><itemref idref="c2"/></spine>'
"</package>",
)
epub.writestr(
"OEBPS/nav.xhtml",
'<?xml version="1.0" encoding="utf-8"?>'
'<html xmlns="http://www.w3.org/1999/xhtml" '
'xmlns:epub="http://www.idpf.org/2007/ops"><head><title>Contents</title>'
'</head><body><nav epub:type="toc"><ol>'
'<li><a href="c1.xhtml">Storm</a></li>'
'<li><a href="c2.xhtml">Harbour</a></li></ol></nav></body></html>',
)
epub.writestr(
"OEBPS/c1.xhtml",
chapter.format(title="Storm", text="The keeper trimmed the wick."),
)
epub.writestr(
"OEBPS/c2.xhtml",
chapter.format(title="Harbour", text="Gulls circled the breakwater."),
)
return dst
52 changes: 50 additions & 2 deletions tests/zim/test_indexing.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,9 +3,14 @@

import libzim.writer # pyright: ignore[reportMissingModuleSource]
import pytest
from pytest_mock import MockerFixture

from zimscraperlib.zim import Archive, Creator
from zimscraperlib.zim.indexing import IndexData, get_pdf_index_data
from zimscraperlib.zim import Archive, Creator, indexing
from zimscraperlib.zim.indexing import (
IndexData,
get_epub_index_data,
get_pdf_index_data,
)
from zimscraperlib.zim.items import StaticItem


Expand Down Expand Up @@ -368,3 +373,46 @@ def test_index_data_wordcount(
IndexData(title="foo", content=content, wordcount=wordcount).get_wordcount()
== expected_wordcount
)


@pytest.mark.parametrize("source", ["content", "fileobj", "filepath"])
def test_get_epub_index_data(
source: str, epub_file: pathlib.Path, mocker: MockerFixture
):
warning = mocker.patch.object(indexing.logger, "warning")
if source == "content":
index_data = get_epub_index_data(content=epub_file.read_bytes())
elif source == "fileobj":
index_data = get_epub_index_data(fileobj=io.BytesIO(epub_file.read_bytes()))
else:
index_data = get_epub_index_data(filepath=epub_file)
assert index_data.get_title() == "Lighthouse Almanac - Ada Quillfeather"
assert "The keeper trimmed the wick." in index_data.get_content()
assert "Gulls circled the breakwater." in index_data.get_content()
assert index_data.has_indexdata()
# MuPDF does not know EPUB 3 but reads it fine, this must not be reported
warning.assert_not_called()


@pytest.mark.parametrize("source", ["content", "fileobj", "filepath"])
def test_indexing_item_epub(
tmp_path: pathlib.Path, epub_file: pathlib.Path, source: str
):
"""An EPUB item can be automatically indexed"""
fpath = tmp_path / "test.zim"
hints: dict[libzim.writer.Hint, int] = {libzim.writer.Hint.FRONT_ARTICLE: True}
if source == "content":
item = StaticItem(content=epub_file.read_bytes(), path="welcome", hints=hints)
elif source == "fileobj":
fileobj = io.BytesIO(epub_file.read_bytes())
item = StaticItem(fileobj=fileobj, path="welcome", hints=hints)
else:
item = StaticItem(filepath=epub_file, path="welcome", hints=hints)
with Creator(fpath, "welcome").config_dev_metadata() as creator:
creator.add_item(item)

reader = Archive(fpath)
assert "welcome" in list(reader.get_suggestions("lighthouse"))
assert "welcome" in list(reader.get_suggestions("quillfeather"))
assert reader.get_search_results_count("breakwater") == 1
assert reader.get_search_results_count("wick") == 1