diff --git a/lib/Epub/Epub/Section.cpp b/lib/Epub/Epub/Section.cpp index be8fe1f2..07f24bc2 100644 --- a/lib/Epub/Epub/Section.cpp +++ b/lib/Epub/Epub/Section.cpp @@ -12,7 +12,7 @@ #include "parsers/ChapterHtmlSlimParser.h" namespace { -constexpr uint8_t SECTION_FILE_VERSION = 25; +constexpr uint8_t SECTION_FILE_VERSION = 26; constexpr uint32_t HEADER_SIZE = sizeof(uint8_t) + // SECTION_FILE_VERSION sizeof(int) + // fontId sizeof(float) + // lineCompression @@ -29,8 +29,11 @@ constexpr uint32_t HEADER_SIZE = sizeof(uint8_t) + // SECTION_FILE_VERSION sizeof(uint32_t) + // anchor map offset sizeof(uint32_t); // paragraph LUT offset -// On-disk paragraph LUT entry: u32 xhtmlByteOffset + u16 paragraphIndex. -constexpr uint32_t PARAGRAPH_LUT_ENTRY_SIZE = sizeof(uint32_t) + sizeof(uint16_t); +// On-disk paragraph LUT entry: u32 xhtmlByteOffset + u16 paragraphIndex + u16 listItemIndex. +// listItemIndex is the running
  • count at page-break time; together with +// paragraphIndex it lets KOReader-supplied

    - and

  • -anchored XPaths snap to +// the exact page on download. +constexpr uint32_t PARAGRAPH_LUT_ENTRY_SIZE = sizeof(uint32_t) + sizeof(uint16_t) + sizeof(uint16_t); inline uint32_t paragraphLutEntryOffset(uint32_t lutStart, uint16_t page) { return lutStart + page * PARAGRAPH_LUT_ENTRY_SIZE; } @@ -471,6 +474,7 @@ bool Section::createSectionFile(const int fontId, const float lineCompression, c for (const auto& entry : paragraphLut) { serialization::writePod(file, entry.xhtmlByteOffset); serialization::writePod(file, entry.paragraphIndex); + serialization::writePod(file, entry.listItemIndex); } // Patch header with final pageCount, lutOffset, anchorMapOffset, and paragraphLutOffset @@ -803,6 +807,42 @@ std::optional Section::getParagraphIndexForPage(const uint16_t page) c return pIdx; } +std::optional Section::getPageForListItemIndex(const uint16_t liIndex) const { + if (liIndex == 0) { + return std::nullopt; + } + + FsFile f; + uint16_t count = 0; + uint32_t lutStart = 0; + if (!readParagraphLutHeader(f, count, lutStart)) { + return std::nullopt; + } + const uint32_t fileSize = f.size(); + + // Mirror getPageForParagraphIndex: each entry stores the running li count at page-break + // time, so the target li first appears on the smallest i where storedLiIdx[i] >= liIndex. + // The listItemIndex field follows xhtmlByteOffset + paragraphIndex within each entry. + for (uint16_t i = 0; i < count; i++) { + const uint32_t entryOffset = paragraphLutEntryOffset(lutStart, i) + sizeof(uint32_t) + sizeof(uint16_t); + const uint64_t requiredOffset = static_cast(entryOffset) + sizeof(uint16_t); + if (requiredOffset > fileSize) { + f.close(); + return std::nullopt; + } + f.seek(entryOffset); + uint16_t pageLiIdx; + serialization::readPod(f, pageLiIdx); + if (pageLiIdx >= liIndex) { + f.close(); + return i; + } + } + + f.close(); + return static_cast(count - 1); +} + std::optional Section::getXhtmlByteOffsetForPage(const uint16_t page) const { FsFile f; uint16_t count = 0; diff --git a/lib/Epub/Epub/Section.h b/lib/Epub/Epub/Section.h index 6c2f451b..cc7c6b0e 100644 --- a/lib/Epub/Epub/Section.h +++ b/lib/Epub/Epub/Section.h @@ -90,6 +90,12 @@ class Section { // Returns nullopt if the paragraph LUT is not available (old cache format). std::optional getPageForParagraphIndex(uint16_t pIndex) const; + // Look up the page number for a running
  • index (1-based, the Nth
  • at any depth + // in the chapter). Used to snap KOReader-supplied list-item XPaths to a precise page + // the same way getPageForParagraphIndex handles

    -anchored XPaths. + // Returns nullopt if the LUT is not available or the index is out of range. + std::optional getPageForListItemIndex(uint16_t liIndex) const; + // Look up the paragraph index for a given page number. // Returns the 1-based paragraph index of the last

    element on or before the page. // Returns nullopt if the paragraph LUT is not available (old cache format). diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp index 10db5834..f6c3f2cd 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp @@ -250,7 +250,7 @@ void ChapterHtmlSlimParser::flushPartWordBuffer() { // Callers must ensure currentPage is non-null and carries content; the helper resets // currentPage to a fresh Page and zeroes currentPageNextY so the caller can keep building. void ChapterHtmlSlimParser::emitPage(uint32_t xhtmlByteOffset) { - paragraphLutPerPage.push_back({xhtmlByteOffset, xpathParagraphIndex}); + paragraphLutPerPage.push_back({xhtmlByteOffset, xpathParagraphIndex, xpathListItemIndex}); completePageFn(std::move(currentPage)); completedPageCount++; currentPage.reset(new Page()); @@ -814,6 +814,13 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* } } + //

  • can appear nested inside