Improving KOReader sync

This commit is contained in:
jpirnay
2026-05-10 15:46:33 +02:00
parent b5ea8d3550
commit fe60b12d86
18 changed files with 252 additions and 54 deletions
@@ -250,7 +250,7 @@ void ChapterHtmlSlimParser::flushPartWordBuffer() {
// Callers must ensure currentPage is non-null and carries content; the helper resets
// currentPage to a fresh Page and zeroes currentPageNextY so the caller can keep building.
void ChapterHtmlSlimParser::emitPage(uint32_t xhtmlByteOffset) {
paragraphLutPerPage.push_back({xhtmlByteOffset, xpathParagraphIndex});
paragraphLutPerPage.push_back({xhtmlByteOffset, xpathParagraphIndex, xpathListItemIndex});
completePageFn(std::move(currentPage));
completedPageCount++;
currentPage.reset(new Page());
@@ -814,6 +814,13 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
}
}
// <li> can appear nested inside <ul>/<ol> at any depth, so count it globally —
// not at body-child level. The running count must match what the runtime reverse
// mapper sees so getPageForListItemIndex can snap a KOReader li XPath to a page.
if (self->xpathBodyDepth >= 0 && strcmp(name, "li") == 0) {
self->xpathListItemIndex++;
}
if (matches(name, SKIP_TAGS, NUM_SKIP_TAGS)) {
// start skip
self->skipUntilDepth = self->depth;
@@ -108,7 +108,11 @@ class ChapterHtmlSlimParser final : public Print {
// Stored per page in the section cache so that XPath p[N] can be resolved to a page
// without reparsing, and current page can generate an XPath without reparsing.
uint16_t xpathParagraphIndex = 0; // current <p> sibling index (1-based)
int xpathBodyDepth = -1; // depth of the <body> element (-1 = not yet seen)
// Running count of <li> elements opened anywhere in the chapter (1-based, any depth).
// Used by the section LUT so KOReader-supplied list-item XPaths can snap to the exact
// page on download, the same way <p>-anchored XPaths use xpathParagraphIndex.
uint16_t xpathListItemIndex = 0;
int xpathBodyDepth = -1; // depth of the <body> element (-1 = not yet seen)
// Byte offset of the most recent direct-body-child element start (any tag at xpathBodyDepth+1).
// Recorded at the same depth condition that increments xpathParagraphIndex, so the stored
// offset is guaranteed to land on a body-child element boundary. This keeps the XPath forward
@@ -119,6 +123,7 @@ class ChapterHtmlSlimParser final : public Print {
struct ParagraphLutEntry {
uint32_t xhtmlByteOffset; // byte offset of most recent body-child element start at page break
uint16_t paragraphIndex; // 1-based <p> index at page completion
uint16_t listItemIndex; // running <li> count at page completion (any depth)
};
std::vector<ParagraphLutEntry> paragraphLutPerPage; // deep LUT: one entry per page