From 7e8bd70f3c710c8c76a4709f164026bcac27f7f8 Mon Sep 17 00:00:00 2001 From: jpirnay Date: Wed, 25 Feb 2026 22:34:11 +0100 Subject: [PATCH] First attempt for proper koreader xpath generation / resolution --- lib/KOReaderSync/ChapterXPathIndexer.cpp | 462 +++++++++++++++++++++++ lib/KOReaderSync/ChapterXPathIndexer.h | 43 +++ lib/KOReaderSync/ProgressMapper.cpp | 100 +++-- lib/KOReaderSync/ProgressMapper.h | 9 +- 4 files changed, 578 insertions(+), 36 deletions(-) create mode 100644 lib/KOReaderSync/ChapterXPathIndexer.cpp create mode 100644 lib/KOReaderSync/ChapterXPathIndexer.h diff --git a/lib/KOReaderSync/ChapterXPathIndexer.cpp b/lib/KOReaderSync/ChapterXPathIndexer.cpp new file mode 100644 index 00000000..478a5e6e --- /dev/null +++ b/lib/KOReaderSync/ChapterXPathIndexer.cpp @@ -0,0 +1,462 @@ +#include "ChapterXPathIndexer.h" + +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +namespace { + +struct XPathAnchor { + size_t textOffset = 0; + std::string xpath; +}; + +struct StackNode { + std::string tag; + int index = 1; + bool hasTextAnchor = false; +}; + +struct ParserState { + explicit ParserState(const int spineIndex) : spineIndex(spineIndex) { siblingCounters.emplace_back(); } + + int spineIndex = 0; + int skipDepth = -1; + size_t totalTextBytes = 0; + + std::vector stack; + std::vector> siblingCounters; + std::vector anchors; + + std::string baseXPath() const { return "/body/DocFragment[" + std::to_string(spineIndex) + "]/body"; } + + static std::string normalizeXPath(const std::string& input) { + if (input.empty()) { + return ""; + } + + std::string out; + out.reserve(input.size()); + for (char c : input) { + const unsigned char uc = static_cast(c); + if (std::isspace(uc)) { + continue; + } + out.push_back(static_cast(std::tolower(uc))); + } + + const std::string textSuffix = "/text()"; + size_t textPos = out.find(textSuffix); + if (textPos != std::string::npos) { + out.erase(textPos); + } + + while (!out.empty() && out.back() == '/') { + out.pop_back(); + } + + return out; + } + + static std::string removeIndices(const std::string& xpath) { + std::string out; + out.reserve(xpath.size()); + + bool inBracket = false; + for (char c : xpath) { + if (c == '[') { + inBracket = true; + continue; + } + if (c == ']') { + inBracket = false; + continue; + } + if (!inBracket) { + out.push_back(c); + } + } + return out; + } + + static int pathDepth(const std::string& xpath) { + int depth = 0; + for (char c : xpath) { + if (c == '/') { + depth++; + } + } + return depth; + } + + bool pickBestAnchorByPath(const std::string& targetPath, const bool ignoreIndices, size_t& outTextOffset, + bool& outExact) const { + if (targetPath.empty() || anchors.empty()) { + return false; + } + + const std::string normalizedTarget = ignoreIndices ? removeIndices(targetPath) : targetPath; + std::string probe = normalizedTarget; + bool exactProbe = true; + + while (!probe.empty()) { + int bestDepth = -1; + size_t bestOffset = 0; + bool found = false; + + for (const auto& anchor : anchors) { + const std::string anchorPath = ignoreIndices ? removeIndices(anchor.xpath) : anchor.xpath; + if (anchorPath == probe) { + const int depth = pathDepth(anchorPath); + if (!found || depth > bestDepth || (depth == bestDepth && anchor.textOffset > bestOffset)) { + found = true; + bestDepth = depth; + bestOffset = anchor.textOffset; + } + } + } + + if (found) { + outTextOffset = bestOffset; + outExact = exactProbe; + return true; + } + + const size_t lastSlash = probe.find_last_of('/'); + if (lastSlash == std::string::npos || lastSlash == 0) { + break; + } + probe.erase(lastSlash); + exactProbe = false; + } + + return false; + } + + static std::string toLower(std::string value) { + for (char& c : value) { + c = static_cast(std::tolower(static_cast(c))); + } + return value; + } + + static bool isSkippableTag(const std::string& tag) { return tag == "head" || tag == "script" || tag == "style"; } + + static bool isWhitespaceOnly(const XML_Char* text, const int len) { + for (int i = 0; i < len; i++) { + if (!std::isspace(static_cast(text[i]))) { + return false; + } + } + return true; + } + + static size_t countVisibleBytes(const XML_Char* text, const int len) { + size_t count = 0; + for (int i = 0; i < len; i++) { + if (!std::isspace(static_cast(text[i]))) { + count++; + } + } + return count; + } + + int bodyDepth() const { + for (int i = static_cast(stack.size()) - 1; i >= 0; i--) { + if (stack[i].tag == "body") { + return i; + } + } + return -1; + } + + bool insideBody() const { return bodyDepth() >= 0; } + + std::string currentXPath() const { + const int bodyIdx = bodyDepth(); + if (bodyIdx < 0) { + return baseXPath(); + } + + std::string xpath = baseXPath(); + for (size_t i = static_cast(bodyIdx + 1); i < stack.size(); i++) { + xpath += "/" + stack[i].tag + "[" + std::to_string(stack[i].index) + "]"; + } + return xpath; + } + + void addAnchorIfNeeded() { + if (!insideBody() || stack.empty()) { + return; + } + + if (!stack.back().hasTextAnchor) { + anchors.push_back({totalTextBytes, currentXPath()}); + stack.back().hasTextAnchor = true; + } else if (anchors.empty() || totalTextBytes - anchors.back().textOffset >= 192) { + const std::string xpath = currentXPath(); + if (anchors.empty() || anchors.back().xpath != xpath) { + anchors.push_back({totalTextBytes, xpath}); + } + } + } + + void onStartElement(const XML_Char* rawName) { + std::string name = toLower(rawName ? rawName : ""); + const size_t depth = stack.size(); + + if (siblingCounters.size() <= depth) { + siblingCounters.resize(depth + 1); + } + const int siblingIndex = ++siblingCounters[depth][name]; + + stack.push_back({name, siblingIndex, false}); + siblingCounters.emplace_back(); + + if (skipDepth < 0 && isSkippableTag(name)) { + skipDepth = static_cast(stack.size()) - 1; + } + } + + void onEndElement() { + if (stack.empty()) { + return; + } + + if (skipDepth == static_cast(stack.size()) - 1) { + skipDepth = -1; + } + + stack.pop_back(); + if (!siblingCounters.empty()) { + siblingCounters.pop_back(); + } + } + + void onCharacterData(const XML_Char* text, const int len) { + if (skipDepth >= 0 || len <= 0 || !insideBody() || isWhitespaceOnly(text, len)) { + return; + } + + addAnchorIfNeeded(); + totalTextBytes += countVisibleBytes(text, len); + } + + std::string chooseXPath(const float intraSpineProgress) const { + if (anchors.empty()) { + return baseXPath(); + } + if (totalTextBytes == 0) { + return anchors.front().xpath; + } + + const float clampedProgress = std::max(0.0f, std::min(1.0f, intraSpineProgress)); + const size_t target = static_cast(clampedProgress * static_cast(totalTextBytes)); + + auto it = std::lower_bound(anchors.begin(), anchors.end(), target, + [](const XPathAnchor& anchor, const size_t value) { return anchor.textOffset < value; }); + + if (it == anchors.end()) { + return anchors.back().xpath; + } + return it->xpath; + } + + bool chooseProgressForXPath(const std::string& xpath, float& outIntraSpineProgress, bool& outExactMatch) const { + if (anchors.empty()) { + return false; + } + + const std::string normalized = normalizeXPath(xpath); + if (normalized.empty()) { + return false; + } + + size_t matchedOffset = 0; + bool exact = false; + bool matched = pickBestAnchorByPath(normalized, false, matchedOffset, exact); + + if (!matched) { + matched = pickBestAnchorByPath(normalized, true, matchedOffset, exact); + } + + if (!matched) { + return false; + } + + outExactMatch = exact; + if (totalTextBytes == 0) { + outIntraSpineProgress = 0.0f; + return true; + } + + outIntraSpineProgress = static_cast(matchedOffset) / static_cast(totalTextBytes); + outIntraSpineProgress = std::max(0.0f, std::min(1.0f, outIntraSpineProgress)); + return true; + } +}; + +void XMLCALL onStartElement(void* userData, const XML_Char* name, const XML_Char**) { + auto* state = static_cast(userData); + state->onStartElement(name); +} + +void XMLCALL onEndElement(void* userData, const XML_Char*) { + auto* state = static_cast(userData); + state->onEndElement(); +} + +void XMLCALL onCharacterData(void* userData, const XML_Char* text, const int len) { + auto* state = static_cast(userData); + state->onCharacterData(text, len); +} + +void XMLCALL onDefaultHandlerExpand(void* userData, const XML_Char* text, const int len) { + auto* state = static_cast(userData); + state->onCharacterData(text, len); +} + +} // namespace + +std::string ChapterXPathIndexer::findXPathForProgress(const std::shared_ptr& epub, const int spineIndex, + const float intraSpineProgress) { + if (!epub || spineIndex < 0 || spineIndex >= epub->getSpineItemsCount()) { + return ""; + } + + const auto spineItem = epub->getSpineItem(spineIndex); + if (spineItem.href.empty()) { + return ""; + } + + size_t chapterSize = 0; + uint8_t* chapterBytes = epub->readItemContentsToBytes(spineItem.href, &chapterSize, false); + if (!chapterBytes || chapterSize == 0) { + free(chapterBytes); + return ""; + } + + ParserState state(spineIndex); + + XML_Parser parser = XML_ParserCreate(nullptr); + if (!parser) { + free(chapterBytes); + LOG_ERR("KOX", "Failed to allocate XML parser for spine=%d", spineIndex); + return ""; + } + + XML_SetUserData(parser, &state); + XML_SetElementHandler(parser, onStartElement, onEndElement); + XML_SetCharacterDataHandler(parser, onCharacterData); + XML_SetDefaultHandlerExpand(parser, onDefaultHandlerExpand); + + const bool parseOk = XML_Parse(parser, reinterpret_cast(chapterBytes), static_cast(chapterSize), + XML_TRUE) != XML_STATUS_ERROR; + + if (!parseOk) { + LOG_ERR("KOX", "XPath parse failed for spine=%d at line %lu: %s", spineIndex, XML_GetCurrentLineNumber(parser), + XML_ErrorString(XML_GetErrorCode(parser))); + } + + XML_ParserFree(parser); + free(chapterBytes); + + if (!parseOk) { + return ""; + } + + return state.chooseXPath(intraSpineProgress); +} + +bool ChapterXPathIndexer::findProgressForXPath(const std::shared_ptr& epub, const int spineIndex, + const std::string& xpath, float& outIntraSpineProgress, + bool& outExactMatch) { + outIntraSpineProgress = 0.0f; + outExactMatch = false; + + if (!epub || spineIndex < 0 || spineIndex >= epub->getSpineItemsCount() || xpath.empty()) { + return false; + } + + const auto spineItem = epub->getSpineItem(spineIndex); + if (spineItem.href.empty()) { + return false; + } + + size_t chapterSize = 0; + uint8_t* chapterBytes = epub->readItemContentsToBytes(spineItem.href, &chapterSize, false); + if (!chapterBytes || chapterSize == 0) { + free(chapterBytes); + return false; + } + + ParserState state(spineIndex); + XML_Parser parser = XML_ParserCreate(nullptr); + if (!parser) { + free(chapterBytes); + LOG_ERR("KOX", "Failed to allocate XML parser for reverse lookup spine=%d", spineIndex); + return false; + } + + XML_SetUserData(parser, &state); + XML_SetElementHandler(parser, onStartElement, onEndElement); + XML_SetCharacterDataHandler(parser, onCharacterData); + XML_SetDefaultHandlerExpand(parser, onDefaultHandlerExpand); + + const bool parseOk = XML_Parse(parser, reinterpret_cast(chapterBytes), static_cast(chapterSize), + XML_TRUE) != XML_STATUS_ERROR; + + if (!parseOk) { + LOG_ERR("KOX", "Reverse XPath parse failed for spine=%d at line %lu: %s", spineIndex, + XML_GetCurrentLineNumber(parser), XML_ErrorString(XML_GetErrorCode(parser))); + } + + XML_ParserFree(parser); + free(chapterBytes); + + if (!parseOk) { + return false; + } + + return state.chooseProgressForXPath(xpath, outIntraSpineProgress, outExactMatch); +} + +bool ChapterXPathIndexer::tryExtractSpineIndexFromXPath(const std::string& xpath, int& outSpineIndex) { + outSpineIndex = -1; + if (xpath.empty()) { + return false; + } + + const std::string normalized = ParserState::normalizeXPath(xpath); + const std::string key = "/docfragment["; + const size_t pos = normalized.find(key); + if (pos == std::string::npos) { + return false; + } + + const size_t start = pos + key.size(); + size_t end = start; + while (end < normalized.size() && std::isdigit(static_cast(normalized[end]))) { + end++; + } + + if (end == start || end >= normalized.size() || normalized[end] != ']') { + return false; + } + + const std::string value = normalized.substr(start, end - start); + const long parsed = std::strtol(value.c_str(), nullptr, 10); + if (parsed < 0 || parsed > std::numeric_limits::max()) { + return false; + } + + outSpineIndex = static_cast(parsed); + return true; +} diff --git a/lib/KOReaderSync/ChapterXPathIndexer.h b/lib/KOReaderSync/ChapterXPathIndexer.h new file mode 100644 index 00000000..e728b881 --- /dev/null +++ b/lib/KOReaderSync/ChapterXPathIndexer.h @@ -0,0 +1,43 @@ +#pragma once + +#include + +#include +#include + +/** + * Builds element-level XPath anchors for a spine item and picks the best match + * for an intra-spine progress value. + */ +class ChapterXPathIndexer { + public: + /** + * @param epub Loaded EPUB instance + * @param spineIndex Current spine item index + * @param intraSpineProgress Position within the spine item [0.0, 1.0] + * @return Best matching XPath, or empty string on failure + */ + static std::string findXPathForProgress(const std::shared_ptr& epub, int spineIndex, float intraSpineProgress); + + /** + * Resolve a KOReader XPath to an intra-spine progress value. + * + * @param epub Loaded EPUB instance + * @param spineIndex Spine item index to parse + * @param xpath Incoming KOReader XPath + * @param outIntraSpineProgress Resolved position within spine [0.0, 1.0] + * @param outExactMatch True when an exact anchor match was found + * @return true if an exact or ancestor match was resolved + */ + static bool findProgressForXPath(const std::shared_ptr& epub, int spineIndex, const std::string& xpath, + float& outIntraSpineProgress, bool& outExactMatch); + + /** + * Parse the DocFragment index from a KOReader-style XPath. + * + * @param xpath KOReader XPath + * @param outSpineIndex Parsed DocFragment index (0-based) + * @return true when DocFragment[...] is present and valid + */ + static bool tryExtractSpineIndexFromXPath(const std::string& xpath, int& outSpineIndex); +}; diff --git a/lib/KOReaderSync/ProgressMapper.cpp b/lib/KOReaderSync/ProgressMapper.cpp index ef542ff4..11a6e6e5 100644 --- a/lib/KOReaderSync/ProgressMapper.cpp +++ b/lib/KOReaderSync/ProgressMapper.cpp @@ -4,6 +4,9 @@ #include +#include "ChapterXPathIndexer.h" + + KOReaderPosition ProgressMapper::toKOReader(const std::shared_ptr& epub, const CrossPointPosition& pos) { KOReaderPosition result; @@ -16,8 +19,13 @@ KOReaderPosition ProgressMapper::toKOReader(const std::shared_ptr& epub, c // Calculate overall book progress (0.0-1.0) result.percentage = epub->calculateProgress(pos.spineIndex, intraSpineProgress); - // Generate XPath with estimated paragraph position based on page - result.xpath = generateXPath(pos.spineIndex, pos.pageNumber, pos.totalPages); + // Generate the best available XPath for the current chapter position. + // Prefer element-level XPaths from a lightweight XHTML reparse; fall back + // to a synthetic chapter-level path if parsing fails. + result.xpath = ChapterXPathIndexer::findXPathForProgress(epub, pos.spineIndex, intraSpineProgress); + if (result.xpath.empty()) { + result.xpath = generateXPath(pos.spineIndex, pos.pageNumber, pos.totalPages); + } // Get chapter info for logging const int tocIndex = epub->getTocIndexForSpineIndex(pos.spineIndex); @@ -36,34 +44,64 @@ CrossPointPosition ProgressMapper::toCrossPoint(const std::shared_ptr& epu result.pageNumber = 0; result.totalPages = 0; - const size_t bookSize = epub->getBookSize(); - if (bookSize == 0) { + if (!epub || epub->getSpineItemsCount() <= 0) { return result; } - // Use percentage-based lookup for both spine and page positioning - // XPath parsing is unreliable since CrossPoint doesn't preserve detailed HTML structure - const size_t targetBytes = static_cast(bookSize * koPos.percentage); - - // Find the spine item that contains this byte position const int spineCount = epub->getSpineItemsCount(); - bool spineFound = false; - for (int i = 0; i < spineCount; i++) { - const size_t cumulativeSize = epub->getCumulativeSpineItemSize(i); - if (cumulativeSize >= targetBytes) { - result.spineIndex = i; - spineFound = true; - break; + + float resolvedIntraSpineProgress = -1.0f; + bool xpathExactMatch = false; + bool usedXPathMapping = false; + + int xpathSpineIndex = -1; + if (ChapterXPathIndexer::tryExtractSpineIndexFromXPath(koPos.xpath, xpathSpineIndex) && xpathSpineIndex >= 0 && + xpathSpineIndex < spineCount) { + float intraFromXPath = 0.0f; + if (ChapterXPathIndexer::findProgressForXPath(epub, xpathSpineIndex, koPos.xpath, intraFromXPath, + xpathExactMatch)) { + result.spineIndex = xpathSpineIndex; + resolvedIntraSpineProgress = intraFromXPath; + usedXPathMapping = true; } } - // If no spine item was found (e.g., targetBytes beyond last cumulative size), - // default to the last spine item so we map to the end of the book instead of the beginning. - if (!spineFound && spineCount > 0) { - result.spineIndex = spineCount - 1; + if (!usedXPathMapping) { + const size_t bookSize = epub->getBookSize(); + if (bookSize == 0) { + return result; + } + + const size_t targetBytes = static_cast(bookSize * koPos.percentage); + + bool spineFound = false; + for (int i = 0; i < spineCount; i++) { + const size_t cumulativeSize = epub->getCumulativeSpineItemSize(i); + if (cumulativeSize >= targetBytes) { + result.spineIndex = i; + spineFound = true; + break; + } + } + + if (!spineFound && spineCount > 0) { + result.spineIndex = spineCount - 1; + } + + if (result.spineIndex < epub->getSpineItemsCount()) { + const size_t prevCumSize = (result.spineIndex > 0) ? epub->getCumulativeSpineItemSize(result.spineIndex - 1) : 0; + const size_t currentCumSize = epub->getCumulativeSpineItemSize(result.spineIndex); + const size_t spineSize = currentCumSize - prevCumSize; + + if (spineSize > 0) { + const size_t bytesIntoSpine = (targetBytes > prevCumSize) ? (targetBytes - prevCumSize) : 0; + resolvedIntraSpineProgress = static_cast(bytesIntoSpine) / static_cast(spineSize); + resolvedIntraSpineProgress = std::max(0.0f, std::min(1.0f, resolvedIntraSpineProgress)); + } + } } - // Estimate page number within the spine item using percentage + // Estimate page number within the selected spine item if (result.spineIndex < epub->getSpineItemsCount()) { const size_t prevCumSize = (result.spineIndex > 0) ? epub->getCumulativeSpineItemSize(result.spineIndex - 1) : 0; const size_t currentCumSize = epub->getCumulativeSpineItemSize(result.spineIndex); @@ -91,24 +129,24 @@ CrossPointPosition ProgressMapper::toCrossPoint(const std::shared_ptr& epu result.totalPages = estimatedTotalPages; - if (spineSize > 0 && estimatedTotalPages > 0) { - const size_t bytesIntoSpine = (targetBytes > prevCumSize) ? (targetBytes - prevCumSize) : 0; - const float intraSpineProgress = static_cast(bytesIntoSpine) / static_cast(spineSize); - const float clampedProgress = std::max(0.0f, std::min(1.0f, intraSpineProgress)); - result.pageNumber = static_cast(clampedProgress * estimatedTotalPages); + if (estimatedTotalPages > 0 && resolvedIntraSpineProgress >= 0.0f) { + const float clampedProgress = std::max(0.0f, std::min(1.0f, resolvedIntraSpineProgress)); + result.pageNumber = static_cast(clampedProgress * static_cast(estimatedTotalPages)); result.pageNumber = std::max(0, std::min(result.pageNumber, estimatedTotalPages - 1)); + } else if (spineSize > 0 && estimatedTotalPages > 0) { + result.pageNumber = 0; } } - LOG_DBG("ProgressMapper", "KOReader -> CrossPoint: %.2f%% at %s -> spine=%d, page=%d", koPos.percentage * 100, - koPos.xpath.c_str(), result.spineIndex, result.pageNumber); + LOG_DBG("ProgressMapper", "KOReader -> CrossPoint: %.2f%% at %s -> spine=%d, page=%d (%s, exact=%s)", + koPos.percentage * 100, koPos.xpath.c_str(), result.spineIndex, result.pageNumber, + usedXPathMapping ? "xpath" : "percentage", xpathExactMatch ? "yes" : "no"); return result; } std::string ProgressMapper::generateXPath(int spineIndex, int pageNumber, int totalPages) { - // Use 0-based DocFragment indices for KOReader - // Use a simple xpath pointing to the DocFragment - KOReader will use the percentage for fine positioning within it - // Avoid specifying paragraph numbers as they may not exist in the target document + // Fallback path when element-level XPath extraction is unavailable. + // Uses 0-based DocFragment indices for KOReader compatibility. return "/body/DocFragment[" + std::to_string(spineIndex) + "]/body"; } diff --git a/lib/KOReaderSync/ProgressMapper.h b/lib/KOReaderSync/ProgressMapper.h index 53ff7696..e4e056e4 100644 --- a/lib/KOReaderSync/ProgressMapper.h +++ b/lib/KOReaderSync/ProgressMapper.h @@ -27,9 +27,9 @@ struct KOReaderPosition { * CrossPoint tracks position as (spineIndex, pageNumber). * KOReader uses XPath-like strings + percentage. * - * Since CrossPoint discards HTML structure during parsing, we generate - * synthetic XPath strings based on spine index, using percentage as the - * primary sync mechanism. + * CrossPoint first tries to extract an element-level XPath by reparsing the + * current spine XHTML and mapping intra-spine progress to text anchors. + * If extraction fails, it falls back to a synthetic chapter-level XPath. */ class ProgressMapper { public: @@ -60,8 +60,7 @@ class ProgressMapper { private: /** * Generate XPath for KOReader compatibility. - * Format: /body/DocFragment[spineIndex+1]/body - * Since CrossPoint doesn't preserve HTML structure, we rely on percentage for positioning. + * Fallback format: /body/DocFragment[spineIndex]/body */ static std::string generateXPath(int spineIndex, int pageNumber, int totalPages); };