fix: Switch to xpath map for paragraph level syncing in KOSync (#1686)
Switch KOReader sync progress mapping from chapter matching to XPath-based mapping. - resolves KOReader positions using real XHTML ancestry paths - supports paragraph-based upload mapping with text offsets where needed - passes the current paragraph index into sync so uploads map back to KOReader more accurately No HTTP client changes are included. No reader-state or resume-flow changes are included. --------- Co-authored-by: jpirnay <jens@pirnay.com>
This commit is contained in:
co-authored by
jpirnay
parent
e8645ed92e
commit
302dea1eea
@@ -0,0 +1,563 @@
|
||||
#include "ChapterXPathResolver.h"
|
||||
|
||||
#include <Logging.h>
|
||||
#include <Print.h>
|
||||
#include <Utf8.h>
|
||||
#include <XmlParserUtils.h>
|
||||
#include <expat.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
#include <string>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace {
|
||||
std::string stripPrefix(const XML_Char* name) {
|
||||
if (!name) {
|
||||
return "";
|
||||
}
|
||||
|
||||
const char* local = std::strrchr(name, ':');
|
||||
return local ? std::string(local + 1) : std::string(name);
|
||||
}
|
||||
|
||||
struct NameCounter {
|
||||
std::string name;
|
||||
int count;
|
||||
};
|
||||
|
||||
struct ParentState {
|
||||
std::vector<NameCounter> children;
|
||||
|
||||
int nextIndex(const std::string& name) {
|
||||
for (auto& child : children) {
|
||||
if (child.name == name) {
|
||||
child.count++;
|
||||
return child.count;
|
||||
}
|
||||
}
|
||||
|
||||
children.push_back({name, 1});
|
||||
return 1;
|
||||
}
|
||||
};
|
||||
|
||||
struct PathSegment {
|
||||
std::string name;
|
||||
int index;
|
||||
};
|
||||
|
||||
std::string buildParagraphXPath(const int spineIndex, const std::vector<PathSegment>& path, const int charOffset) {
|
||||
std::string xpath = "/body/DocFragment[" + std::to_string(spineIndex + 1) + "]/body";
|
||||
for (const auto& segment : path) {
|
||||
xpath += "/" + segment.name + "[" + std::to_string(segment.index) + "]";
|
||||
}
|
||||
if (charOffset > 0) {
|
||||
xpath += "/text()." + std::to_string(charOffset);
|
||||
}
|
||||
return xpath;
|
||||
}
|
||||
|
||||
size_t countUtf8Codepoints(const XML_Char* data, const int len) {
|
||||
if (!data || len <= 0) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
size_t count = 0;
|
||||
const unsigned char* ptr = reinterpret_cast<const unsigned char*>(data);
|
||||
const unsigned char* end = ptr + len;
|
||||
while (ptr < end) {
|
||||
utf8NextCodepoint(&ptr);
|
||||
count++;
|
||||
}
|
||||
|
||||
return count;
|
||||
}
|
||||
|
||||
class ParagraphTextCounter final : public Print {
|
||||
public:
|
||||
ParagraphTextCounter() {
|
||||
parser = XML_ParserCreate(nullptr);
|
||||
if (!parser) {
|
||||
LOG_ERR("KOX", "Failed to create XML parser");
|
||||
return;
|
||||
}
|
||||
|
||||
XML_SetUserData(parser, this);
|
||||
XML_SetElementHandler(parser, &ParagraphTextCounter::startElement, &ParagraphTextCounter::endElement);
|
||||
XML_SetCharacterDataHandler(parser, &ParagraphTextCounter::characterData);
|
||||
}
|
||||
|
||||
~ParagraphTextCounter() override { destroyXmlParser(parser); }
|
||||
|
||||
bool ok() const { return parser != nullptr && parseOk; }
|
||||
|
||||
bool finish() {
|
||||
if (!parser || !parseOk || stopped) {
|
||||
return parseOk;
|
||||
}
|
||||
|
||||
if (XML_Parse(parser, "", 0, XML_TRUE) == XML_STATUS_ERROR) {
|
||||
LOG_ERR("KOX", "Final XML parse error: %s", XML_ErrorString(XML_GetErrorCode(parser)));
|
||||
parseOk = false;
|
||||
}
|
||||
return parseOk;
|
||||
}
|
||||
|
||||
size_t write(uint8_t c) override { return write(&c, 1); }
|
||||
|
||||
size_t write(const uint8_t* buffer, size_t size) override {
|
||||
if (!parser || !parseOk || stopped) {
|
||||
return size;
|
||||
}
|
||||
|
||||
if (XML_Parse(parser, reinterpret_cast<const char*>(buffer), static_cast<int>(size), XML_FALSE) != XML_STATUS_OK) {
|
||||
const enum XML_Error error = XML_GetErrorCode(parser);
|
||||
if (error != XML_ERROR_ABORTED) {
|
||||
LOG_ERR("KOX", "XML parse error: %s", XML_ErrorString(error));
|
||||
parseOk = false;
|
||||
}
|
||||
}
|
||||
|
||||
return size;
|
||||
}
|
||||
|
||||
size_t totalVisibleChars() const { return visibleChars; }
|
||||
|
||||
private:
|
||||
static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char**) {
|
||||
auto* self = static_cast<ParagraphTextCounter*>(userData);
|
||||
self->onStartElement(name);
|
||||
}
|
||||
|
||||
static void XMLCALL endElement(void* userData, const XML_Char* name) {
|
||||
auto* self = static_cast<ParagraphTextCounter*>(userData);
|
||||
self->onEndElement(name);
|
||||
}
|
||||
|
||||
static void XMLCALL characterData(void* userData, const XML_Char* data, const int len) {
|
||||
auto* self = static_cast<ParagraphTextCounter*>(userData);
|
||||
self->onCharacterData(data, len);
|
||||
}
|
||||
|
||||
void onStartElement(const XML_Char* rawName) {
|
||||
const std::string name = stripPrefix(rawName);
|
||||
|
||||
if (!insideBody) {
|
||||
if (name == "body") {
|
||||
insideBody = true;
|
||||
bodyDepth = depth;
|
||||
}
|
||||
depth++;
|
||||
return;
|
||||
}
|
||||
|
||||
if (name == "p") {
|
||||
paragraphDepth++;
|
||||
}
|
||||
depth++;
|
||||
}
|
||||
|
||||
void onEndElement(const XML_Char* rawName) {
|
||||
const std::string name = stripPrefix(rawName);
|
||||
|
||||
depth--;
|
||||
if (!insideBody) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (depth == bodyDepth && name == "body") {
|
||||
insideBody = false;
|
||||
return;
|
||||
}
|
||||
|
||||
if (name == "p" && paragraphDepth > 0) {
|
||||
paragraphDepth--;
|
||||
}
|
||||
}
|
||||
|
||||
void onCharacterData(const XML_Char* data, const int len) {
|
||||
if (!insideBody || paragraphDepth <= 0 || len <= 0) {
|
||||
return;
|
||||
}
|
||||
|
||||
visibleChars += countUtf8Codepoints(data, len);
|
||||
}
|
||||
|
||||
private:
|
||||
XML_Parser parser = nullptr;
|
||||
bool parseOk = true;
|
||||
bool insideBody = false;
|
||||
bool stopped = false;
|
||||
int depth = 0;
|
||||
int bodyDepth = -1;
|
||||
int paragraphDepth = 0;
|
||||
size_t visibleChars = 0;
|
||||
};
|
||||
|
||||
class XPathParagraphResolver final : public Print {
|
||||
public:
|
||||
explicit XPathParagraphResolver(const int targetParagraph) : targetParagraph(targetParagraph) {
|
||||
parser = XML_ParserCreate(nullptr);
|
||||
if (!parser) {
|
||||
LOG_ERR("KOX", "Failed to create XML parser");
|
||||
return;
|
||||
}
|
||||
|
||||
XML_SetUserData(parser, this);
|
||||
XML_SetElementHandler(parser, &XPathParagraphResolver::startElement, &XPathParagraphResolver::endElement);
|
||||
}
|
||||
|
||||
~XPathParagraphResolver() override { destroyXmlParser(parser); }
|
||||
|
||||
bool ok() const { return parser != nullptr && parseOk; }
|
||||
|
||||
bool finish() {
|
||||
if (!parser || !parseOk || stopped) {
|
||||
return parseOk;
|
||||
}
|
||||
|
||||
if (XML_Parse(parser, "", 0, XML_TRUE) == XML_STATUS_ERROR) {
|
||||
LOG_ERR("KOX", "Final XML parse error: %s", XML_ErrorString(XML_GetErrorCode(parser)));
|
||||
parseOk = false;
|
||||
}
|
||||
return parseOk;
|
||||
}
|
||||
|
||||
bool hasMatch() const { return !xpath.empty(); }
|
||||
const std::string& getXPath() const { return xpath; }
|
||||
|
||||
size_t write(uint8_t c) override { return write(&c, 1); }
|
||||
|
||||
size_t write(const uint8_t* buffer, size_t size) override {
|
||||
if (!parser || !parseOk || stopped) {
|
||||
return size;
|
||||
}
|
||||
|
||||
if (XML_Parse(parser, reinterpret_cast<const char*>(buffer), static_cast<int>(size), XML_FALSE) != XML_STATUS_OK) {
|
||||
const enum XML_Error error = XML_GetErrorCode(parser);
|
||||
if (error != XML_ERROR_ABORTED) {
|
||||
LOG_ERR("KOX", "XML parse error: %s", XML_ErrorString(error));
|
||||
parseOk = false;
|
||||
}
|
||||
}
|
||||
|
||||
return size;
|
||||
}
|
||||
|
||||
int spineIndex = 0;
|
||||
|
||||
private:
|
||||
static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char**) {
|
||||
auto* self = static_cast<XPathParagraphResolver*>(userData);
|
||||
self->onStartElement(name);
|
||||
}
|
||||
|
||||
static void XMLCALL endElement(void* userData, const XML_Char* name) {
|
||||
auto* self = static_cast<XPathParagraphResolver*>(userData);
|
||||
self->onEndElement(name);
|
||||
}
|
||||
|
||||
void onStartElement(const XML_Char* rawName) {
|
||||
const std::string name = stripPrefix(rawName);
|
||||
|
||||
if (!insideBody) {
|
||||
if (name == "body") {
|
||||
insideBody = true;
|
||||
bodyDepth = depth;
|
||||
parentStates.emplace_back();
|
||||
}
|
||||
depth++;
|
||||
return;
|
||||
}
|
||||
|
||||
const int siblingIndex = parentStates.back().nextIndex(name);
|
||||
path.push_back({name, siblingIndex});
|
||||
parentStates.emplace_back();
|
||||
|
||||
if (name == "p") {
|
||||
paragraphCount++;
|
||||
if (paragraphCount == targetParagraph) {
|
||||
xpath = buildParagraphXPath(spineIndex, path, 0);
|
||||
stopped = true;
|
||||
XML_StopParser(parser, XML_FALSE);
|
||||
}
|
||||
}
|
||||
|
||||
depth++;
|
||||
}
|
||||
|
||||
void onEndElement(const XML_Char* rawName) {
|
||||
const std::string name = stripPrefix(rawName);
|
||||
|
||||
depth--;
|
||||
if (!insideBody) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (depth == bodyDepth && name == "body") {
|
||||
insideBody = false;
|
||||
parentStates.clear();
|
||||
path.clear();
|
||||
return;
|
||||
}
|
||||
|
||||
if (!path.empty()) {
|
||||
path.pop_back();
|
||||
}
|
||||
if (!parentStates.empty()) {
|
||||
parentStates.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
XML_Parser parser = nullptr;
|
||||
const int targetParagraph;
|
||||
bool parseOk = true;
|
||||
bool insideBody = false;
|
||||
bool stopped = false;
|
||||
int depth = 0;
|
||||
int bodyDepth = -1;
|
||||
int paragraphCount = 0;
|
||||
std::vector<ParentState> parentStates;
|
||||
std::vector<PathSegment> path;
|
||||
std::string xpath;
|
||||
};
|
||||
|
||||
class XPathProgressResolver final : public Print {
|
||||
public:
|
||||
explicit XPathProgressResolver(const size_t targetVisibleChar) : targetVisibleChar(targetVisibleChar) {
|
||||
parser = XML_ParserCreate(nullptr);
|
||||
if (!parser) {
|
||||
LOG_ERR("KOX", "Failed to create XML parser");
|
||||
return;
|
||||
}
|
||||
|
||||
XML_SetUserData(parser, this);
|
||||
XML_SetElementHandler(parser, &XPathProgressResolver::startElement, &XPathProgressResolver::endElement);
|
||||
XML_SetCharacterDataHandler(parser, &XPathProgressResolver::characterData);
|
||||
}
|
||||
|
||||
~XPathProgressResolver() override { destroyXmlParser(parser); }
|
||||
|
||||
bool ok() const { return parser != nullptr && parseOk; }
|
||||
|
||||
bool finish() {
|
||||
if (!parser || !parseOk || stopped) {
|
||||
return parseOk;
|
||||
}
|
||||
|
||||
if (XML_Parse(parser, "", 0, XML_TRUE) == XML_STATUS_ERROR) {
|
||||
LOG_ERR("KOX", "Final XML parse error: %s", XML_ErrorString(XML_GetErrorCode(parser)));
|
||||
parseOk = false;
|
||||
}
|
||||
return parseOk;
|
||||
}
|
||||
|
||||
bool hasMatch() const { return !xpath.empty(); }
|
||||
const std::string& getXPath() const { return xpath; }
|
||||
|
||||
size_t write(uint8_t c) override { return write(&c, 1); }
|
||||
|
||||
size_t write(const uint8_t* buffer, size_t size) override {
|
||||
if (!parser || !parseOk || stopped) {
|
||||
return size;
|
||||
}
|
||||
|
||||
if (XML_Parse(parser, reinterpret_cast<const char*>(buffer), static_cast<int>(size), XML_FALSE) != XML_STATUS_OK) {
|
||||
const enum XML_Error error = XML_GetErrorCode(parser);
|
||||
if (error != XML_ERROR_ABORTED) {
|
||||
LOG_ERR("KOX", "XML parse error: %s", XML_ErrorString(error));
|
||||
parseOk = false;
|
||||
}
|
||||
}
|
||||
|
||||
return size;
|
||||
}
|
||||
|
||||
int spineIndex = 0;
|
||||
|
||||
private:
|
||||
static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char**) {
|
||||
auto* self = static_cast<XPathProgressResolver*>(userData);
|
||||
self->onStartElement(name);
|
||||
}
|
||||
|
||||
static void XMLCALL endElement(void* userData, const XML_Char* name) {
|
||||
auto* self = static_cast<XPathProgressResolver*>(userData);
|
||||
self->onEndElement(name);
|
||||
}
|
||||
|
||||
static void XMLCALL characterData(void* userData, const XML_Char* data, const int len) {
|
||||
auto* self = static_cast<XPathProgressResolver*>(userData);
|
||||
self->onCharacterData(data, len);
|
||||
}
|
||||
|
||||
void onStartElement(const XML_Char* rawName) {
|
||||
const std::string name = stripPrefix(rawName);
|
||||
|
||||
if (!insideBody) {
|
||||
if (name == "body") {
|
||||
insideBody = true;
|
||||
bodyDepth = depth;
|
||||
parentStates.emplace_back();
|
||||
}
|
||||
depth++;
|
||||
return;
|
||||
}
|
||||
|
||||
const int siblingIndex = parentStates.back().nextIndex(name);
|
||||
path.push_back({name, siblingIndex});
|
||||
parentStates.emplace_back();
|
||||
|
||||
if (name == "p") {
|
||||
paragraphDepth++;
|
||||
paragraphVisibleChars = 0;
|
||||
}
|
||||
|
||||
depth++;
|
||||
}
|
||||
|
||||
void onEndElement(const XML_Char* rawName) {
|
||||
const std::string name = stripPrefix(rawName);
|
||||
|
||||
depth--;
|
||||
if (!insideBody) {
|
||||
return;
|
||||
}
|
||||
|
||||
if (depth == bodyDepth && name == "body") {
|
||||
insideBody = false;
|
||||
parentStates.clear();
|
||||
path.clear();
|
||||
return;
|
||||
}
|
||||
|
||||
if (name == "p" && paragraphDepth > 0) {
|
||||
paragraphDepth--;
|
||||
paragraphVisibleChars = 0;
|
||||
}
|
||||
|
||||
if (!path.empty()) {
|
||||
path.pop_back();
|
||||
}
|
||||
if (!parentStates.empty()) {
|
||||
parentStates.pop_back();
|
||||
}
|
||||
}
|
||||
|
||||
void onCharacterData(const XML_Char* data, const int len) {
|
||||
if (!insideBody || paragraphDepth <= 0 || len <= 0 || stopped) {
|
||||
return;
|
||||
}
|
||||
|
||||
const size_t codepointCount = countUtf8Codepoints(data, len);
|
||||
const size_t nextVisibleChars = visibleChars + codepointCount;
|
||||
if (targetVisibleChar <= nextVisibleChars) {
|
||||
const size_t delta = targetVisibleChar - visibleChars;
|
||||
const int charOffset = static_cast<int>(paragraphVisibleChars + delta);
|
||||
xpath = buildParagraphXPath(spineIndex, path, std::max(1, charOffset));
|
||||
stopped = true;
|
||||
XML_StopParser(parser, XML_FALSE);
|
||||
return;
|
||||
}
|
||||
|
||||
visibleChars = nextVisibleChars;
|
||||
paragraphVisibleChars += codepointCount;
|
||||
}
|
||||
|
||||
XML_Parser parser = nullptr;
|
||||
const size_t targetVisibleChar;
|
||||
bool parseOk = true;
|
||||
bool insideBody = false;
|
||||
bool stopped = false;
|
||||
int depth = 0;
|
||||
int bodyDepth = -1;
|
||||
int paragraphDepth = 0;
|
||||
size_t visibleChars = 0;
|
||||
size_t paragraphVisibleChars = 0;
|
||||
std::vector<ParentState> parentStates;
|
||||
std::vector<PathSegment> path;
|
||||
std::string xpath;
|
||||
};
|
||||
} // namespace
|
||||
|
||||
std::string ChapterXPathResolver::findXPathForParagraph(const std::shared_ptr<Epub>& epub, const int spineIndex,
|
||||
const uint16_t paragraphIndex) {
|
||||
if (!epub || paragraphIndex == 0 || spineIndex < 0 || spineIndex >= epub->getSpineItemsCount()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
const auto href = epub->getSpineItem(spineIndex).href;
|
||||
if (href.empty()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
XPathParagraphResolver resolver(paragraphIndex);
|
||||
if (!resolver.ok()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
resolver.spineIndex = spineIndex;
|
||||
if (!epub->readItemContentsToStream(href, resolver, 1024) || !resolver.finish()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
if (resolver.hasMatch()) {
|
||||
LOG_DBG("KOX", "Resolved paragraph %u in spine %d -> %s", paragraphIndex, spineIndex, resolver.getXPath().c_str());
|
||||
return resolver.getXPath();
|
||||
}
|
||||
|
||||
LOG_DBG("KOX", "Paragraph %u not found in spine %d", paragraphIndex, spineIndex);
|
||||
return "";
|
||||
}
|
||||
|
||||
std::string ChapterXPathResolver::findXPathForProgress(const std::shared_ptr<Epub>& epub, const int spineIndex,
|
||||
const float intraSpineProgress) {
|
||||
if (!epub || spineIndex < 0 || spineIndex >= epub->getSpineItemsCount()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
const auto href = epub->getSpineItem(spineIndex).href;
|
||||
if (href.empty()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
if (!(intraSpineProgress > 0.0f)) {
|
||||
return "/body/DocFragment[" + std::to_string(spineIndex + 1) + "]/body";
|
||||
}
|
||||
|
||||
ParagraphTextCounter counter;
|
||||
if (!counter.ok() || !epub->readItemContentsToStream(href, counter, 1024) || !counter.finish()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
const size_t totalVisibleChars = counter.totalVisibleChars();
|
||||
if (totalVisibleChars == 0) {
|
||||
return "";
|
||||
}
|
||||
|
||||
const float clamped = std::max(0.0f, std::min(1.0f, intraSpineProgress));
|
||||
const size_t targetVisibleChar =
|
||||
std::max<size_t>(1, std::min(totalVisibleChars, static_cast<size_t>(std::ceil(clamped * totalVisibleChars))));
|
||||
|
||||
XPathProgressResolver resolver(targetVisibleChar);
|
||||
if (!resolver.ok()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
resolver.spineIndex = spineIndex;
|
||||
if (!epub->readItemContentsToStream(href, resolver, 1024) || !resolver.finish()) {
|
||||
return "";
|
||||
}
|
||||
|
||||
if (resolver.hasMatch()) {
|
||||
LOG_DBG("KOX", "Resolved progress %.3f in spine %d -> %s", intraSpineProgress, spineIndex,
|
||||
resolver.getXPath().c_str());
|
||||
return resolver.getXPath();
|
||||
}
|
||||
|
||||
LOG_DBG("KOX", "Could not resolve progress %.3f in spine %d", intraSpineProgress, spineIndex);
|
||||
return "";
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
#pragma once
|
||||
|
||||
#include <Epub.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <memory>
|
||||
#include <string>
|
||||
|
||||
class ChapterXPathResolver {
|
||||
public:
|
||||
/**
|
||||
* Resolve the Nth paragraph in a spine item to its real XHTML ancestry path.
|
||||
*
|
||||
* Returns a KOReader-compatible path like:
|
||||
* /body/DocFragment[8]/body/div[2]/section[1]/p[4]
|
||||
*
|
||||
* An empty string means parsing failed or the paragraph index was not found.
|
||||
*/
|
||||
static std::string findXPathForParagraph(const std::shared_ptr<Epub>& epub, int spineIndex, uint16_t paragraphIndex);
|
||||
|
||||
/**
|
||||
* Resolve intra-spine progress to a real XHTML ancestry path plus text offset.
|
||||
*
|
||||
* Returns a KOReader-compatible path like:
|
||||
* /body/DocFragment[8]/body/div[2]/section[1]/p[4]/text().96
|
||||
*
|
||||
* An empty string means parsing failed or the location could not be resolved.
|
||||
*/
|
||||
static std::string findXPathForProgress(const std::shared_ptr<Epub>& epub, int spineIndex, float intraSpineProgress);
|
||||
};
|
||||
@@ -2,113 +2,294 @@
|
||||
|
||||
#include <Logging.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
#include <cstring>
|
||||
|
||||
#include "ChapterXPathResolver.h"
|
||||
#include "Epub/htmlEntities.h"
|
||||
#include "Utf8.h"
|
||||
|
||||
namespace {
|
||||
int parseIndex(const std::string& xpath, const char* prefix, bool last = false) {
|
||||
const size_t prefixLen = strlen(prefix);
|
||||
const size_t pos = last ? xpath.rfind(prefix) : xpath.find(prefix);
|
||||
if (pos == std::string::npos) return -1;
|
||||
const size_t numStart = pos + prefixLen;
|
||||
const size_t numEnd = xpath.find(']', numStart);
|
||||
if (numEnd == std::string::npos || numEnd == numStart) return -1;
|
||||
int val = 0;
|
||||
for (size_t i = numStart; i < numEnd; i++) {
|
||||
if (xpath[i] < '0' || xpath[i] > '9') return -1;
|
||||
val = val * 10 + (xpath[i] - '0');
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
int parseCharOffset(const std::string& xpath) {
|
||||
const size_t textPos = xpath.rfind("text()");
|
||||
if (textPos == std::string::npos) return 0;
|
||||
const size_t dotPos = xpath.find('.', textPos);
|
||||
if (dotPos == std::string::npos || dotPos + 1 >= xpath.size()) return 0;
|
||||
int val = 0;
|
||||
for (size_t i = dotPos + 1; i < xpath.size(); i++) {
|
||||
if (xpath[i] < '0' || xpath[i] > '9') return 0;
|
||||
val = val * 10 + (xpath[i] - '0');
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
class ParagraphStreamer final : public Print {
|
||||
size_t bytesWritten = 0;
|
||||
bool globalInTag = false;
|
||||
bool globalInEntity = false;
|
||||
enum { IDLE, SAW_LT, SAW_LT_P } pState = IDLE;
|
||||
static constexpr size_t MAX_ENTITY_SIZE = 16;
|
||||
char entityBuffer[MAX_ENTITY_SIZE] = {};
|
||||
size_t entityLen = 0;
|
||||
|
||||
// Forward mode: count paragraphs at a byte offset
|
||||
size_t fwdTarget;
|
||||
int fwdResult = 0;
|
||||
bool fwdCaptured = false;
|
||||
|
||||
// Reverse mode: find position of Nth paragraph + char offset
|
||||
int revParagraph;
|
||||
int revChar;
|
||||
int pCount = 0;
|
||||
bool revPFound = false;
|
||||
bool revDone = false;
|
||||
int revVisChars = 0; // Visible chars counted WITHIN target paragraph
|
||||
size_t totalVisChars = 0; // Total visible chars in entire file
|
||||
size_t targetVisChars = 0; // Visible chars from start of file to target position
|
||||
|
||||
void onP() {
|
||||
pCount++;
|
||||
if (!revPFound && revParagraph > 0 && pCount >= revParagraph) {
|
||||
revPFound = true;
|
||||
revVisChars = 0;
|
||||
if (revChar <= 0) {
|
||||
targetVisChars = totalVisChars;
|
||||
revDone = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void onVisibleCodepoint() {
|
||||
totalVisChars++;
|
||||
if (revPFound && !revDone) {
|
||||
revVisChars++;
|
||||
if (revVisChars >= revChar) {
|
||||
targetVisChars = totalVisChars;
|
||||
revDone = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void onVisibleText(const char* text) {
|
||||
if (!text) {
|
||||
return;
|
||||
}
|
||||
|
||||
const unsigned char* ptr = reinterpret_cast<const unsigned char*>(text);
|
||||
while (*ptr != 0) {
|
||||
utf8NextCodepoint(&ptr);
|
||||
onVisibleCodepoint();
|
||||
}
|
||||
}
|
||||
|
||||
void flushEntityAsLiteral() {
|
||||
for (size_t i = 0; i < entityLen; i++) {
|
||||
onVisibleCodepoint();
|
||||
}
|
||||
}
|
||||
|
||||
void finishEntity() {
|
||||
entityBuffer[entityLen] = '\0';
|
||||
const char* resolved = lookupHtmlEntity(entityBuffer, entityLen);
|
||||
if (resolved) {
|
||||
onVisibleText(resolved);
|
||||
} else {
|
||||
flushEntityAsLiteral();
|
||||
}
|
||||
globalInEntity = false;
|
||||
entityLen = 0;
|
||||
}
|
||||
|
||||
public:
|
||||
explicit ParagraphStreamer(size_t targetByte) : fwdTarget(targetByte), revParagraph(0), revChar(0) {}
|
||||
ParagraphStreamer(int paragraph, int charOff) : fwdTarget(SIZE_MAX), revParagraph(paragraph), revChar(charOff) {}
|
||||
|
||||
size_t write(uint8_t c) override {
|
||||
if (!fwdCaptured && bytesWritten >= fwdTarget) {
|
||||
fwdResult = pCount;
|
||||
fwdCaptured = true;
|
||||
}
|
||||
bytesWritten++;
|
||||
|
||||
if (globalInEntity) {
|
||||
if (entityLen + 1 < MAX_ENTITY_SIZE) {
|
||||
entityBuffer[entityLen++] = static_cast<char>(c);
|
||||
} else {
|
||||
flushEntityAsLiteral();
|
||||
globalInEntity = false;
|
||||
entityLen = 0;
|
||||
}
|
||||
|
||||
if (globalInEntity) {
|
||||
if (c == ';') {
|
||||
finishEntity();
|
||||
} else if (c == '<' || c == ' ' || c == '\t' || c == '\n' || c == '\r') {
|
||||
flushEntityAsLiteral();
|
||||
globalInEntity = false;
|
||||
entityLen = 0;
|
||||
}
|
||||
}
|
||||
} else if (c == '<') {
|
||||
globalInTag = true;
|
||||
} else if (c == '>') {
|
||||
globalInTag = false;
|
||||
} else if (!globalInTag) {
|
||||
if (c == '&') {
|
||||
globalInEntity = true;
|
||||
entityBuffer[0] = '&';
|
||||
entityLen = 1;
|
||||
} else {
|
||||
const bool startsCodepoint = (c & 0xC0) != 0x80;
|
||||
if (startsCodepoint) {
|
||||
onVisibleCodepoint();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Paragraph detection
|
||||
switch (pState) {
|
||||
case IDLE:
|
||||
if (c == '<') pState = SAW_LT;
|
||||
break;
|
||||
case SAW_LT:
|
||||
pState = (c == 'p' || c == 'P') ? SAW_LT_P : ((c == '<') ? SAW_LT : IDLE);
|
||||
break;
|
||||
case SAW_LT_P:
|
||||
if (c == '>' || c == '/' || c == ' ' || c == '\t' || c == '\n' || c == '\r') onP();
|
||||
pState = (c == '<') ? SAW_LT : IDLE;
|
||||
break;
|
||||
}
|
||||
return 1;
|
||||
}
|
||||
|
||||
size_t write(const uint8_t* buffer, size_t size) override {
|
||||
for (size_t i = 0; i < size; i++) write(buffer[i]);
|
||||
return size;
|
||||
}
|
||||
|
||||
public:
|
||||
int paragraphCount() const { return fwdCaptured ? fwdResult : pCount; }
|
||||
size_t totalBytes() const { return bytesWritten; }
|
||||
bool found() const { return revDone || revPFound; }
|
||||
float progress() const {
|
||||
return totalVisChars > 0 ? static_cast<float>(targetVisChars) / static_cast<float>(totalVisChars) : 0.0f;
|
||||
}
|
||||
};
|
||||
|
||||
bool streamSpine(const std::shared_ptr<Epub>& epub, int spineIndex, ParagraphStreamer& s) {
|
||||
const auto href = epub->getSpineItem(spineIndex).href;
|
||||
return !href.empty() && epub->readItemContentsToStream(href, s, 1024);
|
||||
}
|
||||
} // namespace
|
||||
|
||||
KOReaderPosition ProgressMapper::toKOReader(const std::shared_ptr<Epub>& epub, const CrossPointPosition& pos) {
|
||||
KOReaderPosition result;
|
||||
|
||||
// Calculate page progress within current spine item
|
||||
float intraSpineProgress = 0.0f;
|
||||
if (pos.totalPages > 0) {
|
||||
intraSpineProgress = static_cast<float>(pos.pageNumber) / static_cast<float>(pos.totalPages);
|
||||
float intra = (pos.totalPages > 0) ? static_cast<float>(pos.pageNumber) / static_cast<float>(pos.totalPages) : 0.0f;
|
||||
result.percentage = epub->calculateProgress(pos.spineIndex, intra);
|
||||
if (pos.hasParagraphIndex && pos.paragraphIndex > 0) {
|
||||
result.xpath = ChapterXPathResolver::findXPathForParagraph(epub, pos.spineIndex, pos.paragraphIndex);
|
||||
} else {
|
||||
result.xpath = ChapterXPathResolver::findXPathForProgress(epub, pos.spineIndex, intra);
|
||||
}
|
||||
|
||||
// Calculate overall book progress (0.0-1.0)
|
||||
result.percentage = epub->calculateProgress(pos.spineIndex, intraSpineProgress);
|
||||
|
||||
// Generate XPath with estimated paragraph position based on page
|
||||
result.xpath = generateXPath(pos.spineIndex, pos.pageNumber, pos.totalPages);
|
||||
|
||||
// Get chapter info for logging
|
||||
const int tocIndex = epub->getTocIndexForSpineIndex(pos.spineIndex);
|
||||
const std::string chapterName = (tocIndex >= 0) ? epub->getTocItem(tocIndex).title : "unknown";
|
||||
|
||||
LOG_DBG("ProgressMapper", "CrossPoint -> KOReader: chapter='%s', page=%d/%d -> %.2f%% at %s", chapterName.c_str(),
|
||||
pos.pageNumber, pos.totalPages, result.percentage * 100, result.xpath.c_str());
|
||||
|
||||
if (result.xpath.empty()) {
|
||||
result.xpath = generateXPath(epub, pos.spineIndex, intra);
|
||||
}
|
||||
LOG_DBG("PM", "-> KO: spine=%d page=%d/%d %.2f%% %s", pos.spineIndex, pos.pageNumber, pos.totalPages,
|
||||
result.percentage * 100, result.xpath.c_str());
|
||||
return result;
|
||||
}
|
||||
|
||||
CrossPointPosition ProgressMapper::toCrossPoint(const std::shared_ptr<Epub>& epub, const KOReaderPosition& koPos,
|
||||
int currentSpineIndex, int totalPagesInCurrentSpine) {
|
||||
CrossPointPosition result;
|
||||
result.spineIndex = 0;
|
||||
result.pageNumber = 0;
|
||||
result.totalPages = 0;
|
||||
|
||||
CrossPointPosition result{};
|
||||
const size_t bookSize = epub->getBookSize();
|
||||
if (bookSize == 0) {
|
||||
return result;
|
||||
}
|
||||
if (bookSize == 0) return result;
|
||||
|
||||
// Use percentage-based lookup for both spine and page positioning
|
||||
// XPath parsing is unreliable since CrossPoint doesn't preserve detailed HTML structure
|
||||
const size_t targetBytes = static_cast<size_t>(bookSize * koPos.percentage);
|
||||
|
||||
// Find the spine item that contains this byte position
|
||||
const int spineCount = epub->getSpineItemsCount();
|
||||
bool spineFound = false;
|
||||
for (int i = 0; i < spineCount; i++) {
|
||||
const size_t cumulativeSize = epub->getCumulativeSpineItemSize(i);
|
||||
if (cumulativeSize >= targetBytes) {
|
||||
result.spineIndex = i;
|
||||
spineFound = true;
|
||||
break;
|
||||
}
|
||||
const float clampedPercentage = std::max(0.0f, std::min(1.0f, koPos.percentage));
|
||||
const size_t targetBytes = static_cast<size_t>(static_cast<float>(bookSize) * clampedPercentage);
|
||||
|
||||
const int docFrag = parseIndex(koPos.xpath, "/body/DocFragment[");
|
||||
const int xpathP = parseIndex(koPos.xpath, "/p[", true);
|
||||
const int xpathChar = parseCharOffset(koPos.xpath);
|
||||
const int xpathSpine = (docFrag >= 1) ? (docFrag - 1) : -1;
|
||||
if (xpathP > 0) {
|
||||
result.paragraphIndex = static_cast<uint16_t>(xpathP);
|
||||
result.hasParagraphIndex = true;
|
||||
}
|
||||
|
||||
// If no spine item was found (e.g., targetBytes beyond last cumulative size),
|
||||
// default to the last spine item so we map to the end of the book instead of the beginning.
|
||||
if (!spineFound && spineCount > 0) {
|
||||
result.spineIndex = spineCount - 1;
|
||||
}
|
||||
|
||||
// Estimate page number within the spine item using percentage
|
||||
if (result.spineIndex < epub->getSpineItemsCount()) {
|
||||
const size_t prevCumSize = (result.spineIndex > 0) ? epub->getCumulativeSpineItemSize(result.spineIndex - 1) : 0;
|
||||
const size_t currentCumSize = epub->getCumulativeSpineItemSize(result.spineIndex);
|
||||
const size_t spineSize = currentCumSize - prevCumSize;
|
||||
|
||||
int estimatedTotalPages = 0;
|
||||
|
||||
// If we are in the same spine, use the known total pages
|
||||
if (result.spineIndex == currentSpineIndex && totalPagesInCurrentSpine > 0) {
|
||||
estimatedTotalPages = totalPagesInCurrentSpine;
|
||||
}
|
||||
// Otherwise try to estimate based on density from current spine
|
||||
else if (currentSpineIndex >= 0 && currentSpineIndex < epub->getSpineItemsCount() && totalPagesInCurrentSpine > 0) {
|
||||
const size_t prevCurrCumSize =
|
||||
(currentSpineIndex > 0) ? epub->getCumulativeSpineItemSize(currentSpineIndex - 1) : 0;
|
||||
const size_t currCumSize = epub->getCumulativeSpineItemSize(currentSpineIndex);
|
||||
const size_t currSpineSize = currCumSize - prevCurrCumSize;
|
||||
|
||||
if (currSpineSize > 0) {
|
||||
float ratio = static_cast<float>(spineSize) / static_cast<float>(currSpineSize);
|
||||
estimatedTotalPages = static_cast<int>(totalPagesInCurrentSpine * ratio);
|
||||
if (estimatedTotalPages < 1) estimatedTotalPages = 1;
|
||||
if (xpathSpine >= 0 && xpathSpine < spineCount) {
|
||||
result.spineIndex = xpathSpine;
|
||||
} else {
|
||||
for (int i = 0; i < spineCount; i++) {
|
||||
if (epub->getCumulativeSpineItemSize(i) >= targetBytes) {
|
||||
result.spineIndex = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (result.spineIndex >= spineCount) return result;
|
||||
|
||||
result.totalPages = estimatedTotalPages;
|
||||
const size_t prevCum = (result.spineIndex > 0) ? epub->getCumulativeSpineItemSize(result.spineIndex - 1) : 0;
|
||||
const size_t spineSize = epub->getCumulativeSpineItemSize(result.spineIndex) - prevCum;
|
||||
|
||||
if (spineSize > 0 && estimatedTotalPages > 0) {
|
||||
const size_t bytesIntoSpine = (targetBytes > prevCumSize) ? (targetBytes - prevCumSize) : 0;
|
||||
const float intraSpineProgress = static_cast<float>(bytesIntoSpine) / static_cast<float>(spineSize);
|
||||
const float clampedProgress = std::max(0.0f, std::min(1.0f, intraSpineProgress));
|
||||
result.pageNumber = static_cast<int>(clampedProgress * estimatedTotalPages);
|
||||
result.pageNumber = std::max(0, std::min(result.pageNumber, estimatedTotalPages - 1));
|
||||
if (result.spineIndex == currentSpineIndex && totalPagesInCurrentSpine > 0) {
|
||||
result.totalPages = totalPagesInCurrentSpine;
|
||||
} else if (currentSpineIndex >= 0 && currentSpineIndex < spineCount && totalPagesInCurrentSpine > 0) {
|
||||
const size_t pc = (currentSpineIndex > 0) ? epub->getCumulativeSpineItemSize(currentSpineIndex - 1) : 0;
|
||||
const size_t cs = epub->getCumulativeSpineItemSize(currentSpineIndex) - pc;
|
||||
if (cs > 0)
|
||||
result.totalPages = std::max(
|
||||
1, static_cast<int>(totalPagesInCurrentSpine * static_cast<float>(spineSize) / static_cast<float>(cs)));
|
||||
}
|
||||
if (spineSize == 0 || result.totalPages == 0) return result;
|
||||
|
||||
float intra = 0.0f;
|
||||
if (xpathP > 0) {
|
||||
ParagraphStreamer s(xpathP, xpathChar);
|
||||
if (streamSpine(epub, result.spineIndex, s) && s.found()) {
|
||||
intra = s.progress();
|
||||
LOG_DBG("PM", "XPath p[%d]+%d -> %.1f%%", xpathP, xpathChar, intra * 100);
|
||||
}
|
||||
}
|
||||
if (intra <= 0.0f) {
|
||||
const size_t bytesIn = (targetBytes > prevCum) ? (targetBytes - prevCum) : 0;
|
||||
intra = std::max(0.0f, std::min(1.0f, static_cast<float>(bytesIn) / static_cast<float>(spineSize)));
|
||||
}
|
||||
|
||||
LOG_DBG("ProgressMapper", "KOReader -> CrossPoint: %.2f%% at %s -> spine=%d, page=%d", koPos.percentage * 100,
|
||||
koPos.xpath.c_str(), result.spineIndex, result.pageNumber);
|
||||
|
||||
result.pageNumber = std::max(0, std::min(static_cast<int>(intra * result.totalPages), result.totalPages - 1));
|
||||
LOG_DBG("PM", "<- KO: %.2f%% %s -> spine=%d page=%d/%d", koPos.percentage * 100, koPos.xpath.c_str(),
|
||||
result.spineIndex, result.pageNumber, result.totalPages);
|
||||
return result;
|
||||
}
|
||||
|
||||
std::string ProgressMapper::generateXPath(int spineIndex, int pageNumber, int totalPages) {
|
||||
// Use 0-based DocFragment indices for KOReader
|
||||
// Use a simple xpath pointing to the DocFragment - KOReader will use the percentage for fine positioning within it
|
||||
// Avoid specifying paragraph numbers as they may not exist in the target document
|
||||
return "/body/DocFragment[" + std::to_string(spineIndex) + "]/body";
|
||||
std::string ProgressMapper::generateXPath(const std::shared_ptr<Epub>& epub, int spineIndex, float intra) {
|
||||
const std::string base = "/body/DocFragment[" + std::to_string(spineIndex + 1) + "]/body";
|
||||
if (intra <= 0.0f) return base;
|
||||
|
||||
size_t spineSize = 0;
|
||||
const auto href = epub->getSpineItem(spineIndex).href;
|
||||
if (href.empty() || !epub->getItemSize(href, &spineSize) || spineSize == 0) return base;
|
||||
|
||||
ParagraphStreamer s(static_cast<size_t>(spineSize * std::min(intra, 1.0f)));
|
||||
if (!streamSpine(epub, spineIndex, s)) return base;
|
||||
|
||||
const int p = s.paragraphCount();
|
||||
return (p > 0) ? base + "/p[" + std::to_string(p) + "]" : base;
|
||||
}
|
||||
|
||||
@@ -8,9 +8,11 @@
|
||||
* CrossPoint position representation.
|
||||
*/
|
||||
struct CrossPointPosition {
|
||||
int spineIndex; // Current spine item (chapter) index
|
||||
int pageNumber; // Current page within the spine item
|
||||
int totalPages; // Total pages in the current spine item
|
||||
int spineIndex; // Current spine item (chapter) index
|
||||
int pageNumber; // Current page within the spine item
|
||||
int totalPages; // Total pages in the current spine item
|
||||
uint16_t paragraphIndex = 0; // 1-based synthetic paragraph index from XPath p[N]
|
||||
bool hasParagraphIndex = false; // True when paragraphIndex was resolved from XPath
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -59,9 +61,10 @@ class ProgressMapper {
|
||||
|
||||
private:
|
||||
/**
|
||||
* Generate XPath for KOReader compatibility.
|
||||
* Format: /body/DocFragment[spineIndex+1]/body
|
||||
* Since CrossPoint doesn't preserve HTML structure, we rely on percentage for positioning.
|
||||
* Generate a fallback XPath by streaming the spine item's XHTML and resolving
|
||||
* a paragraph/text position from intra-spine progress.
|
||||
* Produces a full ancestry path such as
|
||||
* /body/DocFragment[3]/body/p[42]/text().17.
|
||||
*/
|
||||
static std::string generateXPath(int spineIndex, int pageNumber, int totalPages);
|
||||
static std::string generateXPath(const std::shared_ptr<Epub>& epub, int spineIndex, float intraSpineProgress);
|
||||
};
|
||||
|
||||
Reference in New Issue
Block a user