From 6daf3fd1bba443d21a9ab64981a60e4ff3c93cc3 Mon Sep 17 00:00:00 2001 From: jpirnay Date: Sat, 28 Mar 2026 20:49:03 +0100 Subject: [PATCH] Add css cache --- lib/Epub/Epub/ParsedText.cpp | 52 ++++++++++------ .../Epub/parsers/ChapterHtmlSlimParser.cpp | 60 ++++++++++++++++--- lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h | 6 ++ 3 files changed, 92 insertions(+), 26 deletions(-) diff --git a/lib/Epub/Epub/ParsedText.cpp b/lib/Epub/Epub/ParsedText.cpp index 043020d7..5219c57e 100644 --- a/lib/Epub/Epub/ParsedText.cpp +++ b/lib/Epub/Epub/ParsedText.cpp @@ -165,6 +165,20 @@ std::vector ParsedText::computeLineBreaks(const GfxRenderer& renderer, c const size_t totalWordCount = words.size(); + // Pre-compute inter-word gaps once so the O(n²) DP inner loop avoids repeated + // codepoint scanning and renderer calls for every (i,j) pair. + // interWordGaps[j] = the spacing between words[j-1] and words[j] (0 for j==0). + std::vector interWordGaps(totalWordCount, 0); + for (size_t j = 1; j < totalWordCount; ++j) { + if (!continuesVec[j]) { + interWordGaps[j] = + renderer.getSpaceAdvance(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]); + } else { + interWordGaps[j] = + renderer.getKerning(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]); + } + } + // DP table to store the minimum badness (cost) of lines starting at index i std::vector dp(totalWordCount); // 'ans[i]' stores the index 'j' of the *last word* in the optimal line starting at 'i' @@ -182,15 +196,7 @@ std::vector ParsedText::computeLineBreaks(const GfxRenderer& renderer, c const int effectivePageWidth = i == 0 ? pageWidth - firstLineIndent : pageWidth; for (size_t j = i; j < totalWordCount; ++j) { - // Add space before word j, unless it's the first word on the line or a continuation - int gap = 0; - if (j > static_cast(i) && !continuesVec[j]) { - gap = - renderer.getSpaceAdvance(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]); - } else if (j > static_cast(i) && continuesVec[j]) { - // Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation) - gap = renderer.getKerning(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]); - } + const int gap = (j > static_cast(i)) ? interWordGaps[j] : 0; currlen += wordWidths[j] + gap; if (currlen > effectivePageWidth) { @@ -284,6 +290,21 @@ std::vector ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& r ? blockStyle.textIndent : 0; + // Pre-compute inter-word gaps to avoid repeated codepoint scanning and renderer + // calls in the inner loop. When hyphenateWordAtIndex inserts a new word, we insert + // a placeholder gap (0) at that position to keep the vector in sync; the remainder + // is always the first word on the next line so its spacing is never used. + std::vector interWordGaps(wordWidths.size(), 0); + for (size_t j = 1; j < wordWidths.size(); ++j) { + if (!continuesVec[j]) { + interWordGaps[j] = + renderer.getSpaceAdvance(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]); + } else { + interWordGaps[j] = + renderer.getKerning(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]); + } + } + std::vector lineBreakIndices; size_t currentIndex = 0; bool isFirstLine = true; @@ -298,15 +319,7 @@ std::vector ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& r // Consume as many words as possible for current line, splitting when prefixes fit while (currentIndex < wordWidths.size()) { const bool isFirstWord = currentIndex == lineStart; - int spacing = 0; - if (!isFirstWord && !continuesVec[currentIndex]) { - spacing = renderer.getSpaceAdvance(fontId, lastCodepoint(words[currentIndex - 1]), - firstCodepoint(words[currentIndex]), wordStyles[currentIndex - 1]); - } else if (!isFirstWord && continuesVec[currentIndex]) { - // Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation) - spacing = renderer.getKerning(fontId, lastCodepoint(words[currentIndex - 1]), - firstCodepoint(words[currentIndex]), wordStyles[currentIndex - 1]); - } + const int spacing = isFirstWord ? 0 : interWordGaps[currentIndex]; const int candidateWidth = spacing + wordWidths[currentIndex]; // Word fits on current line @@ -322,6 +335,9 @@ std::vector ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& r if (availableWidth > 0 && hyphenateWordAtIndex(currentIndex, availableWidth, renderer, fontId, wordWidths, allowFallbackBreaks)) { + // Keep interWordGaps in sync: insert placeholder for the new remainder word. + // The remainder is always the first word on the next line so this slot is never read. + interWordGaps.insert(interWordGaps.begin() + currentIndex + 1, 0); // Prefix now fits; append it to this line and move to next line lineWidth += spacing + wordWidths[currentIndex]; ++currentIndex; diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp index 368a4c60..4e097e1d 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp @@ -186,10 +186,20 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* // before tag-specific branches emit any content or metadata. CssStyle cssStyle; if (self->cssParser) { - cssStyle = self->cssParser->resolveStyle(name, classAttr); + { + std::string cacheKey(name); + cacheKey += '|'; + cacheKey += classAttr; + auto it = self->cssStyleCache_.find(cacheKey); + if (it == self->cssStyleCache_.end()) + it = self->cssStyleCache_.emplace(cacheKey, self->cssParser->resolveStyle(name, classAttr)).first; + cssStyle = it->second; + } if (!styleAttr.empty()) { - CssStyle inlineStyle = CssParser::parseInlineStyle(styleAttr); - cssStyle.applyOver(inlineStyle); + auto it = self->inlineStyleCache_.find(styleAttr); + if (it == self->inlineStyleCache_.end()) + it = self->inlineStyleCache_.emplace(styleAttr, CssParser::parseInlineStyle(styleAttr)).first; + cssStyle.applyOver(it->second); } } @@ -284,9 +294,17 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* // Skip image if CSS display:none if (self->cssParser) { - CssStyle imgDisplayStyle = self->cssParser->resolveStyle("img", classAttr); + std::string imgCacheKey("img|"); + imgCacheKey += classAttr; + auto imgIt = self->cssStyleCache_.find(imgCacheKey); + if (imgIt == self->cssStyleCache_.end()) + imgIt = self->cssStyleCache_.emplace(imgCacheKey, self->cssParser->resolveStyle("img", classAttr)).first; + CssStyle imgDisplayStyle = imgIt->second; if (!styleAttr.empty()) { - imgDisplayStyle.applyOver(CssParser::parseInlineStyle(styleAttr)); + auto it = self->inlineStyleCache_.find(styleAttr); + if (it == self->inlineStyleCache_.end()) + it = self->inlineStyleCache_.emplace(styleAttr, CssParser::parseInlineStyle(styleAttr)).first; + imgDisplayStyle.applyOver(it->second); } if (imgDisplayStyle.hasDisplay() && imgDisplayStyle.display == CssDisplay::None) { self->skipUntilDepth = self->depth; @@ -331,10 +349,19 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* int displayWidth = 0; int displayHeight = 0; const float emSize = static_cast(self->renderer.getFontAscenderSize(self->fontId)); - CssStyle imgStyle = self->cssParser ? self->cssParser->resolveStyle("img", classAttr) : CssStyle{}; + std::string imgCacheKey("img|"); + imgCacheKey += classAttr; + auto imgStyleIt = self->cssParser ? self->cssStyleCache_.find(imgCacheKey) : self->cssStyleCache_.end(); + if (self->cssParser && imgStyleIt == self->cssStyleCache_.end()) + imgStyleIt = + self->cssStyleCache_.emplace(imgCacheKey, self->cssParser->resolveStyle("img", classAttr)).first; + CssStyle imgStyle = self->cssParser ? imgStyleIt->second : CssStyle{}; // Merge inline style (e.g. style="height: 2em") so it overrides stylesheet rules if (!styleAttr.empty()) { - imgStyle.applyOver(CssParser::parseInlineStyle(styleAttr)); + auto it = self->inlineStyleCache_.find(styleAttr); + if (it == self->inlineStyleCache_.end()) + it = self->inlineStyleCache_.emplace(styleAttr, CssParser::parseInlineStyle(styleAttr)).first; + imgStyle.applyOver(it->second); } const bool hasCssHeight = imgStyle.hasImageHeight(); const bool hasCssWidth = imgStyle.hasImageWidth(); @@ -708,6 +735,22 @@ void XMLCALL ChapterHtmlSlimParser::characterData(void* userData, const XML_Char } for (int i = 0; i < len; i++) { + const unsigned char c = static_cast(s[i]); + + // Fast path for plain ASCII word characters (> 0x20 and < 0x80). + // This covers the vast majority of characters in Latin-script text. + // All multi-byte UTF-8 sequences start with a byte >= 0x80, so this + // path is safe to take without any further multi-byte checks. + if (c > 0x20 && c < 0x80) { + if (self->partWordBufferIndex >= MAX_WORD_SIZE) { + // Buffer is full — flush before appending. Pure ASCII means no + // partial multi-byte sequence can be at the boundary. + self->flushPartWordBuffer(); + } + self->partWordBuffer[self->partWordBufferIndex++] = s[i]; + continue; + } + if (isWhitespace(s[i])) { // Currently looking at whitespace, if there's anything in the partWordBuffer, flush it if (self->partWordBufferIndex > 0) { @@ -1047,7 +1090,8 @@ bool ChapterHtmlSlimParser::parseAndBuildPages() { return false; } } while (!done); - LOG_DBG("EHP", "Time to parse and build pages: %lu ms", millis() - chapterStartTime); + const uint32_t totalTimeMs = millis() - chapterStartTime; + LOG_DBG("EHP", "Time to parse and build pages: %lu ms", totalTimeMs); XML_StopParser(parser, XML_FALSE); // Stop any pending processing XML_SetElementHandler(parser, nullptr, nullptr); // Clear callbacks diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h index 1cc0ea39..75413a31 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h @@ -6,6 +6,7 @@ #include #include #include +#include #include #include "../FootnoteEntry.h" @@ -84,6 +85,11 @@ class ChapterHtmlSlimParser { std::vector> pendingFootnotes; // int wordsExtractedInBlock = 0; + // Per-chapter caches: resolveStyle and parseInlineStyle are called for every HTML element; + // caching by (tag|classAttr) and styleAttr avoids repeated string operations and hash lookups. + std::unordered_map cssStyleCache_; + std::unordered_map inlineStyleCache_; + void updateEffectiveInlineStyle(); void startNewTextBlock(const BlockStyle& blockStyle); void flushPartWordBuffer();