diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp index 12c12c30..2621e7b3 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp @@ -7,6 +7,8 @@ #include #include +#include + #include "../../Epub.h" #include "../Page.h" #include "../converters/ImageDecoderFactory.h" @@ -76,6 +78,28 @@ bool isTableStructuralTag(const char* name) { return strcmp(name, "table") == 0 || strcmp(name, "tr") == 0 || strcmp(name, "td") == 0 || strcmp(name, "th") == 0; } +// Calibre sometimes injects empty

...

+// spacers inside running prose. Keep them as paragraph boundaries, but ignore +// their inner text payload (usually NBSP) to avoid no-break-space glue artifacts. +bool isZeroHeightSpacerParagraph(const char* name, const std::string& styleAttr) { + if (strcmp(name, "p") != 0 || styleAttr.empty()) { + return false; + } + + std::string normalized; + normalized.reserve(styleAttr.size()); + for (const char ch : styleAttr) { + if (!isWhitespace(ch)) { + normalized.push_back(static_cast(std::tolower(static_cast(ch)))); + } + } + + const bool hasZeroHeight = normalized.find("height:0") != std::string::npos; + const bool hasZeroMargin = normalized.find("margin:0") != std::string::npos; + const bool hasZeroBorder = normalized.find("border:0") != std::string::npos; + return hasZeroHeight && hasZeroMargin && hasZeroBorder; +} + // Update effective bold/italic/underline based on block style and inline style stack void ChapterHtmlSlimParser::updateEffectiveInlineStyle() { // Start with block-level styles @@ -572,6 +596,13 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* const auto userAlignmentBlockStyle = BlockStyle::fromCssStyle( cssStyle, emSize, static_cast(self->paragraphAlignment), self->viewportWidth); + // Block/header boundaries must flush any buffered trailing word first. + // Otherwise tags like ..."item?"

can carry the final word into the next paragraph. + if (self->partWordBufferIndex > 0 && ((matches(name, HEADER_TAGS, NUM_HEADER_TAGS)) || + (matches(name, BLOCK_TAGS, NUM_BLOCK_TAGS) && strcmp(name, "br") != 0))) { + self->flushPartWordBuffer(); + } + if (matches(name, HEADER_TAGS, NUM_HEADER_TAGS)) { self->currentCssStyle = cssStyle; auto headerBlockStyle = BlockStyle::fromCssStyle(cssStyle, emSize, CssTextAlign::Center, self->viewportWidth); @@ -583,6 +614,17 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* self->boldUntilDepth = std::min(self->boldUntilDepth, self->depth); self->updateEffectiveInlineStyle(); } else if (matches(name, BLOCK_TAGS, NUM_BLOCK_TAGS)) { + if (isZeroHeightSpacerParagraph(name, styleAttr)) { + // Preserve paragraph break semantics for this

, but skip its inner text payload. + self->currentCssStyle = cssStyle; + self->startNewTextBlock(userAlignmentBlockStyle); + self->updateEffectiveInlineStyle(); + + self->skipTextUntilDepth = self->depth; + self->depth += 1; + return; + } + if (strcmp(name, "br") == 0) { if (self->partWordBufferIndex > 0) { // flush word preceding
to currentTextBlock before calling startNewTextBlock @@ -721,6 +763,11 @@ void XMLCALL ChapterHtmlSlimParser::characterData(void* userData, const XML_Char return; } + // Ignore character data inside synthetic zero-height spacer

tags. + if (self->skipTextUntilDepth < self->depth) { + return; + } + // Collect footnote link display text (for the number label) // Skip whitespace and brackets to normalize noterefs like "[1]" → "1" if (self->insideFootnoteLink) { @@ -955,6 +1002,11 @@ void XMLCALL ChapterHtmlSlimParser::endElement(void* userData, const XML_Char* n self->skipUntilDepth = INT_MAX; } + // Leaving zero-height spacer paragraph text-skip scope + if (self->skipTextUntilDepth == self->depth) { + self->skipTextUntilDepth = INT_MAX; + } + if (self->tableDepth == 1 && (strcmp(name, "td") == 0 || strcmp(name, "th") == 0)) { self->nextWordContinues = false; } @@ -1009,9 +1061,11 @@ void XMLCALL ChapterHtmlSlimParser::endElement(void* userData, const XML_Char* n // Margins/padding are preserved so parent element spacing still accumulates correctly. if (self->currentTextBlock && self->currentTextBlock->isEmpty()) { auto style = self->currentTextBlock->getBlockStyle(); - // Keep alignment for synthetic empty
separator blocks so following inline - // text after
inside centered/right-aligned containers preserves alignment. - if (!style.fromBrElement) { + // Keep alignment only when closing the
separator itself so subsequent text + // within the same block container stays aligned. Reset alignment when closing + // other block tags (e.g. div/p) to avoid leaking centered/right alignment globally. + const bool preserveForBrClose = style.fromBrElement && strcmp(name, "br") == 0; + if (!preserveForBrClose) { style.textAlignDefined = false; style.alignment = (self->paragraphAlignment == static_cast(CssTextAlign::None)) ? CssTextAlign::Justify diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h index b690434f..3022dfc8 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h @@ -29,6 +29,7 @@ class ChapterHtmlSlimParser { std::function popupFn; // Popup callback int depth = 0; int skipUntilDepth = INT_MAX; + int skipTextUntilDepth = INT_MAX; // skip character data inside synthetic zero-height spacer

int boldUntilDepth = INT_MAX; int italicUntilDepth = INT_MAX; int underlineUntilDepth = INT_MAX;