feat: add RTL support in epub and txt readers (#1700)

Co-authored-by: Zach Nelson <zach@zdnelson.com>
This commit is contained in:
Uri Tauber
2026-05-29 03:25:17 -04:00
committed by GitHub
co-authored by Zach Nelson
parent cc2079a578
commit f5bc554ae7
27 changed files with 1621 additions and 119 deletions
@@ -43,6 +43,7 @@ INTERVAL_PRESETS = {
(0x1E00, 0x1EFF), (0x2000, 0x206F), (0xFB00, 0xFB06)], (0x1E00, 0x1EFF), (0x2000, 0x206F), (0xFB00, 0xFB06)],
"greek": [(0x0370, 0x03FF), (0x1F00, 0x1FFF)], "greek": [(0x0370, 0x03FF), (0x1F00, 0x1FFF)],
"cyrillic": [(0x0400, 0x04FF), (0x0500, 0x052F)], "cyrillic": [(0x0400, 0x04FF), (0x0500, 0x052F)],
"hebrew": [(0x0590, 0x05FF), (0xFB1D, 0xFB4F)],
"georgian": [(0x10A0, 0x10FF), (0x2D00, 0x2D2F)], "georgian": [(0x10A0, 0x10FF), (0x2D00, 0x2D2F)],
"armenian": [(0x0530, 0x058F)], "armenian": [(0x0530, 0x058F)],
"ethiopic": [(0x1200, 0x137F), (0x1380, 0x139F), (0x2D80, 0x2DDF)], "ethiopic": [(0x1200, 0x137F), (0x1380, 0x139F), (0x2D80, 0x2DDF)],
+275 -83
View File
@@ -1,5 +1,6 @@
#include "ParsedText.h" #include "ParsedText.h"
#include <BidiUtils.h>
#include <GfxRenderer.h> #include <GfxRenderer.h>
#include <Utf8.h> #include <Utf8.h>
@@ -18,6 +19,19 @@ namespace {
// Soft hyphen byte pattern used throughout EPUBs (UTF-8 for U+00AD). // Soft hyphen byte pattern used throughout EPUBs (UTF-8 for U+00AD).
constexpr char SOFT_HYPHEN_UTF8[] = "\xC2\xAD"; constexpr char SOFT_HYPHEN_UTF8[] = "\xC2\xAD";
constexpr size_t SOFT_HYPHEN_BYTES = 2; constexpr size_t SOFT_HYPHEN_BYTES = 2;
// Paragraph-level direction: scan the first N words to find base direction.
constexpr size_t RTL_PARAGRAPH_PROBE_WORDS = 3;
// Per-word: scan enough chars to see through leading neutrals (quotes, numbers)
// before giving up. 64 is a hedge for pathological cases like long numeric tokens.
constexpr int RTL_PER_WORD_PROBE_DEPTH = 64;
// Byte-level pre-check: Hebrew UTF-8 lead bytes 0xD6-0xD7, Arabic/Syriac 0xD8-0xDB.
bool mayContainRtlBytes(const char* str) {
for (const auto* p = reinterpret_cast<const unsigned char*>(str); *p; ++p) {
if (*p >= 0xD6 && *p <= 0xDB) return true;
}
return false;
}
// Returns the first rendered codepoint of a word (skipping leading soft hyphens). // Returns the first rendered codepoint of a word (skipping leading soft hyphens).
uint32_t firstCodepoint(const std::string& word) { uint32_t firstCodepoint(const std::string& word) {
@@ -115,6 +129,8 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
if (underline) { if (underline) {
baseStyle = static_cast<EpdFontFamily::Style>(baseStyle | EpdFontFamily::UNDERLINE); baseStyle = static_cast<EpdFontFamily::Style>(baseStyle | EpdFontFamily::UNDERLINE);
} }
const bool wordStartsRtl = !hasRtlWord && mayContainRtlBytes(word.c_str()) &&
BidiUtils::startsWithRtl(word.c_str(), RTL_PER_WORD_PROBE_DEPTH);
// Already-bold text should stay fully bold; focus splitting would make its suffix regular later. // Already-bold text should stay fully bold; focus splitting would make its suffix regular later.
if (!this->focusReadingEnabled || (baseStyle & EpdFontFamily::BOLD) != 0) { if (!this->focusReadingEnabled || (baseStyle & EpdFontFamily::BOLD) != 0) {
@@ -122,6 +138,9 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
wordStyles.push_back(baseStyle); wordStyles.push_back(baseStyle);
wordContinues.push_back(attachToPrevious); wordContinues.push_back(attachToPrevious);
wordIsFocusSuffix.push_back(false); wordIsFocusSuffix.push_back(false);
if (wordStartsRtl) {
hasRtlWord = true;
}
return; return;
} }
@@ -237,6 +256,17 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
size_t segmentLen = end - segmentStart; size_t segmentLen = end - segmentStart;
std::string_view segment(reinterpret_cast<const char*>(segmentStart), segmentLen); std::string_view segment(reinterpret_cast<const char*>(segmentStart), segmentLen);
processSegment(segment, inWordSegment, isFirstSegment ? attachToPrevious : true); processSegment(segment, inWordSegment, isFirstSegment ? attachToPrevious : true);
if (wordStartsRtl) {
hasRtlWord = true;
}
}
int ParsedText::resolveFirstLineIndent(const bool isFirstLine) const {
if (isFirstLine && blockStyle.textIndentDefined && (blockStyle.textIndent < 0 || !extraParagraphSpacing) &&
isNaturalAlign) {
return blockStyle.textIndent;
}
return 0;
} }
// Consumes data to minimize memory usage // Consumes data to minimize memory usage
void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fontId, const uint16_t viewportWidth, void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fontId, const uint16_t viewportWidth,
@@ -246,6 +276,23 @@ void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fo
return; return;
} }
// Per-paragraph RTL auto-detection: only when CSS/HTML didn't explicitly set direction.
// Explicit dir="ltr" must be respected and not overridden by content heuristic.
if (!blockStyle.directionDefined && hasRtlWord) {
// Check the first few words for RTL letter codepoints (no heap allocation).
const size_t wordsToScan = std::min(words.size(), RTL_PARAGRAPH_PROBE_WORDS);
for (size_t i = 0; i < wordsToScan; ++i) {
if (BidiUtils::startsWithRtl(words[i].c_str(), BidiUtils::RTL_PARAGRAPH_PROBE_DEPTH)) {
blockStyle.isRtl = true;
break;
}
}
}
isNaturalAlign =
blockStyle.alignment == CssTextAlign::Justify ||
(blockStyle.isRtl ? blockStyle.alignment == CssTextAlign::Right : blockStyle.alignment == CssTextAlign::Left);
// Apply fixed transforms before any per-line layout work. // Apply fixed transforms before any per-line layout work.
applyParagraphIndent(); applyParagraphIndent();
@@ -309,15 +356,7 @@ std::vector<size_t> ParsedText::computeLineBreaks(const GfxRenderer& renderer, c
return {}; return {};
} }
// Calculate first line indent (only for left/justified text). const int firstLineIndent = resolveFirstLineIndent(true);
// Positive text-indent (paragraph indent) is suppressed when extraParagraphSpacing is on.
// Negative text-indent (hanging indent, e.g. margin-left:3em; text-indent:-1em) always applies —
// it is structural (positions the bullet/marker), not decorative.
const int firstLineIndent =
blockStyle.textIndentDefined && (blockStyle.textIndent < 0 || !extraParagraphSpacing) &&
(blockStyle.alignment == CssTextAlign::Justify || blockStyle.alignment == CssTextAlign::Left)
? blockStyle.textIndent
: 0;
// Ensure any word that would overflow even as the first entry on a line is split using fallback hyphenation. // Ensure any word that would overflow even as the first entry on a line is split using fallback hyphenation.
for (size_t i = 0; i < wordWidths.size(); ++i) { for (size_t i = 0; i < wordWidths.size(); ++i) {
@@ -431,7 +470,7 @@ void ParsedText::applyParagraphIndent() {
if (blockStyle.textIndentDefined) { if (blockStyle.textIndentDefined) {
// CSS text-indent is explicitly set (even if 0) - don't use fallback EmSpace // CSS text-indent is explicitly set (even if 0) - don't use fallback EmSpace
// The actual indent positioning is handled in extractLine() // The actual indent positioning is handled in extractLine()
} else if (blockStyle.alignment == CssTextAlign::Justify || blockStyle.alignment == CssTextAlign::Left) { } else if (isNaturalAlign) {
// No CSS text-indent defined - use EmSpace fallback for visual indent // No CSS text-indent defined - use EmSpace fallback for visual indent
words.front().insert(0, "\xe2\x80\x83"); words.front().insert(0, "\xe2\x80\x83");
} }
@@ -441,15 +480,7 @@ void ParsedText::applyParagraphIndent() {
std::vector<size_t> ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& renderer, const int fontId, std::vector<size_t> ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& renderer, const int fontId,
const int pageWidth, std::vector<uint16_t>& wordWidths, const int pageWidth, std::vector<uint16_t>& wordWidths,
std::vector<bool>& continuesVec) { std::vector<bool>& continuesVec) {
// Calculate first line indent (only for left/justified text). const int firstLineIndent = resolveFirstLineIndent(true);
// Positive text-indent (paragraph indent) is suppressed when extraParagraphSpacing is on.
// Negative text-indent (hanging indent, e.g. margin-left:3em; text-indent:-1em) always applies —
// it is structural (positions the bullet/marker), not decorative.
const int firstLineIndent =
blockStyle.textIndentDefined && (blockStyle.textIndent < 0 || !extraParagraphSpacing) &&
(blockStyle.alignment == CssTextAlign::Justify || blockStyle.alignment == CssTextAlign::Left)
? blockStyle.textIndent
: 0;
std::vector<size_t> lineBreakIndices; std::vector<size_t> lineBreakIndices;
size_t currentIndex = 0; size_t currentIndex = 0;
@@ -612,16 +643,22 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
const size_t lastBreakAt = breakIndex > 0 ? lineBreakIndices[breakIndex - 1] : 0; const size_t lastBreakAt = breakIndex > 0 ? lineBreakIndices[breakIndex - 1] : 0;
const size_t lineWordCount = lineBreak - lastBreakAt; const size_t lineWordCount = lineBreak - lastBreakAt;
// Calculate first line indent (only for left/justified text). const int firstLineIndent = resolveFirstLineIndent(breakIndex == 0);
// Positive text-indent (paragraph indent) is suppressed when extraParagraphSpacing is on.
// Negative text-indent (hanging indent, e.g. margin-left:3em; text-indent:-1em) always applies — // Build line data by moving from the original vectors using index range
// it is structural (positions the bullet/marker), not decorative. std::vector<std::string> lineWords;
const bool isFirstLine = breakIndex == 0; lineWords.reserve(lineWordCount);
const int firstLineIndent = std::vector<EpdFontFamily::Style> lineWordStyles;
isFirstLine && blockStyle.textIndentDefined && (blockStyle.textIndent < 0 || !extraParagraphSpacing) && lineWordStyles.reserve(lineWordCount);
(blockStyle.alignment == CssTextAlign::Justify || blockStyle.alignment == CssTextAlign::Left)
? blockStyle.textIndent for (size_t i = 0; i < lineWordCount; ++i) {
: 0; std::string word = std::move(words[lastBreakAt + i]);
if (containsSoftHyphen(word)) {
stripSoftHyphensInPlace(word);
}
lineWords.push_back(std::move(word));
lineWordStyles.push_back(wordStyles[lastBreakAt + i]);
}
// Calculate total word width for this line, count actual word gaps, // Calculate total word width for this line, count actual word gaps,
// and accumulate total natural gap widths (including space kerning adjustments). // and accumulate total natural gap widths (including space kerning adjustments).
@@ -634,19 +671,17 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
// Count gaps: each word after the first creates a gap, unless it's a continuation // Count gaps: each word after the first creates a gap, unless it's a continuation
if (wordIdx > 0 && !continuesVec[lastBreakAt + wordIdx]) { if (wordIdx > 0 && !continuesVec[lastBreakAt + wordIdx]) {
actualGapCount++; actualGapCount++;
totalNaturalGaps += totalNaturalGaps += renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx - 1]),
renderer.getSpaceAdvance(fontId, lastCodepoint(words[lastBreakAt + wordIdx - 1]), firstCodepoint(lineWords[wordIdx]), lineWordStyles[wordIdx - 1]);
firstCodepoint(words[lastBreakAt + wordIdx]), wordStyles[lastBreakAt + wordIdx - 1]);
} else if (wordIdx > 0 && continuesVec[lastBreakAt + wordIdx]) { } else if (wordIdx > 0 && continuesVec[lastBreakAt + wordIdx]) {
// Non-breaking space tokens (" " with continues=true) are visible, stretchable spaces — // Non-breaking space tokens (" " with continues=true) are visible, stretchable spaces —
// count them as justifiable gaps so justifyExtra is distributed to them too. // count them as justifiable gaps so justifyExtra is distributed to them too.
if (words[lastBreakAt + wordIdx] == " ") { if (lineWords[wordIdx] == " ") {
actualGapCount++; actualGapCount++;
} }
// Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation) // Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation)
totalNaturalGaps += totalNaturalGaps += renderer.getKerning(fontId, lastCodepoint(lineWords[wordIdx - 1]),
renderer.getKerning(fontId, lastCodepoint(words[lastBreakAt + wordIdx - 1]), firstCodepoint(lineWords[wordIdx]), lineWordStyles[wordIdx - 1]);
firstCodepoint(words[lastBreakAt + wordIdx]), wordStyles[lastBreakAt + wordIdx - 1]);
} }
} }
@@ -654,73 +689,229 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
const int effectivePageWidth = pageWidth - firstLineIndent; const int effectivePageWidth = pageWidth - firstLineIndent;
const bool isLastLine = breakIndex == lineBreakIndices.size() - 1; const bool isLastLine = breakIndex == lineBreakIndices.size() - 1;
// For RTL, implicit/default Left alignment becomes Right alignment.
// Explicit text-align:left must remain left for CSS correctness.
const CssTextAlign effectiveAlignment =
(blockStyle.isRtl && !blockStyle.textAlignDefined && blockStyle.alignment == CssTextAlign::Left)
? CssTextAlign::Right
: blockStyle.alignment;
// For justified text, compute per-gap extra to distribute remaining space evenly // For justified text, compute per-gap extra to distribute remaining space evenly
const int spareSpace = effectivePageWidth - lineWordWidthSum - totalNaturalGaps; const int spareSpace = effectivePageWidth - lineWordWidthSum - totalNaturalGaps;
const int justifyExtra = (blockStyle.alignment == CssTextAlign::Justify && !isLastLine && actualGapCount >= 1) const int justifyExtra = (effectiveAlignment == CssTextAlign::Justify && !isLastLine && actualGapCount >= 1)
? spareSpace / static_cast<int>(actualGapCount) ? spareSpace / static_cast<int>(actualGapCount)
: 0; : 0;
// Calculate initial x position (first line starts at indent for left/justified text; // BiDi processing: reorder words with UAX#9 in full-line context.
// may be negative for hanging indents, e.g. margin-left:3em; text-indent:-1em). visualOrderScratch.clear();
auto xpos = static_cast<int16_t>(firstLineIndent); visualOrderScratch.reserve(lineWordCount);
if (blockStyle.alignment == CssTextAlign::Right) { // Skip expensive visual-order resolution for pure LTR paragraphs that have no RTL words.
xpos = effectivePageWidth - lineWordWidthSum - totalNaturalGaps; const bool shouldResolveVisualOrder = blockStyle.isRtl || hasRtlWord;
} else if (blockStyle.alignment == CssTextAlign::Center) { const bool willReorder =
xpos = (effectivePageWidth - lineWordWidthSum - totalNaturalGaps) / 2; shouldResolveVisualOrder && BidiUtils::computeVisualWordOrder(lineWords, blockStyle.isRtl, visualOrderScratch);
}
// Pre-calculate X positions for words
// Continuation words attach to the previous word with no space before them
std::vector<int16_t> lineXPos; std::vector<int16_t> lineXPos;
lineXPos.reserve(lineWordCount); lineXPos.reserve(lineWordCount);
for (size_t wordIdx = 0; wordIdx < lineWordCount; wordIdx++) { if (willReorder) {
lineXPos.push_back(xpos); reorderedWordsScratch.clear();
reorderedStylesScratch.clear();
reorderedWidthsScratch.clear();
reorderedContinuesScratch.clear();
reorderedFocusSuffixScratch.clear();
reorderedWordsScratch.reserve(visualOrderScratch.size());
reorderedStylesScratch.reserve(visualOrderScratch.size());
reorderedWidthsScratch.reserve(visualOrderScratch.size());
reorderedContinuesScratch.reserve(visualOrderScratch.size());
reorderedFocusSuffixScratch.reserve(visualOrderScratch.size());
const bool nextIsContinuation = wordIdx + 1 < lineWordCount && continuesVec[lastBreakAt + wordIdx + 1]; for (size_t i = 0; i < visualOrderScratch.size(); ++i) {
if (nextIsContinuation) { const uint16_t src = visualOrderScratch[i];
int advance = wordWidths[lastBreakAt + wordIdx]; reorderedWordsScratch.push_back(std::move(lineWords[src]));
// Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation) reorderedStylesScratch.push_back(lineWordStyles[src]);
advance += reorderedWidthsScratch.push_back(wordWidths[lastBreakAt + src]);
renderer.getKerning(fontId, lastCodepoint(words[lastBreakAt + wordIdx]), reorderedFocusSuffixScratch.push_back(wordIsFocusSuffix[lastBreakAt + src]);
firstCodepoint(words[lastBreakAt + wordIdx + 1]), wordStyles[lastBreakAt + wordIdx]);
// Non-breaking space tokens are stretchable — expand them during justification like normal spaces. // Continuation means "no break/gap between two adjacent logical tokens".
if (words[lastBreakAt + wordIdx] == " " && continuesVec[lastBreakAt + wordIdx] && // After visual reordering (common in RTL), an adjacent logical pair can appear
blockStyle.alignment == CssTextAlign::Justify && !isLastLine) { // as either (prev -> curr) or (curr -> prev) in visual order; preserve both.
advance += justifyExtra; bool continues = false;
if (i > 0) {
const size_t prevSrc = visualOrderScratch[i - 1];
const size_t currSrc = src;
const bool forwardAdjacent = currSrc == prevSrc + 1;
const bool reverseAdjacent = prevSrc == currSrc + 1;
if (forwardAdjacent && continuesVec[lastBreakAt + currSrc]) {
continues = true;
} else if (reverseAdjacent && continuesVec[lastBreakAt + prevSrc]) {
continues = true;
}
}
reorderedContinuesScratch.push_back(continues);
}
int reorderedWordWidthSum = 0;
size_t reorderedGapCount = 0;
int reorderedNaturalGaps = 0;
for (size_t wordIdx = 0; wordIdx < reorderedWidthsScratch.size(); wordIdx++) {
reorderedWordWidthSum += reorderedWidthsScratch[wordIdx];
if (wordIdx > 0 && !reorderedContinuesScratch[wordIdx]) {
reorderedGapCount++;
reorderedNaturalGaps += renderer.getSpaceAdvance(fontId, lastCodepoint(reorderedWordsScratch[wordIdx - 1]),
firstCodepoint(reorderedWordsScratch[wordIdx]),
reorderedStylesScratch[wordIdx - 1]);
} else if (wordIdx > 0 && reorderedContinuesScratch[wordIdx]) {
if (reorderedWordsScratch[wordIdx] == " ") {
reorderedGapCount++;
}
reorderedNaturalGaps +=
renderer.getKerning(fontId, lastCodepoint(reorderedWordsScratch[wordIdx - 1]),
firstCodepoint(reorderedWordsScratch[wordIdx]), reorderedStylesScratch[wordIdx - 1]);
}
}
const int reorderedSpare = effectivePageWidth - reorderedWordWidthSum - reorderedNaturalGaps;
const int reorderedJustifyExtra =
(effectiveAlignment == CssTextAlign::Justify && !isLastLine && reorderedGapCount >= 1)
? reorderedSpare / static_cast<int>(reorderedGapCount)
: 0;
const int justifyContribution = (effectiveAlignment == CssTextAlign::Justify && !isLastLine)
? reorderedJustifyExtra * static_cast<int>(reorderedGapCount)
: 0;
const int contentWidth = reorderedWordWidthSum + reorderedNaturalGaps + justifyContribution;
int xpos = 0;
if (blockStyle.isRtl) {
if (effectiveAlignment == CssTextAlign::Right || effectiveAlignment == CssTextAlign::Justify) {
xpos = effectivePageWidth - contentWidth;
} else if (effectiveAlignment == CssTextAlign::Center) {
xpos = (effectivePageWidth - contentWidth) / 2;
} }
xpos += advance;
} else { } else {
int gap = 0; xpos = firstLineIndent;
if (wordIdx + 1 < lineWordCount) { if (effectiveAlignment == CssTextAlign::Right) {
gap = renderer.getSpaceAdvance(fontId, lastCodepoint(words[lastBreakAt + wordIdx]), xpos = effectivePageWidth - contentWidth;
firstCodepoint(words[lastBreakAt + wordIdx + 1]), } else if (effectiveAlignment == CssTextAlign::Center) {
wordStyles[lastBreakAt + wordIdx]); xpos = (effectivePageWidth - contentWidth) / 2;
} }
if (blockStyle.alignment == CssTextAlign::Justify && !isLastLine) { }
gap += justifyExtra;
for (size_t wordIdx = 0; wordIdx < reorderedWidthsScratch.size(); wordIdx++) {
lineXPos.push_back(static_cast<int16_t>(xpos < 0 ? 0 : xpos));
xpos += reorderedWidthsScratch[wordIdx];
const bool nextIsContinuation =
wordIdx + 1 < reorderedWidthsScratch.size() && reorderedContinuesScratch[wordIdx + 1];
if (nextIsContinuation) {
int advance =
renderer.getKerning(fontId, lastCodepoint(reorderedWordsScratch[wordIdx]),
firstCodepoint(reorderedWordsScratch[wordIdx + 1]), reorderedStylesScratch[wordIdx]);
if (reorderedWordsScratch[wordIdx] == " " && reorderedContinuesScratch[wordIdx] &&
effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
advance += reorderedJustifyExtra;
}
xpos += advance;
} else if (wordIdx + 1 < reorderedWidthsScratch.size()) {
int gap = renderer.getSpaceAdvance(fontId, lastCodepoint(reorderedWordsScratch[wordIdx]),
firstCodepoint(reorderedWordsScratch[wordIdx + 1]),
reorderedStylesScratch[wordIdx]);
if (effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
gap += reorderedJustifyExtra;
}
xpos += gap;
}
}
lineWords.swap(reorderedWordsScratch);
lineWordStyles.swap(reorderedStylesScratch);
} else {
// Standard LTR/RTL positioning loop when no visual reordering is needed
if (blockStyle.isRtl) {
// RTL: position words from right to left
auto xpos = static_cast<int>(effectivePageWidth);
if (effectiveAlignment == CssTextAlign::Left) {
// Explicit left alignment in RTL context
xpos = lineWordWidthSum + totalNaturalGaps;
} else if (effectiveAlignment == CssTextAlign::Center) {
xpos = (effectivePageWidth + lineWordWidthSum + totalNaturalGaps) / 2;
}
// For Right and Justify, start from right edge (xpos = effectivePageWidth)
for (size_t wordIdx = 0; wordIdx < lineWordCount; wordIdx++) {
xpos -= wordWidths[lastBreakAt + wordIdx];
lineXPos.push_back(static_cast<int16_t>(xpos < 0 ? 0 : xpos));
const bool nextIsContinuation = wordIdx + 1 < lineWordCount && continuesVec[lastBreakAt + wordIdx + 1];
if (nextIsContinuation) {
// Cross-boundary kerning for continuation words
int advance = renderer.getKerning(fontId, lastCodepoint(lineWords[wordIdx]),
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
if (lineWords[wordIdx] == " " && continuesVec[lastBreakAt + wordIdx] &&
effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
advance += justifyExtra;
}
xpos -= advance;
} else {
int gap = 0;
if (wordIdx + 1 < lineWordCount) {
gap = renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx]),
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
}
if (effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
gap += justifyExtra;
}
xpos -= gap;
}
}
} else {
// LTR: position words from left to right
auto xpos = static_cast<int16_t>(firstLineIndent);
if (effectiveAlignment == CssTextAlign::Right) {
xpos = effectivePageWidth - lineWordWidthSum - totalNaturalGaps;
} else if (effectiveAlignment == CssTextAlign::Center) {
xpos = (effectivePageWidth - lineWordWidthSum - totalNaturalGaps) / 2;
}
for (size_t wordIdx = 0; wordIdx < lineWordCount; wordIdx++) {
lineXPos.push_back(static_cast<int16_t>(xpos < 0 ? 0 : xpos));
const bool nextIsContinuation = wordIdx + 1 < lineWordCount && continuesVec[lastBreakAt + wordIdx + 1];
if (nextIsContinuation) {
int advance = wordWidths[lastBreakAt + wordIdx];
advance += renderer.getKerning(fontId, lastCodepoint(lineWords[wordIdx]),
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
if (lineWords[wordIdx] == " " && continuesVec[lastBreakAt + wordIdx] &&
effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
advance += justifyExtra;
}
xpos += advance;
} else {
int gap = 0;
if (wordIdx + 1 < lineWordCount) {
gap = renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx]),
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
}
if (effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
gap += justifyExtra;
}
xpos += wordWidths[lastBreakAt + wordIdx] + gap;
}
} }
xpos += wordWidths[lastBreakAt + wordIdx] + gap;
} }
} }
// Build line data by moving from the original vectors using index range const auto isFocusSuffixAt = [&](const size_t idx) {
std::vector<std::string> lineWords(std::make_move_iterator(words.begin() + lastBreakAt), return willReorder ? reorderedFocusSuffixScratch[idx] : wordIsFocusSuffix[lastBreakAt + idx];
std::make_move_iterator(words.begin() + lineBreak)); };
std::vector<EpdFontFamily::Style> lineWordStyles(wordStyles.begin() + lastBreakAt, wordStyles.begin() + lineBreak);
for (auto& word : lineWords) {
if (containsSoftHyphen(word)) {
stripSoftHyphensInPlace(word);
}
}
// Fast path: when no word on this line was split for focus reading, skip the merge work // Fast path: when no word on this line was split for focus reading, skip the merge work
// entirely and pass empty boundary/suffixX vectors. TextBlock pays zero per-word RAM cost // entirely and pass empty boundary/suffixX vectors. TextBlock pays zero per-word RAM cost
// for these annotations when the vectors are empty. // for these annotations when the vectors are empty.
bool lineHasFocusSplit = false; bool lineHasFocusSplit = false;
for (size_t i = 0; i < lineWordCount; i++) { for (size_t i = 0; i < lineWordCount; i++) {
if (wordIsFocusSuffix[lastBreakAt + i]) { if (isFocusSuffixAt(i)) {
lineHasFocusSplit = true; lineHasFocusSplit = true;
break; break;
} }
@@ -747,17 +938,18 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
outSuffixX.reserve(lineWordCount); outSuffixX.reserve(lineWordCount);
for (size_t i = 0; i < lineWordCount; i++) { for (size_t i = 0; i < lineWordCount; i++) {
if (wordIsFocusSuffix[lastBreakAt + i] && !outWords.empty()) { if (isFocusSuffixAt(i) && !outWords.empty()) {
// Focus suffix: merge string into the preceding bold-prefix entry. // Focus suffix: merge string into the preceding bold-prefix entry.
outWords.back() += lineWords[i]; outWords.back() += lineWords[i];
} else { } else {
// Normal word: check for a following focus suffix to record the byte boundary. // Normal word: check for a following focus suffix to record the byte boundary.
uint8_t boundary = 0; uint8_t boundary = 0;
uint16_t suffixX = 0; uint16_t suffixX = 0;
if (i + 1 < lineWordCount && wordIsFocusSuffix[lastBreakAt + i + 1]) { if (i + 1 < lineWordCount && isFocusSuffixAt(i + 1)) {
boundary = static_cast<uint8_t>(std::min(lineWords[i].size(), size_t{255})); boundary = static_cast<uint8_t>(std::min(lineWords[i].size(), size_t{255}));
// Suffix x offset = layout-time advance of the bold prefix, already known from xpos table. // Suffix x offset = layout-time advance of the bold prefix, already known from xpos table.
suffixX = static_cast<uint16_t>(lineXPos[i + 1] - lineXPos[i]); const int suffixDelta = static_cast<int>(lineXPos[i + 1]) - static_cast<int>(lineXPos[i]);
suffixX = static_cast<uint16_t>(suffixDelta > 0 ? suffixDelta : 0);
} }
outWords.push_back(std::move(lineWords[i])); outWords.push_back(std::move(lineWords[i]));
outXPos.push_back(lineXPos[i]); outXPos.push_back(lineXPos[i]);
+13 -2
View File
@@ -21,8 +21,17 @@ class ParsedText {
bool extraParagraphSpacing; bool extraParagraphSpacing;
bool hyphenationEnabled; bool hyphenationEnabled;
bool focusReadingEnabled; bool focusReadingEnabled;
bool isNaturalAlign;
bool hasRtlWord;
std::vector<std::string> reorderedWordsScratch;
std::vector<EpdFontFamily::Style> reorderedStylesScratch;
std::vector<uint16_t> reorderedWidthsScratch;
std::vector<bool> reorderedContinuesScratch;
std::vector<bool> reorderedFocusSuffixScratch;
std::vector<uint16_t> visualOrderScratch;
void applyParagraphIndent(); void applyParagraphIndent();
int resolveFirstLineIndent(bool isFirstLine) const;
std::vector<size_t> computeLineBreaks(const GfxRenderer& renderer, int fontId, int pageWidth, std::vector<size_t> computeLineBreaks(const GfxRenderer& renderer, int fontId, int pageWidth,
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec); std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec);
std::vector<size_t> computeHyphenatedLineBreaks(const GfxRenderer& renderer, int fontId, int pageWidth, std::vector<size_t> computeHyphenatedLineBreaks(const GfxRenderer& renderer, int fontId, int pageWidth,
@@ -41,7 +50,9 @@ class ParsedText {
: blockStyle(blockStyle), : blockStyle(blockStyle),
extraParagraphSpacing(extraParagraphSpacing), extraParagraphSpacing(extraParagraphSpacing),
hyphenationEnabled(hyphenationEnabled), hyphenationEnabled(hyphenationEnabled),
focusReadingEnabled(focusReadingEnabled) {} focusReadingEnabled(focusReadingEnabled),
isNaturalAlign(false),
hasRtlWord(false) {}
~ParsedText() = default; ~ParsedText() = default;
void addWord(std::string word, EpdFontFamily::Style fontStyle, bool underline = false, bool attachToPrevious = false); void addWord(std::string word, EpdFontFamily::Style fontStyle, bool underline = false, bool attachToPrevious = false);
@@ -52,4 +63,4 @@ class ParsedText {
void layoutAndExtractLines(const GfxRenderer& renderer, int fontId, uint16_t viewportWidth, void layoutAndExtractLines(const GfxRenderer& renderer, int fontId, uint16_t viewportWidth,
const std::function<void(std::shared_ptr<TextBlock>)>& processLine, const std::function<void(std::shared_ptr<TextBlock>)>& processLine,
bool includeLastLine = true); bool includeLastLine = true);
}; };
+1 -1
View File
@@ -10,7 +10,7 @@
#include "parsers/ChapterHtmlSlimParser.h" #include "parsers/ChapterHtmlSlimParser.h"
namespace { namespace {
constexpr uint8_t SECTION_FILE_VERSION = 24; constexpr uint8_t SECTION_FILE_VERSION = 25;
constexpr uint32_t HEADER_SIZE = sizeof(uint8_t) + sizeof(int) + sizeof(float) + sizeof(bool) + sizeof(uint8_t) + constexpr uint32_t HEADER_SIZE = sizeof(uint8_t) + sizeof(int) + sizeof(float) + sizeof(bool) + sizeof(uint8_t) +
sizeof(uint16_t) + sizeof(uint16_t) + sizeof(uint16_t) + sizeof(bool) + sizeof(bool) + sizeof(uint16_t) + sizeof(uint16_t) + sizeof(uint16_t) + sizeof(bool) + sizeof(bool) +
sizeof(uint8_t) + sizeof(bool) + sizeof(uint32_t) + sizeof(uint32_t) + sizeof(uint8_t) + sizeof(bool) + sizeof(uint32_t) + sizeof(uint32_t) +
+13
View File
@@ -29,6 +29,8 @@ struct BlockStyle {
int16_t textIndent = 0; int16_t textIndent = 0;
bool textIndentDefined = false; // true if text-indent was explicitly set in CSS bool textIndentDefined = false; // true if text-indent was explicitly set in CSS
bool textAlignDefined = false; // true if text-align was explicitly set in CSS bool textAlignDefined = false; // true if text-align was explicitly set in CSS
bool isRtl = false; // true if resolved direction is RTL
bool directionDefined = false; // true if direction was explicitly set in CSS/HTML
// Combined insets (margin + padding) // Combined insets (margin + padding)
[[nodiscard]] int16_t leftInset() const { return marginLeft + paddingLeft; } [[nodiscard]] int16_t leftInset() const { return marginLeft + paddingLeft; }
@@ -84,6 +86,12 @@ struct BlockStyle {
result.paddingBottom = static_cast<int16_t>(child.paddingBottom + paddingBottom); result.paddingBottom = static_cast<int16_t>(child.paddingBottom + paddingBottom);
} }
// Direction is not axis-specific. Inherit from parent when child doesn't define it.
if (!child.directionDefined && directionDefined) {
result.isRtl = isRtl;
result.directionDefined = true;
}
return result; return result;
} }
@@ -119,6 +127,11 @@ struct BlockStyle {
} else { } else {
blockStyle.alignment = paragraphAlignment; blockStyle.alignment = paragraphAlignment;
} }
// RTL direction from CSS/HTML
if (cssStyle.hasDirection()) {
blockStyle.isRtl = (cssStyle.direction == CssTextDirection::Rtl);
blockStyle.directionDefined = true;
}
return blockStyle; return blockStyle;
} }
}; };
+12 -5
View File
@@ -1,5 +1,6 @@
#include "TextBlock.h" #include "TextBlock.h"
#include <BidiUtils.h>
#include <GfxRenderer.h> #include <GfxRenderer.h>
#include <Logging.h> #include <Logging.h>
#include <Serialization.h> #include <Serialization.h>
@@ -22,6 +23,8 @@ void TextBlock::render(const GfxRenderer& renderer, const int fontId, const int
for (size_t i = 0; i < words.size(); i++) { for (size_t i = 0; i < words.size(); i++) {
const int wordX = wordXpos[i] + x; const int wordX = wordXpos[i] + x;
const EpdFontFamily::Style currentStyle = wordStyles[i]; const EpdFontFamily::Style currentStyle = wordStyles[i];
const auto baseDir = static_cast<BidiUtils::BidiBaseDir>(
BidiUtils::detectParagraphLevel(words[i].c_str(), blockStyle.isRtl ? 1 : 0));
const uint8_t boundary = hasFocus ? wordFocusBoundary[i] : 0; const uint8_t boundary = hasFocus ? wordFocusBoundary[i] : 0;
// SUP/SUB shift the baseline passed to drawText; the glyph is also scaled 50% inside // SUP/SUB shift the baseline passed to drawText; the glyph is also scaled 50% inside
@@ -48,16 +51,16 @@ void TextBlock::render(const GfxRenderer& renderer, const int fontId, const int
const size_t boldLen = std::min<size_t>({static_cast<size_t>(boundary), words[i].size(), sizeof(boldBuf) - 1}); const size_t boldLen = std::min<size_t>({static_cast<size_t>(boundary), words[i].size(), sizeof(boldBuf) - 1});
memcpy(boldBuf, words[i].c_str(), boldLen); memcpy(boldBuf, words[i].c_str(), boldLen);
boldBuf[boldLen] = '\0'; boldBuf[boldLen] = '\0';
renderer.drawText(fontId, wordX, wordY, boldBuf, true, boldStyle); renderer.drawText(fontId, wordX, wordY, boldBuf, true, boldStyle, baseDir);
const int suffixX = wordX + wordFocusSuffixX[i]; const int suffixX = wordX + wordFocusSuffixX[i];
renderer.drawText(fontId, suffixX, wordY, words[i].c_str() + boldLen, true, currentStyle); renderer.drawText(fontId, suffixX, wordY, words[i].c_str() + boldLen, true, currentStyle, baseDir);
} else { } else {
renderer.drawText(fontId, wordX, wordY, words[i].c_str(), true, currentStyle); renderer.drawText(fontId, wordX, wordY, words[i].c_str(), true, currentStyle, baseDir);
} }
if ((currentStyle & EpdFontFamily::UNDERLINE) != 0) { if ((currentStyle & EpdFontFamily::UNDERLINE) != 0) {
const std::string& w = words[i]; const std::string& w = words[i];
const int fullWordWidth = renderer.getTextWidth(fontId, w.c_str(), currentStyle); const int fullWordWidth = renderer.getTextWidth(fontId, w.c_str(), currentStyle, baseDir);
// y is the top of the text line; add ascender to reach baseline, then offset 2px below // y is the top of the text line; add ascender to reach baseline, then offset 2px below
const int underlineY = wordY + ascender + 2; const int underlineY = wordY + ascender + 2;
@@ -69,7 +72,7 @@ void TextBlock::render(const GfxRenderer& renderer, const int fontId, const int
static_cast<uint8_t>(w[2]) == 0x83) { static_cast<uint8_t>(w[2]) == 0x83) {
const char* visiblePtr = w.c_str() + 3; const char* visiblePtr = w.c_str() + 3;
const int prefixWidth = renderer.getTextAdvanceX(fontId, "\xe2\x80\x83", currentStyle); const int prefixWidth = renderer.getTextAdvanceX(fontId, "\xe2\x80\x83", currentStyle);
const int visibleWidth = renderer.getTextWidth(fontId, visiblePtr, currentStyle); const int visibleWidth = renderer.getTextWidth(fontId, visiblePtr, currentStyle, baseDir);
startX = wordX + prefixWidth; startX = wordX + prefixWidth;
underlineWidth = visibleWidth; underlineWidth = visibleWidth;
} }
@@ -118,6 +121,8 @@ bool TextBlock::serialize(HalFile& file) const {
serialization::writePod(file, blockStyle.paddingRight); serialization::writePod(file, blockStyle.paddingRight);
serialization::writePod(file, blockStyle.textIndent); serialization::writePod(file, blockStyle.textIndent);
serialization::writePod(file, blockStyle.textIndentDefined); serialization::writePod(file, blockStyle.textIndentDefined);
serialization::writePod(file, blockStyle.isRtl);
serialization::writePod(file, blockStyle.directionDefined);
return true; return true;
} }
@@ -171,6 +176,8 @@ std::unique_ptr<TextBlock> TextBlock::deserialize(HalFile& file) {
serialization::readPod(file, blockStyle.paddingRight); serialization::readPod(file, blockStyle.paddingRight);
serialization::readPod(file, blockStyle.textIndent); serialization::readPod(file, blockStyle.textIndent);
serialization::readPod(file, blockStyle.textIndentDefined); serialization::readPod(file, blockStyle.textIndentDefined);
serialization::readPod(file, blockStyle.isRtl);
serialization::readPod(file, blockStyle.directionDefined);
return std::unique_ptr<TextBlock>(new TextBlock(std::move(words), std::move(wordXpos), std::move(wordStyles), return std::unique_ptr<TextBlock>(new TextBlock(std::move(words), std::move(wordXpos), std::move(wordStyles),
std::move(wordFocusBoundary), std::move(wordFocusSuffixX), std::move(wordFocusBoundary), std::move(wordFocusSuffixX),
+21 -3
View File
@@ -344,6 +344,15 @@ void CssParser::parseDeclarationIntoStyle(const std::string& decl, CssStyle& sty
const std::string_view displayValue = stripTrailingImportant(propValueBuf); const std::string_view displayValue = stripTrailingImportant(propValueBuf);
style.display = (displayValue == "none") ? CssDisplay::None : CssDisplay::Block; style.display = (displayValue == "none") ? CssDisplay::None : CssDisplay::Block;
style.defined.display = 1; style.defined.display = 1;
} else if (propNameBuf == "direction") {
const std::string_view directionValue = stripTrailingImportant(propValueBuf);
if (directionValue == "rtl") {
style.direction = CssTextDirection::Rtl;
style.defined.direction = 1;
} else if (directionValue == "ltr") {
style.direction = CssTextDirection::Ltr;
style.defined.direction = 1;
}
} else if (propNameBuf == "vertical-align") { } else if (propNameBuf == "vertical-align") {
const std::string v = normalized(propValueBuf); const std::string v = normalized(propValueBuf);
if (v == "super") { if (v == "super") {
@@ -710,6 +719,7 @@ bool CssParser::saveToCache() const {
file.write(static_cast<uint8_t>(style.fontStyle)); file.write(static_cast<uint8_t>(style.fontStyle));
file.write(static_cast<uint8_t>(style.fontWeight)); file.write(static_cast<uint8_t>(style.fontWeight));
file.write(static_cast<uint8_t>(style.textDecoration)); file.write(static_cast<uint8_t>(style.textDecoration));
file.write(static_cast<uint8_t>(style.direction));
// Write CssLength fields (value + unit) // Write CssLength fields (value + unit)
auto writeLength = [&file](const CssLength& len) { auto writeLength = [&file](const CssLength& len) {
@@ -731,7 +741,7 @@ bool CssParser::saveToCache() const {
file.write(static_cast<uint8_t>(style.display)); file.write(static_cast<uint8_t>(style.display));
file.write(static_cast<uint8_t>(style.verticalAlign)); file.write(static_cast<uint8_t>(style.verticalAlign));
// Write defined flags as uint16_t // Write defined flags as uint32_t
uint32_t definedBits = 0; uint32_t definedBits = 0;
if (style.defined.textAlign) definedBits |= 1 << 0; if (style.defined.textAlign) definedBits |= 1 << 0;
if (style.defined.fontStyle) definedBits |= 1 << 1; if (style.defined.fontStyle) definedBits |= 1 << 1;
@@ -749,7 +759,8 @@ bool CssParser::saveToCache() const {
if (style.defined.imageHeight) definedBits |= 1 << 13; if (style.defined.imageHeight) definedBits |= 1 << 13;
if (style.defined.imageWidth) definedBits |= 1 << 14; if (style.defined.imageWidth) definedBits |= 1 << 14;
if (style.defined.display) definedBits |= 1 << 15; if (style.defined.display) definedBits |= 1 << 15;
if (style.defined.verticalAlign) definedBits |= 1 << 16; if (style.defined.direction) definedBits |= 1 << 16;
if (style.defined.verticalAlign) definedBits |= 1 << 17;
file.write(reinterpret_cast<const uint8_t*>(&definedBits), sizeof(definedBits)); file.write(reinterpret_cast<const uint8_t*>(&definedBits), sizeof(definedBits));
} }
@@ -862,6 +873,12 @@ bool CssParser::loadFromCache() {
} }
style.textDecoration = static_cast<CssTextDecoration>(enumVal); style.textDecoration = static_cast<CssTextDecoration>(enumVal);
if (file.read(&enumVal, 1) != 1) {
rulesBySelector_.clear();
return false;
}
style.direction = static_cast<CssTextDirection>(enumVal);
// Read CssLength fields // Read CssLength fields
auto readLength = [&file](CssLength& len) -> bool { auto readLength = [&file](CssLength& len) -> bool {
if (file.read(&len.value, sizeof(len.value)) != sizeof(len.value)) { if (file.read(&len.value, sizeof(len.value)) != sizeof(len.value)) {
@@ -921,7 +938,8 @@ bool CssParser::loadFromCache() {
style.defined.imageHeight = (definedBits & 1 << 13) != 0; style.defined.imageHeight = (definedBits & 1 << 13) != 0;
style.defined.imageWidth = (definedBits & 1 << 14) != 0; style.defined.imageWidth = (definedBits & 1 << 14) != 0;
style.defined.display = (definedBits & 1 << 15) != 0; style.defined.display = (definedBits & 1 << 15) != 0;
style.defined.verticalAlign = (definedBits & 1 << 16) != 0; style.defined.direction = (definedBits & 1 << 16) != 0;
style.defined.verticalAlign = (definedBits & 1 << 17) != 0;
rulesBySelector_[selector] = style; rulesBySelector_[selector] = style;
} }
+1 -1
View File
@@ -31,7 +31,7 @@
class CssParser { class CssParser {
public: public:
// Bump when CSS cache format or rules change; section caches are invalidated when this changes // Bump when CSS cache format or rules change; section caches are invalidated when this changes
static constexpr uint8_t CSS_CACHE_VERSION = 5; static constexpr uint8_t CSS_CACHE_VERSION = 6;
explicit CssParser(std::string cachePath) : cachePath(std::move(cachePath)) {} explicit CssParser(std::string cachePath) : cachePath(std::move(cachePath)) {}
~CssParser() = default; ~CssParser() = default;
+16 -2
View File
@@ -5,6 +5,7 @@
// Matches order of PARAGRAPH_ALIGNMENT in CrossPointSettings // Matches order of PARAGRAPH_ALIGNMENT in CrossPointSettings
enum class CssTextAlign : uint8_t { Justify = 0, Left = 1, Center = 2, Right = 3, None = 4 }; enum class CssTextAlign : uint8_t { Justify = 0, Left = 1, Center = 2, Right = 3, None = 4 };
enum class CssUnit : uint8_t { Pixels = 0, Em = 1, Rem = 2, Points = 3, Percent = 4 }; enum class CssUnit : uint8_t { Pixels = 0, Em = 1, Rem = 2, Points = 3, Percent = 4 };
enum class CssTextDirection : uint8_t { Ltr = 0, Rtl = 1 };
// Represents a CSS length value with its unit, allowing deferred resolution to pixels // Represents a CSS length value with its unit, allowing deferred resolution to pixels
struct CssLength { struct CssLength {
@@ -78,6 +79,7 @@ struct CssPropertyFlags {
uint16_t imageHeight : 1; uint16_t imageHeight : 1;
uint16_t imageWidth : 1; uint16_t imageWidth : 1;
uint16_t display : 1; uint16_t display : 1;
uint16_t direction : 1;
uint16_t verticalAlign : 1; uint16_t verticalAlign : 1;
CssPropertyFlags() CssPropertyFlags()
@@ -97,22 +99,27 @@ struct CssPropertyFlags {
imageHeight(0), imageHeight(0),
imageWidth(0), imageWidth(0),
display(0), display(0),
direction(0),
verticalAlign(0) {} verticalAlign(0) {}
[[nodiscard]] bool anySet() const { [[nodiscard]] bool anySet() const {
return textAlign || fontStyle || fontWeight || textDecoration || textIndent || marginTop || marginBottom || return textAlign || fontStyle || fontWeight || textDecoration || textIndent || marginTop || marginBottom ||
marginLeft || marginRight || paddingTop || paddingBottom || paddingLeft || paddingRight || imageHeight || marginLeft || marginRight || paddingTop || paddingBottom || paddingLeft || paddingRight || imageHeight ||
imageWidth || display || verticalAlign; imageWidth || display || direction || verticalAlign;
} }
void clearAll() { void clearAll() {
textAlign = fontStyle = fontWeight = textDecoration = textIndent = 0; textAlign = fontStyle = fontWeight = textDecoration = textIndent = 0;
marginTop = marginBottom = marginLeft = marginRight = 0; marginTop = marginBottom = marginLeft = marginRight = 0;
paddingTop = paddingBottom = paddingLeft = paddingRight = 0; paddingTop = paddingBottom = paddingLeft = paddingRight = 0;
imageHeight = imageWidth = display = verticalAlign = 0; imageHeight = imageWidth = display = direction = verticalAlign = 0;
} }
}; };
// Cache serializes defined flags as uint32_t with bit indices 0..17.
static_assert(sizeof(CssPropertyFlags) <= sizeof(uint32_t),
"CssPropertyFlags exceeds 32 bits; update cache read/write in CssParser.cpp");
// Represents a collection of CSS style properties // Represents a collection of CSS style properties
// Only stores properties relevant to e-ink text rendering // Only stores properties relevant to e-ink text rendering
// Length values are stored as CssLength (value + unit) for deferred resolution // Length values are stored as CssLength (value + unit) for deferred resolution
@@ -121,6 +128,7 @@ struct CssStyle {
CssFontStyle fontStyle = CssFontStyle::Normal; CssFontStyle fontStyle = CssFontStyle::Normal;
CssFontWeight fontWeight = CssFontWeight::Normal; CssFontWeight fontWeight = CssFontWeight::Normal;
CssTextDecoration textDecoration = CssTextDecoration::None; CssTextDecoration textDecoration = CssTextDecoration::None;
CssTextDirection direction = CssTextDirection::Ltr;
CssLength textIndent; // First-line indent (deferred resolution) CssLength textIndent; // First-line indent (deferred resolution)
CssLength marginTop; // Vertical spacing before block CssLength marginTop; // Vertical spacing before block
@@ -205,6 +213,10 @@ struct CssStyle {
display = base.display; display = base.display;
defined.display = 1; defined.display = 1;
} }
if (base.hasDirection()) {
direction = base.direction;
defined.direction = 1;
}
if (base.hasVerticalAlign()) { if (base.hasVerticalAlign()) {
verticalAlign = base.verticalAlign; verticalAlign = base.verticalAlign;
defined.verticalAlign = 1; defined.verticalAlign = 1;
@@ -227,6 +239,7 @@ struct CssStyle {
[[nodiscard]] bool hasImageHeight() const { return defined.imageHeight; } [[nodiscard]] bool hasImageHeight() const { return defined.imageHeight; }
[[nodiscard]] bool hasImageWidth() const { return defined.imageWidth; } [[nodiscard]] bool hasImageWidth() const { return defined.imageWidth; }
[[nodiscard]] bool hasDisplay() const { return defined.display; } [[nodiscard]] bool hasDisplay() const { return defined.display; }
[[nodiscard]] bool hasDirection() const { return defined.direction; }
[[nodiscard]] bool hasVerticalAlign() const { return defined.verticalAlign; } [[nodiscard]] bool hasVerticalAlign() const { return defined.verticalAlign; }
void reset() { void reset() {
@@ -234,6 +247,7 @@ struct CssStyle {
fontStyle = CssFontStyle::Normal; fontStyle = CssFontStyle::Normal;
fontWeight = CssFontWeight::Normal; fontWeight = CssFontWeight::Normal;
textDecoration = CssTextDecoration::None; textDecoration = CssTextDecoration::None;
direction = CssTextDirection::Ltr;
textIndent = CssLength{}; textIndent = CssLength{};
marginTop = marginBottom = marginLeft = marginRight = CssLength{}; marginTop = marginBottom = marginLeft = marginRight = CssLength{};
paddingTop = paddingBottom = paddingLeft = paddingRight = CssLength{}; paddingTop = paddingBottom = paddingLeft = paddingRight = CssLength{};
@@ -67,6 +67,13 @@ bool isTableStructuralTag(const char* name) {
return strcmp(name, "table") == 0 || strcmp(name, "tr") == 0 || strcmp(name, "td") == 0 || strcmp(name, "th") == 0; return strcmp(name, "table") == 0 || strcmp(name, "tr") == 0 || strcmp(name, "td") == 0 || strcmp(name, "th") == 0;
} }
void ChapterHtmlSlimParser::applyDirectionToEntry(StyleStackEntry& entry, const CssStyle& css) {
if (css.hasDirection()) {
entry.hasDirection = true;
entry.direction = css.direction;
}
}
// Update effective bold/italic/underline based on block style and inline style stack // Update effective bold/italic/underline based on block style and inline style stack
void ChapterHtmlSlimParser::updateEffectiveInlineStyle() { void ChapterHtmlSlimParser::updateEffectiveInlineStyle() {
// Start with block-level styles // Start with block-level styles
@@ -74,6 +81,8 @@ void ChapterHtmlSlimParser::updateEffectiveInlineStyle() {
effectiveItalic = currentCssStyle.hasFontStyle() && currentCssStyle.fontStyle == CssFontStyle::Italic; effectiveItalic = currentCssStyle.hasFontStyle() && currentCssStyle.fontStyle == CssFontStyle::Italic;
effectiveUnderline = effectiveUnderline =
currentCssStyle.hasTextDecoration() && currentCssStyle.textDecoration == CssTextDecoration::Underline; currentCssStyle.hasTextDecoration() && currentCssStyle.textDecoration == CssTextDecoration::Underline;
effectiveDirectionDefined = currentCssStyle.hasDirection();
effectiveDirection = currentCssStyle.direction;
effectiveSup = false; effectiveSup = false;
effectiveSub = false; effectiveSub = false;
@@ -88,6 +97,10 @@ void ChapterHtmlSlimParser::updateEffectiveInlineStyle() {
if (entry.hasUnderline) { if (entry.hasUnderline) {
effectiveUnderline = entry.underline; effectiveUnderline = entry.underline;
} }
if (entry.hasDirection) {
effectiveDirectionDefined = true;
effectiveDirection = entry.direction;
}
if (entry.hasSup) { if (entry.hasSup) {
effectiveSup = entry.sup; effectiveSup = entry.sup;
if (entry.sup) effectiveSub = false; if (entry.sup) effectiveSub = false;
@@ -97,6 +110,19 @@ void ChapterHtmlSlimParser::updateEffectiveInlineStyle() {
if (entry.sub) effectiveSup = false; if (entry.sub) effectiveSup = false;
} }
} }
// Keep inherited direction in the active empty text block so upcoming block starts
// can inherit from non-block ancestors such as <html dir="rtl"> / <body dir="rtl">.
if (currentTextBlock && currentTextBlock->isEmpty()) {
auto& style = currentTextBlock->getBlockStyle();
if (effectiveDirectionDefined) {
style.directionDefined = true;
style.isRtl = (effectiveDirection == CssTextDirection::Rtl);
} else {
style.directionDefined = false;
style.isRtl = false;
}
}
} }
void ChapterHtmlSlimParser::flushPendingAnchor() { void ChapterHtmlSlimParser::flushPendingAnchor() {
@@ -254,9 +280,10 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
self->xpathListItemIndex++; self->xpathListItemIndex++;
} }
// Extract class, style, and id attributes // Extract class, style, id, and dir attributes for CSS/RTL processing
std::string classAttr; std::string classAttr;
std::string styleAttr; std::string styleAttr;
std::string dirAttr;
if (atts != nullptr) { if (atts != nullptr) {
for (int i = 0; atts[i]; i += 2) { for (int i = 0; atts[i]; i += 2) {
if (strcmp(atts[i], "class") == 0) { if (strcmp(atts[i], "class") == 0) {
@@ -267,6 +294,8 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
// Defer both anchor recording and TOC page breaks until startNewTextBlock, // Defer both anchor recording and TOC page breaks until startNewTextBlock,
// after the previous block is flushed to pages via makePages(). // after the previous block is flushed to pages via makePages().
self->pendingAnchorId = atts[i + 1]; self->pendingAnchorId = atts[i + 1];
} else if (strcmp(atts[i], "dir") == 0) {
dirAttr = atts[i + 1];
} }
} }
} }
@@ -286,6 +315,24 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
} }
} }
// HTML dir attribute overrides CSS direction (case-insensitive per HTML spec)
if (!dirAttr.empty()) {
if (strcasecmp(dirAttr.c_str(), "rtl") == 0) {
cssStyle.direction = CssTextDirection::Rtl;
cssStyle.defined.direction = 1;
} else if (strcasecmp(dirAttr.c_str(), "ltr") == 0) {
cssStyle.direction = CssTextDirection::Ltr;
cssStyle.defined.direction = 1;
}
}
// Direction is inherited in HTML/CSS. If this element does not define one, carry
// the currently active inherited direction into its computed style.
if (!cssStyle.hasDirection() && self->effectiveDirectionDefined) {
cssStyle.direction = self->effectiveDirection;
cssStyle.defined.direction = 1;
}
// Skip elements with display:none before all fast paths (tables, links, etc.). // Skip elements with display:none before all fast paths (tables, links, etc.).
if (cssStyle.hasDisplay() && cssStyle.display == CssDisplay::None) { if (cssStyle.hasDisplay() && cssStyle.display == CssDisplay::None) {
self->skipUntilDepth = self->depth; self->skipUntilDepth = self->depth;
@@ -686,6 +733,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
entry.depth = self->depth; entry.depth = self->depth;
entry.hasUnderline = true; entry.hasUnderline = true;
entry.underline = true; entry.underline = true;
ChapterHtmlSlimParser::applyDirectionToEntry(entry, cssStyle);
self->inlineStyleStack.push_back(entry); self->inlineStyleStack.push_back(entry);
self->updateEffectiveInlineStyle(); self->updateEffectiveInlineStyle();
@@ -770,6 +818,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
entry.hasItalic = true; entry.hasItalic = true;
entry.italic = cssStyle.fontStyle == CssFontStyle::Italic; entry.italic = cssStyle.fontStyle == CssFontStyle::Italic;
} }
ChapterHtmlSlimParser::applyDirectionToEntry(entry, cssStyle);
self->inlineStyleStack.push_back(entry); self->inlineStyleStack.push_back(entry);
self->updateEffectiveInlineStyle(); self->updateEffectiveInlineStyle();
} else if (matches(name, BOLD_TAGS, std::size(BOLD_TAGS))) { } else if (matches(name, BOLD_TAGS, std::size(BOLD_TAGS))) {
@@ -792,6 +841,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
entry.hasUnderline = true; entry.hasUnderline = true;
entry.underline = cssStyle.textDecoration == CssTextDecoration::Underline; entry.underline = cssStyle.textDecoration == CssTextDecoration::Underline;
} }
ChapterHtmlSlimParser::applyDirectionToEntry(entry, cssStyle);
self->inlineStyleStack.push_back(entry); self->inlineStyleStack.push_back(entry);
self->updateEffectiveInlineStyle(); self->updateEffectiveInlineStyle();
} else if (matches(name, ITALIC_TAGS, std::size(ITALIC_TAGS))) { } else if (matches(name, ITALIC_TAGS, std::size(ITALIC_TAGS))) {
@@ -814,6 +864,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
entry.hasUnderline = true; entry.hasUnderline = true;
entry.underline = cssStyle.textDecoration == CssTextDecoration::Underline; entry.underline = cssStyle.textDecoration == CssTextDecoration::Underline;
} }
ChapterHtmlSlimParser::applyDirectionToEntry(entry, cssStyle);
self->inlineStyleStack.push_back(entry); self->inlineStyleStack.push_back(entry);
self->updateEffectiveInlineStyle(); self->updateEffectiveInlineStyle();
} else if (strcmp(name, "sup") == 0 || strcmp(name, "sub") == 0) { } else if (strcmp(name, "sup") == 0 || strcmp(name, "sub") == 0) {
@@ -835,7 +886,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
} else if (strcmp(name, "span") == 0 || !isHeaderOrBlock(name)) { } else if (strcmp(name, "span") == 0 || !isHeaderOrBlock(name)) {
// Handle span and other inline elements for CSS styling // Handle span and other inline elements for CSS styling
if (cssStyle.hasFontWeight() || cssStyle.hasFontStyle() || cssStyle.hasTextDecoration() || if (cssStyle.hasFontWeight() || cssStyle.hasFontStyle() || cssStyle.hasTextDecoration() ||
cssStyle.hasVerticalAlign()) { cssStyle.hasDirection() || cssStyle.hasVerticalAlign()) {
// Flush buffer before style change so preceding text gets current style // Flush buffer before style change so preceding text gets current style
if (self->partWordBufferIndex > 0) { if (self->partWordBufferIndex > 0) {
self->flushPartWordBuffer(); self->flushPartWordBuffer();
@@ -855,6 +906,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char*
entry.hasUnderline = true; entry.hasUnderline = true;
entry.underline = cssStyle.textDecoration == CssTextDecoration::Underline; entry.underline = cssStyle.textDecoration == CssTextDecoration::Underline;
} }
ChapterHtmlSlimParser::applyDirectionToEntry(entry, cssStyle);
if (cssStyle.hasVerticalAlign()) { if (cssStyle.hasVerticalAlign()) {
if (cssStyle.verticalAlign == CssVerticalAlign::Super) { if (cssStyle.verticalAlign == CssVerticalAlign::Super) {
entry.hasSup = true; entry.hasSup = true;
@@ -61,6 +61,8 @@ class ChapterHtmlSlimParser {
bool hasBold = false, bold = false; bool hasBold = false, bold = false;
bool hasItalic = false, italic = false; bool hasItalic = false, italic = false;
bool hasUnderline = false, underline = false; bool hasUnderline = false, underline = false;
bool hasDirection = false;
CssTextDirection direction = CssTextDirection::Ltr;
bool hasSup = false, sup = false; bool hasSup = false, sup = false;
bool hasSub = false, sub = false; bool hasSub = false, sub = false;
}; };
@@ -70,6 +72,8 @@ class ChapterHtmlSlimParser {
bool effectiveBold = false; bool effectiveBold = false;
bool effectiveItalic = false; bool effectiveItalic = false;
bool effectiveUnderline = false; bool effectiveUnderline = false;
bool effectiveDirectionDefined = false;
CssTextDirection effectiveDirection = CssTextDirection::Ltr;
bool effectiveSup = false; bool effectiveSup = false;
bool effectiveSub = false; bool effectiveSub = false;
int tableDepth = 0; int tableDepth = 0;
@@ -97,6 +101,7 @@ class ChapterHtmlSlimParser {
void flushPendingAnchor(); void flushPendingAnchor();
void flushPartWordBuffer(); void flushPartWordBuffer();
void makePages(); void makePages();
static void applyDirectionToEntry(StyleStackEntry& entry, const CssStyle& css);
void emitHorizontalRule(const BlockStyle& blockStyle); void emitHorizontalRule(const BlockStyle& blockStyle);
// XML callbacks // XML callbacks
static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char** atts); static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char** atts);
+69 -14
View File
@@ -1,5 +1,6 @@
#include "GfxRenderer.h" #include "GfxRenderer.h"
#include <BidiUtils.h>
#include <FontDecompressor.h> #include <FontDecompressor.h>
#include <HalGPIO.h> #include <HalGPIO.h>
#include <Logging.h> #include <Logging.h>
@@ -21,6 +22,10 @@ uint8_t resolveSdCardStyle(const SdCardFont& font, const EpdFontFamily::Style st
} }
} // namespace } // namespace
namespace {
const char* resolveVisualText(const char* text, std::string& visualBuffer, BidiUtils::BidiBaseDir baseDir);
} // namespace
const uint8_t* GfxRenderer::getGlyphBitmap(const EpdFontData* fontData, const EpdGlyph* glyph) const { const uint8_t* GfxRenderer::getGlyphBitmap(const EpdFontData* fontData, const EpdGlyph* glyph) const {
if (fontData->groups != nullptr) { if (fontData->groups != nullptr) {
auto* fd = fontCacheManager_ ? fontCacheManager_->getDecompressor() : nullptr; auto* fd = fontCacheManager_ ? fontCacheManager_->getDecompressor() : nullptr;
@@ -352,26 +357,42 @@ void GfxRenderer::drawPixel(const int x, const int y, const bool state) const {
} }
} }
int GfxRenderer::getTextWidth(const int fontId, const char* text, const EpdFontFamily::Style style) const { int GfxRenderer::getTextWidth(const int fontId, const char* text, const EpdFontFamily::Style style,
const BidiUtils::BidiBaseDir baseDir) const {
if (text == nullptr || *text == '\0') {
return 0;
}
const auto fontIt = fontMap.find(fontId); const auto fontIt = fontMap.find(fontId);
if (fontIt == fontMap.end()) { if (fontIt == fontMap.end()) {
LOG_ERR("GFX", "Font %d not found", fontId); LOG_ERR("GFX", "Font %d not found", fontId);
return 0; return 0;
} }
std::string visual;
const char* renderedText = resolveVisualText(text, visual, baseDir);
int w = 0, h = 0; int w = 0, h = 0;
fontIt->second.getTextDimensions(text, &w, &h, style); fontIt->second.getTextDimensions(renderedText, &w, &h, style);
return w; return w;
} }
void GfxRenderer::drawCenteredText(const int fontId, const int y, const char* text, const bool black, void GfxRenderer::drawCenteredText(const int fontId, const int y, const char* text, const bool black,
const EpdFontFamily::Style style) const { const EpdFontFamily::Style style, const BidiUtils::BidiBaseDir baseDir) const {
const int x = (getScreenWidth() - getTextWidth(fontId, text, style)) / 2; const int x = (getScreenWidth() - getTextWidth(fontId, text, style, baseDir)) / 2;
drawText(fontId, x, y, text, black, style); drawText(fontId, x, y, text, black, style, baseDir);
} }
void GfxRenderer::drawText(const int fontId, const int x, const int y, const char* text, const bool black, void GfxRenderer::drawText(const int fontId, const int x, const int y, const char* text, const bool black,
const EpdFontFamily::Style style) const { const EpdFontFamily::Style style, const BidiUtils::BidiBaseDir baseDir) const {
// cannot draw a NULL / empty string
if (text == nullptr || *text == '\0') {
return;
}
std::string visual;
const char* renderedText = resolveVisualText(text, visual, baseDir);
const int yPos = y + getFontAscenderSize(fontId); const int yPos = y + getFontAscenderSize(fontId);
int lastBaseX = x; int lastBaseX = x;
int lastBaseLeft = 0; int lastBaseLeft = 0;
@@ -379,13 +400,8 @@ void GfxRenderer::drawText(const int fontId, const int x, const int y, const cha
int lastBaseTop = 0; int lastBaseTop = 0;
int32_t prevAdvanceFP = 0; // 12.4 fixed-point: prev glyph's advance + next kern for snap int32_t prevAdvanceFP = 0; // 12.4 fixed-point: prev glyph's advance + next kern for snap
// cannot draw a NULL / empty string
if (text == nullptr || *text == '\0') {
return;
}
if (fontCacheManager_ && fontCacheManager_->isScanning()) { if (fontCacheManager_ && fontCacheManager_->isScanning()) {
fontCacheManager_->recordText(text, fontId, style); fontCacheManager_->recordText(renderedText, fontId, style);
return; return;
} }
@@ -396,9 +412,16 @@ void GfxRenderer::drawText(const int fontId, const int x, const int y, const cha
} }
const auto& font = fontIt->second; const auto& font = fontIt->second;
const char* textCursor = renderedText;
uint32_t cp; uint32_t cp;
uint32_t prevCp = 0; uint32_t prevCp = 0;
while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text)))) { while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&textCursor)))) {
// Skip Hebrew Niqqud (vowel marks)
// Temporary: avoid adding Niqqud to built-in fonts. Remove when custom fonts are supported.
if (cp >= 0x0591 && cp <= 0x05C7) {
continue;
}
if (utf8IsCombiningMark(cp)) { if (utf8IsCombiningMark(cp)) {
const EpdGlyph* combiningGlyph = font.getGlyph(cp, style); const EpdGlyph* combiningGlyph = font.getGlyph(cp, style);
if (!combiningGlyph) continue; if (!combiningGlyph) continue;
@@ -409,7 +432,7 @@ void GfxRenderer::drawText(const int fontId, const int x, const int y, const cha
continue; continue;
} }
cp = font.applyLigatures(cp, text, style); cp = font.applyLigatures(cp, textCursor, style);
// Differential rounding: snap (previous advance + current kern) as one unit so // Differential rounding: snap (previous advance + current kern) as one unit so
// identical character pairs always produce the same pixel step regardless of // identical character pairs always produce the same pixel step regardless of
@@ -443,6 +466,32 @@ void GfxRenderer::drawText(const int fontId, const int x, const int y, const cha
} }
} }
namespace {
const char* resolveVisualText(const char* text, std::string& visualBuffer, const BidiUtils::BidiBaseDir baseDir) {
if (!text || *text == '\0') return text;
if (baseDir != BidiUtils::BidiBaseDir::RTL) {
// Byte-level scan: skip BiDi when no RTL script lead bytes are present.
// Hebrew UTF-8 lead bytes: 0xD6-0xD7; Arabic/Syriac: 0xD8-0xDB.
// This covers all RTL content without false negatives and avoids triggering
// the full UAX#9 algorithm for Latin-extended, em-dashes, accented text, etc.
bool hasRtlBytes = false;
for (const unsigned char* q = reinterpret_cast<const unsigned char*>(text); *q; ++q) {
if (*q >= 0xD6 && *q <= 0xDB) {
hasRtlBytes = true;
break;
}
}
if (!hasRtlBytes) return text;
}
if (BidiUtils::applyBidiVisual(text, visualBuffer, static_cast<int>(baseDir)) && !visualBuffer.empty()) {
return visualBuffer.c_str();
}
return text;
}
} // namespace
void GfxRenderer::drawLine(int x1, int y1, int x2, int y2, const bool state) const { void GfxRenderer::drawLine(int x1, int y1, int x2, int y2, const bool state) const {
if (fontCacheManager_ && fontCacheManager_->isScanning()) return; if (fontCacheManager_ && fontCacheManager_->isScanning()) return;
if (x1 == x2) { if (x1 == x2) {
@@ -1481,6 +1530,12 @@ void GfxRenderer::drawTextRotated90CW(const int fontId, const int x, const int y
uint32_t cp; uint32_t cp;
uint32_t prevCp = 0; uint32_t prevCp = 0;
while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text)))) { while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text)))) {
// Skip Hebrew Niqqud (vowel marks)
// Temporary: avoid adding Niqqud to built-in fonts. Remove when custom fonts are supported.
if (cp >= 0x0591 && cp <= 0x05C7) {
continue;
}
if (utf8IsCombiningMark(cp)) { if (utf8IsCombiningMark(cp)) {
const EpdGlyph* combiningGlyph = font.getGlyph(cp, style); const EpdGlyph* combiningGlyph = font.getGlyph(cp, style);
if (!combiningGlyph) continue; if (!combiningGlyph) continue;
+14 -3
View File
@@ -3,6 +3,14 @@
#include <EpdFontFamily.h> #include <EpdFontFamily.h>
#include <HalDisplay.h> #include <HalDisplay.h>
namespace BidiUtils {
// Paragraph base direction for the Unicode BiDi algorithm (UAX#9).
// AUTO: scan text for first strong directional character (P2/P3 rules)
// LTR: force left-to-right paragraph embedding level
// RTL: force right-to-left paragraph embedding level
enum class BidiBaseDir : signed char { AUTO = -1, LTR = 0, RTL = 1 };
} // namespace BidiUtils
class FontCacheManager; class FontCacheManager;
class SdCardFont; class SdCardFont;
@@ -175,11 +183,14 @@ class GfxRenderer {
void fillPolygon(const int* xPoints, const int* yPoints, int numPoints, bool state = true) const; void fillPolygon(const int* xPoints, const int* yPoints, int numPoints, bool state = true) const;
// Text // Text
int getTextWidth(int fontId, const char* text, EpdFontFamily::Style style = EpdFontFamily::REGULAR) const; int getTextWidth(int fontId, const char* text, EpdFontFamily::Style style = EpdFontFamily::REGULAR,
BidiUtils::BidiBaseDir baseDir = BidiUtils::BidiBaseDir::AUTO) const;
void drawCenteredText(int fontId, int y, const char* text, bool black = true, void drawCenteredText(int fontId, int y, const char* text, bool black = true,
EpdFontFamily::Style style = EpdFontFamily::REGULAR) const; EpdFontFamily::Style style = EpdFontFamily::REGULAR,
BidiUtils::BidiBaseDir baseDir = BidiUtils::BidiBaseDir::AUTO) const;
void drawText(int fontId, int x, int y, const char* text, bool black = true, void drawText(int fontId, int x, int y, const char* text, bool black = true,
EpdFontFamily::Style style = EpdFontFamily::REGULAR) const; EpdFontFamily::Style style = EpdFontFamily::REGULAR,
BidiUtils::BidiBaseDir baseDir = BidiUtils::BidiBaseDir::AUTO) const;
int getSpaceWidth(int fontId, EpdFontFamily::Style style = EpdFontFamily::REGULAR) const; int getSpaceWidth(int fontId, EpdFontFamily::Style style = EpdFontFamily::REGULAR) const;
/// Returns the total inter-word advance: fp4::toPixel(spaceAdvance + kern(leftCp,' ') + kern(' ',rightCp)). /// Returns the total inter-word advance: fp4::toPixel(spaceAdvance + kern(leftCp,' ') + kern(' ',rightCp)).
/// Using a single snap avoids the +/-1 px rounding error that arises when space advance and kern are /// Using a single snap avoids the +/-1 px rounding error that arises when space advance and kern are
+234
View File
@@ -0,0 +1,234 @@
#include "BidiUtils.h"
extern "C" {
#include "minibidi.h"
}
#undef when
#undef otherwise
#include <Logging.h>
#include <Utf8.h>
#include <cstring>
namespace {
bool isNaturalDirectionClass(const uchar cls) {
switch (cls) {
case L:
case R:
case AL:
case EN:
case AN:
return true;
default:
return false;
}
}
} // namespace
namespace BidiUtils {
bool startsWithRtl(const char* utf8, int maxStrongChars) {
if (!utf8 || maxStrongChars <= 0) return false;
auto* p = reinterpret_cast<const unsigned char*>(utf8);
int checked = 0;
while (*p) {
const uint32_t cp = utf8NextCodepoint(&p);
if (!cp || cp == REPLACEMENT_GLYPH) break;
const uchar cls = bidi_class(cp);
if (cls == R || cls == AL) return true;
if (cls == L) return false;
checked++;
if (checked >= maxStrongChars) break;
}
return false;
}
int detectParagraphLevel(const char* utf8, const int fallbackLevel, const int maxStrongChars) {
if (!utf8 || maxStrongChars <= 0) return fallbackLevel & 1;
auto* p = reinterpret_cast<const unsigned char*>(utf8);
int checked = 0;
while (*p) {
const uint32_t cp = utf8NextCodepoint(&p);
if (!cp || cp == REPLACEMENT_GLYPH) break;
const uchar cls = bidi_class(cp);
if (cls == R || cls == AL) return 1;
if (cls == L) return 0;
checked++;
if (checked >= maxStrongChars) break;
}
return fallbackLevel & 1;
}
bool applyBidiVisual(const char* utf8, std::string& out, int paragraphLevel) {
if (!utf8 || !*utf8) return false;
static bidi_char line[BIDI_MAX_LINE];
int count = 0;
auto* p = reinterpret_cast<const unsigned char*>(utf8);
while (*p) {
if (count >= BIDI_MAX_LINE) {
LOG_DBG("BIDI", "applyBidiVisual: input exceeds BIDI_MAX_LINE (%d chars), returning unprocessed", BIDI_MAX_LINE);
return false;
}
const uint32_t cp = utf8NextCodepoint(&p);
if (!cp || cp == REPLACEMENT_GLYPH) break;
line[count].origwc = line[count].wc = cp;
line[count].index = static_cast<uint16_t>(count);
count++;
}
if (!count) return false;
const bool autodir = (paragraphLevel < 0);
const int level = autodir ? 0 : (paragraphLevel & 1);
do_bidi(autodir, level, line, count);
out.clear();
out.reserve(std::strlen(utf8));
for (int i = 0; i < count; i++) {
utf8AppendCodepoint(line[i].wc, out);
}
return true;
}
bool computeVisualWordOrder(const std::vector<std::string>& words, bool paragraphIsRtl,
std::vector<uint16_t>& visualOrder) {
visualOrder.clear();
const size_t nWords = words.size();
if (nWords <= 1 || nWords > BIDI_MAX_LINE) return false;
static bidi_char line[BIDI_MAX_LINE];
int count = 0;
bool truncated = false;
for (size_t w = 0; w < nWords && !truncated; w++) {
auto* p = reinterpret_cast<const unsigned char*>(words[w].c_str());
while (*p) {
if (count >= BIDI_MAX_LINE) {
truncated = true;
break;
}
const uint32_t cp = utf8NextCodepoint(&p);
if (!cp || cp == REPLACEMENT_GLYPH) break;
line[count].origwc = line[count].wc = cp;
line[count].index = static_cast<uint16_t>(w);
count++;
}
if (!truncated && w + 1 < nWords) {
if (count >= BIDI_MAX_LINE) {
truncated = true;
break;
}
line[count].origwc = line[count].wc = ' ';
line[count].index = static_cast<uint16_t>(nWords);
count++;
}
}
if (truncated || count == 0) return false;
// Fast-path for homogeneous lines: skip UAX#9 if there's no mixing.
bool hasL = false, hasR = false;
for (int i = 0; i < count; i++) {
uchar bc = bidi_class(line[i].wc);
if (bc == L || bc == EN || bc == AN)
hasL = true;
else if (bc == R || bc == AL)
hasR = true;
}
// Purely LTR line in RTL paragraph: identity order, but we might still need to reorder
// if some characters are mirrored or neutral resolution differs.
// Actually, UAX#9 rule L1/L2 says purely LTR in RTL para stays as is (identity).
// Purely RTL line: just reverse the words.
if (!hasL && hasR && paragraphIsRtl) {
visualOrder.reserve(nWords);
for (int i = static_cast<int>(nWords) - 1; i >= 0; i--) {
visualOrder.push_back(static_cast<uint16_t>(i));
}
return true;
}
if (!hasR) {
if (!paragraphIsRtl) {
// Pure LTR in LTR paragraph: nothing to do.
return false;
}
// Pure LTR in RTL paragraph: no word reordering, but must use the
// willReorder (left-to-right) positioning path, not the RTL right-to-left path.
visualOrder.reserve(nWords);
for (size_t i = 0; i < nWords; i++) {
visualOrder.push_back(static_cast<uint16_t>(i));
}
return true;
}
do_bidi(/*autodir=*/false, paragraphIsRtl ? 1 : 0, line, count);
uint16_t firstAny[BIDI_MAX_LINE];
uint16_t firstNatural[BIDI_MAX_LINE];
for (size_t w = 0; w < nWords; w++) {
firstAny[w] = UINT16_MAX;
firstNatural[w] = UINT16_MAX;
}
for (int i = 0; i < count; i++) {
const uint16_t w = line[i].index;
if (w >= nWords) continue;
if (firstAny[w] == UINT16_MAX) {
firstAny[w] = static_cast<uint16_t>(i);
}
if (firstNatural[w] == UINT16_MAX && isNaturalDirectionClass(bidi_class(line[i].wc))) {
firstNatural[w] = static_cast<uint16_t>(i);
}
}
visualOrder.reserve(nWords);
for (int i = 0; i < count; i++) {
const uint16_t w = line[i].index;
if (w >= nWords) continue;
const uint16_t anchor = firstNatural[w] != UINT16_MAX ? firstNatural[w] : firstAny[w];
if (anchor == UINT16_MAX) {
visualOrder.clear();
return false;
}
if (anchor == static_cast<uint16_t>(i)) {
visualOrder.push_back(w);
}
}
if (visualOrder.size() != nWords) {
visualOrder.clear();
return false;
}
// Check if the order is exactly the same as the original input
bool needsReorder = false;
for (size_t i = 0; i < nWords; i++) {
if (visualOrder[i] != i) {
needsReorder = true;
break;
}
}
if (!needsReorder) {
visualOrder.clear();
return false;
}
return true;
}
} // namespace BidiUtils
+22
View File
@@ -0,0 +1,22 @@
#pragma once
#include <cstdint>
#include <string>
#include <vector>
namespace BidiUtils {
// Paragraph-level P2/P3: scan the first N strong chars per word to find base direction.
inline constexpr int RTL_PARAGRAPH_PROBE_DEPTH = 5;
bool startsWithRtl(const char* utf8, int maxStrongChars = RTL_PARAGRAPH_PROBE_DEPTH);
int detectParagraphLevel(const char* utf8, int fallbackLevel = 0, int maxStrongChars = 64);
// paragraphLevel: -1 = auto-detect, 0 = LTR, 1 = RTL
bool applyBidiVisual(const char* utf8, std::string& out, int paragraphLevel = -1);
bool computeVisualWordOrder(const std::vector<std::string>& words, bool paragraphIsRtl,
std::vector<uint16_t>& visualOrder);
} // namespace BidiUtils
+39
View File
@@ -0,0 +1,39 @@
/* bidi_pairs.t unified mirror + bracket table for CrossPoint.
*
* Replaces both mirroring.t and brackets.t. canonical.t is dropped
* (fullwidth brackets are not used in Hebrew epub content).
*
* Each entry: {from, to, bracket_type}
* bracket_type == BRACKo : `from` is an opening bracket
* bracket_type == BRACKc : `from` is a closing bracket; `to` = opener
* bracket_type == BRACKx : not a bracket pair mirrored by rule L4 only
*
* mirror(c) always returns `to` for any entry where c==from
* bracket(c) returns 0 for BRACKx; c for BRACKo; `to` (opener) for BRACKc
*
* Sorted ascending by `from` (binary search).
*/
/* ASCII brackets — both bracket pairs AND L4 mirrors */
{0x0028, 0x0029, BRACKo}, /* ( */
{0x0029, 0x0028, BRACKc}, /* ) */
{0x003C, 0x003E, BRACKo}, /* < */
{0x003E, 0x003C, BRACKc}, /* > */
{0x005B, 0x005D, BRACKo}, /* [ */
{0x005D, 0x005B, BRACKc}, /* ] */
{0x007B, 0x007D, BRACKo}, /* { */
{0x007D, 0x007B, BRACKc}, /* } */
/* Angle quotation marks — L4 mirror only */
{0x00AB, 0x00BB, BRACKx}, /* « → » */
{0x00BB, 0x00AB, BRACKx}, /* » → « */
/* Curly quotes — L4 mirror only */
{0x2018, 0x2019, BRACKx}, /* ' → ' */
{0x2019, 0x2018, BRACKx}, /* ' → ' */
{0x201C, 0x201D, BRACKx}, /* " → " */
{0x201D, 0x201C, BRACKx}, /* " → " */
/* Single angle quotes */
{0x2039, 0x203A, BRACKo}, /* */
{0x203A, 0x2039, BRACKc}, /* */
+115
View File
@@ -0,0 +1,115 @@
/* bidiclasses.t — bidi class table for CrossPoint Hebrew/English epub.
*
* Coverage rationale:
* Hebrew + English is the primary target. However, CrossPoint renders
* Latin and Cyrillic scripts for many other languages, so these MUST be
* classified as L (not fall through to ON) to avoid regression when they
* appear adjacent to Hebrew runs.
*
* Scripts NOT in this table fall through to ON correct per UAX#9 for
* scripts CrossPoint's fonts don't support (CJK, Arabic, Devanagari, etc.)
* ON is the right class for "unknown" it behaves neutrally.
*
* Entries sorted ascending by first (binary search requirement).
*/
/* ── ASCII C0 controls ────────────────────────────────────────────────── */
{0x0000, 0x0008, BN},
{0x0009, 0x0009, S},
{0x000A, 0x000A, B},
{0x000B, 0x000B, S},
{0x000C, 0x000C, WS},
{0x000D, 0x000D, B},
{0x000E, 0x001B, BN},
{0x001C, 0x001E, B},
{0x001F, 0x001F, S},
{0x0020, 0x0020, WS},
/* ── ASCII punctuation: number-adjacent classes ─────────────────────── */
{0x0023, 0x0025, ET}, /* # $ % */
{0x002B, 0x002B, ES}, /* + */
{0x002C, 0x002C, CS}, /* , */
{0x002D, 0x002D, ES}, /* - */
{0x002E, 0x002F, CS}, /* . / */
{0x0030, 0x0039, EN}, /* 0-9 */
{0x003A, 0x003A, CS}, /* : */
/* ── Basic Latin letters ─────────────────────────────────────────────── */
{0x0041, 0x005A, L}, /* A-Z */
{0x0061, 0x007A, L}, /* a-z */
/* ── C1 / BN */
{0x007F, 0x0084, BN},
{0x0085, 0x0085, B},
{0x0086, 0x009F, BN},
/* ── Latin-1 supplement ─────────────────────────────────────────────── */
{0x00A0, 0x00A0, CS}, /* non-breaking space */
{0x00A2, 0x00A5, ET}, /* ¢ £ ¤ ¥ */
{0x00AA, 0x00AA, L},
{0x00AD, 0x00AD, BN}, /* soft hyphen */
{0x00B0, 0x00B1, ET}, /* ° ± */
{0x00B2, 0x00B3, EN}, /* ² ³ */
{0x00B5, 0x00B5, L},
{0x00B9, 0x00B9, EN}, /* ¹ */
{0x00BA, 0x00BA, L},
{0x00C0, 0x00D6, L},
{0x00D8, 0x00F6, L},
{0x00F8, 0x02B8, L}, /* Latin Extended-A/B, IPA, Spacing Modifiers
covers: Polish, Czech, Slovak, Turkish, etc. */
/* ── Combining Diacritical Marks (NSM) ──────────────────────────────── */
/* Needed for decomposed Latin characters (some epubs use NFD/NFKD form) */
{0x0300, 0x036F, NSM},
/* ── Cyrillic (L) ────────────────────────────────────────────────────── */
/* Required: CrossPoint supports Russian, Ukrainian, Bulgarian, etc.
Without these, Cyrillic chars fall to ON, breaking mixed Hebrew+Russian. */
{0x0400, 0x04FF, L}, /* Cyrillic */
{0x0500, 0x052F, L}, /* Cyrillic Supplement */
/* ── Hebrew vowel points / cantillation (NSM) */
/* Do NOT remove: niqqud must be NSM or pointed Hebrew breaks after reorder */
{0x0591, 0x05A1, NSM},
{0x05A3, 0x05B9, NSM},
{0x05BB, 0x05BD, NSM},
{0x05BE, 0x05BE, R}, /* maqaf (Hebrew hyphen) */
{0x05BF, 0x05BF, NSM},
{0x05C0, 0x05C0, R}, /* paseq */
{0x05C1, 0x05C2, NSM},
{0x05C3, 0x05C3, R}, /* sof pasuq */
{0x05C4, 0x05C4, NSM},
/* ── Hebrew letters ─────────────────────────────────────────────────── */
{0x05D0, 0x05EA, R}, /* alef … tav */
{0x05F0, 0x05F4, R}, /* alternative forms + geresh/gershayim */
/* ── Latin Extended Additional (L) ─────────────────────────────────── */
/* Covers accented chars for Vietnamese, Welsh, Romanian, etc.
Not currently rendered by CrossPoint fonts, but costs only 2 table rows. */
{0x1E00, 0x1EFF, L},
/* ── Unicode directional format characters ─────────────────────────── */
/* All must be present — UBA X-rules depend on them */
{0x200B, 0x200D, BN}, /* ZWSP, ZWNJ, ZWJ */
{0x200E, 0x200E, L}, /* LEFT-TO-RIGHT MARK */
{0x200F, 0x200F, R}, /* RIGHT-TO-LEFT MARK */
{0x2028, 0x2028, WS},
{0x2029, 0x2029, B},
{0x202A, 0x202A, LRE},
{0x202B, 0x202B, RLE},
{0x202C, 0x202C, PDF},
{0x202D, 0x202D, LRO},
{0x202E, 0x202E, RLO},
{0x202F, 0x202F, WS}, /* narrow no-break space */
{0x2060, 0x2063, BN},
/* Unicode 6.3 isolate markers */
{0x2066, 0x2066, LRI},
{0x2067, 0x2067, RLI},
{0x2068, 0x2068, FSI},
{0x2069, 0x2069, PDI},
{0x206A, 0x206F, BN},
/* ── Byte Order Mark ────────────────────────────────────────────────── */
{0xFEFF, 0xFEFF, BN},
+565
View File
@@ -0,0 +1,565 @@
/*
* minibidi.c — Unicode Bidirectional Algorithm (UAX #9) for CrossPoint/ESP32C3
*
* Original author: Ahmad Khalifa (www.arabeyes.org, MIT licence)
* Mintty changes: Thomas Wolff (rules N0, W7/L1/X9 fixes, isolates)
*
* UAX #9: https://www.unicode.org/reports/tr9/
*/
#include "minibidi.h"
#define leastGreaterOdd(x) (((x) + 1) | 1)
#define leastGreaterEven(x) (((x) + 2) & ~1)
/* ═══════════════════════════════════════════════════════════════════════
* flip_runs / find_run (UAX#9 rule L2)
* ═══════════════════════════════════════════════════════════════════════ */
static int find_run(uchar* levels, int start, int count, int tlevel) {
for (int i = start; i < count; i++)
if (tlevel <= levels[i]) return i;
return count;
}
static void flip_runs(bidi_char* from, uchar* levels, int tlevel, int count) {
int i = 0, j = 0;
while (i < count && j < count) {
i = j = find_run(levels, i, count, tlevel);
while (i < count && tlevel <= levels[i]) i++;
for (int k = i - 1; k > j; k--, j++) {
bidi_char tmp = from[k];
from[k] = from[j];
from[j] = tmp;
}
}
}
/* ═══════════════════════════════════════════════════════════════════════
* bidi_class()
* ═══════════════════════════════════════════════════════════════════════ */
uchar bidi_class(ucschar ch) {
static const struct {
ucschar first, last;
uchar type;
} lookup[] = {
#include "bidiclasses.t"
};
int i = -1, j = lengthof(lookup);
while (j - i > 1) {
int k = (i + j) / 2;
if (ch < lookup[k].first)
j = k;
else if (ch > lookup[k].last)
i = k;
else
return lookup[k].type;
}
return ON; /* correct UAX#9 fallback for unlisted characters */
}
/* ═══════════════════════════════════════════════════════════════════════
* Character class predicates
* ═══════════════════════════════════════════════════════════════════════ */
bool is_rtl_class(uchar bc) {
const int mask = (1 << R) | (1 << AL) | (1 << RLE) | (1 << RLO) | (1 << RLI) | (1 << FSI);
return (mask >> bc) & 1;
}
static inline bool is_NI(uchar bc) {
const int mask = (1 << B) | (1 << S) | (1 << WS) | (1 << ON) | (1 << FSI) | (1 << LRI) | (1 << RLI) | (1 << PDI);
return (mask >> bc) & 1;
}
/* ═══════════════════════════════════════════════════════════════════════
* Unified bracket + mirror table (bidi_pairs.t)
*
* Replaces both brackets.t and mirroring.t. canonical.t is dropped.
* ═══════════════════════════════════════════════════════════════════════ */
enum { BRACKx = 0, BRACKo = 1, BRACKc = 2 };
typedef struct {
ucschar from, to;
uchar bracket; /* BRACKo / BRACKc / BRACKx */
} bidi_pair;
static const bidi_pair pairs[] = {
#include "bidi_pairs.t"
};
/* Binary search over the pairs table */
static const bidi_pair* find_pair(ucschar c) {
int i = -1, j = lengthof(pairs);
while (j - i > 1) {
int k = (i + j) / 2;
if (c == pairs[k].from)
return &pairs[k];
else if (c < pairs[k].from)
j = k;
else
i = k;
}
return NULL;
}
/*
* bracket(c):
* 0 → not a bracket
* c → opening bracket
* opener → closing bracket (returns the matching opener)
*/
static ucschar bracket(ucschar c) {
const bidi_pair* p = find_pair(c);
if (!p || p->bracket == BRACKx) return 0;
return (p->bracket == BRACKo) ? c : p->to;
}
/*
* mirror(c): returns the mirrored form for rule L4,
* or c unchanged if not in the table.
*/
ucschar mirror(ucschar c) {
const bidi_pair* p = find_pair(c);
return p ? p->to : c;
}
/* ═══════════════════════════════════════════════════════════════════════
* Directional Status Stack
* (replaces GCC nested functions — ESP32C3 has no executable stack)
* ═══════════════════════════════════════════════════════════════════════ */
typedef struct {
uchar emb[BIDI_MAX_LINE + 1];
uchar ovr[BIDI_MAX_LINE + 1];
bool isol[BIDI_MAX_LINE + 1];
int top;
} DirStatusStack;
static inline void dss_init(DirStatusStack* s) { s->top = -1; }
static inline int dss_count(const DirStatusStack* s) { return s->top + 1; }
static inline void dss_push(DirStatusStack* s, uchar emb, uchar ovr, bool isol) {
if (s->top < BIDI_MAX_LINE) {
++s->top;
s->emb[s->top] = emb;
s->ovr[s->top] = ovr;
s->isol[s->top] = isol;
}
}
static inline void dss_pop(DirStatusStack* s, uchar* emb, uchar* ovr, bool* isol) {
if (s->top >= 0) s->top--;
if (s->top >= 0) {
*emb = s->emb[s->top];
*ovr = s->ovr[s->top];
*isol = s->isol[s->top];
} else {
/* Stack underflow: return safe defaults (should not happen in valid input) */
*emb = 0; /* LTR base level */
*ovr = ON; /* No override */
*isol = false; /* No isolate */
}
}
/* ═══════════════════════════════════════════════════════════════════════
* do_bidi() — The main UAX#9 algorithm
* ═══════════════════════════════════════════════════════════════════════ */
int do_bidi(bool autodir, int paragraphLevel, bidi_char* line, int count) {
if (count > BIDI_MAX_LINE) count = BIDI_MAX_LINE;
uchar currentEmbedding, currentOverride;
bool currentIsolate;
int i, j;
/* Fixed-size working arrays — no VLAs, no heap */
uchar types[BIDI_MAX_LINE];
uchar levels[BIDI_MAX_LINE];
bool skip[BIDI_MAX_LINE];
/* ── P2/P3: detect paragraph level ── */
int isolateLevel = 0, resLevel = -1;
bool hasRTL = false;
for (i = 0; i < count; i++) {
uchar type = bidi_class(line[i].wc);
if (type == LRI || type == RLI || type == FSI) {
hasRTL = true;
isolateLevel++;
} else if (type == PDI) {
hasRTL = true;
if (isolateLevel > 0) isolateLevel--;
} else if (isolateLevel == 0) {
if (type == R || type == AL) {
hasRTL = true;
if (resLevel < 0) resLevel = 1;
break;
} else if (type == RLE || type == LRE || type == RLO || type == LRO || type == PDF) {
hasRTL = true;
if (resLevel >= 0) break;
} else if (type == L) {
if (resLevel < 0) resLevel = 0;
} else if (type == AN)
hasRTL = true;
}
}
if (autodir) {
if (resLevel >= 0) paragraphLevel = resLevel;
} else
resLevel = paragraphLevel;
/* Fast path: pure LTR line with LTR paragraph — nothing to reorder */
if (!hasRTL && !paragraphLevel) return 0;
/* ── X1X8: compute embedding levels ── */
currentEmbedding = (uchar)paragraphLevel;
currentOverride = ON;
currentIsolate = false;
isolateLevel = 0;
DirStatusStack dss;
dss_init(&dss);
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
for (i = 0; i < count; i++) {
uchar tempType = bidi_class(line[i].wc);
levels[i] = currentEmbedding;
/* FSI: look-ahead to resolve direction */
if (tempType == FSI) {
int lvl = 0;
tempType = LRI;
for (int k = i + 1; k < count; k++) {
uchar kt = bidi_class(line[k].wc);
if (kt == FSI || kt == RLI || kt == LRI)
lvl++;
else if (kt == PDI) {
if (lvl)
lvl--;
else
break;
} else if (kt == R || kt == AL) {
tempType = RLI;
break;
} else if (kt == L)
break;
}
}
switch (tempType) {
when RLE : currentEmbedding = leastGreaterOdd(currentEmbedding);
currentOverride = ON;
currentIsolate = false;
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
when LRE : currentEmbedding = leastGreaterEven(currentEmbedding);
currentOverride = ON;
currentIsolate = false;
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
when RLO : currentEmbedding = leastGreaterOdd(currentEmbedding);
currentOverride = R;
currentIsolate = false;
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
when LRO : currentEmbedding = leastGreaterEven(currentEmbedding);
currentOverride = L;
currentIsolate = false;
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
when RLI : if (currentOverride != ON) tempType = currentOverride;
currentEmbedding = leastGreaterOdd(currentEmbedding);
isolateLevel++;
currentOverride = ON;
currentIsolate = true;
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
when LRI : if (currentOverride != ON) tempType = currentOverride;
currentEmbedding = leastGreaterEven(currentEmbedding);
isolateLevel++;
currentOverride = ON;
currentIsolate = true;
dss_push(&dss, currentEmbedding, currentOverride, currentIsolate);
when PDF : if (!currentIsolate && dss_count(&dss) >= 2)
dss_pop(&dss, &currentEmbedding, &currentOverride, &currentIsolate);
levels[i] = currentEmbedding;
when PDI : if (isolateLevel > 0) {
while (!currentIsolate && dss_count(&dss) > 0)
dss_pop(&dss, &currentEmbedding, &currentOverride, &currentIsolate);
dss_pop(&dss, &currentEmbedding, &currentOverride, &currentIsolate);
isolateLevel--;
}
if (currentOverride != ON) tempType = currentOverride;
levels[i] = currentEmbedding;
when WS : case S:
if (currentOverride != ON) tempType = currentOverride;
otherwise : if (currentOverride != ON) tempType = currentOverride;
}
types[i] = tempType;
}
/* ── X9: mask format chars as NSM (Wolff fix: NSM not BN) ── */
for (i = 0; i < count; i++) {
switch (types[i]) {
when RLE : case LRE:
case RLO:
case LRO:
case PDF:
case BN:
types[i] = NSM;
skip[i] = true;
otherwise:
skip[i] = false;
}
}
/* ── W1: NSM inherits type of previous char (or sor) ── */
if (types[0] == NSM) types[0] = (paragraphLevel & 1) ? R : L;
for (i = 1; i < count; i++) {
if (types[i] == NSM) {
switch (types[i - 1]) {
when LRI : case RLI:
case FSI:
case PDI:
types[i] = ON;
otherwise : types[i] = types[i - 1];
}
}
}
/* ── W2: EN after AL → AN ── */
for (i = 0; i < count; i++) {
if (types[i] == EN) {
for (j = i - 1; j >= 0; j--) {
uchar t = types[j];
if (t == AL) {
types[i] = AN;
break;
}
if (t == R || t == L) break;
}
}
}
/* ── W3: AL → R ── */
for (i = 0; i < count; i++)
if (types[i] == AL) types[i] = R;
/* ── W4: single ES/CS between same numerals → that numeral type ── */
for (i = 1; i + 1 < count; i++) {
if (types[i] == ES || types[i] == CS) {
int prev = i - 1;
while (prev >= 0 && skip[prev]) prev--;
int next = i + 1;
while (next < count && skip[next]) next++;
if (prev >= 0 && next < count) {
if (types[i] == ES && types[prev] == EN && types[next] == EN) types[i] = EN;
if (types[i] == CS) {
if (types[prev] == EN && types[next] == EN) types[i] = EN;
if (types[prev] == AN && types[next] == AN) types[i] = AN;
}
}
}
}
/* ── W5: ET adjacent to EN → EN (forward pass) ── */
for (i = 0; i < count; i++) {
if (skip[i] || types[i] != ET) continue;
for (j = i; j < count; j++) {
if (skip[j]) continue;
if (types[j] == ET) continue;
if (types[j] == EN) types[i] = EN;
break;
}
}
/* W5 backward pass */
for (i = count - 1; i >= 0; i--) {
if (skip[i] || types[i] != ET) continue;
for (j = i; j >= 0; j--) {
if (skip[j]) continue;
if (types[j] == ET) continue;
if (types[j] == EN) types[i] = EN;
break;
}
}
/* ── W6: remaining ES, ET, CS → ON ── */
for (i = 0; i < count; i++)
if (types[i] == ES || types[i] == ET || types[i] == CS) types[i] = ON;
/* ── W7: EN after last strong L (back to sor) → L ── */
{
uchar last_strong = (paragraphLevel & 1) ? R : L;
for (i = 0; i < count; i++) {
if (skip[i]) continue;
if (types[i] == L || types[i] == R) last_strong = types[i];
if (types[i] == EN && last_strong == L) types[i] = L;
}
}
/* ── N0: bracket pair handling ── */
{
uchar e = (paragraphLevel & 1) ? R : L;
uchar o = (e == L) ? R : L;
#define BRACKET_STACK 63
struct {
ucschar opener;
int pos;
} openers[BRACKET_STACK];
int opener_top = 0;
for (i = 0; i < count; i++) {
if (skip[i]) continue;
ucschar bc = bracket(line[i].wc);
if (!bc) continue;
if (bc == line[i].wc) {
/* Opening bracket */
if (opener_top < BRACKET_STACK) {
openers[opener_top].opener = line[i].wc;
openers[opener_top].pos = i;
opener_top++;
}
} else {
/* Closing bracket: find matching opener */
int k;
for (k = opener_top - 1; k >= 0; k--)
if (openers[k].opener == bc) break;
if (k < 0) continue;
int open_pos = openers[k].pos;
opener_top = k;
bool found_e = false, found_o = false;
for (int m = open_pos + 1; m < i; m++) {
if (skip[m]) continue;
uchar t = types[m];
if (t == EN || t == AN) t = R;
if (t == R || t == AL) {
if (e == R)
found_e = true;
else
found_o = true;
} else if (t == L) {
if (e == L)
found_e = true;
else
found_o = true;
}
}
uchar dir;
if (found_e) {
dir = e;
} else if (found_o) {
uchar ctx = e;
for (int m = open_pos - 1; m >= 0; m--) {
if (skip[m]) continue;
uchar t = types[m];
if (t == EN || t == AN) t = R;
if (t == R || t == AL) {
ctx = R;
break;
} else if (t == L) {
ctx = L;
break;
}
}
dir = (ctx == o) ? o : e;
} else {
continue;
}
types[open_pos] = dir;
types[i] = dir;
for (int m = open_pos + 1; m < i; m++)
if (is_NI(types[m])) types[m] = dir;
}
}
#undef BRACKET_STACK
}
/* ── N1: NI between same-direction strongs → that direction ── */
for (i = 0; i < count; i++) {
if (skip[i] || !is_NI(types[i])) continue;
int end = i;
while (end + 1 < count && (skip[end + 1] || is_NI(types[end + 1]))) end++;
uchar prev_strong = (paragraphLevel & 1) ? R : L;
for (j = i - 1; j >= 0; j--) {
if (skip[j]) continue;
uchar t = types[j];
if (t == EN || t == AN) t = R;
if (t == R || t == L) {
prev_strong = t;
break;
}
}
uchar next_strong = (paragraphLevel & 1) ? R : L;
for (j = end + 1; j < count; j++) {
if (skip[j]) continue;
uchar t = types[j];
if (t == EN || t == AN) t = R;
if (t == R || t == L) {
next_strong = t;
break;
}
}
if (prev_strong == next_strong)
for (j = i; j <= end; j++) types[j] = prev_strong;
i = end;
}
/* ── N2: remaining NI → embedding direction ── */
for (i = 0; i < count; i++)
if (is_NI(types[i])) types[i] = (levels[i] & 1) ? R : L;
/* ── I1/I2: adjust levels ── */
for (i = 0; i < count; i++) {
if (skip[i]) continue;
if ((levels[i] & 1) == 0) {
if (types[i] == R)
levels[i] += 1;
else if (types[i] == AN || types[i] == EN)
levels[i] += 2;
} else {
if (types[i] == L || types[i] == EN || types[i] == AN) levels[i] += 1;
}
}
/* ── L1: reset trailing/segment whitespace to paragraph level ── */
for (i = count - 1; i >= 0; i--) {
if (skip[i]) continue;
uchar t = types[i];
if (t == WS || t == S || t == B)
levels[i] = (uchar)paragraphLevel;
else
break;
}
for (i = 0; i < count; i++) {
if (types[i] == S) {
levels[i] = (uchar)paragraphLevel;
for (j = i - 1; j >= 0; j--) {
if (skip[j]) continue;
if (types[j] == WS || types[j] == BN)
levels[j] = (uchar)paragraphLevel;
else
break;
}
}
}
/* ── L2: reverse from highest level down to lowest odd ── */
uchar max_level = (uchar)paragraphLevel, min_odd = 255;
for (i = 0; i < count; i++) {
if (levels[i] > max_level) max_level = levels[i];
if ((levels[i] & 1) && levels[i] < min_odd) min_odd = levels[i];
}
for (int level = max_level; level >= (int)min_odd; level--) flip_runs(line, levels, level, count);
/* ── L4: mirror characters in RTL runs ── */
for (i = 0; i < count; i++)
if (levels[i] & 1) line[i].wc = mirror(line[i].wc);
return paragraphLevel;
}
+115
View File
@@ -0,0 +1,115 @@
#ifndef MINIBIDI_H
#define MINIBIDI_H
/*
* minibidi.h — standalone header for ESP32C3 BiDi calculations
*
* Derived from [mintty](https://github.com/mintty/mintty/) (Thomas Wolff, MIT licence).
* Stripped of: Arabic shaping, box-drawing mirror, terminal dependencies,
* GCC nested functions, VLAs, and non-Hebrew/English Unicode data.
*/
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
/* ── Basic types ─────────────────────────────────────────────────────── */
typedef uint8_t uchar;
typedef uint32_t ucschar; /* Unicode codepoint; BMP-only content fits uint16_t
but uint32_t is safer and ESP32C3 is 32-bit anyway */
/* ── Convenience macros ──────────────────────────────────────────────── */
#define lengthof(a) ((int)(sizeof(a) / sizeof(*(a))))
/* PuTTY/mintty switch-case style — kept for readability of algorithm */
#define when \
break; \
case
#define otherwise \
break; \
default
/* Maximum line length the algorithm will process.
Adjust to your actual screen width. Stack cost = ~5×MAX bytes. */
#define BIDI_MAX_LINE 128
/* ── bidi_char ───────────────────────────────────────────────────────── */
/* origwc: the codepoint as it came from the epub text stream
wc: working codepoint (may be replaced by mirrored form after do_bidi)
index: original logical position, so the caller can reorder glyphs */
typedef struct {
ucschar origwc;
ucschar wc;
uint16_t index;
} bidi_char;
/* ── Bidi character classes (UAX #9) ────────────────────────────────── */
enum {
L, /* Left-to-Right */
LRE, /* Left-to-Right Embedding */
LRO, /* Left-to-Right Override */
R, /* Right-to-Left */
AL, /* Right-to-Left Arabic */
RLE, /* Right-to-Left Embedding */
RLO, /* Right-to-Left Override */
PDF, /* Pop Directional Format */
EN, /* European Number */
ES, /* European Number Separator */
ET, /* European Number Terminator */
AN, /* Arabic Number */
CS, /* Common Number Separator */
NSM, /* Non-Spacing Mark */
BN, /* Boundary Neutral */
B, /* Paragraph Separator */
S, /* Segment Separator */
WS, /* Whitespace */
ON, /* Other Neutrals */
/* Unicode 6.3 isolate types */
LRI, /* Left-to-Right Isolate */
RLI, /* Right-to-Left Isolate */
FSI, /* First Strong Isolate */
PDI, /* Pop Directional Isolate */
};
/* ── Public API ──────────────────────────────────────────────────────── */
/*
* bidi_class(ch)
* Returns the UAX#9 bidi class of Unicode codepoint ch.
* Unknown characters return ON (correct per spec).
*/
uchar bidi_class(ucschar ch);
/*
* is_rtl_class(bc)
* Returns true if bidi class bc can cause RTL reordering.
* Use to fast-skip lines with no RTL content.
*/
bool is_rtl_class(uchar bc);
/*
* mirror(ch)
* Returns the mirrored form of Unicode codepoint ch for UAX#9 rule L4.
* If no mirror exists, returns ch unchanged.
*/
ucschar mirror(ucschar ch);
/*
* do_bidi(autodir, paragraphLevel, line, count)
*
* Applies UAX#9 Bidirectional Algorithm (rules PL) to `line[0..count-1]`.
* Reorders the array in-place; sets line[i].wc to the mirrored form where
* required (rule L4). Returns the resolved paragraph level (0=LTR, 1=RTL),
* or 0 if the line was left-to-right and no reordering was done.
*
* autodir: true → detect paragraph direction from content (P2/P3)
* false → use paragraphLevel as-is
* paragraphLevel: 0 = LTR, 1 = RTL. Ignored when autodir=true unless
* the content has no strong type (used as fallback).
*
* count must be ≤ BIDI_MAX_LINE; lines longer than that are silently
* truncated to BIDI_MAX_LINE before processing.
*/
int do_bidi(bool autodir, int paragraphLevel, bidi_char* line, int count);
#endif /* MINIBIDI_H */
+18
View File
@@ -56,6 +56,24 @@ uint32_t utf8NextCodepoint(const unsigned char** string) {
return cp; return cp;
} }
void utf8AppendCodepoint(uint32_t cp, std::string& out) {
if (cp < 0x80) {
out += static_cast<char>(cp);
} else if (cp < 0x800) {
out += static_cast<char>(0xC0 | (cp >> 6));
out += static_cast<char>(0x80 | (cp & 0x3F));
} else if (cp < 0x10000) {
out += static_cast<char>(0xE0 | (cp >> 12));
out += static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
out += static_cast<char>(0x80 | (cp & 0x3F));
} else {
out += static_cast<char>(0xF0 | (cp >> 18));
out += static_cast<char>(0x80 | ((cp >> 12) & 0x3F));
out += static_cast<char>(0x80 | ((cp >> 6) & 0x3F));
out += static_cast<char>(0x80 | (cp & 0x3F));
}
}
int utf8SafeTruncateBuffer(const char* buf, int len) { int utf8SafeTruncateBuffer(const char* buf, int len) {
if (len <= 0) return 0; if (len <= 0) return 0;
+2
View File
@@ -5,6 +5,8 @@
#define REPLACEMENT_GLYPH 0xFFFD #define REPLACEMENT_GLYPH 0xFFFD
uint32_t utf8NextCodepoint(const unsigned char** string); uint32_t utf8NextCodepoint(const unsigned char** string);
// Appends a Unicode codepoint to a std::string in UTF-8 encoding.
void utf8AppendCodepoint(uint32_t cp, std::string& out);
// Remove the last UTF-8 codepoint from a std::string and return the new size. // Remove the last UTF-8 codepoint from a std::string and return the new size.
size_t utf8RemoveLastChar(std::string& str); size_t utf8RemoveLastChar(std::string& str);
// Truncate string by removing N UTF-8 codepoints from the end. // Truncate string by removing N UTF-8 codepoints from the end.
+9 -3
View File
@@ -1,5 +1,6 @@
#include "TxtReaderActivity.h" #include "TxtReaderActivity.h"
#include <BidiUtils.h>
#include <FontCacheManager.h> #include <FontCacheManager.h>
#include <GfxRenderer.h> #include <GfxRenderer.h>
#include <HalStorage.h> #include <HalStorage.h>
@@ -358,20 +359,25 @@ void TxtReaderActivity::renderPage() {
for (const auto& line : currentPageLines) { for (const auto& line : currentPageLines) {
if (!line.empty()) { if (!line.empty()) {
int x = cachedOrientedMarginLeft; int x = cachedOrientedMarginLeft;
const bool lineIsRtl = BidiUtils::startsWithRtl(line.c_str(), BidiUtils::RTL_PARAGRAPH_PROBE_DEPTH);
uint8_t effectiveAlignment = cachedParagraphAlignment;
if (lineIsRtl && (effectiveAlignment == CrossPointSettings::LEFT_ALIGN ||
effectiveAlignment == CrossPointSettings::JUSTIFIED)) {
effectiveAlignment = CrossPointSettings::RIGHT_ALIGN;
}
const int textWidth = renderer.getTextAdvanceX(cachedFontId, line.c_str(), EpdFontFamily::REGULAR);
// Apply text alignment // Apply text alignment
switch (cachedParagraphAlignment) { switch (effectiveAlignment) {
case CrossPointSettings::LEFT_ALIGN: case CrossPointSettings::LEFT_ALIGN:
default: default:
// x already set to left margin // x already set to left margin
break; break;
case CrossPointSettings::CENTER_ALIGN: { case CrossPointSettings::CENTER_ALIGN: {
int textWidth = renderer.getTextAdvanceX(cachedFontId, line.c_str(), EpdFontFamily::REGULAR);
x = cachedOrientedMarginLeft + (contentWidth - textWidth) / 2; x = cachedOrientedMarginLeft + (contentWidth - textWidth) / 2;
break; break;
} }
case CrossPointSettings::RIGHT_ALIGN: { case CrossPointSettings::RIGHT_ALIGN: {
int textWidth = renderer.getTextAdvanceX(cachedFontId, line.c_str(), EpdFontFamily::REGULAR);
x = cachedOrientedMarginLeft + contentWidth - textWidth; x = cachedOrientedMarginLeft + contentWidth - textWidth;
break; break;
} }
Binary file not shown.
Binary file not shown.

After

Width:  |  Height:  |  Size: 47 KiB

Binary file not shown.

After

Width:  |  Height:  |  Size: 47 KiB

+7
View File
@@ -0,0 +1,7 @@
# Simple Hebrew-English Test EPUB
If you modify any code involving RTL logic, please perform a visual regression check: Open the EPUB with reader font `NotoSansHebrew` (Medium), and compare the rendering against the reference screenshots.
Binary file not shown.