feat: Support for Korean line breaks and glyph spacing (#2288)
Co-authored-by: Uri Tauber <uritaube@gmail.com>
This commit is contained in:
co-authored by
Uri Tauber
parent
d4069aeae5
commit
22f3575064
+241
-32
@@ -24,6 +24,7 @@ constexpr size_t RTL_PARAGRAPH_PROBE_WORDS = 3;
|
||||
// Per-word: scan enough chars to see through leading neutrals (quotes, numbers)
|
||||
// before giving up. 64 is a hedge for pathological cases like long numeric tokens.
|
||||
constexpr int RTL_PER_WORD_PROBE_DEPTH = 64;
|
||||
constexpr size_t MIN_JUSTIFY_GAPS = 1;
|
||||
|
||||
// Byte-level pre-check: Hebrew UTF-8 lead bytes 0xD6-0xD7, Arabic/Syriac 0xD8-0xDB.
|
||||
bool mayContainRtlBytes(const char* str) {
|
||||
@@ -57,6 +58,134 @@ uint32_t lastCodepoint(const std::string& word) {
|
||||
|
||||
bool containsSoftHyphen(const std::string& word) { return word.find(SOFT_HYPHEN_UTF8) != std::string::npos; }
|
||||
|
||||
bool isNoBreakBeforeCjkPunctuation(const uint32_t cp) {
|
||||
switch (cp) {
|
||||
case '.':
|
||||
case ',':
|
||||
case ':':
|
||||
case ';':
|
||||
case '!':
|
||||
case '?':
|
||||
case ')':
|
||||
case ']':
|
||||
case '}':
|
||||
case 0x00BB: // »
|
||||
case 0x2019: // ’
|
||||
case 0x201D: // ”
|
||||
case 0x3001: // 、
|
||||
case 0x3002: // 。
|
||||
case 0x3009: // 〉
|
||||
case 0x300B: // 》
|
||||
case 0x300D: // 」
|
||||
case 0x300F: // 』
|
||||
case 0x3011: // 】
|
||||
case 0x3015: // 〕
|
||||
case 0x3017: // 〗
|
||||
case 0x3019: // 〙
|
||||
case 0x301B: // 〛
|
||||
case 0xFF01: // !
|
||||
case 0xFF09: // )
|
||||
case 0xFF0C: // ,
|
||||
case 0xFF0E: // .
|
||||
case 0xFF1A: // :
|
||||
case 0xFF1B: // ;
|
||||
case 0xFF1F: // ?
|
||||
case 0xFF3D: // ]
|
||||
case 0xFF5D: // }
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool isNoBreakAfterCjkPunctuation(const uint32_t cp) {
|
||||
switch (cp) {
|
||||
case '(':
|
||||
case '[':
|
||||
case '{':
|
||||
case 0x00AB: // «
|
||||
case 0x2018: // ‘
|
||||
case 0x201C: // “
|
||||
case 0x3008: // 〈
|
||||
case 0x300A: // 《
|
||||
case 0x300C: // 「
|
||||
case 0x300E: // 『
|
||||
case 0x3010: // 【
|
||||
case 0x3014: // 〔
|
||||
case 0x3016: // 〖
|
||||
case 0x3018: // 〘
|
||||
case 0x301A: // 〚
|
||||
case 0xFF08: // (
|
||||
case 0xFF3B: // [
|
||||
case 0xFF5B: // {
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
bool containsCjkBreakableCodepoint(const std::string& text) {
|
||||
const auto* ptr = reinterpret_cast<const unsigned char*>(text.c_str());
|
||||
while (*ptr) {
|
||||
const uint32_t cp = utf8NextCodepoint(&ptr);
|
||||
if (utf8IsCjkBreakable(cp)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool hasCjkBreakOpportunityBetween(const uint32_t leftCp, const uint32_t rightCp) {
|
||||
if (!utf8IsCjkBreakable(leftCp) && !utf8IsCjkBreakable(rightCp)) return false;
|
||||
if (isNoBreakAfterCjkPunctuation(leftCp) || isNoBreakBeforeCjkPunctuation(rightCp)) return false;
|
||||
if (utf8IsCombiningMark(rightCp)) return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
std::vector<size_t> cjkCharacterBreakByteOffsets(const std::string& text) {
|
||||
struct CodepointBoundary {
|
||||
uint32_t cp;
|
||||
size_t endOffset;
|
||||
};
|
||||
|
||||
std::vector<CodepointBoundary> codepoints;
|
||||
codepoints.reserve(text.size());
|
||||
bool hasCjkBreakable = false;
|
||||
|
||||
const auto* ptr = reinterpret_cast<const unsigned char*>(text.c_str());
|
||||
const auto* const start = ptr;
|
||||
while (*ptr) {
|
||||
const uint32_t cp = utf8NextCodepoint(&ptr);
|
||||
if (cp == 0) break;
|
||||
if (utf8IsCjkBreakable(cp)) {
|
||||
hasCjkBreakable = true;
|
||||
}
|
||||
codepoints.push_back({cp, static_cast<size_t>(ptr - start)});
|
||||
}
|
||||
|
||||
if (!hasCjkBreakable || codepoints.size() < 2) return {};
|
||||
|
||||
std::vector<size_t> allowedOffsets;
|
||||
allowedOffsets.reserve(codepoints.size() - 1);
|
||||
for (size_t i = 0; i + 1 < codepoints.size(); ++i) {
|
||||
const uint32_t current = codepoints[i].cp;
|
||||
const uint32_t next = codepoints[i + 1].cp;
|
||||
if (!hasCjkBreakOpportunityBetween(current, next)) continue;
|
||||
allowedOffsets.push_back(codepoints[i].endOffset);
|
||||
}
|
||||
return allowedOffsets;
|
||||
}
|
||||
|
||||
int computeJustifyExtra(const int spareSpace, const size_t gapCount) {
|
||||
if (gapCount < MIN_JUSTIFY_GAPS || spareSpace <= 0) return 0;
|
||||
// Distribute the spare space evenly across gaps. Do NOT bail out to 0 when the
|
||||
// per-gap stretch is large: a sparse line (few words on a wide page) legitimately
|
||||
// needs big gaps to reach the margin. Returning 0 there disables justification for
|
||||
// that line, leaving it right-aligned (RTL) / left-aligned (LTR) — the mismatched
|
||||
// alignment bug. Match the un-capped behavior of the old code.
|
||||
return spareSpace / static_cast<int>(gapCount);
|
||||
}
|
||||
|
||||
// Removes every soft hyphen in-place so rendered glyphs match measured widths.
|
||||
void stripSoftHyphensInPlace(std::string& word) {
|
||||
size_t pos = 0;
|
||||
@@ -132,12 +261,54 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
|
||||
const bool wordStartsRtl = !hasRtlWord && mayContainRtlBytes(word.c_str()) &&
|
||||
BidiUtils::startsWithRtl(word.c_str(), RTL_PER_WORD_PROBE_DEPTH);
|
||||
|
||||
const auto pushToken = [&](std::string token, const bool continues, const bool noSpaceBefore,
|
||||
const bool isFocusSuffix) {
|
||||
words.push_back(std::move(token));
|
||||
wordStyles.push_back(baseStyle);
|
||||
wordContinues.push_back(continues);
|
||||
wordNoSpaceBefore.push_back(noSpaceBefore);
|
||||
wordIsFocusSuffix.push_back(isFocusSuffix);
|
||||
};
|
||||
|
||||
bool effectiveAttachToPrevious = attachToPrevious;
|
||||
bool effectiveNoSpaceBefore = false;
|
||||
if (attachToPrevious && !words.empty() &&
|
||||
hasCjkBreakOpportunityBetween(lastCodepoint(words.back()), firstCodepoint(word))) {
|
||||
effectiveAttachToPrevious = false;
|
||||
effectiveNoSpaceBefore = true;
|
||||
}
|
||||
|
||||
if (auto breakOffsets = cjkCharacterBreakByteOffsets(word); !breakOffsets.empty()) {
|
||||
bool firstToken = true;
|
||||
size_t tokenStart = 0;
|
||||
for (const size_t breakOffset : breakOffsets) {
|
||||
if (breakOffset <= tokenStart || breakOffset > word.size()) continue;
|
||||
pushToken(word.substr(tokenStart, breakOffset - tokenStart), firstToken ? effectiveAttachToPrevious : false,
|
||||
firstToken ? effectiveNoSpaceBefore : true, false);
|
||||
firstToken = false;
|
||||
tokenStart = breakOffset;
|
||||
}
|
||||
if (tokenStart < word.size()) {
|
||||
pushToken(word.substr(tokenStart), firstToken ? effectiveAttachToPrevious : false,
|
||||
firstToken ? effectiveNoSpaceBefore : true, false);
|
||||
}
|
||||
if (wordStartsRtl) {
|
||||
hasRtlWord = true;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (containsCjkBreakableCodepoint(word)) {
|
||||
pushToken(std::move(word), effectiveAttachToPrevious, effectiveNoSpaceBefore, false);
|
||||
if (wordStartsRtl) {
|
||||
hasRtlWord = true;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
// Already-bold text should stay fully bold; focus splitting would make its suffix regular later.
|
||||
if (!this->focusReadingEnabled || (baseStyle & EpdFontFamily::BOLD) != 0) {
|
||||
words.push_back(std::move(word));
|
||||
wordStyles.push_back(baseStyle);
|
||||
wordContinues.push_back(attachToPrevious);
|
||||
wordIsFocusSuffix.push_back(false);
|
||||
pushToken(std::move(word), effectiveAttachToPrevious, effectiveNoSpaceBefore, false);
|
||||
if (wordStartsRtl) {
|
||||
hasRtlWord = true;
|
||||
}
|
||||
@@ -166,17 +337,19 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
|
||||
words.reserve(newCapacity);
|
||||
wordStyles.reserve(newCapacity);
|
||||
wordContinues.reserve(newCapacity);
|
||||
wordNoSpaceBefore.reserve(newCapacity);
|
||||
wordIsFocusSuffix.reserve(newCapacity);
|
||||
}
|
||||
|
||||
// Lambda helper to process and push individual sub-segments of the string
|
||||
// Use std::string_view to avoid heap allocations when slicing
|
||||
auto processSegment = [&](std::string_view segment, bool isWord, bool attach) {
|
||||
auto processSegment = [&](std::string_view segment, bool isWord, bool attach, bool noSpaceBefore) {
|
||||
if (!isWord) {
|
||||
// Punctuation and Numbers stay regular
|
||||
words.emplace_back(segment);
|
||||
wordStyles.push_back(baseStyle);
|
||||
wordContinues.push_back(attach);
|
||||
wordNoSpaceBefore.push_back(noSpaceBefore);
|
||||
wordIsFocusSuffix.push_back(false);
|
||||
} else {
|
||||
size_t charCount = 0;
|
||||
@@ -198,6 +371,7 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
|
||||
words.emplace_back(segment);
|
||||
wordStyles.push_back(static_cast<EpdFontFamily::Style>(baseStyle | EpdFontFamily::BOLD));
|
||||
wordContinues.push_back(attach);
|
||||
wordNoSpaceBefore.push_back(noSpaceBefore);
|
||||
wordIsFocusSuffix.push_back(false);
|
||||
} else {
|
||||
countPtr = reinterpret_cast<const unsigned char*>(segment.data());
|
||||
@@ -210,12 +384,14 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
|
||||
words.emplace_back(segment.substr(0, splitByteOffset));
|
||||
wordStyles.push_back(static_cast<EpdFontFamily::Style>(baseStyle | EpdFontFamily::BOLD));
|
||||
wordContinues.push_back(attach);
|
||||
wordNoSpaceBefore.push_back(noSpaceBefore);
|
||||
wordIsFocusSuffix.push_back(false);
|
||||
|
||||
// Regular suffix - marked so extractLine can merge it back into single TextBlock entry
|
||||
words.emplace_back(segment.substr(splitByteOffset));
|
||||
wordStyles.push_back(baseStyle);
|
||||
wordContinues.push_back(true);
|
||||
wordNoSpaceBefore.push_back(false);
|
||||
wordIsFocusSuffix.push_back(true);
|
||||
}
|
||||
}
|
||||
@@ -243,7 +419,8 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
|
||||
|
||||
// Only the very first segment inherits the original attachToPrevious flag.
|
||||
// Every subsequent segment MUST attach=true so it glues seamlessly to the prefix.
|
||||
processSegment(segment, inWordSegment, isFirstSegment ? attachToPrevious : true);
|
||||
processSegment(segment, inWordSegment, isFirstSegment ? effectiveAttachToPrevious : true,
|
||||
isFirstSegment ? effectiveNoSpaceBefore : false);
|
||||
|
||||
// Setup for the next segment
|
||||
segmentStart = currentCpStart;
|
||||
@@ -255,7 +432,8 @@ void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle,
|
||||
// Process the final remaining segment
|
||||
size_t segmentLen = end - segmentStart;
|
||||
std::string_view segment(reinterpret_cast<const char*>(segmentStart), segmentLen);
|
||||
processSegment(segment, inWordSegment, isFirstSegment ? attachToPrevious : true);
|
||||
processSegment(segment, inWordSegment, isFirstSegment ? effectiveAttachToPrevious : true,
|
||||
isFirstSegment ? effectiveNoSpaceBefore : false);
|
||||
if (wordStartsRtl) {
|
||||
hasRtlWord = true;
|
||||
}
|
||||
@@ -324,14 +502,16 @@ void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fo
|
||||
std::vector<size_t> lineBreakIndices;
|
||||
if (hyphenationEnabled) {
|
||||
// Use greedy layout that can split words mid-loop when a hyphenated prefix fits.
|
||||
lineBreakIndices = computeHyphenatedLineBreaks(renderer, fontId, pageWidth, wordWidths, wordContinues);
|
||||
lineBreakIndices =
|
||||
computeHyphenatedLineBreaks(renderer, fontId, pageWidth, wordWidths, wordContinues, wordNoSpaceBefore);
|
||||
} else {
|
||||
lineBreakIndices = computeLineBreaks(renderer, fontId, pageWidth, wordWidths, wordContinues);
|
||||
lineBreakIndices = computeLineBreaks(renderer, fontId, pageWidth, wordWidths, wordContinues, wordNoSpaceBefore);
|
||||
}
|
||||
const size_t lineCount = includeLastLine ? lineBreakIndices.size() : lineBreakIndices.size() - 1;
|
||||
|
||||
for (size_t i = 0; i < lineCount; ++i) {
|
||||
extractLine(i, pageWidth, wordWidths, wordContinues, lineBreakIndices, processLine, renderer, fontId);
|
||||
extractLine(i, pageWidth, wordWidths, wordContinues, wordNoSpaceBefore, lineBreakIndices, processLine, renderer,
|
||||
fontId);
|
||||
}
|
||||
|
||||
// Remove consumed words so size() reflects only remaining words
|
||||
@@ -340,6 +520,7 @@ void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fo
|
||||
words.erase(words.begin(), words.begin() + consumed);
|
||||
wordStyles.erase(wordStyles.begin(), wordStyles.begin() + consumed);
|
||||
wordContinues.erase(wordContinues.begin(), wordContinues.begin() + consumed);
|
||||
wordNoSpaceBefore.erase(wordNoSpaceBefore.begin(), wordNoSpaceBefore.begin() + consumed);
|
||||
wordIsFocusSuffix.erase(wordIsFocusSuffix.begin(), wordIsFocusSuffix.begin() + consumed);
|
||||
}
|
||||
}
|
||||
@@ -356,7 +537,8 @@ std::vector<uint16_t> ParsedText::calculateWordWidths(const GfxRenderer& rendere
|
||||
}
|
||||
|
||||
std::vector<size_t> ParsedText::computeLineBreaks(const GfxRenderer& renderer, const int fontId, const int pageWidth,
|
||||
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec) {
|
||||
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec,
|
||||
std::vector<bool>& noSpaceBeforeVec) {
|
||||
if (words.empty()) {
|
||||
return {};
|
||||
}
|
||||
@@ -395,7 +577,9 @@ std::vector<size_t> ParsedText::computeLineBreaks(const GfxRenderer& renderer, c
|
||||
for (size_t j = i; j < totalWordCount; ++j) {
|
||||
// Add space before word j, unless it's the first word on the line or a continuation
|
||||
int gap = 0;
|
||||
if (j > static_cast<size_t>(i) && !continuesVec[j]) {
|
||||
if (j > static_cast<size_t>(i) && noSpaceBeforeVec[j]) {
|
||||
gap = 0;
|
||||
} else if (j > static_cast<size_t>(i) && !continuesVec[j]) {
|
||||
gap =
|
||||
renderer.getSpaceAdvance(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]);
|
||||
} else if (j > static_cast<size_t>(i) && continuesVec[j]) {
|
||||
@@ -470,7 +654,8 @@ std::vector<size_t> ParsedText::computeLineBreaks(const GfxRenderer& renderer, c
|
||||
// Builds break indices while opportunistically splitting the word that would overflow the current line.
|
||||
std::vector<size_t> ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& renderer, const int fontId,
|
||||
const int pageWidth, std::vector<uint16_t>& wordWidths,
|
||||
std::vector<bool>& continuesVec) {
|
||||
std::vector<bool>& continuesVec,
|
||||
std::vector<bool>& noSpaceBeforeVec) {
|
||||
const int firstLineIndent = resolveFirstLineIndent(true, renderer, fontId);
|
||||
|
||||
std::vector<size_t> lineBreakIndices;
|
||||
@@ -488,7 +673,9 @@ std::vector<size_t> ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& r
|
||||
while (currentIndex < wordWidths.size()) {
|
||||
const bool isFirstWord = currentIndex == lineStart;
|
||||
int spacing = 0;
|
||||
if (!isFirstWord && !continuesVec[currentIndex]) {
|
||||
if (!isFirstWord && noSpaceBeforeVec[currentIndex]) {
|
||||
spacing = 0;
|
||||
} else if (!isFirstWord && !continuesVec[currentIndex]) {
|
||||
spacing = renderer.getSpaceAdvance(fontId, lastCodepoint(words[currentIndex - 1]),
|
||||
firstCodepoint(words[currentIndex]), wordStyles[currentIndex - 1]);
|
||||
} else if (!isFirstWord && continuesVec[currentIndex]) {
|
||||
@@ -618,6 +805,7 @@ bool ParsedText::hyphenateWordAtIndex(const size_t wordIndex, const int availabl
|
||||
// line, while "kilometer" moves to the next line.
|
||||
// wordContinues[wordIndex] is intentionally left unchanged — the prefix keeps its original attachment.
|
||||
wordContinues.insert(wordContinues.begin() + wordIndex + 1, false);
|
||||
wordNoSpaceBefore.insert(wordNoSpaceBefore.begin() + wordIndex + 1, false);
|
||||
|
||||
// Update cached widths to reflect the new prefix/remainder pairing.
|
||||
wordWidths[wordIndex] = static_cast<uint16_t>(chosenWidth);
|
||||
@@ -627,7 +815,8 @@ bool ParsedText::hyphenateWordAtIndex(const size_t wordIndex, const int availabl
|
||||
}
|
||||
|
||||
void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const std::vector<uint16_t>& wordWidths,
|
||||
const std::vector<bool>& continuesVec, const std::vector<size_t>& lineBreakIndices,
|
||||
const std::vector<bool>& continuesVec, const std::vector<bool>& noSpaceBeforeVec,
|
||||
const std::vector<size_t>& lineBreakIndices,
|
||||
const std::function<void(std::shared_ptr<TextBlock>)>& processLine,
|
||||
const GfxRenderer& renderer, const int fontId) {
|
||||
const size_t lineBreak = lineBreakIndices[breakIndex];
|
||||
@@ -660,7 +849,11 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
for (size_t wordIdx = 0; wordIdx < lineWordCount; wordIdx++) {
|
||||
lineWordWidthSum += wordWidths[lastBreakAt + wordIdx];
|
||||
// Count gaps: each word after the first creates a gap, unless it's a continuation
|
||||
if (wordIdx > 0 && !continuesVec[lastBreakAt + wordIdx]) {
|
||||
if (wordIdx > 0 && noSpaceBeforeVec[lastBreakAt + wordIdx]) {
|
||||
// Unicode break opportunity with no inserted Latin-style space. It is still
|
||||
// a stretchable gap for justified CJK/Korean text.
|
||||
actualGapCount++;
|
||||
} else if (wordIdx > 0 && !continuesVec[lastBreakAt + wordIdx]) {
|
||||
actualGapCount++;
|
||||
totalNaturalGaps += renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx - 1]),
|
||||
firstCodepoint(lineWords[wordIdx]), lineWordStyles[wordIdx - 1]);
|
||||
@@ -689,8 +882,8 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
|
||||
// For justified text, compute per-gap extra to distribute remaining space evenly
|
||||
const int spareSpace = effectivePageWidth - lineWordWidthSum - totalNaturalGaps;
|
||||
const int justifyExtra = (effectiveAlignment == CssTextAlign::Justify && !isLastLine && actualGapCount >= 1)
|
||||
? spareSpace / static_cast<int>(actualGapCount)
|
||||
const int justifyExtra = (effectiveAlignment == CssTextAlign::Justify && !isLastLine)
|
||||
? computeJustifyExtra(spareSpace, actualGapCount)
|
||||
: 0;
|
||||
|
||||
// BiDi processing: reorder words with UAX#9 in full-line context.
|
||||
@@ -709,11 +902,13 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
reorderedStylesScratch.clear();
|
||||
reorderedWidthsScratch.clear();
|
||||
reorderedContinuesScratch.clear();
|
||||
reorderedNoSpaceBeforeScratch.clear();
|
||||
reorderedFocusSuffixScratch.clear();
|
||||
reorderedWordsScratch.reserve(visualOrderScratch.size());
|
||||
reorderedStylesScratch.reserve(visualOrderScratch.size());
|
||||
reorderedWidthsScratch.reserve(visualOrderScratch.size());
|
||||
reorderedContinuesScratch.reserve(visualOrderScratch.size());
|
||||
reorderedNoSpaceBeforeScratch.reserve(visualOrderScratch.size());
|
||||
reorderedFocusSuffixScratch.reserve(visualOrderScratch.size());
|
||||
|
||||
for (size_t i = 0; i < visualOrderScratch.size(); ++i) {
|
||||
@@ -740,6 +935,7 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
}
|
||||
}
|
||||
reorderedContinuesScratch.push_back(continues);
|
||||
reorderedNoSpaceBeforeScratch.push_back(!continues && noSpaceBeforeVec[lastBreakAt + src]);
|
||||
}
|
||||
|
||||
int reorderedWordWidthSum = 0;
|
||||
@@ -747,7 +943,11 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
int reorderedNaturalGaps = 0;
|
||||
for (size_t wordIdx = 0; wordIdx < reorderedWidthsScratch.size(); wordIdx++) {
|
||||
reorderedWordWidthSum += reorderedWidthsScratch[wordIdx];
|
||||
if (wordIdx > 0 && !reorderedContinuesScratch[wordIdx]) {
|
||||
if (wordIdx > 0 && reorderedNoSpaceBeforeScratch[wordIdx]) {
|
||||
// Unicode break opportunity with no inserted Latin-style space. It is still
|
||||
// a stretchable gap for justified CJK/Korean text.
|
||||
reorderedGapCount++;
|
||||
} else if (wordIdx > 0 && !reorderedContinuesScratch[wordIdx]) {
|
||||
reorderedGapCount++;
|
||||
reorderedNaturalGaps += renderer.getSpaceAdvance(fontId, lastCodepoint(reorderedWordsScratch[wordIdx - 1]),
|
||||
firstCodepoint(reorderedWordsScratch[wordIdx]),
|
||||
@@ -763,10 +963,9 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
}
|
||||
|
||||
const int reorderedSpare = effectivePageWidth - reorderedWordWidthSum - reorderedNaturalGaps;
|
||||
const int reorderedJustifyExtra =
|
||||
(effectiveAlignment == CssTextAlign::Justify && !isLastLine && reorderedGapCount >= 1)
|
||||
? reorderedSpare / static_cast<int>(reorderedGapCount)
|
||||
: 0;
|
||||
const int reorderedJustifyExtra = (effectiveAlignment == CssTextAlign::Justify && !isLastLine)
|
||||
? computeJustifyExtra(reorderedSpare, reorderedGapCount)
|
||||
: 0;
|
||||
|
||||
const int justifyContribution = (effectiveAlignment == CssTextAlign::Justify && !isLastLine)
|
||||
? reorderedJustifyExtra * static_cast<int>(reorderedGapCount)
|
||||
@@ -805,9 +1004,11 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
}
|
||||
xpos += advance;
|
||||
} else if (wordIdx + 1 < reorderedWidthsScratch.size()) {
|
||||
int gap = renderer.getSpaceAdvance(fontId, lastCodepoint(reorderedWordsScratch[wordIdx]),
|
||||
firstCodepoint(reorderedWordsScratch[wordIdx + 1]),
|
||||
reorderedStylesScratch[wordIdx]);
|
||||
const bool nextNoSpace = reorderedNoSpaceBeforeScratch[wordIdx + 1];
|
||||
int gap = nextNoSpace ? 0
|
||||
: renderer.getSpaceAdvance(fontId, lastCodepoint(reorderedWordsScratch[wordIdx]),
|
||||
firstCodepoint(reorderedWordsScratch[wordIdx + 1]),
|
||||
reorderedStylesScratch[wordIdx]);
|
||||
if (effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
|
||||
gap += reorderedJustifyExtra;
|
||||
}
|
||||
@@ -846,11 +1047,15 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
xpos -= advance;
|
||||
} else {
|
||||
int gap = 0;
|
||||
bool nextNoSpace = false;
|
||||
if (wordIdx + 1 < lineWordCount) {
|
||||
gap = renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx]),
|
||||
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
|
||||
nextNoSpace = noSpaceBeforeVec[lastBreakAt + wordIdx + 1];
|
||||
gap = nextNoSpace
|
||||
? 0
|
||||
: renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx]),
|
||||
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
|
||||
}
|
||||
if (effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
|
||||
if (wordIdx + 1 < lineWordCount && effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
|
||||
gap += justifyExtra;
|
||||
}
|
||||
xpos -= gap;
|
||||
@@ -880,11 +1085,15 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
xpos += advance;
|
||||
} else {
|
||||
int gap = 0;
|
||||
bool nextNoSpace = false;
|
||||
if (wordIdx + 1 < lineWordCount) {
|
||||
gap = renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx]),
|
||||
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
|
||||
nextNoSpace = noSpaceBeforeVec[lastBreakAt + wordIdx + 1];
|
||||
gap = nextNoSpace
|
||||
? 0
|
||||
: renderer.getSpaceAdvance(fontId, lastCodepoint(lineWords[wordIdx]),
|
||||
firstCodepoint(lineWords[wordIdx + 1]), lineWordStyles[wordIdx]);
|
||||
}
|
||||
if (effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
|
||||
if (wordIdx + 1 < lineWordCount && effectiveAlignment == CssTextAlign::Justify && !isLastLine) {
|
||||
gap += justifyExtra;
|
||||
}
|
||||
xpos += wordWidths[lastBreakAt + wordIdx] + gap;
|
||||
|
||||
@@ -15,7 +15,8 @@ class GfxRenderer;
|
||||
class ParsedText {
|
||||
std::vector<std::string> words;
|
||||
std::vector<EpdFontFamily::Style> wordStyles;
|
||||
std::vector<bool> wordContinues; // true = word attaches to previous (no space before it)
|
||||
std::vector<bool> wordContinues; // true = word attaches to previous with no break
|
||||
std::vector<bool> wordNoSpaceBefore; // true = may break before token, but no synthetic space when joined
|
||||
std::vector<bool> wordIsFocusSuffix; // true = token is the regular tail of a focus bold-prefix split
|
||||
BlockStyle blockStyle;
|
||||
bool extraParagraphSpacing;
|
||||
@@ -27,18 +28,22 @@ class ParsedText {
|
||||
std::vector<EpdFontFamily::Style> reorderedStylesScratch;
|
||||
std::vector<uint16_t> reorderedWidthsScratch;
|
||||
std::vector<bool> reorderedContinuesScratch;
|
||||
std::vector<bool> reorderedNoSpaceBeforeScratch;
|
||||
std::vector<bool> reorderedFocusSuffixScratch;
|
||||
std::vector<uint16_t> visualOrderScratch;
|
||||
|
||||
int resolveFirstLineIndent(bool isFirstLine, const GfxRenderer& renderer, int fontId) const;
|
||||
std::vector<size_t> computeLineBreaks(const GfxRenderer& renderer, int fontId, int pageWidth,
|
||||
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec);
|
||||
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec,
|
||||
std::vector<bool>& noSpaceBeforeVec);
|
||||
std::vector<size_t> computeHyphenatedLineBreaks(const GfxRenderer& renderer, int fontId, int pageWidth,
|
||||
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec);
|
||||
std::vector<uint16_t>& wordWidths, std::vector<bool>& continuesVec,
|
||||
std::vector<bool>& noSpaceBeforeVec);
|
||||
bool hyphenateWordAtIndex(size_t wordIndex, int availableWidth, const GfxRenderer& renderer, int fontId,
|
||||
std::vector<uint16_t>& wordWidths, bool allowFallbackBreaks);
|
||||
void extractLine(size_t breakIndex, int pageWidth, const std::vector<uint16_t>& wordWidths,
|
||||
const std::vector<bool>& continuesVec, const std::vector<size_t>& lineBreakIndices,
|
||||
const std::vector<bool>& continuesVec, const std::vector<bool>& noSpaceBeforeVec,
|
||||
const std::vector<size_t>& lineBreakIndices,
|
||||
const std::function<void(std::shared_ptr<TextBlock>)>& processLine, const GfxRenderer& renderer,
|
||||
int fontId);
|
||||
std::vector<uint16_t> calculateWordWidths(const GfxRenderer& renderer, int fontId);
|
||||
|
||||
Reference in New Issue
Block a user