+106
-35
@@ -11,6 +11,7 @@
|
|||||||
#include <set>
|
#include <set>
|
||||||
#include <vector>
|
#include <vector>
|
||||||
|
|
||||||
|
#include "hyphenation/HyphenationCommon.h"
|
||||||
#include "hyphenation/Hyphenator.h"
|
#include "hyphenation/Hyphenator.h"
|
||||||
|
|
||||||
constexpr int MAX_COST = std::numeric_limits<int>::max();
|
constexpr int MAX_COST = std::numeric_limits<int>::max();
|
||||||
@@ -128,12 +129,66 @@ constexpr int kBionicReadingMinBoldPrefix = 1;
|
|||||||
constexpr int kBionicReadingBoldPrefixNumerator = 1;
|
constexpr int kBionicReadingBoldPrefixNumerator = 1;
|
||||||
constexpr int kBionicReadingBoldPrefixDenominator = 2;
|
constexpr int kBionicReadingBoldPrefixDenominator = 2;
|
||||||
|
|
||||||
|
struct TokenSpan {
|
||||||
|
size_t start;
|
||||||
|
size_t end;
|
||||||
|
bool isWord;
|
||||||
|
};
|
||||||
|
|
||||||
static int computeBionicBoldPrefixCount(const int codepointCount) {
|
static int computeBionicBoldPrefixCount(const int codepointCount) {
|
||||||
return std::max(kBionicReadingMinBoldPrefix,
|
return std::max(kBionicReadingMinBoldPrefix,
|
||||||
(codepointCount * kBionicReadingBoldPrefixNumerator + kBionicReadingBoldPrefixDenominator - 1) /
|
(codepointCount * kBionicReadingBoldPrefixNumerator + kBionicReadingBoldPrefixDenominator - 1) /
|
||||||
kBionicReadingBoldPrefixDenominator);
|
kBionicReadingBoldPrefixDenominator);
|
||||||
}
|
}
|
||||||
|
|
||||||
|
static bool isBionicWordCodepoint(const uint32_t cp) {
|
||||||
|
if (cp == 0) {
|
||||||
|
return false;
|
||||||
|
}
|
||||||
|
if (utf8IsCombiningMark(cp)) {
|
||||||
|
return true;
|
||||||
|
}
|
||||||
|
return isAlphabetic(cp) || isAsciiDigit(cp) || isApostrophe(cp);
|
||||||
|
}
|
||||||
|
|
||||||
|
// Split a word token into contiguous spans of "word-like" characters and non-word characters.
|
||||||
|
// This avoids applying bionic bolding to punctuation, digits-only runs, or other separators.
|
||||||
|
// Only spans marked as word-like are eligible for the bionic prefix transform.
|
||||||
|
static std::vector<TokenSpan> tokenizeBionicWord(const std::string& word) {
|
||||||
|
std::vector<TokenSpan> spans;
|
||||||
|
spans.reserve(2);
|
||||||
|
|
||||||
|
const unsigned char* base = reinterpret_cast<const unsigned char*>(word.c_str());
|
||||||
|
const unsigned char* ptr = base;
|
||||||
|
const unsigned char* segmentStart = ptr;
|
||||||
|
bool currentIsWord = false;
|
||||||
|
bool haveCurrent = false;
|
||||||
|
|
||||||
|
while (true) {
|
||||||
|
const unsigned char* cpStart = ptr;
|
||||||
|
uint32_t cp = utf8NextCodepoint(&ptr);
|
||||||
|
if (cp == 0) {
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
|
||||||
|
bool cpIsWord = isBionicWordCodepoint(cp);
|
||||||
|
if (!haveCurrent) {
|
||||||
|
currentIsWord = cpIsWord;
|
||||||
|
haveCurrent = true;
|
||||||
|
} else if (!utf8IsCombiningMark(cp) && cpIsWord != currentIsWord) {
|
||||||
|
spans.push_back({static_cast<size_t>(segmentStart - base), static_cast<size_t>(cpStart - base), currentIsWord});
|
||||||
|
segmentStart = cpStart;
|
||||||
|
currentIsWord = cpIsWord;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
if (haveCurrent) {
|
||||||
|
spans.push_back({static_cast<size_t>(segmentStart - base), word.size(), currentIsWord});
|
||||||
|
}
|
||||||
|
|
||||||
|
return spans;
|
||||||
|
}
|
||||||
|
|
||||||
} // namespace
|
} // namespace
|
||||||
|
|
||||||
void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle, const bool underline,
|
void ParsedText::addWord(std::string word, const EpdFontFamily::Style fontStyle, const bool underline,
|
||||||
@@ -533,49 +588,65 @@ void ParsedText::applyBionicReadingTransform() {
|
|||||||
transformedContinues.reserve(wordContinues.size() * 2);
|
transformedContinues.reserve(wordContinues.size() * 2);
|
||||||
|
|
||||||
for (size_t i = 0; i < words.size(); ++i) {
|
for (size_t i = 0; i < words.size(); ++i) {
|
||||||
const std::string& word = words[i];
|
std::string source = std::move(words[i]);
|
||||||
const auto originalStyle = wordStyles[i];
|
const auto originalStyle = wordStyles[i];
|
||||||
const bool attachToPrevious = wordContinues[i];
|
const bool originalAttachToPrevious = wordContinues[i];
|
||||||
|
const char* raw = source.c_str();
|
||||||
|
|
||||||
const unsigned char* ptr = reinterpret_cast<const unsigned char*>(word.c_str());
|
const auto spans = tokenizeBionicWord(source);
|
||||||
int codepointCount = 0;
|
if (spans.empty()) {
|
||||||
while (utf8NextCodepoint(&ptr)) {
|
|
||||||
codepointCount++;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (codepointCount < kBionicReadingMinCodepoints) {
|
|
||||||
transformedWords.push_back(word);
|
|
||||||
transformedStyles.push_back(originalStyle);
|
|
||||||
transformedContinues.push_back(attachToPrevious);
|
|
||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
const int boldPrefixCount = computeBionicBoldPrefixCount(codepointCount);
|
bool attachToPrevious = originalAttachToPrevious;
|
||||||
ptr = reinterpret_cast<const unsigned char*>(word.c_str());
|
for (size_t spanIndex = 0; spanIndex < spans.size(); ++spanIndex) {
|
||||||
const unsigned char* prefixEnd = ptr;
|
const TokenSpan span = spans[spanIndex];
|
||||||
for (int j = 0; j < boldPrefixCount && *prefixEnd; ++j) {
|
const size_t spanLength = span.end - span.start;
|
||||||
utf8NextCodepoint(&prefixEnd);
|
std::string token;
|
||||||
}
|
if (spans.size() == 1 && spanIndex == 0) {
|
||||||
const size_t prefixByteCount =
|
token = std::move(source);
|
||||||
static_cast<size_t>(prefixEnd - reinterpret_cast<const unsigned char*>(word.c_str()));
|
} else {
|
||||||
if (prefixByteCount >= word.size()) {
|
token.assign(raw + span.start, spanLength);
|
||||||
transformedWords.push_back(word);
|
}
|
||||||
|
|
||||||
|
if (span.isWord) {
|
||||||
|
const unsigned char* ptr = reinterpret_cast<const unsigned char*>(token.c_str());
|
||||||
|
int codepointCount = 0;
|
||||||
|
while (utf8NextCodepoint(&ptr)) {
|
||||||
|
codepointCount++;
|
||||||
|
}
|
||||||
|
|
||||||
|
if (codepointCount >= kBionicReadingMinCodepoints) {
|
||||||
|
const int boldPrefixCount = computeBionicBoldPrefixCount(codepointCount);
|
||||||
|
ptr = reinterpret_cast<const unsigned char*>(token.c_str());
|
||||||
|
const unsigned char* prefixEnd = ptr;
|
||||||
|
for (int j = 0; j < boldPrefixCount && *prefixEnd; ++j) {
|
||||||
|
utf8NextCodepoint(&prefixEnd);
|
||||||
|
}
|
||||||
|
const size_t prefixByteCount =
|
||||||
|
static_cast<size_t>(prefixEnd - reinterpret_cast<const unsigned char*>(token.c_str()));
|
||||||
|
if (prefixByteCount < token.size()) {
|
||||||
|
std::string suffix(reinterpret_cast<const char*>(prefixEnd), token.size() - prefixByteCount);
|
||||||
|
token.resize(prefixByteCount);
|
||||||
|
const auto boldStyle = static_cast<EpdFontFamily::Style>(originalStyle | EpdFontFamily::BOLD);
|
||||||
|
transformedWords.push_back(std::move(token));
|
||||||
|
transformedStyles.push_back(boldStyle);
|
||||||
|
transformedContinues.push_back(attachToPrevious);
|
||||||
|
|
||||||
|
transformedWords.push_back(std::move(suffix));
|
||||||
|
transformedStyles.push_back(originalStyle);
|
||||||
|
transformedContinues.push_back(true);
|
||||||
|
attachToPrevious = true;
|
||||||
|
continue;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
|
transformedWords.push_back(std::move(token));
|
||||||
transformedStyles.push_back(originalStyle);
|
transformedStyles.push_back(originalStyle);
|
||||||
transformedContinues.push_back(attachToPrevious);
|
transformedContinues.push_back(attachToPrevious);
|
||||||
continue;
|
attachToPrevious = true;
|
||||||
}
|
}
|
||||||
|
|
||||||
const std::string prefix(word.data(), prefixByteCount);
|
|
||||||
const std::string suffix(word.data() + prefixByteCount, word.size() - prefixByteCount);
|
|
||||||
const auto boldStyle = static_cast<EpdFontFamily::Style>(originalStyle | EpdFontFamily::BOLD);
|
|
||||||
|
|
||||||
transformedWords.push_back(prefix);
|
|
||||||
transformedStyles.push_back(boldStyle);
|
|
||||||
transformedContinues.push_back(attachToPrevious);
|
|
||||||
|
|
||||||
transformedWords.push_back(suffix);
|
|
||||||
transformedStyles.push_back(originalStyle);
|
|
||||||
transformedContinues.push_back(true);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
words = std::move(transformedWords);
|
words = std::move(transformedWords);
|
||||||
|
|||||||
Reference in New Issue
Block a user