feat: Support for kerning and ligatures (#873)
## Summary **What is the goal of this PR?** Improved typesetting, including [kerning](https://en.wikipedia.org/wiki/Kerning) and [ligatures](https://en.wikipedia.org/wiki/Ligature_(writing)#Latin_alphabet). **What changes are included?** - The script to convert built-in fonts now adds kerning and ligature information to the generated font headers. - Epub page layout calculates proper kerning spaces and makes ligature substitutions according to the selected font.    ## Additional Context - I am not a typography expert. - The implementation has been reworked from the earlier version, so it is no longer necessary to omit Open Dyslexic, and kerning data now covers all fonts, styles, and codepoints for which we include bitmap data. - Claude Opus 4.6 helped with a lot of this. - There's an included test epub document with lots of kerning and ligature examples, shown in the photos. **_After some time to mature, I think this change is in decent shape to merge and get people testing._** After opening this PR I came across #660, which overlaps in adding ligature support. --- ### AI Usage While CrossPoint doesn't have restrictions on AI tools in contributing, please be transparent about their usage as it helps set the right context for reviewers. Did you use AI tools to help write this code? _**YES, Claude Opus 4.6**_ --------- Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
+92
-5
@@ -20,26 +20,36 @@ void EpdFont::getTextBounds(const char* string, const int startX, const int star
|
||||
int lastBaseX = startX;
|
||||
int lastBaseAdvance = 0;
|
||||
int lastBaseTop = 0;
|
||||
bool hasBaseGlyph = false;
|
||||
constexpr int MIN_COMBINING_GAP_PX = 1;
|
||||
uint32_t cp;
|
||||
uint32_t prevCp = 0;
|
||||
while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&string)))) {
|
||||
const bool isCombining = utf8IsCombiningMark(cp);
|
||||
|
||||
if (!isCombining) {
|
||||
cp = applyLigatures(cp, string);
|
||||
}
|
||||
|
||||
const EpdGlyph* glyph = getGlyph(cp);
|
||||
if (!glyph) {
|
||||
// TODO: Better handle this?
|
||||
prevCp = 0;
|
||||
continue;
|
||||
}
|
||||
|
||||
const bool isCombining = utf8IsCombiningMark(cp);
|
||||
int raiseBy = 0;
|
||||
if (isCombining && hasBaseGlyph) {
|
||||
if (isCombining) {
|
||||
const int currentGap = glyph->top - glyph->height - lastBaseTop;
|
||||
if (currentGap < MIN_COMBINING_GAP_PX) {
|
||||
raiseBy = MIN_COMBINING_GAP_PX - currentGap;
|
||||
}
|
||||
}
|
||||
|
||||
const int glyphBaseX = (isCombining && hasBaseGlyph) ? (lastBaseX + lastBaseAdvance / 2) : cursorX;
|
||||
if (!isCombining && prevCp != 0) {
|
||||
cursorX += getKerning(prevCp, cp);
|
||||
}
|
||||
|
||||
const int glyphBaseX = isCombining ? (lastBaseX + lastBaseAdvance / 2) : cursorX;
|
||||
const int glyphBaseY = cursorY - raiseBy;
|
||||
|
||||
*minX = std::min(*minX, glyphBaseX + glyph->left);
|
||||
@@ -51,8 +61,8 @@ void EpdFont::getTextBounds(const char* string, const int startX, const int star
|
||||
lastBaseX = cursorX;
|
||||
lastBaseAdvance = glyph->advanceX;
|
||||
lastBaseTop = glyph->top;
|
||||
hasBaseGlyph = true;
|
||||
cursorX += glyph->advanceX;
|
||||
prevCp = cp;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -66,6 +76,83 @@ void EpdFont::getTextDimensions(const char* string, int* w, int* h) const {
|
||||
*h = maxY - minY;
|
||||
}
|
||||
|
||||
static uint8_t lookupKernClass(const EpdKernClassEntry* entries, const uint16_t count, const uint32_t cp) {
|
||||
if (!entries || count == 0 || cp > 0xFFFF) {
|
||||
return 0;
|
||||
}
|
||||
const auto target = static_cast<uint16_t>(cp);
|
||||
int left = 0;
|
||||
int right = static_cast<int>(count) - 1;
|
||||
while (left <= right) {
|
||||
const int mid = left + (right - left) / 2;
|
||||
const uint16_t midCp = entries[mid].codepoint;
|
||||
if (midCp == target) {
|
||||
return entries[mid].classId;
|
||||
}
|
||||
if (midCp < target) {
|
||||
left = mid + 1;
|
||||
} else {
|
||||
right = mid - 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int8_t EpdFont::getKerning(const uint32_t leftCp, const uint32_t rightCp) const {
|
||||
if (!data->kernMatrix) {
|
||||
return 0;
|
||||
}
|
||||
const uint8_t lc = lookupKernClass(data->kernLeftClasses, data->kernLeftEntryCount, leftCp);
|
||||
if (lc == 0) return 0;
|
||||
const uint8_t rc = lookupKernClass(data->kernRightClasses, data->kernRightEntryCount, rightCp);
|
||||
if (rc == 0) return 0;
|
||||
return data->kernMatrix[(lc - 1) * data->kernRightClassCount + (rc - 1)];
|
||||
}
|
||||
|
||||
uint32_t EpdFont::getLigature(const uint32_t leftCp, const uint32_t rightCp) const {
|
||||
const auto* pairs = data->ligaturePairs;
|
||||
const auto count = data->ligaturePairCount;
|
||||
if (!pairs || count == 0 || leftCp > 0xFFFF || rightCp > 0xFFFF) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
const uint32_t key = (leftCp << 16) | rightCp;
|
||||
int left = 0;
|
||||
int right = static_cast<int>(count) - 1;
|
||||
|
||||
while (left <= right) {
|
||||
const int mid = left + (right - left) / 2;
|
||||
const uint32_t midKey = pairs[mid].pair;
|
||||
if (midKey == key) {
|
||||
return pairs[mid].ligatureCp;
|
||||
}
|
||||
if (midKey < key) {
|
||||
left = mid + 1;
|
||||
} else {
|
||||
right = mid - 1;
|
||||
}
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
uint32_t EpdFont::applyLigatures(uint32_t cp, const char*& text) const {
|
||||
if (!data->ligaturePairs || data->ligaturePairCount == 0) {
|
||||
return cp;
|
||||
}
|
||||
while (true) {
|
||||
const auto saved = reinterpret_cast<const uint8_t*>(text);
|
||||
const uint32_t nextCp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text));
|
||||
if (nextCp == 0) break;
|
||||
const uint32_t lig = getLigature(cp, nextCp);
|
||||
if (lig == 0) {
|
||||
text = reinterpret_cast<const char*>(saved);
|
||||
break;
|
||||
}
|
||||
cp = lig;
|
||||
}
|
||||
return cp;
|
||||
}
|
||||
|
||||
const EpdGlyph* EpdFont::getGlyph(const uint32_t cp) const {
|
||||
const EpdUnicodeInterval* intervals = data->intervals;
|
||||
const int count = data->intervalCount;
|
||||
|
||||
@@ -11,4 +11,16 @@ class EpdFont {
|
||||
void getTextDimensions(const char* string, int* w, int* h) const;
|
||||
|
||||
const EpdGlyph* getGlyph(uint32_t cp) const;
|
||||
|
||||
/// Returns the kerning adjustment (in pixels) between two codepoints.
|
||||
/// Returns 0 if no kerning data exists for the pair.
|
||||
int8_t getKerning(uint32_t leftCp, uint32_t rightCp) const;
|
||||
|
||||
/// Returns the ligature codepoint for a pair, or 0 if no ligature exists.
|
||||
uint32_t getLigature(uint32_t leftCp, uint32_t rightCp) const;
|
||||
|
||||
/// Greedily applies ligature substitutions starting from cp, consuming
|
||||
/// as many following codepoints from text as possible. Returns the
|
||||
/// (possibly substituted) codepoint; advances text past consumed chars.
|
||||
uint32_t applyLigatures(uint32_t cp, const char*& text) const;
|
||||
};
|
||||
|
||||
@@ -31,6 +31,20 @@ typedef struct {
|
||||
uint32_t offset; ///< Index of the first code point into the glyph array
|
||||
} EpdUnicodeInterval;
|
||||
|
||||
/// Maps a codepoint to a kerning class ID, sorted by codepoint for binary search.
|
||||
/// Class IDs are 1-based; codepoints not in the table have implicit class 0 (no kerning).
|
||||
typedef struct {
|
||||
uint16_t codepoint; ///< Unicode codepoint
|
||||
uint8_t classId; ///< 1-based kerning class ID
|
||||
} __attribute__((packed)) EpdKernClassEntry;
|
||||
|
||||
/// Ligature substitution for a specific glyph pair, sorted by `pair` for binary search.
|
||||
/// `pair` encodes (leftCodepoint << 16 | rightCodepoint) for single-key lookup.
|
||||
typedef struct {
|
||||
uint32_t pair; ///< Packed codepoint pair (left << 16 | right)
|
||||
uint32_t ligatureCp; ///< Codepoint of the replacement ligature glyph
|
||||
} __attribute__((packed)) EpdLigaturePair;
|
||||
|
||||
/// Data stored for FONT AS A WHOLE
|
||||
typedef struct {
|
||||
const uint8_t* bitmap; ///< Glyph bitmaps, concatenated
|
||||
@@ -43,4 +57,13 @@ typedef struct {
|
||||
bool is2Bit;
|
||||
const EpdFontGroup* groups; ///< NULL for uncompressed fonts
|
||||
uint16_t groupCount; ///< 0 for uncompressed fonts
|
||||
const EpdKernClassEntry* kernLeftClasses; ///< Sorted left-side class map (nullptr if none)
|
||||
const EpdKernClassEntry* kernRightClasses; ///< Sorted right-side class map (nullptr if none)
|
||||
const int8_t* kernMatrix; ///< Flat leftClassCount x rightClassCount matrix
|
||||
uint16_t kernLeftEntryCount; ///< Entries in kernLeftClasses
|
||||
uint16_t kernRightEntryCount; ///< Entries in kernRightClasses
|
||||
uint8_t kernLeftClassCount; ///< Number of distinct left classes (matrix rows)
|
||||
uint8_t kernRightClassCount; ///< Number of distinct right classes (matrix cols)
|
||||
const EpdLigaturePair* ligaturePairs; ///< Sorted ligature pair table (nullptr if none)
|
||||
uint32_t ligaturePairCount; ///< Number of entries in ligaturePairs
|
||||
} EpdFontData;
|
||||
|
||||
@@ -26,4 +26,12 @@ const EpdFontData* EpdFontFamily::getData(const Style style) const { return getF
|
||||
|
||||
const EpdGlyph* EpdFontFamily::getGlyph(const uint32_t cp, const Style style) const {
|
||||
return getFont(style)->getGlyph(cp);
|
||||
};
|
||||
}
|
||||
|
||||
int8_t EpdFontFamily::getKerning(const uint32_t leftCp, const uint32_t rightCp, const Style style) const {
|
||||
return getFont(style)->getKerning(leftCp, rightCp);
|
||||
}
|
||||
|
||||
uint32_t EpdFontFamily::applyLigatures(const uint32_t cp, const char*& text, const Style style) const {
|
||||
return getFont(style)->applyLigatures(cp, text);
|
||||
}
|
||||
|
||||
@@ -12,6 +12,8 @@ class EpdFontFamily {
|
||||
void getTextDimensions(const char* string, int* w, int* h, Style style = REGULAR) const;
|
||||
const EpdFontData* getData(Style style = REGULAR) const;
|
||||
const EpdGlyph* getGlyph(uint32_t cp, Style style = REGULAR) const;
|
||||
int8_t getKerning(uint32_t leftCp, uint32_t rightCp, Style style = REGULAR) const;
|
||||
uint32_t applyLigatures(uint32_t cp, const char*& text, Style style = REGULAR) const;
|
||||
|
||||
private:
|
||||
const EpdFont* regular;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -6,6 +6,7 @@ import re
|
||||
import math
|
||||
import argparse
|
||||
from collections import namedtuple
|
||||
from fontTools.ttLib import TTFont
|
||||
|
||||
# Originally from https://github.com/vroland/epdiy
|
||||
|
||||
@@ -106,6 +107,9 @@ intervals = [
|
||||
# (0xFE30, 0xFE4F),
|
||||
# # CJK Compatibility Ideographs
|
||||
# (0xF900, 0xFAFF),
|
||||
### Alphabetic Presentation Forms (Latin ligatures) ###
|
||||
# ff, fi, fl, ffi, ffl, long-st, st
|
||||
(0xFB00, 0xFB06),
|
||||
### Specials
|
||||
# Replacement Character
|
||||
(0xFFFD, 0xFFFD),
|
||||
@@ -134,7 +138,6 @@ def load_glyph(code_point):
|
||||
face.load_glyph(glyph_index, load_flags)
|
||||
return face
|
||||
face_index += 1
|
||||
print(f"code point {code_point} ({hex(code_point)}) not found in font stack!", file=sys.stderr)
|
||||
return None
|
||||
|
||||
unmerged_intervals = sorted(intervals + add_ints)
|
||||
@@ -275,6 +278,375 @@ for index, glyph in enumerate(all_glyphs):
|
||||
glyph_data.extend([b for b in packed])
|
||||
glyph_props.append(props)
|
||||
|
||||
# --- Kerning pair extraction ---
|
||||
# Modern fonts store kerning in the OpenType GPOS table, which FreeType's
|
||||
# get_kerning() does not read. We use fonttools to parse both the legacy
|
||||
# kern table and the GPOS 'kern' feature (PairPos lookups, including
|
||||
# Extension wrappers).
|
||||
|
||||
COMBINING_MARKS_START = 0x0300
|
||||
COMBINING_MARKS_END = 0x036F
|
||||
all_codepoints = [g.code_point for g in glyph_props]
|
||||
kernable_codepoints = set(cp for cp in all_codepoints
|
||||
if not (COMBINING_MARKS_START <= cp <= COMBINING_MARKS_END))
|
||||
|
||||
# Map each kernable codepoint to the font-stack index that serves it
|
||||
# (same priority logic as load_glyph).
|
||||
cp_to_face_idx = {}
|
||||
for cp in kernable_codepoints:
|
||||
for face_idx, f in enumerate(font_stack):
|
||||
if f.get_char_index(cp) > 0:
|
||||
cp_to_face_idx[cp] = face_idx
|
||||
break
|
||||
|
||||
# Group codepoints by face index
|
||||
face_idx_cps = {}
|
||||
for cp, fi in cp_to_face_idx.items():
|
||||
face_idx_cps.setdefault(fi, set()).add(cp)
|
||||
|
||||
def _extract_pairpos_subtable(subtable, glyph_to_cp, raw_kern):
|
||||
"""Extract kerning from a PairPos subtable (Format 1 or 2)."""
|
||||
if subtable.Format == 1:
|
||||
# Individual pairs
|
||||
for i, coverage_glyph in enumerate(subtable.Coverage.glyphs):
|
||||
if coverage_glyph not in glyph_to_cp:
|
||||
continue
|
||||
pair_set = subtable.PairSet[i]
|
||||
for pvr in pair_set.PairValueRecord:
|
||||
if pvr.SecondGlyph not in glyph_to_cp:
|
||||
continue
|
||||
xa = 0
|
||||
if hasattr(pvr, 'Value1') and pvr.Value1:
|
||||
xa = getattr(pvr.Value1, 'XAdvance', 0) or 0
|
||||
if xa != 0:
|
||||
key = (coverage_glyph, pvr.SecondGlyph)
|
||||
raw_kern[key] = raw_kern.get(key, 0) + xa
|
||||
elif subtable.Format == 2:
|
||||
# Class-based pairs
|
||||
class_def1 = subtable.ClassDef1.classDefs if subtable.ClassDef1 else {}
|
||||
class_def2 = subtable.ClassDef2.classDefs if subtable.ClassDef2 else {}
|
||||
coverage_set = set(subtable.Coverage.glyphs)
|
||||
for left_glyph in glyph_to_cp:
|
||||
if left_glyph not in coverage_set:
|
||||
continue
|
||||
c1 = class_def1.get(left_glyph, 0)
|
||||
if c1 >= len(subtable.Class1Record):
|
||||
continue
|
||||
class1_rec = subtable.Class1Record[c1]
|
||||
for right_glyph in glyph_to_cp:
|
||||
c2 = class_def2.get(right_glyph, 0)
|
||||
if c2 >= len(class1_rec.Class2Record):
|
||||
continue
|
||||
c2_rec = class1_rec.Class2Record[c2]
|
||||
xa = 0
|
||||
if hasattr(c2_rec, 'Value1') and c2_rec.Value1:
|
||||
xa = getattr(c2_rec.Value1, 'XAdvance', 0) or 0
|
||||
if xa != 0:
|
||||
key = (left_glyph, right_glyph)
|
||||
raw_kern[key] = raw_kern.get(key, 0) + xa
|
||||
|
||||
def extract_kerning_fonttools(font_path, codepoints, ppem):
|
||||
"""Extract kerning pairs from a font file using fonttools.
|
||||
|
||||
Returns dict of {(leftCp, rightCp): pixel_adjust} for the given
|
||||
codepoints. Values are scaled from font design units to integer
|
||||
pixels at ppem.
|
||||
"""
|
||||
font = TTFont(font_path)
|
||||
units_per_em = font['head'].unitsPerEm
|
||||
cmap = font.getBestCmap() or {}
|
||||
|
||||
# Build glyph_name -> codepoint map (only for requested codepoints)
|
||||
glyph_to_cp = {}
|
||||
for cp in codepoints:
|
||||
gname = cmap.get(cp)
|
||||
if gname:
|
||||
glyph_to_cp[gname] = cp
|
||||
|
||||
# Collect raw kerning values in font design units
|
||||
raw_kern = {} # (left_glyph_name, right_glyph_name) -> design_units
|
||||
|
||||
# 1. Legacy kern table
|
||||
if 'kern' in font:
|
||||
for subtable in font['kern'].kernTables:
|
||||
if hasattr(subtable, 'kernTable'):
|
||||
for (lg, rg), val in subtable.kernTable.items():
|
||||
if lg in glyph_to_cp and rg in glyph_to_cp:
|
||||
raw_kern[(lg, rg)] = raw_kern.get((lg, rg), 0) + val
|
||||
|
||||
# 2. GPOS 'kern' feature
|
||||
if 'GPOS' in font:
|
||||
gpos = font['GPOS'].table
|
||||
kern_lookup_indices = set()
|
||||
if gpos.FeatureList:
|
||||
for fr in gpos.FeatureList.FeatureRecord:
|
||||
if fr.FeatureTag == 'kern':
|
||||
kern_lookup_indices.update(fr.Feature.LookupListIndex)
|
||||
for li in kern_lookup_indices:
|
||||
lookup = gpos.LookupList.Lookup[li]
|
||||
for st in lookup.SubTable:
|
||||
actual = st
|
||||
# Unwrap Extension (lookup type 9) wrappers
|
||||
if lookup.LookupType == 9 and hasattr(st, 'ExtSubTable'):
|
||||
actual = st.ExtSubTable
|
||||
if hasattr(actual, 'Format'):
|
||||
_extract_pairpos_subtable(actual, glyph_to_cp, raw_kern)
|
||||
|
||||
font.close()
|
||||
|
||||
# Scale design-unit values to pixels
|
||||
scale = ppem / units_per_em
|
||||
result = {} # (leftCp, rightCp) -> adjust
|
||||
for (lg, rg), du in raw_kern.items():
|
||||
lcp = glyph_to_cp[lg]
|
||||
rcp = glyph_to_cp[rg]
|
||||
adjust = int(math.floor(du * scale))
|
||||
if adjust != 0:
|
||||
adjust = max(-128, min(127, adjust))
|
||||
result[(lcp, rcp)] = adjust
|
||||
return result
|
||||
|
||||
# The ppem used by the existing glyph rasterization:
|
||||
# face.set_char_size(size << 6, size << 6, 150, 150)
|
||||
# means size_pt at 150 DPI -> ppem = size * 150 / 72
|
||||
ppem = size * 150.0 / 72.0
|
||||
|
||||
kern_map = {} # (leftCp, rightCp) -> adjust
|
||||
for face_idx, cps in face_idx_cps.items():
|
||||
font_path = args.fontstack[face_idx]
|
||||
kern_map.update(extract_kerning_fonttools(font_path, cps, ppem))
|
||||
|
||||
print(f"kerning: {len(kern_map)} pairs extracted", file=sys.stderr)
|
||||
|
||||
# --- Derive class-based kerning from pairs ---
|
||||
kern_left_classes = [] # list of (codepoint, classId)
|
||||
kern_right_classes = [] # list of (codepoint, classId)
|
||||
kern_matrix = [] # flat list of int8_t values
|
||||
kern_left_class_count = 0
|
||||
kern_right_class_count = 0
|
||||
|
||||
if kern_map:
|
||||
all_left_cps = {lcp for lcp, _ in kern_map}
|
||||
all_right_cps = {rcp for _, rcp in kern_map}
|
||||
|
||||
sorted_right_cps = sorted(all_right_cps)
|
||||
sorted_left_cps = sorted(all_left_cps)
|
||||
|
||||
# Group left codepoints by identical adjustment row
|
||||
left_profile_to_class = {}
|
||||
left_class_map = {}
|
||||
left_class_id = 1
|
||||
for lcp in sorted(all_left_cps):
|
||||
row = tuple(kern_map.get((lcp, rcp), 0) for rcp in sorted_right_cps)
|
||||
if row not in left_profile_to_class:
|
||||
left_profile_to_class[row] = left_class_id
|
||||
left_class_id += 1
|
||||
left_class_map[lcp] = left_profile_to_class[row]
|
||||
|
||||
# Group right codepoints by identical adjustment column
|
||||
right_profile_to_class = {}
|
||||
right_class_map = {}
|
||||
right_class_id = 1
|
||||
for rcp in sorted(all_right_cps):
|
||||
col = tuple(kern_map.get((lcp, rcp), 0) for lcp in sorted_left_cps)
|
||||
if col not in right_profile_to_class:
|
||||
right_profile_to_class[col] = right_class_id
|
||||
right_class_id += 1
|
||||
right_class_map[rcp] = right_profile_to_class[col]
|
||||
|
||||
kern_left_class_count = left_class_id - 1
|
||||
kern_right_class_count = right_class_id - 1
|
||||
|
||||
if kern_left_class_count > 255 or kern_right_class_count > 255:
|
||||
print(f"WARNING: kerning class count exceeds uint8_t range "
|
||||
f"(left={kern_left_class_count}, right={kern_right_class_count})",
|
||||
file=sys.stderr)
|
||||
|
||||
# Build the class x class matrix
|
||||
kern_matrix = [0] * (kern_left_class_count * kern_right_class_count)
|
||||
for (lcp, rcp), adjust in kern_map.items():
|
||||
lc = left_class_map[lcp] - 1
|
||||
rc = right_class_map[rcp] - 1
|
||||
kern_matrix[lc * kern_right_class_count + rc] = adjust
|
||||
|
||||
# Build sorted class entry lists
|
||||
kern_left_classes = sorted(left_class_map.items())
|
||||
kern_right_classes = sorted(right_class_map.items())
|
||||
|
||||
matrix_size = kern_left_class_count * kern_right_class_count
|
||||
entries_size = (len(kern_left_classes) + len(kern_right_classes)) * 3
|
||||
print(f"kerning: {kern_left_class_count} left classes, {kern_right_class_count} right classes, "
|
||||
f"{matrix_size + entries_size} bytes", file=sys.stderr)
|
||||
|
||||
# --- Ligature pair extraction ---
|
||||
# Parse the OpenType GSUB table for LigatureSubst (type 4) lookups.
|
||||
# Multi-character ligatures (3+ codepoints) are decomposed into chained
|
||||
# pairs when an intermediate ligature exists (e.g., ffi = ff + i where ff
|
||||
# is itself a ligature). Only pairs where both input codepoints and the
|
||||
# output codepoint are in the generated glyph set are included.
|
||||
|
||||
all_codepoints_set = set(all_codepoints)
|
||||
|
||||
# Standard Unicode ligature codepoints for known input sequences.
|
||||
# Used as a fallback when the GSUB substitute glyph has no cmap entry.
|
||||
STANDARD_LIGATURE_MAP = {
|
||||
(0x66, 0x66): 0xFB00, # ff
|
||||
(0x66, 0x69): 0xFB01, # fi
|
||||
(0x66, 0x6C): 0xFB02, # fl
|
||||
(0x66, 0x66, 0x69): 0xFB03, # ffi
|
||||
(0x66, 0x66, 0x6C): 0xFB04, # ffl
|
||||
(0x17F, 0x74): 0xFB05, # long-s + t
|
||||
(0x73, 0x74): 0xFB06, # st
|
||||
}
|
||||
|
||||
def extract_ligatures_fonttools(font_path, codepoints):
|
||||
"""Extract ligature substitution pairs from a font file using fonttools.
|
||||
|
||||
Returns list of (packed_pair, ligature_codepoint) for the given codepoints.
|
||||
Multi-character ligatures are decomposed into chained pairs.
|
||||
"""
|
||||
font = TTFont(font_path)
|
||||
cmap = font.getBestCmap() or {}
|
||||
|
||||
# Build glyph_name -> codepoint and codepoint -> glyph_name maps
|
||||
glyph_to_cp = {}
|
||||
cp_to_glyph = {}
|
||||
for cp, gname in cmap.items():
|
||||
glyph_to_cp[gname] = cp
|
||||
cp_to_glyph[cp] = gname
|
||||
|
||||
# Collect raw ligature rules: (sequence_of_codepoints) -> ligature_codepoint
|
||||
raw_ligatures = {} # tuple of codepoints -> ligature codepoint
|
||||
|
||||
if 'GSUB' in font:
|
||||
gsub = font['GSUB'].table
|
||||
|
||||
# Find lookup indices for ligature features.
|
||||
# Currently extracts 'liga' (standard) and 'rlig' (required) only.
|
||||
# To also extract discretionary or historical ligatures, add:
|
||||
# 'dlig' - Discretionary Ligatures (e.g., ft, st in Bookerly)
|
||||
# 'hlig' - Historical Ligatures (e.g., long-s+t in OpenDyslexic)
|
||||
# These are off by default in standard text renderers.
|
||||
LIGATURE_FEATURES = ('liga', 'rlig')
|
||||
liga_lookup_indices = set()
|
||||
if gsub.FeatureList:
|
||||
for fr in gsub.FeatureList.FeatureRecord:
|
||||
if fr.FeatureTag in LIGATURE_FEATURES:
|
||||
liga_lookup_indices.update(fr.Feature.LookupListIndex)
|
||||
|
||||
for li in liga_lookup_indices:
|
||||
lookup = gsub.LookupList.Lookup[li]
|
||||
for st in lookup.SubTable:
|
||||
actual = st
|
||||
# Unwrap Extension (lookup type 7) wrappers
|
||||
if lookup.LookupType == 7 and hasattr(st, 'ExtSubTable'):
|
||||
actual = st.ExtSubTable
|
||||
# LigatureSubst is lookup type 4
|
||||
if not hasattr(actual, 'ligatures'):
|
||||
continue
|
||||
for first_glyph, ligature_list in actual.ligatures.items():
|
||||
if first_glyph not in glyph_to_cp:
|
||||
continue
|
||||
first_cp = glyph_to_cp[first_glyph]
|
||||
for lig in ligature_list:
|
||||
# lig.Component is a list of subsequent glyph names
|
||||
# lig.LigGlyph is the substitute glyph name
|
||||
component_cps = []
|
||||
valid = True
|
||||
for comp_glyph in lig.Component:
|
||||
if comp_glyph not in glyph_to_cp:
|
||||
valid = False
|
||||
break
|
||||
component_cps.append(glyph_to_cp[comp_glyph])
|
||||
if not valid:
|
||||
continue
|
||||
seq = tuple([first_cp] + component_cps)
|
||||
if lig.LigGlyph in glyph_to_cp:
|
||||
lig_cp = glyph_to_cp[lig.LigGlyph]
|
||||
elif seq in STANDARD_LIGATURE_MAP:
|
||||
lig_cp = STANDARD_LIGATURE_MAP[seq]
|
||||
else:
|
||||
seq_str = ', '.join(f'U+{cp:04X}' for cp in seq)
|
||||
print(f"ligatures: WARNING: dropping ligature ({seq_str}) -> "
|
||||
f"glyph '{lig.LigGlyph}': output glyph has no cmap entry "
|
||||
f"and input sequence is not in STANDARD_LIGATURE_MAP",
|
||||
file=sys.stderr)
|
||||
continue
|
||||
raw_ligatures[seq] = lig_cp
|
||||
|
||||
font.close()
|
||||
|
||||
# Filter: only keep ligatures where all input and output codepoints are
|
||||
# in our generated glyph set
|
||||
filtered = {}
|
||||
for seq, lig_cp in raw_ligatures.items():
|
||||
if lig_cp not in codepoints and lig_cp not in all_codepoints_set:
|
||||
continue
|
||||
if all(cp in codepoints for cp in seq):
|
||||
filtered[seq] = lig_cp
|
||||
|
||||
# Decompose into chained pairs
|
||||
# For 2-codepoint sequences: direct pair (a, b) -> lig
|
||||
# For 3+ codepoint sequences: chain through intermediates
|
||||
# e.g., (f, f, i) -> ffi requires (f, f) -> ff to exist,
|
||||
# then we add (ff, i) -> ffi
|
||||
pairs = []
|
||||
# First pass: collect all 2-codepoint ligatures
|
||||
two_char = {seq: lig_cp for seq, lig_cp in filtered.items() if len(seq) == 2}
|
||||
for seq, lig_cp in two_char.items():
|
||||
packed = (seq[0] << 16) | seq[1]
|
||||
pairs.append((packed, lig_cp))
|
||||
|
||||
# Second pass: decompose 3+ codepoint ligatures into chained pairs
|
||||
for seq, lig_cp in filtered.items():
|
||||
if len(seq) < 3:
|
||||
continue
|
||||
# Try to find an intermediate: check if the first N-1 codepoints
|
||||
# form a known ligature, then chain (intermediate, last) -> lig
|
||||
prefix = seq[:-1]
|
||||
last_cp = seq[-1]
|
||||
if prefix in filtered:
|
||||
intermediate_cp = filtered[prefix]
|
||||
packed = (intermediate_cp << 16) | last_cp
|
||||
pairs.append((packed, lig_cp))
|
||||
else:
|
||||
print(f"ligatures: skipping {len(seq)}-char ligature "
|
||||
f"({', '.join(f'U+{cp:04X}' for cp in seq)}) -> U+{lig_cp:04X}: "
|
||||
f"no intermediate ligature for prefix", file=sys.stderr)
|
||||
|
||||
return pairs
|
||||
|
||||
ligature_codepoints = set(cp for cp in all_codepoints
|
||||
if not (COMBINING_MARKS_START <= cp <= COMBINING_MARKS_END))
|
||||
|
||||
# Map ligature codepoints to the font-stack index that serves them
|
||||
lig_cp_to_face_idx = {}
|
||||
for cp in ligature_codepoints:
|
||||
for face_idx, f in enumerate(font_stack):
|
||||
if f.get_char_index(cp) > 0:
|
||||
lig_cp_to_face_idx[cp] = face_idx
|
||||
break
|
||||
|
||||
# Group by face index
|
||||
lig_face_idx_cps = {}
|
||||
for cp, fi in lig_cp_to_face_idx.items():
|
||||
lig_face_idx_cps.setdefault(fi, set()).add(cp)
|
||||
|
||||
ligature_pairs = []
|
||||
for face_idx, cps in lig_face_idx_cps.items():
|
||||
font_path = args.fontstack[face_idx]
|
||||
ligature_pairs.extend(extract_ligatures_fonttools(font_path, cps))
|
||||
|
||||
# Deduplicate (keep first occurrence) and sort
|
||||
seen_lig_keys = set()
|
||||
unique_ligature_pairs = []
|
||||
for packed, lig_cp in ligature_pairs:
|
||||
if packed not in seen_lig_keys:
|
||||
seen_lig_keys.add(packed)
|
||||
unique_ligature_pairs.append((packed, lig_cp))
|
||||
ligature_pairs = sorted(unique_ligature_pairs, key=lambda p: p[0])
|
||||
print(f"ligatures: {len(ligature_pairs)} pairs extracted", file=sys.stderr)
|
||||
|
||||
compress = args.compress
|
||||
|
||||
# Build groups for compression
|
||||
@@ -294,6 +666,7 @@ if compress:
|
||||
(0x20A0, 0x20CF), # Currency Symbols
|
||||
(0x2190, 0x21FF), # Arrows
|
||||
(0x2200, 0x22FF), # Math Operators
|
||||
(0xFB00, 0xFB06), # Alphabetic Presentation Forms (ligatures)
|
||||
(0xFFFD, 0xFFFD), # Replacement Character
|
||||
]
|
||||
|
||||
@@ -385,9 +758,14 @@ else:
|
||||
print (" " + " ".join(f"0x{b:02X}," for b in c))
|
||||
print ("};\n");
|
||||
|
||||
def cp_label(cp):
|
||||
if cp == 0x5C:
|
||||
return '<backslash>'
|
||||
return chr(cp) if 0x20 < cp < 0x7F else f'U+{cp:04X}'
|
||||
|
||||
print(f"static const EpdGlyph {font_name}Glyphs[] = {{")
|
||||
for i, g in enumerate(glyph_props):
|
||||
print (" { " + ", ".join([f"{a}" for a in list(g[:-1])]),"},", f"// {chr(g.code_point) if g.code_point != 92 else '<backslash>'}")
|
||||
print (" { " + ", ".join([f"{a}" for a in list(g[:-1])]),"},", f"// {cp_label(g.code_point)}")
|
||||
print ("};\n");
|
||||
|
||||
print(f"static const EpdUnicodeInterval {font_name}Intervals[] = {{")
|
||||
@@ -405,6 +783,30 @@ if compress:
|
||||
compressed_offset += len(compressed)
|
||||
print("};\n")
|
||||
|
||||
if kern_map:
|
||||
print(f"static const EpdKernClassEntry {font_name}KernLeftClasses[] = {{")
|
||||
for cp, cls in kern_left_classes:
|
||||
print(f" {{ 0x{cp:04X}, {cls} }}, // {cp_label(cp)}")
|
||||
print("};\n")
|
||||
|
||||
print(f"static const EpdKernClassEntry {font_name}KernRightClasses[] = {{")
|
||||
for cp, cls in kern_right_classes:
|
||||
print(f" {{ 0x{cp:04X}, {cls} }}, // {cp_label(cp)}")
|
||||
print("};\n")
|
||||
|
||||
print(f"static const int8_t {font_name}KernMatrix[] = {{")
|
||||
for row in range(kern_left_class_count):
|
||||
row_start = row * kern_right_class_count
|
||||
row_vals = kern_matrix[row_start:row_start + kern_right_class_count]
|
||||
print(" " + ", ".join(f"{v:4d}" for v in row_vals) + ",")
|
||||
print("};\n")
|
||||
|
||||
if ligature_pairs:
|
||||
print(f"static const EpdLigaturePair {font_name}LigaturePairs[] = {{")
|
||||
for packed_pair, lig_cp in ligature_pairs:
|
||||
print(f" {{ 0x{packed_pair:08X}, 0x{lig_cp:04X} }}, // {cp_label(packed_pair >> 16)} {cp_label(packed_pair & 0xFFFF)} -> {cp_label(lig_cp)}")
|
||||
print("};\n")
|
||||
|
||||
print(f"static const EpdFontData {font_name} = {{")
|
||||
print(f" {font_name}Bitmaps,")
|
||||
print(f" {font_name}Glyphs,")
|
||||
@@ -420,4 +822,26 @@ if compress:
|
||||
else:
|
||||
print(f" nullptr,")
|
||||
print(f" 0,")
|
||||
if kern_map:
|
||||
print(f" {font_name}KernLeftClasses,")
|
||||
print(f" {font_name}KernRightClasses,")
|
||||
print(f" {font_name}KernMatrix,")
|
||||
print(f" {len(kern_left_classes)},")
|
||||
print(f" {len(kern_right_classes)},")
|
||||
print(f" {kern_left_class_count},")
|
||||
print(f" {kern_right_class_count},")
|
||||
else:
|
||||
print(f" nullptr,")
|
||||
print(f" nullptr,")
|
||||
print(f" nullptr,")
|
||||
print(f" 0,")
|
||||
print(f" 0,")
|
||||
print(f" 0,")
|
||||
print(f" 0,")
|
||||
if ligature_pairs:
|
||||
print(f" {font_name}LigaturePairs,")
|
||||
print(f" {len(ligature_pairs)},")
|
||||
else:
|
||||
print(f" nullptr,")
|
||||
print(f" 0,")
|
||||
print("};")
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "ParsedText.h"
|
||||
|
||||
#include <GfxRenderer.h>
|
||||
#include <Utf8.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
@@ -18,6 +19,28 @@ namespace {
|
||||
constexpr char SOFT_HYPHEN_UTF8[] = "\xC2\xAD";
|
||||
constexpr size_t SOFT_HYPHEN_BYTES = 2;
|
||||
|
||||
// Returns the first rendered codepoint of a word (skipping leading soft hyphens).
|
||||
uint32_t firstCodepoint(const std::string& word) {
|
||||
const auto* ptr = reinterpret_cast<const unsigned char*>(word.c_str());
|
||||
while (true) {
|
||||
const uint32_t cp = utf8NextCodepoint(&ptr);
|
||||
if (cp == 0) return 0;
|
||||
if (cp != 0x00AD) return cp; // skip soft hyphens
|
||||
}
|
||||
}
|
||||
|
||||
// Returns the last codepoint of a word by scanning backward for the start of the last UTF-8 sequence.
|
||||
uint32_t lastCodepoint(const std::string& word) {
|
||||
if (word.empty()) return 0;
|
||||
// UTF-8 continuation bytes start with 10xxxxxx; scan backward to find the leading byte.
|
||||
size_t i = word.size() - 1;
|
||||
while (i > 0 && (static_cast<uint8_t>(word[i]) & 0xC0) == 0x80) {
|
||||
--i;
|
||||
}
|
||||
const auto* ptr = reinterpret_cast<const unsigned char*>(word.c_str() + i);
|
||||
return utf8NextCodepoint(&ptr);
|
||||
}
|
||||
|
||||
bool containsSoftHyphen(const std::string& word) { return word.find(SOFT_HYPHEN_UTF8) != std::string::npos; }
|
||||
|
||||
// Removes every soft hyphen in-place so rendered glyphs match measured widths.
|
||||
@@ -29,7 +52,7 @@ void stripSoftHyphensInPlace(std::string& word) {
|
||||
}
|
||||
|
||||
// Returns the advance width for a word while ignoring soft hyphen glyphs and optionally appending a visible hyphen.
|
||||
// Uses advance width (sum of glyph advances) rather than bounding box width so that italic glyph overhangs
|
||||
// Uses advance width (sum of glyph advances + kerning) rather than bounding box width so that italic glyph overhangs
|
||||
// don't inflate inter-word spacing.
|
||||
uint16_t measureWordWidth(const GfxRenderer& renderer, const int fontId, const std::string& word,
|
||||
const EpdFontFamily::Style style, const bool appendHyphen = false) {
|
||||
@@ -78,7 +101,7 @@ void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fo
|
||||
applyParagraphIndent();
|
||||
|
||||
const int pageWidth = viewportWidth;
|
||||
const int spaceWidth = renderer.getSpaceWidth(fontId);
|
||||
const int spaceWidth = renderer.getSpaceWidth(fontId, EpdFontFamily::REGULAR);
|
||||
auto wordWidths = calculateWordWidths(renderer, fontId);
|
||||
|
||||
std::vector<size_t> lineBreakIndices;
|
||||
@@ -91,7 +114,7 @@ void ParsedText::layoutAndExtractLines(const GfxRenderer& renderer, const int fo
|
||||
const size_t lineCount = includeLastLine ? lineBreakIndices.size() : lineBreakIndices.size() - 1;
|
||||
|
||||
for (size_t i = 0; i < lineCount; ++i) {
|
||||
extractLine(i, pageWidth, spaceWidth, wordWidths, wordContinues, lineBreakIndices, processLine);
|
||||
extractLine(i, pageWidth, spaceWidth, wordWidths, wordContinues, lineBreakIndices, processLine, renderer, fontId);
|
||||
}
|
||||
|
||||
// Remove consumed words so size() reflects only remaining words
|
||||
@@ -159,7 +182,15 @@ std::vector<size_t> ParsedText::computeLineBreaks(const GfxRenderer& renderer, c
|
||||
|
||||
for (size_t j = i; j < totalWordCount; ++j) {
|
||||
// Add space before word j, unless it's the first word on the line or a continuation
|
||||
const int gap = j > static_cast<size_t>(i) && !continuesVec[j] ? spaceWidth : 0;
|
||||
int gap = 0;
|
||||
if (j > static_cast<size_t>(i) && !continuesVec[j]) {
|
||||
gap = spaceWidth;
|
||||
gap += renderer.getSpaceKernAdjust(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]),
|
||||
wordStyles[j - 1]);
|
||||
} else if (j > static_cast<size_t>(i) && continuesVec[j]) {
|
||||
// Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation)
|
||||
gap = renderer.getKerning(fontId, lastCodepoint(words[j - 1]), firstCodepoint(words[j]), wordStyles[j - 1]);
|
||||
}
|
||||
currlen += wordWidths[j] + gap;
|
||||
|
||||
if (currlen > effectivePageWidth) {
|
||||
@@ -265,7 +296,16 @@ std::vector<size_t> ParsedText::computeHyphenatedLineBreaks(const GfxRenderer& r
|
||||
// Consume as many words as possible for current line, splitting when prefixes fit
|
||||
while (currentIndex < wordWidths.size()) {
|
||||
const bool isFirstWord = currentIndex == lineStart;
|
||||
const int spacing = isFirstWord || continuesVec[currentIndex] ? 0 : spaceWidth;
|
||||
int spacing = 0;
|
||||
if (!isFirstWord && !continuesVec[currentIndex]) {
|
||||
spacing = spaceWidth;
|
||||
spacing += renderer.getSpaceKernAdjust(fontId, lastCodepoint(words[currentIndex - 1]),
|
||||
firstCodepoint(words[currentIndex]), wordStyles[currentIndex - 1]);
|
||||
} else if (!isFirstWord && continuesVec[currentIndex]) {
|
||||
// Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation)
|
||||
spacing = renderer.getKerning(fontId, lastCodepoint(words[currentIndex - 1]),
|
||||
firstCodepoint(words[currentIndex]), wordStyles[currentIndex - 1]);
|
||||
}
|
||||
const int candidateWidth = spacing + wordWidths[currentIndex];
|
||||
|
||||
// Word fits on current line
|
||||
@@ -397,7 +437,8 @@ bool ParsedText::hyphenateWordAtIndex(const size_t wordIndex, const int availabl
|
||||
void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const int spaceWidth,
|
||||
const std::vector<uint16_t>& wordWidths, const std::vector<bool>& continuesVec,
|
||||
const std::vector<size_t>& lineBreakIndices,
|
||||
const std::function<void(std::shared_ptr<TextBlock>)>& processLine) {
|
||||
const std::function<void(std::shared_ptr<TextBlock>)>& processLine,
|
||||
const GfxRenderer& renderer, const int fontId) {
|
||||
const size_t lineBreak = lineBreakIndices[breakIndex];
|
||||
const size_t lastBreakAt = breakIndex > 0 ? lineBreakIndices[breakIndex - 1] : 0;
|
||||
const size_t lineWordCount = lineBreak - lastBreakAt;
|
||||
@@ -410,37 +451,46 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
? blockStyle.textIndent
|
||||
: 0;
|
||||
|
||||
// Calculate total word width for this line and count actual word gaps
|
||||
// (continuation words attach to previous word with no gap)
|
||||
// Calculate total word width for this line, count actual word gaps,
|
||||
// and accumulate total natural gap widths (including space kerning adjustments).
|
||||
int lineWordWidthSum = 0;
|
||||
size_t actualGapCount = 0;
|
||||
int totalNaturalGaps = 0;
|
||||
|
||||
for (size_t wordIdx = 0; wordIdx < lineWordCount; wordIdx++) {
|
||||
lineWordWidthSum += wordWidths[lastBreakAt + wordIdx];
|
||||
// Count gaps: each word after the first creates a gap, unless it's a continuation
|
||||
if (wordIdx > 0 && !continuesVec[lastBreakAt + wordIdx]) {
|
||||
actualGapCount++;
|
||||
int naturalGap = spaceWidth;
|
||||
naturalGap += renderer.getSpaceKernAdjust(fontId, lastCodepoint(words[lastBreakAt + wordIdx - 1]),
|
||||
firstCodepoint(words[lastBreakAt + wordIdx]),
|
||||
wordStyles[lastBreakAt + wordIdx - 1]);
|
||||
totalNaturalGaps += naturalGap;
|
||||
} else if (wordIdx > 0 && continuesVec[lastBreakAt + wordIdx]) {
|
||||
// Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation)
|
||||
totalNaturalGaps +=
|
||||
renderer.getKerning(fontId, lastCodepoint(words[lastBreakAt + wordIdx - 1]),
|
||||
firstCodepoint(words[lastBreakAt + wordIdx]), wordStyles[lastBreakAt + wordIdx - 1]);
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate spacing (account for indent reducing effective page width on first line)
|
||||
const int effectivePageWidth = pageWidth - firstLineIndent;
|
||||
const int spareSpace = effectivePageWidth - lineWordWidthSum;
|
||||
|
||||
int spacing = spaceWidth;
|
||||
const bool isLastLine = breakIndex == lineBreakIndices.size() - 1;
|
||||
|
||||
// For justified text, calculate spacing based on actual gap count
|
||||
if (blockStyle.alignment == CssTextAlign::Justify && !isLastLine && actualGapCount >= 1) {
|
||||
spacing = spareSpace / static_cast<int>(actualGapCount);
|
||||
}
|
||||
// For justified text, compute per-gap extra to distribute remaining space evenly
|
||||
const int spareSpace = effectivePageWidth - lineWordWidthSum - totalNaturalGaps;
|
||||
const int justifyExtra = (blockStyle.alignment == CssTextAlign::Justify && !isLastLine && actualGapCount >= 1)
|
||||
? spareSpace / static_cast<int>(actualGapCount)
|
||||
: 0;
|
||||
|
||||
// Calculate initial x position (first line starts at indent for left/justified text)
|
||||
auto xpos = static_cast<uint16_t>(firstLineIndent);
|
||||
if (blockStyle.alignment == CssTextAlign::Right) {
|
||||
xpos = spareSpace - static_cast<int>(actualGapCount) * spaceWidth;
|
||||
xpos = effectivePageWidth - lineWordWidthSum - totalNaturalGaps;
|
||||
} else if (blockStyle.alignment == CssTextAlign::Center) {
|
||||
xpos = (spareSpace - static_cast<int>(actualGapCount) * spaceWidth) / 2;
|
||||
xpos = (effectivePageWidth - lineWordWidthSum - totalNaturalGaps) / 2;
|
||||
}
|
||||
|
||||
// Pre-calculate X positions for words
|
||||
@@ -449,14 +499,28 @@ void ParsedText::extractLine(const size_t breakIndex, const int pageWidth, const
|
||||
lineXPos.reserve(lineWordCount);
|
||||
|
||||
for (size_t wordIdx = 0; wordIdx < lineWordCount; wordIdx++) {
|
||||
const uint16_t currentWordWidth = wordWidths[lastBreakAt + wordIdx];
|
||||
|
||||
lineXPos.push_back(xpos);
|
||||
|
||||
// Add spacing after this word, unless the next word is a continuation
|
||||
const bool nextIsContinuation = wordIdx + 1 < lineWordCount && continuesVec[lastBreakAt + wordIdx + 1];
|
||||
|
||||
xpos += currentWordWidth + (nextIsContinuation ? 0 : spacing);
|
||||
if (nextIsContinuation) {
|
||||
int advance = wordWidths[lastBreakAt + wordIdx];
|
||||
// Cross-boundary kerning for continuation words (e.g. nonbreaking spaces, attached punctuation)
|
||||
advance +=
|
||||
renderer.getKerning(fontId, lastCodepoint(words[lastBreakAt + wordIdx]),
|
||||
firstCodepoint(words[lastBreakAt + wordIdx + 1]), wordStyles[lastBreakAt + wordIdx]);
|
||||
xpos += advance;
|
||||
} else {
|
||||
int gap = spaceWidth;
|
||||
if (wordIdx + 1 < lineWordCount) {
|
||||
gap += renderer.getSpaceKernAdjust(fontId, lastCodepoint(words[lastBreakAt + wordIdx]),
|
||||
firstCodepoint(words[lastBreakAt + wordIdx + 1]),
|
||||
wordStyles[lastBreakAt + wordIdx]);
|
||||
}
|
||||
if (blockStyle.alignment == CssTextAlign::Justify && !isLastLine) {
|
||||
gap += justifyExtra;
|
||||
}
|
||||
xpos += wordWidths[lastBreakAt + wordIdx] + gap;
|
||||
}
|
||||
}
|
||||
|
||||
// Build line data by moving from the original vectors using index range
|
||||
|
||||
@@ -30,7 +30,8 @@ class ParsedText {
|
||||
std::vector<uint16_t>& wordWidths, bool allowFallbackBreaks);
|
||||
void extractLine(size_t breakIndex, int pageWidth, int spaceWidth, const std::vector<uint16_t>& wordWidths,
|
||||
const std::vector<bool>& continuesVec, const std::vector<size_t>& lineBreakIndices,
|
||||
const std::function<void(std::shared_ptr<TextBlock>)>& processLine);
|
||||
const std::function<void(std::shared_ptr<TextBlock>)>& processLine, const GfxRenderer& renderer,
|
||||
int fontId);
|
||||
std::vector<uint16_t> calculateWordWidths(const GfxRenderer& renderer, int fontId);
|
||||
|
||||
public:
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
#include "parsers/ChapterHtmlSlimParser.h"
|
||||
|
||||
namespace {
|
||||
constexpr uint8_t SECTION_FILE_VERSION = 13;
|
||||
constexpr uint8_t SECTION_FILE_VERSION = 14;
|
||||
constexpr uint32_t HEADER_SIZE = sizeof(uint8_t) + sizeof(int) + sizeof(float) + sizeof(bool) + sizeof(uint8_t) +
|
||||
sizeof(uint16_t) + sizeof(uint16_t) + sizeof(uint16_t) + sizeof(bool) + sizeof(bool) +
|
||||
sizeof(uint32_t);
|
||||
|
||||
@@ -153,13 +153,11 @@ static void renderCharImpl(const GfxRenderer& renderer, GfxRenderer::RenderMode
|
||||
}
|
||||
}
|
||||
|
||||
if (!utf8IsCombiningMark(cp)) {
|
||||
if constexpr (rotation == TextRotation::Rotated90CW) {
|
||||
*cursorY -= glyph->advanceX;
|
||||
} else {
|
||||
*cursorX += glyph->advanceX;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// IMPORTANT: This function is in critical rendering path and is called for every pixel. Please keep it as simple and
|
||||
@@ -209,12 +207,11 @@ void GfxRenderer::drawCenteredText(const int fontId, const int y, const char* te
|
||||
void GfxRenderer::drawText(const int fontId, const int x, const int y, const char* text, const bool black,
|
||||
const EpdFontFamily::Style style) const {
|
||||
int yPos = y + getFontAscenderSize(fontId);
|
||||
int xpos = x;
|
||||
int xPos = x;
|
||||
int lastBaseX = x;
|
||||
int lastBaseY = yPos;
|
||||
int lastBaseAdvance = 0;
|
||||
int lastBaseTop = 0;
|
||||
bool hasBaseGlyph = false;
|
||||
|
||||
// cannot draw a NULL / empty string
|
||||
if (text == nullptr || *text == '\0') {
|
||||
@@ -230,8 +227,9 @@ void GfxRenderer::drawText(const int fontId, const int x, const int y, const cha
|
||||
constexpr int MIN_COMBINING_GAP_PX = 1;
|
||||
|
||||
uint32_t cp;
|
||||
uint32_t prevCp = 0;
|
||||
while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text)))) {
|
||||
if (utf8IsCombiningMark(cp) && hasBaseGlyph) {
|
||||
if (utf8IsCombiningMark(cp)) {
|
||||
const EpdGlyph* combiningGlyph = font.getGlyph(cp, style);
|
||||
int raiseBy = 0;
|
||||
if (combiningGlyph) {
|
||||
@@ -247,16 +245,20 @@ void GfxRenderer::drawText(const int fontId, const int x, const int y, const cha
|
||||
continue;
|
||||
}
|
||||
|
||||
cp = font.applyLigatures(cp, text, style);
|
||||
if (prevCp != 0) {
|
||||
xPos += font.getKerning(prevCp, cp, style);
|
||||
}
|
||||
|
||||
const EpdGlyph* glyph = font.getGlyph(cp, style);
|
||||
if (!utf8IsCombiningMark(cp)) {
|
||||
lastBaseX = xpos;
|
||||
|
||||
lastBaseX = xPos;
|
||||
lastBaseY = yPos;
|
||||
lastBaseAdvance = glyph ? glyph->advanceX : 0;
|
||||
lastBaseTop = glyph ? glyph->top : 0;
|
||||
hasBaseGlyph = true;
|
||||
}
|
||||
|
||||
renderChar(font, cp, &xpos, &yPos, black, style);
|
||||
renderChar(font, cp, &xPos, &yPos, black, style);
|
||||
prevCp = cp;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -882,7 +884,22 @@ int GfxRenderer::getSpaceWidth(const int fontId, const EpdFontFamily::Style styl
|
||||
return spaceGlyph ? spaceGlyph->advanceX : 0;
|
||||
}
|
||||
|
||||
int GfxRenderer::getTextAdvanceX(const int fontId, const char* text, const EpdFontFamily::Style style) const {
|
||||
int GfxRenderer::getSpaceKernAdjust(const int fontId, const uint32_t leftCp, const uint32_t rightCp,
|
||||
const EpdFontFamily::Style style) const {
|
||||
const auto fontIt = fontMap.find(fontId);
|
||||
if (fontIt == fontMap.end()) return 0;
|
||||
const auto& font = fontIt->second;
|
||||
return font.getKerning(leftCp, ' ', style) + font.getKerning(' ', rightCp, style);
|
||||
}
|
||||
|
||||
int GfxRenderer::getKerning(const int fontId, const uint32_t leftCp, const uint32_t rightCp,
|
||||
const EpdFontFamily::Style style) const {
|
||||
const auto fontIt = fontMap.find(fontId);
|
||||
if (fontIt == fontMap.end()) return 0;
|
||||
return fontIt->second.getKerning(leftCp, rightCp, style);
|
||||
}
|
||||
|
||||
int GfxRenderer::getTextAdvanceX(const int fontId, const char* text, EpdFontFamily::Style style) const {
|
||||
const auto fontIt = fontMap.find(fontId);
|
||||
if (fontIt == fontMap.end()) {
|
||||
LOG_ERR("GFX", "Font %d not found", fontId);
|
||||
@@ -890,14 +907,20 @@ int GfxRenderer::getTextAdvanceX(const int fontId, const char* text, const EpdFo
|
||||
}
|
||||
|
||||
uint32_t cp;
|
||||
uint32_t prevCp = 0;
|
||||
int width = 0;
|
||||
const auto& font = fontIt->second;
|
||||
while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text)))) {
|
||||
if (utf8IsCombiningMark(cp)) {
|
||||
continue;
|
||||
}
|
||||
cp = font.applyLigatures(cp, text, style);
|
||||
if (prevCp != 0) {
|
||||
width += font.getKerning(prevCp, cp, style);
|
||||
}
|
||||
const EpdGlyph* glyph = font.getGlyph(cp, style);
|
||||
if (glyph) width += glyph->advanceX;
|
||||
prevCp = cp;
|
||||
}
|
||||
return width;
|
||||
}
|
||||
@@ -952,12 +975,12 @@ void GfxRenderer::drawTextRotated90CW(const int fontId, const int x, const int y
|
||||
int lastBaseY = y;
|
||||
int lastBaseAdvance = 0;
|
||||
int lastBaseTop = 0;
|
||||
bool hasBaseGlyph = false;
|
||||
constexpr int MIN_COMBINING_GAP_PX = 1;
|
||||
|
||||
uint32_t cp;
|
||||
uint32_t prevCp = 0;
|
||||
while ((cp = utf8NextCodepoint(reinterpret_cast<const uint8_t**>(&text)))) {
|
||||
if (utf8IsCombiningMark(cp) && hasBaseGlyph) {
|
||||
if (utf8IsCombiningMark(cp)) {
|
||||
const EpdGlyph* combiningGlyph = font.getGlyph(cp, style);
|
||||
int raiseBy = 0;
|
||||
if (combiningGlyph) {
|
||||
@@ -973,16 +996,20 @@ void GfxRenderer::drawTextRotated90CW(const int fontId, const int x, const int y
|
||||
continue;
|
||||
}
|
||||
|
||||
cp = font.applyLigatures(cp, text, style);
|
||||
if (prevCp != 0) {
|
||||
yPos -= font.getKerning(prevCp, cp, style);
|
||||
}
|
||||
|
||||
const EpdGlyph* glyph = font.getGlyph(cp, style);
|
||||
if (!utf8IsCombiningMark(cp)) {
|
||||
|
||||
lastBaseX = xPos;
|
||||
lastBaseY = yPos;
|
||||
lastBaseAdvance = glyph ? glyph->advanceX : 0;
|
||||
lastBaseTop = glyph ? glyph->top : 0;
|
||||
hasBaseGlyph = true;
|
||||
}
|
||||
|
||||
renderCharImpl<TextRotation::Rotated90CW>(*this, renderMode, font, cp, &xPos, &yPos, black, style);
|
||||
prevCp = cp;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -110,6 +110,11 @@ class GfxRenderer {
|
||||
void drawText(int fontId, int x, int y, const char* text, bool black = true,
|
||||
EpdFontFamily::Style style = EpdFontFamily::REGULAR) const;
|
||||
int getSpaceWidth(int fontId, EpdFontFamily::Style style = EpdFontFamily::REGULAR) const;
|
||||
/// Returns the kerning adjustment for a space between two codepoints:
|
||||
/// kern(leftCp, ' ') + kern(' ', rightCp). Returns 0 if kerning is unavailable.
|
||||
int getSpaceKernAdjust(int fontId, uint32_t leftCp, uint32_t rightCp, EpdFontFamily::Style style) const;
|
||||
/// Returns the kerning adjustment between two adjacent codepoints.
|
||||
int getKerning(int fontId, uint32_t leftCp, uint32_t rightCp, EpdFontFamily::Style style) const;
|
||||
int getTextAdvanceX(int fontId, const char* text, EpdFontFamily::Style style) const;
|
||||
int getFontAscenderSize(int fontId) const;
|
||||
int getLineHeight(int fontId) const;
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -510,7 +510,7 @@ void BaseTheme::drawRecentBookCover(GfxRenderer& renderer, Rect rect, const std:
|
||||
std::string currentLine;
|
||||
// Extra padding inside the card so text doesn't hug the border
|
||||
const int maxLineWidth = bookWidth - 40;
|
||||
const int spaceWidth = renderer.getSpaceWidth(UI_12_FONT_ID);
|
||||
const int spaceWidth = renderer.getSpaceWidth(UI_12_FONT_ID, EpdFontFamily::REGULAR);
|
||||
|
||||
for (auto& i : words) {
|
||||
// If we just hit the line limit (3), stop processing words
|
||||
|
||||
Binary file not shown.
Reference in New Issue
Block a user