#!/usr/bin/env python3 """Generate .cpfont binary files for SD card font loading. Outputs binary .cpfont files containing glyph metadata and uncompressed 2-bit bitmaps, matching the EpdFontData/EpdGlyph/EpdUnicodeInterval struct layout on the ESP32-C3 (little-endian, RISC-V). Usage: # Single file with specific presets python fontconvert_sdcard.py \\ --intervals latin-ext,greek,cyrillic \\ --size 14 --style regular \\ NotoSans-Regular.ttf \\ -o NotoSansExt_14.cpfont # All 4 sizes at once python fontconvert_sdcard.py \\ --intervals cjk \\ --sizes 12,14,16,18 --style regular \\ NotoSansCJKsc-Regular.otf \\ --output-dir NotoSansCJK/ """ from __future__ import annotations import struct import sys import os import re import math import argparse from collections import namedtuple from cpfont_version import CPFONT_VERSION # --- Unicode interval presets --- INTERVAL_PRESETS = { "ascii": [(0x0020, 0x007E)], "latin1": [(0x0080, 0x00FF)], "latin-ext": [(0x0020, 0x007E), (0x0080, 0x00FF), (0x0100, 0x024F), (0x1E00, 0x1EFF), (0x2000, 0x206F), (0xFB00, 0xFB06)], "greek": [(0x0370, 0x03FF), (0x1F00, 0x1FFF)], "cyrillic": [(0x0400, 0x04FF), (0x0500, 0x052F)], "georgian": [(0x10A0, 0x10FF), (0x2D00, 0x2D2F)], "armenian": [(0x0530, 0x058F)], "ethiopic": [(0x1200, 0x137F), (0x1380, 0x139F), (0x2D80, 0x2DDF)], "vietnamese": [(0x01A0, 0x01B0), (0x1EA0, 0x1EF9)], "punctuation": [(0x2000, 0x206F)], "cjk": [(0x3000, 0x303F), (0x3040, 0x309F), (0x30A0, 0x30FF), (0x4E00, 0x9FFF), (0xF900, 0xFAFF), (0xFF00, 0xFFEF)], "hangul": [(0xAC00, 0xD7AF), (0x1100, 0x11FF), (0x3130, 0x318F)], "cherokee": [(0x13A0, 0x13FF), (0xAB70, 0xABBF)], "tifinagh": [(0x2D30, 0x2D7F)], # Symbol blocks commonly seen in scifi/popsci/literary fiction. "symbols": [(0x2070, 0x209F), (0x20A0, 0x20CF), (0x2150, 0x218F), (0x2190, 0x21FF), (0x2200, 0x22FF), (0x2500, 0x257F), (0x25A0, 0x25FF), (0x2600, 0x26FF), (0x2700, 0x27BF)], # Composite preset for English-language literary fiction including scifi/popsci. # Greek for physics terms, math operators, geometric shapes, uncommon # dialogue punctuation, CJK quote marks, miscellaneous symbols (♪♫♬), dingbats. "reading": [(0x0020, 0x024F), (0x0300, 0x036F), (0x0370, 0x03FF), (0x0400, 0x04FF), (0x1E00, 0x1EFF), (0x2000, 0x206F), (0x2070, 0x209F), (0x20A0, 0x20CF), (0x2150, 0x218F), (0x2190, 0x21FF), (0x2200, 0x22FF), (0x2500, 0x257F), (0x25A0, 0x25FF), (0x2600, 0x26FF), (0x2700, 0x27BF), (0x2900, 0x29FF), (0x2E00, 0x2E7F), (0x3000, 0x303F), (0xFB00, 0xFB06)], # Matches the built-in font intervals from fontconvert.py exactly "builtin": [(0x0000, 0x007F), (0x0080, 0x00FF), (0x0100, 0x017F), (0x01A0, 0x01A1), (0x01AF, 0x01B0), (0x01C4, 0x021F), (0x0300, 0x036F), (0x0400, 0x04FF), (0x1EA0, 0x1EF9), (0x2000, 0x206F), (0x20A0, 0x20CF), (0x2070, 0x209F), (0x2190, 0x21FF), (0x2200, 0x22FF), (0xFB00, 0xFB06)], } # Regex for parsing unnamed hex range intervals: (0xSTART-0xEND) _HEX_RANGE_PATTERN = re.compile(r'^\(0x([0-9a-fA-F]+)-0x([0-9a-fA-F]+)\)$') def parse_hex_range(s: str) -> tuple[int, int] | None: match = _HEX_RANGE_PATTERN.fullmatch(s) if not match: return None start_hex, end_hex = match.groups() start, end = int(start_hex, 16), int(end_hex, 16) # Validating Unicode range bounds. if start > end or end > 0x10FFFF: return None return start, end def resolve_intervals(preset_str): """Resolve comma-separated preset names into a merged, sorted, deduplicated interval list.""" all_intervals = [] for name in preset_str.split(","): name = name.strip().lower() unnamed_interval = parse_hex_range(name) if name not in INTERVAL_PRESETS and unnamed_interval is None: print(f"Error: unknown interval preset '{name}'", file=sys.stderr) print(f"Available presets: {', '.join(sorted(INTERVAL_PRESETS.keys()))}", file=sys.stderr) print("You can also specify unnamed hex ranges like (0x2100-0x214F)", file=sys.stderr) sys.exit(1) if unnamed_interval is not None: all_intervals.append(unnamed_interval) else: all_intervals.extend(INTERVAL_PRESETS[name]) # Always add replacement character all_intervals.append((0xFFFD, 0xFFFD)) # Sort and merge overlapping/adjacent intervals all_intervals.sort() merged = [] for start, end in all_intervals: if merged and start <= merged[-1][1] + 1: merged[-1] = (merged[-1][0], max(merged[-1][1], end)) else: merged.append((start, end)) return merged GlyphProps = namedtuple("GlyphProps", [ "width", "height", "advance_x", "left", "top", "data_length", "data_offset", "code_point" ]) # Intermediate data from rasterizing one font style StyleRasterData = namedtuple("StyleRasterData", [ "style_id", # 0=regular, 1=bold, 2=italic, 3=bolditalic "intervals", # validated intervals [(start, end), ...] "all_glyphs", # [(GlyphProps, packed_bytes), ...] "total_bitmap_size", # int "advanceY", "ascender", "descender", "kern_left_classes", "kern_right_classes", "kern_matrix", "kern_left_class_count", "kern_right_class_count", "ligature_pairs", ]) def norm_floor(val): return int(math.floor(val / (1 << 6))) def norm_ceil(val): return int(math.ceil(val / (1 << 6))) # Fixed-point (fp4) output conventions (must match EpdFontData.h / fp4 namespace): # # advanceX 12.4 unsigned fixed-point (uint16_t). # 12 integer bits, 4 fractional bits = 1/16-pixel resolution. # Encoded from FreeType's 16.16 linearHoriAdvance. # # kernMatrix 4.4 signed fixed-point (int8_t). # 4 integer bits, 4 fractional bits = 1/16-pixel resolution. # Range: -8.0 to +7.9375 pixels. # Encoded from font design-unit kerning values. # # Both share 4 fractional bits so the renderer can add them directly into a # single int32_t accumulator and defer rounding until pixel placement. def fp4_from_ft16_16(val): """Convert FreeType 16.16 fixed-point to 12.4 fixed-point with rounding.""" return (val + (1 << 11)) >> 12 def fp4_from_design_units(du, scale): """Convert a font design-unit value to 4.4 fixed-point, clamped to int8_t. Multiplies by scale (ppem / units_per_em) and shifts into 4 fractional bits. The result is rounded to nearest and clamped to [-128, 127]. """ raw = round(du * scale * 16) return max(-128, min(127, raw)) # Standard Unicode ligature codepoints for known input sequences. # Used as a fallback when the GSUB substitute glyph has no cmap entry. STANDARD_LIGATURE_MAP = { (0x66, 0x66): 0xFB00, # ff (0x66, 0x69): 0xFB01, # fi (0x66, 0x6C): 0xFB02, # fl (0x66, 0x66, 0x69): 0xFB03, # ffi (0x66, 0x66, 0x6C): 0xFB04, # ffl (0x17F, 0x74): 0xFB05, # long-s + t (0x73, 0x74): 0xFB06, # st } def _extract_pairpos_subtable(subtable, glyph_to_cp, raw_kern): """Extract kerning from a PairPos subtable (Format 1 or 2).""" if subtable.Format == 1: # Individual pairs for i, coverage_glyph in enumerate(subtable.Coverage.glyphs): if coverage_glyph not in glyph_to_cp: continue pair_set = subtable.PairSet[i] for pvr in pair_set.PairValueRecord: if pvr.SecondGlyph not in glyph_to_cp: continue xa = 0 if hasattr(pvr, 'Value1') and pvr.Value1: xa = getattr(pvr.Value1, 'XAdvance', 0) or 0 if xa != 0: key = (coverage_glyph, pvr.SecondGlyph) raw_kern[key] = raw_kern.get(key, 0) + xa elif subtable.Format == 2: # Class-based pairs — iterate by class, not by glyph, to avoid # O(glyphs²) explosion for CJK fonts with many requested glyphs. class_def1 = subtable.ClassDef1.classDefs if subtable.ClassDef1 else {} class_def2 = subtable.ClassDef2.classDefs if subtable.ClassDef2 else {} coverage_set = set(subtable.Coverage.glyphs) # Build reverse mappings: class_id -> list of glyph names left_by_class = {} # only glyphs in coverage AND glyph_to_cp for glyph in glyph_to_cp: if glyph not in coverage_set: continue c1 = class_def1.get(glyph, 0) left_by_class.setdefault(c1, []).append(glyph) right_by_class = {} # all glyphs in glyph_to_cp for glyph in glyph_to_cp: c2 = class_def2.get(glyph, 0) right_by_class.setdefault(c2, []).append(glyph) # Iterate class pairs (typically << glyph pairs) for c1, class1_rec in enumerate(subtable.Class1Record): if c1 not in left_by_class: continue for c2, c2_rec in enumerate(class1_rec.Class2Record): xa = 0 if hasattr(c2_rec, 'Value1') and c2_rec.Value1: xa = getattr(c2_rec.Value1, 'XAdvance', 0) or 0 if xa == 0: continue if c2 not in right_by_class: continue for lg in left_by_class[c1]: for rg in right_by_class[c2]: key = (lg, rg) raw_kern[key] = raw_kern.get(key, 0) + xa def extract_kerning_fonttools(font_path, codepoints, ppem): """Extract kerning pairs from a font file using fonttools. Returns dict of {(leftCp, rightCp): pixel_adjust} for the given codepoints. Values are scaled from font design units to integer pixels at ppem. """ from fontTools.ttLib import TTFont font = TTFont(font_path) units_per_em = font['head'].unitsPerEm cmap = font.getBestCmap() or {} # Build glyph_name -> [codepoints] map (preserves aliases where multiple # codepoints share a glyph, e.g. space/nbsp) glyph_to_cps = {} for cp in codepoints: gname = cmap.get(cp) if gname: glyph_to_cps.setdefault(gname, []).append(cp) # Flat dict for membership checks and subtable extraction (uses keys only) glyph_to_cp = glyph_to_cps # Collect raw kerning values in font design units raw_kern = {} # (left_glyph_name, right_glyph_name) -> design_units # 1. Legacy kern table if 'kern' in font: for subtable in font['kern'].kernTables: if hasattr(subtable, 'kernTable'): for (lg, rg), val in subtable.kernTable.items(): if lg in glyph_to_cp and rg in glyph_to_cp: raw_kern[(lg, rg)] = raw_kern.get((lg, rg), 0) + val # 2. GPOS 'kern' feature if 'GPOS' in font: gpos = font['GPOS'].table kern_lookup_indices = set() if gpos.FeatureList: for fr in gpos.FeatureList.FeatureRecord: if fr.FeatureTag == 'kern': kern_lookup_indices.update(fr.Feature.LookupListIndex) for li in kern_lookup_indices: lookup = gpos.LookupList.Lookup[li] for st in lookup.SubTable: actual = st # Unwrap Extension (lookup type 9) wrappers. After unwrapping, # `lookup.LookupType` is still 9, so we must look at the # *effective* type carried on the extension subtable to know # whether `actual` is a PairPos table. if lookup.LookupType == 9 and hasattr(st, 'ExtSubTable'): actual = st.ExtSubTable effective_type = getattr(st, 'ExtensionLookupType', lookup.LookupType) if hasattr(actual, 'Format'): # _extract_pairpos_subtable assumes a Type-2 (PairPos) # subtable. Other lookup types reachable through the kern # feature (cursive attachment, mark-to-mark, contextual, # etc.) have a different shape and crash inside the # extractor. Skip them with a debug note rather than # aborting the whole build. Modern fonts often ship kern # via Extension-wrapped PairPos, so checking the effective # type instead of the outer type is what makes those # lookups actually reach the extractor. if effective_type == 2: _extract_pairpos_subtable(actual, glyph_to_cp, raw_kern) else: print(f" Debug: skipping unsupported GPOS kern lookupType=" f"{effective_type} (outer={lookup.LookupType}, Format={actual.Format})", file=sys.stderr) font.close() # Scale design-unit kerning values to 4.4 fixed-point pixels. # Expand glyph aliases: if multiple codepoints share a glyph, emit kern # pairs for all codepoint combinations. scale = ppem / units_per_em result = {} # (leftCp, rightCp) -> 4.4 fixed-point adjust for (lg, rg), du in raw_kern.items(): adjust = fp4_from_design_units(du, scale) if adjust != 0: for lcp in glyph_to_cps[lg]: for rcp in glyph_to_cps[rg]: result[(lcp, rcp)] = adjust return result def derive_kern_classes(kern_map): """Derive class-based kerning from a pair map. Returns (kern_left_classes, kern_right_classes, kern_matrix, kern_left_class_count, kern_right_class_count) where: - kern_left_classes: sorted list of (codepoint, classId) tuples - kern_right_classes: sorted list of (codepoint, classId) tuples - kern_matrix: flat list of int8 values (left_class_count * right_class_count) - kern_left_class_count: number of distinct left classes - kern_right_class_count: number of distinct right classes """ if not kern_map: return [], [], [], 0, 0 all_left_cps = {lcp for lcp, _ in kern_map} all_right_cps = {rcp for _, rcp in kern_map} sorted_right_cps = sorted(all_right_cps) sorted_left_cps = sorted(all_left_cps) # Group left codepoints by identical adjustment row left_profile_to_class = {} left_class_map = {} left_class_id = 1 for lcp in sorted(all_left_cps): row = tuple(kern_map.get((lcp, rcp), 0) for rcp in sorted_right_cps) if row not in left_profile_to_class: left_profile_to_class[row] = left_class_id left_class_id += 1 left_class_map[lcp] = left_profile_to_class[row] # Group right codepoints by identical adjustment column right_profile_to_class = {} right_class_map = {} right_class_id = 1 for rcp in sorted(all_right_cps): col = tuple(kern_map.get((lcp, rcp), 0) for lcp in sorted_left_cps) if col not in right_profile_to_class: right_profile_to_class[col] = right_class_id right_class_id += 1 right_class_map[rcp] = right_profile_to_class[col] kern_left_class_count = left_class_id - 1 kern_right_class_count = right_class_id - 1 if kern_left_class_count > 255 or kern_right_class_count > 255: print(f"WARNING: kerning class count exceeds uint8_t range " f"(left={kern_left_class_count}, right={kern_right_class_count}), " f"dropping kerning for this style", file=sys.stderr) return ([], [], [], 0, 0) # Build the class x class matrix kern_matrix = [0] * (kern_left_class_count * kern_right_class_count) for (lcp, rcp), adjust in kern_map.items(): lc = left_class_map[lcp] - 1 rc = right_class_map[rcp] - 1 kern_matrix[lc * kern_right_class_count + rc] = adjust # Build sorted class entry lists kern_left_classes = sorted(left_class_map.items()) kern_right_classes = sorted(right_class_map.items()) return (kern_left_classes, kern_right_classes, kern_matrix, kern_left_class_count, kern_right_class_count) def extract_ligatures_fonttools(font_path, codepoints): """Extract ligature substitution pairs from a font file using fonttools. Returns list of (packed_pair, ligature_codepoint) for the given codepoints. Multi-character ligatures are decomposed into chained pairs. """ from fontTools.ttLib import TTFont font = TTFont(font_path) cmap = font.getBestCmap() or {} # Build glyph_name -> codepoint and codepoint -> glyph_name maps glyph_to_cp = {} cp_to_glyph = {} for cp, gname in cmap.items(): glyph_to_cp[gname] = cp cp_to_glyph[cp] = gname # Collect raw ligature rules: (sequence_of_codepoints) -> ligature_codepoint raw_ligatures = {} # tuple of codepoints -> ligature codepoint if 'GSUB' in font: gsub = font['GSUB'].table LIGATURE_FEATURES = ('liga', 'rlig') liga_lookup_indices = set() if gsub.FeatureList: for fr in gsub.FeatureList.FeatureRecord: if fr.FeatureTag in LIGATURE_FEATURES: liga_lookup_indices.update(fr.Feature.LookupListIndex) for li in liga_lookup_indices: lookup = gsub.LookupList.Lookup[li] for st in lookup.SubTable: actual = st # Unwrap Extension (lookup type 7) wrappers if lookup.LookupType == 7 and hasattr(st, 'ExtSubTable'): actual = st.ExtSubTable # LigatureSubst is lookup type 4 if not hasattr(actual, 'ligatures'): continue for first_glyph, ligature_list in actual.ligatures.items(): if first_glyph not in glyph_to_cp: continue first_cp = glyph_to_cp[first_glyph] for lig in ligature_list: component_cps = [] valid = True for comp_glyph in lig.Component: if comp_glyph not in glyph_to_cp: valid = False break component_cps.append(glyph_to_cp[comp_glyph]) if not valid: continue seq = tuple([first_cp] + component_cps) if lig.LigGlyph in glyph_to_cp: lig_cp = glyph_to_cp[lig.LigGlyph] elif seq in STANDARD_LIGATURE_MAP: lig_cp = STANDARD_LIGATURE_MAP[seq] else: seq_str = ', '.join(f'U+{cp:04X}' for cp in seq) print(f"ligatures: WARNING: dropping ligature ({seq_str}) -> " f"glyph '{lig.LigGlyph}': output glyph has no cmap entry " f"and input sequence is not in STANDARD_LIGATURE_MAP", file=sys.stderr) continue raw_ligatures[seq] = lig_cp font.close() # Filter: only keep ligatures where all input and output codepoints are # in our generated glyph set, and all codepoints fit in 16 bits. # # The on-disk format packs each component as a uint16 (the 3+ chained # path packs `intermediate_cp << 16 | last_cp`, where `intermediate_cp` # is the lig_cp of the prefix). Dropping any seq with an SMP cp here — # plus any lig_cp > 0xFFFF — means every cp that reaches `packed = … << # 16 | …` below is already 16-bit safe, including the chained path # (intermediate_cp = filtered[prefix] is filtered too). codepoints_set = set(codepoints) filtered = {} for seq, lig_cp in raw_ligatures.items(): if lig_cp not in codepoints_set or lig_cp > 0xFFFF: continue if any(cp > 0xFFFF for cp in seq): continue if all(cp in codepoints_set for cp in seq): filtered[seq] = lig_cp # Decompose into chained pairs pairs = [] # First pass: collect all 2-codepoint ligatures two_char = {seq: lig_cp for seq, lig_cp in filtered.items() if len(seq) == 2} for seq, lig_cp in two_char.items(): packed = (seq[0] << 16) | seq[1] pairs.append((packed, lig_cp)) # Second pass: decompose 3+ codepoint ligatures into chained pairs for seq, lig_cp in filtered.items(): if len(seq) < 3: continue prefix = seq[:-1] last_cp = seq[-1] if prefix in filtered: intermediate_cp = filtered[prefix] packed = (intermediate_cp << 16) | last_cp pairs.append((packed, lig_cp)) else: print(f"ligatures: skipping {len(seq)}-char ligature " f"({', '.join(f'U+{cp:04X}' for cp in seq)}) -> U+{lig_cp:04X}: " f"no intermediate ligature for prefix", file=sys.stderr) # Sort by packed pair key — on-device lookup uses binary search pairs.sort(key=lambda p: p[0]) return pairs def rasterize_font_style(fontfile, size, intervals, style_id=0, force_autohint=False): """Rasterize all glyphs for one font style. Returns StyleRasterData.""" import freetype style_names = {0: "regular", 1: "bold", 2: "italic", 3: "bolditalic"} style_label = style_names.get(style_id, str(style_id)) face = freetype.Face(fontfile) # Set font size at 150 DPI (matching fontconvert.py) BEFORE any glyph load. # load_glyph() with FT_LOAD_RENDER renders at the active size, so calling # it before set_char_size() would waste work at the default size and risk # Invalid_Size_Handle on some fonts. face.set_char_size(size << 6, size << 6, 150, 150) load_flags = freetype.FT_LOAD_RENDER if force_autohint: load_flags |= freetype.FT_LOAD_FORCE_AUTOHINT def load_glyph(code_point): glyph_index = face.get_char_index(code_point) if glyph_index > 0: face.load_glyph(glyph_index, load_flags) return face return None # Validate intervals: remove codepoints not present in the font. # Only check glyph existence via get_char_index — do NOT call # load_glyph here, as that triggers FT_LOAD_RENDER at the target # DPI and doubles total rasterization time for no benefit. print(f" [{style_label}] Validating intervals against font...", file=sys.stderr) validated_intervals = [] for i_start, i_end in intervals: start = i_start for code_point in range(i_start, i_end + 1): if face.get_char_index(code_point) == 0: if start < code_point: validated_intervals.append((start, code_point - 1)) start = code_point + 1 if start <= i_end: validated_intervals.append((start, i_end)) intervals = validated_intervals total_glyphs = sum(end - start + 1 for start, end in intervals) print(f" [{style_label}] Validated: {len(intervals)} intervals, {total_glyphs} glyphs", file=sys.stderr) # Rasterize all glyphs total_bitmap_size = 0 all_glyphs = [] for i_start, i_end in intervals: for code_point in range(i_start, i_end + 1): f = load_glyph(code_point) if f is None: glyph = GlyphProps(0, 0, 0, 0, 0, 0, total_bitmap_size, code_point) all_glyphs.append((glyph, b'')) continue bitmap = f.glyph.bitmap # Build 4-bit greyscale bitmap (same logic as fontconvert.py). # # FreeType returns the buffer with bitmap.pitch as the row stride # in bytes, which can be negative when the bitmap is stored # bottom-up. Iterating bitmap.buffer linearly assumes # pitch == width and a top-down layout — that holds in the common # case but breaks on padded or flipped bitmaps and corrupts the # output. Walk by (row, col) using the real pitch instead. # # Cache bitmap.buffer in a local — ctypes struct field access # creates a new Python wrapper object each time, so re-evaluating # it per pixel is catastrophically slow. pixels4g = [] px = 0 buf = bitmap.buffer abs_pitch = abs(bitmap.pitch) for y in range(bitmap.rows): row_offset = y * abs_pitch if bitmap.pitch >= 0 else (bitmap.rows - 1 - y) * abs_pitch for x in range(bitmap.width): v = buf[row_offset + x] if x % 2 == 0: px = (v >> 4) else: px = px | (v & 0xF0) pixels4g.append(px) px = 0 if bitmap.width % 2 > 0: pixels4g.append(px) px = 0 # Downsample to 2-bit bitmap pixels2b = [] px = 0 pitch = (bitmap.width // 2) + (bitmap.width % 2) for y in range(bitmap.rows): for x in range(bitmap.width): px = px << 2 bm = pixels4g[y * pitch + (x // 2)] bm = (bm >> ((x % 2) * 4)) & 0xF if bm >= 12: px += 3 elif bm >= 8: px += 2 elif bm >= 4: px += 1 if (y * bitmap.width + x) % 4 == 3: pixels2b.append(px) px = 0 if (bitmap.width * bitmap.rows) % 4 != 0: # Outer parens are for clarity: in Python `*` binds tighter # than `<<`, so the original `px << (4 - … % 4) * 2` already # evaluates as `px << ((4 - … % 4) * 2)`. Match the explicit # bracketing here so the shift width is obvious at a glance, # mirroring the inner-loop style in fontconvert.py. px = px << ((4 - (bitmap.width * bitmap.rows) % 4) * 2) pixels2b.append(px) packed = bytes(pixels2b) glyph = GlyphProps( width=bitmap.width, height=bitmap.rows, advance_x=fp4_from_ft16_16(f.glyph.linearHoriAdvance), left=f.glyph.bitmap_left, top=f.glyph.bitmap_top, data_length=len(packed), data_offset=total_bitmap_size, code_point=code_point, ) total_bitmap_size += len(packed) all_glyphs.append((glyph, packed)) # Get font metrics from pipe character (same heuristic as fontconvert.py) load_glyph(ord('|')) advanceY = norm_ceil(face.size.height) ascender = norm_ceil(face.size.ascender) descender = norm_floor(face.size.descender) print(f" [{style_label}] Metrics: advanceY={advanceY}, ascender={ascender}, descender={descender}", file=sys.stderr) print(f" [{style_label}] Bitmap: {total_bitmap_size} bytes ({total_bitmap_size / 1024:.1f} KB)", file=sys.stderr) # --- Extract kerning and ligatures --- ppem = size * 150.0 / 72.0 all_cps = set(g.code_point for g, _ in all_glyphs) kern_map = extract_kerning_fonttools(fontfile, all_cps, ppem) # SMP codepoints (> U+FFFF) cannot be stored in the uint16 kern codepoint # field; drop them before class derivation to avoid a downstream # struct.error when packing the binary kern tables. kern_map = {(lcp, rcp): v for (lcp, rcp), v in kern_map.items() if lcp <= 0xFFFF and rcp <= 0xFFFF} print(f" [{style_label}] Kerning: {len(kern_map)} pairs extracted", file=sys.stderr) (kern_left_classes, kern_right_classes, kern_matrix, kern_left_class_count, kern_right_class_count) = derive_kern_classes(kern_map) if kern_map: matrix_size = kern_left_class_count * kern_right_class_count entries_size = (len(kern_left_classes) + len(kern_right_classes)) * 3 print(f" [{style_label}] Kerning classes: {kern_left_class_count} left, {kern_right_class_count} right, " f"{matrix_size + entries_size} bytes", file=sys.stderr) # SMP codepoints in ligature inputs / outputs are filtered inside # extract_ligatures_fonttools (see the codepoints_set filter), so every # entry returned here is already 16-bit safe. ligature_pairs = extract_ligatures_fonttools(fontfile, all_cps) if len(ligature_pairs) > 255: print(f" [{style_label}] WARNING: {len(ligature_pairs)} ligature pairs exceeds uint8_t max (255), truncating", file=sys.stderr) ligature_pairs = ligature_pairs[:255] print(f" [{style_label}] Ligatures: {len(ligature_pairs)} pairs", file=sys.stderr) return StyleRasterData( style_id=style_id, intervals=intervals, all_glyphs=all_glyphs, total_bitmap_size=total_bitmap_size, advanceY=advanceY, ascender=ascender, descender=descender, kern_left_classes=kern_left_classes, kern_right_classes=kern_right_classes, kern_matrix=kern_matrix, kern_left_class_count=kern_left_class_count, kern_right_class_count=kern_right_class_count, ligature_pairs=ligature_pairs, ) # --- Binary packing helpers --- # EpdGlyph struct: 16 bytes, little-endian GLYPH_STRUCT_FORMAT = " StyleRasterData for style_id in sorted(style_fonts.keys()): fontfile = style_fonts[style_id] print(f" Rasterizing style {style_id}...", file=sys.stderr) raster_data[style_id] = rasterize_font_style( fontfile, size, intervals, style_id=style_id, force_autohint=force_autohint) # Pack binary sections for each style packed_sections = {} # style_id -> tuple of section bytearrays for style_id, sd in raster_data.items(): packed_sections[style_id] = pack_style_sections(sd) # Calculate data offsets (after header + TOC) data_start = HEADER_SIZE + style_count * STYLE_TOC_ENTRY_SIZE current_offset = data_start style_offsets = {} # style_id -> absolute file offset for style_id in sorted(packed_sections.keys()): style_offsets[style_id] = current_offset current_offset += style_sections_total_size(packed_sections[style_id]) # Build global header # V4 header: magic(8) + version(2) + flags(2) + styleCount(1) + reserved(19) = 32 header = struct.pack("<8sHHB19s", MAGIC, CPFONT_VERSION, flags, style_count, bytes(19)) assert len(header) == HEADER_SIZE # Build style TOC entries # Each entry: styleId(1) + pad(3) + intervalCount(4) + glyphCount(4) + # advanceY(1) + ascender(2) + descender(2) + kernL(2) + kernR(2) + # kernLCls(1) + kernRCls(1) + ligCount(1) + dataOffset(4) + reserved(4) = 32 STYLE_TOC_FORMAT = " 255: print(f"ERROR: advanceY ({sd.advanceY}) exceeds uint8 range for " f"style {style_id} size {size}. This likely means the font " f"size is too large for this format.", file=sys.stderr) sys.exit(1) toc_data += struct.pack(STYLE_TOC_FORMAT, style_id, len(sd.intervals), len(sd.all_glyphs), sd.advanceY, sd.ascender, sd.descender, len(sd.kern_left_classes), len(sd.kern_right_classes), sd.kern_left_class_count, sd.kern_right_class_count, len(sd.ligature_pairs), style_offsets[style_id]) # Write output os.makedirs(os.path.dirname(output_path) if os.path.dirname(output_path) else ".", exist_ok=True) total_file_size = 0 with open(output_path, "wb") as f: f.write(header) f.write(toc_data) for style_id in sorted(packed_sections.keys()): for section in packed_sections[style_id]: f.write(section) total_file_size = f.tell() # Print summary print(f" Output: {output_path} (v4, {style_count} styles)", file=sys.stderr) print(f" Header+TOC: {HEADER_SIZE + len(toc_data)} bytes", file=sys.stderr) for style_id in sorted(raster_data.keys()): sd = raster_data[style_id] secs = packed_sections[style_id] style_names = {0: "regular", 1: "bold", 2: "italic", 3: "bolditalic"} sname = style_names.get(style_id, str(style_id)) ssize = style_sections_total_size(secs) print(f" {sname}: {len(sd.all_glyphs)} glyphs, {len(sd.intervals)} intervals, " f"{ssize} bytes", file=sys.stderr) print(f" Total: {total_file_size} bytes ({total_file_size / 1024 / 1024:.2f} MB)", file=sys.stderr) return total_file_size def main(): parser = argparse.ArgumentParser( description="Generate .cpfont files for SD card font loading.", formatter_class=argparse.RawDescriptionHelpFormatter, epilog=f"Available interval presets: {', '.join(sorted(INTERVAL_PRESETS.keys()))}" ) # Font file (positional, optional for multi-style mode) parser.add_argument("fontfile", nargs="?", default=None, help="Path to the font file (single-style mode).") parser.add_argument("--intervals", dest="intervals", help="Comma-separated interval presets (e.g., 'latin-ext,greek,cyrillic').") parser.add_argument("--size", type=int, dest="size", help="Single font size to generate.") parser.add_argument("--sizes", dest="sizes", help="Comma-separated sizes (e.g., '12,14,16,18').") parser.add_argument("--style", dest="style", default="regular", choices=["regular", "bold", "italic", "bolditalic"], help="Font style for single-style mode (default: regular).") parser.add_argument("--name", dest="name", help="Font family name for output filenames (default: derived from font filename).") parser.add_argument("--force-autohint", dest="force_autohint", action="store_true", help="Force FreeType auto-hinter instead of native font hinting.") parser.add_argument("-o", "--output", dest="output", help="Output file path (for single-size mode).") parser.add_argument("--output-dir", dest="output_dir", help="Output directory for multi-size mode.") parser.add_argument("--list-presets", action="store_true", help="List available interval presets and exit.") # Multi-style mode: per-style font file arguments (generates v4 .cpfont) parser.add_argument("--regular", dest="font_regular", help="Font file for regular style (enables multi-style v4 mode).") parser.add_argument("--bold", dest="font_bold", help="Font file for bold style.") parser.add_argument("--italic", dest="font_italic", help="Font file for italic style.") parser.add_argument("--bolditalic", dest="font_bolditalic", help="Font file for bold-italic style.") args = parser.parse_args() if args.list_presets: print("Available interval presets:") for name, ranges in sorted(INTERVAL_PRESETS.items()): total = sum(e - s + 1 for s, e in ranges) print(f" {name:15s} {len(ranges)} range(s), ~{total} codepoints") sys.exit(0) # Detect multi-style mode style_fonts = {} if args.font_regular: style_fonts[0] = args.font_regular if args.font_bold: style_fonts[1] = args.font_bold if args.font_italic: style_fonts[2] = args.font_italic if args.font_bolditalic: style_fonts[3] = args.font_bolditalic is_multistyle = len(style_fonts) > 0 fontfile = args.fontfile # Require --intervals if not args.intervals: print("Error: --intervals is required (e.g., --intervals latin-ext,greek,cyrillic)", file=sys.stderr) print(f"Available presets: {', '.join(sorted(INTERVAL_PRESETS.keys()))}", file=sys.stderr) sys.exit(1) intervals = resolve_intervals(args.intervals) # Determine sizes if args.sizes: sizes = [int(s.strip()) for s in args.sizes.split(",")] elif args.size: sizes = [args.size] else: print("Error: --size or --sizes is required", file=sys.stderr) sys.exit(1) # Validate early: single-style mode requires a font file if not is_multistyle and not fontfile: print("Error: fontfile is required in single-style mode", file=sys.stderr) sys.exit(1) # Determine font name if args.name: font_name = args.name elif is_multistyle: # Derive from the regular font file ref_file = style_fonts[min(style_fonts.keys())] base = os.path.splitext(os.path.basename(ref_file))[0] for suffix in ["-Regular", "-Bold", "-Italic", "-BoldItalic", "-regular", "-bold", "-italic", "-bolditalic"]: if base.endswith(suffix): base = base[:-len(suffix)] break font_name = base else: base = os.path.splitext(os.path.basename(fontfile))[0] for suffix in ["-Regular", "-Bold", "-Italic", "-BoldItalic", "-regular", "-bold", "-italic", "-bolditalic"]: if base.endswith(suffix): base = base[:-len(suffix)] break font_name = base if not is_multistyle: # Single font file provided: wrap as a single-style v4 font style_map = {"regular": 0, "bold": 1, "italic": 2, "bolditalic": 3} style_fonts[style_map[args.style]] = fontfile # Always generate v4 format if args.output and len(sizes) != 1: print("Error: --output can only be used with a single size", file=sys.stderr) sys.exit(1) output_dir = args.output_dir if args.output_dir else f"{font_name}/" total_size = 0 for sz in sizes: if args.output and len(sizes) == 1: output_path = args.output else: filename = f"{font_name}_{sz}.cpfont" output_path = os.path.join(output_dir, filename) print(f"Generating {output_path} (size {sz}, {len(style_fonts)} style(s), v4)...", file=sys.stderr) total_size += generate_cpfont_multistyle( style_fonts, sz, intervals, output_path, force_autohint=args.force_autohint) print(f"\nTotal: {len(sizes)} files, {total_size / 1024 / 1024:.2f} MB", file=sys.stderr) if __name__ == "__main__": main()