diff --git a/lib/Epub/Epub/BookMetadataCache.cpp b/lib/Epub/Epub/BookMetadataCache.cpp index 593ca70d..aa708ede 100644 --- a/lib/Epub/Epub/BookMetadataCache.cpp +++ b/lib/Epub/Epub/BookMetadataCache.cpp @@ -3,13 +3,15 @@ #include #include #include +#include +#include -#include +#include #include "FsHelpers.h" namespace { -constexpr uint8_t BOOK_CACHE_VERSION = 7; +constexpr uint8_t BOOK_CACHE_VERSION = 8; constexpr char bookBinFile[] = "/book.bin"; constexpr char tmpSpineBinFile[] = "/spine.bin.tmp"; constexpr char tmpTocBinFile[] = "/toc.bin.tmp"; @@ -52,11 +54,12 @@ bool BookMetadataCache::beginTocPass() { spineHrefIndex.clear(); spineHrefIndex.resize(spineCount); spineFile.seek(0); + SpineEntry scratch; for (int i = 0; i < spineCount; i++) { - auto entry = readSpineEntry(spineFile); + readSpineEntry(spineFile, scratch); SpineHrefIndexEntry idx; - idx.hrefHash = fnvHash64(entry.href); - idx.hrefLen = static_cast(entry.href.size()); + idx.hrefHash = fnvHash64(scratch.href); + idx.hrefLen = static_cast(scratch.href.size()); idx.spineIndex = static_cast(i); spineHrefIndex[i] = idx; } @@ -97,6 +100,8 @@ bool BookMetadataCache::endWrite() { } bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMetadata& metadata) { + LOG_DBG("BMC", "buildBookBin start: free=%lu contig=%lu", static_cast(esp_get_free_heap_size()), + static_cast(heap_caps_get_largest_free_block(MALLOC_CAP_8BIT | MALLOC_CAP_DEFAULT))); // Open all three files, writing to meta, reading from spine and toc if (!Storage.openFileForWrite("BMC", cachePath + bookBinFile, bookFile)) { return false; @@ -140,11 +145,18 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta serialization::writeString(bookFile, metadata.seriesIndex); serialization::writeString(bookFile, metadata.description); + // Scratch entries reused across all read loops below. Their internal + // std::strings keep the capacity of the longest href/title encountered, so + // each subsequent readSpineEntry / readTocEntry call reuses that allocation + // instead of churning hundreds of small heap blocks during the build. + SpineEntry spineScratch; + TocEntry tocScratch; + // Loop through spine entries, writing LUT positions spineFile.seek(0); for (int i = 0; i < spineCount; i++) { uint32_t pos = spineFile.position(); - auto spineEntry = readSpineEntry(spineFile); + readSpineEntry(spineFile, spineScratch); serialization::writePod(bookFile, pos + lutOffset + lutSize); } @@ -152,7 +164,7 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta tocFile.seek(0); for (int i = 0; i < tocCount; i++) { uint32_t pos = tocFile.position(); - auto tocEntry = readTocEntry(tocFile); + readTocEntry(tocFile, tocScratch); serialization::writePod(bookFile, pos + lutOffset + lutSize + static_cast(spineFile.position())); } @@ -163,14 +175,14 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta // Also count distinct spines referenced by the TOC so tocReliable can be persisted in the // header below — without this, every first-page load on a large book pays an O(tocCount) // seek-heavy scan in Epub::hasReliableToc(). - std::deque spineToTocIndex(spineCount, -1); + std::vector spineToTocIndex(spineCount, -1); int distinctSpinesReferenced = 0; tocFile.seek(0); for (int j = 0; j < tocCount; j++) { - auto tocEntry = readTocEntry(tocFile); - if (tocEntry.spineIndex >= 0 && tocEntry.spineIndex < spineCount) { - if (spineToTocIndex[tocEntry.spineIndex] == -1) { - spineToTocIndex[tocEntry.spineIndex] = static_cast(j); + readTocEntry(tocFile, tocScratch); + if (tocScratch.spineIndex >= 0 && tocScratch.spineIndex < spineCount) { + if (spineToTocIndex[tocScratch.spineIndex] == -1) { + spineToTocIndex[tocScratch.spineIndex] = static_cast(j); distinctSpinesReferenced++; } } @@ -201,23 +213,24 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta // This is O(n*log(m)) instead of O(n*m) while avoiding memory exhaustion. // See: https://github.com/crosspoint-reader/crosspoint-reader/issues/134 - std::deque spineSizes; + std::vector spineSizes; bool useBatchSizes = false; if (spineCount >= LARGE_SPINE_THRESHOLD) { LOG_DBG("BMC", "Using batch size lookup for %d spine items", spineCount); - std::deque targets; + std::vector targets; targets.resize(spineCount); + std::string pathScratch; spineFile.seek(0); for (int i = 0; i < spineCount; i++) { - auto entry = readSpineEntry(spineFile); - std::string path = FsHelpers::normalisePath(entry.href); + readSpineEntry(spineFile, spineScratch); + FsHelpers::normalisePath(spineScratch.href, pathScratch); ZipFile::SizeTarget t; - t.hash = ZipFile::fnvHash64(path.c_str(), path.size()); - t.len = static_cast(path.size()); + t.hash = ZipFile::fnvHash64(pathScratch.c_str(), pathScratch.size()); + t.len = static_cast(pathScratch.size()); t.index = static_cast(i); targets[i] = t; } @@ -239,41 +252,42 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta uint32_t cumSize = 0; spineFile.seek(0); int lastSpineTocIndex = -1; + std::string pathScratch; for (int i = 0; i < spineCount; i++) { - auto spineEntry = readSpineEntry(spineFile); + readSpineEntry(spineFile, spineScratch); - spineEntry.tocIndex = spineToTocIndex[i]; + spineScratch.tocIndex = spineToTocIndex[i]; // Not a huge deal if we don't fine a TOC entry for the spine entry, this is expected behaviour for EPUBs // Logging here is for debugging - if (spineEntry.tocIndex == -1) { + if (spineScratch.tocIndex == -1) { LOG_DBG("BMC", "Warning: Could not find TOC entry for spine item %d: %s, using title from last section", i, - spineEntry.href.c_str()); - spineEntry.tocIndex = lastSpineTocIndex; + spineScratch.href.c_str()); + spineScratch.tocIndex = lastSpineTocIndex; } - lastSpineTocIndex = spineEntry.tocIndex; + lastSpineTocIndex = spineScratch.tocIndex; size_t itemSize = 0; if (useBatchSizes) { itemSize = spineSizes[i]; if (itemSize == 0) { - const std::string path = FsHelpers::normalisePath(spineEntry.href); - if (!zip.getInflatedFileSize(path.c_str(), &itemSize)) { - LOG_ERR("BMC", "Warning: Could not get size for spine item: %s", path.c_str()); + FsHelpers::normalisePath(spineScratch.href, pathScratch); + if (!zip.getInflatedFileSize(pathScratch.c_str(), &itemSize)) { + LOG_ERR("BMC", "Warning: Could not get size for spine item: %s", pathScratch.c_str()); } } } else { - const std::string path = FsHelpers::normalisePath(spineEntry.href); - if (!zip.getInflatedFileSize(path.c_str(), &itemSize)) { - LOG_ERR("BMC", "Warning: Could not get size for spine item: %s", path.c_str()); + FsHelpers::normalisePath(spineScratch.href, pathScratch); + if (!zip.getInflatedFileSize(pathScratch.c_str(), &itemSize)) { + LOG_ERR("BMC", "Warning: Could not get size for spine item: %s", pathScratch.c_str()); } } cumSize += itemSize; - spineEntry.cumulativeSize = cumSize; + spineScratch.cumulativeSize = cumSize; // Write out spine data to book.bin - writeSpineEntry(bookFile, spineEntry); + writeSpineEntry(bookFile, spineScratch); } // Close opened zip file zip.close(); @@ -281,8 +295,8 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta // Loop through toc entries from toc file writing to book.bin tocFile.seek(0); for (int i = 0; i < tocCount; i++) { - auto tocEntry = readTocEntry(tocFile); - writeTocEntry(bookFile, tocEntry); + readTocEntry(tocFile, tocScratch); + writeTocEntry(bookFile, tocScratch); } // Patch tocReliable placeholder in header A @@ -294,6 +308,8 @@ bool BookMetadataCache::buildBookBin(const std::string& epubPath, const BookMeta tocFile.close(); LOG_DBG("BMC", "Successfully built book.bin"); + LOG_DBG("BMC", "buildBookBin end: free=%lu contig=%lu", static_cast(esp_get_free_heap_size()), + static_cast(heap_caps_get_largest_free_block(MALLOC_CAP_8BIT | MALLOC_CAP_DEFAULT))); return true; } @@ -460,20 +476,28 @@ BookMetadataCache::TocEntry BookMetadataCache::getTocEntry(const int index) { return readTocEntry(bookFile); } +void BookMetadataCache::readSpineEntry(FsFile& file, SpineEntry& out) const { + serialization::readString(file, out.href); + serialization::readPod(file, out.cumulativeSize); + serialization::readPod(file, out.tocIndex); +} + +void BookMetadataCache::readTocEntry(FsFile& file, TocEntry& out) const { + serialization::readString(file, out.title); + serialization::readString(file, out.href); + serialization::readString(file, out.anchor); + serialization::readPod(file, out.level); + serialization::readPod(file, out.spineIndex); +} + BookMetadataCache::SpineEntry BookMetadataCache::readSpineEntry(FsFile& file) const { SpineEntry entry; - serialization::readString(file, entry.href); - serialization::readPod(file, entry.cumulativeSize); - serialization::readPod(file, entry.tocIndex); + readSpineEntry(file, entry); return entry; } BookMetadataCache::TocEntry BookMetadataCache::readTocEntry(FsFile& file) const { TocEntry entry; - serialization::readString(file, entry.title); - serialization::readString(file, entry.href); - serialization::readString(file, entry.anchor); - serialization::readPod(file, entry.level); - serialization::readPod(file, entry.spineIndex); + readTocEntry(file, entry); return entry; } diff --git a/lib/Epub/Epub/BookMetadataCache.h b/lib/Epub/Epub/BookMetadataCache.h index fa54d9b0..b32fe9ef 100644 --- a/lib/Epub/Epub/BookMetadataCache.h +++ b/lib/Epub/Epub/BookMetadataCache.h @@ -3,8 +3,8 @@ #include #include -#include #include +#include class BookMetadataCache { public: @@ -65,7 +65,7 @@ class BookMetadataCache { uint16_t hrefLen; // length for collision reduction int16_t spineIndex; }; - std::deque spineHrefIndex; + std::vector spineHrefIndex; bool useSpineHrefIndex = false; static constexpr uint16_t LARGE_SPINE_THRESHOLD = 400; @@ -84,6 +84,11 @@ class BookMetadataCache { uint32_t writeTocEntry(FsFile& file, const TocEntry& entry) const; SpineEntry readSpineEntry(FsFile& file) const; TocEntry readTocEntry(FsFile& file) const; + // Out-parameter overloads reuse the caller's string capacity inside hot + // build loops, eliminating per-iteration std::string allocation churn that + // would otherwise fragment the heap during book.bin construction. + void readSpineEntry(FsFile& file, SpineEntry& out) const; + void readTocEntry(FsFile& file, TocEntry& out) const; public: BookMetadata coreMetadata; diff --git a/lib/FsHelpers/FsHelpers.cpp b/lib/FsHelpers/FsHelpers.cpp index 08ca4460..ba18cb2a 100644 --- a/lib/FsHelpers/FsHelpers.cpp +++ b/lib/FsHelpers/FsHelpers.cpp @@ -2,43 +2,46 @@ #include #include -#include namespace FsHelpers { -std::string normalisePath(const std::string& path) { - std::vector components; - std::string component; - - for (const auto c : path) { - if (c == '/') { - if (!component.empty()) { - if (component == "..") { - if (!components.empty()) { - components.pop_back(); - } - } else { - components.push_back(component); - } - component.clear(); - } +// Process a finalised component in-place, appending it to `out` (preceded by +// '/' if `out` is non-empty) or popping the last component for "..". Used by +// both normalisePath overloads so the parsing rules stay in one place. +static void appendOrPopComponent(std::string& out, const char* compData, size_t compLen) { + if (compLen == 0) return; + if (compLen == 1 && compData[0] == '.') return; + if (compLen == 2 && compData[0] == '.' && compData[1] == '.') { + if (out.empty()) return; + const auto lastSlash = out.find_last_of('/'); + if (lastSlash == std::string::npos) { + out.clear(); } else { - component += c; + out.resize(lastSlash); + } + return; + } + if (!out.empty()) { + out.push_back('/'); + } + out.append(compData, compLen); +} + +void normalisePath(const std::string& path, std::string& out) { + out.clear(); + size_t componentStart = 0; + for (size_t i = 0; i < path.size(); i++) { + if (path[i] == '/') { + appendOrPopComponent(out, path.data() + componentStart, i - componentStart); + componentStart = i + 1; } } + appendOrPopComponent(out, path.data() + componentStart, path.size() - componentStart); +} - if (!component.empty()) { - components.push_back(component); - } - +std::string normalisePath(const std::string& path) { std::string result; - for (const auto& c : components) { - if (!result.empty()) { - result += "/"; - } - result += c; - } - + normalisePath(path, result); return result; } diff --git a/lib/FsHelpers/FsHelpers.h b/lib/FsHelpers/FsHelpers.h index a2113512..b1aa0c41 100644 --- a/lib/FsHelpers/FsHelpers.h +++ b/lib/FsHelpers/FsHelpers.h @@ -7,6 +7,10 @@ namespace FsHelpers { std::string normalisePath(const std::string& path); +// Out-parameter overload that reuses `out`'s capacity and performs the +// normalisation in-place without allocating a temporary components vector. +// Use inside hot loops to keep heap fragmentation bounded. +void normalisePath(const std::string& path, std::string& out); /** * Check if the given filename ends with the specified extension (case-insensitive). diff --git a/lib/ZipFile/ZipFile.cpp b/lib/ZipFile/ZipFile.cpp index bdf5c0e2..c91f2d4f 100644 --- a/lib/ZipFile/ZipFile.cpp +++ b/lib/ZipFile/ZipFile.cpp @@ -295,7 +295,7 @@ bool ZipFile::getInflatedFileSize(const char* filename, size_t* size) { return true; } -int ZipFile::fillUncompressedSizes(std::deque& targets, std::deque& sizes) { +int ZipFile::fillUncompressedSizes(const std::vector& targets, std::vector& sizes) { if (targets.empty()) { return 0; } diff --git a/lib/ZipFile/ZipFile.h b/lib/ZipFile/ZipFile.h index 60c97a4c..76d0af31 100644 --- a/lib/ZipFile/ZipFile.h +++ b/lib/ZipFile/ZipFile.h @@ -1,9 +1,9 @@ #pragma once #include -#include #include #include +#include class ZipFile { public: @@ -64,7 +64,7 @@ class ZipFile { // Batch lookup: scan ZIP central dir once and fill sizes for matching targets. // targets must be sorted by (hash, len). sizes[target.index] receives uncompressedSize. // Returns number of targets matched. - int fillUncompressedSizes(std::deque& targets, std::deque& sizes); + int fillUncompressedSizes(const std::vector& targets, std::vector& sizes); // Due to the memory required to run each of these, it is recommended to not preopen the zip file for multiple // These functions will open and close the zip as needed uint8_t* readFileToMemory(const char* filename, size_t* size = nullptr, bool trailingNullByte = false);