## Summary * **What is the goal of this PR?** * This PR ports over Crossink's handling of large EPUBs that helps prevent crashes during open after the book metadata cache is built. * **What changes are included?** * Removes the post-indexing ZIP-wide CSS discovery pass that built an in-memory map of every ZIP entry. * Reuses `content.opf` parsing to collect declared CSS files without writing spine entries again. * Temporarily releases the loaded book metadata cache while rebuilding CSS for cached books. * Parses CSS before reloading `book.bin` after a fresh cache build, leaving more heap available during CSS rule parsing. ## Additional Context * User reported their EPUB opening fine on Crossink but would crash on Crosspoint. Verified this claim on my own devices. * The crash this addresses happened after `book.bin` was successfully built, when CSS discovery allocated a large `unordered_map` for ~3k EPUB ZIP entries. * Tradeoff: CSS files not declared in `content.opf` are no longer discovered by scanning the full ZIP. This avoids the high-risk memory allocation but improperly formatted EPUBs (ones that don't declare their CSS styles in `content.opf` will render without styling and fallback to inline styles. * User provided epub that was crashing prior to this change: https://www.mediafire.com/file/g57ea4mj13iunvh/Quang+%C3%82m+Chi+Ngo%E1%BA%A1i+-+Nh%C4%A9+C%C4%83n.epub/file --- ### AI Usage While CrossPoint doesn't have restrictions on AI tools in contributing, please be transparent about their usage as it helps set the right context for reviewers. Did you use AI tools to help write this code? _**< YES >**_
130 lines
3.7 KiB
C++
130 lines
3.7 KiB
C++
#pragma once
|
|
#include <HalStorage.h>
|
|
|
|
#include <deque>
|
|
#include <string>
|
|
#include <string_view>
|
|
#include <unordered_map>
|
|
|
|
class ZipFile {
|
|
public:
|
|
struct FileStatSlim {
|
|
uint16_t method; // Compression method
|
|
uint32_t compressedSize; // Compressed size
|
|
uint32_t uncompressedSize; // Uncompressed size
|
|
uint32_t localHeaderOffset; // Offset of local file header
|
|
};
|
|
|
|
struct ZipDetails {
|
|
uint32_t centralDirOffset;
|
|
uint16_t totalEntries;
|
|
bool isSet;
|
|
};
|
|
|
|
// Target for batch uncompressed size lookup (sorted by hash, then len)
|
|
struct SizeTarget {
|
|
uint64_t hash; // FNV-1a 64-bit hash of normalized path
|
|
uint16_t len; // Length of path for collision reduction
|
|
uint16_t index; // Caller's index (e.g. spine index)
|
|
};
|
|
|
|
// FNV-1a 64-bit hash computed from char buffer (no std::string allocation)
|
|
static uint64_t fnvHash64(const char* s, size_t len) {
|
|
uint64_t hash = 14695981039346656037ull;
|
|
for (size_t i = 0; i < len; i++) {
|
|
hash ^= static_cast<uint8_t>(s[i]);
|
|
hash *= 1099511628211ull;
|
|
}
|
|
return hash;
|
|
}
|
|
|
|
private:
|
|
const std::string& filePath;
|
|
HalFile file;
|
|
ZipDetails zipDetails = {0, 0, false};
|
|
std::unordered_map<std::string, FileStatSlim> fileStatSlimCache;
|
|
|
|
// Cursor for sequential central-dir scanning optimization
|
|
uint32_t lastCentralDirPos = 0;
|
|
bool lastCentralDirPosValid = false;
|
|
|
|
bool loadFileStatSlim(const char* filename, FileStatSlim* fileStat);
|
|
long getDataOffset(const FileStatSlim& fileStat);
|
|
bool loadZipDetails();
|
|
|
|
public:
|
|
explicit ZipFile(const std::string& filePath) : filePath(filePath) {}
|
|
~ZipFile() = default;
|
|
// Zip file can be opened and closed by hand in order to allow for quick calculation of inflated file size
|
|
// It is NOT recommended to pre-open it for any kind of inflation due to memory constraints
|
|
bool isOpen() const { return !!file; }
|
|
bool open();
|
|
bool close();
|
|
bool loadAllFileStatSlims();
|
|
bool getInflatedFileSize(const char* filename, size_t* size);
|
|
// Batch lookup: scan ZIP central dir once and fill sizes for matching targets.
|
|
// targets must be sorted by (hash, len). sizes[target.index] receives uncompressedSize.
|
|
// Returns number of targets matched.
|
|
int fillUncompressedSizes(std::deque<SizeTarget>& targets, std::deque<uint32_t>& sizes);
|
|
// Due to the memory required to run each of these, it is recommended to not preopen the zip file for multiple
|
|
// These functions will open and close the zip as needed
|
|
uint8_t* readFileToMemory(const char* filename, size_t* size = nullptr, bool trailingNullByte = false);
|
|
bool readFileToStream(const char* filename, Print& out, size_t chunkSize);
|
|
|
|
template <typename F>
|
|
bool enumerateFilePaths(F&& callback) {
|
|
if (!fileStatSlimCache.empty()) {
|
|
for (const auto& entry : fileStatSlimCache) {
|
|
callback(std::string_view{entry.first});
|
|
}
|
|
return true;
|
|
}
|
|
|
|
const bool wasOpen = isOpen();
|
|
if (!wasOpen && !open()) {
|
|
return false;
|
|
}
|
|
|
|
if (!loadZipDetails()) {
|
|
if (!wasOpen) {
|
|
close();
|
|
}
|
|
return false;
|
|
}
|
|
|
|
file.seek(zipDetails.centralDirOffset);
|
|
|
|
uint32_t sig;
|
|
char itemName[256];
|
|
|
|
while (file.available()) {
|
|
file.read(&sig, 4);
|
|
if (sig != 0x02014b50) {
|
|
break;
|
|
}
|
|
|
|
file.seekCur(24);
|
|
uint16_t nameLen, m, k;
|
|
file.read(&nameLen, 2);
|
|
file.read(&m, 2);
|
|
file.read(&k, 2);
|
|
file.seekCur(12);
|
|
|
|
if (nameLen < sizeof(itemName)) {
|
|
file.read(itemName, nameLen);
|
|
itemName[nameLen] = '\0';
|
|
callback(std::string_view{itemName, nameLen});
|
|
} else {
|
|
file.seekCur(nameLen);
|
|
}
|
|
|
|
file.seekCur(m + k);
|
|
}
|
|
|
|
if (!wasOpen) {
|
|
close();
|
|
}
|
|
return true;
|
|
}
|
|
};
|