## Summary * **What is the goal of this PR?** * Fix EPUBs where internal file references are written in URL-style escaped form, like spaces appearing as `%20`, so the reader can find the right files instead of treating those references as missing. * **What changes are included?** * Added a small shared helper that converts those escaped EPUB-internal paths back into their normal filenames before we try to look them up. * Applied that cleanup step across the EPUB parsing flow wherever we resolve internal references, including cover images, manifest items, TOC links, spine entries, and inline HTML images. * Kept the change narrowly focused on EPUB-internal asset resolution rather than changing broader URL or networking behavior. ## Additional Context * The user-facing bug here is that some books package their internal filenames in an escaped form, so a file like `Chapter 1.xhtml` may be referenced more like `Chapter%201.xhtml`. The reader was treating that escaped text as the literal filename, which means otherwise-valid books could lose images, covers, or chapter targets because the lookup no longer matched the real file inside the EPUB. * Risk is intentionally low. The helper only rewrites valid `%XX` escape sequences and leaves malformed input alone, so it should improve compatibility with escaped filenames without broadening the parser’s behavior in unrelated cases. ## Local Testing Performed * This was tested on my device with the user-provided optimized epub that was not rendering images within the text prior to this fix (cover image and chapter headers were rendering fine): [orv_main_baseline.epub.zip](https://github.com/user-attachments/files/28529545/orv_main_baseline.epub.zip) * This was also tested by the user with a local build and the affected epub and confirmed to be working ## Steps for Testing * Try to open the affected epub (linked above) or any epub that has similar percent-encoding on a build prior to this fix. * Apply this fix, clear book cache, and re-open the affected book. * Images should render properly. --- ### AI Usage While CrossPoint doesn't have restrictions on AI tools in contributing, please be transparent about their usage as it helps set the right context for reviewers. Did you use AI tools to help write this code? _**< YES >**_
189 lines
5.2 KiB
C++
189 lines
5.2 KiB
C++
#include "FsHelpers.h"
|
|
|
|
#include <algorithm>
|
|
#include <cctype>
|
|
#include <cstring>
|
|
#include <vector>
|
|
|
|
namespace FsHelpers {
|
|
|
|
namespace {
|
|
bool isHexDigit(const char c) { return (c >= '0' && c <= '9') || (c >= 'a' && c <= 'f') || (c >= 'A' && c <= 'F'); }
|
|
|
|
uint8_t hexValue(const char c) {
|
|
if (c >= '0' && c <= '9') return static_cast<uint8_t>(c - '0');
|
|
if (c >= 'a' && c <= 'f') return static_cast<uint8_t>(10 + (c - 'a'));
|
|
return static_cast<uint8_t>(10 + (c - 'A'));
|
|
}
|
|
} // namespace
|
|
|
|
std::string decodeUriEscapes(const std::string& path) {
|
|
std::string decoded;
|
|
decoded.reserve(path.size());
|
|
|
|
for (size_t i = 0; i < path.size(); i++) {
|
|
if (path[i] == '%' && i + 2 < path.size() && isHexDigit(path[i + 1]) && isHexDigit(path[i + 2])) {
|
|
const uint8_t value = static_cast<uint8_t>((hexValue(path[i + 1]) << 4) | hexValue(path[i + 2]));
|
|
decoded += static_cast<char>(value);
|
|
i += 2;
|
|
continue;
|
|
}
|
|
|
|
decoded += path[i];
|
|
}
|
|
|
|
return decoded;
|
|
}
|
|
|
|
std::string normalisePath(const std::string& path) {
|
|
std::vector<std::string> components;
|
|
std::string component;
|
|
|
|
for (const auto c : path) {
|
|
if (c == '/') {
|
|
if (!component.empty()) {
|
|
if (component == "..") {
|
|
if (!components.empty()) {
|
|
components.pop_back();
|
|
}
|
|
} else {
|
|
components.push_back(component);
|
|
}
|
|
component.clear();
|
|
}
|
|
} else {
|
|
component += c;
|
|
}
|
|
}
|
|
|
|
if (!component.empty()) {
|
|
components.push_back(component);
|
|
}
|
|
|
|
std::string result;
|
|
for (const auto& c : components) {
|
|
if (!result.empty()) {
|
|
result += "/";
|
|
}
|
|
result += c;
|
|
}
|
|
|
|
return result;
|
|
}
|
|
|
|
void sortFileList(std::vector<std::string>& strs) {
|
|
std::sort(begin(strs), end(strs), [](const std::string& str1, const std::string& str2) {
|
|
// Directories first
|
|
bool isDir1 = str1.back() == '/';
|
|
bool isDir2 = str2.back() == '/';
|
|
if (isDir1 != isDir2) return isDir1;
|
|
|
|
// Start naive natural sort
|
|
const char* s1 = str1.c_str();
|
|
const char* s2 = str2.c_str();
|
|
|
|
// Iterate while both strings have characters
|
|
while (*s1 && *s2) {
|
|
// Check if both are at the start of a number
|
|
if (isdigit(*s1) && isdigit(*s2)) {
|
|
// Skip leading zeros and track them
|
|
while (*s1 == '0') s1++;
|
|
while (*s2 == '0') s2++;
|
|
|
|
// Count digits to compare lengths first
|
|
int len1 = 0, len2 = 0;
|
|
while (isdigit(s1[len1])) len1++;
|
|
while (isdigit(s2[len2])) len2++;
|
|
|
|
// Different length so return smaller integer value
|
|
if (len1 != len2) return len1 < len2;
|
|
|
|
// Same length so compare digit by digit
|
|
for (int i = 0; i < len1; i++) {
|
|
if (s1[i] != s2[i]) return s1[i] < s2[i];
|
|
}
|
|
|
|
// Numbers equal so advance pointers
|
|
s1 += len1;
|
|
s2 += len2;
|
|
} else {
|
|
// Regular case-insensitive character comparison
|
|
char c1 = tolower(*s1);
|
|
char c2 = tolower(*s2);
|
|
if (c1 != c2) return c1 < c2;
|
|
s1++;
|
|
s2++;
|
|
}
|
|
}
|
|
|
|
// One string is prefix of other
|
|
return *s1 == '\0' && *s2 != '\0';
|
|
});
|
|
}
|
|
|
|
bool checkFileExtension(std::string_view fileName, const char* extension) {
|
|
const size_t extLen = strlen(extension);
|
|
if (fileName.length() < extLen) {
|
|
return false;
|
|
}
|
|
|
|
const size_t offset = fileName.length() - extLen;
|
|
for (size_t i = 0; i < extLen; i++) {
|
|
if (tolower(static_cast<unsigned char>(fileName[offset + i])) !=
|
|
tolower(static_cast<unsigned char>(extension[i]))) {
|
|
return false;
|
|
}
|
|
}
|
|
return true;
|
|
}
|
|
|
|
bool hasJpgExtension(std::string_view fileName) {
|
|
return checkFileExtension(fileName, ".jpg") || checkFileExtension(fileName, ".jpeg");
|
|
}
|
|
|
|
bool hasPngExtension(std::string_view fileName) { return checkFileExtension(fileName, ".png"); }
|
|
|
|
bool hasBmpExtension(std::string_view fileName) { return checkFileExtension(fileName, ".bmp"); }
|
|
|
|
bool hasGifExtension(std::string_view fileName) { return checkFileExtension(fileName, ".gif"); }
|
|
|
|
bool hasEpubExtension(std::string_view fileName) { return checkFileExtension(fileName, ".epub"); }
|
|
|
|
bool hasXtcExtension(std::string_view fileName) {
|
|
return checkFileExtension(fileName, ".xtc") || checkFileExtension(fileName, ".xtch");
|
|
}
|
|
|
|
bool hasTxtExtension(std::string_view fileName) { return checkFileExtension(fileName, ".txt"); }
|
|
|
|
bool hasMarkdownExtension(std::string_view fileName) { return checkFileExtension(fileName, ".md"); }
|
|
|
|
bool hasCssExtension(std::string_view fileName) { return checkFileExtension(fileName, ".css"); }
|
|
|
|
std::string extractFolderPath(const std::string& filePath) {
|
|
const auto lastSlash = filePath.find_last_of('/');
|
|
if (lastSlash == std::string::npos || lastSlash == 0) {
|
|
return "/";
|
|
}
|
|
return filePath.substr(0, lastSlash);
|
|
}
|
|
|
|
void sanitizePathComponentForFat32(const char* input, char* output, size_t maxLen) {
|
|
if (maxLen == 0) {
|
|
return;
|
|
}
|
|
|
|
size_t i = 0;
|
|
for (; i < maxLen - 1 && input[i] != '\0'; i++) {
|
|
const char c = input[i];
|
|
if (c == '\\' || c == '/' || c == ':' || c == '*' || c == '?' || c == '"' || c == '<' || c == '>' || c == '|' ||
|
|
c == ' ' || (c > 0x00 && c <= 0x1f)) {
|
|
output[i] = '-';
|
|
} else {
|
|
output[i] = c;
|
|
}
|
|
}
|
|
output[i] = '\0';
|
|
}
|
|
|
|
} // namespace FsHelpers
|