From 302dea1eea737bb6771c227163f3e81068bcf2f8 Mon Sep 17 00:00:00 2001 From: Justin Mitchell Date: Mon, 20 Apr 2026 13:47:41 -0400 Subject: [PATCH] fix: Switch to xpath map for paragraph level syncing in KOSync (#1686) Switch KOReader sync progress mapping from chapter matching to XPath-based mapping. - resolves KOReader positions using real XHTML ancestry paths - supports paragraph-based upload mapping with text offsets where needed - passes the current paragraph index into sync so uploads map back to KOReader more accurately No HTTP client changes are included. No reader-state or resume-flow changes are included. --------- Co-authored-by: jpirnay --- lib/Epub/Epub/Section.cpp | 111 +++- lib/Epub/Epub/Section.h | 6 + .../Epub/parsers/ChapterHtmlSlimParser.cpp | 10 +- lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h | 5 +- lib/KOReaderSync/ChapterXPathResolver.cpp | 563 ++++++++++++++++++ lib/KOReaderSync/ChapterXPathResolver.h | 30 + lib/KOReaderSync/ProgressMapper.cpp | 349 ++++++++--- lib/KOReaderSync/ProgressMapper.h | 17 +- src/activities/reader/EpubReaderActivity.cpp | 47 +- src/activities/reader/EpubReaderActivity.h | 3 + .../reader/KOReaderSyncActivity.cpp | 29 +- src/activities/reader/KOReaderSyncActivity.h | 6 +- 12 files changed, 1058 insertions(+), 118 deletions(-) create mode 100644 lib/KOReaderSync/ChapterXPathResolver.cpp create mode 100644 lib/KOReaderSync/ChapterXPathResolver.h diff --git a/lib/Epub/Epub/Section.cpp b/lib/Epub/Epub/Section.cpp index da74ce648..f93457916 100644 --- a/lib/Epub/Epub/Section.cpp +++ b/lib/Epub/Epub/Section.cpp @@ -10,10 +10,15 @@ #include "parsers/ChapterHtmlSlimParser.h" namespace { -constexpr uint8_t SECTION_FILE_VERSION = 19; +constexpr uint8_t SECTION_FILE_VERSION = 20; constexpr uint32_t HEADER_SIZE = sizeof(uint8_t) + sizeof(int) + sizeof(float) + sizeof(bool) + sizeof(uint8_t) + sizeof(uint16_t) + sizeof(uint16_t) + sizeof(uint16_t) + sizeof(bool) + sizeof(bool) + - sizeof(uint8_t) + sizeof(uint32_t) + sizeof(uint32_t); + sizeof(uint8_t) + sizeof(uint32_t) + sizeof(uint32_t) + sizeof(uint32_t); + +struct PageLutEntry { + uint32_t fileOffset; + uint16_t paragraphIndex; +}; } // namespace uint32_t Section::onPageComplete(std::unique_ptr page) { @@ -44,7 +49,8 @@ void Section::writeSectionFileHeader(const int fontId, const float lineCompressi static_assert(HEADER_SIZE == sizeof(SECTION_FILE_VERSION) + sizeof(fontId) + sizeof(lineCompression) + sizeof(extraParagraphSpacing) + sizeof(paragraphAlignment) + sizeof(viewportWidth) + sizeof(viewportHeight) + sizeof(pageCount) + sizeof(hyphenationEnabled) + - sizeof(embeddedStyle) + sizeof(imageRendering) + sizeof(uint32_t) + sizeof(uint32_t), + sizeof(embeddedStyle) + sizeof(imageRendering) + sizeof(uint32_t) + + sizeof(uint32_t) + sizeof(uint32_t), "Header size mismatch"); serialization::writePod(file, SECTION_FILE_VERSION); serialization::writePod(file, fontId); @@ -59,6 +65,7 @@ void Section::writeSectionFileHeader(const int fontId, const float lineCompressi serialization::writePod(file, pageCount); // Placeholder for page count (will be initially 0, patched later) serialization::writePod(file, static_cast(0)); // Placeholder for LUT offset (patched later) serialization::writePod(file, static_cast(0)); // Placeholder for anchor map offset (patched later) + serialization::writePod(file, static_cast(0)); // Placeholder for paragraph LUT offset (patched later) } bool Section::loadSectionFile(const int fontId, const float lineCompression, const bool extraParagraphSpacing, @@ -190,7 +197,7 @@ bool Section::createSectionFile(const int fontId, const float lineCompression, c } writeSectionFileHeader(fontId, lineCompression, extraParagraphSpacing, paragraphAlignment, viewportWidth, viewportHeight, hyphenationEnabled, embeddedStyle, imageRendering); - std::vector lut = {}; + std::vector lut = {}; // Derive the content base directory and image cache path prefix for the parser size_t lastSlash = localPath.find_last_of('/'); @@ -210,7 +217,9 @@ bool Section::createSectionFile(const int fontId, const float lineCompression, c ChapterHtmlSlimParser visitor( epub, tmpHtmlPath, renderer, fontId, lineCompression, extraParagraphSpacing, paragraphAlignment, viewportWidth, viewportHeight, hyphenationEnabled, - [this, &lut](std::unique_ptr page) { lut.emplace_back(this->onPageComplete(std::move(page))); }, + [this, &lut](std::unique_ptr page, const uint16_t paragraphIndex) { + lut.push_back({this->onPageComplete(std::move(page)), paragraphIndex}); + }, embeddedStyle, contentBase, imageBasePath, imageRendering, popupFn, cssParser); Hyphenator::setPreferredLanguage(epub->getLanguage()); success = visitor.parseAndBuildPages(); @@ -230,12 +239,12 @@ bool Section::createSectionFile(const int fontId, const float lineCompression, c const uint32_t lutOffset = file.position(); bool hasFailedLutRecords = false; // Write LUT - for (const uint32_t& pos : lut) { - if (pos == 0) { + for (const auto& entry : lut) { + if (entry.fileOffset == 0) { hasFailedLutRecords = true; break; } - serialization::writePod(file, pos); + serialization::writePod(file, entry.fileOffset); } if (hasFailedLutRecords) { @@ -255,11 +264,18 @@ bool Section::createSectionFile(const int fontId, const float lineCompression, c serialization::writePod(file, page); } - // Patch header with final pageCount, lutOffset, and anchorMapOffset - file.seek(HEADER_SIZE - sizeof(uint32_t) * 2 - sizeof(pageCount)); + const uint32_t paragraphLutOffset = file.position(); + serialization::writePod(file, static_cast(lut.size())); + for (const auto& entry : lut) { + serialization::writePod(file, entry.paragraphIndex); + } + + // Patch header with final pageCount, lutOffset, anchorMapOffset, and paragraphLutOffset + file.seek(HEADER_SIZE - sizeof(uint32_t) * 3 - sizeof(pageCount)); serialization::writePod(file, pageCount); serialization::writePod(file, lutOffset); serialization::writePod(file, anchorMapOffset); + serialization::writePod(file, paragraphLutOffset); // Explicit close() required: member variable persists beyond function scope file.close(); if (cssParser) { @@ -273,7 +289,7 @@ std::unique_ptr Section::loadPageFromSectionFile() { return nullptr; } - file.seek(HEADER_SIZE - sizeof(uint32_t) * 2); + file.seek(HEADER_SIZE - sizeof(uint32_t) * 3); uint32_t lutOffset; serialization::readPod(file, lutOffset); file.seek(lutOffset + sizeof(uint32_t) * currentPage); @@ -294,7 +310,7 @@ std::optional Section::getPageForAnchor(const std::string& anchor) con } const uint32_t fileSize = f.size(); - f.seek(HEADER_SIZE - sizeof(uint32_t)); + f.seek(HEADER_SIZE - sizeof(uint32_t) * 2); uint32_t anchorMapOffset; serialization::readPod(f, anchorMapOffset); if (anchorMapOffset == 0 || anchorMapOffset >= fileSize) { @@ -316,3 +332,74 @@ std::optional Section::getPageForAnchor(const std::string& anchor) con return std::nullopt; } + +std::optional Section::getPageForParagraphIndex(const uint16_t pIndex) const { + FsFile f; + if (!Storage.openFileForRead("SCT", filePath, f)) { + return std::nullopt; + } + + const uint32_t fileSize = f.size(); + f.seek(HEADER_SIZE - sizeof(uint32_t)); + uint32_t paragraphLutOffset; + serialization::readPod(f, paragraphLutOffset); + if (paragraphLutOffset == 0 || paragraphLutOffset >= fileSize) { + return std::nullopt; + } + + f.seek(paragraphLutOffset); + uint16_t count; + serialization::readPod(f, count); + if (count == 0) { + return std::nullopt; + } + + const uint32_t lutEnd = paragraphLutOffset + sizeof(uint16_t) + count * sizeof(uint16_t); + if (lutEnd > fileSize) { + return std::nullopt; + } + + uint16_t resultPage = count - 1; + for (uint16_t i = 0; i < count; i++) { + uint16_t pagePIdx; + serialization::readPod(f, pagePIdx); + if (pagePIdx >= pIndex) { + resultPage = i; + break; + } + } + + return resultPage; +} + +std::optional Section::getParagraphIndexForPage(const uint16_t page) const { + FsFile f; + if (!Storage.openFileForRead("SCT", filePath, f)) { + return std::nullopt; + } + + const uint32_t fileSize = f.size(); + f.seek(HEADER_SIZE - sizeof(uint32_t)); + uint32_t paragraphLutOffset; + serialization::readPod(f, paragraphLutOffset); + if (paragraphLutOffset == 0 || paragraphLutOffset >= fileSize) { + return std::nullopt; + } + + f.seek(paragraphLutOffset); + uint16_t count; + serialization::readPod(f, count); + if (count == 0 || page >= count) { + return std::nullopt; + } + + const uint32_t entryEnd = paragraphLutOffset + sizeof(uint16_t) + (page + 1) * sizeof(uint16_t); + if (entryEnd > fileSize) { + return std::nullopt; + } + + f.seek(paragraphLutOffset + sizeof(uint16_t) + page * sizeof(uint16_t)); + uint16_t pIdx; + serialization::readPod(f, pIdx); + return pIdx; +} diff --git a/lib/Epub/Epub/Section.h b/lib/Epub/Epub/Section.h index 6f002c44d..9870c7b7e 100644 --- a/lib/Epub/Epub/Section.h +++ b/lib/Epub/Epub/Section.h @@ -42,4 +42,10 @@ class Section { // Look up the page number for an anchor id from the section cache file. std::optional getPageForAnchor(const std::string& anchor) const; + + // Look up the page number for a synthetic paragraph index from XPath p[N]. + std::optional getPageForParagraphIndex(uint16_t pIndex) const; + + // Look up the synthetic paragraph index for the given rendered page. + std::optional getParagraphIndexForPage(uint16_t page) const; }; diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp index 9a957d616..93a28104e 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.cpp @@ -163,6 +163,10 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* return; } + if (strcmp(name, "p") == 0) { + self->xpathParagraphIndex++; + } + // Extract class, style, and id attributes std::string classAttr; std::string styleAttr; @@ -428,7 +432,7 @@ void XMLCALL ChapterHtmlSlimParser::startElement(void* userData, const XML_Char* // Create page for image - only break if image won't fit remaining space if (self->currentPage && !self->currentPage->elements.empty() && (self->currentPageNextY + displayHeight > self->viewportHeight)) { - self->completePageFn(std::move(self->currentPage)); + self->completePageFn(std::move(self->currentPage), self->xpathParagraphIndex); self->completedPageCount++; self->currentPage.reset(new Page()); if (!self->currentPage) { @@ -1066,7 +1070,7 @@ bool ChapterHtmlSlimParser::parseAndBuildPages() { anchorData.push_back({std::move(pendingAnchorId), static_cast(completedPageCount)}); pendingAnchorId.clear(); } - completePageFn(std::move(currentPage)); + completePageFn(std::move(currentPage), xpathParagraphIndex); completedPageCount++; currentPage.reset(); currentTextBlock.reset(); @@ -1084,7 +1088,7 @@ void ChapterHtmlSlimParser::addLineToPage(std::shared_ptr line) { } if (currentPageNextY + lineHeight > viewportHeight) { - completePageFn(std::move(currentPage)); + completePageFn(std::move(currentPage), xpathParagraphIndex); completedPageCount++; currentPage.reset(new Page()); currentPageNextY = 0; diff --git a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h index 1cc0ea390..31c8ecc2a 100644 --- a/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h +++ b/lib/Epub/Epub/parsers/ChapterHtmlSlimParser.h @@ -25,7 +25,7 @@ class ChapterHtmlSlimParser { std::shared_ptr epub; const std::string& filepath; GfxRenderer& renderer; - std::function)> completePageFn; + std::function, uint16_t)> completePageFn; std::function popupFn; // Popup callback int depth = 0; int skipUntilDepth = INT_MAX; @@ -74,6 +74,7 @@ class ChapterHtmlSlimParser { int completedPageCount = 0; std::vector> anchorData; std::string pendingAnchorId; // deferred until after previous text block is flushed + uint16_t xpathParagraphIndex = 0; // Footnote link tracking bool insideFootnoteLink = false; @@ -99,7 +100,7 @@ class ChapterHtmlSlimParser { const int fontId, const float lineCompression, const bool extraParagraphSpacing, const uint8_t paragraphAlignment, const uint16_t viewportWidth, const uint16_t viewportHeight, const bool hyphenationEnabled, - const std::function)>& completePageFn, + const std::function, uint16_t)>& completePageFn, const bool embeddedStyle, const std::string& contentBase, const std::string& imageBasePath, const uint8_t imageRendering = 0, const std::function& popupFn = nullptr, const CssParser* cssParser = nullptr) diff --git a/lib/KOReaderSync/ChapterXPathResolver.cpp b/lib/KOReaderSync/ChapterXPathResolver.cpp new file mode 100644 index 000000000..5751a1dbe --- /dev/null +++ b/lib/KOReaderSync/ChapterXPathResolver.cpp @@ -0,0 +1,563 @@ +#include "ChapterXPathResolver.h" + +#include +#include +#include +#include +#include + +#include +#include +#include +#include +#include +#include + +namespace { +std::string stripPrefix(const XML_Char* name) { + if (!name) { + return ""; + } + + const char* local = std::strrchr(name, ':'); + return local ? std::string(local + 1) : std::string(name); +} + +struct NameCounter { + std::string name; + int count; +}; + +struct ParentState { + std::vector children; + + int nextIndex(const std::string& name) { + for (auto& child : children) { + if (child.name == name) { + child.count++; + return child.count; + } + } + + children.push_back({name, 1}); + return 1; + } +}; + +struct PathSegment { + std::string name; + int index; +}; + +std::string buildParagraphXPath(const int spineIndex, const std::vector& path, const int charOffset) { + std::string xpath = "/body/DocFragment[" + std::to_string(spineIndex + 1) + "]/body"; + for (const auto& segment : path) { + xpath += "/" + segment.name + "[" + std::to_string(segment.index) + "]"; + } + if (charOffset > 0) { + xpath += "/text()." + std::to_string(charOffset); + } + return xpath; +} + +size_t countUtf8Codepoints(const XML_Char* data, const int len) { + if (!data || len <= 0) { + return 0; + } + + size_t count = 0; + const unsigned char* ptr = reinterpret_cast(data); + const unsigned char* end = ptr + len; + while (ptr < end) { + utf8NextCodepoint(&ptr); + count++; + } + + return count; +} + +class ParagraphTextCounter final : public Print { + public: + ParagraphTextCounter() { + parser = XML_ParserCreate(nullptr); + if (!parser) { + LOG_ERR("KOX", "Failed to create XML parser"); + return; + } + + XML_SetUserData(parser, this); + XML_SetElementHandler(parser, &ParagraphTextCounter::startElement, &ParagraphTextCounter::endElement); + XML_SetCharacterDataHandler(parser, &ParagraphTextCounter::characterData); + } + + ~ParagraphTextCounter() override { destroyXmlParser(parser); } + + bool ok() const { return parser != nullptr && parseOk; } + + bool finish() { + if (!parser || !parseOk || stopped) { + return parseOk; + } + + if (XML_Parse(parser, "", 0, XML_TRUE) == XML_STATUS_ERROR) { + LOG_ERR("KOX", "Final XML parse error: %s", XML_ErrorString(XML_GetErrorCode(parser))); + parseOk = false; + } + return parseOk; + } + + size_t write(uint8_t c) override { return write(&c, 1); } + + size_t write(const uint8_t* buffer, size_t size) override { + if (!parser || !parseOk || stopped) { + return size; + } + + if (XML_Parse(parser, reinterpret_cast(buffer), static_cast(size), XML_FALSE) != XML_STATUS_OK) { + const enum XML_Error error = XML_GetErrorCode(parser); + if (error != XML_ERROR_ABORTED) { + LOG_ERR("KOX", "XML parse error: %s", XML_ErrorString(error)); + parseOk = false; + } + } + + return size; + } + + size_t totalVisibleChars() const { return visibleChars; } + + private: + static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char**) { + auto* self = static_cast(userData); + self->onStartElement(name); + } + + static void XMLCALL endElement(void* userData, const XML_Char* name) { + auto* self = static_cast(userData); + self->onEndElement(name); + } + + static void XMLCALL characterData(void* userData, const XML_Char* data, const int len) { + auto* self = static_cast(userData); + self->onCharacterData(data, len); + } + + void onStartElement(const XML_Char* rawName) { + const std::string name = stripPrefix(rawName); + + if (!insideBody) { + if (name == "body") { + insideBody = true; + bodyDepth = depth; + } + depth++; + return; + } + + if (name == "p") { + paragraphDepth++; + } + depth++; + } + + void onEndElement(const XML_Char* rawName) { + const std::string name = stripPrefix(rawName); + + depth--; + if (!insideBody) { + return; + } + + if (depth == bodyDepth && name == "body") { + insideBody = false; + return; + } + + if (name == "p" && paragraphDepth > 0) { + paragraphDepth--; + } + } + + void onCharacterData(const XML_Char* data, const int len) { + if (!insideBody || paragraphDepth <= 0 || len <= 0) { + return; + } + + visibleChars += countUtf8Codepoints(data, len); + } + + private: + XML_Parser parser = nullptr; + bool parseOk = true; + bool insideBody = false; + bool stopped = false; + int depth = 0; + int bodyDepth = -1; + int paragraphDepth = 0; + size_t visibleChars = 0; +}; + +class XPathParagraphResolver final : public Print { + public: + explicit XPathParagraphResolver(const int targetParagraph) : targetParagraph(targetParagraph) { + parser = XML_ParserCreate(nullptr); + if (!parser) { + LOG_ERR("KOX", "Failed to create XML parser"); + return; + } + + XML_SetUserData(parser, this); + XML_SetElementHandler(parser, &XPathParagraphResolver::startElement, &XPathParagraphResolver::endElement); + } + + ~XPathParagraphResolver() override { destroyXmlParser(parser); } + + bool ok() const { return parser != nullptr && parseOk; } + + bool finish() { + if (!parser || !parseOk || stopped) { + return parseOk; + } + + if (XML_Parse(parser, "", 0, XML_TRUE) == XML_STATUS_ERROR) { + LOG_ERR("KOX", "Final XML parse error: %s", XML_ErrorString(XML_GetErrorCode(parser))); + parseOk = false; + } + return parseOk; + } + + bool hasMatch() const { return !xpath.empty(); } + const std::string& getXPath() const { return xpath; } + + size_t write(uint8_t c) override { return write(&c, 1); } + + size_t write(const uint8_t* buffer, size_t size) override { + if (!parser || !parseOk || stopped) { + return size; + } + + if (XML_Parse(parser, reinterpret_cast(buffer), static_cast(size), XML_FALSE) != XML_STATUS_OK) { + const enum XML_Error error = XML_GetErrorCode(parser); + if (error != XML_ERROR_ABORTED) { + LOG_ERR("KOX", "XML parse error: %s", XML_ErrorString(error)); + parseOk = false; + } + } + + return size; + } + + int spineIndex = 0; + + private: + static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char**) { + auto* self = static_cast(userData); + self->onStartElement(name); + } + + static void XMLCALL endElement(void* userData, const XML_Char* name) { + auto* self = static_cast(userData); + self->onEndElement(name); + } + + void onStartElement(const XML_Char* rawName) { + const std::string name = stripPrefix(rawName); + + if (!insideBody) { + if (name == "body") { + insideBody = true; + bodyDepth = depth; + parentStates.emplace_back(); + } + depth++; + return; + } + + const int siblingIndex = parentStates.back().nextIndex(name); + path.push_back({name, siblingIndex}); + parentStates.emplace_back(); + + if (name == "p") { + paragraphCount++; + if (paragraphCount == targetParagraph) { + xpath = buildParagraphXPath(spineIndex, path, 0); + stopped = true; + XML_StopParser(parser, XML_FALSE); + } + } + + depth++; + } + + void onEndElement(const XML_Char* rawName) { + const std::string name = stripPrefix(rawName); + + depth--; + if (!insideBody) { + return; + } + + if (depth == bodyDepth && name == "body") { + insideBody = false; + parentStates.clear(); + path.clear(); + return; + } + + if (!path.empty()) { + path.pop_back(); + } + if (!parentStates.empty()) { + parentStates.pop_back(); + } + } + + XML_Parser parser = nullptr; + const int targetParagraph; + bool parseOk = true; + bool insideBody = false; + bool stopped = false; + int depth = 0; + int bodyDepth = -1; + int paragraphCount = 0; + std::vector parentStates; + std::vector path; + std::string xpath; +}; + +class XPathProgressResolver final : public Print { + public: + explicit XPathProgressResolver(const size_t targetVisibleChar) : targetVisibleChar(targetVisibleChar) { + parser = XML_ParserCreate(nullptr); + if (!parser) { + LOG_ERR("KOX", "Failed to create XML parser"); + return; + } + + XML_SetUserData(parser, this); + XML_SetElementHandler(parser, &XPathProgressResolver::startElement, &XPathProgressResolver::endElement); + XML_SetCharacterDataHandler(parser, &XPathProgressResolver::characterData); + } + + ~XPathProgressResolver() override { destroyXmlParser(parser); } + + bool ok() const { return parser != nullptr && parseOk; } + + bool finish() { + if (!parser || !parseOk || stopped) { + return parseOk; + } + + if (XML_Parse(parser, "", 0, XML_TRUE) == XML_STATUS_ERROR) { + LOG_ERR("KOX", "Final XML parse error: %s", XML_ErrorString(XML_GetErrorCode(parser))); + parseOk = false; + } + return parseOk; + } + + bool hasMatch() const { return !xpath.empty(); } + const std::string& getXPath() const { return xpath; } + + size_t write(uint8_t c) override { return write(&c, 1); } + + size_t write(const uint8_t* buffer, size_t size) override { + if (!parser || !parseOk || stopped) { + return size; + } + + if (XML_Parse(parser, reinterpret_cast(buffer), static_cast(size), XML_FALSE) != XML_STATUS_OK) { + const enum XML_Error error = XML_GetErrorCode(parser); + if (error != XML_ERROR_ABORTED) { + LOG_ERR("KOX", "XML parse error: %s", XML_ErrorString(error)); + parseOk = false; + } + } + + return size; + } + + int spineIndex = 0; + + private: + static void XMLCALL startElement(void* userData, const XML_Char* name, const XML_Char**) { + auto* self = static_cast(userData); + self->onStartElement(name); + } + + static void XMLCALL endElement(void* userData, const XML_Char* name) { + auto* self = static_cast(userData); + self->onEndElement(name); + } + + static void XMLCALL characterData(void* userData, const XML_Char* data, const int len) { + auto* self = static_cast(userData); + self->onCharacterData(data, len); + } + + void onStartElement(const XML_Char* rawName) { + const std::string name = stripPrefix(rawName); + + if (!insideBody) { + if (name == "body") { + insideBody = true; + bodyDepth = depth; + parentStates.emplace_back(); + } + depth++; + return; + } + + const int siblingIndex = parentStates.back().nextIndex(name); + path.push_back({name, siblingIndex}); + parentStates.emplace_back(); + + if (name == "p") { + paragraphDepth++; + paragraphVisibleChars = 0; + } + + depth++; + } + + void onEndElement(const XML_Char* rawName) { + const std::string name = stripPrefix(rawName); + + depth--; + if (!insideBody) { + return; + } + + if (depth == bodyDepth && name == "body") { + insideBody = false; + parentStates.clear(); + path.clear(); + return; + } + + if (name == "p" && paragraphDepth > 0) { + paragraphDepth--; + paragraphVisibleChars = 0; + } + + if (!path.empty()) { + path.pop_back(); + } + if (!parentStates.empty()) { + parentStates.pop_back(); + } + } + + void onCharacterData(const XML_Char* data, const int len) { + if (!insideBody || paragraphDepth <= 0 || len <= 0 || stopped) { + return; + } + + const size_t codepointCount = countUtf8Codepoints(data, len); + const size_t nextVisibleChars = visibleChars + codepointCount; + if (targetVisibleChar <= nextVisibleChars) { + const size_t delta = targetVisibleChar - visibleChars; + const int charOffset = static_cast(paragraphVisibleChars + delta); + xpath = buildParagraphXPath(spineIndex, path, std::max(1, charOffset)); + stopped = true; + XML_StopParser(parser, XML_FALSE); + return; + } + + visibleChars = nextVisibleChars; + paragraphVisibleChars += codepointCount; + } + + XML_Parser parser = nullptr; + const size_t targetVisibleChar; + bool parseOk = true; + bool insideBody = false; + bool stopped = false; + int depth = 0; + int bodyDepth = -1; + int paragraphDepth = 0; + size_t visibleChars = 0; + size_t paragraphVisibleChars = 0; + std::vector parentStates; + std::vector path; + std::string xpath; +}; +} // namespace + +std::string ChapterXPathResolver::findXPathForParagraph(const std::shared_ptr& epub, const int spineIndex, + const uint16_t paragraphIndex) { + if (!epub || paragraphIndex == 0 || spineIndex < 0 || spineIndex >= epub->getSpineItemsCount()) { + return ""; + } + + const auto href = epub->getSpineItem(spineIndex).href; + if (href.empty()) { + return ""; + } + + XPathParagraphResolver resolver(paragraphIndex); + if (!resolver.ok()) { + return ""; + } + + resolver.spineIndex = spineIndex; + if (!epub->readItemContentsToStream(href, resolver, 1024) || !resolver.finish()) { + return ""; + } + + if (resolver.hasMatch()) { + LOG_DBG("KOX", "Resolved paragraph %u in spine %d -> %s", paragraphIndex, spineIndex, resolver.getXPath().c_str()); + return resolver.getXPath(); + } + + LOG_DBG("KOX", "Paragraph %u not found in spine %d", paragraphIndex, spineIndex); + return ""; +} + +std::string ChapterXPathResolver::findXPathForProgress(const std::shared_ptr& epub, const int spineIndex, + const float intraSpineProgress) { + if (!epub || spineIndex < 0 || spineIndex >= epub->getSpineItemsCount()) { + return ""; + } + + const auto href = epub->getSpineItem(spineIndex).href; + if (href.empty()) { + return ""; + } + + if (!(intraSpineProgress > 0.0f)) { + return "/body/DocFragment[" + std::to_string(spineIndex + 1) + "]/body"; + } + + ParagraphTextCounter counter; + if (!counter.ok() || !epub->readItemContentsToStream(href, counter, 1024) || !counter.finish()) { + return ""; + } + + const size_t totalVisibleChars = counter.totalVisibleChars(); + if (totalVisibleChars == 0) { + return ""; + } + + const float clamped = std::max(0.0f, std::min(1.0f, intraSpineProgress)); + const size_t targetVisibleChar = + std::max(1, std::min(totalVisibleChars, static_cast(std::ceil(clamped * totalVisibleChars)))); + + XPathProgressResolver resolver(targetVisibleChar); + if (!resolver.ok()) { + return ""; + } + + resolver.spineIndex = spineIndex; + if (!epub->readItemContentsToStream(href, resolver, 1024) || !resolver.finish()) { + return ""; + } + + if (resolver.hasMatch()) { + LOG_DBG("KOX", "Resolved progress %.3f in spine %d -> %s", intraSpineProgress, spineIndex, + resolver.getXPath().c_str()); + return resolver.getXPath(); + } + + LOG_DBG("KOX", "Could not resolve progress %.3f in spine %d", intraSpineProgress, spineIndex); + return ""; +} diff --git a/lib/KOReaderSync/ChapterXPathResolver.h b/lib/KOReaderSync/ChapterXPathResolver.h new file mode 100644 index 000000000..e8d1f4bb7 --- /dev/null +++ b/lib/KOReaderSync/ChapterXPathResolver.h @@ -0,0 +1,30 @@ +#pragma once + +#include + +#include +#include +#include + +class ChapterXPathResolver { + public: + /** + * Resolve the Nth paragraph in a spine item to its real XHTML ancestry path. + * + * Returns a KOReader-compatible path like: + * /body/DocFragment[8]/body/div[2]/section[1]/p[4] + * + * An empty string means parsing failed or the paragraph index was not found. + */ + static std::string findXPathForParagraph(const std::shared_ptr& epub, int spineIndex, uint16_t paragraphIndex); + + /** + * Resolve intra-spine progress to a real XHTML ancestry path plus text offset. + * + * Returns a KOReader-compatible path like: + * /body/DocFragment[8]/body/div[2]/section[1]/p[4]/text().96 + * + * An empty string means parsing failed or the location could not be resolved. + */ + static std::string findXPathForProgress(const std::shared_ptr& epub, int spineIndex, float intraSpineProgress); +}; diff --git a/lib/KOReaderSync/ProgressMapper.cpp b/lib/KOReaderSync/ProgressMapper.cpp index ef542ff40..10633a30b 100644 --- a/lib/KOReaderSync/ProgressMapper.cpp +++ b/lib/KOReaderSync/ProgressMapper.cpp @@ -2,113 +2,294 @@ #include +#include #include +#include + +#include "ChapterXPathResolver.h" +#include "Epub/htmlEntities.h" +#include "Utf8.h" + +namespace { +int parseIndex(const std::string& xpath, const char* prefix, bool last = false) { + const size_t prefixLen = strlen(prefix); + const size_t pos = last ? xpath.rfind(prefix) : xpath.find(prefix); + if (pos == std::string::npos) return -1; + const size_t numStart = pos + prefixLen; + const size_t numEnd = xpath.find(']', numStart); + if (numEnd == std::string::npos || numEnd == numStart) return -1; + int val = 0; + for (size_t i = numStart; i < numEnd; i++) { + if (xpath[i] < '0' || xpath[i] > '9') return -1; + val = val * 10 + (xpath[i] - '0'); + } + return val; +} + +int parseCharOffset(const std::string& xpath) { + const size_t textPos = xpath.rfind("text()"); + if (textPos == std::string::npos) return 0; + const size_t dotPos = xpath.find('.', textPos); + if (dotPos == std::string::npos || dotPos + 1 >= xpath.size()) return 0; + int val = 0; + for (size_t i = dotPos + 1; i < xpath.size(); i++) { + if (xpath[i] < '0' || xpath[i] > '9') return 0; + val = val * 10 + (xpath[i] - '0'); + } + return val; +} + +class ParagraphStreamer final : public Print { + size_t bytesWritten = 0; + bool globalInTag = false; + bool globalInEntity = false; + enum { IDLE, SAW_LT, SAW_LT_P } pState = IDLE; + static constexpr size_t MAX_ENTITY_SIZE = 16; + char entityBuffer[MAX_ENTITY_SIZE] = {}; + size_t entityLen = 0; + + // Forward mode: count paragraphs at a byte offset + size_t fwdTarget; + int fwdResult = 0; + bool fwdCaptured = false; + + // Reverse mode: find position of Nth paragraph + char offset + int revParagraph; + int revChar; + int pCount = 0; + bool revPFound = false; + bool revDone = false; + int revVisChars = 0; // Visible chars counted WITHIN target paragraph + size_t totalVisChars = 0; // Total visible chars in entire file + size_t targetVisChars = 0; // Visible chars from start of file to target position + + void onP() { + pCount++; + if (!revPFound && revParagraph > 0 && pCount >= revParagraph) { + revPFound = true; + revVisChars = 0; + if (revChar <= 0) { + targetVisChars = totalVisChars; + revDone = true; + } + } + } + + void onVisibleCodepoint() { + totalVisChars++; + if (revPFound && !revDone) { + revVisChars++; + if (revVisChars >= revChar) { + targetVisChars = totalVisChars; + revDone = true; + } + } + } + + void onVisibleText(const char* text) { + if (!text) { + return; + } + + const unsigned char* ptr = reinterpret_cast(text); + while (*ptr != 0) { + utf8NextCodepoint(&ptr); + onVisibleCodepoint(); + } + } + + void flushEntityAsLiteral() { + for (size_t i = 0; i < entityLen; i++) { + onVisibleCodepoint(); + } + } + + void finishEntity() { + entityBuffer[entityLen] = '\0'; + const char* resolved = lookupHtmlEntity(entityBuffer, entityLen); + if (resolved) { + onVisibleText(resolved); + } else { + flushEntityAsLiteral(); + } + globalInEntity = false; + entityLen = 0; + } + + public: + explicit ParagraphStreamer(size_t targetByte) : fwdTarget(targetByte), revParagraph(0), revChar(0) {} + ParagraphStreamer(int paragraph, int charOff) : fwdTarget(SIZE_MAX), revParagraph(paragraph), revChar(charOff) {} + + size_t write(uint8_t c) override { + if (!fwdCaptured && bytesWritten >= fwdTarget) { + fwdResult = pCount; + fwdCaptured = true; + } + bytesWritten++; + + if (globalInEntity) { + if (entityLen + 1 < MAX_ENTITY_SIZE) { + entityBuffer[entityLen++] = static_cast(c); + } else { + flushEntityAsLiteral(); + globalInEntity = false; + entityLen = 0; + } + + if (globalInEntity) { + if (c == ';') { + finishEntity(); + } else if (c == '<' || c == ' ' || c == '\t' || c == '\n' || c == '\r') { + flushEntityAsLiteral(); + globalInEntity = false; + entityLen = 0; + } + } + } else if (c == '<') { + globalInTag = true; + } else if (c == '>') { + globalInTag = false; + } else if (!globalInTag) { + if (c == '&') { + globalInEntity = true; + entityBuffer[0] = '&'; + entityLen = 1; + } else { + const bool startsCodepoint = (c & 0xC0) != 0x80; + if (startsCodepoint) { + onVisibleCodepoint(); + } + } + } + + // Paragraph detection + switch (pState) { + case IDLE: + if (c == '<') pState = SAW_LT; + break; + case SAW_LT: + pState = (c == 'p' || c == 'P') ? SAW_LT_P : ((c == '<') ? SAW_LT : IDLE); + break; + case SAW_LT_P: + if (c == '>' || c == '/' || c == ' ' || c == '\t' || c == '\n' || c == '\r') onP(); + pState = (c == '<') ? SAW_LT : IDLE; + break; + } + return 1; + } + + size_t write(const uint8_t* buffer, size_t size) override { + for (size_t i = 0; i < size; i++) write(buffer[i]); + return size; + } + + public: + int paragraphCount() const { return fwdCaptured ? fwdResult : pCount; } + size_t totalBytes() const { return bytesWritten; } + bool found() const { return revDone || revPFound; } + float progress() const { + return totalVisChars > 0 ? static_cast(targetVisChars) / static_cast(totalVisChars) : 0.0f; + } +}; + +bool streamSpine(const std::shared_ptr& epub, int spineIndex, ParagraphStreamer& s) { + const auto href = epub->getSpineItem(spineIndex).href; + return !href.empty() && epub->readItemContentsToStream(href, s, 1024); +} +} // namespace KOReaderPosition ProgressMapper::toKOReader(const std::shared_ptr& epub, const CrossPointPosition& pos) { KOReaderPosition result; - - // Calculate page progress within current spine item - float intraSpineProgress = 0.0f; - if (pos.totalPages > 0) { - intraSpineProgress = static_cast(pos.pageNumber) / static_cast(pos.totalPages); + float intra = (pos.totalPages > 0) ? static_cast(pos.pageNumber) / static_cast(pos.totalPages) : 0.0f; + result.percentage = epub->calculateProgress(pos.spineIndex, intra); + if (pos.hasParagraphIndex && pos.paragraphIndex > 0) { + result.xpath = ChapterXPathResolver::findXPathForParagraph(epub, pos.spineIndex, pos.paragraphIndex); + } else { + result.xpath = ChapterXPathResolver::findXPathForProgress(epub, pos.spineIndex, intra); } - - // Calculate overall book progress (0.0-1.0) - result.percentage = epub->calculateProgress(pos.spineIndex, intraSpineProgress); - - // Generate XPath with estimated paragraph position based on page - result.xpath = generateXPath(pos.spineIndex, pos.pageNumber, pos.totalPages); - - // Get chapter info for logging - const int tocIndex = epub->getTocIndexForSpineIndex(pos.spineIndex); - const std::string chapterName = (tocIndex >= 0) ? epub->getTocItem(tocIndex).title : "unknown"; - - LOG_DBG("ProgressMapper", "CrossPoint -> KOReader: chapter='%s', page=%d/%d -> %.2f%% at %s", chapterName.c_str(), - pos.pageNumber, pos.totalPages, result.percentage * 100, result.xpath.c_str()); - + if (result.xpath.empty()) { + result.xpath = generateXPath(epub, pos.spineIndex, intra); + } + LOG_DBG("PM", "-> KO: spine=%d page=%d/%d %.2f%% %s", pos.spineIndex, pos.pageNumber, pos.totalPages, + result.percentage * 100, result.xpath.c_str()); return result; } CrossPointPosition ProgressMapper::toCrossPoint(const std::shared_ptr& epub, const KOReaderPosition& koPos, int currentSpineIndex, int totalPagesInCurrentSpine) { - CrossPointPosition result; - result.spineIndex = 0; - result.pageNumber = 0; - result.totalPages = 0; - + CrossPointPosition result{}; const size_t bookSize = epub->getBookSize(); - if (bookSize == 0) { - return result; - } + if (bookSize == 0) return result; - // Use percentage-based lookup for both spine and page positioning - // XPath parsing is unreliable since CrossPoint doesn't preserve detailed HTML structure - const size_t targetBytes = static_cast(bookSize * koPos.percentage); - - // Find the spine item that contains this byte position const int spineCount = epub->getSpineItemsCount(); - bool spineFound = false; - for (int i = 0; i < spineCount; i++) { - const size_t cumulativeSize = epub->getCumulativeSpineItemSize(i); - if (cumulativeSize >= targetBytes) { - result.spineIndex = i; - spineFound = true; - break; - } + const float clampedPercentage = std::max(0.0f, std::min(1.0f, koPos.percentage)); + const size_t targetBytes = static_cast(static_cast(bookSize) * clampedPercentage); + + const int docFrag = parseIndex(koPos.xpath, "/body/DocFragment["); + const int xpathP = parseIndex(koPos.xpath, "/p[", true); + const int xpathChar = parseCharOffset(koPos.xpath); + const int xpathSpine = (docFrag >= 1) ? (docFrag - 1) : -1; + if (xpathP > 0) { + result.paragraphIndex = static_cast(xpathP); + result.hasParagraphIndex = true; } - // If no spine item was found (e.g., targetBytes beyond last cumulative size), - // default to the last spine item so we map to the end of the book instead of the beginning. - if (!spineFound && spineCount > 0) { - result.spineIndex = spineCount - 1; - } - - // Estimate page number within the spine item using percentage - if (result.spineIndex < epub->getSpineItemsCount()) { - const size_t prevCumSize = (result.spineIndex > 0) ? epub->getCumulativeSpineItemSize(result.spineIndex - 1) : 0; - const size_t currentCumSize = epub->getCumulativeSpineItemSize(result.spineIndex); - const size_t spineSize = currentCumSize - prevCumSize; - - int estimatedTotalPages = 0; - - // If we are in the same spine, use the known total pages - if (result.spineIndex == currentSpineIndex && totalPagesInCurrentSpine > 0) { - estimatedTotalPages = totalPagesInCurrentSpine; - } - // Otherwise try to estimate based on density from current spine - else if (currentSpineIndex >= 0 && currentSpineIndex < epub->getSpineItemsCount() && totalPagesInCurrentSpine > 0) { - const size_t prevCurrCumSize = - (currentSpineIndex > 0) ? epub->getCumulativeSpineItemSize(currentSpineIndex - 1) : 0; - const size_t currCumSize = epub->getCumulativeSpineItemSize(currentSpineIndex); - const size_t currSpineSize = currCumSize - prevCurrCumSize; - - if (currSpineSize > 0) { - float ratio = static_cast(spineSize) / static_cast(currSpineSize); - estimatedTotalPages = static_cast(totalPagesInCurrentSpine * ratio); - if (estimatedTotalPages < 1) estimatedTotalPages = 1; + if (xpathSpine >= 0 && xpathSpine < spineCount) { + result.spineIndex = xpathSpine; + } else { + for (int i = 0; i < spineCount; i++) { + if (epub->getCumulativeSpineItemSize(i) >= targetBytes) { + result.spineIndex = i; + break; } } + } + if (result.spineIndex >= spineCount) return result; - result.totalPages = estimatedTotalPages; + const size_t prevCum = (result.spineIndex > 0) ? epub->getCumulativeSpineItemSize(result.spineIndex - 1) : 0; + const size_t spineSize = epub->getCumulativeSpineItemSize(result.spineIndex) - prevCum; - if (spineSize > 0 && estimatedTotalPages > 0) { - const size_t bytesIntoSpine = (targetBytes > prevCumSize) ? (targetBytes - prevCumSize) : 0; - const float intraSpineProgress = static_cast(bytesIntoSpine) / static_cast(spineSize); - const float clampedProgress = std::max(0.0f, std::min(1.0f, intraSpineProgress)); - result.pageNumber = static_cast(clampedProgress * estimatedTotalPages); - result.pageNumber = std::max(0, std::min(result.pageNumber, estimatedTotalPages - 1)); + if (result.spineIndex == currentSpineIndex && totalPagesInCurrentSpine > 0) { + result.totalPages = totalPagesInCurrentSpine; + } else if (currentSpineIndex >= 0 && currentSpineIndex < spineCount && totalPagesInCurrentSpine > 0) { + const size_t pc = (currentSpineIndex > 0) ? epub->getCumulativeSpineItemSize(currentSpineIndex - 1) : 0; + const size_t cs = epub->getCumulativeSpineItemSize(currentSpineIndex) - pc; + if (cs > 0) + result.totalPages = std::max( + 1, static_cast(totalPagesInCurrentSpine * static_cast(spineSize) / static_cast(cs))); + } + if (spineSize == 0 || result.totalPages == 0) return result; + + float intra = 0.0f; + if (xpathP > 0) { + ParagraphStreamer s(xpathP, xpathChar); + if (streamSpine(epub, result.spineIndex, s) && s.found()) { + intra = s.progress(); + LOG_DBG("PM", "XPath p[%d]+%d -> %.1f%%", xpathP, xpathChar, intra * 100); } } + if (intra <= 0.0f) { + const size_t bytesIn = (targetBytes > prevCum) ? (targetBytes - prevCum) : 0; + intra = std::max(0.0f, std::min(1.0f, static_cast(bytesIn) / static_cast(spineSize))); + } - LOG_DBG("ProgressMapper", "KOReader -> CrossPoint: %.2f%% at %s -> spine=%d, page=%d", koPos.percentage * 100, - koPos.xpath.c_str(), result.spineIndex, result.pageNumber); - + result.pageNumber = std::max(0, std::min(static_cast(intra * result.totalPages), result.totalPages - 1)); + LOG_DBG("PM", "<- KO: %.2f%% %s -> spine=%d page=%d/%d", koPos.percentage * 100, koPos.xpath.c_str(), + result.spineIndex, result.pageNumber, result.totalPages); return result; } -std::string ProgressMapper::generateXPath(int spineIndex, int pageNumber, int totalPages) { - // Use 0-based DocFragment indices for KOReader - // Use a simple xpath pointing to the DocFragment - KOReader will use the percentage for fine positioning within it - // Avoid specifying paragraph numbers as they may not exist in the target document - return "/body/DocFragment[" + std::to_string(spineIndex) + "]/body"; +std::string ProgressMapper::generateXPath(const std::shared_ptr& epub, int spineIndex, float intra) { + const std::string base = "/body/DocFragment[" + std::to_string(spineIndex + 1) + "]/body"; + if (intra <= 0.0f) return base; + + size_t spineSize = 0; + const auto href = epub->getSpineItem(spineIndex).href; + if (href.empty() || !epub->getItemSize(href, &spineSize) || spineSize == 0) return base; + + ParagraphStreamer s(static_cast(spineSize * std::min(intra, 1.0f))); + if (!streamSpine(epub, spineIndex, s)) return base; + + const int p = s.paragraphCount(); + return (p > 0) ? base + "/p[" + std::to_string(p) + "]" : base; } diff --git a/lib/KOReaderSync/ProgressMapper.h b/lib/KOReaderSync/ProgressMapper.h index 53ff76966..d262622e3 100644 --- a/lib/KOReaderSync/ProgressMapper.h +++ b/lib/KOReaderSync/ProgressMapper.h @@ -8,9 +8,11 @@ * CrossPoint position representation. */ struct CrossPointPosition { - int spineIndex; // Current spine item (chapter) index - int pageNumber; // Current page within the spine item - int totalPages; // Total pages in the current spine item + int spineIndex; // Current spine item (chapter) index + int pageNumber; // Current page within the spine item + int totalPages; // Total pages in the current spine item + uint16_t paragraphIndex = 0; // 1-based synthetic paragraph index from XPath p[N] + bool hasParagraphIndex = false; // True when paragraphIndex was resolved from XPath }; /** @@ -59,9 +61,10 @@ class ProgressMapper { private: /** - * Generate XPath for KOReader compatibility. - * Format: /body/DocFragment[spineIndex+1]/body - * Since CrossPoint doesn't preserve HTML structure, we rely on percentage for positioning. + * Generate a fallback XPath by streaming the spine item's XHTML and resolving + * a paragraph/text position from intra-spine progress. + * Produces a full ancestry path such as + * /body/DocFragment[3]/body/p[42]/text().17. */ - static std::string generateXPath(int spineIndex, int pageNumber, int totalPages); + static std::string generateXPath(const std::shared_ptr& epub, int spineIndex, float intraSpineProgress); }; diff --git a/src/activities/reader/EpubReaderActivity.cpp b/src/activities/reader/EpubReaderActivity.cpp index 21e6f82e2..bd700b761 100644 --- a/src/activities/reader/EpubReaderActivity.cpp +++ b/src/activities/reader/EpubReaderActivity.cpp @@ -10,6 +10,8 @@ #include #include +#include + #include "CrossPointSettings.h" #include "CrossPointState.h" #include "EpubReaderChapterSelectionActivity.h" @@ -63,6 +65,13 @@ void EpubReaderActivity::onEnter() { if (dataSize == 4 || dataSize == 6) { currentSpineIndex = data[0] + (data[1] << 8); nextPageNumber = data[2] + (data[3] << 8); + if (nextPageNumber == UINT16_MAX) { + // UINT16_MAX is an in-memory navigation sentinel for "open previous + // chapter on its last page". It should never be treated as persisted + // resume state after sleep or reopen. + LOG_DBG("ERS", "Ignoring stale last-page sentinel from progress cache"); + nextPageNumber = 0; + } cachedSpineIndex = currentSpineIndex; LOG_DBG("ERS", "Loaded cache: %d, %d", currentSpineIndex, nextPageNumber); } @@ -186,7 +195,8 @@ void EpubReaderActivity::loop() { onGoHome(); } else { currentSpineIndex = epub->getSpineItemsCount() - 1; - nextPageNumber = UINT16_MAX; + nextPageNumber = 0; + pendingPageJump = std::numeric_limits::max(); requestUpdate(); } return; @@ -390,11 +400,19 @@ void EpubReaderActivity::onReaderMenuConfirm(EpubReaderMenuActivity::MenuAction } case EpubReaderMenuActivity::MenuAction::SYNC: { if (KOREADER_STORE.hasCredentials()) { - const int currentPage = section ? section->currentPage : 0; - const int totalPages = section ? section->pageCount : 0; + const int currentPage = section ? section->currentPage : nextPageNumber; + const int totalPages = section ? section->pageCount : cachedChapterTotalPageCount; + std::optional paragraphIndex; + if (section && currentPage >= 0 && currentPage < section->pageCount) { + const uint16_t paragraphPage = + currentPage > 0 ? static_cast(currentPage - 1) : static_cast(currentPage); + if (const auto pIdx = section->getParagraphIndexForPage(paragraphPage)) { + paragraphIndex = *pIdx; + } + } startActivityForResult( std::make_unique(renderer, mappedInput, epub, epub->getPath(), currentSpineIndex, - currentPage, totalPages), + currentPage, totalPages, paragraphIndex), [this](const ActivityResult& result) { if (!result.isCancelled) { const auto& sync = std::get(result.data); @@ -402,6 +420,9 @@ void EpubReaderActivity::onReaderMenuConfirm(EpubReaderMenuActivity::MenuAction RenderLock lock(*this); currentSpineIndex = sync.spineIndex; nextPageNumber = sync.page; + cachedChapterTotalPageCount = 0; // Prevent rescaling sync page + pendingPageJump.reset(); + saveProgress(currentSpineIndex, nextPageNumber, 0); section.reset(); } } @@ -484,7 +505,8 @@ void EpubReaderActivity::pageTurn(bool isForwardTurn) { // We don't want to delete the section mid-render, so grab the semaphore { RenderLock lock(*this); - nextPageNumber = UINT16_MAX; + nextPageNumber = 0; + pendingPageJump = std::numeric_limits::max(); currentSpineIndex--; section.reset(); } @@ -566,10 +588,21 @@ void EpubReaderActivity::render(RenderLock&& lock) { LOG_DBG("ERS", "Cache found, skipping build..."); } - if (nextPageNumber == UINT16_MAX) { - section->currentPage = section->pageCount - 1; + if (pendingPageJump.has_value()) { + if (*pendingPageJump >= section->pageCount && section->pageCount > 0) { + section->currentPage = section->pageCount - 1; + } else { + section->currentPage = *pendingPageJump; + } + pendingPageJump.reset(); } else { section->currentPage = nextPageNumber; + if (section->currentPage < 0) { + section->currentPage = 0; + } else if (section->currentPage >= section->pageCount && section->pageCount > 0) { + LOG_DBG("ERS", "Clamping cached page %d to %d", section->currentPage, section->pageCount - 1); + section->currentPage = section->pageCount - 1; + } } if (!pendingAnchor.empty()) { diff --git a/src/activities/reader/EpubReaderActivity.h b/src/activities/reader/EpubReaderActivity.h index bc6c83540..d786ffed5 100644 --- a/src/activities/reader/EpubReaderActivity.h +++ b/src/activities/reader/EpubReaderActivity.h @@ -3,6 +3,8 @@ #include #include +#include + #include "EpubReaderMenuActivity.h" #include "activities/Activity.h" @@ -11,6 +13,7 @@ class EpubReaderActivity final : public Activity { std::unique_ptr
section = nullptr; int currentSpineIndex = 0; int nextPageNumber = 0; + std::optional pendingPageJump; // Set when navigating to a footnote href with a fragment (e.g. #note1). // Cleared on the next render after the new section loads and resolves it to a page. std::string pendingAnchor; diff --git a/src/activities/reader/KOReaderSyncActivity.cpp b/src/activities/reader/KOReaderSyncActivity.cpp index 4df71e939..9dc2e9be2 100644 --- a/src/activities/reader/KOReaderSyncActivity.cpp +++ b/src/activities/reader/KOReaderSyncActivity.cpp @@ -6,6 +6,7 @@ #include #include +#include "Epub/Section.h" #include "KOReaderCredentialStore.h" #include "KOReaderDocumentId.h" #include "MappedInputManager.h" @@ -14,6 +15,16 @@ #include "fontIds.h" namespace { +CrossPointPosition makeLocalPositionWithParagraph(const int spineIndex, const int page, const int totalPages, + const std::optional& paragraphIndex) { + CrossPointPosition pos = {spineIndex, page, totalPages}; + if (paragraphIndex.has_value()) { + pos.paragraphIndex = *paragraphIndex; + pos.hasParagraphIndex = true; + } + return pos; +} + void syncTimeWithNTP() { // Stop SNTP if already running (can't reconfigure while running) if (esp_sntp_enabled()) { @@ -135,8 +146,21 @@ void KOReaderSyncActivity::performSync() { KOReaderPosition koPos = {remoteProgress.progress, remoteProgress.percentage}; remotePosition = ProgressMapper::toCrossPoint(epub, koPos, currentSpineIndex, totalPagesInSpine); + // If XPath carried a paragraph index, refine the page using the section cache's + // per-page paragraph LUT instead of anchor matching. + if (remotePosition.hasParagraphIndex) { + Section tempSection(epub, remotePosition.spineIndex, renderer); + const auto paragraphPage = tempSection.getPageForParagraphIndex(remotePosition.paragraphIndex); + if (paragraphPage.has_value()) { + LOG_DBG("KOSync", "Paragraph %u resolved to page %d (was %d)", remotePosition.paragraphIndex, *paragraphPage, + remotePosition.pageNumber); + remotePosition.pageNumber = *paragraphPage; + } + } + // Calculate local progress in KOReader format (for display) - CrossPointPosition localPos = {currentSpineIndex, currentPage, totalPagesInSpine}; + CrossPointPosition localPos = + makeLocalPositionWithParagraph(currentSpineIndex, currentPage, totalPagesInSpine, currentParagraphIndex); localProgress = ProgressMapper::toKOReader(epub, localPos); { @@ -162,7 +186,8 @@ void KOReaderSyncActivity::performUpload() { requestUpdateAndWait(); // Convert current position to KOReader format - CrossPointPosition localPos = {currentSpineIndex, currentPage, totalPagesInSpine}; + CrossPointPosition localPos = + makeLocalPositionWithParagraph(currentSpineIndex, currentPage, totalPagesInSpine, currentParagraphIndex); KOReaderPosition koPos = ProgressMapper::toKOReader(epub, localPos); KOReaderProgress progress; diff --git a/src/activities/reader/KOReaderSyncActivity.h b/src/activities/reader/KOReaderSyncActivity.h index afd331ece..29010699b 100644 --- a/src/activities/reader/KOReaderSyncActivity.h +++ b/src/activities/reader/KOReaderSyncActivity.h @@ -3,6 +3,7 @@ #include #include +#include #include "KOReaderSyncClient.h" #include "ProgressMapper.h" @@ -22,13 +23,15 @@ class KOReaderSyncActivity final : public Activity { public: explicit KOReaderSyncActivity(GfxRenderer& renderer, MappedInputManager& mappedInput, const std::shared_ptr& epub, const std::string& epubPath, int currentSpineIndex, - int currentPage, int totalPagesInSpine) + int currentPage, int totalPagesInSpine, + std::optional currentParagraphIndex = std::nullopt) : Activity("KOReaderSync", renderer, mappedInput), epub(epub), epubPath(epubPath), currentSpineIndex(currentSpineIndex), currentPage(currentPage), totalPagesInSpine(totalPagesInSpine), + currentParagraphIndex(currentParagraphIndex), remoteProgress{}, remotePosition{}, localProgress{} {} @@ -58,6 +61,7 @@ class KOReaderSyncActivity final : public Activity { int currentSpineIndex; int currentPage; int totalPagesInSpine; + std::optional currentParagraphIndex; State state = WIFI_SELECTION; std::string statusMessage;