Pdf text reflow (#36)

* Implemented PDF reflow mode by introducing a mechanism to convert PDF content to Markdown/HTML for viewing in the EPUB reader.

Specific changes include:
- Added `PdfReflowGenerator` and `PdfToMarkdownGenerator` to handle PDF text extraction and conversion to reflowable formats.
- Updated `MainViewModel` with `toggleReflowMode` logic to switch between original PDF and reflowed views.
- Modified `RecentFileEntity` and `RecentFileDao` to persist user reflow preferences, including a Room database migration (v12 to v13).
- Updated `PdfViewerScreen` and `EpubReaderControls` to include UI options for toggling reflow mode.
- Integrated reflow preference check into the book opening workflow to automatically load the preferred view.
- Updated `AppNavigation` and `EpubReaderScreen` to support the new view switching state.

* Implemented background processing and incremental loading for PDF reflow mode.

- Added `reflowProgress` to `MainViewModel` to track and display PDF-to-Markdown conversion progress in the UI.
- Refactored `PdfToMarkdownGenerator` to generate a skeleton EPUB structure immediately while processing page content (text and images) asynchronously.
- Switched PDF text extraction to use `PDFBox` with optimized memory settings and JPEG compression for images.
- Implemented priority page processing in reflow mode, starting with the user's current page.
- Added "Clear Reflow Cache" debug option to the Home Screen.
- Enhanced `BookPaginator` to support lazy loading of chapter content from disk and improved cache hit detection.

* Refactored PDF Reflow Mode to generate standalone Markdown files instead of temporary EPUB books.

* perf(reflow): optimize PDF-to-Markdown conversion and fix viewing lag

- Re-architected PdfToMarkdownGenerator to use a single-pass stream (O(N) complexity), fixing performance bottlenecks and timeouts on large PDFs.
- Implemented "Virtual Chaptering" in SingleFileImporter for Markdown files to split content into page-level HTML files, eliminating UI lag during reading.
- Simplified ReflowWorker to delegate progress tracking and looping to the generator.
- Enhanced PdfViewerScreen with a prominent top-bar progress indicator and a completion snackbar with an "OPEN" action.

* Optimized EPUB parsing performance and fixed PDF viewer UI layout.

- Optimized `EpubParser` by implementing parallel chapter parsing using coroutines and a semaphore to limit concurrency.
- Reduced memory usage in `EpubParser` and `SingleFileImporter` by no longer storing full HTML content in memory for chapters.
- Updated `EpubXMLFileParser` to support an existing `Document` object to avoid redundant Jsoup parsing.
- Fixed an issue in `PdfViewerScreen` where the snackbar was appearing under the bottom app bar.
This commit is contained in:
Aryan 2026-03-07 16:10:34 +05:30 committed by GitHub
parent 8f52549c19
commit 1e879eb604
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
21 changed files with 936 additions and 165 deletions

View file

@ -36,6 +36,10 @@ import java.net.URLDecoder
import java.nio.file.Paths
import java.util.UUID
import java.util.zip.ZipFile
import kotlinx.coroutines.async
import kotlinx.coroutines.awaitAll
import kotlinx.coroutines.sync.Semaphore
import kotlinx.coroutines.sync.withPermit
class EpubParser(private val context: Context) {
data class EpubDocument(
@ -472,94 +476,102 @@ class EpubParser(private val context: Context) {
return UUID.randomUUID().toString()
}
private fun parseUsingSpine(
private suspend fun parseUsingSpine(
spine: Node,
manifestItems: Map<String, EpubManifestItem>,
filesContentMap: Map<String, EpubFile>,
ncxMetadataMap: Map<String, NcxMetadata>
): List<EpubChapter> {
var chapterCounter = 0
val tempChapters = mutableListOf<TempEpubChapter>()
): List<EpubChapter> = withContext(Dispatchers.Default) {
val parsingSemaphore = Semaphore(6)
spine.selectChildTag("itemref")
val spineItems = spine.selectChildTag("itemref")
.ifEmpty { spine.selectChildTag("opf:itemref") }
.mapNotNull { manifestItems[it.getAttribute("idref")] }
.forEach { item ->
val fileBytes = filesContentMap[item.absPath]?.data
if (fileBytes != null) {
if (item.mediaType.startsWith("application/xhtml+xml") ||
item.mediaType.startsWith("text/html") ||
item.absPath.endsWith(".html", ignoreCase = true) ||
item.absPath.endsWith(".xhtml", ignoreCase = true) ||
item.absPath.endsWith(".xml", ignoreCase = true)
val deferredChapters = spineItems.mapIndexed { index, itemRef ->
async {
parsingSemaphore.withPermit {
val idRef = itemRef.getAttribute("idref")
val item = manifestItems[idRef] ?: return@withPermit null
val fileBytes = filesContentMap[item.absPath]?.data ?: return@withPermit null
val mediaType = item.mediaType
val absPath = item.absPath
if (mediaType.startsWith("application/xhtml+xml") ||
mediaType.startsWith("text/html") ||
absPath.endsWith(".html", ignoreCase = true) ||
absPath.endsWith(".xhtml", ignoreCase = true) ||
absPath.endsWith(".xml", ignoreCase = true)
) {
val rawHtml = String(fileBytes, Charsets.UTF_8)
val plainText = Jsoup.parse(rawHtml).text()
val document = Jsoup.parse(rawHtml)
val plainText = document.text()
val parser = EpubXMLFileParser(
fileRelativePath = item.absPath,
fileRelativePath = absPath,
data = fileBytes,
fragmentId = null
)
val res = parser.parseForTitleAndPath()
val res = parser.parseForTitleAndPath(document)
val chapterTitleFromHtml = res.title
val ncxKey = item.absPath.substringBefore('#')
val ncxKey = absPath.substringBefore('#')
val ncxData = ncxMetadataMap[ncxKey]
val isEffectiveInToc = if (ncxMetadataMap.isNotEmpty()) {
ncxData != null
} else {
true
}
val finalChapterTitle = if (ncxData != null && ncxData.title.isNotBlank()) {
ncxData.title
} else {
Timber.d("No NCX title for ${item.absPath}, using HTML title: '$chapterTitleFromHtml'")
chapterTitleFromHtml
}
val finalDepth = ncxData?.depth ?: 0
chapterCounter++
tempChapters.add(
TempEpubChapter(
url = item.absPath,
title = finalChapterTitle,
htmlFilePath = res.effectiveHtmlPath,
chapterIndex = chapterCounter,
plainTextContent = plainText,
htmlContent = rawHtml,
depth = finalDepth,
isInToc = isEffectiveInToc
)
TempEpubChapter(
url = absPath,
title = finalChapterTitle,
htmlFilePath = res.effectiveHtmlPath,
chapterIndex = index + 1,
plainTextContent = plainText,
htmlContent = "", // OPTIMIZATION: Don't store HTML in memory, it's on disk
depth = finalDepth,
isInToc = isEffectiveInToc
)
} else if (item.mediaType.startsWith("image/")) {
} else if (mediaType.startsWith("image/")) {
// Image handling remains similar, but usually small enough
val htmlContent = """
<!DOCTYPE html><html style="margin:0;padding:0;height:100%;"><head><title>Image</title></head><body style="margin:0;padding:0;height:100%;text-align:center;"><img src="${item.absPath}" alt="Image from spine" style="object-fit:contain;width:100%;height:100%;"/></body></html>
<!DOCTYPE html><html style="margin:0;padding:0;height:100%;"><head><title>Image</title></head><body style="margin:0;padding:0;height:100%;text-align:center;"><img src="$absPath" alt="Image from spine" style="object-fit:contain;width:100%;height:100%;"/></body></html>
""".trimIndent()
val ncxKey = item.absPath.substringBefore('#')
val ncxKey = absPath.substringBefore('#')
val ncxData = ncxMetadataMap[ncxKey]
val isEffectiveInToc = if (ncxMetadataMap.isNotEmpty()) ncxData != null else true
chapterCounter++
tempChapters.add(
TempEpubChapter(
url = item.absPath,
title = ncxData?.title ?: "Image",
htmlFilePath = item.absPath,
chapterIndex = chapterCounter,
plainTextContent = "[Image]",
htmlContent = htmlContent,
depth = ncxData?.depth ?: 0,
isInToc = isEffectiveInToc
)
TempEpubChapter(
url = absPath,
title = ncxData?.title ?: "Image",
htmlFilePath = absPath,
chapterIndex = index + 1,
plainTextContent = "[Image]",
htmlContent = htmlContent,
depth = ncxData?.depth ?: 0,
isInToc = isEffectiveInToc
)
} else {
null
}
}
}
}
return tempChapters.map { tempChapter ->
val tempChapters = deferredChapters.toList().awaitAll().filterNotNull()
return@withContext tempChapters.map { tempChapter ->
EpubChapter(
chapterId = generateId(),
absPath = tempChapter.url,
@ -573,7 +585,6 @@ class EpubParser(private val context: Context) {
}.filter { it.htmlFilePath.isNotBlank() }
}
private fun parseEpubImages(
manifestItems: Map<String, EpubManifestItem>,
filesContentMap: Map<String, EpubFile>,