diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt index 33eab118..6f314a46 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt @@ -1,8 +1,14 @@ +@file:OptIn(ExperimentalCoroutinesApi::class) + package ua.acclorite.book_story.data.parser.epub import android.net.Uri import android.util.Log import kotlinx.coroutines.Dispatchers +import kotlinx.coroutines.ExperimentalCoroutinesApi +import kotlinx.coroutines.async +import kotlinx.coroutines.awaitAll +import kotlinx.coroutines.coroutineScope import kotlinx.coroutines.withContext import kotlinx.coroutines.yield import org.jsoup.Jsoup @@ -13,13 +19,19 @@ import ua.acclorite.book_story.domain.model.Chapter import ua.acclorite.book_story.domain.model.ChapterWithText import ua.acclorite.book_story.domain.util.Resource import ua.acclorite.book_story.domain.util.UIText +import ua.acclorite.book_story.presentation.core.util.addAll import ua.acclorite.book_story.presentation.core.util.clearMarkdown import java.io.File +import java.util.concurrent.ConcurrentLinkedQueue import java.util.zip.ZipEntry import java.util.zip.ZipFile import javax.inject.Inject +import kotlin.collections.set private const val EPUB_TAG = "EPUB Parser" +private typealias Title = String + +private val dispatcher = Dispatchers.IO.limitedParallelism(2) class EpubTextParser @Inject constructor( private val documentParser: DocumentParser @@ -35,18 +47,30 @@ class EpubTextParser @Inject constructor( withContext(Dispatchers.IO) { ZipFile(file).use { zip -> - yield() - - zip.entries().asSequence().find { entry -> + val tocEntry = zip.entries().toList().find { entry -> entry.name.endsWith("toc.ncx", ignoreCase = true) - }.apply { - parseEpub(tocEntry = this, zip = zip).let { - if (it == null) { - Log.e(EPUB_TAG, "Could not parse EPUB.") - return@withContext - } - chapters.addAll(it) + } + val opfEntry = zip.entries().toList().find { entry -> + entry.name.endsWith("content.opf", ignoreCase = true) + } + + val chapterEntries = zip.getChapterEntries(opfEntry) + val chapterTitleEntries = zip.getChapterTitleMapFromToc(tocEntry) + + Log.i(EPUB_TAG, "TOC Entry: ${tocEntry?.name ?: "no toc.ncx"}") + Log.i(EPUB_TAG, "OPF Entry: ${opfEntry?.name ?: "no content.opf"}") + Log.i(EPUB_TAG, "Chapter entries, size: ${chapterEntries.size}") + Log.i(EPUB_TAG, "Title entries, size: ${chapterTitleEntries?.size}") + + zip.parseEpub( + chapterEntries = chapterEntries, + chapterTitleEntries = chapterTitleEntries + ).let { + if (it == null || it.isEmpty()) { + Log.e(EPUB_TAG, "Could not parse EPUB (null or empty).") + return@withContext } + chapters.addAll(it) } } } @@ -74,118 +98,218 @@ class EpubTextParser @Inject constructor( * Parses text and chapters from EPUB. * Uses toc.ncx(if present) to retrieve titles, otherwise uses first line as title. * + * @param chapterTitleEntries Titles extracted from toc.ncx. + * @param chapterEntries [ZipEntry]s to parse. + * * @return Null if could not parse. */ - private suspend fun parseEpub(tocEntry: ZipEntry?, zip: ZipFile): List? { - Log.i(EPUB_TAG, "TOC Entry: ${tocEntry?.name ?: "NO TOC"}") - - yield() + @OptIn(ExperimentalCoroutinesApi::class) + private suspend fun ZipFile.parseEpub( + chapterEntries: List, + chapterTitleEntries: Map>? + ): List? { val chapters = mutableListOf() - var chapterTextIndex = -1 + coroutineScope { + val unformattedChapters = ConcurrentLinkedQueue() - val tocContent = tocEntry?.let { - withContext(Dispatchers.IO) { - zip.getInputStream(it) - }.bufferedReader().use { it.readText() } - } - val tocDocument = tocContent?.let { Jsoup.parse(it) } - val chaptersTitles = tocDocument.run { - if (this == null) return@run null - var titles = mutableMapOf>() + // Asynchronously getting all chapters with text + val jobs = chapterEntries.mapIndexed { index, entry -> + async(dispatcher) { + yield() - select("navPoint").forEach { navPoint -> - val title = navPoint.selectFirst("navLabel > text")?.text()?.trim() - ?: return@forEach - val source = navPoint.selectFirst("content")?.attr("src")?.trim().let { - if (it == null) return@forEach - Uri.parse(it).path ?: it - }.substringAfterLast("/") + unformattedChapters.parseZipEntry( + zip = this@parseEpub, + index = index, + entry = entry, + chapterTitleMap = chapterTitleEntries + ) - titles[source] = (titles[source] ?: emptyList()) + title - } - - titles - } - - yield() - - zip.entries().asSequence().sortedBy { - it.name.filter { it.isDigit() }.toIntOrNull() - }.forEach { entry -> - yield() - - if ( - !entry.name.endsWith(".xhtml") - && !entry.name.endsWith(".html") - && !entry.name.endsWith(".htm") - ) return@forEach - - yield() - - val content = zip.getInputStream(entry).bufferedReader().use { it.readText() } - var chapter = documentParser.run { - Jsoup.parse(content).parseDocument() - } - - if (chapter.isEmpty()) { - Log.w(EPUB_TAG, "Chapter ${entry.name} is empty.") - return@forEach - } - - val chapterTitle = getChapterTitleFromToc( - chapterSource = entry.name, - chaptersTitles = chaptersTitles - ).run { - if (this != null) { - return@run this + yield() } - chapter.first().clearMarkdown() } + jobs.awaitAll() - chapter = chapter.dropWhile { - it.clearMarkdown().lowercase() == chapterTitle.lowercase() + // Sorting chapters in correct order + chapters.addAll { + var textIndex = -1 + unformattedChapters.toList() + .sortedBy { it.chapter.index } + .mapIndexed { index, item -> + item.copy( + chapter = item.chapter.copy( + index = index, + startIndex = textIndex + 1, + endIndex = textIndex + item.text.size + ) + ).also { textIndex += item.text.size } + } } - - if (chapter.isEmpty()) { - Log.w(EPUB_TAG, "Chapter ${entry.name} is empty.") - return@forEach - } - - yield() - - chapters.add( - ChapterWithText( - chapter = Chapter( - index = chapters.size, - title = chapterTitle, - startIndex = chapterTextIndex + 1, - endIndex = chapterTextIndex + chapter.size - ), - text = chapter - ) - ) - chapterTextIndex += chapter.size } - yield() - if (chapters.isEmpty()) { - Log.e(EPUB_TAG, "Could not parse file without toc.ncx") return null } return chapters } + /** + * Parses [entry] to get it's text and chapter. + * Adds parsed entry in [ConcurrentLinkedQueue]. + * + * @param zip [ZipFile] of the [entry]. + * @param index Index of the [entry]. + * @param entry [ZipEntry]. + * @param chapterTitleMap Titles from [getChapterTitleMapFromToc]. + */ + private suspend fun ConcurrentLinkedQueue.parseZipEntry( + zip: ZipFile, + index: Int, + entry: ZipEntry, + chapterTitleMap: Map>? + ) { + // Getting all text + val content = zip.getInputStream(entry).bufferedReader().use { it.readText() } + var chapter = documentParser.run { + Jsoup.parse(content).parseDocument() + } + + if (chapter.isEmpty()) { + Log.w(EPUB_TAG, "Chapter ${entry.name} is empty.") + return + } + + // Getting title and removing first line (if matches title) + val chapterTitle = getChapterTitleFromToc( + chapterSource = entry.name, + chapterTitleMap = chapterTitleMap + ).run { + if (this != null) { + return@run this + } + chapter.first().clearMarkdown() + }.also { title -> + chapter = chapter.dropWhile { line -> + line.clearMarkdown().lowercase() == title.lowercase() + } + } + + if (chapter.isEmpty()) { + Log.w(EPUB_TAG, "Chapter ${entry.name} is empty.") + return + } + + add( + ChapterWithText( + Chapter( + index = index, + title = chapterTitle, + startIndex = 0, + endIndex = 0 + ), + text = chapter + ) + ) + } + + /** + * Getting all titles from [tocEntry]. + * + * @return null if [tocEntry] is null. + */ + private suspend fun ZipFile.getChapterTitleMapFromToc( + tocEntry: ZipEntry? + ): Map>? { + val tocContent = tocEntry?.let { + withContext(Dispatchers.IO) { + getInputStream(it) + }.bufferedReader().use { it.readText() } + } + val tocDocument = tocContent?.let { Jsoup.parse(it) } + + if (tocDocument == null) return null + var titleMap = mutableMapOf>() + + tocDocument.select("navPoint").forEach { navPoint -> + val title = navPoint.selectFirst("navLabel > text")?.text()?.trim() + ?: return@forEach + val source = navPoint.selectFirst("content")?.attr("src")?.trim() + .let { + if (it == null) return@forEach + Uri.parse(it).path ?: it + }.substringAfterLast("/") + + titleMap[source] = (titleMap[source] ?: emptyList()) + title + } + + return titleMap + } + + /** + * Getting title from [chapterTitleMap]. + * + * @return Null if did not find matching chapters to the [chapterSource]. + */ private fun getChapterTitleFromToc( chapterSource: String, - chaptersTitles: Map>? + chapterTitleMap: Map>? ): String? { - if (chaptersTitles.isNullOrEmpty()) return null - return chaptersTitles + if (chapterTitleMap.isNullOrEmpty()) return null + return chapterTitleMap .getOrElse(chapterSource.substringAfterLast("/")) { null } ?.joinToString(separator = " / ") ?.ifBlank { null } } + + /** + * Getting all chapter entries. + * If [opfEntry] is not null, then getting chapters from Spine. + * If [opfEntry] is null, then getting chapters from the whole [ZipFile] and manually sorting them. + * + * @param opfEntry OPF entry. May be null. + * + * @return List of chapter entries in correct order (do not reorder). + */ + private fun ZipFile.getChapterEntries(opfEntry: ZipEntry?): List { + opfEntry.let { opfEntry -> + if (opfEntry == null) { + return@let + } + + val opfContent = getInputStream(opfEntry).bufferedReader().use { + it.readText() + } + val document = Jsoup.parse(opfContent) + val zipEntries = entries().toList() + + val manifestItems = document.select("manifest > item").associate { + it.attr("id") to it.attr("href") + } + + document.select("spine > itemref").mapNotNull { itemRef -> + val spineId = itemRef.attr("idref") + val chapterSource = manifestItems[spineId]?.substringAfterLast('/')?.lowercase() + ?: return@mapNotNull null + + zipEntries.find { entry -> + entry.name.substringAfterLast('/').lowercase() == chapterSource + } + }.also { entries -> + if (entries.isEmpty()) return@let + + Log.i(EPUB_TAG, "Successfully parsed OPF to get entries from spine.") + return entries + } + } + + Log.w(EPUB_TAG, "Could not parse OPF, manual filtering.") + return entries().toList().filter { entry -> + listOf(".html", ".htm", ".xhtml").any { + entry.name.endsWith(it, ignoreCase = true) + } + }.sortedBy { + it.name.filter { char -> char.isDigit() }.toBigIntegerOrNull() + } + } } \ No newline at end of file diff --git a/app/src/main/java/ua/acclorite/book_story/presentation/core/util/Extensions.kt b/app/src/main/java/ua/acclorite/book_story/presentation/core/util/Extensions.kt index 55bac524..eb409f34 100644 --- a/app/src/main/java/ua/acclorite/book_story/presentation/core/util/Extensions.kt +++ b/app/src/main/java/ua/acclorite/book_story/presentation/core/util/Extensions.kt @@ -71,4 +71,8 @@ fun Modifier.noRippleClickable( fun String.clearMarkdown(): String { return replace(Regex("_|\\*\\*"), "") +} + +fun MutableList.addAll(calculation: () -> List) { + addAll(calculation()) } \ No newline at end of file