diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/DocumentParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/DocumentParser.kt new file mode 100644 index 00000000..b5efb63a --- /dev/null +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/DocumentParser.kt @@ -0,0 +1,52 @@ +package ua.acclorite.book_story.data.parser + +import kotlinx.coroutines.yield +import org.jsoup.nodes.Document +import javax.inject.Inject + +class DocumentParser @Inject constructor() { + /** + * Parses document to get it's text. + * If [fragment] is not null, searches document for specific [fragment]. + * + * @return Parsed text line by line. + */ + suspend fun Document.parseDocument(fragment: String?): List { + val lines = mutableListOf() + + yield() + + body() + .select("p") + .apply { + forEach { element -> + yield() + + val cleanedText = element.html().replace(Regex("\\n+"), " ") + element.html(cleanedText) + } + + append("\n") + } + + yield() + + body() + .run { + fragment?.let { return@run getElementById(it) ?: this } + this + } + .wholeText() + .lines() + .forEach { line -> + yield() + if (line.isNotBlank()) { + lines.add(line.trim()) + } + } + + yield() + + return lines + } +} \ No newline at end of file diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt index c988692a..09355b26 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/epub/EpubTextParser.kt @@ -4,8 +4,10 @@ import android.net.Uri import android.util.Log import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.withContext +import kotlinx.coroutines.yield import org.jsoup.Jsoup import ua.acclorite.book_story.R +import ua.acclorite.book_story.data.parser.DocumentParser import ua.acclorite.book_story.data.parser.TextParser import ua.acclorite.book_story.domain.model.Chapter import ua.acclorite.book_story.domain.model.ChapterWithText @@ -18,7 +20,9 @@ import javax.inject.Inject private const val EPUB_TAG = "EPUB Parser" -class EpubTextParser @Inject constructor() : TextParser { +class EpubTextParser @Inject constructor( + private val documentParser: DocumentParser +) : TextParser { override suspend fun parse(file: File): Resource> { Log.i(EPUB_TAG, "Started EPUB parsing: ${file.name}.") @@ -26,8 +30,12 @@ class EpubTextParser @Inject constructor() : TextParser { return try { val chapters = mutableListOf() + yield() + withContext(Dispatchers.IO) { ZipFile(file).use { zip -> + yield() + zip.entries().asSequence().find { entry -> entry.name.endsWith("toc.ncx", ignoreCase = true) }.apply { @@ -45,10 +53,14 @@ class EpubTextParser @Inject constructor() : TextParser { chapters.addAll(this) } + + yield() } } } + yield() + if (chapters.isEmpty()) { return Resource.Error(UIText.StringResource(R.string.error_file_empty)) } @@ -71,12 +83,16 @@ class EpubTextParser @Inject constructor() : TextParser { * * @return Null if could not parse. */ - private fun parseWithoutToc(zip: ZipFile): List? { + private suspend fun parseWithoutToc(zip: ZipFile): List? { val chapters = mutableListOf() var chapterTextIndex = -1 var chapterIndex = 1 + yield() + zip.entries().asSequence().sortedBy { it.name }.forEach { entry -> + yield() + if ( !entry.name.endsWith(".xhtml") && !entry.name.endsWith(".html") @@ -85,7 +101,13 @@ class EpubTextParser @Inject constructor() : TextParser { || entry.name.endsWith("container.xml") ) return@forEach - val chapter = zip.parseDocument(entry = entry, fragment = null) + val content = zip.getInputStream(entry) + .bufferedReader() + .use { + it.readText() + } + + val chapter = documentParser.run { Jsoup.parse(content).parseDocument(fragment = null) } if (chapter.isEmpty()) { Log.w(EPUB_TAG, "Chapter ${entry.name} is empty.") return@forEach @@ -106,6 +128,8 @@ class EpubTextParser @Inject constructor() : TextParser { chapterIndex++ } + yield() + if (chapters.isEmpty()) { Log.e(EPUB_TAG, "Could not parse file without toc.ncx") return null @@ -119,7 +143,7 @@ class EpubTextParser @Inject constructor() : TextParser { * * @return Null if could not parse toc.ncx. */ - private fun parseWithToc(tocEntry: ZipEntry, zip: ZipFile): List? { + private suspend fun parseWithToc(tocEntry: ZipEntry, zip: ZipFile): List? { Log.i(EPUB_TAG, "TOC Entry: ${tocEntry.name}") val chapters = mutableListOf() @@ -127,12 +151,18 @@ class EpubTextParser @Inject constructor() : TextParser { var chapterTextIndex = -1 var chapterIndex = 1 - val tocContent = zip.getInputStream(tocEntry) - .bufferedReader() - .use { it.readText() } + yield() + + val tocContent = withContext(Dispatchers.IO) { + zip.getInputStream(tocEntry) + }.bufferedReader().use { it.readText() } val tocDocument = Jsoup.parse(tocContent) + yield() + tocDocument.select("navPoint").forEach { navPoint -> + yield() + val chapterTitle = navPoint.selectFirst("navLabel > text")?.text()?.trim() ?: "Chapter $chapterIndex" val chapterSrc = navPoint.selectFirst("content")?.attr("src")?.trim() @@ -154,11 +184,18 @@ class EpubTextParser @Inject constructor() : TextParser { return null } - val chapter = zip.parseDocument( - entry = this, - fragment = chapterSrc.second - ).dropWhile { - it == chapterTitle // Remove chapter title if present + val content = zip.getInputStream(this) + .bufferedReader() + .use { + it.readText() + } + + val chapter = documentParser.run { + Jsoup.parse(content).parseDocument( + fragment = chapterSrc.second + ).dropWhile { + it == chapterTitle // Remove chapter title if present + } } if (chapter.isEmpty()) { Log.w(EPUB_TAG, "Chapter $chapterTitle is empty.") @@ -182,6 +219,8 @@ class EpubTextParser @Inject constructor() : TextParser { } } + yield() + if (chapters.isEmpty()) { Log.e(EPUB_TAG, "Could not parse text with toc.ncx") return null @@ -194,44 +233,4 @@ class EpubTextParser @Inject constructor() : TextParser { return chapters } - - /** - * Parses [entry] to get it's text. - * - * @return Parsed text line by line. Can have line break issues due to bad [entry] formatting. - */ - private fun ZipFile.parseDocument(entry: ZipEntry, fragment: String?): List { - val lines = mutableListOf() - val content = getInputStream(entry) - .bufferedReader() - .use { - it.readText() - } - - val document = Jsoup.parse(content) - document - .body() - .select("p") - .append("\n") - .forEach { element -> - val cleanedText = element.html().replace(Regex("\\n+"), " ") - element.html(cleanedText) - } - - document - .body() - .run { - fragment?.let { return@run getElementById(it) ?: this } - this - } - .wholeText() - .lines() - .forEach { line -> - if (line.isNotBlank()) { - lines.add(line.trim()) - } - } - - return lines - } } \ No newline at end of file diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/fb2/Fb2TextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/fb2/Fb2TextParser.kt index 1cc74e05..dffc7d89 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/fb2/Fb2TextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/fb2/Fb2TextParser.kt @@ -3,6 +3,7 @@ package ua.acclorite.book_story.data.parser.fb2 import android.util.Log import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.withContext +import kotlinx.coroutines.yield import org.w3c.dom.Element import org.w3c.dom.NodeList import ua.acclorite.book_story.R @@ -38,11 +39,15 @@ class Fb2TextParser @Inject constructor() : TextParser { ) } + yield() + val unformattedLines = mutableListOf() val bodyNode = bodyNodes.item(0) as Element val paragraphNodes = bodyNode.getElementsByTagName("p") for (element in paragraphNodes.asList()) { + yield() + if (element.textContent.isBlank()) { continue } @@ -52,9 +57,13 @@ class Fb2TextParser @Inject constructor() : TextParser { ) } + yield() + val lines = mutableListOf() unformattedLines.forEachIndexed { index, string -> try { + yield() + val line = string.trim() if (index == 0) { @@ -96,10 +105,15 @@ class Fb2TextParser @Inject constructor() : TextParser { } } + yield() + lines.forEach { line -> + yield() formattedLines.add(line.trim()) } + yield() + if (formattedLines.isEmpty()) { return Resource.Error(UIText.StringResource(R.string.error_file_empty)) } diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/htm/HtmTextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/htm/HtmTextParser.kt index 3206b63d..ee513b21 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/htm/HtmTextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/htm/HtmTextParser.kt @@ -1,8 +1,10 @@ package ua.acclorite.book_story.data.parser.htm import android.util.Log +import kotlinx.coroutines.yield import org.jsoup.Jsoup import ua.acclorite.book_story.R +import ua.acclorite.book_story.data.parser.DocumentParser import ua.acclorite.book_story.data.parser.TextParser import ua.acclorite.book_story.domain.model.Chapter import ua.acclorite.book_story.domain.model.ChapterWithText @@ -13,26 +15,17 @@ import javax.inject.Inject private const val HTM_TAG = "HTM Parser" -class HtmTextParser @Inject constructor() : TextParser { +class HtmTextParser @Inject constructor( + private val documentParser: DocumentParser +) : TextParser { override suspend fun parse(file: File): Resource> { Log.i(HTM_TAG, "Started HTM parsing: ${file.name}.") return try { - val lines = mutableListOf() + val lines = documentParser.run { Jsoup.parse(file).parseDocument(null) } - val document = Jsoup.parse(file) - document.select("p").append("\n") - document.select("head > title").remove() - - document - .wholeText() - .lines() - .forEach { line -> - if (line.isNotBlank()) { - lines.add(line.trim()) - } - } + yield() if (lines.isEmpty()) { return Resource.Error(UIText.StringResource(R.string.error_file_empty)) diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/html/HtmlTextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/html/HtmlTextParser.kt index 4f4ced00..e4fddf89 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/html/HtmlTextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/html/HtmlTextParser.kt @@ -1,8 +1,10 @@ package ua.acclorite.book_story.data.parser.html import android.util.Log +import kotlinx.coroutines.yield import org.jsoup.Jsoup import ua.acclorite.book_story.R +import ua.acclorite.book_story.data.parser.DocumentParser import ua.acclorite.book_story.data.parser.TextParser import ua.acclorite.book_story.domain.model.Chapter import ua.acclorite.book_story.domain.model.ChapterWithText @@ -13,26 +15,17 @@ import javax.inject.Inject private const val HTML_TAG = "HTML Parser" -class HtmlTextParser @Inject constructor() : TextParser { +class HtmlTextParser @Inject constructor( + private val documentParser: DocumentParser +) : TextParser { override suspend fun parse(file: File): Resource> { Log.i(HTML_TAG, "Started HTML parsing: ${file.name}.") return try { - val lines = mutableListOf() + val lines = documentParser.run { Jsoup.parse(file).parseDocument(null) } - val document = Jsoup.parse(file) - document.select("p").append("\n") - document.select("head > title").remove() - - document - .wholeText() - .lines() - .forEach { line -> - if (line.isNotBlank()) { - lines.add(line.trim()) - } - } + yield() if (lines.isEmpty()) { return Resource.Error(UIText.StringResource(R.string.error_file_empty)) diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/pdf/PdfTextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/pdf/PdfTextParser.kt index 124c601a..58aa1c04 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/pdf/PdfTextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/pdf/PdfTextParser.kt @@ -3,6 +3,7 @@ package ua.acclorite.book_story.data.parser.pdf import android.util.Log import com.tom_roush.pdfbox.pdmodel.PDDocument import com.tom_roush.pdfbox.text.PDFTextStripper +import kotlinx.coroutines.yield import ua.acclorite.book_story.R import ua.acclorite.book_story.data.parser.TextParser import ua.acclorite.book_story.domain.model.ChapterWithText @@ -20,18 +21,24 @@ class PdfTextParser @Inject constructor() : TextParser { Log.i(PDF_TAG, "Started PDF parsing: ${file.name}.") return try { - val document = PDDocument.load(file) - val strings = mutableListOf() + yield() + + val oldText: String val pdfStripper = PDFTextStripper() pdfStripper.paragraphStart = "
" - val oldText = pdfStripper.getText(document) - .replace("\r", "") + PDDocument.load(file).use { + oldText = pdfStripper.getText(it) + .replace("\r", "") + } - document.close() + yield() + val strings = mutableListOf() val text = oldText.filterIndexed { index, c -> + yield() + if (c == ' ') { oldText[index - 1] != ' ' } else { @@ -39,12 +46,18 @@ class PdfTextParser @Inject constructor() : TextParser { } } + yield() + val unformattedLines = text.split("${pdfStripper.paragraphStart}|\\n".toRegex()) .filter { it.isNotBlank() } + yield() + val lines = mutableListOf() unformattedLines.forEachIndexed { index, string -> try { + yield() + val line = string.trim() if (index == 0) { @@ -86,10 +99,15 @@ class PdfTextParser @Inject constructor() : TextParser { } } + yield() + lines.forEach { line -> + yield() strings.add(line.trim()) } + yield() + if (strings.isEmpty()) { return Resource.Error(UIText.StringResource(R.string.error_file_empty)) } diff --git a/app/src/main/java/ua/acclorite/book_story/data/parser/txt/TxtTextParser.kt b/app/src/main/java/ua/acclorite/book_story/data/parser/txt/TxtTextParser.kt index 42f38d92..89063ebb 100644 --- a/app/src/main/java/ua/acclorite/book_story/data/parser/txt/TxtTextParser.kt +++ b/app/src/main/java/ua/acclorite/book_story/data/parser/txt/TxtTextParser.kt @@ -3,6 +3,7 @@ package ua.acclorite.book_story.data.parser.txt import android.util.Log import kotlinx.coroutines.Dispatchers import kotlinx.coroutines.withContext +import kotlinx.coroutines.yield import ua.acclorite.book_story.R import ua.acclorite.book_story.data.parser.TextParser import ua.acclorite.book_story.domain.model.ChapterWithText @@ -34,6 +35,8 @@ class TxtTextParser @Inject constructor() : TextParser { } } + yield() + if (lines.isEmpty()) { return Resource.Error(UIText.StringResource(R.string.error_file_empty)) }