commit 62dca60b0bd3bba8bf74e6543ba635f15be9e72e equwal <truex@equwal.com> 2026-09-20 22:48:03 -0700 Read epub pages as XHTML; add an on-device job test A test on a real device found subtitles with text from the wrong passage. The speech model heard the correct words. The epub reader lost two thirds of the book: the HTML parser takes a self-closing <title/> as an open title and reads the start of a large page as its text. BookText now reads each page as XML first. The HTML parser stays as the fallback. OnDeviceJobTest runs the full job on hardware without the screen: decode, transcribe, align, write the .srt, then a cached second run. The media is local only. ExcerptTest reproduces the device failure on the JVM.
.gitignore | 3 + app/build.gradle.kts | 3 + .../kotlin/space/subread/app/OnDeviceJobTest.kt | 79 ++++++++++++++++++++++ .../src/main/kotlin/space/subread/core/BookText.kt | 13 ++-- .../src/test/kotlin/space/subread/core/BookTest.kt | 17 +++++ .../test/kotlin/space/subread/core/ExcerptTest.kt | 41 +++++++++++ gradle/libs.versions.toml | 4 ++ 7 files changed, 156 insertions(+), 4 deletions(-)
diff --git a/.gitignore b/.gitignore index ee37da8..607f8d1 100644 --- a/.gitignore +++ b/.gitignore @@ -12,5 +12,8 @@ captures/ # that generated them. Tests skip this directory when it is missing. core/src/test/resources/golden-local/ +# On-device test media. Bring your own pair; the test skips without it. +app/src/androidTest/assets/local/ + # Downloaded at build time, never committed. app/src/main/assets/models/ diff --git a/app/build.gradle.kts b/app/build.gradle.kts index fb0509e..c3fb81b 100644 --- a/app/build.gradle.kts +++ b/app/build.gradle.kts @@ -22,6 +22,7 @@ android { targetSdk = 36 versionCode = (providers.gradleProperty("versionCode").orNull ?: "1").toInt() versionName = providers.gradleProperty("versionName").orNull ?: "0.1.0" + testInstrumentationRunner = "androidx.test.runner.AndroidJUnitRunner" // Every phone worth transcribing on is 64-bit ARM; the speech library // is built for ARMv8.2 specifically (see src/main/cpp/CMakeLists.txt). @@ -82,4 +83,6 @@ dependencies { implementation(libs.androidx.lifecycle.runtime.ktx) implementation(libs.androidx.lifecycle.runtime.compose) implementation(libs.kotlinx.coroutines.android) + androidTestImplementation(libs.androidx.test.runner) + androidTestImplementation(libs.androidx.test.junit) } diff --git a/app/src/androidTest/kotlin/space/subread/app/OnDeviceJobTest.kt b/app/src/androidTest/kotlin/space/subread/app/OnDeviceJobTest.kt new file mode 100644 index 0000000..23f11bf --- /dev/null +++ b/app/src/androidTest/kotlin/space/subread/app/OnDeviceJobTest.kt @@ -0,0 +1,79 @@ +package space.subread.app + +import android.net.Uri +import android.util.Log +import androidx.test.ext.junit.runners.AndroidJUnit4 +import androidx.test.platform.app.InstrumentationRegistry +import org.junit.Assert.assertEquals +import org.junit.Assert.assertNotNull +import org.junit.Assert.assertTrue +import org.junit.Assume.assumeTrue +import org.junit.Test +import org.junit.runner.RunWith +import space.subread.app.job.Job +import space.subread.app.job.Phase +import space.subread.app.job.TranscriptStore +import java.io.File + +/** + * The whole job on real hardware: decode, transcribe with the native speech + * model, align, write the subtitles. No screen is needed. + * + * The media is not in the repository. Put an audio file and its book in + * app/src/androidTest/assets/local/ (gitignored); the test skips without them. + */ +@RunWith(AndroidJUnit4::class) +class OnDeviceJobTest { + + @Test + fun anAudiobookAndItsBookBecomeSubtitles() { + val instrumentation = InstrumentationRegistry.getInstrumentation() + val names = instrumentation.context.assets.list("local").orEmpty() + val audioName = names.firstOrNull { it.endsWith(".mp3") || it.endsWith(".m4b") || it.endsWith(".m4a") } + val bookName = names.firstOrNull { it.endsWith(".epub") || it.endsWith(".txt") } + assumeTrue("no local test media", audioName != null && bookName != null) + + val app = instrumentation.targetContext + fun stage(name: String): Uri { + val out = File(app.cacheDir, name) + instrumentation.context.assets.open("local/$name").use { src -> out.outputStream().use { src.copyTo(it) } } + return Uri.fromFile(out) + } + val audio = stage(audioName!!) + val book = stage(bookName!!) + TranscriptStore.forAudio(app, audio).reset() // measure a cold run, not a cached one + + val started = System.nanoTime() + Job.run(app, audio, book, "auto") + val seconds = (System.nanoTime() - started) / 1e9 + + val status = Job.status.value + Log.i("SubReadTest", "phase=${status.phase} detail=${status.detail} cues=${status.cues} " + + "match=${status.matchRate} dropped=${status.paragraphsDropped} seconds=%.1f".format(seconds)) + assertEquals(status.detail, Phase.DONE, status.phase) + assertNotNull(status.srt) + val srt = status.srt!!.readText() + Log.i("SubReadTest", "language=${TranscriptStore.forAudio(app, audio).language} srt head:\n" + + srt.lineSequence().take(12).joinToString("\n")) + // Side by side: what the model heard, and the book text given to that cue. + val heard = TranscriptStore.forAudio(app, audio).segments() + val cues = srt.trim().split("\n\n").map { it.lines().getOrElse(2) { "" } } + heard.forEach { s -> + Log.i("SubReadSeg", org.json.JSONObject().put("text", s.text).put("start", s.start).put("end", s.end).toString()) + } + heard.take(14).forEachIndexed { i, s -> + Log.i("SubReadTest", "%6.2f-%6.2f heard: %s | book: %s".format(s.start, s.end, s.text, cues.getOrElse(i) { "" })) + } + assertTrue("only ${status.cues} cues", status.cues >= 5) + assertTrue("match rate ${status.matchRate}", (status.matchRate ?: 0.0) > 0.8) + assertTrue(srt.startsWith("1\n00:00:")) + + // A second run reuses the saved transcript: seconds, not minutes. + val again = System.nanoTime() + Job.run(app, audio, book, "auto") + val cached = (System.nanoTime() - again) / 1e9 + Log.i("SubReadTest", "cached rerun seconds=%.1f".format(cached)) + assertEquals(Phase.DONE, Job.status.value.phase) + assertTrue("cached rerun took ${cached}s", cached < seconds / 2 + 5) + } +} diff --git a/core/src/main/kotlin/space/subread/core/BookText.kt b/core/src/main/kotlin/space/subread/core/BookText.kt index dd6ca4a..f8a89df 100644 --- a/core/src/main/kotlin/space/subread/core/BookText.kt +++ b/core/src/main/kotlin/space/subread/core/BookText.kt @@ -73,15 +73,20 @@ object BookText { } private fun page(bytes: ByteArray): List<String> { - val doc = Jsoup.parse(String(bytes, Charsets.UTF_8)) + // An epub page is XHTML, so read it as XML first. To an HTML parser a + // self-closing <title/> never closes, and it takes the whole page as the + // title text. The HTML parser stays as the fallback for malformed pages. + val source = String(bytes, Charsets.UTF_8) + val xml = Jsoup.parse(source, "", Parser.xmlParser()) + val doc = if (xml.selectFirst("body") != null) xml else Jsoup.parse(source) // Furigana would otherwise be read twice: once as kanji, once as kana. doc.select("rt, rp").remove() - val body = doc.body() + val body = doc.selectFirst("body") ?: doc val blocks = body.select(BLOCKS) // A quote holding paragraphs would yield its text twice; keep the leaves. .filter { it.select(BLOCKS).size == 1 } - val source = if (blocks.isEmpty()) listOf(body) else blocks - return source.map { it.text().trim() }.filter { it.isNotEmpty() } + val parts = if (blocks.isEmpty()) listOf(body) else blocks + return parts.map { it.text().trim() }.filter { it.isNotEmpty() } } // ---------------------------------------------------------------- aozora diff --git a/core/src/test/kotlin/space/subread/core/BookTest.kt b/core/src/test/kotlin/space/subread/core/BookTest.kt index b0a242e..ae75837 100644 --- a/core/src/test/kotlin/space/subread/core/BookTest.kt +++ b/core/src/test/kotlin/space/subread/core/BookTest.kt @@ -129,6 +129,23 @@ class BookTest { ) } + /** + * Found on a real device: two thirds of a real epub disappeared, and the + * subtitles got text from the wrong passage. An HTML parser takes a + * self-closing <title/> as an open title, reads the start of the page as + * its text, and the first paragraphs are lost. The page must be large for + * the defect to show: a small page parses correctly. + */ + @Test + fun aSelfClosingTitleDoesNotSwallowTheStartOfALargePage() { + val paragraphs = (1..120).map { n -> "Paragraph $n. " + "word ".repeat(40).trim() } + val bytes = epub("a.xhtml" to + "<?xml version='1.0' encoding='UTF-8'?><html xmlns='http://www.w3.org/1999/xhtml'><head><title/>" + + "<link rel='stylesheet' href='s.css' type='text/css'/></head><body><span id='x'>" + + paragraphs.joinToString("") { "<p class='p1'>$it</p>" } + "</span></body></html>") + assertEquals(paragraphs, BookText.read(ByteArrayInputStream(bytes), "x.epub")) + } + @Test fun epubWithoutAUsablePackageFileStillYieldsItsText() { val bytes = epub("b.html" to page("<p>two</p>"), "a.html" to page("<p>one</p>")) diff --git a/core/src/test/kotlin/space/subread/core/ExcerptTest.kt b/core/src/test/kotlin/space/subread/core/ExcerptTest.kt new file mode 100644 index 0000000..6b39e20 --- /dev/null +++ b/core/src/test/kotlin/space/subread/core/ExcerptTest.kt @@ -0,0 +1,41 @@ +package space.subread.core + +import kotlinx.serialization.json.Json +import kotlinx.serialization.json.double +import kotlinx.serialization.json.jsonArray +import kotlinx.serialization.json.jsonObject +import kotlinx.serialization.json.jsonPrimitive +import org.junit.Assert.assertEquals +import org.junit.Assert.assertTrue +import org.junit.Assume.assumeTrue +import org.junit.Test +import java.io.File + +/** + * A short excerpt from the middle of a novel, against the whole epub. This is + * the case that went wrong on a real device. Local only: the book is in + * copyright. Needs golden-local/excerpt/ with the epub and the device transcript. + */ +class ExcerptTest { + @Test + fun anExcerptFindsItsPlaceInTheWholeBook() { + val dir = javaClass.classLoader.getResource("golden-local/excerpt")?.let { File(it.toURI()) } + assumeTrue("no local excerpt fixture", dir != null && File(dir, "moskva.epub").exists()) + val json = Json.parseToJsonElement(File(dir, "moskva_device.json").readText()).jsonObject + val transcript = json.getValue("transcript").jsonArray.map { + val o = it.jsonObject + TranscriptSegment(o.getValue("text").jsonPrimitive.content, + o.getValue("start").jsonPrimitive.double, o.getValue("end").jsonPrimitive.double) + } + val reference = json.getValue("paragraphs").jsonArray.map { it.jsonPrimitive.content } + + val paragraphs = BookText.read(File(dir, "moskva.epub")) + println("paragraphs: ours ${paragraphs.size}, reference reader ${reference.size}; longest ours ${paragraphs.maxOf { it.length }}") + assertEquals(reference.size, paragraphs.size) + + val result = BookAligner.align(transcript, paragraphs, Language.of("ru")) + result.cues.take(4).forEach { println(it.text) } + assertTrue(result.cues[0].text, result.cues[0].text.startsWith("Чемоданчик я все-таки взял с собой")) + assertEquals("в ожидании заказа.", result.cues[1].text) + } +} diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml index 3c0f364..3019534 100644 --- a/gradle/libs.versions.toml +++ b/gradle/libs.versions.toml @@ -9,6 +9,8 @@ kotlinxCoroutines = "1.10.2" kotlinxSerializationJson = "1.9.0" jsoup = "1.21.2" junit = "4.13.2" +androidxTestRunner = "1.7.0" +androidxTestJunit = "1.3.0" [libraries] androidx-core-ktx = { group = "androidx.core", name = "core-ktx", version.ref = "coreKtx" } @@ -25,6 +27,8 @@ kotlinx-coroutines-android = { group = "org.jetbrains.kotlinx", name = "kotlinx- kotlinx-serialization-json = { group = "org.jetbrains.kotlinx", name = "kotlinx-serialization-json", version.ref = "kotlinxSerializationJson" } jsoup = { group = "org.jsoup", name = "jsoup", version.ref = "jsoup" } junit = { group = "junit", name = "junit", version.ref = "junit" } +androidx-test-runner = { group = "androidx.test", name = "runner", version.ref = "androidxTestRunner" } +androidx-test-junit = { group = "androidx.test.ext", name = "junit", version.ref = "androidxTestJunit" } [plugins] android-application = { id = "com.android.application", version.ref = "agp" }