Recently Written · git

subread-android

SubRead for Android: times an audiobook against its ebook on the device

git clone https://github.com/equwal/subread-android

Log | Files | Refs


commit 62dca60b0bd3bba8bf74e6543ba635f15be9e72e
equwal <truex@equwal.com>
2026-09-20 22:48:03 -0700

Read epub pages as XHTML; add an on-device job test

A test on a real device found subtitles with text from the wrong passage.
The speech model heard the correct words. The epub reader lost two thirds
of the book: the HTML parser takes a self-closing <title/> as an open
title and reads the start of a large page as its text. BookText now reads
each page as XML first. The HTML parser stays as the fallback.

OnDeviceJobTest runs the full job on hardware without the screen: decode,
transcribe, align, write the .srt, then a cached second run. The media is
local only. ExcerptTest reproduces the device failure on the JVM.

 .gitignore                                         |  3 +
 app/build.gradle.kts                               |  3 +
 .../kotlin/space/subread/app/OnDeviceJobTest.kt    | 79 ++++++++++++++++++++++
 .../src/main/kotlin/space/subread/core/BookText.kt | 13 ++--
 .../src/test/kotlin/space/subread/core/BookTest.kt | 17 +++++
 .../test/kotlin/space/subread/core/ExcerptTest.kt  | 41 +++++++++++
 gradle/libs.versions.toml                          |  4 ++
 7 files changed, 156 insertions(+), 4 deletions(-)
diff --git a/.gitignore b/.gitignore
index ee37da8..607f8d1 100644
--- a/.gitignore
+++ b/.gitignore
@@ -12,5 +12,8 @@ captures/
 # that generated them. Tests skip this directory when it is missing.
 core/src/test/resources/golden-local/
 
+# On-device test media. Bring your own pair; the test skips without it.
+app/src/androidTest/assets/local/
+
 # Downloaded at build time, never committed.
 app/src/main/assets/models/
diff --git a/app/build.gradle.kts b/app/build.gradle.kts
index fb0509e..c3fb81b 100644
--- a/app/build.gradle.kts
+++ b/app/build.gradle.kts
@@ -22,6 +22,7 @@ android {
         targetSdk = 36
         versionCode = (providers.gradleProperty("versionCode").orNull ?: "1").toInt()
         versionName = providers.gradleProperty("versionName").orNull ?: "0.1.0"
+        testInstrumentationRunner = "androidx.test.runner.AndroidJUnitRunner"
 
         // Every phone worth transcribing on is 64-bit ARM; the speech library
         // is built for ARMv8.2 specifically (see src/main/cpp/CMakeLists.txt).
@@ -82,4 +83,6 @@ dependencies {
     implementation(libs.androidx.lifecycle.runtime.ktx)
     implementation(libs.androidx.lifecycle.runtime.compose)
     implementation(libs.kotlinx.coroutines.android)
+    androidTestImplementation(libs.androidx.test.runner)
+    androidTestImplementation(libs.androidx.test.junit)
 }
diff --git a/app/src/androidTest/kotlin/space/subread/app/OnDeviceJobTest.kt b/app/src/androidTest/kotlin/space/subread/app/OnDeviceJobTest.kt
new file mode 100644
index 0000000..23f11bf
--- /dev/null
+++ b/app/src/androidTest/kotlin/space/subread/app/OnDeviceJobTest.kt
@@ -0,0 +1,79 @@
+package space.subread.app
+
+import android.net.Uri
+import android.util.Log
+import androidx.test.ext.junit.runners.AndroidJUnit4
+import androidx.test.platform.app.InstrumentationRegistry
+import org.junit.Assert.assertEquals
+import org.junit.Assert.assertNotNull
+import org.junit.Assert.assertTrue
+import org.junit.Assume.assumeTrue
+import org.junit.Test
+import org.junit.runner.RunWith
+import space.subread.app.job.Job
+import space.subread.app.job.Phase
+import space.subread.app.job.TranscriptStore
+import java.io.File
+
+/**
+ * The whole job on real hardware: decode, transcribe with the native speech
+ * model, align, write the subtitles. No screen is needed.
+ *
+ * The media is not in the repository. Put an audio file and its book in
+ * app/src/androidTest/assets/local/ (gitignored); the test skips without them.
+ */
+@RunWith(AndroidJUnit4::class)
+class OnDeviceJobTest {
+
+    @Test
+    fun anAudiobookAndItsBookBecomeSubtitles() {
+        val instrumentation = InstrumentationRegistry.getInstrumentation()
+        val names = instrumentation.context.assets.list("local").orEmpty()
+        val audioName = names.firstOrNull { it.endsWith(".mp3") || it.endsWith(".m4b") || it.endsWith(".m4a") }
+        val bookName = names.firstOrNull { it.endsWith(".epub") || it.endsWith(".txt") }
+        assumeTrue("no local test media", audioName != null && bookName != null)
+
+        val app = instrumentation.targetContext
+        fun stage(name: String): Uri {
+            val out = File(app.cacheDir, name)
+            instrumentation.context.assets.open("local/$name").use { src -> out.outputStream().use { src.copyTo(it) } }
+            return Uri.fromFile(out)
+        }
+        val audio = stage(audioName!!)
+        val book = stage(bookName!!)
+        TranscriptStore.forAudio(app, audio).reset()   // measure a cold run, not a cached one
+
+        val started = System.nanoTime()
+        Job.run(app, audio, book, "auto")
+        val seconds = (System.nanoTime() - started) / 1e9
+
+        val status = Job.status.value
+        Log.i("SubReadTest", "phase=${status.phase} detail=${status.detail} cues=${status.cues} " +
+            "match=${status.matchRate} dropped=${status.paragraphsDropped} seconds=%.1f".format(seconds))
+        assertEquals(status.detail, Phase.DONE, status.phase)
+        assertNotNull(status.srt)
+        val srt = status.srt!!.readText()
+        Log.i("SubReadTest", "language=${TranscriptStore.forAudio(app, audio).language} srt head:\n" +
+            srt.lineSequence().take(12).joinToString("\n"))
+        // Side by side: what the model heard, and the book text given to that cue.
+        val heard = TranscriptStore.forAudio(app, audio).segments()
+        val cues = srt.trim().split("\n\n").map { it.lines().getOrElse(2) { "" } }
+        heard.forEach { s ->
+            Log.i("SubReadSeg", org.json.JSONObject().put("text", s.text).put("start", s.start).put("end", s.end).toString())
+        }
+        heard.take(14).forEachIndexed { i, s ->
+            Log.i("SubReadTest", "%6.2f-%6.2f heard: %s | book: %s".format(s.start, s.end, s.text, cues.getOrElse(i) { "" }))
+        }
+        assertTrue("only ${status.cues} cues", status.cues >= 5)
+        assertTrue("match rate ${status.matchRate}", (status.matchRate ?: 0.0) > 0.8)
+        assertTrue(srt.startsWith("1\n00:00:"))
+
+        // A second run reuses the saved transcript: seconds, not minutes.
+        val again = System.nanoTime()
+        Job.run(app, audio, book, "auto")
+        val cached = (System.nanoTime() - again) / 1e9
+        Log.i("SubReadTest", "cached rerun seconds=%.1f".format(cached))
+        assertEquals(Phase.DONE, Job.status.value.phase)
+        assertTrue("cached rerun took ${cached}s", cached < seconds / 2 + 5)
+    }
+}
diff --git a/core/src/main/kotlin/space/subread/core/BookText.kt b/core/src/main/kotlin/space/subread/core/BookText.kt
index dd6ca4a..f8a89df 100644
--- a/core/src/main/kotlin/space/subread/core/BookText.kt
+++ b/core/src/main/kotlin/space/subread/core/BookText.kt
@@ -73,15 +73,20 @@ object BookText {
     }
 
     private fun page(bytes: ByteArray): List<String> {
-        val doc = Jsoup.parse(String(bytes, Charsets.UTF_8))
+        // An epub page is XHTML, so read it as XML first. To an HTML parser a
+        // self-closing <title/> never closes, and it takes the whole page as the
+        // title text. The HTML parser stays as the fallback for malformed pages.
+        val source = String(bytes, Charsets.UTF_8)
+        val xml = Jsoup.parse(source, "", Parser.xmlParser())
+        val doc = if (xml.selectFirst("body") != null) xml else Jsoup.parse(source)
         // Furigana would otherwise be read twice: once as kanji, once as kana.
         doc.select("rt, rp").remove()
-        val body = doc.body()
+        val body = doc.selectFirst("body") ?: doc
         val blocks = body.select(BLOCKS)
             // A quote holding paragraphs would yield its text twice; keep the leaves.
             .filter { it.select(BLOCKS).size == 1 }
-        val source = if (blocks.isEmpty()) listOf(body) else blocks
-        return source.map { it.text().trim() }.filter { it.isNotEmpty() }
+        val parts = if (blocks.isEmpty()) listOf(body) else blocks
+        return parts.map { it.text().trim() }.filter { it.isNotEmpty() }
     }
 
     // ---------------------------------------------------------------- aozora
diff --git a/core/src/test/kotlin/space/subread/core/BookTest.kt b/core/src/test/kotlin/space/subread/core/BookTest.kt
index b0a242e..ae75837 100644
--- a/core/src/test/kotlin/space/subread/core/BookTest.kt
+++ b/core/src/test/kotlin/space/subread/core/BookTest.kt
@@ -129,6 +129,23 @@ class BookTest {
         )
     }
 
+    /**
+     * Found on a real device: two thirds of a real epub disappeared, and the
+     * subtitles got text from the wrong passage. An HTML parser takes a
+     * self-closing <title/> as an open title, reads the start of the page as
+     * its text, and the first paragraphs are lost. The page must be large for
+     * the defect to show: a small page parses correctly.
+     */
+    @Test
+    fun aSelfClosingTitleDoesNotSwallowTheStartOfALargePage() {
+        val paragraphs = (1..120).map { n -> "Paragraph $n. " + "word ".repeat(40).trim() }
+        val bytes = epub("a.xhtml" to
+            "<?xml version='1.0' encoding='UTF-8'?><html xmlns='http://www.w3.org/1999/xhtml'><head><title/>" +
+            "<link rel='stylesheet' href='s.css' type='text/css'/></head><body><span id='x'>" +
+            paragraphs.joinToString("") { "<p class='p1'>$it</p>" } + "</span></body></html>")
+        assertEquals(paragraphs, BookText.read(ByteArrayInputStream(bytes), "x.epub"))
+    }
+
     @Test
     fun epubWithoutAUsablePackageFileStillYieldsItsText() {
         val bytes = epub("b.html" to page("<p>two</p>"), "a.html" to page("<p>one</p>"))
diff --git a/core/src/test/kotlin/space/subread/core/ExcerptTest.kt b/core/src/test/kotlin/space/subread/core/ExcerptTest.kt
new file mode 100644
index 0000000..6b39e20
--- /dev/null
+++ b/core/src/test/kotlin/space/subread/core/ExcerptTest.kt
@@ -0,0 +1,41 @@
+package space.subread.core
+
+import kotlinx.serialization.json.Json
+import kotlinx.serialization.json.double
+import kotlinx.serialization.json.jsonArray
+import kotlinx.serialization.json.jsonObject
+import kotlinx.serialization.json.jsonPrimitive
+import org.junit.Assert.assertEquals
+import org.junit.Assert.assertTrue
+import org.junit.Assume.assumeTrue
+import org.junit.Test
+import java.io.File
+
+/**
+ * A short excerpt from the middle of a novel, against the whole epub. This is
+ * the case that went wrong on a real device. Local only: the book is in
+ * copyright. Needs golden-local/excerpt/ with the epub and the device transcript.
+ */
+class ExcerptTest {
+    @Test
+    fun anExcerptFindsItsPlaceInTheWholeBook() {
+        val dir = javaClass.classLoader.getResource("golden-local/excerpt")?.let { File(it.toURI()) }
+        assumeTrue("no local excerpt fixture", dir != null && File(dir, "moskva.epub").exists())
+        val json = Json.parseToJsonElement(File(dir, "moskva_device.json").readText()).jsonObject
+        val transcript = json.getValue("transcript").jsonArray.map {
+            val o = it.jsonObject
+            TranscriptSegment(o.getValue("text").jsonPrimitive.content,
+                o.getValue("start").jsonPrimitive.double, o.getValue("end").jsonPrimitive.double)
+        }
+        val reference = json.getValue("paragraphs").jsonArray.map { it.jsonPrimitive.content }
+
+        val paragraphs = BookText.read(File(dir, "moskva.epub"))
+        println("paragraphs: ours ${paragraphs.size}, reference reader ${reference.size}; longest ours ${paragraphs.maxOf { it.length }}")
+        assertEquals(reference.size, paragraphs.size)
+
+        val result = BookAligner.align(transcript, paragraphs, Language.of("ru"))
+        result.cues.take(4).forEach { println(it.text) }
+        assertTrue(result.cues[0].text, result.cues[0].text.startsWith("Чемоданчик я все-таки взял с собой"))
+        assertEquals("в ожидании заказа.", result.cues[1].text)
+    }
+}
diff --git a/gradle/libs.versions.toml b/gradle/libs.versions.toml
index 3c0f364..3019534 100644
--- a/gradle/libs.versions.toml
+++ b/gradle/libs.versions.toml
@@ -9,6 +9,8 @@ kotlinxCoroutines = "1.10.2"
 kotlinxSerializationJson = "1.9.0"
 jsoup = "1.21.2"
 junit = "4.13.2"
+androidxTestRunner = "1.7.0"
+androidxTestJunit = "1.3.0"
 
 [libraries]
 androidx-core-ktx = { group = "androidx.core", name = "core-ktx", version.ref = "coreKtx" }
@@ -25,6 +27,8 @@ kotlinx-coroutines-android = { group = "org.jetbrains.kotlinx", name = "kotlinx-
 kotlinx-serialization-json = { group = "org.jetbrains.kotlinx", name = "kotlinx-serialization-json", version.ref = "kotlinxSerializationJson" }
 jsoup = { group = "org.jsoup", name = "jsoup", version.ref = "jsoup" }
 junit = { group = "junit", name = "junit", version.ref = "junit" }
+androidx-test-runner = { group = "androidx.test", name = "runner", version.ref = "androidxTestRunner" }
+androidx-test-junit = { group = "androidx.test.ext", name = "junit", version.ref = "androidxTestJunit" }
 
 [plugins]
 android-application = { id = "com.android.application", version.ref = "agp" }