app/src/main/kotlin/space/subread/app/job/TranscriptStore.kt (4487 bytes)
1 package space.subread.app.job 2 3 import android.content.Context 4 import android.net.Uri 5 import android.provider.OpenableColumns 6 import org.json.JSONObject 7 import space.subread.core.TranscriptSegment 8 import java.io.File 9 import java.security.MessageDigest 10 11 /** 12 * What has been transcribed so far, on disk, per audio file. 13 * 14 * Transcribing a book takes hours and Android is entitled to kill the process 15 * at any point in them. Every finished chunk is appended here, so a restart 16 * picks up at the last chunk rather than at the beginning - and aligning the 17 * same audio against a different edition of the book costs seconds, not hours, 18 * because the transcript depends on the audio alone. 19 */ 20 class TranscriptStore private constructor(private val dir: File) { 21 22 private val segmentsFile = File(dir, "segments.jsonl") 23 private val stateFile = File(dir, "state.json") 24 25 /** Audio before this point (seconds) is transcribed and saved. */ 26 var doneUntil: Double = 0.0 27 private set 28 var language: String? = null 29 private set 30 var complete: Boolean = false 31 private set 32 33 init { 34 dir.mkdirs() 35 if (stateFile.exists()) runCatching { 36 val o = JSONObject(stateFile.readText()) 37 doneUntil = o.optDouble("doneUntil", 0.0) 38 language = o.optString("language").ifEmpty { null } 39 complete = o.optBoolean("complete", false) 40 } 41 // A chunk's segments are written before its state. Killed in between, 42 // the file holds segments past doneUntil that are about to be 43 // transcribed a second time; remove them now or they end up doubled. 44 if (segmentsFile.exists()) { 45 val kept = segmentsFile.readLines().filter { line -> 46 runCatching { JSONObject(line).getDouble("s") < doneUntil }.getOrDefault(false) 47 } 48 segmentsFile.writeText(kept.joinToString("") { it + "\n" }) 49 } 50 } 51 52 fun segments(): List<TranscriptSegment> { 53 if (!segmentsFile.exists()) return emptyList() 54 return segmentsFile.readLines().mapNotNull { line -> 55 runCatching { 56 val o = JSONObject(line) 57 TranscriptSegment(o.getString("t"), o.getDouble("s"), o.getDouble("e")) 58 }.getOrNull() 59 } 60 } 61 62 fun append(chunk: List<TranscriptSegment>, until: Double, language: String?) { 63 segmentsFile.appendText(chunk.joinToString("") { s -> 64 JSONObject().put("s", s.start).put("e", s.end).put("t", s.text).toString() + "\n" 65 }) 66 doneUntil = until 67 if (language != null) this.language = language 68 save() 69 } 70 71 fun finish() { 72 complete = true 73 save() 74 } 75 76 fun reset() { 77 segmentsFile.delete() 78 stateFile.delete() 79 doneUntil = 0.0 80 language = null 81 complete = false 82 } 83 84 private fun save() { 85 val tmp = File(dir, "state.json.tmp") 86 tmp.writeText( 87 JSONObject().put("doneUntil", doneUntil).put("language", language ?: "") 88 .put("complete", complete).toString() 89 ) 90 tmp.renameTo(stateFile) 91 } 92 93 companion object { 94 fun forAudio(context: Context, audio: Uri): TranscriptStore { 95 val (name, size) = describe(context, audio) 96 // Name and size, not the URI: the same file picked again through a 97 // different route gets a different URI and should still resume. 98 val key = MessageDigest.getInstance("SHA-1").digest("$name|$size".toByteArray()) 99 .joinToString("") { "%02x".format(it) }.take(20) 100 return TranscriptStore(File(context.filesDir, "transcripts/$key")) 101 } 102 103 /** Display name and byte size of a picked document. */ 104 fun describe(context: Context, uri: Uri): Pair<String, Long> { 105 var name = uri.lastPathSegment ?: "file" 106 var size = -1L 107 runCatching { 108 context.contentResolver.query(uri, null, null, null, null)?.use { c -> 109 if (c.moveToFirst()) { 110 val n = c.getColumnIndex(OpenableColumns.DISPLAY_NAME) 111 val s = c.getColumnIndex(OpenableColumns.SIZE) 112 if (n >= 0 && !c.isNull(n)) name = c.getString(n) 113 if (s >= 0 && !c.isNull(s)) size = c.getLong(s) 114 } 115 } 116 } 117 return name to size 118 } 119 } 120 }