commit 95cfc6087391a2451dd6aa2d7049cbfda3a1677f equwal <truex@equwal.com> 2026-09-21 05:09:21 -0700 Add pure Lua core: SRT parser, cue index, play clock, text tools The core has no KOReader dependency, so it runs under busted or under the fallback runner in spec/run.lua.
.busted | 12 +++++ spec/clock_spec.lua | 126 +++++++++++++++++++++++++++++++++++++++++++ spec/cues_spec.lua | 140 +++++++++++++++++++++++++++++++++++++++++++++++ spec/run.lua | 45 ++++++++++++++++ spec/srt_spec.lua | 131 ++++++++++++++++++++++++++++++++++++++++++++ spec/text_spec.lua | 74 +++++++++++++++++++++++++ subread/clock.lua | 113 ++++++++++++++++++++++++++++++++++++++ subread/cues.lua | 143 ++++++++++++++++++++++++++++++++++++++++++++++++ subread/srt.lua | 120 +++++++++++++++++++++++++++++++++++++++++ subread/text.lua | 153 ++++++++++++++++++++++++++++++++++++++++++++++++++++ 10 files changed, 1057 insertions(+)
diff --git a/.busted b/.busted new file mode 100644 index 0000000..f5bdd27 --- /dev/null +++ b/.busted @@ -0,0 +1,12 @@ +-- busted configuration. +-- Run the tests from this folder with: busted +-- Without a Lua interpreter with busted, use: lua spec/run.lua +return { + _all = { + lpath = "./?.lua;./?/init.lua", + }, + default = { + ROOT = { "spec" }, + pattern = "_spec%.lua$", + }, +} diff --git a/spec/clock_spec.lua b/spec/clock_spec.lua new file mode 100644 index 0000000..1a3894d --- /dev/null +++ b/spec/clock_spec.lua @@ -0,0 +1,126 @@ +local Clock = require("subread.clock") + +describe("SubRead clock", function() + it("starts paused at zero", function() + local c = Clock.new() + assert.is_false(c:isRunning()) + assert.equals(0, c:getPosition(1000)) + end) + + it("does not move while paused", function() + local c = Clock.new{ position = 30 } + assert.equals(30, c:getPosition(1000)) + assert.equals(30, c:getPosition(9999)) + end) + + it("moves with real time when running", function() + local c = Clock.new() + c:start(100) + assert.near(0, c:getPosition(100), 1e-9) + assert.near(5, c:getPosition(105), 1e-9) + end) + + it("keeps the position over a pause", function() + local c = Clock.new() + c:start(100) + c:pause(110) + assert.near(10, c:getPosition(500), 1e-9) + c:start(500) + assert.near(12, c:getPosition(502), 1e-9) + end) + + it("applies the speed", function() + local c = Clock.new{ speed = 2.0 } + c:start(0) + assert.near(20, c:getPosition(10), 1e-9) + end) + + it("keeps the position when the speed changes", function() + local c = Clock.new() + c:start(0) + c:setSpeed(3.0, 10) + assert.near(10, c:getPosition(10), 1e-9) + assert.near(40, c:getPosition(20), 1e-9) + end) + + it("holds the speed inside its limits", function() + local c = Clock.new() + c:setSpeed(0.1, 0) + assert.equals(Clock.SPEED_MIN, c.speed) + c:setSpeed(99, 0) + assert.equals(Clock.SPEED_MAX, c.speed) + end) + + it("takes the offset off the cue time", function() + local c = Clock.new{ position = 100, offset = 20 } + assert.equals(100, c:getPosition(0)) + assert.equals(80, c:getCueTime(0)) + c:setOffset(-5) + assert.equals(105, c:getCueTime(0)) + end) + + it("seeks to an absolute position", function() + local c = Clock.new() + c:start(0) + c:seek(60, 10) + assert.near(60, c:getPosition(10), 1e-9) + assert.near(65, c:getPosition(15), 1e-9) + assert.is_true(c:isRunning()) + end) + + it("never seeks before zero", function() + local c = Clock.new{ position = 5 } + c:seek(-10, 0) + assert.equals(0, c.position) + end) + + it("skips forward and back", function() + local c = Clock.new{ position = 100 } + c:skip(10, 0) + assert.equals(110, c:getPosition(0)) + c:skip(-10, 0) + assert.equals(100, c:getPosition(0)) + end) + + it("seeks by cue time through the offset", function() + local c = Clock.new{ offset = 20 } + c:seekCueTime(50, 0) + assert.equals(70, c:getPosition(0)) + assert.equals(50, c:getCueTime(0)) + end) + + it("toggles between running and paused", function() + local c = Clock.new() + assert.is_true(c:toggle(0)) + assert.is_false(c:toggle(5)) + assert.equals(5, c:getPosition(100)) + end) + + it("tells how long until a cue time", function() + local c = Clock.new{ speed = 2.0, offset = 10 } + c:start(0) + -- Cue time 30 is player position 40, which comes after 20 real seconds. + assert.near(20, c:realSecondsUntilCueTime(30, 0), 1e-9) + -- Cue time -20 is player position -10, which is already past. + assert.is_nil(c:realSecondsUntilCueTime(-20, 0)) + end) + + it("gives no wait time while paused", function() + local c = Clock.new() + assert.is_nil(c:realSecondsUntilCueTime(100, 0)) + end) + + it("keeps position and cue time in step over any sequence", function() + local c = Clock.new{ speed = 1.5, offset = 3 } + local now = 0 + c:start(now) + for step = 1, 40 do + now = now + 0.25 + if step % 7 == 0 then c:toggle(now) end + if step % 11 == 0 then c:skip(-10, now) end + if step % 13 == 0 then c:setSpeed(0.5 + (step % 5) / 2, now) end + assert.near(c:getPosition(now) - c.offset, c:getCueTime(now), 1e-9) + assert.is_true(c:getPosition(now) >= 0) + end + end) +end) diff --git a/spec/cues_spec.lua b/spec/cues_spec.lua new file mode 100644 index 0000000..d288fd3 --- /dev/null +++ b/spec/cues_spec.lua @@ -0,0 +1,140 @@ +local Cues = require("subread.cues") +local Srt = require("subread.srt") + +local function build(triples) + local list = {} + for _, t in ipairs(triples) do + list[#list + 1] = { + start = t[1], stop = t[2], text = t[3], + norm = t[3], no_place = false, + } + end + return Cues.new(list) +end + +local SIMPLE = { + { 0, 2, "alpha one" }, + { 2, 4, "beta two" }, + { 4, 6, "gamma three" }, + { 8, 10, "delta four" }, -- a gap from 6 to 8 +} + +describe("SubRead cues", function() + it("counts the cues", function() + assert.equals(4, build(SIMPLE):count()) + assert.equals(0, Cues.new({}):count()) + end) + + it("finds the last cue that started", function() + local index = build(SIMPLE) + assert.equals(0, index:lastStartedAt(-1)) + assert.equals(1, index:lastStartedAt(0)) + assert.equals(1, index:lastStartedAt(1.9)) + assert.equals(3, index:lastStartedAt(7)) + assert.equals(4, index:lastStartedAt(100)) + end) + + it("finds the cue that is on", function() + local index = build(SIMPLE) + assert.equals(1, index:findByTime(0)) + assert.equals(1, index:findByTime(1.99)) + assert.equals(2, index:findByTime(2)) + assert.equals(4, index:findByTime(9)) + end) + + it("returns nil in a gap and outside the file", function() + local index = build(SIMPLE) + assert.is_nil(index:findByTime(7)) + assert.is_nil(index:findByTime(-1)) + assert.is_nil(index:findByTime(11)) + end) + + it("agrees with a slow reference for every time", function() + local index = build(SIMPLE) + local function reference(t) + local best + for i, cue in ipairs(index.list) do + if cue.start <= t and t < cue.stop then best = i end + end + return best + end + for step = -20, 240 do + local t = step / 20 + assert.equals(reference(t), index:findByTime(t)) + end + end) + + it("lets the latest overlapping cue win", function() + local index = build({ + { 0, 100, "long background cue" }, + { 10, 12, "short cue on top" }, + }) + assert.equals(1, index:findByTime(5)) + assert.equals(2, index:findByTime(11)) + -- After the short cue ends the long one is still on. + assert.equals(1, index:findByTime(20)) + end) + + it("finds the next cue start", function() + local index = build(SIMPLE) + assert.equals(1, index:nextAfter(-5)) + assert.equals(2, index:nextAfter(0)) + assert.equals(4, index:nextAfter(6)) + assert.is_nil(index:nextAfter(8)) + end) + + it("steps over cues that share a start time", function() + local index = build({ { 0, 1, "a" }, { 5, 6, "b" }, { 5, 7, "c" }, { 9, 10, "d" } }) + assert.equals(4, index:nextAfter(5)) + end) + + it("finds the nearest cue for a start time", function() + local index = build(SIMPLE) + assert.equals(1, index:findNearest(-3)) + assert.equals(3, index:findNearest(6.5)) -- gap, nearer the cue before + assert.equals(4, index:findNearest(7.9)) -- gap, nearer the cue after + assert.equals(4, index:findNearest(50)) + assert.is_nil(Cues.new({}):findNearest(1)) + end) + + it("finds the cue that holds a piece of text", function() + local index = build(SIMPLE) + assert.equals(2, index:findIndexContaining("beta")) + assert.is_nil(index:findIndexContaining("missing")) + assert.is_nil(index:findIndexContaining("")) + end) + + it("starts the text search at the given index", function() + local index = build({ { 0, 1, "same words" }, { 1, 2, "same words" } }) + assert.equals(1, index:findIndexContaining("same")) + assert.equals(2, index:findIndexContaining("same", 2)) + end) + + it("treats the needle as plain text, not a pattern", function() + local index = build({ { 0, 1, "cost is 5 % of it" } }) + assert.equals(1, index:findIndexContaining("5 %")) + end) + + it("finds the cue for a page of book text", function() + local index = build(SIMPLE) + -- Book text with other white space than the cue text. + assert.equals(2, index:findIndexForText("beta\n two and more words")) + assert.is_nil(index:findIndexForText("nothing here at all matches")) + assert.is_nil(index:findIndexForText("")) + end) + + it("skips over page text that no cue holds", function() + local index = build({ { 0, 1, "the real sentence of the book" } }) + -- The page starts with a running head that is not in any cue. + local page = "CHAPTER ONE 12 the real sentence of the book" + assert.equals(1, index:findIndexForText(page)) + end) + + it("builds from a parsed file", function() + local data = "1\n00:00:00,000 --> 00:00:01,000\nhello there friend\n" + local index = Cues.new(Srt.parse(data)) + assert.equals(1, index:count()) + assert.equals(1, index:findByTime(0.5)) + assert.equals("hello there friend", index:get(1).norm) + end) +end) diff --git a/spec/run.lua b/spec/run.lua new file mode 100644 index 0000000..c383bd6 --- /dev/null +++ b/spec/run.lua @@ -0,0 +1,45 @@ +-- Fallback test runner for a plain Lua interpreter. +-- Run it from the plugin folder: lua spec/run.lua +-- It gives the small part of the busted API that the spec files use, so the +-- same files also run under busted. + +local here = (arg and arg[0] or "spec/run.lua"):gsub("[^/\\]*$", "") +package.path = table.concat({ here .. "../?.lua", "./?.lua", package.path }, ";") + +local failed, passed, path = 0, 0, {} + +local function deepEqual(a, b) + if a == b then return true end + if type(a) ~= "table" or type(b) ~= "table" then return false end + for k, v in pairs(a) do if not deepEqual(v, b[k]) then return false end end + for k in pairs(b) do if a[k] == nil then return false end end + return true +end + +local function check(ok, message) + if not ok then error(message or "assertion failed", 3) end +end + +local A = { equals = function(a, b) check(a == b, tostring(a) .. " ~= " .. tostring(b)) end, + same = function(a, b) check(deepEqual(a, b), "tables differ") end, + is_nil = function(a) check(a == nil, "expected nil, got " .. tostring(a)) end, + is_not_nil = function(a) check(a ~= nil, "expected a value, got nil") end, + is_true = function(a) check(a == true, "expected true, got " .. tostring(a)) end, + is_false = function(a) check(a == false, "expected false, got " .. tostring(a)) end, + near = function(a, b, tol) check(math.abs(a - b) <= tol, tostring(a) .. " not near " .. tostring(b)) end } +A.are_equal = A.equals +_G.assert = setmetatable(A, { __call = function(_, ok, message) check(ok, message) return ok end }) + +function _G.describe(name, body) path[#path + 1] = name; body(); path[#path] = nil end +function _G.it(name, body) + local title = table.concat(path, " / ") .. " / " .. name + local ok, err = pcall(body) + if ok then passed = passed + 1 else failed = failed + 1; print("FAIL " .. title .. ": " .. tostring(err)) end +end +_G.setup, _G.teardown, _G.before_each, _G.after_each = function(f) f() end, function() end, nil, nil + +for _, name in ipairs({ "text", "srt", "cues", "clock" }) do + dofile(here .. name .. "_spec.lua") +end +print(string.format("%d passed, %d failed", passed, failed)) +os.exit(failed == 0 and 0 or 1) diff --git a/spec/srt_spec.lua b/spec/srt_spec.lua new file mode 100644 index 0000000..68d28b8 --- /dev/null +++ b/spec/srt_spec.lua @@ -0,0 +1,131 @@ +local Srt = require("subread.srt") + +local SAMPLE = table.concat({ + "1", + "00:00:01,000 --> 00:00:03,500", + "It was the best of times,", + "", + "2", + "00:00:03,500 --> 00:00:06,000", + "it was the worst of times,", + "it was the age of wisdom,", + "", + "3", + "00:00:06,000 --> 00:00:08,000", + "\239\188\138 chapter announcement", + "", +}, "\n") + +describe("SubRead srt", function() + it("reads a time line with a comma", function() + local start, stop = Srt.parseTimeLine("00:01:02,500 --> 00:01:04,250") + assert.near(62.5, start, 1e-6) + assert.near(64.25, stop, 1e-6) + end) + + it("reads a time line with a full stop", function() + local start, stop = Srt.parseTimeLine("01:00:00.000 --> 01:00:01.000") + assert.near(3600, start, 1e-6) + assert.near(3601, stop, 1e-6) + end) + + it("scales the fraction by its length", function() + local start = Srt.parseTimeLine("00:00:00,5 --> 00:00:01,0") + assert.near(0.5, start, 1e-6) + end) + + it("rejects a line that is not a time line", function() + assert.is_nil(Srt.parseTimeLine("2")) + assert.is_nil(Srt.parseTimeLine("some cue text")) + end) + + it("splits CRLF, CR and LF the same way", function() + assert.same({ "a", "b", "c" }, Srt.splitLines("a\r\nb\rc")) + end) + + it("parses a whole file", function() + local cues = Srt.parse(SAMPLE) + assert.equals(3, #cues) + assert.near(1.0, cues[1].start, 1e-6) + assert.near(3.5, cues[1].stop, 1e-6) + assert.equals("It was the best of times,", cues[1].text) + end) + + it("joins the lines of a cue", function() + local cues = Srt.parse(SAMPLE) + assert.equals("it was the worst of times,\nit was the age of wisdom,", + cues[2].text) + assert.equals("it was the worst of times, it was the age of wisdom,", + cues[2].norm) + end) + + it("marks the cue that has no place in the book", function() + local cues = Srt.parse(SAMPLE) + assert.is_false(cues[1].no_place) + assert.is_true(cues[3].no_place) + end) + + it("accepts a byte order mark and CRLF", function() + local data = "\239\187\191" .. SAMPLE:gsub("\n", "\r\n") + assert.equals(3, #Srt.parse(data)) + end) + + it("skips a block that has no time line", function() + local data = "header junk\n\n99\nnot a time\n\n" .. SAMPLE + assert.equals(3, #Srt.parse(data)) + end) + + it("accepts a missing cue number", function() + local data = "00:00:01,000 --> 00:00:02,000\nfirst\n\n" .. + "00:00:02,000 --> 00:00:03,000\nsecond\n" + local cues = Srt.parse(data) + assert.equals(2, #cues) + assert.equals("second", cues[2].text) + end) + + it("ends a cue on the next time line when a blank line is missing", function() + local data = "00:00:01,000 --> 00:00:02,000\nfirst\n" .. + "00:00:02,000 --> 00:00:03,000\nsecond\n" + local cues = Srt.parse(data) + assert.equals(2, #cues) + assert.equals("first", cues[1].text) + end) + + it("drops a cue with no text", function() + local data = "00:00:01,000 --> 00:00:02,000\n\n" .. + "00:00:02,000 --> 00:00:03,000\nkept\n" + local cues = Srt.parse(data) + assert.equals(1, #cues) + assert.equals("kept", cues[1].text) + end) + + it("sorts the cues by start time", function() + local data = "00:00:09,000 --> 00:00:10,000\nlate\n\n" .. + "00:00:01,000 --> 00:00:02,000\nearly\n" + local cues = Srt.parse(data) + assert.equals("early", cues[1].text) + assert.equals("late", cues[2].text) + end) + + it("returns an empty list for empty input", function() + assert.same({}, Srt.parse("")) + assert.same({}, Srt.parse(nil)) + end) + + it("keeps the start times in order for any input", function() + local blocks = {} + local seed = 7 + for i = 1, 60 do + seed = (seed * 1103515245 + 12345) % 2147483648 + local start = seed % 1000 + blocks[#blocks + 1] = string.format( + "%d\n00:00:%02d,%03d --> 00:00:%02d,%03d\ncue %d\n", + i, start % 60, start % 1000, (start + 2) % 60, start % 1000, i) + end + local cues = Srt.parse(table.concat(blocks, "\n")) + assert.equals(60, #cues) + for i = 2, #cues do + assert.is_true(cues[i - 1].start <= cues[i].start) + end + end) +end) diff --git a/spec/text_spec.lua b/spec/text_spec.lua new file mode 100644 index 0000000..2f5cf58 --- /dev/null +++ b/spec/text_spec.lua @@ -0,0 +1,74 @@ +local Text = require("subread.text") + +describe("SubRead text", function() + it("removes a byte order mark", function() + assert.equals("abc", Text.stripBOM("\239\187\191abc")) + assert.equals("abc", Text.stripBOM("abc")) + end) + + it("collapses every kind of space", function() + assert.equals("a b", Text.normalize("a \t\n b")) + assert.equals("a b", Text.normalize("a\194\160b")) -- no-break space + assert.equals("a b", Text.normalize("a\227\128\128b")) -- ideographic space + assert.equals("ab", Text.normalize("a\194\173b")) -- soft hyphen removed + end) + + it("trims the ends", function() + assert.equals("hello", Text.normalize(" hello \r\n")) + assert.equals("", Text.normalize(" ")) + assert.equals("", Text.normalize(nil)) + end) + + it("counts UTF-8 characters", function() + assert.equals(3, Text.len("abc")) + assert.equals(3, Text.len("\230\188\162\229\173\151a")) -- 漢字a + assert.equals(0, Text.len("")) + end) + + it("cuts on character boundaries", function() + local s = "\230\188\162\229\173\151a" -- 漢字a + assert.equals("\230\188\162", Text.sub(s, 1, 1)) + assert.equals("\229\173\151a", Text.sub(s, 2, 3)) + assert.equals(s, Text.sub(s, 1, 99)) + assert.equals("", Text.sub(s, 4, 9)) + end) + + it("keeps sub and len consistent for every prefix", function() + local s = "a\230\188\162b\227\129\130c" -- a漢bあc + for i = 1, Text.len(s) do + assert.equals(i, Text.len(Text.sub(s, 1, i))) + end + end) + + it("marks a cue that has no place in the book", function() + assert.is_true(Text.hasNoPlace("\239\188\138 narrator note")) + assert.is_true(Text.hasNoPlace(" ")) + assert.is_false(Text.hasNoPlace("real book text")) + end) + + it("makes anchors, longest first, without repeats", function() + local long = string.rep("x", 40) + local anchors = Text.anchors(long) + assert.equals(2, #anchors) + assert.equals(24, Text.len(anchors[1])) + assert.equals(12, Text.len(anchors[2])) + end) + + it("gives one anchor for a short cue", function() + assert.same({ "yes" }, Text.anchors("yes")) + assert.same({ "eight ch" }, Text.anchors(" eight ch ")) + end) + + it("makes anchors that are prefixes of the normalised cue", function() + local cue = "The quick brown fox jumps over the lazy dog" + local norm = Text.normalize(cue) + for _, anchor in ipairs(Text.anchors(cue)) do + assert.equals(anchor, Text.sub(norm, 1, Text.len(anchor))) + end + end) + + it("ellipsizes long text only", function() + assert.equals("short", Text.ellipsize("short", 10)) + assert.equals("abcde\226\128\166", Text.ellipsize("abcdefgh", 5)) + end) +end) diff --git a/subread/clock.lua b/subread/clock.lua new file mode 100644 index 0000000..61ff2a8 --- /dev/null +++ b/subread/clock.lua @@ -0,0 +1,113 @@ +--[[-- +Play clock for SubRead. + +Pure Lua. No KOReader dependency, so it can be unit tested on a PC. + +The caller gives the time. `now` is a monotonic time in seconds, from any +source. The clock keeps: + * position: where the audio player is, in seconds, + * speed: how fast the player runs, 1.0 is normal, + * offset: how much later the narration is than the subtitle times. + An audio file with a 20 s intro has offset 20. +The cue time is the position minus the offset. +--]]-- + +local Clock = {} +Clock.__index = Clock + +Clock.SPEED_MIN = 0.5 +Clock.SPEED_MAX = 3.0 + +--- Makes a new clock. It is paused. +function Clock.new(opts) + opts = opts or {} + local self = setmetatable({}, Clock) + self.position = opts.position or 0 + self.speed = opts.speed or 1.0 + self.offset = opts.offset or 0 + self.running = false + self.started_at = nil + return self +end + +function Clock:isRunning() + return self.running +end + +--- Returns the player position, in seconds. +function Clock:getPosition(now) + if not self.running then return self.position end + return self.position + (now - self.started_at) * self.speed +end + +--- Returns the time to look up in the cue index, in seconds. +function Clock:getCueTime(now) + return self:getPosition(now) - self.offset +end + +--- Starts the clock at the current position. +function Clock:start(now) + if self.running then return end + self.started_at = now + self.running = true +end + +--- Stops the clock and keeps the position. +function Clock:pause(now) + if not self.running then return end + self.position = self:getPosition(now) + self.running = false + self.started_at = nil +end + +--- Starts or stops the clock. Returns the new state. +function Clock:toggle(now) + if self.running then + self:pause(now) + else + self:start(now) + end + return self.running +end + +--- Moves the player position to an absolute value. +function Clock:seek(position, now) + if position < 0 then position = 0 end + self.position = position + if self.running then self.started_at = now end +end + +--- Moves the player position by a number of seconds. +function Clock:skip(delta, now) + self:seek(self:getPosition(now) + delta, now) +end + +--- Moves the player position to the given cue time. +function Clock:seekCueTime(cue_time, now) + self:seek(cue_time + self.offset, now) +end + +--- Sets the speed. The position is kept. +function Clock:setSpeed(speed, now) + if speed < Clock.SPEED_MIN then speed = Clock.SPEED_MIN end + if speed > Clock.SPEED_MAX then speed = Clock.SPEED_MAX end + self.position = self:getPosition(now) + if self.running then self.started_at = now end + self.speed = speed +end + +--- Sets the offset. The position is kept, so the cue time moves. +function Clock:setOffset(offset) + self.offset = offset +end + +--- Returns the real seconds until the clock reaches the given cue time. +-- Returns nil when the clock is paused or the time is already past. +function Clock:realSecondsUntilCueTime(cue_time, now) + if not self.running then return nil end + local delta = (cue_time + self.offset) - self:getPosition(now) + if delta <= 0 then return nil end + return delta / self.speed +end + +return Clock diff --git a/subread/cues.lua b/subread/cues.lua new file mode 100644 index 0000000..695291f --- /dev/null +++ b/subread/cues.lua @@ -0,0 +1,143 @@ +--[[-- +Cue index for SubRead. + +Pure Lua. No KOReader dependency, so it can be unit tested on a PC. + +The index holds the cues in start order and answers three questions: + * which cue is on at time t, + * which cue starts next after time t, + * which cue holds a piece of text. +--]]-- + +local Text = require("subread.text") + +local Cues = {} +Cues.__index = Cues + +--- Builds an index from a cue array. The array must be sorted by start time. +function Cues.new(list) + local self = setmetatable({}, Cues) + self.list = list or {} + self.max_duration = 0 + for _, cue in ipairs(self.list) do + local duration = cue.stop - cue.start + if duration > self.max_duration then + self.max_duration = duration + end + end + return self +end + +function Cues:count() + return #self.list +end + +function Cues:get(index) + return self.list[index] +end + +--- Returns the index of the last cue that starts at or before t. +-- Returns 0 when every cue starts after t. +function Cues:lastStartedAt(t) + local low, high, answer = 1, #self.list, 0 + while low <= high do + local mid = math.floor((low + high) / 2) + if self.list[mid].start <= t then + answer = mid + low = mid + 1 + else + high = mid - 1 + end + end + return answer +end + +--- Returns the index of the cue that is on at time t, or nil. +-- Cues can overlap. The latest cue that is still on wins. +-- The walk back stops after the longest cue in the file, so the cost is bound. +function Cues:findByTime(t) + local index = self:lastStartedAt(t) + while index >= 1 do + local cue = self.list[index] + if t - cue.start > self.max_duration then break end + if cue.stop > t then return index end + index = index - 1 + end + return nil +end + +--- Returns the index of the first cue that starts after t, or nil. +function Cues:nextAfter(t) + local index = self:lastStartedAt(t) + 1 + -- Cues with the same start time can follow the one that lastStartedAt + -- found, so step over every cue that does not start after t. + while self.list[index] and self.list[index].start <= t do + index = index + 1 + end + if self.list[index] then return index end + return nil +end + +--- Returns the index of the cue that is on at t, or of the nearest cue. +-- Use it to choose a start time. Returns nil for an empty index. +function Cues:findNearest(t) + if #self.list == 0 then return nil end + local on = self:findByTime(t) + if on then return on end + local before = self:lastStartedAt(t) + local after = before + 1 + if before < 1 then return 1 end + if not self.list[after] then return before end + local gap_before = t - self.list[before].stop + local gap_after = self.list[after].start - t + if gap_after < gap_before then return after end + return before +end + +--- Returns the index of the first cue whose text holds the needle, or nil. +-- The needle must already be normalised. The search is a plain substring +-- search, not a pattern match. +-- @int from optional first index to look at (default 1) +function Cues:findIndexContaining(needle, from) + if not needle or needle == "" then return nil end + for index = from or 1, #self.list do + if self.list[index].norm:find(needle, 1, true) then + return index + end + end + return nil +end + +-- Lengths, in characters, of the needles cut out of a page of the book. +-- The long needle comes first, because a long match is more sure. The short +-- needle is the fallback for a cue that is shorter than the long needle. +Cues.NEEDLE_LENGTHS = { 12, 6 } +-- Number of needles tried for each length. A needle can fail because the book +-- text holds ruby text or a running head that the cue text does not hold. +Cues.NEEDLE_TRIES = 8 +-- Distance, in characters, between two needles. +Cues.NEEDLE_STEP = 8 + +--- Returns the index of the first cue that holds a piece of the given text. +-- Use it to answer "which cue does this page start with?". +-- @string haystack book text, not yet normalised +function Cues:findIndexForText(haystack) + local norm = Text.normalize(haystack) + local count = Text.len(norm) + if count == 0 then return nil end + for _, length in ipairs(Cues.NEEDLE_LENGTHS) do + for try = 0, Cues.NEEDLE_TRIES - 1 do + local first = 1 + try * Cues.NEEDLE_STEP + if first > count then break end + local last = first + length - 1 + if last > count then last = count end + if last - first + 1 >= length then + local index = self:findIndexContaining(Text.sub(norm, first, last)) + if index then return index end + end + end + end + return nil +end + +return Cues diff --git a/subread/srt.lua b/subread/srt.lua new file mode 100644 index 0000000..63d8a21 --- /dev/null +++ b/subread/srt.lua @@ -0,0 +1,120 @@ +--[[-- +SubRip (.srt) parser for SubRead. + +Pure Lua. No KOReader dependency, so it can be unit tested on a PC. + +The parser accepts: + * a UTF-8 byte order mark, + * CRLF, LF and CR line ends, + * "," or "." between the seconds and the fraction, + * a missing or wrong cue number, + * cue text on more than one line. +Blocks without a time stamp line are skipped. +The result is sorted by start time. +--]]-- + +local Text = require("subread.text") + +local Srt = {} + +-- One time stamp: hours, minutes, seconds, fraction. +local STAMP = "(%d+):(%d+):(%d+)[,%.](%d+)" + +--- Converts one time stamp to seconds. +-- The fraction keeps its own scale, so "5" means 0.5 s and "500" means 0.5 s. +local function stampToSeconds(h, m, s, frac) + local value = tonumber(h) * 3600 + tonumber(m) * 60 + tonumber(s) + local denominator = 10 ^ #frac + return value + tonumber(frac) / denominator +end + +--- Reads a "start --> stop" line. +-- @return start seconds, stop seconds, or nil when the line is not a time line +function Srt.parseTimeLine(line) + local h1, m1, s1, f1, h2, m2, s2, f2 = + line:match(STAMP .. "%s*%-%-+>%s*" .. STAMP) + if not h1 then return nil end + return stampToSeconds(h1, m1, s1, f1), stampToSeconds(h2, m2, s2, f2) +end + +--- Splits text into lines. CRLF and CR both become a line end. +function Srt.splitLines(s) + s = s:gsub("\r\n", "\n"):gsub("\r", "\n") + local lines = {} + local from = 1 + while true do + local at = s:find("\n", from, true) + if not at then + lines[#lines + 1] = s:sub(from) + break + end + lines[#lines + 1] = s:sub(from, at - 1) + from = at + 1 + end + return lines +end + +--- Parses the content of a .srt file. +-- @string data the whole file +-- @return array of cues { start, stop, text, norm, no_place }, sorted by start +function Srt.parse(data) + local cues = {} + if not data or data == "" then return cues end + local lines = Srt.splitLines(Text.stripBOM(data)) + + local i, count = 1, #lines + while i <= count do + local start, stop = Srt.parseTimeLine(lines[i]) + if start then + i = i + 1 + local parts = {} + while i <= count and lines[i]:match("%S") do + -- A new time line ends the cue as well: it means the file has + -- no blank line between the blocks. + if Srt.parseTimeLine(lines[i]) then break end + parts[#parts + 1] = lines[i] + i = i + 1 + end + local text = table.concat(parts, "\n") + local norm = Text.normalize(text) + if norm ~= "" then + cues[#cues + 1] = { + start = start, + stop = stop, + text = text, + norm = norm, + no_place = Text.hasNoPlace(norm), + } + end + else + i = i + 1 + end + end + + -- A stable sort by start time. table.sort is not stable, so the original + -- order is the second key. + for index, cue in ipairs(cues) do + cue.order = index + end + table.sort(cues, function(a, b) + if a.start ~= b.start then return a.start < b.start end + return a.order < b.order + end) + for _, cue in ipairs(cues) do + cue.order = nil + end + return cues +end + +--- Reads and parses a .srt file. +-- @return array of cues, or nil plus an error message +function Srt.parseFile(path) + local file, err = io.open(path, "rb") + if not file then return nil, err end + local data = file:read("*a") + file:close() + if not data then return nil, "empty file" end + return Srt.parse(data) +end + +return Srt diff --git a/subread/text.lua b/subread/text.lua new file mode 100644 index 0000000..648c34a --- /dev/null +++ b/subread/text.lua @@ -0,0 +1,153 @@ +--[[-- +Text normalisation for SubRead. + +Pure Lua. No KOReader dependency, so it can be unit tested on a PC. + +The cue text of a SubRead subtitle is a slice of the book text, but the +white space is not always the same. This module makes a canonical form of a +string, so cue text and book text can be compared. +--]]-- + +local Text = {} + +-- Byte sequences of the space characters that UTF-8 books use. +-- Lua patterns do not know UTF-8, so the sequences are listed. +local UNICODE_SPACES = { + "\194\160", -- U+00A0 no-break space + "\226\128\128", -- U+2000 en quad + "\226\128\129", -- U+2001 em quad + "\226\128\130", -- U+2002 en space + "\226\128\131", -- U+2003 em space + "\226\128\132", -- U+2004 three-per-em space + "\226\128\133", -- U+2005 four-per-em space + "\226\128\134", -- U+2006 six-per-em space + "\226\128\135", -- U+2007 figure space + "\226\128\136", -- U+2008 punctuation space + "\226\128\137", -- U+2009 thin space + "\226\128\138", -- U+200A hair space + "\226\128\139", -- U+200B zero width space + "\226\128\168", -- U+2028 line separator + "\226\128\169", -- U+2029 paragraph separator + "\227\128\128", -- U+3000 ideographic space + "\239\187\191", -- U+FEFF byte order mark / zero width no-break space +} + +-- Byte sequences that are removed, not replaced by a space. +local REMOVED = { + "\194\173", -- U+00AD soft hyphen + "\226\128\140", -- U+200C zero width non-joiner + "\226\128\141", -- U+200D zero width joiner +} + +-- A cue that starts with this character has no place in the book. +Text.NO_PLACE_MARK = "\239\188\138" -- U+FF0A fullwidth asterisk + +--- Removes a UTF-8 byte order mark from the start of a string. +function Text.stripBOM(s) + if s:sub(1, 3) == "\239\187\191" then + return s:sub(4) + end + return s +end + +--- Makes the canonical form of a string. +-- All space characters become one ASCII space. The ends are trimmed. +function Text.normalize(s) + if not s or s == "" then return "" end + s = Text.stripBOM(s) + for _, seq in ipairs(REMOVED) do + s = s:gsub(seq, "") + end + for _, seq in ipairs(UNICODE_SPACES) do + s = s:gsub(seq, " ") + end + s = s:gsub("%s+", " ") + s = s:gsub("^ ", ""):gsub(" $", "") + return s +end + +--- Returns the byte offset of each UTF-8 character, plus the end offset. +-- offsets[i] is the first byte of character i. offsets[#offsets] is #s + 1. +-- The loop does not use a Lua pattern, because an embedded zero byte in a +-- pattern is not safe in Lua 5.1. +function Text.charOffsets(s) + local offsets = {} + local i, n = 1, #s + while i <= n do + offsets[#offsets + 1] = i + local b = s:byte(i) + local size = 1 + if b >= 0xF0 then size = 4 + elseif b >= 0xE0 then size = 3 + elseif b >= 0xC0 then size = 2 + end + i = i + size + end + offsets[#offsets + 1] = n + 1 + return offsets +end + +--- Counts the UTF-8 characters in a string. +function Text.len(s) + return #Text.charOffsets(s) - 1 +end + +--- Returns characters first..last of a string. Both limits are inclusive. +function Text.sub(s, first, last) + local offsets = Text.charOffsets(s) + local count = #offsets - 1 + if first < 1 then first = 1 end + if last > count then last = count end + if first > last then return "" end + return s:sub(offsets[first], offsets[last + 1] - 1) +end + +--- True if the cue text has no place in the book. +function Text.hasNoPlace(s) + local n = Text.normalize(s) + return n == "" or n:sub(1, #Text.NO_PLACE_MARK) == Text.NO_PLACE_MARK +end + +-- Lengths, in characters, of the search anchors. +-- A long anchor is more unique, but a small difference between the cue text +-- and the book text makes it fail. A short anchor almost always matches, and +-- the caller keeps the document order, so a wrong hit is improbable. +-- Two anchors only: each failed search reads the book to its end, so the +-- number of tries controls the worst case cost. +Text.ANCHOR_LENGTHS = { 24, 12 } +Text.ANCHOR_MIN_LENGTH = 6 + +--- Returns the search anchors for a cue text, longest first. +-- The result never holds the same string two times. +function Text.anchors(s, lengths) + lengths = lengths or Text.ANCHOR_LENGTHS + local norm = Text.normalize(s) + if norm == "" then return {} end + local count = Text.len(norm) + if count < Text.ANCHOR_MIN_LENGTH then + return { norm } + end + local out, seen = {}, {} + for _, want in ipairs(lengths) do + local take = want + if take > count then take = count end + if take >= Text.ANCHOR_MIN_LENGTH then + local anchor = Text.sub(norm, 1, take) + if not seen[anchor] then + seen[anchor] = true + out[#out + 1] = anchor + end + end + end + if #out == 0 then out[1] = norm end + return out +end + +--- Cuts a string to a maximum number of characters, for a dialog title. +function Text.ellipsize(s, max_chars) + local norm = Text.normalize(s) + if Text.len(norm) <= max_chars then return norm end + return Text.sub(norm, 1, max_chars) .. "\226\128\166" -- U+2026 ellipsis +end + +return Text