Recently Written · git

subread.koplugin

KOReader plugin: the book follows the narration of an audiobook, from an .srt made by subread.space

git clone https://github.com/equwal/subread.koplugin

Log | Files | Refs


subread/text.lua (5668 bytes)

1 --[[--
2 Text normalisation for SubRead.
3 
4 Pure Lua. No KOReader dependency, so it can be unit tested on a PC.
5 
6 The cue text of a SubRead subtitle is a slice of the book text, but the
7 white space is not always the same. This module makes a canonical form of a
8 string, so cue text and book text can be compared.
9 --]]--
10 
11 local Text = {}
12 
13 -- Byte sequences of the space characters that UTF-8 books use.
14 -- Lua patterns do not know UTF-8, so the sequences are listed.
15 local UNICODE_SPACES = {
16     "\194\160",      -- U+00A0 no-break space
17     "\226\128\128",  -- U+2000 en quad
18     "\226\128\129",  -- U+2001 em quad
19     "\226\128\130",  -- U+2002 en space
20     "\226\128\131",  -- U+2003 em space
21     "\226\128\132",  -- U+2004 three-per-em space
22     "\226\128\133",  -- U+2005 four-per-em space
23     "\226\128\134",  -- U+2006 six-per-em space
24     "\226\128\135",  -- U+2007 figure space
25     "\226\128\136",  -- U+2008 punctuation space
26     "\226\128\137",  -- U+2009 thin space
27     "\226\128\138",  -- U+200A hair space
28     "\226\128\139",  -- U+200B zero width space
29     "\226\128\168",  -- U+2028 line separator
30     "\226\128\169",  -- U+2029 paragraph separator
31     "\227\128\128",  -- U+3000 ideographic space
32     "\239\187\191",  -- U+FEFF byte order mark / zero width no-break space
33 }
34 
35 -- Byte sequences that are removed, not replaced by a space.
36 local REMOVED = {
37     "\194\173",      -- U+00AD soft hyphen
38     "\226\128\140",  -- U+200C zero width non-joiner
39     "\226\128\141",  -- U+200D zero width joiner
40 }
41 
42 -- A cue that starts with this character has no place in the book.
43 Text.NO_PLACE_MARK = "\239\188\138" -- U+FF0A fullwidth asterisk
44 
45 --- Removes a UTF-8 byte order mark from the start of a string.
46 function Text.stripBOM(s)
47     if s:sub(1, 3) == "\239\187\191" then
48         return s:sub(4)
49     end
50     return s
51 end
52 
53 --- Makes the canonical form of a string.
54 -- All space characters become one ASCII space. The ends are trimmed.
55 function Text.normalize(s)
56     if not s or s == "" then return "" end
57     s = Text.stripBOM(s)
58     for _, seq in ipairs(REMOVED) do
59         s = s:gsub(seq, "")
60     end
61     for _, seq in ipairs(UNICODE_SPACES) do
62         s = s:gsub(seq, " ")
63     end
64     s = s:gsub("%s+", " ")
65     s = s:gsub("^ ", ""):gsub(" $", "")
66     return s
67 end
68 
69 --- Returns the byte offset of each UTF-8 character, plus the end offset.
70 -- offsets[i] is the first byte of character i. offsets[#offsets] is #s + 1.
71 -- The loop does not use a Lua pattern, because an embedded zero byte in a
72 -- pattern is not safe in Lua 5.1.
73 function Text.charOffsets(s)
74     local offsets = {}
75     local i, n = 1, #s
76     while i <= n do
77         offsets[#offsets + 1] = i
78         local b = s:byte(i)
79         local size = 1
80         if b >= 0xF0 then size = 4
81         elseif b >= 0xE0 then size = 3
82         elseif b >= 0xC0 then size = 2
83         end
84         i = i + size
85     end
86     offsets[#offsets + 1] = n + 1
87     return offsets
88 end
89 
90 --- Counts the UTF-8 characters in a string.
91 function Text.len(s)
92     return #Text.charOffsets(s) - 1
93 end
94 
95 --- Returns characters first..last of a string. Both limits are inclusive.
96 function Text.sub(s, first, last)
97     local offsets = Text.charOffsets(s)
98     local count = #offsets - 1
99     if first < 1 then first = 1 end
100     if last > count then last = count end
101     if first > last then return "" end
102     return s:sub(offsets[first], offsets[last + 1] - 1)
103 end
104 
105 --- True if the cue text has no place in the book.
106 function Text.hasNoPlace(s)
107     local n = Text.normalize(s)
108     return n == "" or n:sub(1, #Text.NO_PLACE_MARK) == Text.NO_PLACE_MARK
109 end
110 
111 -- Lengths, in characters, of the search anchors.
112 -- A long anchor is more unique, but a small difference between the cue text
113 -- and the book text makes it fail. A short anchor almost always matches, and
114 -- the caller keeps the document order, so a wrong hit is improbable.
115 -- Two anchors only: each failed search reads the book to its end, so the
116 -- number of tries controls the worst case cost.
117 Text.ANCHOR_LENGTHS = { 24, 12 }
118 Text.ANCHOR_MIN_LENGTH = 6
119 
120 --- Returns the search anchors for a cue text, longest first.
121 -- The result never holds the same string two times.
122 function Text.anchors(s, lengths)
123     lengths = lengths or Text.ANCHOR_LENGTHS
124     local norm = Text.normalize(s)
125     if norm == "" then return {} end
126     local count = Text.len(norm)
127     if count < Text.ANCHOR_MIN_LENGTH then
128         return { norm }
129     end
130     local out, seen = {}, {}
131     for _, want in ipairs(lengths) do
132         local take = want
133         if take > count then take = count end
134         if take >= Text.ANCHOR_MIN_LENGTH then
135             local anchor = Text.sub(norm, 1, take)
136             if not seen[anchor] then
137                 seen[anchor] = true
138                 out[#out + 1] = anchor
139             end
140         end
141     end
142     if #out == 0 then out[1] = norm end
143     return out
144 end
145 
146 -- Length, in characters, of the tail that finds the end of a cue.
147 Text.TAIL_LENGTH = 12
148 
149 --- Returns the last characters of a cue text, to find its end in the book.
150 -- Returns nil when the text is not longer than the tail: the anchor then
151 -- already covers the whole cue.
152 function Text.tail(s, length)
153     length = length or Text.TAIL_LENGTH
154     local norm = Text.normalize(s)
155     local count = Text.len(norm)
156     if count <= length then return nil end
157     return Text.sub(norm, count - length + 1, count)
158 end
159 
160 --- Cuts a string to a maximum number of characters, for a dialog title.
161 function Text.ellipsize(s, max_chars)
162     local norm = Text.normalize(s)
163     if Text.len(norm) <= max_chars then return norm end
164     return Text.sub(norm, 1, max_chars) .. "\226\128\166" -- U+2026 ellipsis
165 end
166 
167 return Text