subread/text.lua (5668 bytes)
1 --[[-- 2 Text normalisation for SubRead. 3 4 Pure Lua. No KOReader dependency, so it can be unit tested on a PC. 5 6 The cue text of a SubRead subtitle is a slice of the book text, but the 7 white space is not always the same. This module makes a canonical form of a 8 string, so cue text and book text can be compared. 9 --]]-- 10 11 local Text = {} 12 13 -- Byte sequences of the space characters that UTF-8 books use. 14 -- Lua patterns do not know UTF-8, so the sequences are listed. 15 local UNICODE_SPACES = { 16 "\194\160", -- U+00A0 no-break space 17 "\226\128\128", -- U+2000 en quad 18 "\226\128\129", -- U+2001 em quad 19 "\226\128\130", -- U+2002 en space 20 "\226\128\131", -- U+2003 em space 21 "\226\128\132", -- U+2004 three-per-em space 22 "\226\128\133", -- U+2005 four-per-em space 23 "\226\128\134", -- U+2006 six-per-em space 24 "\226\128\135", -- U+2007 figure space 25 "\226\128\136", -- U+2008 punctuation space 26 "\226\128\137", -- U+2009 thin space 27 "\226\128\138", -- U+200A hair space 28 "\226\128\139", -- U+200B zero width space 29 "\226\128\168", -- U+2028 line separator 30 "\226\128\169", -- U+2029 paragraph separator 31 "\227\128\128", -- U+3000 ideographic space 32 "\239\187\191", -- U+FEFF byte order mark / zero width no-break space 33 } 34 35 -- Byte sequences that are removed, not replaced by a space. 36 local REMOVED = { 37 "\194\173", -- U+00AD soft hyphen 38 "\226\128\140", -- U+200C zero width non-joiner 39 "\226\128\141", -- U+200D zero width joiner 40 } 41 42 -- A cue that starts with this character has no place in the book. 43 Text.NO_PLACE_MARK = "\239\188\138" -- U+FF0A fullwidth asterisk 44 45 --- Removes a UTF-8 byte order mark from the start of a string. 46 function Text.stripBOM(s) 47 if s:sub(1, 3) == "\239\187\191" then 48 return s:sub(4) 49 end 50 return s 51 end 52 53 --- Makes the canonical form of a string. 54 -- All space characters become one ASCII space. The ends are trimmed. 55 function Text.normalize(s) 56 if not s or s == "" then return "" end 57 s = Text.stripBOM(s) 58 for _, seq in ipairs(REMOVED) do 59 s = s:gsub(seq, "") 60 end 61 for _, seq in ipairs(UNICODE_SPACES) do 62 s = s:gsub(seq, " ") 63 end 64 s = s:gsub("%s+", " ") 65 s = s:gsub("^ ", ""):gsub(" $", "") 66 return s 67 end 68 69 --- Returns the byte offset of each UTF-8 character, plus the end offset. 70 -- offsets[i] is the first byte of character i. offsets[#offsets] is #s + 1. 71 -- The loop does not use a Lua pattern, because an embedded zero byte in a 72 -- pattern is not safe in Lua 5.1. 73 function Text.charOffsets(s) 74 local offsets = {} 75 local i, n = 1, #s 76 while i <= n do 77 offsets[#offsets + 1] = i 78 local b = s:byte(i) 79 local size = 1 80 if b >= 0xF0 then size = 4 81 elseif b >= 0xE0 then size = 3 82 elseif b >= 0xC0 then size = 2 83 end 84 i = i + size 85 end 86 offsets[#offsets + 1] = n + 1 87 return offsets 88 end 89 90 --- Counts the UTF-8 characters in a string. 91 function Text.len(s) 92 return #Text.charOffsets(s) - 1 93 end 94 95 --- Returns characters first..last of a string. Both limits are inclusive. 96 function Text.sub(s, first, last) 97 local offsets = Text.charOffsets(s) 98 local count = #offsets - 1 99 if first < 1 then first = 1 end 100 if last > count then last = count end 101 if first > last then return "" end 102 return s:sub(offsets[first], offsets[last + 1] - 1) 103 end 104 105 --- True if the cue text has no place in the book. 106 function Text.hasNoPlace(s) 107 local n = Text.normalize(s) 108 return n == "" or n:sub(1, #Text.NO_PLACE_MARK) == Text.NO_PLACE_MARK 109 end 110 111 -- Lengths, in characters, of the search anchors. 112 -- A long anchor is more unique, but a small difference between the cue text 113 -- and the book text makes it fail. A short anchor almost always matches, and 114 -- the caller keeps the document order, so a wrong hit is improbable. 115 -- Two anchors only: each failed search reads the book to its end, so the 116 -- number of tries controls the worst case cost. 117 Text.ANCHOR_LENGTHS = { 24, 12 } 118 Text.ANCHOR_MIN_LENGTH = 6 119 120 --- Returns the search anchors for a cue text, longest first. 121 -- The result never holds the same string two times. 122 function Text.anchors(s, lengths) 123 lengths = lengths or Text.ANCHOR_LENGTHS 124 local norm = Text.normalize(s) 125 if norm == "" then return {} end 126 local count = Text.len(norm) 127 if count < Text.ANCHOR_MIN_LENGTH then 128 return { norm } 129 end 130 local out, seen = {}, {} 131 for _, want in ipairs(lengths) do 132 local take = want 133 if take > count then take = count end 134 if take >= Text.ANCHOR_MIN_LENGTH then 135 local anchor = Text.sub(norm, 1, take) 136 if not seen[anchor] then 137 seen[anchor] = true 138 out[#out + 1] = anchor 139 end 140 end 141 end 142 if #out == 0 then out[1] = norm end 143 return out 144 end 145 146 -- Length, in characters, of the tail that finds the end of a cue. 147 Text.TAIL_LENGTH = 12 148 149 --- Returns the last characters of a cue text, to find its end in the book. 150 -- Returns nil when the text is not longer than the tail: the anchor then 151 -- already covers the whole cue. 152 function Text.tail(s, length) 153 length = length or Text.TAIL_LENGTH 154 local norm = Text.normalize(s) 155 local count = Text.len(norm) 156 if count <= length then return nil end 157 return Text.sub(norm, count - length + 1, count) 158 end 159 160 --- Cuts a string to a maximum number of characters, for a dialog title. 161 function Text.ellipsize(s, max_chars) 162 local norm = Text.normalize(s) 163 if Text.len(norm) <= max_chars then return norm end 164 return Text.sub(norm, 1, max_chars) .. "\226\128\166" -- U+2026 ellipsis 165 end 166 167 return Text