Recently Written · git

subread.koplugin

KOReader plugin: the book follows the narration of an audiobook, from an .srt made by subread.space

git clone https://github.com/equwal/subread.koplugin

Log | Files | Refs


spec/text_spec.lua (2854 bytes)

1 local Text = require("subread.text")
2 
3 describe("SubRead text", function()
4     it("removes a byte order mark", function()
5         assert.equals("abc", Text.stripBOM("\239\187\191abc"))
6         assert.equals("abc", Text.stripBOM("abc"))
7     end)
8 
9     it("collapses every kind of space", function()
10         assert.equals("a b", Text.normalize("a \t\n b"))
11         assert.equals("a b", Text.normalize("a\194\160b"))         -- no-break space
12         assert.equals("a b", Text.normalize("a\227\128\128b"))     -- ideographic space
13         assert.equals("ab", Text.normalize("a\194\173b"))          -- soft hyphen removed
14     end)
15 
16     it("trims the ends", function()
17         assert.equals("hello", Text.normalize("   hello \r\n"))
18         assert.equals("", Text.normalize("   "))
19         assert.equals("", Text.normalize(nil))
20     end)
21 
22     it("counts UTF-8 characters", function()
23         assert.equals(3, Text.len("abc"))
24         assert.equals(3, Text.len("\230\188\162\229\173\151a")) -- 漢字a
25         assert.equals(0, Text.len(""))
26     end)
27 
28     it("cuts on character boundaries", function()
29         local s = "\230\188\162\229\173\151a" -- 漢字a
30         assert.equals("\230\188\162", Text.sub(s, 1, 1))
31         assert.equals("\229\173\151a", Text.sub(s, 2, 3))
32         assert.equals(s, Text.sub(s, 1, 99))
33         assert.equals("", Text.sub(s, 4, 9))
34     end)
35 
36     it("keeps sub and len consistent for every prefix", function()
37         local s = "a\230\188\162b\227\129\130c" -- a漢bあc
38         for i = 1, Text.len(s) do
39             assert.equals(i, Text.len(Text.sub(s, 1, i)))
40         end
41     end)
42 
43     it("marks a cue that has no place in the book", function()
44         assert.is_true(Text.hasNoPlace("\239\188\138 narrator note"))
45         assert.is_true(Text.hasNoPlace("  "))
46         assert.is_false(Text.hasNoPlace("real book text"))
47     end)
48 
49     it("makes anchors, longest first, without repeats", function()
50         local long = string.rep("x", 40)
51         local anchors = Text.anchors(long)
52         assert.equals(2, #anchors)
53         assert.equals(24, Text.len(anchors[1]))
54         assert.equals(12, Text.len(anchors[2]))
55     end)
56 
57     it("gives one anchor for a short cue", function()
58         assert.same({ "yes" }, Text.anchors("yes"))
59         assert.same({ "eight ch" }, Text.anchors("  eight   ch "))
60     end)
61 
62     it("makes anchors that are prefixes of the normalised cue", function()
63         local cue = "The quick brown fox jumps over the lazy dog"
64         local norm = Text.normalize(cue)
65         for _, anchor in ipairs(Text.anchors(cue)) do
66             assert.equals(anchor, Text.sub(norm, 1, Text.len(anchor)))
67         end
68     end)
69 
70     it("ellipsizes long text only", function()
71         assert.equals("short", Text.ellipsize("short", 10))
72         assert.equals("abcde\226\128\166", Text.ellipsize("abcdefgh", 5))
73     end)
74 end)