Recently Written · git

recentlywritten

my writings (done recently)

git clone https://github.com/equwal/recentlywritten

Log | Files | Refs


build.sh (9305 bytes)

1 #!/bin/sh
2 # build.sh — converts posts/*.md and pages/*.md to site/ using pandoc
3 # Usage: sh build.sh [--no-deploy]
4 
5 set -e
6 
7 POSTS_DIR="posts"
8 PAGES_DIR="pages"
9 SITE_DIR="site"
10 STATIC_DIR="static"
11 SITE_URL="https://recentlywritten.com"
12 DEPLOY_HOST="root@recentlywritten.com"
13 DEPLOY_PATH="/var/www/recentlywritten/"
14 FEED_SIZE=10
15 
16 DEPLOY=yes
17 [ "$1" = "--no-deploy" ] && DEPLOY=no
18 
19 WORK="${TMPDIR:-/tmp}/rw-build.$$"
20 mkdir -p "$WORK"
21 trap 'rm -rf "$WORK"' EXIT INT TERM
22 
23 # ── setup ──────────────────────────────────────────────────
24 rm -rf "$SITE_DIR"
25 mkdir -p "$SITE_DIR"
26 cp style.css "$SITE_DIR/style.css"
27 # Images and downloads the recovered posts link to as static/...
28 [ -d "$STATIC_DIR" ] && cp -R "$STATIC_DIR" "$SITE_DIR/$STATIC_DIR"
29 
30 # pandoc HTML template for individual posts.
31 # Kept as a relative path in the working directory on purpose: under cygwin
32 # the pandoc on PATH is the native Windows build, which cannot resolve a
33 # POSIX path like /tmp/..., but does resolve a path relative to the cwd.
34 TEMPLATE=".build-template.html"
35 trap 'rm -rf "$WORK" "$TEMPLATE"' EXIT INT TERM
36 cat > "$TEMPLATE" << 'TMPL'
37 <!DOCTYPE html>
38 <html lang="en">
39 <head>
40   <meta charset="UTF-8" />
41   <meta name="viewport" content="width=device-width, initial-scale=1" />
42   <title>$title$ — Recently Written</title>
43   <meta property="og:type" content="article" />
44   <meta property="og:site_name" content="Recently Written" />
45   <meta property="og:title" content="$title$" />
46   <meta property="og:url" content="$url$" />
47   <meta name="twitter:card" content="summary" />
48   <link rel="stylesheet" href="style.css" />
49   <link rel="alternate" type="application/rss+xml" title="Recently Written" href="rss.xml" />
50 </head>
51 <body>
52 <div id="container">
53 
54   <div id="top">
55     <a class="site-title" href="index.html">Recently Written</a>
56     <nav>
57       <a href="index.html">Home</a>
58       <a href="lair.html">code</a>
59       <a href="git/index.html">git</a>
60       <a href="esperanto.html">esperanto</a>
61       <a href="call.html">contact</a>
62       <a href="rss.xml">rss</a>
63     </nav>
64   </div>
65 
66   <h1 class="post-title">$title$</h1>
67 
68   $body$
69 
70   <div id="footer">
71     <a href="index.html">← Home</a> ·
72     <a href="https://github.com/equwal">Github</a>
73   </div>
74 
75 </div>
76 </body>
77 </html>
78 TMPL
79 
80 # Read one `key: value` line out of a file's front matter, dropping the
81 # surrounding quotes that titles containing a colon have to be written with.
82 meta() {
83     sed -n "s/^$2: *//p" "$1" | head -1 | sed -e 's/^"//' -e 's/"$//' -e 's/\\"/"/g'
84 }
85 
86 # Escape the characters that must not appear raw in generated HTML.
87 esc() {
88     sed -e 's/&/\&amp;/g' -e 's/</\&lt;/g' -e 's/>/\&gt;/g'
89 }
90 
91 # render <markdown> <output> <title>
92 render() {
93     pandoc \
94         --from markdown \
95         --to html5 \
96         --template "$TEMPLATE" \
97         --metadata title="$3" \
98         --metadata url="$SITE_URL/$(basename "$2")" \
99         --output "$2" \
100         "$1"
101 }
102 
103 # ── build each post ─────────────────────────────────────────
104 : > "$WORK/posts.tsv"
105 for md in "$POSTS_DIR"/*.md; do
106     [ -e "$md" ] || continue
107     slug=$(basename "$md" .md)
108     title=$(meta "$md" title)
109     date=$(meta "$md" date)
110     # Recovered posts carry `order`: their position on the old front page,
111     # the only surviving record of their sequence. Everything else sorts by
112     # date. The two never interleave, since an order is three digits ("051")
113     # and a date leads with its year ("2026-..."), so new posts land on top.
114     order=$(meta "$md" order)
115     [ -z "$order" ] && order="$date"
116     render "$md" "$SITE_DIR/${slug}.html" "$title"
117     printf '%s\t%s\t%s\t%s\n' "$order" "$date" "$slug" "$title" >> "$WORK/posts.tsv"
118 done
119 
120 # Dates order the posts but are never shown on the site: most recovered ones
121 # are approximations taken from archive crawls, good enough to sort by and
122 # not good enough to publish. C collation keeps the sort independent of the
123 # system language.
124 LC_ALL=C sort -r "$WORK/posts.tsv" > "$WORK/sorted.tsv"
125 
126 # ── build each standalone page ──────────────────────────────
127 # Pages are dateless and stay out of the chronological list; these are the
128 # hub pages the old site linked from its nav (lair, esperanto, call, ...).
129 : > "$WORK/pages.tsv"
130 for md in "$PAGES_DIR"/*.md; do
131     [ -e "$md" ] || continue
132     slug=$(basename "$md" .md)
133     title=$(meta "$md" title)
134     render "$md" "$SITE_DIR/${slug}.html" "$title"
135     printf '%s\t%s\n' "$slug" "$title" >> "$WORK/pages.tsv"
136 done
137 
138 # ── list fragments for the index ────────────────────────────
139 TAB=$(printf '\t')
140 
141 while IFS="$TAB" read -r order date slug title; do
142     [ -z "$slug" ] && continue
143     printf '<li><a href="%s.html">%s</a></li>\n' \
144         "$slug" "$(printf '%s' "$title" | esc)"
145 done < "$WORK/sorted.tsv" > "$WORK/post-items.html"
146 
147 LC_ALL=C sort -f "$WORK/pages.tsv" | while IFS="$TAB" read -r slug title; do
148     [ -z "$slug" ] && continue
149     printf '<li><a href="%s.html">%s</a></li>\n' \
150         "$slug" "$(printf '%s' "$title" | esc)"
151 done > "$WORK/page-items.html"
152 
153 # ── build index ─────────────────────────────────────────────
154 # The fragments are read from disk rather than substituted into an awk
155 # variable, so titles containing & or \ survive intact.
156 awk -v postfile="$WORK/post-items.html" -v pagefile="$WORK/page-items.html" '
157     /<!-- POST_LIST -->/ {
158         print "<ul class=\"post-list\">"
159         while ((getline line < postfile) > 0) print line
160         close(postfile)
161         print "</ul>"
162         next
163     }
164     /<!-- PAGE_LIST -->/ {
165         print "<ul class=\"page-list\">"
166         while ((getline line < pagefile) > 0) print line
167         close(pagefile)
168         print "</ul>"
169         next
170     }
171     { print }
172 ' index.html > "$SITE_DIR/index.html"
173 
174 # ── feed ─────────────────────────────────────────────────────
175 # Only the newest FEED_SIZE posts go in, and that is deliberate as well as
176 # conventional: a feed reader shows each item's date, and the newest posts are
177 # the ones whose dates are exact rather than recovered from a crawl.
178 
179 # RFC 822, as RSS requires. The C locale matters: without it the day and
180 # month names come out in the system language. `date -d` is GNU; where it is
181 # missing the item simply goes out without a pubDate.
182 feed_date() {
183     LC_ALL=C date -u -d "$1" '+%a, %d %b %Y 00:00:00 +0000' 2>/dev/null
184 }
185 
186 # A feed reader shows the HTML away from the site, so links relative to it
187 # need the site put in front of them.
188 absolute() {
189     sed -E \
190         -e 's@(href|src)="/([^/"][^"]*)"@\1="'"$SITE_URL"'/\2"@g' \
191         -e 's@(href|src)="([^"/#][^":]*)"@\1="'"$SITE_URL"'/\2"@g'
192 }
193 
194 {
195     printf '<?xml version="1.0" encoding="UTF-8"?>\n'
196     printf '<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">\n<channel>\n'
197     printf '  <title>Recently Written</title>\n'
198     printf '  <link>%s/</link>\n' "$SITE_URL"
199     printf '  <description>A site full of things which I have recently written.</description>\n'
200     printf '  <language>en-us</language>\n'
201     printf '  <atom:link href="%s/rss.xml" rel="self" type="application/rss+xml" />\n' "$SITE_URL"
202     head -n "$FEED_SIZE" "$WORK/sorted.tsv" | while IFS="$TAB" read -r order date slug title; do
203         url="$SITE_URL/$slug.html"
204         printf '  <item>\n'
205         printf '    <title>%s</title>\n' "$(printf '%s' "$title" | esc)"
206         printf '    <link>%s</link>\n' "$url"
207         printf '    <guid isPermaLink="true">%s</guid>\n' "$url"
208         pub=$(feed_date "$date") && [ -n "$pub" ] &&
209             printf '    <pubDate>%s</pubDate>\n' "$pub"
210         printf '    <description><![CDATA['
211         pandoc --from markdown --to html5 "$POSTS_DIR/$slug.md" |
212             absolute | sed 's/]]>/]]]]><![CDATA[>/g'
213         printf ']]></description>\n'
214         printf '  </item>\n'
215     done
216     printf '</channel>\n</rss>\n'
217 } > "$SITE_DIR/rss.xml"
218 
219 echo "Built $(ls "$SITE_DIR"/*.html | wc -l | tr -d ' ') pages → $SITE_DIR/"
220 echo "  posts: $(wc -l < "$WORK/posts.tsv" | tr -d ' ')   pages: $(wc -l < "$WORK/pages.tsv" | tr -d ' ')   feed: $(grep -c '<item>' "$SITE_DIR/rss.xml") items"
221 
222 # ── deploy ───────────────────────────────────────────────────
223 # git/ and git.html share this web root but are published by deploy-git.sh,
224 # not built here. Excluding them keeps --delete from treating them as stale:
225 # rsync never deletes an excluded path on the receiving side.
226 # static/book/ holds book files that stay out of the repo (see .gitignore).
227 # They exist only on the server, so the same rule protects them.
228 if [ "$DEPLOY" = yes ]; then
229     rsync -avzP --delete --exclude=/git/ --exclude=/git.html \
230         --exclude=/static/book/ \
231         "$SITE_DIR/" "$DEPLOY_HOST:$DEPLOY_PATH"
232     ssh "$DEPLOY_HOST" "chmod -R a+rX $DEPLOY_PATH"
233 fi