build.sh (9305 bytes)
1 #!/bin/sh 2 # build.sh — converts posts/*.md and pages/*.md to site/ using pandoc 3 # Usage: sh build.sh [--no-deploy] 4 5 set -e 6 7 POSTS_DIR="posts" 8 PAGES_DIR="pages" 9 SITE_DIR="site" 10 STATIC_DIR="static" 11 SITE_URL="https://recentlywritten.com" 12 DEPLOY_HOST="root@recentlywritten.com" 13 DEPLOY_PATH="/var/www/recentlywritten/" 14 FEED_SIZE=10 15 16 DEPLOY=yes 17 [ "$1" = "--no-deploy" ] && DEPLOY=no 18 19 WORK="${TMPDIR:-/tmp}/rw-build.$$" 20 mkdir -p "$WORK" 21 trap 'rm -rf "$WORK"' EXIT INT TERM 22 23 # ── setup ────────────────────────────────────────────────── 24 rm -rf "$SITE_DIR" 25 mkdir -p "$SITE_DIR" 26 cp style.css "$SITE_DIR/style.css" 27 # Images and downloads the recovered posts link to as static/... 28 [ -d "$STATIC_DIR" ] && cp -R "$STATIC_DIR" "$SITE_DIR/$STATIC_DIR" 29 30 # pandoc HTML template for individual posts. 31 # Kept as a relative path in the working directory on purpose: under cygwin 32 # the pandoc on PATH is the native Windows build, which cannot resolve a 33 # POSIX path like /tmp/..., but does resolve a path relative to the cwd. 34 TEMPLATE=".build-template.html" 35 trap 'rm -rf "$WORK" "$TEMPLATE"' EXIT INT TERM 36 cat > "$TEMPLATE" << 'TMPL' 37 <!DOCTYPE html> 38 <html lang="en"> 39 <head> 40 <meta charset="UTF-8" /> 41 <meta name="viewport" content="width=device-width, initial-scale=1" /> 42 <title>$title$ — Recently Written</title> 43 <meta property="og:type" content="article" /> 44 <meta property="og:site_name" content="Recently Written" /> 45 <meta property="og:title" content="$title$" /> 46 <meta property="og:url" content="$url$" /> 47 <meta name="twitter:card" content="summary" /> 48 <link rel="stylesheet" href="style.css" /> 49 <link rel="alternate" type="application/rss+xml" title="Recently Written" href="rss.xml" /> 50 </head> 51 <body> 52 <div id="container"> 53 54 <div id="top"> 55 <a class="site-title" href="index.html">Recently Written</a> 56 <nav> 57 <a href="index.html">Home</a> 58 <a href="lair.html">code</a> 59 <a href="git/index.html">git</a> 60 <a href="esperanto.html">esperanto</a> 61 <a href="call.html">contact</a> 62 <a href="rss.xml">rss</a> 63 </nav> 64 </div> 65 66 <h1 class="post-title">$title$</h1> 67 68 $body$ 69 70 <div id="footer"> 71 <a href="index.html">← Home</a> · 72 <a href="https://github.com/equwal">Github</a> 73 </div> 74 75 </div> 76 </body> 77 </html> 78 TMPL 79 80 # Read one `key: value` line out of a file's front matter, dropping the 81 # surrounding quotes that titles containing a colon have to be written with. 82 meta() { 83 sed -n "s/^$2: *//p" "$1" | head -1 | sed -e 's/^"//' -e 's/"$//' -e 's/\\"/"/g' 84 } 85 86 # Escape the characters that must not appear raw in generated HTML. 87 esc() { 88 sed -e 's/&/\&/g' -e 's/</\</g' -e 's/>/\>/g' 89 } 90 91 # render <markdown> <output> <title> 92 render() { 93 pandoc \ 94 --from markdown \ 95 --to html5 \ 96 --template "$TEMPLATE" \ 97 --metadata title="$3" \ 98 --metadata url="$SITE_URL/$(basename "$2")" \ 99 --output "$2" \ 100 "$1" 101 } 102 103 # ── build each post ───────────────────────────────────────── 104 : > "$WORK/posts.tsv" 105 for md in "$POSTS_DIR"/*.md; do 106 [ -e "$md" ] || continue 107 slug=$(basename "$md" .md) 108 title=$(meta "$md" title) 109 date=$(meta "$md" date) 110 # Recovered posts carry `order`: their position on the old front page, 111 # the only surviving record of their sequence. Everything else sorts by 112 # date. The two never interleave, since an order is three digits ("051") 113 # and a date leads with its year ("2026-..."), so new posts land on top. 114 order=$(meta "$md" order) 115 [ -z "$order" ] && order="$date" 116 render "$md" "$SITE_DIR/${slug}.html" "$title" 117 printf '%s\t%s\t%s\t%s\n' "$order" "$date" "$slug" "$title" >> "$WORK/posts.tsv" 118 done 119 120 # Dates order the posts but are never shown on the site: most recovered ones 121 # are approximations taken from archive crawls, good enough to sort by and 122 # not good enough to publish. C collation keeps the sort independent of the 123 # system language. 124 LC_ALL=C sort -r "$WORK/posts.tsv" > "$WORK/sorted.tsv" 125 126 # ── build each standalone page ────────────────────────────── 127 # Pages are dateless and stay out of the chronological list; these are the 128 # hub pages the old site linked from its nav (lair, esperanto, call, ...). 129 : > "$WORK/pages.tsv" 130 for md in "$PAGES_DIR"/*.md; do 131 [ -e "$md" ] || continue 132 slug=$(basename "$md" .md) 133 title=$(meta "$md" title) 134 render "$md" "$SITE_DIR/${slug}.html" "$title" 135 printf '%s\t%s\n' "$slug" "$title" >> "$WORK/pages.tsv" 136 done 137 138 # ── list fragments for the index ──────────────────────────── 139 TAB=$(printf '\t') 140 141 while IFS="$TAB" read -r order date slug title; do 142 [ -z "$slug" ] && continue 143 printf '<li><a href="%s.html">%s</a></li>\n' \ 144 "$slug" "$(printf '%s' "$title" | esc)" 145 done < "$WORK/sorted.tsv" > "$WORK/post-items.html" 146 147 LC_ALL=C sort -f "$WORK/pages.tsv" | while IFS="$TAB" read -r slug title; do 148 [ -z "$slug" ] && continue 149 printf '<li><a href="%s.html">%s</a></li>\n' \ 150 "$slug" "$(printf '%s' "$title" | esc)" 151 done > "$WORK/page-items.html" 152 153 # ── build index ───────────────────────────────────────────── 154 # The fragments are read from disk rather than substituted into an awk 155 # variable, so titles containing & or \ survive intact. 156 awk -v postfile="$WORK/post-items.html" -v pagefile="$WORK/page-items.html" ' 157 /<!-- POST_LIST -->/ { 158 print "<ul class=\"post-list\">" 159 while ((getline line < postfile) > 0) print line 160 close(postfile) 161 print "</ul>" 162 next 163 } 164 /<!-- PAGE_LIST -->/ { 165 print "<ul class=\"page-list\">" 166 while ((getline line < pagefile) > 0) print line 167 close(pagefile) 168 print "</ul>" 169 next 170 } 171 { print } 172 ' index.html > "$SITE_DIR/index.html" 173 174 # ── feed ───────────────────────────────────────────────────── 175 # Only the newest FEED_SIZE posts go in, and that is deliberate as well as 176 # conventional: a feed reader shows each item's date, and the newest posts are 177 # the ones whose dates are exact rather than recovered from a crawl. 178 179 # RFC 822, as RSS requires. The C locale matters: without it the day and 180 # month names come out in the system language. `date -d` is GNU; where it is 181 # missing the item simply goes out without a pubDate. 182 feed_date() { 183 LC_ALL=C date -u -d "$1" '+%a, %d %b %Y 00:00:00 +0000' 2>/dev/null 184 } 185 186 # A feed reader shows the HTML away from the site, so links relative to it 187 # need the site put in front of them. 188 absolute() { 189 sed -E \ 190 -e 's@(href|src)="/([^/"][^"]*)"@\1="'"$SITE_URL"'/\2"@g' \ 191 -e 's@(href|src)="([^"/#][^":]*)"@\1="'"$SITE_URL"'/\2"@g' 192 } 193 194 { 195 printf '<?xml version="1.0" encoding="UTF-8"?>\n' 196 printf '<rss version="2.0" xmlns:atom="http://www.w3.org/2005/Atom">\n<channel>\n' 197 printf ' <title>Recently Written</title>\n' 198 printf ' <link>%s/</link>\n' "$SITE_URL" 199 printf ' <description>A site full of things which I have recently written.</description>\n' 200 printf ' <language>en-us</language>\n' 201 printf ' <atom:link href="%s/rss.xml" rel="self" type="application/rss+xml" />\n' "$SITE_URL" 202 head -n "$FEED_SIZE" "$WORK/sorted.tsv" | while IFS="$TAB" read -r order date slug title; do 203 url="$SITE_URL/$slug.html" 204 printf ' <item>\n' 205 printf ' <title>%s</title>\n' "$(printf '%s' "$title" | esc)" 206 printf ' <link>%s</link>\n' "$url" 207 printf ' <guid isPermaLink="true">%s</guid>\n' "$url" 208 pub=$(feed_date "$date") && [ -n "$pub" ] && 209 printf ' <pubDate>%s</pubDate>\n' "$pub" 210 printf ' <description><![CDATA[' 211 pandoc --from markdown --to html5 "$POSTS_DIR/$slug.md" | 212 absolute | sed 's/]]>/]]]]><![CDATA[>/g' 213 printf ']]></description>\n' 214 printf ' </item>\n' 215 done 216 printf '</channel>\n</rss>\n' 217 } > "$SITE_DIR/rss.xml" 218 219 echo "Built $(ls "$SITE_DIR"/*.html | wc -l | tr -d ' ') pages → $SITE_DIR/" 220 echo " posts: $(wc -l < "$WORK/posts.tsv" | tr -d ' ') pages: $(wc -l < "$WORK/pages.tsv" | tr -d ' ') feed: $(grep -c '<item>' "$SITE_DIR/rss.xml") items" 221 222 # ── deploy ─────────────────────────────────────────────────── 223 # git/ and git.html share this web root but are published by deploy-git.sh, 224 # not built here. Excluding them keeps --delete from treating them as stale: 225 # rsync never deletes an excluded path on the receiving side. 226 # static/book/ holds book files that stay out of the repo (see .gitignore). 227 # They exist only on the server, so the same rule protects them. 228 if [ "$DEPLOY" = yes ]; then 229 rsync -avzP --delete --exclude=/git/ --exclude=/git.html \ 230 --exclude=/static/book/ \ 231 "$SITE_DIR/" "$DEPLOY_HOST:$DEPLOY_PATH" 232 ssh "$DEPLOY_HOST" "chmod -R a+rX $DEPLOY_PATH" 233 fi