| git.druid.rocks | index | druid520 | wd | wd.sh |
wd.sh
#!/bin/sh
# wd - minimal wikidot cli, purely for scp-wiki.wikidot.com and
# wanderers-library.wikidot.com. login, read, write, upload, status.
# nothing else.
#
# why this exists instead of browsh/a real browser: wikidot has no public
# write api, but its *private* ajax-module-connector.php (what the site's own
# js calls) is plain form-post -> json, and login/edit/upload are each just
# one or two such calls. reverse-engineered against pyscp's proven
# implementation (github.com/anqxyr/pyscp, MIT), not guessed.
#
# csrf: wikidot checks that a cookie named wikidot_token7 matches a same-named
# post field. it does NOT check the value against anything server-side - any
# value works as long as both copies match. we hardcode 123456 everywhere,
# same as every other wikidot client does.
#
# session: logging in at www.wikidot.com sets a WIKIDOT_SESSION_ID cookie
# scoped to the whole .wikidot.com domain, so one login covers both target
# sites - no per-site login step.
#
# deps: curl (required), links (optional, nicer `read` rendering - falls
# back to a plain tag-strip if missing).
set -eu
# -l anywhere as the first argument: re-run ourselves with the rest of the
# args, page combined stdout+stderr through less. stdin is untouched by the
# pipe (only fd 1/2 are redirected) so login's password prompt still reads
# straight from the real terminal - only the *output* gets paged.
if [ "${1:-}" = "-l" ]; then
shift
"$0" "$@" 2>&1 | less
exit $?
fi
WD_HOME="${WD_HOME:-$HOME/.wd}"
JAR="$WD_HOME/cookies"
TOKEN='wikidot_token7=123456'
die() { echo "wd: $*" >&2; exit 1; }
usage() {
cat <<'EOF'
usage:
wd [-l] login prompt for wikidot username/password
wd [-l] logout drop the saved session
wd [-l] read <site> <page> [-s] read a page (rendered text; -s = raw wikitext source)
wd [-l] write <site> <page> [file] [-t title] [-m comment]
create/edit a page. no file: opens the
current source in $EDITOR and asks
before saving. title defaults to the
page's current title (or its slug, for
a new page) if -t is omitted.
wd [-l] upload <site> <page> <file> attach a local file to a page
wd [-l] status logged-in-as check + every page you've
posted on both sites (empty if not
logged in - it can't know who "you"
are without a saved login)
-l pages the command's output through less (stdin, e.g. login's password
prompt, still goes straight to your terminal).
<site> is "scp", "wl", or any *.wikidot.com domain.
scp -> scp-wiki.wikidot.com
wl -> wanderers-library.wikidot.com
<page> is the page slug, e.g. scp-173, or component:foo.
EOF
exit 1
}
site_host() {
case "$1" in
scp) echo "scp-wiki.wikidot.com" ;;
wl|wanderers|wanderers-library) echo "wanderers-library.wikidot.com" ;;
*.wikidot.com) echo "$1" ;;
*) die "unknown site '$1' (use scp, wl, or a *.wikidot.com host)" ;;
esac
}
need_jar() {
[ -s "$JAR" ] || die "not logged in - run 'wd login' first"
}
# jfield JSON KEY - pull a flat ("key":"val" or "key":123) field's value.
# only safe for simple scalar fields, never for the big "body" blob.
jfield() {
printf '%s' "$1" | sed -n 's/.*"'"$2"'":"\{0,1\}\([^",}]*\)"\{0,1\}[,}].*/\1/p' | head -n1
}
# jbody JSON KEY - pull a json STRING field's value and unescape it, the
# proper way (walking the string tracking backslash-escapes), because wikidot's
# json puts other fields (jsInclude, callbackIndex, ...) after "body", so a
# naive "match to the last quote" sed pattern grabs way too much.
jbody() {
# JSTR/JKEY go through ENVIRON, not -v: awk's -v does its own backslash
# escape decoding on the assigned value, which corrupts json's own \"
# \/ \n before this loop ever sees them.
JSTR="$1" JKEY="$2" awk '
BEGIN {
s = ENVIRON["JSTR"]
key = ENVIRON["JKEY"]
pat = "\"" key "\":\""
i = index(s, pat)
if (i == 0) exit 1
i += length(pat)
n = length(s)
esc = 0
out = ""
for (j = i; j <= n; j++) {
c = substr(s, j, 1)
if (esc) {
if (c == "n") out = out "\n"
else if (c == "t") out = out "\t"
else out = out c
esc = 0
} else if (c == "\\") {
esc = 1
} else if (c == "\"") {
break
} else {
out = out c
}
}
printf "%s", out
}'
}
# strip the wikidot-source html wrapper (<br />, tags, entities) down to
# plain wikitext, once the json string itself has been unescaped.
source_unwrap() {
sed -e 's/<br[^>]*>/\n/g' -e 's/<[^>]*>//g' \
-e 's/>/>/g' -e 's/</</g' -e 's/"/"/g' -e "s/'/'/g" \
-e 's/&/\&/g'
}
# module HOST FIELDS... - post to ajax-module-connector.php, print raw json.
module() {
host="$1"; shift
curl -sS --max-time 30 -b "$JAR" -b "$TOKEN" -c "$JAR" "$@" \
-d "$TOKEN" "https://$host/ajax-module-connector.php"
}
cmd_login() {
printf 'wikidot username: '
read -r user
printf 'wikidot password: '
stty -echo 2>/dev/null || true
read -r pass
stty echo 2>/dev/null || true
echo
mkdir -p "$WD_HOME"
chmod 700 "$WD_HOME"
resp=$(curl -sS --max-time 30 -b "$TOKEN" -c "$JAR" \
--data-urlencode "login=$user" \
--data-urlencode "password=$pass" \
-d "action=Login2Action" -d "event=login" -d "$TOKEN" \
"https://www.wikidot.com/default--flow/login__LoginPopupScreen")
unset pass
chmod 600 "$JAR" 2>/dev/null || true
if printf '%s' "$resp" | grep -qi "login and password do not match\|alert-danger"; then
rm -f "$JAR"
die "login failed - wrong username/password"
fi
# saved alongside the cookie jar so a later `wd status` (a separate
# invocation - there's no other state kept between runs) knows whose
# posts to list without asking again.
printf '%s\n' "$user" > "$WD_HOME/user"
chmod 600 "$WD_HOME/user" 2>/dev/null || true
echo "logged in as $user (session saved to $JAR)"
}
cmd_logout() {
rm -f "$JAR" "$WD_HOME/user"
echo "logged out"
}
# session_user - print the saved username, but only if the session cookie
# still actually authenticates. a jar file existing isn't proof by itself:
# wikidot sessions expire server-side same as any other. there's no
# dedicated whoami api, so this checks the login-status widget every
# wikidot.com page embeds - it says "sign in"/"create account" when
# anonymous, something else (the account name) when not.
session_user() {
[ -s "$JAR" ] && [ -s "$WD_HOME/user" ] || return 1
html=$(curl -sS --max-time 15 -b "$JAR" -b "$TOKEN" "https://www.wikidot.com/" 2>/dev/null)
case "$html" in
*login-status-sign-in*) return 1 ;;
esac
cat "$WD_HOME/user"
}
# list_pages HOST USER - every page USER has authored on HOST. walks
# list/ListPagesModule's pagination (100/page) until it runs out.
#
# wikidot's own pagination has a real quirk, confirmed live: ask for a
# page number past the actual last page and it silently RE-SERVES page 1
# instead of coming back empty. so two separate stop conditions, both
# needed: a short page (fewer rows than perPage) proves it was the last
# one; and if a page is ever byte-identical to the one before it, that's
# wikidot re-serving page 1 out of range, not 100 more real posts, so it's
# discarded rather than printed. capped at 20 pages (2000 posts) either
# way, so a lookup can't loop forever if something about the response
# ever stops matching this.
list_pages() {
host="$1"; user="$2"; p=1; per=100; prev=""
while [ "$p" -le 20 ]; do
resp=$(module "$host" -d "moduleName=list/ListPagesModule" \
--data-urlencode "createdBy=$user" -d "order=created_at desc" \
-d "perPage=$per" -d "p=$p" \
--data-urlencode "module_body=%%title%% (%%fullname%%) - %%created_at%%")
[ "$(jfield "$resp" status)" = "ok" ] || break
# pull only what's actually inside a <p>...</p> (one per result,
# always on a single line here) rather than tag-stripping the
# whole response body - a later page's "page N of M ... next"
# pager widget lives in that same body and isn't valid to just
# blanket-strip around, it has to be excluded structurally.
body=$(jbody "$resp" body | sed -n 's/^.*<p>\(.*\)<\/p>.*$/\1/p' \
| sed -e 's/<[^>]*>//g' -e 's/>/>/g' -e 's/</</g' \
-e 's/ / /g' -e 's/&/\&/g' \
| sed -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//' | sed '/^$/d')
[ -n "$body" ] || break
[ "$body" = "$prev" ] && break
printf '%s\n' "$body"
prev="$body"
n=$(printf '%s\n' "$body" | wc -l)
[ "$n" -ge "$per" ] || break
p=$((p + 1))
done
}
cmd_status() {
user=$(session_user) || { echo "not logged in"; return 0; }
echo "logged in as $user"
for alias in scp wl; do
host=$(site_host "$alias")
echo
echo "posts on $host:"
lp=$(list_pages "$host" "$user")
if [ -n "$lp" ]; then
printf '%s\n' "$lp" | sed 's/^/ /'
else
echo " (none found)"
fi
done
}
# page_id HOST PAGE - scrape the numeric page id out of the plain page html.
# empty output means the page doesn't exist yet (fine for `write`, an error
# for `read`/`upload`).
page_id() {
curl -sS --max-time 30 -b "$JAR" -b "$TOKEN" "https://$1/$2" 2>/dev/null \
| sed -n 's/.*pageId = \([0-9]*\);.*/\1/p' | head -n1
}
# current_title HOST PAGE - scrape the page's displayed title, e.g. for
# `write` to default to when -t isn't given (never send a blank title over
# an existing one - savePage sets whatever title it's given, it doesn't
# leave an omitted one alone).
current_title() {
curl -sS --max-time 30 -b "$JAR" -b "$TOKEN" "https://$1/$2" 2>/dev/null \
| sed -n '/id="page-title"/,/<\/div>/p' \
| sed -e 's/<[^>]*>//g' -e 's/^[[:space:]]*//' -e 's/[[:space:]]*$//' \
| sed '/^$/d' | head -n1
}
# fetch_source HOST PAGE_ID - raw wikitext source of an existing page, via
# the same authenticated viewsource call `read -s` uses. shared so `write`'s
# edit-in-place mode starts from real content, not a blank file.
fetch_source() {
resp=$(module "$1" -d "moduleName=viewsource/ViewSourceModule" -d "page_id=$2")
[ "$(jfield "$resp" status)" = "ok" ] || die "viewsource failed: $(jfield "$resp" message)"
jbody "$resp" body | source_unwrap | sed '1{/^Page source$/d}'
}
cmd_read() {
[ $# -ge 2 ] || usage
host=$(site_host "$1"); page="$2"; src=0
[ "${3:-}" = "-s" ] && src=1
if [ "$src" = 1 ]; then
pid=$(page_id "$host" "$page")
[ -n "$pid" ] || die "no such page: $page on $host"
fetch_source "$host" "$pid"
elif command -v links >/dev/null 2>&1; then
# links -dump wants an actual file, not stdin - it needs to seek
# to sniff the content type.
tmp=$(mktemp)
trap 'rm -f "$tmp"' EXIT
curl -sS --max-time 30 -b "$JAR" -b "$TOKEN" "https://$host/$page" -o "$tmp"
links -dump "$tmp"
rm -f "$tmp"
trap - EXIT
else
curl -sS --max-time 30 -b "$JAR" -b "$TOKEN" "https://$host/$page" \
| sed -e '/<script/,/<\/script>/d' -e '/<style/,/<\/style>/d' \
-e 's/<[^>]*>//g' -e 's/>/>/g' -e 's/</</g' -e 's/&/\&/g' \
| sed '/^[[:space:]]*$/d'
fi
}
cmd_write() {
[ $# -ge 2 ] || usage
host=$(site_host "$1"); page="$2"; shift 2
need_jar
# a bare leading non-option arg, if present, is the source file. if
# it's missing entirely we fall into edit-in-place mode below instead
# of demanding one, since "open it, change it, save it" is the common
# case and shouldn't need a scratch file made by hand every time.
file=""
if [ $# -gt 0 ] && [ "${1#-}" = "$1" ]; then
file="$1"; shift
fi
title=""; title_set=0; comment="wd edit"
while [ $# -gt 0 ]; do
case "$1" in
-t) title="$2"; title_set=1; shift 2 ;;
-m) comment="$2"; shift 2 ;;
*) die "unknown option: $1" ;;
esac
done
[ -z "$file" ] || [ -f "$file" ] || die "no such file: $file"
pid=$(page_id "$host" "$page")
# never send a blank title over an existing one, and give a new page
# something better than blank too - savePage sets exactly what it's
# given, it doesn't leave an omitted title alone.
if [ "$title_set" = 0 ]; then
if [ -n "$pid" ]; then
title=$(current_title "$host" "$page")
else
title="$page"
fi
fi
tmpfile=""
if [ -z "$file" ]; then
tmpfile=$(mktemp)
trap 'rm -f "$tmpfile"' EXIT
[ -n "$pid" ] && fetch_source "$host" "$pid" > "$tmpfile"
"${EDITOR:-vi}" "$tmpfile"
[ -s "$tmpfile" ] || die "empty after editing, not saving"
printf 'save %s on %s? [y/N] ' "$page" "$host"
read -r ans
case "$ans" in y|Y|yes|YES) ;; *) echo "aborted"; exit 0 ;; esac
file="$tmpfile"
fi
lock=$(module "$host" -d "moduleName=edit/PageEditModule" -d "mode=page" \
-d "wiki_page=$page" ${pid:+-d "page_id=$pid"} -d "force_lock=true")
[ "$(jfield "$lock" status)" = "ok" ] || die "could not lock page for edit: $(jfield "$lock" message)"
lock_id=$(jfield "$lock" lock_id)
lock_secret=$(jfield "$lock" lock_secret)
rev_id=$(jfield "$lock" page_revision_id)
[ -n "$lock_id" ] && [ -n "$lock_secret" ] || die "lock response missing lock_id/lock_secret - wikidot may have changed its api"
save=$(curl -sS --max-time 30 -b "$JAR" -b "$TOKEN" -c "$JAR" \
-d "moduleName=Empty" -d "action=WikiPageAction" -d "event=savePage" \
-d "mode=page" -d "wiki_page=$page" ${pid:+-d "page_id=$pid"} \
-d "lock_id=$lock_id" --data-urlencode "lock_secret=$lock_secret" \
${rev_id:+-d "revision_id=$rev_id"} \
--data-urlencode "title=$title" \
--data-urlencode "comments=$comment" \
--data-urlencode "source@$file" \
-d "$TOKEN" "https://$host/ajax-module-connector.php")
[ -z "$tmpfile" ] || { rm -f "$tmpfile"; trap - EXIT; }
[ "$(jfield "$save" status)" = "ok" ] || die "save failed: $(jfield "$save" message)"
echo "saved $page on $host"
}
cmd_upload() {
[ $# -ge 3 ] || usage
host=$(site_host "$1"); page="$2"; file="$3"
[ -f "$file" ] || die "no such file: $file"
need_jar
pid=$(page_id "$host" "$page")
[ -n "$pid" ] || die "no such page: $page on $host (create it first with 'wd write')"
resp=$(curl -sS --max-time 120 -b "$JAR" -b "$TOKEN" -c "$JAR" \
-F "action=FileAction" -F "event=uploadFile" -F "page_id=$pid" \
-F "MAX_FILE_SIZE=52428800" -F "userfile=@$file" \
"https://$host/default--flow/files__UploadTarget")
if printf '%s' "$resp" | grep -qi 'error\|denied'; then
die "upload may have failed - server said: $(printf '%s' "$resp" | sed -e 's/<[^>]*>//g' | tr -s ' \n' ' ')"
fi
echo "uploaded $file to $page on $host"
}
[ $# -ge 1 ] || usage
cmd="$1"; shift
case "$cmd" in
login) cmd_login "$@" ;;
logout) cmd_logout "$@" ;;
read) cmd_read "$@" ;;
write) cmd_write "$@" ;;
upload) cmd_upload "$@" ;;
status) cmd_status "$@" ;;
*) usage ;;
esac