Hello Navi

Tech, Security & Personal Notes

Challenge

JS Obfuscation. FTW!

Find the password:

索引名 JS Obfuscation. FTW!,难度 moderate。密码比较逻辑藏在一个运行时由十六进制字符串数组加 String.fromCharCode(...) 生成出来的 <button> 里。要先把混淆还原出来。

Solution

1
2
3
$ curl -s -b "$HTS_COOKIE" \
-H "Referer: https://www.hackthissite.org/missions/javascript/7/" \
"https://www.hackthissite.org/missions/javascript/7/" -o lvl7.html
1
var _0x4e9d=["\x66\x72\x6F\x6D\x43\x68\x61\x72\x43\x6F\x64\x65","\x77\x72\x69\x74\x65"];document[_0x4e9d[0x1]](String[_0x4e9d[0x0]](0x3c,0x62,0x75,0x74,0x74,0x6f,0x6e,0x20,0x6f,0x6e,0x63,0x6c,0x69,0x63,0x6b,0x3d,0x27,0x6a,0x61,0x76,0x61,0x73,0x63,0x72,0x69,0x70,0x74,0x3a,0x69,0x66,0x20,0x28,0x64,0x6f,0x63,0x75,0x6d,0x65,0x6e,0x74,0x2e,0x67,0x65,0x74,0x45,0x6c,0x65,0x6d,0x65,0x6e,0x74,0x42,0x79,0x49,0x64,0x28,0x22,0x70,0x61,0x73,0x73,0x22,0x29,0x2e,0x76,0x61,0x6c,0x75,0x65,0x3d,0x3d,0x22,0x6a,0x30,0x30,0x77,0x31,0x6e,0x22,0x29,0x7b,0x61,0x6c,0x65,0x72,0x74,0x28,0x22,0x59,0x6f,0x75,0x20,0x57,0x49,0x4e,0x21,0x22,0x29,0x3b,0x77,0x69,0x6e,0x64,0x6f,0x77,0x2e,0x6c,0x6f,0x63,0x61,0x74,0x69,0x6f,0x6e,0x20,0x2b,0x3d,0x20,0x22,0x3f,0x6c,0x76,0x6c,0x5f,0x70,0x61,0x73,0x73,0x77,0x6f,0x72,0x64,0x3d,0x22,0x2b,0x64,0x6f,0x63,0x75,0x6d,0x65,0x6e,0x74,0x2e,0x67,0x65,0x74,0x45,0x6c,0x65,0x6d,0x65,0x6e,0x74,0x42,0x79,0x49,0x64,0x28,0x22,0x70,0x61,0x73,0x73,0x22,0x29,0x2e,0x76,0x61,0x6c,0x75,0x65,0x7d,0x65,0x6c,0x73,0x65,0x20,0x7b,0x61,0x6c,0x65,0x72,0x74,0x28,0x22,0x57,0x52,0x4f,0x4e,0x47,0x21,0x20,0x54,0x72,0x79,0x20,0x61,0x67,0x61,0x69,0x6e,0x21,0x22,0x29,0x7d,0x27,0x3e,0x43,0x68,0x65,0x63,0x6b,0x20,0x50,0x61,0x73,0x73,0x77,0x6f,0x72,0x64,0x3c,0x2f,0x62,0x75,0x74,0x74,0x6f,0x6e,0x3e));
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
#!/usr/bin/env python3
"""Recover the JavaScript Mission 7 password from the obfuscated source.

The level ships two hex-string arrays and rebuilds a <button> at run time by
turning a numeric char-code list into HTML:

var _0x4e9d=["\\x66\\x72\\x6F\\x6D\\x43\\x68\\x61\\x72\\x43\\x6F\\x64\\x65",
"\\x77\\x72\\x69\\x74\\x65"];

Step 1 unescapes the \\xNN literals into identifier names
("fromCharCode","write"); step 2 turns the numeric char-code list into the
button HTML and reads the compared literal out of it -- the password.
"""
import re

src = open("lvl7.html", encoding="utf-8", errors="replace").read()

# Step 1: \xNN string literals -> identifier names
arr = re.search(r"_0x4e9d=\[(.*?)\]", src, re.S).group(1)
names = [bytes(b, "latin1").decode("unicode_escape")
for b in re.findall(r'"((?:\\x[0-9a-fA-F]{2})+)"', arr)]
print("identifiers:", names)

# Step 2: rebuild the button HTML from the String.fromCharCode() code list
call = re.search(r"String\[_0x4e9d\[0x0\]\]\(([^)]*)\)", src, re.S).group(1)
codes = [int(x, 16) for x in re.findall(r"0x([0-9a-fA-F]+)", call)]
button = "".join(chr(c) for c in codes)
print("button HTML:")
print(button)

# Step 3: the compared literal inside the button is the password
password = re.search(r'value=="([^"]+)"', button).group(1)
print("password:", password)
1
2
3
4
5
$ uv run python decode7.py
identifiers: ['fromCharCode', 'write']
button HTML:
<button onclick='javascript:if (document.getElementById("pass").value=="j00w1n"){alert("You WIN!");window.location += "?lvl_password="+document.getElementById("pass").value}else {alert("WRONG! Try again!")}'>Check Password</button>
password: j00w1n
j00w1n

Challenge

Fiftysixer decided to try his hand at javascript! All was going well until he realized that he forgot to remove the unused code, which resulted in a confusing mess. He didn't mind, in fact, he did his best to make it even MORE confusing!

Find the password:

索引名 go go away .js,难度 Weird。主页面只留一段自相矛盾的误导脚本,真正的比较藏在外部文件 checkpass.js 里。

Solution

  • 关卡页 https://www.hackthissite.org/missions/javascript/6/ 通过 <script src="/missions/javascript/6/checkpass.js"></script> 引入外部脚本。
  • 页面内联脚本故意留下大量未使用且自相矛盾的代码:check() 里出现 "hack_this_site" 字面量和跳向 about:blank 的分支,这些都被真链路旁置。
  • 表单真正调用的是内联的 checkpassw(this.value),它把输入写进 RawrRawr 再调用外部脚本里的 checkpass()。

拉取主页面与外部脚本:

1
2
3
4
5
6
$ curl -s -b "$HTS_COOKIE" \
-H "Referer: https://www.hackthissite.org/missions/javascript/6/" \
"https://www.hackthissite.org/missions/javascript/6/" -o lvl6.html
$ curl -s -b "$HTS_COOKIE" \
-H "Referer: https://www.hackthissite.org/missions/javascript/6/" \
"https://www.hackthissite.org/missions/javascript/6/checkpass.js" -o checkpass.js
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
RawrRawr = "moo";
function check(x) {
"+RawrRawr+" == "hack_this_site";
if (x == "" + RawrRawr + "") {
alert("Rawr! win!");
window.location = "about:blank";
} else {
alert("Rawr, nope, try again!");
}
}

function checkpassw(moo) {
RawrRawr = moo;
checkpass(RawrRawr);
}

check() 里那条 "+RawrRawr+" == "hack_this_site" 是一条无副作用的裸表达式语句,about:blank 分支也永远不会给站点送出密码:表单并不调用 check()。真正的入口是 checkpassw(moo) → checkpass(RawrRawr),而 checkpass 定义在外部文件里。这正是题面说的忘了删掉没用的代码。

1
2
3
4
5
6
7
8
9
10
11
12
dairycow = "moo";
moo = "pwns";
rawr = "moo";

function checkpass(pass) {
if (pass == rawr + " " + moo) {
alert("How did you do that??? Good job!");
window.location = "../../../missions/javascript/6/?lvl_password=" + pass;
} else {
alert("Nope, try again");
}
}

决定性的比较是 pass == rawr+" "+moo:rawr="moo"、moo="pwns",中间夹一个空格字面量 " "。dairycow="moo" 在本文件里没被使用

从抓下来的 checkpass.js 里把变量逐个读出来再按比较式拼接:

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
#!/usr/bin/env python3
"""Recover the JavaScript Mission 6 password from the external script.

index.php loads /missions/javascript/6/checkpass.js; the winning comparison
there is pass == rawr+" "+moo with dairycow="moo", rawr="moo", moo="pwns".

The inline check() (RawrRawr / "hack_this_site" / about:blank) is dead code;
the form calls checkpassw() which forwards to the external checkpass().
The password is exactly "rawr + ' ' + moo" -- the single space included.
"""
import re
import urllib.parse


def main():
js = open("checkpass.js", encoding="utf-8", errors="replace").read()
vals = dict(re.findall(r'(\w+)\s*=\s*"([^"]*)"', js))
missing = {name for name in ("rawr", "moo") if name not in vals}
if missing:
raise SystemExit("missing required assignment(s): " + ", ".join(sorted(missing)))
password = "%s %s" % (vals["rawr"], vals["moo"])
print("rawr :", vals["rawr"])
print("moo :", vals["moo"])
print("password :", repr(password))
print("query : lvl_password=" + urllib.parse.quote(password))


if __name__ == "__main__":
main()
1
2
3
4
5
$ uv run python decode6.py
rawr : moo
moo : pwns
password : 'moo pwns'
query : lvl_password=moo%20pwns

Challenge

Javascript Mission 5: Uhm, faith spelled runescape wrong?

索引名为 Escape!,难度 easy。页面上只有一个密码框和 Check Password 按钮,密码被一层百分号编码包着,用 unescape() 在浏览器里解出来。

Solution

  • 页面没有任何把密码送去服务端校验的请求:比对全在客户端的 check() 里完成,命中后用 window.location = "../../../missions/javascript/5/?lvl_password="+x 回跳。

拉取页面源码:

1
2
3
$ curl -s -b "$HTS_COOKIE" \
-H "Referer: https://www.hackthissite.org/missions/javascript/5/" \
"https://www.hackthissite.org/missions/javascript/5/" -o lvl5.html

页面内联脚本(业务部分全文):

1
2
3
4
5
6
7
8
9
moo = unescape("%69%6C%6F%76%65%6D%6F%6F");
function check(x) {
if (x == moo) {
alert("Ahh.. so that's what she means");
window.location = "../../../missions/javascript/5/?lvl_password=" + x;
} else {
alert("Nope... try again!");
}
}

moo 直接由 unescape(...) 得到,check(x) 只做一次相等比较,相等就带着 x 回跳。

unescape() 是早期的百分号解码器,%XX 中的 XX 是字符的十六进制 ASCII 码。逐个翻译:

  • %69 = 0x69 = 105 = i
  • %6C = 0x6C = 108 = l
  • %6F = 0x6F = 111 = o
  • %76 = 0x76 = 118 = v
  • %65 = 0x65 = 101 = e
  • %6D = 0x6D = 109 = m
  • %6F = 0x6F = 111 = o
  • %6F = 0x6F = 111 = o
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
#!/usr/bin/env python3
"""Recover the JavaScript Mission 5 password from the fetched page source.

The page assigns the expected value at load time:

moo = unescape('%69%6C%6F%76%65%6D%6F%6F');

`unescape()` is the legacy percent-decoder, so the password is just that
literal decoded once. urllib.parse.unquote is the modern equivalent.
"""
import re
import urllib.parse


def main():
src = open("lvl5.html", encoding="utf-8", errors="replace").read()
match = re.search(r"unescape\('([^']*)'\)", src)
if match is None:
raise SystemExit("could not find an unescape() literal in lvl5.html")
encoded = match.group(1)
password = urllib.parse.unquote(encoded)
print("encoded :", encoded)
print("decoded :", repr(password))


if __name__ == "__main__":
main()
1
2
3
$ uv run python decode5.py
encoded : %69%6C%6F%76%65%6D%6F%6F
decoded : 'ilovemoo'
1
2
3
4
$ curl -s -o /dev/null -w '%{http_code}\n' -b "$HTS_COOKIE" \
-H "Referer: https://www.hackthissite.org/missions/javascript/5/" \
"https://www.hackthissite.org/missions/javascript/5/?lvl_password=ilovemoo"
200

Challenge

Faith is trying to trick you.... The checker has a dead comparison that looks like the answer; the real password is a one-line variable.

第四关页面开头写着 Faith is trying to trick you...(Faith 意在误导)。校验函数里有一行看似答案的废弃比较,真正的密码只是一个变量。

Solution

用带登录态的会话取关卡页:

1
2
$ curl -s -b 'HackThisSite=<mission-cookie>' \
'https://www.hackthissite.org/missions/javascript/4/'

页面里的校验函数:

1
2
3
4
5
6
7
8
9
10
RawrRawr = "moo";
function check(x) {
"" + RawrRawr + "" == "hack_this_site";
if (x == "" + RawrRawr + "") {
alert("Rawr! win!");
window.location = "../../../missions/javascript/4/?lvl_password=" + x;
} else {
alert("Rawr, fail.");
}
}
  • RawrRawr = "moo" 是全局赋值,值就写在源码里。
  • ""+RawrRawr+"" == "hack_this_site"; 是干扰语句。它确实计算了一个字符串拼接("" + "moo" + "" 得到 "moo"),也确实拿它和 "hack_this_site" 做了比较,但这个比较的布尔结果被直接丢弃、没有赋给任何变量、也没有参与后面的 if。用 Node 单独跑这一行就能看到它恒为 false:
1
2
$ node -e 'var RawrRawr = "moo"; console.log(""+RawrRawr+"" == "hack_this_site")'
false
  • 真正生效的条件是下一行的 if (x == ""+RawrRawr+"")。"" + RawrRawr + "" 只是把变量转成字符串,结果还是 "moo",所以比较等价于 x == "moo"。与空串拼接不会改变值:这是题目用来混淆视觉的,不是用来加密的。
1
2
$ node -e 'var RawrRawr = "moo"; function check(x){ return x == ""+RawrRawr+""; } console.log(check("moo"))'
true
  • 放行动作是把 x 拼进 URL 回跳:window.location = "../../../missions/javascript/4/?lvl_password="+x。和前面几关一样,客户端只负责拼 URL,拼接结果完全可以手工构造。

Verify

1
2
3
$ curl -s -b 'HackThisSite=<mission-cookie>' \
-e 'https://www.hackthissite.org/missions/javascript/4/' \
'https://www.hackthissite.org/missions/javascript/4/?lvl_password=moo'

Challenge

Math time!. The page hands you the checker source in a textarea; the password is a length, not a secret string.

第三关 Math time!:页面直接把校验源码放在一个 textarea 里,密码是一段算出来的长度,不是一个固定字符串。

Solution

用带登录态的会话取关卡页:

1
2
$ curl -s -b 'HackThisSite=<mission-cookie>' \
'https://www.hackthissite.org/missions/javascript/3/'

页面里的 textarea 给出这段源码:

1
2
3
4
5
6
7
8
9
10
11
12
var foo = 5 + 6 * 7;
var bar = foo % 8;
var moo = bar * 2;
var rar = moo / 3;
function check(x) {
if (x.length == moo) {
alert("win!");
window.location += "?lvl_password=" + x;
} else {
alert("Fail D:");
}
}
  • foo = 5 + 6 * 7。乘法优先:6 * 7 = 42,5 + 42 = 47。
  • bar = foo % 8。47 % 8 = 7(47 = 5 * 8 + 7)。
  • moo = bar * 2。7 * 2 = 14。
  • rar = moo / 3。14 / 3 = 4.666…。这一行算完就没被用过,是干扰项:校验里只引用了 moo。

用 Node 运行一遍确认:

1
2
$ node -e 'var foo = 5 + 6 * 7, bar = foo % 8, moo = bar * 2, rar = moo / 3; console.log(foo, bar, moo, rar)'
47 7 14 4.666666666666667

校验条件是 x.length == moo,也就是 x.length == 14。

Verify

提交一个 14 字符的串:

1
2
3
$ curl -s -b 'HackThisSite=<mission-cookie>' \
-e 'https://www.hackthissite.org/missions/javascript/3/' \
'https://www.hackthissite.org/missions/javascript/3/?lvl_password=aaaaaaaaaaaaaa'

Challenge

Disable Javascript. Loading the page immediately kicks you to a fail page; the win link is hidden in the same HTML.

第二关 Disable Javascript:页面加载后立即跳转到失败页,通向通关的链接就在同一份 HTML 里。

Solution

用带登录态的会话取关卡页:

1
2
$ curl -s -b 'HackThisSite=<mission-cookie>' \
'https://www.hackthissite.org/missions/javascript/2/'

服务端返回的正文里,开头只有一条跳转脚本:

1
2
3
4
<script>
window.location =
"http://www.hackthissite.org/missions/javascript/2/fail.php";
</script>

同一份 HTML 的下方藏着一个锚点:

1
2
3
<a href="/missions/javascript/2/index.php?challengePass=<串>"
>Click here to win.</a
>

抓页面并提交

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
#!/usr/bin/env python3
"""HackThisSite JavaScript 2 ("Disable Javascript").

The level page only runs `window.location = ".../2/fail.php"`; the win link is
hidden in the same HTML. index.php regenerates the challengePass token on
every load, so fetch and submit must happen back to back in one session:

usage: HTS_COOKIE=<live-cookie> ./solve.py
"""
import os
import re
import sys
import urllib.parse
import urllib.request

BASE = "https://www.hackthissite.org/missions/javascript/2/"
COOKIE = os.environ.get("HTS_COOKIE", "<mission-cookie>")

def get(url, referer):
req = urllib.request.Request(url)
req.add_header("Cookie", "HackThisSite=" + COOKIE)
req.add_header("Referer", referer)
with urllib.request.urlopen(req) as rsp:
return rsp.read().decode("latin-1")

def main():
page = get(BASE, BASE)
match = re.search(r'challengePass=([^"&]+)', page)
if not match:
sys.exit("no challengePass token in page")
token = match.group(1)
print("token:", token)
win_url = BASE + "index.php?" + urllib.parse.urlencode({"challengePass": token})
body = get(win_url, BASE)
print("submitted:", win_url)
print(body[:200])

if __name__ == "__main__":
main()

urllib.parse.urlencode 会把串里的 @、%、$、* 这些字符正确转义进查询串。服务端接受的等价 curl 形式是:

1
2
3
4
$ curl -s -b 'HackThisSite=<mission-cookie>' \
-e 'https://www.hackthissite.org/missions/javascript/2/' \
-G 'https://www.hackthissite.org/missions/javascript/2/index.php' \
--data-urlencode 'challengePass=EK@1%I'

Challenge

Idiot Test. The page is a single password box and a button; the check runs entirely in the browser.

第一关 Idiot Test:页面只有一个密码输入框和一个提交按钮,点按钮时在浏览器里执行页内 JS 校验。

页面结构是 input#pass 加一个 <button onclick="javascript:check(document.getElementById('pass').value)">。按钮不提交任何表单,点击只是把输入框的值交给页内函数 check()。

Solution

用带登录态的会话取关卡页源码(HackThisSite 是短期凭据,这里写成占位符):

1
2
$ curl -s -b 'HackThisSite=<mission-cookie>' \
'https://www.hackthissite.org/missions/javascript/1/'

页面里负责校验的函数:

1
2
3
4
5
6
7
8
function check(x) {
if (x == "cookies") {
alert("win!");
window.location += "?lvl_password=" + x;
} else {
alert("Fail D:");
}
}

Step 1: 为什么这段 JS 能被绕过

  • check() 把输入和字面量 "cookies" 直接比较,密码就硬编码在客户端 JavaScript 里。浏览器拿到的源码就是全部逻辑,查看源码或 devtools 即可读出,不需要任何猜测。
  • 真正决定通关的动作是成功分支里的 window.location += "?lvl_password=" + x。它把密码追加到当前 URL 上,再作为一个普通 GET 回跳给服务端。也就是说,客户端 JS 只决定是否发起这次跳转;跳转本身可以手工构造,无需运行页面。
  • 用 Node 把这段逻辑提取出来运行一遍,即可确认比较结果:
1
2
$ node -e 'function check(x){ return x == "cookies"; } console.log(check("cookies"))'
true

Verify

把 cookies 填进输入框点按钮,等价于对关卡页发起这次请求:

1
2
3
$ curl -s -b 'HackThisSite=<mission-cookie>' \
-e 'https://www.hackthissite.org/missions/javascript/1/' \
'https://www.hackthissite.org/missions/javascript/1/?lvl_password=cookies'

服务端靠 Referer 判断请求来自关卡页:同一 URL 不带 Referer 时不计完成,补上 -e 指向的关卡页地址后即计入完成。

Challenge

Level 12 — String manipulation

页面给出一串随机字符串。把所有数字取出来,按质数/合数分类(都是一位数,0 和 1 不计):先求所有合数之和,再求所有质数之和,两者相乘得到一个乘积。然后取字符串里前 25 个非数字字符,把每个字符的 ASCII 值加一(例:# → $),把这 25 个字符与乘积首尾相接连成答案。限时 5 秒。

Your answer should look like this: oc{lujxdpb%jvqrt{luruudtx140224

Solution

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
#!/usr/bin/env python3
"""Solve HackThisSite Programming Mission 12."""

import html
import os
import re

import requests

BASE = "https://www.hackthissite.org"
LEVEL_URL = f"{BASE}/missions/prog/12/"
USER_AGENT = (
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
)
PRIMES = {2, 3, 5, 7}
COMPOSITES = {4, 6, 8, 9}


def make_session():
cookie = os.environ.get("HTS_COOKIE", "").strip().strip("'\"")
if not cookie:
raise SystemExit("HTS_COOKIE is not set")
client = requests.Session()
client.headers.update(
{
"User-Agent": USER_AGENT,
"Cookie": cookie,
"Referer": f"{BASE}/missions/programming/",
"Accept-Language": "en-US,en;q=0.9",
}
)
return client


def body_text(source):
source = re.sub(r"<script.*?</script>", "", source, flags=re.S)
source = re.sub(r"<br\s*/?>", "\n", source)
source = re.sub(r"</(?:p|div|tr)>", "\n", source)
source = re.sub(r"<[^>]+>", " ", source)
source = html.unescape(source)
return re.sub(r"\n\s*\n+", "\n", re.sub(r"[ \t]+", " ", source))


def parse(page_text):
match = re.search(r'<input type="text" value="([^"]+)"', page_text)
if match is None:
raise SystemExit("random string not found in page")
return match.group(1)


def solve(value):
digits = [int(char) for char in value if char.isdigit()]
composite_sum = sum(digit for digit in digits if digit in COMPOSITES)
prime_sum = sum(digit for digit in digits if digit in PRIMES)
product = composite_sum * prime_sum
first25 = [char for char in value if not char.isdigit()][:25]
shifted = "".join(chr(ord(char) + 1) for char in first25)
return shifted + str(product), (composite_sum, prime_sum, product)


def verdict(response_text):
text = body_text(response_text).lower()
success = any(
marker in text
for marker in (
"congratulation",
"you have completed",
"completed this",
"successfully",
"correct",
"well done",
"mission accomplished",
"complete!",
)
)
failure = any(
marker in text
for marker in (
"wrong",
"incorrect",
"not correct",
"try again",
"failed",
"too late",
"time is up",
"sorry",
)
)
if success and not failure:
return True
if failure and not success:
return False
return None


def submit(client, answer):
response = client.post(
f"{LEVEL_URL}index.php",
data={"solution": answer, "submitbutton": "submit"},
headers={"Referer": LEVEL_URL},
timeout=30,
)
response.raise_for_status()
return verdict(response.text)


def main():
client = make_session()
page = client.get(LEVEL_URL, timeout=20)
page.raise_for_status()
value = parse(page.text)
answer, info = solve(value)
composite_sum, prime_sum, product = info
print("input prefix:", value[:60])
print(
"sums: composite=%d prime=%d product=%d"
% (composite_sum, prime_sum, product)
)
print("answer:", answer)
print("verdict:", submit(client, answer))


if __name__ == "__main__":
main()

Challenge

Level 11 — Reverse Ascii Shift

页面给出一串随机生成的 % 分隔的 ASCII 码和一个 Shift 值。把每个码按 Shift 反向还原,得到的字符串就是答案(Decoded ASCII)。限时 3 秒。

This string was randomly generated. It will not be recognizable text. You have 3 seconds to take the information from the website, and apply that to your algorithm.

Solution

  • 实例页 https://www.hackthissite.org/missions/prog/11/ 的正文里有三行:Generated String: 89%71%64%55%42%89%84%37%57%、Shift: 3、Decoded ASCII。
  • 表单只有一个 solution 字段,submit (remaining time: 3 seconds)
1
2
3
4
5
6
Level 11
This string was randomly generated. It will not be recognizable text. You have 3 seconds to take the
information from the website, and apply that to your algorithm.
Generated String: 89%71%64%55%42%89%84%37%57%
Shift: 3
Decoded ASCII

% 只是分隔符,每个数字是一个被平移过的字符码。题面用词是 reverse(反向还原),所以先尝试 chr(code - shift);实测该方向被服务端直接接受:

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
#!/usr/bin/env python3
"""Solve HackThisSite Programming Mission 11."""

import html
import os
import re

import requests

BASE = "https://www.hackthissite.org"
LEVEL_URL = f"{BASE}/missions/prog/11/"
USER_AGENT = (
"Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 "
"(KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36"
)


def make_session():
cookie = os.environ.get("HTS_COOKIE", "").strip().strip("'\"")
if not cookie:
raise SystemExit("HTS_COOKIE is not set")
client = requests.Session()
client.headers.update(
{
"User-Agent": USER_AGENT,
"Cookie": cookie,
"Referer": f"{BASE}/missions/programming/",
"Accept-Language": "en-US,en;q=0.9",
}
)
return client


def body_text(source):
source = re.sub(r"<script.*?</script>", "", source, flags=re.S)
source = re.sub(r"<br\s*/?>", "\n", source)
source = re.sub(r"</(?:p|div|tr)>", "\n", source)
source = re.sub(r"<[^>]+>", " ", source)
source = html.unescape(source)
return re.sub(r"\n\s*\n+", "\n", re.sub(r"[ \t]+", " ", source))


def parse(page_text):
section = page_text.split("Generated String:", 1)[1].split("Shift:", 1)[0]
codes = [int(value) for value in re.findall(r"\d+", section)]
match = re.search(r"Shift:\s*(\d+)", page_text)
if match is None:
raise SystemExit("shift value not found")
return codes, int(match.group(1))


def decode(codes, shift, direction):
return "".join(chr(code + direction * shift) for code in codes)


def verdict(response_text):
text = body_text(response_text).lower()
success = any(
marker in text
for marker in (
"congratulation",
"you have completed",
"completed this",
"successfully",
"correct",
"well done",
"mission accomplished",
"complete!",
)
)
failure = any(
marker in text
for marker in (
"wrong",
"incorrect",
"not correct",
"try again",
"failed",
"too late",
"time is up",
"sorry",
)
)
if success and not failure:
return True
if failure and not success:
return False
return None


def submit(client, answer):
response = client.post(
f"{LEVEL_URL}index.php",
data={"solution": answer, "submitbutton": "submit"},
headers={"Referer": LEVEL_URL},
timeout=30,
)
response.raise_for_status()
return verdict(response.text), response.text


def main():
client = make_session()
page = client.get(LEVEL_URL, timeout=20)
page.raise_for_status()
codes, shift = parse(body_text(page.text))
for direction in (-1, 1):
answer = decode(codes, shift, direction)
print(f"try shift {direction * shift:+d} -> {answer!r}")
ok, _ = submit(client, answer)
print("verdict:", ok)
if ok:
return
if direction == -1:
page = client.get(LEVEL_URL, timeout=20)
page.raise_for_status()
codes, shift = parse(body_text(page.text))
print("retry codes:", codes, "shift", shift)
raise SystemExit("both directions rejected")


if __name__ == "__main__":
main()

Challenge

This level is hybrid of Programming/Stego missions, The main purpose being: Get the data from the image, and get the answer from that data. The data itself is encoded and encrypted multiple times, in the end is a 10 character password. The password consists of two characters, one upper case and one lower case, repeating in random order. Example: XyyXyXXyXy In order to get this data you must brute-force the encrypted hashes you get in each step. To do this within the time limit you must write a smart brute forcing method, good luck. Save this image to get started.

这道题是 Programming 和 Stego 的混合题:从图片里取出数据,再从数据里取出答案。数据被多层编码和加密,最后是一个 10 字符的密码。密码由两个字符组成(一个大写一个小写),随机交替排列,例如 XyyXyXXyXy。每一步都会拿到加密过的 hash,必须在限时内用足够聪明的方式爆破。把图片存下来开始吧。

实例页 https://www.hackthissite.org/missions/prog/10/ 里图片地址是随机的 image.php?<随机数>,表单只有 solution 一个字段,限时 45 秒;图片按实例随机生成,只有带 cookie 的 session 能取到本实例的这张图:

1
2
3
<form name="submitform" action="/missions/prog/10/index.php" method="POST">
<input size="75" name="solution">
<input name="submitbutton" type="submit" value="Submit (remaining time: 45 seconds)">

45 秒意味着不能人工看图片,整个链路 fetch → 下载 → 解码 → 爆破 → 提交必须在一次脚本运行里执行完毕。

Solution

Step 1: 图片静态检查

先按常规隐写方法检查一遍。

1
2
3
4
5
6
7
8
9
10
11
12
$ file image1.png
image1.png: PNG image data, 255 x 128, 8-bit/color RGB, non-interlaced

$ binwalk image1.png
DECIMAL HEXADECIMAL DESCRIPTION
0 0x0 PNG image, total size: 1845 bytes

$ strings -n 4 image1.png
IHDR
IDATx
)Ei$'
LERD

binwalk 没有发现任何附加文件,strings 只有 zlib 压缩残渣。再用结构解析确认 PNG chunk 列表和尾部数据:

1
2
3
4
5
6
7
8
9
10
11
import struct

d = open('image1.png', 'rb').read()
i = 8
while i < len(d):
ln = struct.unpack('>I', d[i:i + 4])[0]
typ = d[i + 4:i + 8].decode('latin1')
print(typ, ln, 'at', i)
i += 12 + ln
print('total', len(d), 'end at', i)
print('trailing:', d[i:][:100])
1
2
3
4
5
IHDR 13 at 8
IDAT 1788 at 33
IEND 0 at 1833
total 1845 end at 1845
trailing: b''

只有 IHDR / IDAT / IEND,没有 tEXt,IEND 之后也没有数据。数据只可能在像素里。

Step 2: 像素统计:找到离群像素

统计一张图里的颜色分布:

1
2
3
4
5
6
7
8
9
10
11
12
from PIL import Image
import collections

im = Image.open('image1.png').convert('RGB')
w, h = im.size
px = im.load()
cols = collections.Counter(px[x, y] for y in range(h) for x in range(w))
print(im.mode, im.size)
for c, n in cols.most_common(6):
print(c, n)
print('unique colors:', len(cols))
print('R values', set(c[0] for c in cols), 'G', set(c[1] for c in cols))
1
2
3
4
5
6
7
8
9
RGB (255, 128)
(0, 0, 56) 244
(0, 0, 57) 244
(0, 0, 58) 244
(0, 0, 59) 244
(0, 0, 60) 244
(0, 0, 62) 244
unique colors: 221
R values {0} G {0, 5, 6, 7, 8, 9, 10}

几个关键观察:

  • 绝大多数像素是 (0, 0, b),也就是只有 B 通道在变化,R 恒为 0。
  • B 通道是一条三角波渐变:每一行从左到右 ±1 地走,到端点反射。这是纯装饰性背景。
  • G 通道基本全是 0,但有 88 个像素非零,而且每行最多一个。这些就是被写入的离群像素。

再看非零 G 像素的行分布,可以看到它落在几乎每一行上、每行只有一个:

1
2
nonzero G count 88
[(90, 0, 9), (84, 1, 6), (77, 2, 6), (53, 3, 10), (78, 4, 10), (106, 5, 6), ...]

(90, 0, 9) 的含义是:第 0 行、第 90 列那个像素的 G 值是 9。G 的值本身不携带信息(只有 5~10 这几个取值,像是随机噪声),真正携带信息的是这个像素的 x 坐标。

顺带排除了 LSB 路线,对 R/G/B 三个通道分别取最低位再按 8 位组字节,得到的只是渐变的固定重复模式,不是明文:

1
2
B 0 printable! UUUUUUUUUUUUUUUUV.................UUUUUUUUUUUUUUUUj................UU...
B 1 printable! UUUUUUUUUV.................UUUUUUUUUUUUUUUUj................UU...

逐行找那个离群像素,把它的 x 当字符码取出来,按行序拼接:

1
2
marker offsets (first 20): [90, 84, 77, 53, 78, 106, 85, 48, 89, 106, 81, 50, 90, 87, 82, 106, 78, 84, 107, 121]
layer 1: ZTM5NjU0YjQ2ZWRjNTkyODRjNDBiYWQ5Njg4N2VjNTQxZTQ3NmY5NmQ2YjdlODFiM2RmZjkyNjAxZWViY2ZlYw==

88 个字符、以 == 结尾、全部落在 base64 字母表内:第一层是 base64 得到了确认。

1
2
layer 2 (base64): b'e39654b46edc59284c40bad96887ec541e476f96d6b7e81b3dff92601eebcfec'
layer 3 (digest): e39654b46edc59284c40bad96887ec541e476f96d6b7e81b3dff92601eebcfec (64 hex chars)

base64 解码后正好是 64 个 hex 字符,即一个 SHA-256。这就对上了题面说的 brute-force the encrypted hashes。

题面已经把搜索空间交代得很清楚:密码 10 位,由恰好两个字符(一个小写、一个大写)随机交替组成。所以不需要字典,穷举即可:

  • 大小写字母对:26 × 26 = 676 种
  • 10 个位置的排列:2¹⁰ = 1024 种
  • 总计 676 × 1024 = 692224 个候选

这个规模在 CPython 里不到一秒就能执行完毕,完全不需要 hashcat 或 GPU:

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
import hashlib
import string
import time

digest = 'e39654b46edc59284c40bad96887ec541e476f96d6b7e81b3dff92601eebcfec'
t0 = time.time()
n = 0
for lo in string.ascii_lowercase:
for up in string.ascii_uppercase:
for mask in range(1 << 10):
pw = ''.join(up if (mask >> i) & 1 else lo for i in range(10))
n += 1
if hashlib.sha256(pw.encode()).hexdigest() == digest:
print('found', pw, 'after', n, 'hashes in', round(time.time() - t0, 2), 's')
raise SystemExit
1
found fffffDDffD after 136801 hashes in 0.18 s

Step 3: 渐变通道随机

把 G 非零 = marker 硬编码了进去,换一个实例就得到:

1
2
3
4
[2/4] rows carrying a marker: 128/128
marker offsets (first 20): [1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]
layer 1: \x01\x00\x00\x00...
[3/4] layer 2 (base64): b''

128 行全都有 marker、坐标都在 0 附近,说明解析结果不可用。重新抓下这张图(image.php?21124)分析:

1
2
3
row0 first 15: [(0, 0, 0), (1, 0, 0), (2, 0, 0), ..., (14, 0, 0)]
R nonzero: 32327 G nonzero: 0
gradient channel 0 {0: 104, 1: 1, 2: 7}

原来这个实例把渐变画在了 R 通道,G 恒为 0。所以 G 非零 这个判据在第一张图成立纯属巧合:那张图渐变在 B 通道,于是 R 和 G 都是非渐变通道,离群像素里恰好有 G 的分量。

正确的判据应该是和渐变通道无关的:先认出哪条通道在当渐变通道(取值种类最多的那条),然后逐行找在另外两条通道上有非零值的那个像素。这样两种实例都成立:

1
2
3
4
5
6
gradient channel 0 {0: 104, 1: 1, 2: 7}   # 渐变在 R
[2/4] rows carrying a marker: 88/128
layer 1: NzQ2ODEwNzJiZmNjOGVmZGYxMGExODQ0NmI2MjFhNDJiMWI3OWNhMzNhZGFmOTA2MGE0MDFjNmViODVmODUzZA==
[3/4] layer 2 (base64): b'74681072bfcc8efdf10a18446b621a42b1b79ca33adaf9060a401c6eb85f853d'
[4/4] cracked sha256 after 369893 hashes in 0.48s
pw ('nnXnnXXXnn', 'sha256')

用 tobytes() 直接按字节做也行,比 im.load() 少一轮 PIL 调用的开销:

1
2
raw = im.tobytes()
at = lambda x, y: raw[3 * (y * w + x):3 * (y * w + x) + 3]

Script

把四层串成一个进程,一次运行完成 fetch → 下载图 → 解码 → 爆破 → 提交。cookie 从环境变量读取,不落盘:

1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
#!/usr/bin/env python3
"""HackThisSite Programming mission 10 -- Automated Steganography (45 s limit).

Pipeline (all in one process, one instance fetch):

1. GET https://www.hackthissite.org/missions/prog/10/ -> form + random image URL
2. GET .../missions/prog/10/image.php?<rand> -> 255x128 RGB PNG
3. The PNG is a 255x128 triangle-wave gradient drawn in ONE randomly
chosen channel (R, G or B). The grader hides one byte per row in the
*x position* of the single pixel that also carries a non-zero value in
one of the other two channels:
char = chr(x_of_the_off_gradient_pixel)
Only 88 of the 128 rows carry a marker; concatenating those chars in row
order yields the base64 string.
4. base64-decode -> 64 hex chars -> a SHA-256 digest.
5. The answer is a 10-character password built from exactly two distinct
characters (one lower case, one upper case) in random order, so it is
found by hashing all 26*26 pairs x 2^10 orderings and comparing to the
digest (~0.7 s in CPython; no external tools needed).
6. POST the password back to .../missions/prog/10/index.php (field `solution`).

Usage:
cd <hts-workspace>
export HTS_COOKIE='HackThisSite=...'
uv run python challenges/hts-prog/10/solve.py # solve + submit
uv run python challenges/hts-prog/10/solve.py --dry-run # decode only

The cookie is read from HTS_COOKIE at runtime and is never written to disk.
"""

import argparse
import base64
import hashlib
import io
import os
import re
import string
import sys
import time

from PIL import Image
import requests

sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__))))
from common import BASE, body_text, fetch_level, session, submit # noqa: E402

LEVEL = 10
IMAGE_RE = re.compile(r'src="(/missions/prog/10/image\.php[^"]*)"')
ALGOS = ("sha256", "sha512", "sha1", "md5", "sha384", "sha224")

def find_image_url(page_text):
m = IMAGE_RE.search(page_text)
if not m:
raise SystemExit("image.php URL not found on level page (not logged in?)")
return m.group(1)

def download_image(s, url):
r = s.get(BASE + url, timeout=20)
r.raise_for_status()
return r.content

def extract_base64(png_bytes):
"""Return (base64_string, per-row marker offsets).

The instance renders a triangle-wave gradient in ONE randomly chosen
channel (R, G or B) and hides one byte per row in the x-offset of the
single pixel that also carries a value in another channel. Detect the
gradient channel as the one with the most distinct values, then take,
for every row, the x of the first pixel whose other channels are non-zero.
"""
im = Image.open(io.BytesIO(png_bytes)).convert("RGB")
w, h = im.size
raw = im.tobytes()
print(f"[1/4] image: {w}x{h} {im.mode}")
at = lambda x, y: raw[3 * (y * w + x):3 * (y * w + x) + 3]

spread = {c: len({at(x, y)[c] for y in range(h) for x in range(w)})
for c in range(3)}
grad = max(spread, key=lambda c: spread[c])
others = [c for c in range(3) if c != grad]
print(f" gradient channel: {'RGB'[grad]} (distinct values {spread})")

xs = []
for y in range(h):
hits = [x for x in range(w)
if any(at(x, y)[c] != 0 for c in others)]
if hits:
xs.append(hits[0])
s = "".join(chr(x) for x in xs)
print(f"[2/4] rows carrying a marker: {len(xs)}/{h}")
print(f" marker offsets (first 20): {xs[:20]}")
print(f" layer 1 (chars from offsets): {s}")
return s, xs

def decode_layer1(b64_str):
pad = b64_str + "=" * (-len(b64_str) % 4)
raw = base64.b64decode(pad)
digest = raw.decode("ascii").strip()
print(f"[3/4] layer 2 (base64): {raw!r}")
if not re.fullmatch(r"[0-9a-f]+", digest):
raise SystemExit("decoded value is not a hex digest")
print(f" layer 3 (digest): {digest} ({len(digest)} hex chars)")
return digest

def candidates():
"""Every 10-char string built from one lower + one upper case letter."""
for lo in string.ascii_lowercase:
for up in string.ascii_uppercase:
for mask in range(1 << 10):
yield "".join(up if (mask >> i) & 1 else lo for i in range(10))

def brute_force(digest):
"""Hash candidates algo-major so the common case (sha256) is ~0.5 s."""
t0 = time.time()
for algo in ALGOS:
tried = 0
for pw in candidates():
tried += 1
if hashlib.new(algo, pw.encode()).hexdigest() == digest:
print(f"[4/4] cracked {algo} after {tried} hashes "
f"in {time.time() - t0:.2f}s")
return pw, algo
raise SystemExit("no password matched (unexpected hash algorithm?)")

def main():
ap = argparse.ArgumentParser()
ap.add_argument("--dry-run", action="store_true",
help="decode only, do not submit")
args = ap.parse_args()

t_start = time.time()
s = session()
page = fetch_level(s, LEVEL)
print("[0/4] level page fetched "
f"({len(page)} bytes, {time.time() - t_start:.2f}s)")

url = find_image_url(page)
print(f" image URL: {url}")
png = download_image(s, url)
print(f" image bytes: {len(png)} ({time.time() - t_start:.2f}s)")

b64_str, _ = extract_base64(png)
digest = decode_layer1(b64_str)
password, algo = brute_force(digest)
print(f" PASSWORD: {password!r} (distinct chars: {sorted(set(password))})")

if args.dry_run:
print(f"dry run: not submitting. total {time.time() - t_start:.2f}s")
return

ok, resp = submit(s, LEVEL, password)
print(f" submitted, verdict={ok} total {time.time() - t_start:.2f}s")
print(body_text(resp)[-600:])
if ok is not True:
sys.exit(1)

if __name__ == "__main__":
main()