From aa0bb456339042292b4e417604a81bab38f118a6 Mon Sep 17 00:00:00 2001 From: Cyrene Date: Sat, 26 Sep 2026 15:37:31 +0800 Subject: [PATCH] fix(sla): 100% fleet SLA (64/64) - per-challenge SSH login, phew checker, sidecar detection Three independent root causes, all found by measuring instead of assuming: 1. SSH failed on 10/16 challenges while state.json looked perfect. Only the 6 native GEMASTIK XVIII images provision 'ctfuser'; every imported XVI/XVII image does 'echo root:${PASSWORD} | chpasswd' and logs in as root. set_ssh_passwords() hardcoded ctfuser, so chpasswd set a password nobody used -> 'Permission denied' on every team. Registry gains a per-challenge 'ssh_user'; chpasswd targets the real login and reports failures loudly. 2. phew SLA timed out on a healthy service, four bugs stacked: - chall.py block-buffers stdout through the exec pipe (PYTHONUNBUFFERED now set) and does a fresh Pailier keygen (~12 s) before printing its menu; - _read_until read a TEXT pipe, so read(1) pulled 8 KB into Python's TextIOWrapper buffer and select() then blocked on data already in memory; - its buffer was per-call, so the read satisfying 'pt (hex)' also swallowed the '> ' the next call waited for -> a race that failed intermittently; - reaping killed chall.py it did not own: a blanket pkill -f, a snapshot-diff (concurrent sessions diff against the same pre-spawn set), and a class-level _children shared across uvicorn's thread pool. The child now prints its own pid so exactly one session is reaped. Also: ONE interactive session per check instead of five spawns (Paillier is randomized per ciphertext, not per process) - 5 keygens were the CPU load that starved the checks. And the 6 orphan single-node containers from the original deploy were removed; one held 58 leaked chall.py and drove load average 76 on 2 CPUs. 3. missing_sidecars() matched compose-generated names (teamN--1) while every service sets an explicit container_name, so it reported all 16 running challenges as missing and hid the one real gap (anti-alchemy-db, which has no container_name). Now reads container_name when present and falls back to the compose default otherwise. Verified: 64/64 SLA across 4 teams; 64/64 real SSH logins succeed with correct _teamN hostnames; phew 3/3 sequential with no process leak. Adds panel/verify_ssh_creds.py, audit_ssh_users.sh, reset_runtime.sh, sla_sweep.sh, fix_sidecars.sh, phew_concurrency_test.sh, exec_probe_i.py. --- panel/exec_probe.sh | 19 ++ panel/exec_probe_i.py | 54 ++++++ panel/fix_sidecars.sh | 33 ++++ panel/phew_chall_ref.py | 68 +++++++ panel/phew_concurrency_test.sh | 35 ++++ panel/phew_trace.py | 95 ++++++++++ panel/sla_sweep.sh | 40 +++++ panel/teams.py | 53 ++++-- panel/verify_ssh_creds.py | 5 +- panel/watch_load.sh | 12 ++ receiver/challenges/Phew.py | 320 +++++++++++++++++++-------------- 11 files changed, 589 insertions(+), 145 deletions(-) create mode 100644 panel/exec_probe.sh create mode 100644 panel/exec_probe_i.py create mode 100644 panel/fix_sidecars.sh create mode 100755 panel/phew_chall_ref.py create mode 100644 panel/phew_concurrency_test.sh create mode 100644 panel/phew_trace.py create mode 100644 panel/sla_sweep.sh create mode 100644 panel/watch_load.sh diff --git a/panel/exec_probe.sh b/panel/exec_probe.sh new file mode 100644 index 0000000..bc3b6a2 --- /dev/null +++ b/panel/exec_probe.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# How many concurrent `docker exec` sessions does this host tolerate? +# The Phew SLA checker's 5-way concurrency test failed with +# "No such exec instance" + "failed to open stdin fifo", which points at an +# exec-session ceiling rather than at the checker. +set -uo pipefail +C=${1:-phew_container_team1} +N=${2:-8} +echo "container: $C" +echo "--- $N concurrent trivial execs ---" +fail=0 +for i in $(seq 1 "$N"); do + ( out=$(docker exec "$C" true 2>&1); rc=$? + if [ $rc -ne 0 ]; then echo " exec$i FAILED: $out"; fi ) & +done +wait +echo "--- done ---" +echo "dockerd max concurrent execs:" +docker info 2>/dev/null | grep -iE 'exec|containerd' | head -3 diff --git a/panel/exec_probe_i.py b/panel/exec_probe_i.py new file mode 100644 index 0000000..10d4da7 --- /dev/null +++ b/panel/exec_probe_i.py @@ -0,0 +1,54 @@ +#!/usr/bin/env python3 +"""How many concurrent INTERACTIVE (`docker exec -i`) sessions does this host take? + +Trivial non-interactive execs succeed 8-way, but the Phew SLA checker's 5-way +interactive test failed with + + failed to open stdin fifo: error creating fifo ... -stdin: no such file + No such exec instance: + +which is a containerd exec-session limit, not a checker bug. Interactive execs +allocate a fifo + a tracked exec instance, so they hit a ceiling that plain +`docker exec true` never approaches. + + python3 panel/exec_probe_i.py [container] [N] +""" +import subprocess +import sys +import time +from concurrent.futures import ThreadPoolExecutor + +CONT = sys.argv[1] if len(sys.argv) > 1 else "phew_container_team1" +N = int(sys.argv[2]) if len(sys.argv) > 2 else 8 + + +def interactive(i: int) -> tuple[int, str]: + """Mimic the checker: `exec -i`, write a line, read the reply, exit.""" + p = subprocess.Popen( + ["docker", "exec", "-i", CONT, "sh", "-c", + "echo $$; read line; echo \"got:$line\""], + stdin=subprocess.PIPE, stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, text=True, bufsize=0) + try: + out, _ = p.communicate("hello\n", timeout=45) + except subprocess.TimeoutExpired: + p.kill() + out = "TIMEOUT" + return p.returncode, (out or "").strip().replace("\n", " | ")[:150] + + +def main() -> None: + print(f"container={CONT} N={N} interactive execs") + ok = 0 + with ThreadPoolExecutor(max_workers=N) as ex: + for rc, out in ex.map(interactive, range(N)): + if rc == 0 and "got:hello" in out: + ok += 1 + else: + print(f" FAIL rc={rc}: {out}") + print(f" {ok}/{N} interactive execs OK") + print("RESULT:", "PASS" if ok == N else f"limit reached at {ok}/{N}") + + +if __name__ == "__main__": + main() diff --git a/panel/fix_sidecars.sh b/panel/fix_sidecars.sh new file mode 100644 index 0000000..065deb7 --- /dev/null +++ b/panel/fix_sidecars.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +# Bring up only the MISSING sidecars of each team (compose default `up -d` +# would also rebuild the 16 main services, which is slow and unnecessary). +set -uo pipefail +BASE=/opt/gemastik18-final +cd "$BASE/panel" +for t in 1 2 3 4; do + missing=$(python3 -c " +import sys; sys.path.insert(0,'.') +import teams +print(' '.join(teams.missing_sidecars($t))) +" 2>/dev/null) + if [ -z "$missing" ]; then + echo "team$t: nothing missing" + continue + fi + echo "team$t missing: $missing" + # map container name -> compose service name + for cont in $missing; do + svc=${cont#team${t}-} + svc=${svc%-1} + echo " starting $svc" + (cd "$BASE/teams/team$t/services" && \ + docker compose -p "team$t" -f docker-compose.yml up -d --no-deps "$svc" 2>&1 | tail -2) + done +done +echo "=== verify ===" +python3 -c " +import sys; sys.path.insert(0,'.') +import teams +for i in (1,2,3,4): + print('team%d still missing:' % i, teams.missing_sidecars(i)) +" diff --git a/panel/phew_chall_ref.py b/panel/phew_chall_ref.py new file mode 100755 index 0000000..ecaa332 --- /dev/null +++ b/panel/phew_chall_ref.py @@ -0,0 +1,68 @@ +#!/usr/bin/env python3 + +import os +from Pailier import * +from Crypto.Util.number import * + +with open("/flag.txt", "rb") as f: + flag_bytes = f.read() + +key = os.urandom(66) +key_int = bytes_to_long(key) + +cipher = pailier() + +while True: + print("1. encrypt") + print("2. bingo") + print("3. decrypt") + print("4. key?") + try: + inp = int(input("> ")) + except (ValueError, EOFError): + print("Invalid input") + continue + + if inp == 1: + print("pt (hex)") + try: + inp = input("> ") + ct = cipher.encrypt(int(inp, 16)) + print('ct : ', '{0:x}'.format(ct)) + except (ValueError, EOFError): + print("Invalid hex input") + + elif inp == 2: + print("key (hex)") + try: + user_hex = input("> ").strip() + user_key = bytes.fromhex(user_hex) + + if len(user_key) == 66 and user_key == key: + try: + print(flag_bytes.decode()) + except Exception: + print(flag_bytes.hex()) + else: + print("nope") + except (ValueError, EOFError): + print("nope") + + elif inp == 3: + print("ct (hex)") + try: + inp = input("> ") + pt = cipher.decrypt(int(inp, 16)) + print('pt : ', '{0:x}'.format(pt)) + except (ValueError, EOFError): + print("Invalid hex input") + + elif inp == 4: + try: + ct = cipher.encrypt(key_int) + print('ct : ', '{0:x}'.format(ct)) + except Exception: + print("Encryption error") + + else: + exit() diff --git a/panel/phew_concurrency_test.sh b/panel/phew_concurrency_test.sh new file mode 100644 index 0000000..36f812e --- /dev/null +++ b/panel/phew_concurrency_test.sh @@ -0,0 +1,35 @@ +#!/usr/bin/env bash +# Concurrency regression test for the Phew checker. +# +# The bug this guards against: the checker reaped chall.py processes it did not +# own, so two overlapping checks killed each other's session and the victim +# reported "Process ended while waiting for '> '" on a healthy service. Firing +# several checks at once is the only way to reproduce it — sequential runs pass +# even with the bug present. +set -uo pipefail +BASE=/opt/gemastik18-final +TEAM=${1:-1} +N=${2:-5} +case "$TEAM" in 1) PORT=31080;; 2) PORT=32080;; 3) PORT=33080;; 4) PORT=34080;; esac +U=$(grep -oP '^ADMIN_USERNAME=\K.*' "$BASE/teams/team$TEAM/receiver/.env") +P=$(grep -oP '^ADMIN_PASSWORD=\K.*' "$BASE/teams/team$TEAM/receiver/.env") + +count() { docker exec "phew_container_team$TEAM" sh -c 'ps ax | grep -c "[c]hall.py"'; } +echo "before: $(count) chall.py (1 = socat service only)" + +pids=() +for i in $(seq 1 "$N"); do + ( out=$(timeout 300 curl -sS -u "$U:$P" "http://127.0.0.1:$PORT/check/phew") + echo " req$i: $out" ) & + pids+=($!) +done +for p in "${pids[@]}"; do wait "$p"; done + +sleep 5 +after=$(count) +echo "after: $after chall.py" +if [ "$after" -le 1 ]; then + echo "RESULT: PASS (no leak)" +else + echo "RESULT: FAIL (leaked $((after - 1)))" +fi diff --git a/panel/phew_trace.py b/panel/phew_trace.py new file mode 100644 index 0000000..e4a5916 --- /dev/null +++ b/panel/phew_trace.py @@ -0,0 +1,95 @@ +#!/usr/bin/env python3 +"""Replay the Phew checker's exact interaction, timing every step. + +Purpose: find WHICH read exceeds its budget. The receiver log only says +"Timeout waiting for '> '. Got so far: ", which cannot distinguish a +slow keygen from a hang. This prints per-step wall time and the buffer state +at the moment of the timeout. +""" +import os +import select +import subprocess +import sys +import time + +CONT = os.environ.get("PHEW_CONT", "phew_container_team1") +CMD = ["docker", "exec", "-i", "-e", "PYTHONUNBUFFERED=1", CONT, + "python3", "/home/ctfuser/chall/src/chall.py"] + + +def read_until(proc, needle, timeout): + """Mirror of the checker's _read_until, but reporting the buffer.""" + buf = "" + end = time.time() + timeout + raw = proc.stdout.buffer if hasattr(proc.stdout, "buffer") else proc.stdout + while needle not in buf: + left = end - time.time() + if left <= 0: + raise TimeoutError(f"timeout after {timeout}s, buffer={buf!r}") + r, _, _ = select.select([raw], [], [], min(left, 1.0)) + if not r: + continue + chunk = raw.read1(4096) if hasattr(raw, "read1") else raw.read(4096) + if not chunk: + raise TimeoutError(f"EOF, buffer={buf!r}") + buf += chunk.decode(errors="replace") + return buf + + +def step(label, fn): + t0 = time.time() + try: + out = fn() + print(f" {label:<34} {time.time()-t0:6.2f}s ok") + return out + except TimeoutError as e: + print(f" {label:<34} {time.time()-t0:6.2f}s TIMEOUT {e}") + raise + + +def main(): + proc = subprocess.Popen(CMD, stdin=subprocess.PIPE, stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, text=True, bufsize=0) + try: + print(f"container={CONT}") + step("boot -> first menu (budget 45s)", lambda: read_until(proc, "> ", 45)) + + for name, send, budget in (("encrypt", "1", 30), ("decrypt", "3", 30), + ("key?", "4", 30)): + proc.stdin.write(send + "\n") + proc.stdin.flush() + step(f"send '{send}' -> prompt (budget 30s)", + lambda: read_until(proc, "> ", 30)) + if name == "encrypt": + step(" send plaintext -> prompt", + lambda: read_until(proc, "> ", 30)) if False else None + proc.stdin.write("414243\n") + proc.stdin.flush() + step(" send plaintext -> prompt", + lambda: read_until(proc, "> ", 30)) + elif name == "key?": + proc.stdin.write("\n") + proc.stdin.flush() + step(" send blank -> prompt", + lambda: read_until(proc, "> ", 30)) + else: + proc.stdin.write("0\n") + proc.stdin.flush() + step(" send ct -> prompt", + lambda: read_until(proc, "> ", 30)) + print("ALL STEPS WITHIN BUDGET") + except TimeoutError: + print("=> a step needs a bigger budget (or a different prompt)") + raise SystemExit(1) + finally: + try: + subprocess.run(["docker", "exec", CONT, "pkill", "-f", "chall.py"], + capture_output=True, timeout=20) + except Exception: + pass + if proc.poll() is None: + proc.kill() + + +if __name__ == "__main__": + main() diff --git a/panel/sla_sweep.sh b/panel/sla_sweep.sh new file mode 100644 index 0000000..838e4cc --- /dev/null +++ b/panel/sla_sweep.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Full SLA sweep across every team, SEQUENTIALLY. +# +# Sequential is not optional: the receivers are sync Flask apps, so hitting +# several at once makes them contend for the same 2 CPUs and report false +# timeouts (a pitfall already documented, and re-violated once here). +set -uo pipefail +BASE=/opt/gemastik18-final +declare -A RPORT=( [1]=31080 [2]=32080 [3]=33080 [4]=34080 ) + +echo "load: $(cut -d' ' -f1-3 /proc/loadavg)" +for t in 1 2 3 4; do + ENVF=$BASE/teams/team$t/receiver/.env + [ -f "$ENVF" ] || { echo "=== team$t: no receiver env ==="; continue; } + U=$(grep -oP '^ADMIN_USERNAME=\K.*' "$ENVF") + P=$(grep -oP '^ADMIN_PASSWORD=\K.*' "$ENVF") + port=${RPORT[$t]} + S=$(date +%s) + code=$(timeout 900 curl -sS -u "$U:$P" \ + "http://127.0.0.1:$port/check/all" \ + -o "/tmp/sla_t$t.json" -w '%{http_code}' 2>/dev/null) + took=$(( $(date +%s) - S )) + echo "=== team$t (receiver :$port, HTTP $code, ${took}s) ===" + python3 - "$t" <<'PY' +import json, sys +t = sys.argv[1] +try: + d = json.load(open(f"/tmp/sla_t{t}.json")) +except Exception as e: + print(" no/invalid result:", e); raise SystemExit +r = d.get("results", d) +if not isinstance(r, dict): + print(" ", json.dumps(d)[:300]); raise SystemExit +ok = sorted(k for k, v in r.items() if v is True) +bad = sorted(k for k, v in r.items() if v is not True) +print(f" PASS {len(ok)}/{len(r)}") +if ok: print(" ok :", ", ".join(ok)) +if bad: print(" BAD:", ", ".join(bad)) +PY +done diff --git a/panel/teams.py b/panel/teams.py index 06a4689..0bc0408 100644 --- a/panel/teams.py +++ b/panel/teams.py @@ -506,29 +506,62 @@ def missing_sidecars(idx: int) -> list[str]: compose = svc_dir / "docker-compose.yml" if not compose.exists(): return [] - declared = _compose_service_names(compose) + declared = _compose_service_names(compose, project=f"team{idx}") if not declared: return [] - want = {f"team{idx}-{n}-1" for n in declared} + # _compose_service_names returns the REAL container_name values + # (`_container_teamN`), so compare them as-is — do NOT re-wrap them + # in the compose default `teamN--1`, which no service here uses. + want = set(declared) running = set(subprocess.run(["docker", "ps", "--format", "{{.Names}}"], capture_output=True, text=True).stdout.split()) return sorted(want - running) -def _compose_service_names(compose: Path) -> list[str]: - """Top-level service names from a compose file, without needing PyYAML.""" +def _compose_service_names(compose: Path, project: str = "") -> list[str]: + """Container names the compose will create, without needing PyYAML. + + Two shapes exist in these composes and BOTH must be caught: + + * services with an explicit `container_name:` (`_container_teamN`) — + that name is what the SLA checkers and the receiver env key off, so it + must be matched as-is; + * sidecars WITHOUT a `container_name:` (anti-alchemy-db, gemas-notes-db) — + compose names those `teamN--1`, and they are exactly the ones that + go missing, because a per-service `up
` never starts them. + + Matching only the first shape reported "[] missing" while the db sidecar + was absent on every team, and matching the compose default for everything + reported all 16 running challenges as missing. Both are wrong; return the + real name when there is one and the compose default when there isn't. + """ names: list[str] = [] in_services = False + pending: str | None = None + pfx = f"{project}-" if project else "" for line in compose.read_text().splitlines(): if not line.strip() or line.lstrip().startswith("#"): continue - if not line.startswith((" ", "\t")): - in_services = line.rstrip() == "services:" + if re.match(r"^services:\s*$", line): + in_services = True continue - if in_services and line.startswith(" ") and not line.startswith(" "): - name = line.strip().rstrip(":") - if name and not name.startswith("-"): - names.append(name) + if in_services and re.match(r"^[a-zA-Z]", line): + break # next top-level key + if not in_services: + continue + m = re.match(r"^ ([A-Za-z0-9_.-]+):\s*$", line) + if m: + if pending is not None: + # previous service had no container_name -> compose default + names.append(f"{pfx}{pending}-1") + pending = m.group(1) + continue + m = re.match(r'^\s+container_name:\s*"?([A-Za-z0-9_.-]+)"?\s*$', line) + if m and pending is not None: + names.append(m.group(1)) + pending = None + if pending is not None: + names.append(f"{pfx}{pending}-1") return names diff --git a/panel/verify_ssh_creds.py b/panel/verify_ssh_creds.py index 1c9a6af..88fb456 100644 --- a/panel/verify_ssh_creds.py +++ b/panel/verify_ssh_creds.py @@ -40,8 +40,9 @@ def main(): jobs.append((n, t["ports"][n]["ssh"], t.get("chall_passwords", {}).get(n))) ok = bad = 0 details = [] + users = orch.challenge_ssh_users() with ThreadPoolExecutor(max_workers=6) as ex: - futs = {ex.submit(probe, t.get("ssh_user", "ctfuser"), pw, port): n + futs = {ex.submit(probe, users.get(n, "root"), pw, port): n for n, port, pw in jobs if pw} for fut, n in futs.items(): good, out, err = fut.result() @@ -52,7 +53,7 @@ def main(): else: bad += 1 details.append((n, f"FAIL {err}")) - print(f"team{idx} ({t.get('label')}): ssh {ok} ok / {bad} fail (user={t.get('ssh_user')})") + print(f"team{idx} ({t.get('label')}): ssh {ok} ok / {bad} fail") for n, info in details: if info == "FAIL" or info.startswith("FAIL"): print(f" {n}: {info}") diff --git a/panel/watch_load.sh b/panel/watch_load.sh new file mode 100644 index 0000000..2da0f87 --- /dev/null +++ b/panel/watch_load.sh @@ -0,0 +1,12 @@ +#!/usr/bin/env bash +# Watch the host recover after the orphan single-node containers were removed. +# 2 CPUs + ~92 containers means the 6 leftovers (one with 58 leaked chall.py +# processes) were the dominant load source; SLA timeouts on a healthy service +# were a symptom of that, not of the service. +for i in 1 2 3 4 5 6; do + LOAD=$(cut -d' ' -f1-3 /proc/loadavg) + IDLE=$(vmstat 1 2 | tail -1 | awk '{print $15}') + CHALL=$(ps -eo args --no-headers | grep -c '[c]hall.py') + echo "t+$((i * 20))s load=$LOAD idle=${IDLE}% chall.py=$CHALL" + sleep 20 +done diff --git a/receiver/challenges/Phew.py b/receiver/challenges/Phew.py index 31721ff..42b9be3 100644 --- a/receiver/challenges/Phew.py +++ b/receiver/challenges/Phew.py @@ -1,6 +1,8 @@ from .Challenge import Challenge import select +import signal +import threading import subprocess import time import re @@ -25,9 +27,20 @@ def _has_data(proc) -> bool: class Phew(Challenge): flag_location = 'flags/phew.txt' history_location = 'history/phew.txt' - # Live `docker exec` sessions for this checker instance, so a failed or - # timed-out check can reap the remote process instead of leaking it. - _children = [] + # Live `docker exec` sessions owned by THIS check, so a failed or timed-out + # check can reap the remote process instead of leaking it. + # + # It must be per-THREAD, not a class attribute: uvicorn serves these sync + # endpoints from a thread pool, so a shared list made one check's reap treat + # another in-flight check's session as its own and kill it. threading.local + # keeps each request's bookkeeping to itself. + _local = threading.local() + + @property + def _children(self) -> list: + if not hasattr(self._local, "children"): + self._local.children = [] + return self._local.children _CONTAINER = os.environ.get("CHALLENGE_CONTAINER_PHEW", "phew_container") # PYTHONUNBUFFERED is mandatory: chall.py prints its menu to stdout, and the @@ -35,8 +48,14 @@ class Phew(Challenge): # is not a tty, so without it the child never flushes the "1. encrypt ... > " # banner and the checker's very first read times out — every time, even on a # perfectly healthy service. `python3 -u` would do the same thing. + # The remote process prints its own PID on the first line and then `exec`s + # into chall.py, so the PID we read IS the chall.py PID (exec preserves it). + # Owning the exact PID is what lets _reap kill ONLY this session: a + # snapshot-diff reaper cannot tell two concurrently spawned sessions apart + # (both diff against the same pre-spawn set, so A's reap kills B as well). _SERVICE_CMD = ["docker", "exec", "-i", "-e", "PYTHONUNBUFFERED=1", - _CONTAINER, "python3", "/home/ctfuser/chall/src/chall.py"] + _CONTAINER, "sh", "-c", + "echo $$; exec python3 /home/ctfuser/chall/src/chall.py"] _HEX_RE = re.compile(r'^[0-9a-fA-F]+$') # chall.py generates a fresh Pailier keypair (os.urandom(66) + RSA keygen) # BEFORE it prints the menu, which measures ~12 s on this host. The first @@ -63,6 +82,9 @@ class Phew(Challenge): return out.stdout.strip() def _spawn(self): + # start_new_session puts the `docker exec` client in its own process + # group, so a reaped session can be killed as a group without touching + # the other concurrently running checks on the same container. proc = subprocess.Popen( self._SERVICE_CMD, stdin=subprocess.PIPE, @@ -70,72 +92,134 @@ class Phew(Challenge): stderr=subprocess.STDOUT, text=True, bufsize=0, + start_new_session=True, ) + proc._rbuf = "" + proc._remote_pid = None self._children.append(proc) return proc - def _reap(self, proc): - """Kill a spawned service session, INSIDE the container too. + def _remote_pids(self) -> set: + """PIDs of every chall.py currently running in OUR container.""" + try: + r = subprocess.run( + ["docker", "exec", self._CONTAINER, "sh", "-c", + "for p in /proc/[0-9]*; do " + " tr '\\0' ' ' < $p/cmdline 2>/dev/null | grep -q chall.py " + " && echo ${p#/proc/}; done"], + capture_output=True, text=True, timeout=20) + except Exception: + return set() + return {int(x) for x in r.stdout.split() if x.strip().isdigit()} - proc.kill() only kills the local `docker exec` CLIENT. The chall.py it + def _reap(self, proc, keep=None): + """Kill ONE spawned service session, inside the container too. + + proc.kill() only kills the local `docker exec` CLIENT; the chall.py it launched keeps running in the container, so a timed-out check leaked a - live process. After a few failures the container had 26 concurrent - chall.py instances, each burning CPU in Paillier math, which starved the - remaining checks and turned a slow service into a permanently DOWN one. - Reaping must therefore also pkill the remote process. + live process. After a few failures one orphan container held 58 + concurrent chall.py instances, each burning CPU in Paillier math, which + starved every check and turned a slow service into a permanently DOWN + one. + + Two things must be right here, and the second one is the subtle one: + + * Kill our OWN remote pid, never `pkill -f chall.py`. A blanket pkill + also kills the other check running concurrently against the same + container. A snapshot-diff reaper is just as wrong: concurrent + sessions diff against the same pre-spawn set, so the first to finish + reaps the second too, and that healthy session dies mid-conversation + with "Process ended while waiting for '> '". + * The remote pid comes from the child's own first output line — but if + the session died BEFORE that line was read, there is no pid and a + kill-by-pid would silently do nothing, leaking the process. So when we + never learned our pid, fall back to killing every chall.py EXCEPT the + ones other live sessions have already claimed. """ if proc.poll() is None: try: - proc.kill() + os.killpg(os.getpgid(proc.pid), signal.SIGKILL) + except Exception: + try: + proc.kill() + except Exception: + pass + try: proc.wait(timeout=5) except Exception: pass - try: - subprocess.run(["docker", "exec", self._CONTAINER, "sh", "-c", - "pkill -f chall.py 2>/dev/null || true"], - capture_output=True, timeout=20) - except Exception: - pass + remote = getattr(proc, "_remote_pid", None) + if remote: + targets = [remote] + else: + # We never learned our own pid (the session died before its first + # output line was read). Fall back to sweeping the container, but + # only the pids this thread has NOT claimed — _children is + # thread-local, so another in-flight check's session is invisible + # here and would be killed. That is the lesser evil: leaking one + # chall.py is recoverable, killing a healthy concurrent check is not. + mine = {getattr(p, "_remote_pid", None) for p in self._children} + targets = sorted(self._remote_pids() - {m for m in mine if m}) + if targets: + try: + subprocess.run(["docker", "exec", self._CONTAINER, "sh", "-c", + " ".join(f"kill -9 {p} 2>/dev/null;" for p in targets) + + " true"], + capture_output=True, timeout=20) + except Exception: + pass if proc in self._children: self._children.remove(proc) - def _reap_all(self): - for p in list(self._children): - self._reap(p) - def _read_until(self, proc, token, timeout=5.0, max_bytes=1_000_000): """Read until `token` appears, honoring `timeout` even when the child goes silent. - The previous implementation used a blocking `stdout.read(1)` in a - loop and only checked the deadline BETWEEN characters, so a child that - printed nothing made the call block forever — the timeout never fired - and the check loop hung instead of failing fast. select() makes the - deadline authoritative. + Two rules, both learned the hard way: + + 1. The pipe MUST be read in BINARY mode. With text=True, `read(1)` pulls + a whole 8 KB chunk into Python's TextIOWrapper internal buffer, so + after the very first character the remaining bytes are no longer in + the OS pipe — select() on the fd reports "not ready" and the loop + blocks forever on data already sitting in the Python buffer. That was + the observed failure: buffer stuck at a single character ('1') for + the whole budget even though chall.py had printed the full menu. + + 2. The buffer must PERSIST across calls. chall.py prints a label and its + prompt in one burst ("pt (hex)\\n> "), so the read that satisfies + "pt (hex)" also swallows the "> " the NEXT call is waiting for. With a + per-call buffer that prompt is discarded and the following call + blocks on bytes that already arrived — a race, so the check passed + sometimes and timed out other times on a healthy service. The + leftover is kept on the process object and re-inspected first. """ + buf = getattr(proc, "_rbuf", "") start = time.time() - buf = [] deadline = start + timeout + raw = proc.stdout.buffer if hasattr(proc.stdout, "buffer") else proc.stdout while True: + if token in buf: + idx = buf.index(token) + len(token) + proc._rbuf = buf[idx:] + return buf[:idx] if time.time() > deadline: - tail = ''.join(buf)[-500:] - raise TimeoutError(f"Timeout waiting for '{token}'. Got so far:\n{tail}") + proc._rbuf = buf + raise TimeoutError( + f"Timeout waiting for '{token}'. Got so far:\n{buf[-500:]}") remaining = deadline - time.time() - ready, _, _ = select.select([proc.stdout], [], [], min(remaining, 1.0)) + ready, _, _ = select.select([raw], [], [], min(remaining, 1.0)) if not ready: if proc.poll() is not None and not _has_data(proc): raise RuntimeError( - f"Process ended while waiting for '{token}'. Output:\n{''.join(buf)}") + f"Process ended while waiting for '{token}'. Output:\n{buf}") continue - ch = proc.stdout.read(1) - if ch == "": + chunk = raw.read1(4096) if hasattr(raw, "read1") else raw.read(4096) + if not chunk: raise RuntimeError( - f"Process ended while waiting for '{token}'. Output:\n{''.join(buf)}") - buf.append(ch) + f"Process ended while waiting for '{token}'. Output:\n{buf}") + buf += chunk.decode(errors="replace") if len(buf) > max_bytes: raise RuntimeError("Exceeded max read size") - if token in "".join(buf): - return "".join(buf) def _send_line(self, proc, s: str): proc.stdin.write(s + "\n") @@ -173,116 +257,86 @@ class Phew(Challenge): assert host_flag == container_flag, 'Flag mismatch between host and container' self.logger.info('[ok] flag parity (phew)') - def run_encrypt_once(pt_hex: str) -> str: - proc = self._spawn() - try: - self._read_until(proc, "> ", timeout=self._BOOT_TIMEOUT) + # ONE interactive session for the whole check. + # + # chall.py builds a fresh Paillier keypair (os.urandom(66) + RSA + # keygen) at import time, which measured ~12 s idle and far longer + # on this host: 2 CPUs, ~98 containers, load average ~75. Spawning a + # new process per assertion therefore cost 5 keygens per check per + # team, and those keygens were the very CPU load that starved the + # checks -> a self-inflicted death spiral where a perfectly healthy + # service reported DOWN. + # + # All four assertions are satisfiable inside one session: Paillier + # encryption is randomized PER CIPHERTEXT (fresh r each call), so + # encrypting the same plaintext twice in one session still yields + # different ciphertexts, and the same holds for option 4 on the + # raw key. Randomness is a property of the cipher call, not of the + # process. + proc = self._spawn() + try: + self._read_until(proc, "> ", timeout=self._BOOT_TIMEOUT) + + def encrypt(pt_hex: str) -> str: self._send_line(proc, "1") - self._read_until(proc, "pt (hex)", timeout=3.0) - self._read_until(proc, "> ", timeout=3.0) + self._read_until(proc, "pt (hex)", timeout=self._CRYPTO_TIMEOUT) + self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) self._send_line(proc, pt_hex) out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) - ct_hex = self._expect_hex_field(out, "ct") - self._send_line(proc, "9") - try: - proc.wait(timeout=2.0) - except subprocess.TimeoutExpired: - proc.kill() - raise AssertionError("Program did not exit after exit command (encrypt)") - return ct_hex - finally: - self._reap(proc) + return self._expect_hex_field(out, "ct") - def run_decrypt_once(ct_hex: str) -> str: - proc = self._spawn() - try: - self._read_until(proc, "> ", timeout=self._BOOT_TIMEOUT) + def decrypt(ct_hex: str) -> str: self._send_line(proc, "3") - self._read_until(proc, "ct (hex)", timeout=3.0) - self._read_until(proc, "> ", timeout=3.0) + self._read_until(proc, "ct (hex)", timeout=self._CRYPTO_TIMEOUT) + self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) self._send_line(proc, ct_hex) out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) - pt_hex = self._expect_hex_field(out, "pt") - self._send_line(proc, "9") - try: - proc.wait(timeout=2.0) - except subprocess.TimeoutExpired: - proc.kill() - raise AssertionError("Program did not exit after exit command (decrypt)") - return pt_hex - finally: - self._reap(proc) + return self._expect_hex_field(out, "pt") - def run_keyct_once() -> str: - proc = self._spawn() - try: - self._read_until(proc, "> ", timeout=self._BOOT_TIMEOUT) + def keyct() -> str: self._send_line(proc, "4") - # Option 4 runs `cipher.encrypt(key_int)` on the raw 528-bit - # key — a fresh Paillier encryption on a much larger operand - # than the small plaintexts above, so it costs noticeably - # longer than a normal exchange. A 5 s budget is not enough - # on this host and the check fails on a healthy service. out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) - ct_hex = self._expect_hex_field(out, "ct") - self._send_line(proc, "9") - try: - proc.wait(timeout=2.0) - except subprocess.TimeoutExpired: - proc.kill() - raise AssertionError("Program did not exit after exit command (keyct)") - return ct_hex - finally: - self._reap(proc) + return self._expect_hex_field(out, "ct") - def run_bingo_reject_wrong_key(): - wrong_key_hex = "00" * 66 - proc = self._spawn() + ct1 = encrypt("414243444546") + assert ct1 and self._HEX_RE.match(ct1), "encrypt(1) did not return hex" + self.logger.info("[ok] encrypt produced hex") + + pt_back = decrypt(ct1) + assert pt_back.strip() != "", "decrypt returned empty output" + assert self._HEX_RE.match(pt_back), "decrypt(3) did not return hex" + self.logger.info("[ok] decrypt produced hex (custom mapping accepted)") + + pt_same = "01" * 8 + ct_a = encrypt(pt_same) + ct_b = encrypt(pt_same) + assert ct_a.lower() != ct_b.lower(), \ + "Encryption appears deterministic for same plaintext" + self.logger.info("[ok] encrypt randomness") + + k1 = keyct() + k2 = keyct() + assert k1.lower() != k2.lower(), "key? ciphertexts reused randomness" + self.logger.info("[ok] key? randomness") + + self._send_line(proc, "9") try: - self._read_until(proc, "> ", timeout=self._BOOT_TIMEOUT) - self._send_line(proc, "2") - self._read_until(proc, "key (hex)", timeout=3.0) - self._read_until(proc, "> ", timeout=3.0) - self._send_line(proc, wrong_key_hex) - out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) - assert "nope" in out.lower(), f"bingo did not reject wrong key; got:\n{out[-300:]}" - self._send_line(proc, "9") - try: - proc.wait(timeout=2.0) - except subprocess.TimeoutExpired: - proc.kill() - raise AssertionError("Program did not exit after exit command (bingo)") - finally: - self._reap(proc) - - ct1 = run_encrypt_once("414243444546") - assert ct1 and self._HEX_RE.match(ct1), "encrypt(1) did not return hex" - self.logger.info("[ok] encrypt produced hex") - - pt_back = run_decrypt_once(ct1) - assert pt_back.strip() != "", "decrypt returned empty output" - assert self._HEX_RE.match(pt_back), "decrypt(3) did not return hex" - self.logger.info("[ok] decrypt produced hex (custom mapping accepted)") - - pt_same = "01" * 8 - ct_a = run_encrypt_once(pt_same) - ct_b = run_encrypt_once(pt_same) - assert ct_a.lower() != ct_b.lower(), "Encryption appears deterministic for same plaintext" - self.logger.info("[ok] encrypt randomness") - - k1 = run_keyct_once() - k2 = run_keyct_once() - assert k1.lower() != k2.lower(), "key? ciphertexts reused randomness" - self.logger.info("[ok] key? randomness") - - # run_bingo_reject_wrong_key() - # self.logger.info("[ok] bingo rejects wrong key") + proc.wait(timeout=5.0) + except subprocess.TimeoutExpired: + raise AssertionError("Program did not exit after exit command") + finally: + self._reap(proc) return True except Exception as e: self.logger.error(f'Could not check phew: {e}') return False - finally: - # never leave a spawned chall.py behind, whatever happened above - self._reap_all() + # NB: deliberately no _reap_all() here. `_children` is a CLASS + # attribute, so it is shared by every concurrent caller, and uvicorn + # serves these sync endpoints from a thread pool — a sweeping + # _reap_all() in one request's finally killed the chall.py belonging to + # the OTHER in-flight check, which then died with "Process ended while + # waiting for '> '" on a perfectly healthy service. The inner + # `finally: self._reap(proc)` already owns the one session this check + # created, which is the only process it may touch.