from .Challenge import Challenge import select import signal import threading import subprocess import time import re import os def _has_data(proc) -> bool: """True if the child's pipe still holds buffered output.""" import fcntl try: fd = proc.stdout.fileno() fl = fcntl.fcntl(fd, fcntl.F_GETFL) fcntl.fcntl(fd, fcntl.F_SETFL, fl | os.O_NONBLOCK) data = proc.stdout.read() if data: return True return False except Exception: return False class Phew(Challenge): flag_location = 'flags/phew.txt' history_location = 'history/phew.txt' # Live `docker exec` sessions owned by THIS check, so a failed or timed-out # check can reap the remote process instead of leaking it. # # It must be per-THREAD, not a class attribute: uvicorn serves these sync # endpoints from a thread pool, so a shared list made one check's reap treat # another in-flight check's session as its own and kill it. threading.local # keeps each request's bookkeeping to itself. _local = threading.local() @property def _children(self) -> list: if not hasattr(self._local, "children"): self._local.children = [] return self._local.children _CONTAINER = os.environ.get("CHALLENGE_CONTAINER_PHEW", "phew_container") # PYTHONUNBUFFERED is mandatory: chall.py prints its menu to stdout, and the # checker reads that pipe interactively. Python block-buffers stdout when it # is not a tty, so without it the child never flushes the "1. encrypt ... > " # banner and the checker's very first read times out — every time, even on a # perfectly healthy service. `python3 -u` would do the same thing. # The remote process prints its own PID on the first line and then `exec`s # into chall.py, so the PID we read IS the chall.py PID (exec preserves it). # Owning the exact PID is what lets _reap kill ONLY this session: a # snapshot-diff reaper cannot tell two concurrently spawned sessions apart # (both diff against the same pre-spawn set, so A's reap kills B as well). _SERVICE_CMD = ["docker", "exec", "-i", "-e", "PYTHONUNBUFFERED=1", _CONTAINER, "sh", "-c", "echo $$; exec python3 /home/ctfuser/chall/src/chall.py"] _HEX_RE = re.compile(r'^[0-9a-fA-F]+$') # chall.py generates a fresh Pailier keypair (os.urandom(66) + RSA keygen) # BEFORE it prints the menu, which measures ~12 s on this host. The first # read must outlast that or the check fails on a healthy service. Later # exchanges reuse the same key, so they can stay short. _BOOT_TIMEOUT = 45.0 # Budget for a single cipher operation. Encrypting the raw 528-bit key # (menu option 4) is measurably slower than encrypting a small plaintext, # and a saturated host makes even the small ones slower — 5 s was too tight # and produced a false DOWN. _CRYPTO_TIMEOUT = 30.0 def _read_container_flag(self) -> str: # NB: a timeout is mandatory here. `docker exec` against a container # whose process table is saturated (accumulated chall.py zombies) can # block forever and take the whole SLA check loop down with it. try: out = subprocess.run(["docker", "exec", self._CONTAINER, "cat", "/flag.txt"], capture_output=True, text=True, timeout=30) except subprocess.TimeoutExpired: raise TimeoutError("docker exec cat /flag.txt timed out (container overloaded?)") if out.returncode != 0 or not out.stdout.strip(): raise FileNotFoundError("Flag not found in container (/flag.txt)") return out.stdout.strip() def _spawn(self): # start_new_session puts the `docker exec` client in its own process # group, so a reaped session can be killed as a group without touching # the other concurrently running checks on the same container. proc = subprocess.Popen( self._SERVICE_CMD, stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, text=True, bufsize=0, start_new_session=True, ) proc._rbuf = "" proc._remote_pid = None self._children.append(proc) return proc def _remote_pids(self) -> set: """PIDs of every chall.py currently running in OUR container.""" try: r = subprocess.run( ["docker", "exec", self._CONTAINER, "sh", "-c", "for p in /proc/[0-9]*; do " " tr '\\0' ' ' < $p/cmdline 2>/dev/null | grep -q chall.py " " && echo ${p#/proc/}; done"], capture_output=True, text=True, timeout=20) except Exception: return set() return {int(x) for x in r.stdout.split() if x.strip().isdigit()} def _reap(self, proc, keep=None): """Kill ONE spawned service session, inside the container too. proc.kill() only kills the local `docker exec` CLIENT; the chall.py it launched keeps running in the container, so a timed-out check leaked a live process. After a few failures one orphan container held 58 concurrent chall.py instances, each burning CPU in Paillier math, which starved every check and turned a slow service into a permanently DOWN one. Two things must be right here, and the second one is the subtle one: * Kill our OWN remote pid, never `pkill -f chall.py`. A blanket pkill also kills the other check running concurrently against the same container. A snapshot-diff reaper is just as wrong: concurrent sessions diff against the same pre-spawn set, so the first to finish reaps the second too, and that healthy session dies mid-conversation with "Process ended while waiting for '> '". * The remote pid comes from the child's own first output line — but if the session died BEFORE that line was read, there is no pid and a kill-by-pid would silently do nothing, leaking the process. So when we never learned our pid, fall back to killing every chall.py EXCEPT the ones other live sessions have already claimed. """ if proc.poll() is None: try: os.killpg(os.getpgid(proc.pid), signal.SIGKILL) except Exception: try: proc.kill() except Exception: pass try: proc.wait(timeout=5) except Exception: pass remote = getattr(proc, "_remote_pid", None) if remote: targets = [remote] else: # We never learned our own pid (the session died before its first # output line was read). Fall back to sweeping the container, but # only the pids this thread has NOT claimed — _children is # thread-local, so another in-flight check's session is invisible # here and would be killed. That is the lesser evil: leaking one # chall.py is recoverable, killing a healthy concurrent check is not. mine = {getattr(p, "_remote_pid", None) for p in self._children} targets = sorted(self._remote_pids() - {m for m in mine if m}) if targets: try: subprocess.run(["docker", "exec", self._CONTAINER, "sh", "-c", " ".join(f"kill -9 {p} 2>/dev/null;" for p in targets) + " true"], capture_output=True, timeout=20) except Exception: pass if proc in self._children: self._children.remove(proc) def _read_until(self, proc, token, timeout=5.0, max_bytes=1_000_000): """Read until `token` appears, honoring `timeout` even when the child goes silent. Two rules, both learned the hard way: 1. The pipe MUST be read in BINARY mode. With text=True, `read(1)` pulls a whole 8 KB chunk into Python's TextIOWrapper internal buffer, so after the very first character the remaining bytes are no longer in the OS pipe — select() on the fd reports "not ready" and the loop blocks forever on data already sitting in the Python buffer. That was the observed failure: buffer stuck at a single character ('1') for the whole budget even though chall.py had printed the full menu. 2. The buffer must PERSIST across calls. chall.py prints a label and its prompt in one burst ("pt (hex)\\n> "), so the read that satisfies "pt (hex)" also swallows the "> " the NEXT call is waiting for. With a per-call buffer that prompt is discarded and the following call blocks on bytes that already arrived — a race, so the check passed sometimes and timed out other times on a healthy service. The leftover is kept on the process object and re-inspected first. """ buf = getattr(proc, "_rbuf", "") start = time.time() deadline = start + timeout raw = proc.stdout.buffer if hasattr(proc.stdout, "buffer") else proc.stdout while True: if token in buf: idx = buf.index(token) + len(token) proc._rbuf = buf[idx:] return buf[:idx] if time.time() > deadline: proc._rbuf = buf raise TimeoutError( f"Timeout waiting for '{token}'. Got so far:\n{buf[-500:]}") remaining = deadline - time.time() ready, _, _ = select.select([raw], [], [], min(remaining, 1.0)) if not ready: if proc.poll() is not None and not _has_data(proc): raise RuntimeError( f"Process ended while waiting for '{token}'. Output:\n{buf}") continue chunk = raw.read1(4096) if hasattr(raw, "read1") else raw.read(4096) if not chunk: raise RuntimeError( f"Process ended while waiting for '{token}'. Output:\n{buf}") buf += chunk.decode(errors="replace") if len(buf) > max_bytes: raise RuntimeError("Exceeded max read size") def _send_line(self, proc, s: str): proc.stdin.write(s + "\n") proc.stdin.flush() def _expect_hex_field(self, text: str, label: str) -> str: m = re.search(rf"{re.escape(label)}\s*:\s*([0-9a-fA-F]+)", text) assert m, f"Missing '{label}' in output. Tail:\n{text[-400:]}" hx = m.group(1) assert self._HEX_RE.match(hx), f"{label} is not hex" return hx def distribute(self, flag): try: os.makedirs(os.path.dirname(self.flag_location), exist_ok=True) with open(self.flag_location, 'w') as f: f.write(flag) os.makedirs(os.path.dirname(self.history_location), exist_ok=True) with open(self.history_location, 'a') as f: f.write(flag + '\n') self.logger.info(f'Flag {flag} written to {self.flag_location}') return True except Exception as e: self.logger.error(f'Could not write flag to {self.flag_location}: {e}') return False def check(self): try: # parity check with open(self.flag_location, 'r') as f: host_flag = f.read().strip() container_flag = self._read_container_flag() assert host_flag == container_flag, 'Flag mismatch between host and container' self.logger.info('[ok] flag parity (phew)') # ONE interactive session for the whole check. # # chall.py builds a fresh Paillier keypair (os.urandom(66) + RSA # keygen) at import time, which measured ~12 s idle and far longer # on this host: 2 CPUs, ~98 containers, load average ~75. Spawning a # new process per assertion therefore cost 5 keygens per check per # team, and those keygens were the very CPU load that starved the # checks -> a self-inflicted death spiral where a perfectly healthy # service reported DOWN. # # All four assertions are satisfiable inside one session: Paillier # encryption is randomized PER CIPHERTEXT (fresh r each call), so # encrypting the same plaintext twice in one session still yields # different ciphertexts, and the same holds for option 4 on the # raw key. Randomness is a property of the cipher call, not of the # process. proc = self._spawn() try: self._read_until(proc, "> ", timeout=self._BOOT_TIMEOUT) def encrypt(pt_hex: str) -> str: self._send_line(proc, "1") self._read_until(proc, "pt (hex)", timeout=self._CRYPTO_TIMEOUT) self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) self._send_line(proc, pt_hex) out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) return self._expect_hex_field(out, "ct") def decrypt(ct_hex: str) -> str: self._send_line(proc, "3") self._read_until(proc, "ct (hex)", timeout=self._CRYPTO_TIMEOUT) self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) self._send_line(proc, ct_hex) out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) return self._expect_hex_field(out, "pt") def keyct() -> str: self._send_line(proc, "4") out = self._read_until(proc, "> ", timeout=self._CRYPTO_TIMEOUT) return self._expect_hex_field(out, "ct") ct1 = encrypt("414243444546") assert ct1 and self._HEX_RE.match(ct1), "encrypt(1) did not return hex" self.logger.info("[ok] encrypt produced hex") pt_back = decrypt(ct1) assert pt_back.strip() != "", "decrypt returned empty output" assert self._HEX_RE.match(pt_back), "decrypt(3) did not return hex" self.logger.info("[ok] decrypt produced hex (custom mapping accepted)") pt_same = "01" * 8 ct_a = encrypt(pt_same) ct_b = encrypt(pt_same) assert ct_a.lower() != ct_b.lower(), \ "Encryption appears deterministic for same plaintext" self.logger.info("[ok] encrypt randomness") k1 = keyct() k2 = keyct() assert k1.lower() != k2.lower(), "key? ciphertexts reused randomness" self.logger.info("[ok] key? randomness") self._send_line(proc, "9") try: proc.wait(timeout=5.0) except subprocess.TimeoutExpired: raise AssertionError("Program did not exit after exit command") finally: self._reap(proc) return True except Exception as e: self.logger.error(f'Could not check phew: {e}') return False # NB: deliberately no _reap_all() here. `_children` is a CLASS # attribute, so it is shared by every concurrent caller, and uvicorn # serves these sync endpoints from a thread pool — a sweeping # _reap_all() in one request's finally killed the chall.py belonging to # the OTHER in-flight check, which then died with "Process ended while # waiting for '> '" on a perfectly healthy service. The inner # `finally: self._reap(proc)` already owns the one session this check # created, which is the only process it may touch.