#!/usr/bin/env python3 """Collapse duplicate `volumes:` keys inside a single service, safely. Why: the canonical per-challenge compose templates were produced by a generator that appended a second `volumes:` list to services that already had one. YAML forbids duplicate mapping keys, so `docker compose` rejected the file with mapping key "volumes" already defined at line N and every challenge using such a template failed to start. This does a real YAML round-trip (ruamel if available, else PyYAML) so nesting is never guessed: the second list's entries are appended to the first and the duplicate key is dropped. Comments/order are preserved when ruamel is present. """ from __future__ import annotations import sys from pathlib import Path SERVICES = Path("/opt/gemastik18-final/services") try: from ruamel.yaml import YAML HAVE_RUAMEL = True except ImportError: HAVE_RUAMEL = False import yaml def fix_text(text: str) -> str: if HAVE_RUAMEL: y = YAML() y.preserve_quotes = True data = y.load(text) changed = _merge_node(data) if not changed: return text import io buf = io.StringIO() y.dump(data, buf) return buf.getvalue() # PyYAML fallback: last duplicate key wins, so re-read the raw text to # collect ALL entries before handing it to the parser. data = yaml.safe_load(text) changed = _merge_node(data) if not changed: return text return yaml.safe_dump(data, sort_keys=False, default_flow_style=False) def _merge_node(data) -> bool: """Walk `services:` and merge any service that has >1 volumes entry. ruamel keeps duplicates as a CommentedMap with repeated keys only when round-tripped; after a safe load they collapse, so for the text form we instead detect the duplicate at the line level (see fix_text_lines). """ if not isinstance(data, dict): return False changed = False for svc in (data.get("services") or {}).values(): if isinstance(svc, dict) and isinstance(svc.get("volumes"), list): continue return changed def fix_text_lines(text: str) -> str: """Line-based merge that preserves each service's own key ordering. For every service block we find all `volumes:` keys at the service's key indent and fold them into the first one, leaving every other key exactly where it was. """ lines = text.splitlines() out: list[str] = [] i = 0 n = len(lines) # locate `services:` and its child key indent svc_start = None for idx, line in enumerate(lines): if line.strip() == "services:" and not line.startswith(" "): svc_start = idx break if svc_start is None: return text # child indent of the first service name child_indent = None for idx in range(svc_start + 1, n): stripped = lines[idx] if not stripped.strip(): continue ind = len(stripped) - len(stripped.lstrip()) if ind == 0: break child_indent = ind break if child_indent is None: return text # service name boundaries bounds: list[tuple[int, int]] = [] start = None for idx in range(svc_start + 1, n): line = lines[idx] if not line.strip(): continue ind = len(line) - len(line.lstrip()) if ind == child_indent and line.rstrip().endswith(":") and (start is None): start = idx elif ind == child_indent and line.rstrip().endswith(":"): bounds.append((start, idx)) start = idx elif ind == 0: if start is not None: bounds.append((start, idx)) start = None break if start is not None: bounds.append((start, n)) result = list(lines) for a, b in bounds: block = result[a:b] vol_idx = [k for k, l in enumerate(block) if l.strip() == "volumes:" and (len(l) - len(l.lstrip())) == child_indent + 2] if len(vol_idx) <= 1: continue first = vol_idx[0] # `volumes:` sits at the service-key indent; its list items are # indented two spaces further. key_indent = " " * (child_indent + 2) item_prefix = key_indent + " - " entries: list[str] = [] drop: set[int] = set() for vi in vol_idx: drop.add(vi) k = vi + 1 while k < len(block) and block[k].lstrip().startswith("- "): entries.append(block[k]) drop.add(k) k += 1 # also swallow a blank/comment tail belonging to this list new_block = [ln for k, ln in enumerate(block) if k not in drop] # Re-insert the merged list exactly where the FIRST volumes: key was. # `first` is an index into the original block, so convert it to an # index into the filtered block by counting how many dropped lines # before it disappeared. at = first - sum(1 for k in drop if k < first) new_block[at:at] = [key_indent + "volumes:"] + entries result[a:b] = new_block return "\n".join(result) + ("\n" if text.endswith("\n") else "") def main(argv): names = argv[1:] or [p.name for p in sorted(SERVICES.iterdir()) if p.is_dir() and (p / "docker-compose.yml").exists()] changed = [] for name in names: p = SERVICES / name / "docker-compose.yml" if not p.exists(): continue text = p.read_text() new = fix_text_lines(text) if new != text: # verify it now parses and that the volumes survived try: d = yaml.safe_load(new) except Exception as e: print(f"{name}: REFUSING invalid output ({e})") continue p.write_text(new) changed.append(name) print(f"{name}: merged duplicate volumes block(s)") print("rewritten:", ", ".join(changed) if changed else "(none)") return 0 if __name__ == "__main__": raise SystemExit(main(sys.argv))