Skip to content

Commit 79ca41d

Browse files
committed
chore: harden temporary assignment scan
1 parent d2bf477 commit 79ca41d

1 file changed

Lines changed: 160 additions & 0 deletions

File tree

Lines changed: 160 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,160 @@
1+
name: Temporary WCA assignment scan v2
2+
3+
on:
4+
pull_request:
5+
paths:
6+
- .github/workflows/wca-assignment-scan-v2.yml
7+
8+
permissions:
9+
contents: read
10+
11+
jobs:
12+
scan:
13+
runs-on: ubuntu-latest
14+
timeout-minutes: 20
15+
steps:
16+
- name: Scan public WCIF assignment data
17+
shell: python
18+
run: |
19+
import csv, json, os, time, urllib.error, urllib.parse, urllib.request
20+
from collections import Counter
21+
from concurrent.futures import ThreadPoolExecutor, as_completed
22+
from datetime import date
23+
24+
START = date.fromisoformat("2026-04-30")
25+
END = date.fromisoformat("2026-07-30")
26+
API = "https://www.worldcubeassociation.org/api/v0"
27+
HEADERS = {"User-Agent": "competitiongroups.com assignment usage analysis"}
28+
OUT = "assignment-scan-v2"
29+
os.makedirs(OUT, exist_ok=True)
30+
31+
def get_json(url, attempts=4):
32+
last = None
33+
for attempt in range(attempts):
34+
try:
35+
request = urllib.request.Request(url, headers=HEADERS)
36+
with urllib.request.urlopen(request, timeout=30) as response:
37+
return json.loads(response.read().decode(response.headers.get_content_charset() or "utf-8"))
38+
except (urllib.error.HTTPError, urllib.error.URLError, TimeoutError) as error:
39+
last = error
40+
if getattr(error, "code", None) in (404, 422):
41+
raise
42+
time.sleep(min(2 ** attempt, 8))
43+
raise last
44+
45+
competitions = []
46+
page_signatures = set()
47+
for page in range(1, 101):
48+
query = urllib.parse.urlencode({"start": START.isoformat(), "end": END.isoformat(), "page": page})
49+
batch = get_json(f"{API}/competitions?{query}")
50+
if not batch:
51+
break
52+
signature = tuple(item["id"] for item in batch)
53+
if signature in page_signatures:
54+
print(f"Duplicate page detected at page {page}; stopping pagination")
55+
break
56+
page_signatures.add(signature)
57+
competitions.extend(batch)
58+
print(f"Competition page {page}: {len(batch)}", flush=True)
59+
if len(batch) < 25:
60+
break
61+
else:
62+
raise RuntimeError("Competition pagination exceeded 100 pages")
63+
64+
competitions = {
65+
item["id"]: item for item in competitions
66+
if START <= date.fromisoformat(item["start_date"]) <= END
67+
}
68+
competitions = sorted(competitions.values(), key=lambda item: (item["start_date"], item["id"]))
69+
print(f"Competitions in final date range: {len(competitions)}", flush=True)
70+
71+
def scan(item):
72+
cid = item["id"]
73+
row = {
74+
"competition_id": cid,
75+
"name": item.get("name", ""),
76+
"start_date": item.get("start_date", ""),
77+
"end_date": item.get("end_date", ""),
78+
"country_iso2": item.get("country_iso2", ""),
79+
"city": item.get("city", ""),
80+
"cancelled": bool(item.get("cancelled_at")),
81+
"persons_total": 0,
82+
"persons_with_assignments": 0,
83+
"assignment_count": 0,
84+
"competitor_assignment_count": 0,
85+
"staff_assignment_count": 0,
86+
"other_assignment_count": 0,
87+
"assignment_codes": "",
88+
"has_person_assignments": False,
89+
"has_competitor_assignments": False,
90+
"has_staff_assignments": False,
91+
"wcif_status": "ok",
92+
"inference": "no assignment data",
93+
"wca_url": item.get("url") or f"https://www.worldcubeassociation.org/competitions/{cid}",
94+
"competitiongroups_url": f"https://www.competitiongroups.com/competitions/{cid}",
95+
}
96+
try:
97+
wcif = get_json(f"{API}/competitions/{cid}/wcif/public")
98+
persons = wcif.get("persons") or []
99+
row["persons_total"] = len(persons)
100+
codes = Counter()
101+
for person in persons:
102+
assignments = person.get("assignments") or []
103+
row["persons_with_assignments"] += bool(assignments)
104+
for assignment in assignments:
105+
codes[assignment.get("assignmentCode") or "unknown"] += 1
106+
row["assignment_count"] = sum(codes.values())
107+
row["competitor_assignment_count"] = codes.get("competitor", 0)
108+
row["staff_assignment_count"] = sum(v for k, v in codes.items() if k.startswith("staff-"))
109+
row["other_assignment_count"] = row["assignment_count"] - row["competitor_assignment_count"] - row["staff_assignment_count"]
110+
row["assignment_codes"] = ";".join(f"{k}:{v}" for k, v in sorted(codes.items()))
111+
row["has_person_assignments"] = row["assignment_count"] > 0
112+
row["has_competitor_assignments"] = row["competitor_assignment_count"] > 0
113+
row["has_staff_assignments"] = row["staff_assignment_count"] > 0
114+
if row["has_person_assignments"]:
115+
row["inference"] = "potential Competition Groups user: assignment data present"
116+
except urllib.error.HTTPError as error:
117+
row["wcif_status"] = f"http_{error.code}"
118+
except Exception as error:
119+
row["wcif_status"] = f"error:{type(error).__name__}"
120+
return row
121+
122+
rows = []
123+
with ThreadPoolExecutor(max_workers=20) as executor:
124+
futures = [executor.submit(scan, item) for item in competitions]
125+
for index, future in enumerate(as_completed(futures), 1):
126+
rows.append(future.result())
127+
if index % 50 == 0 or index == len(futures):
128+
print(f"Scanned {index}/{len(futures)}", flush=True)
129+
130+
rows.sort(key=lambda item: (item["start_date"], item["competition_id"]))
131+
assignment_rows = [row for row in rows if row["has_person_assignments"]]
132+
fields = list(rows[0]) if rows else []
133+
for filename, output_rows in (("all_competitions.csv", rows), ("competitions_with_assignments.csv", assignment_rows)):
134+
with open(f"{OUT}/{filename}", "w", newline="", encoding="utf-8") as output:
135+
writer = csv.DictWriter(output, fieldnames=fields)
136+
writer.writeheader(); writer.writerows(output_rows)
137+
138+
summary = {
139+
"date_window": {"start": START.isoformat(), "end": END.isoformat(), "basis": "competition start_date"},
140+
"competitions_total": len(rows),
141+
"competitions_cancelled": sum(row["cancelled"] for row in rows),
142+
"competitions_with_assignments": len(assignment_rows),
143+
"assignment_coverage_percent": round(100 * len(assignment_rows) / len(rows), 2) if rows else 0,
144+
"competitions_with_competitor_assignments": sum(row["has_competitor_assignments"] for row in rows),
145+
"competitions_with_staff_assignments": sum(row["has_staff_assignments"] for row in rows),
146+
"total_person_assignments": sum(row["assignment_count"] for row in rows),
147+
"wcif_failures": dict(Counter(row["wcif_status"] for row in rows if row["wcif_status"] != "ok")),
148+
"by_country": dict(Counter(row["country_iso2"] for row in assignment_rows).most_common()),
149+
"by_month": dict(sorted(Counter(row["start_date"][:7] for row in assignment_rows).items())),
150+
}
151+
with open(f"{OUT}/summary.json", "w", encoding="utf-8") as output:
152+
json.dump(summary, output, indent=2, ensure_ascii=False)
153+
print(json.dumps(summary, indent=2), flush=True)
154+
155+
- name: Upload scan results
156+
uses: actions/upload-artifact@v4
157+
with:
158+
name: wca-assignment-scan-v2-2026-07-30
159+
path: assignment-scan-v2/
160+
retention-days: 7

0 commit comments

Comments
 (0)