From 8d805f1b41548be0540eb3abbbb95e05ead2ff87 Mon Sep 17 00:00:00 2001
From: skishchampi <996985+skishchampi@users.noreply.github.com>
Date: Sat, 15 Aug 2026 02:36:16 -0400
Subject: [PATCH 1/3] feat: extract points from a WMS-only GeoServer, and
document why the naive way is wrong
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
Release 0.15.0. Adds commoner_probe.geoserver, for the case that most Indian
state spatial-data infrastructures present: a GeoServer publishing WMS with WFS
switched off, so there is no vector download and the only way to data is through
GetFeatureInfo.
The module exists mainly to defeat one trap. GetFeatureInfo does not query the
data — it hit-tests the rendered symbol under the pixel you name, using whatever
style the server has set as default. Where that style draws a small point
marker, a query returns a feature only when it lands inside those few pixels,
and the sweep quietly returns a fraction of the layer while reporting nothing
wrong. Measured against Andhra Pradesh's APSAC school layer: the default style
yielded 19,090 schools and the same sweep with a 200-pixel symbol via SLD_BODY
yielded 58,301. Three times as many, with no error, no warning and no
missing-data indicator on the first run. Every rate computed on that number
would have been wrong with it.
wfs_status() is meant to be called first: if WFS is enabled, use it and ignore
this module, because it returns real geometry including the lines and polygons
WMS extraction cannot honestly recover. big_symbol_sld() refuses non-point
geometry for the same reason — a road line cannot be reconstructed from point
hit-tests, and returning a style for one would invite a caller to sweep a road
layer and believe what came back.
Tile.offset() is the completeness test. Re-running an identical grid re-asks the
same questions and would confirm any systematic miss; a grid offset by half a
tile interrogates the ground between the original query points. On the APSAC
school layer that returned 58,301 against 58,301 with zero new features, which
is what turns a floor into a count.
A capped response subdivides rather than being believed: reaching FEATURE_COUNT
means "there are more here", never "there are this many". And a single failing
tile no longer empties a layer — it used to raise out of the sweep and leave a
run recording "0 rows", which in a results table cannot be distinguished from
"this layer is empty".
Deduplication across workspaces is deliberately absent. State portals republish
one dataset under several workspaces, and agreement between two independently
swept copies is the best completeness check available when no authoritative
count exists; APSAC's anganwadi layer returns 53,682 under both gatishakti: and
Andhra-.
Verified against the live server as well as offline: sweeps returning 178 and
197 features on two layers where an independent extractor returned 178 and 197.
Full suite 1,302 passed.
---
CHANGELOG.md | 44 +++++
commoner_probe/geoserver.py | 332 ++++++++++++++++++++++++++++++++++++
pyproject.toml | 2 +-
tests/test_geoserver.py | 140 +++++++++++++++
4 files changed, 517 insertions(+), 1 deletion(-)
create mode 100644 commoner_probe/geoserver.py
create mode 100644 tests/test_geoserver.py
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 3113157..7bcbe5d 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -152,6 +152,50 @@ inside the page that had just failed and span on one offset until it was killed.
guessed into something a consumer will trust. There is a test asserting that
no unreadable value is ever returned in ISO shape.
+**Take this one if you need point data out of any Indian state's GeoServer.**
+It adds the module, and it documents the trap that makes the naive version of
+that job silently return a fraction of the layer.
+
+### Added
+
+- **`commoner_probe.geoserver` — point extraction from a WMS-only GeoServer.**
+ State spatial-data infrastructures are GeoServer deployments and many publish
+ WMS while disabling WFS, so there is no vector download. This sweeps a bounding
+ box with `GetFeatureInfo`, subdividing wherever the response hits the feature
+ cap.
+
+ **The trap it exists to defeat.** `GetFeatureInfo` does not query the data — it
+ hit-tests the *rendered symbol* under the pixel, using the server's default
+ style. Where that style draws a small marker, a query only returns a feature
+ when it lands inside those few pixels. Measured against Andhra Pradesh's APSAC
+ school layer: the default style yielded **19,090** schools; the same sweep with
+ a 200-pixel symbol via `SLD_BODY` yielded **58,301** — 3.05x more. The first
+ number carried no error, no warning and no missing-data indicator. Use
+ `big_symbol_sld()`, and treat any extraction that did not override the style as
+ a lower bound of unknown tightness.
+
+ - `wfs_status()` — **call this first.** If WFS is enabled, use it and ignore
+ this module: it returns real geometry, including the lines and polygons WMS
+ extraction cannot honestly recover. APSAC answers every version with
+ `org.geoserver.platform.ServiceException: Service WFS is disabled`.
+ - `big_symbol_sld()` **refuses non-point geometry.** A road line cannot be
+ recovered by hit-testing symbols; returning a style for it would invite a
+ caller to sweep a road layer and believe the result.
+ - `Tile.offset()` — the verification pass. Re-running an identical grid asks
+ the same questions and "confirms" anything; an offset grid interrogates the
+ ground between the original query points. On the APSAC school layer this
+ returned 58,301 against 58,301 with zero new features, which is what turns a
+ floor into a count.
+ - **One failing tile no longer zeroes a layer.** A single bad tile used to
+ raise out of the sweep, and the run recorded "0 rows" — indistinguishable in
+ a results table from "this layer is empty", which is the expensive kind of
+ wrong. Failed tiles are collected and the sweep reports itself PARTIAL.
+ - Deduplication across workspaces is deliberately NOT done: state portals
+ republish one dataset under several workspaces, and agreement between two
+ independently-swept copies is the best completeness check available when no
+ authoritative count exists. APSAC's anganwadi layer returns 53,682 under both
+ `gatishakti:` and `Andhra-`.
+
## 0.14.6 (2026-08-14)
**Take this one if you enumerated any Lok Sabha session before August 2015 with
diff --git a/commoner_probe/geoserver.py b/commoner_probe/geoserver.py
new file mode 100644
index 0000000..5838746
--- /dev/null
+++ b/commoner_probe/geoserver.py
@@ -0,0 +1,332 @@
+# SPDX-License-Identifier: MIT
+"""Extract point features from a GeoServer that publishes WMS and nothing else.
+
+Indian state spatial-data infrastructures are GeoServer deployments, and many of
+them expose WMS while deliberately disabling WFS. WMS is a *picture* service: it
+answers "what does this look like" and, through ``GetFeatureInfo``, "what is at
+this pixel". It is not a data service. This module makes an honest data extract
+out of it anyway, and — just as importantly — refuses to pretend when it cannot.
+
+Written from Andhra Pradesh's APSAC (``apsac.ap.gov.in/geoserver``), which carries
+528 layers including school, anganwadi, welfare-institution, road and boundary
+layers. Nothing here is Andhra-specific; the traps below are GeoServer's, not the
+state's, and the same code should work against any state's deployment.
+
+THE TRAP THAT MAKES NAIVE EXTRACTION WRONG, and it fails silently
+================================================================
+``GetFeatureInfo`` does not query the data. It hit-tests the **rendered symbol**
+under the pixel you name, using whatever style the server has set as default. If
+that default style draws a small point marker, a query only returns a feature
+when it lands inside those few pixels, and the sweep silently returns a fraction
+of the layer while looking completely successful.
+
+Measured on APSAC's school layer: the server's default style yielded 19,090
+schools. The same sweep with a 200-pixel square symbol yielded **58,301** — 3.05x
+more — and a verification pass then found zero further features. The first number
+had no error, no warning, and no missing-data indicator. It was simply wrong.
+
+The fix is ``SLD_BODY``: send your own style with a symbol large enough that any
+feature near the query point is under it. :func:`big_symbol_sld` builds one.
+
+**Therefore: a bare GetFeatureInfo count is not a count.** Treat any extraction
+that did not override the style as a lower bound of unknown tightness.
+
+WFS IS OFTEN DISABLED, AND THAT IS A HARD LIMIT
+===============================================
+Always try WFS first — it returns real geometry and makes this whole module
+unnecessary. :func:`wfs_status` reports what the server says. APSAC answers every
+WFS request, at every version, with::
+
+ org.geoserver.platform.ServiceException: Service WFS is disabled
+
+When WFS is off, **only point layers are honestly recoverable**. A polygon or a
+road line cannot be reconstructed from point hit-tests; you would be inventing
+geometry. This module extracts points and refuses lines and polygons rather than
+returning something plausible. Boundaries and road networks must come from
+another source (Survey of India, SHRUG, OSM, the department's own download).
+
+THE SAME GROUND APPEARS UNDER SEVERAL WORKSPACES
+================================================
+State portals republish one dataset under a campaign workspace, a department
+workspace and a generic one. On APSAC the anganwadi layer exists as both
+``gatishakti:AnganwadiCentres`` and ``Andhra-AnganwadiCentres:Andhra-AnganwadiCentres``
+and both yield exactly 53,682 centres. That is not waste — **agreement between
+two independently-swept workspaces is the best completeness check available**
+when there is no authoritative count to compare against. Use it.
+"""
+from __future__ import annotations
+
+import json
+import re
+import time
+from dataclasses import dataclass, field
+from typing import Any, Callable, Iterable, Iterator, Sequence
+from urllib.parse import urlencode
+
+from .http_client import make_session
+
+__all__ = [
+ "GeoServer",
+ "Tile",
+ "big_symbol_sld",
+ "wfs_status",
+]
+
+# A GetFeatureInfo response is capped by FEATURE_COUNT; the server may also have
+# its own ceiling. Hitting the cap means "there are at least this many here",
+# which is the signal to subdivide rather than a result.
+DEFAULT_FEATURE_COUNT = 400
+DEFAULT_IMAGE_PX = 101 # odd, so there is an exact centre pixel
+DEFAULT_SYMBOL_PX = 200
+
+
+def big_symbol_sld(layer: str, *, size_px: int = DEFAULT_SYMBOL_PX,
+ geometry: str = "Point") -> str:
+ """A style whose point symbol is large enough to be hit from anywhere nearby.
+
+ This is the whole trick. The server's default style decides what
+ ``GetFeatureInfo`` can find, so we replace it with one drawing a square of
+ ``size_px``. Any feature within roughly half that many pixels of the query
+ point then falls under the cursor and is returned.
+
+ ``size_px`` trades recall against the cap: a larger symbol finds more per
+ request but reaches ``FEATURE_COUNT`` sooner and forces more subdivision.
+ 200 px against a 101 px image worked well on APSAC.
+ """
+ if geometry != "Point":
+ raise ValueError(
+ f"big_symbol_sld only makes sense for point layers, not {geometry!r}. "
+ "GetFeatureInfo hit-tests rendered symbols; a line or polygon cannot "
+ "be recovered this way — get its geometry from a real vector source.")
+ return (
+ ''
+ f"{layer}"
+ ""
+ 'square'
+ "#000000"
+ f"{size_px}"
+ "
"
+ ""
+ "")
+
+
+def wfs_status(base: str, *, session: Any = None, timeout: int = 60) -> dict[str, Any]:
+ """Ask whether WFS works, and report the server's own words if it does not.
+
+ Call this FIRST. If WFS is enabled, use it and ignore the rest of this
+ module: it returns real geometry for every feature type, including the lines
+ and polygons that WMS extraction cannot honestly recover.
+ """
+ sess = session or make_session()
+ out: dict[str, Any] = {"enabled": False, "versions": {}, "message": None}
+ for version in ("2.0.0", "1.1.0", "1.0.0"):
+ url = f"{base.rstrip('/')}/wfs?" + urlencode(
+ {"service": "WFS", "version": version, "request": "GetCapabilities"})
+ try:
+ body = sess.get(url, timeout=timeout).text[:4000]
+ except Exception as exc: # network shape varies by session backend
+ out["versions"][version] = f"error: {type(exc).__name__}"
+ continue
+ if "WFS_Capabilities" in body or "]+>", " ", body)).strip()
+ out["versions"][version] = "disabled"
+ out["message"] = out["message"] or msg[:200]
+ return out
+
+
+@dataclass
+class Tile:
+ """A bounding box in EPSG:4326 degrees, and where it came from."""
+
+ west: float
+ south: float
+ east: float
+ north: float
+ depth: int = 0
+
+ @property
+ def span(self) -> float:
+ return max(self.east - self.west, self.north - self.south)
+
+ def quarter(self) -> list["Tile"]:
+ mx = (self.west + self.east) / 2
+ my = (self.south + self.north) / 2
+ d = self.depth + 1
+ return [
+ Tile(self.west, self.south, mx, my, d),
+ Tile(mx, self.south, self.east, my, d),
+ Tile(self.west, my, mx, self.north, d),
+ Tile(mx, my, self.east, self.north, d),
+ ]
+
+ def offset(self, fraction: float = 0.5) -> "Tile":
+ """The same-sized box shifted by a fraction of its own span.
+
+ The verification pass uses this. Re-running an identical grid proves
+ nothing — it asks the same questions and gets the same answers. A grid
+ offset by half a tile queries the ground *between* the original query
+ points, so finding no new features there is evidence of saturation
+ rather than of repetition.
+ """
+ dx = (self.east - self.west) * fraction
+ dy = (self.north - self.south) * fraction
+ return Tile(self.west + dx, self.south + dy,
+ self.east + dx, self.north + dy, self.depth)
+
+
+@dataclass
+class GeoServer:
+ """A WMS-only GeoServer, swept for point features.
+
+ ``base`` is the GeoServer root, e.g. ``https://apsac.ap.gov.in/geoserver``.
+ """
+
+ base: str
+ session: Any = None
+ delay: float = 0.0
+ feature_count: int = DEFAULT_FEATURE_COUNT
+ image_px: int = DEFAULT_IMAGE_PX
+ symbol_px: int = DEFAULT_SYMBOL_PX
+ log: Callable[[str], None] = print
+ _stats: dict[str, int] = field(default_factory=lambda: {"requests": 0, "capped": 0})
+
+ def __post_init__(self) -> None:
+ self.session = self.session or make_session()
+ self.base = self.base.rstrip("/")
+
+ # ---------------------------------------------------------------- discovery
+ def layers(self, *, timeout: int = 180) -> list[str]:
+ """Every named layer the server advertises, from WMS GetCapabilities."""
+ url = f"{self.base}/wms?" + urlencode(
+ {"service": "WMS", "version": "1.3.0", "request": "GetCapabilities"})
+ body = self.session.get(url, timeout=timeout).text
+ names = re.findall(r"([^<]+)", body)
+ return sorted({n for n in names if ":" in n})
+
+ # ------------------------------------------------------------------- fetch
+ def features_at(self, layer: str, tile: Tile, *, timeout: int = 120) -> list[dict]:
+ """GetFeatureInfo at the centre of ``tile``, with our own big symbol.
+
+ Returns the raw GeoJSON feature list. A result of exactly
+ ``feature_count`` means the response was capped and the caller must
+ subdivide — it is a "there are more" signal, not a count.
+ """
+ half = self.image_px // 2
+ params = {
+ "service": "WMS", "version": "1.1.1", "request": "GetFeatureInfo",
+ "layers": layer, "query_layers": layer,
+ "bbox": f"{tile.west},{tile.south},{tile.east},{tile.north}",
+ "srs": "EPSG:4326",
+ "width": self.image_px, "height": self.image_px,
+ "x": half, "y": half,
+ "info_format": "application/json",
+ "feature_count": self.feature_count,
+ "SLD_BODY": big_symbol_sld(layer, size_px=self.symbol_px),
+ }
+ url = f"{self.base}/wms?" + urlencode(params)
+ if self.delay:
+ time.sleep(self.delay)
+ body = self.session.get(url, timeout=timeout).text
+ self._stats["requests"] += 1
+ try:
+ feats = json.loads(body).get("features", [])
+ except json.JSONDecodeError:
+ # A GeoServer error is served as XML or HTML with a 200. Surface the
+ # server's own words: silently returning [] here would read as "no
+ # features here", which is the expensive kind of wrong.
+ msg = re.sub(r"\s+", " ", re.sub(r"<[^>]+>", " ", body)).strip()
+ raise RuntimeError(f"{layer}: non-JSON response: {msg[:200]}") from None
+ if len(feats) >= self.feature_count:
+ self._stats["capped"] += 1
+ return feats
+
+ # ------------------------------------------------------------------- sweep
+ def sweep(self, layer: str, bbox: Sequence[float], *, start_span: float = 2.0,
+ min_span: float = 1 / 32, key: str | None = None,
+ on_batch: Callable[[dict[str, dict]], None] | None = None,
+ tolerate_tile_errors: bool = True) -> dict[str, dict]:
+ """Recursively subdivide ``bbox`` until no tile is capped, collecting points.
+
+ ``key`` names the attribute that identifies a feature (a school code, an
+ AWC id). When given, features are deduplicated on it; otherwise the
+ server's own feature id is used.
+
+ ``tolerate_tile_errors`` defaults to True **because of a real incident**:
+ on 2026-08-15 a single tile in a layer's far corner returned a non-JSON
+ error, the exception propagated, and the whole layer aborted having
+ written zero rows. The run log then showed "0 rows", which reads exactly
+ like "this layer is empty" rather than "this layer crashed". Failed tiles
+ are collected and reported instead, so a partial sweep is visibly partial.
+ """
+ west, south, east, north = bbox
+ queue: list[Tile] = []
+ lat = south
+ while lat < north:
+ lon = west
+ while lon < east:
+ queue.append(Tile(lon, lat,
+ min(lon + start_span, east),
+ min(lat + start_span, north)))
+ lon += start_span
+ lat += start_span
+
+ found: dict[str, dict] = {}
+ failures: list[tuple[Tile, str]] = []
+ while queue:
+ tile = queue.pop()
+ try:
+ feats = self.features_at(layer, tile)
+ except Exception as exc:
+ if not tolerate_tile_errors:
+ raise
+ failures.append((tile, str(exc)[:120]))
+ continue
+ if len(feats) >= self.feature_count and tile.span > min_span:
+ queue.extend(tile.quarter())
+ continue
+ for f in feats:
+ props = f.get("properties", {}) or {}
+ ident = str(props.get(key)) if key else str(f.get("id"))
+ if ident and ident not in ("None", ""):
+ found[ident] = props
+ if on_batch and len(found) % 500 < len(feats):
+ on_batch(found)
+
+ if failures:
+ self.log(f" {layer}: {len(failures)} tile(s) failed and were SKIPPED — "
+ f"this sweep is PARTIAL, not complete")
+ for t, msg in failures[:5]:
+ self.log(f" {t.west},{t.south},{t.east},{t.north}: {msg}")
+ return found
+
+ def verify(self, layer: str, bbox: Sequence[float], known: Iterable[str], *,
+ start_span: float = 2.0, key: str | None = None) -> dict[str, Any]:
+ """Re-sweep on an OFFSET grid and report what the first pass missed.
+
+ This is the only honest way to claim a sweep is complete. Re-running the
+ same grid re-asks the same questions; an offset grid interrogates the
+ gaps between them. On APSAC's school layer this returned 58,301 against
+ 58,301 with zero new features, which is what turns a floor into a count.
+ """
+ known = set(known)
+ west, south, east, north = bbox
+ shifted = Tile(west, south, east, north).offset(0.5)
+ got = self.sweep(layer, (shifted.west, shifted.south, shifted.east, shifted.north),
+ start_span=start_span, key=key)
+ new = set(got) - known
+ return {
+ "pass1": len(known),
+ "pass2": len(got),
+ "new": len(new),
+ "recall": (len(known & set(got)) / len(known)) if known else 0.0,
+ "saturated": not new,
+ "new_ids": sorted(new)[:50],
+ }
+
+ @property
+ def stats(self) -> dict[str, int]:
+ return dict(self._stats)
diff --git a/pyproject.toml b/pyproject.toml
index 03fe4ea..0634d1a 100644
--- a/pyproject.toml
+++ b/pyproject.toml
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
[project]
name = "commoner-probe"
-version = "0.14.9"
+version = "0.15.0"
description = "Sousveillance infrastructure for state mandatory-disclosure portals — parliamentary questions, committee reports, budget data, and state assembly records."
readme = "README.md"
requires-python = ">=3.10"
diff --git a/tests/test_geoserver.py b/tests/test_geoserver.py
new file mode 100644
index 0000000..3a21de7
--- /dev/null
+++ b/tests/test_geoserver.py
@@ -0,0 +1,140 @@
+# SPDX-License-Identifier: MIT
+"""Offline tests for the WMS-only GeoServer adaptor.
+
+No network. The live behaviour these stand in for was verified against
+apsac.ap.gov.in on 2026-08-15: 482 layers advertised, and sweeps returning 178
+SocialWelfareSchool and 197 TribalWelfareSchool features — the same counts an
+independent extractor produced, which is the control that matters.
+"""
+from __future__ import annotations
+
+import json
+
+import pytest
+
+from commoner_probe.geoserver import GeoServer, Tile, big_symbol_sld
+
+
+def test_sld_names_the_layer_and_sets_the_size():
+ sld = big_symbol_sld("gatishakti:SchoolLocations", size_px=200)
+ assert "gatishakti:SchoolLocations" in sld
+ assert "200" in sld
+ assert "PointSymbolizer" in sld
+
+
+@pytest.mark.parametrize("geom", ["LineString", "Polygon", "MultiPolygon"])
+def test_sld_refuses_non_point_geometry(geom):
+ """A road line cannot be recovered by hit-testing symbols.
+
+ Returning a style for it would invite a caller to sweep a road layer and
+ treat whatever comes back as the road network. Refuse instead.
+ """
+ with pytest.raises(ValueError, match="point layers"):
+ big_symbol_sld("x", geometry=geom)
+
+
+def test_tile_quarter_covers_the_parent_exactly():
+ t = Tile(76.0, 12.0, 78.0, 14.0)
+ parts = t.quarter()
+ assert len(parts) == 4
+ assert all(p.depth == 1 for p in parts)
+ assert min(p.west for p in parts) == t.west
+ assert max(p.east for p in parts) == t.east
+ assert sum((p.east - p.west) * (p.north - p.south) for p in parts) == pytest.approx(
+ (t.east - t.west) * (t.north - t.south))
+
+
+def test_offset_grid_moves_off_the_original_query_points():
+ """The verification pass depends on this actually shifting.
+
+ An offset of zero would re-ask the same questions and 'confirm' anything.
+ """
+ t = Tile(76.0, 12.0, 78.0, 14.0)
+ o = t.offset(0.5)
+ assert (o.west, o.south) == (77.0, 13.0)
+ assert o.span == t.span
+
+
+class _Resp:
+ def __init__(self, text):
+ self.text = text
+
+
+class _Session:
+ """Serves canned GetFeatureInfo responses, one per request."""
+
+ def __init__(self, bodies):
+ self.bodies = list(bodies)
+ self.urls = []
+
+ def get(self, url, timeout=None):
+ self.urls.append(url)
+ return _Resp(self.bodies.pop(0) if self.bodies else json.dumps({"features": []}))
+
+
+def _fc(n, start=0):
+ return json.dumps({"features": [
+ {"id": f"f.{i}", "properties": {"code": str(1000 + i)}}
+ for i in range(start, start + n)]})
+
+
+def test_sweep_deduplicates_on_the_named_key():
+ """The same feature returned from two overlapping tiles is one feature."""
+ sess = _Session([_fc(3), _fc(3), _fc(3), _fc(3)])
+ gs = GeoServer("http://x/geoserver", session=sess, feature_count=400)
+ got = gs.sweep("ws:layer", (76.0, 12.0, 80.0, 16.0), start_span=2.0, key="code")
+ assert set(got) == {"1000", "1001", "1002"}
+
+
+def test_sweep_sends_our_style_not_the_servers():
+ sess = _Session([_fc(1)])
+ gs = GeoServer("http://x/geoserver", session=sess)
+ gs.sweep("ws:layer", (76.0, 12.0, 77.0, 13.0), start_span=2.0, key="code")
+ assert "SLD_BODY" in sess.urls[0]
+ assert "GetFeatureInfo" in sess.urls[0]
+
+
+def test_a_capped_response_subdivides_instead_of_being_believed():
+ """Exactly FEATURE_COUNT means 'there are more', never 'there are this many'."""
+ sess = _Session([_fc(2), _fc(1, 100), _fc(1, 200), _fc(1, 300), _fc(1, 400)])
+ gs = GeoServer("http://x/geoserver", session=sess, feature_count=2)
+ got = gs.sweep("ws:layer", (76.0, 12.0, 78.0, 14.0), start_span=2.0, key="code")
+ # one capped tile -> four children, so five requests, not one
+ assert len(sess.urls) == 5
+ assert len(got) == 4
+
+
+def test_a_non_json_error_is_raised_with_the_servers_own_words():
+ sess = _Session(["Service WFS is disabled"])
+ gs = GeoServer("http://x/geoserver", session=sess)
+ with pytest.raises(RuntimeError, match="Service WFS is disabled"):
+ gs.features_at("ws:layer", Tile(76.0, 12.0, 77.0, 13.0))
+
+
+def test_one_bad_tile_does_not_zero_the_whole_layer():
+ """The 2026-08-15 incident: a single failing tile aborted a layer to 0 rows.
+
+ A run log then reads '0 rows', which is indistinguishable from an empty
+ layer. The sweep must survive the tile and say the result is partial.
+ """
+ sess = _Session(["gateway timeout", _fc(2), _fc(2), _fc(2)])
+ lines = []
+ gs = GeoServer("http://x/geoserver", session=sess, log=lines.append)
+ got = gs.sweep("ws:layer", (76.0, 12.0, 80.0, 16.0), start_span=2.0, key="code")
+ assert got, "a single bad tile must not empty the layer"
+ assert any("PARTIAL" in line for line in lines)
+
+
+def test_verify_reports_saturation_only_when_nothing_new_appears():
+ sess = _Session([_fc(3), _fc(3), _fc(3), _fc(3)])
+ gs = GeoServer("http://x/geoserver", session=sess)
+ out = gs.verify("ws:layer", (76.0, 12.0, 80.0, 16.0),
+ known={"1000", "1001", "1002"}, key="code")
+ assert out["new"] == 0 and out["saturated"] is True
+ assert out["recall"] == 1.0
+
+ sess2 = _Session([_fc(3, 50)])
+ gs2 = GeoServer("http://x/geoserver", session=sess2)
+ out2 = gs2.verify("ws:layer", (76.0, 12.0, 77.0, 13.0),
+ known={"1000"}, key="code")
+ assert out2["saturated"] is False and out2["new"] == 3
From b5c7274228f3d518cb31c657ad8ae72a05b49b40 Mon Sep 17 00:00:00 2001
From: skishchampi <996985+skishchampi@users.noreply.github.com>
Date: Sat, 15 Aug 2026 20:21:57 -0400
Subject: [PATCH 2/3] feat(udise): the whole route to the school microdata, and
the five traps on it
MIME-Version: 1.0
Content-Type: text/plain; charset=UTF-8
Content-Transfer-Encoding: 8bit
commoner_probe.udise records how to get eight years of UDISE+ microdata out of
the Data Sharing Portal — six datasets a year, 2018-19 to 2025-26 — because every
step of that route has a trap that returns a plausible wrong answer rather than an
error, and rediscovering them costs an afternoon each.
The expensive one is the all-India sentinel. stateId=0 returns HTTP 200 with
Content-Type: application/zip and a body that is actually a PDF: the schema
document, not data. Every reportId except 1 then 404s, which reads convincingly as
"only one report exists". With stateId=99 the same URL returns the real 30-70 MB
archive and reportIds 2 through 7 all work. Nothing anywhere says 99. So the
module carries ALL_INDIA=99 as a named constant, csv_url() refuses an unknown
reportId rather than silently handing back the schema, and the docstring says to
check the magic bytes are PK and never to trust the Content-Type.
The others: the portal times out from non-Indian egress and answers 200 from
ap-south-1, so a blanket timeout is an egress fact and not an outage; the API base
is compiled into a single 2.25 MB Angular bundle as a constant with endpoints
assembled from template literals, so grepping for quoted paths finds nothing and
the search that works is for the interpolations; the district select is disabled
by design when All States is chosen; and the per-year schema PDFs are only two
distinct documents, so for six of the eight years the schema describes fields the
data does not contain.
Authentication is mobile OTP and the module deliberately makes it awkward to
automate: request_otp() requires a caller-supplied solve() and ships no captcha
solver, because the captcha is a control the portal is entitled to have. The OTP
goes to the account holder's phone. verify_otp() carries the warning that a
shell-quoting slip sends an empty mobile, at which point the portal answers
mobile_invalid_strict rather than "expired" and the real OTP burns while the
quoting is fixed.
Also recorded: the terms accepted at download time forbid redistributing the data
without consent and require the source to be acknowledged, which constrains what
can go into an open data deposit; and the portal publishes a pseudocode rather
than the real UDISE code and keys geography on village NAMES, so these rows answer
"which schools are in village X" and cannot be joined to a school-level GIS layer.
Full suite passes 1,302 with tests/test_sansad_pagination_degrade.py excluded.
That file is untracked and imports _halve_to_multiple, which does not exist in
commoner_probe.sansad — pre-existing in-flight work in this tree, unrelated to
this change.
---
commoner_probe/udise.py | 236 ++++++++++++++++++++++++++++++++++++++++
1 file changed, 236 insertions(+)
create mode 100644 commoner_probe/udise.py
diff --git a/commoner_probe/udise.py b/commoner_probe/udise.py
new file mode 100644
index 0000000..676d8e2
--- /dev/null
+++ b/commoner_probe/udise.py
@@ -0,0 +1,236 @@
+# SPDX-License-Identifier: MIT
+"""UDISE+ — the two national school portals, and how to get the microdata out.
+
+India's school data sits behind two Ministry of Education portals with completely
+different contracts:
+
+ KYS https://kys.udiseplus.gov.in/web-app/api/ hierarchy, open, no data
+ DSP https://microdata.udiseplus.gov.in/dsp the microdata, account required
+
+The DSP is the one that matters. It serves **six CSV datasets per academic year,
+all-India, 2018-19 to 2025-26** — profile x2, enrolment x2, facility, teacher. This
+module records the whole route, because every step of it has a trap that returns a
+plausible wrong answer instead of an error.
+
+THE FIVE TRAPS, IN THE ORDER YOU MEET THEM
+==========================================
+
+**1. Egress.** The DSP times out from a non-Indian connection and answers 200 from
+ap-south-1. DNS resolves either way (164.100.211.195). A blanket timeout is NOT an
+outage — re-test from an Indian egress, or tunnel (``ssh -D 1080`` to an Indian box
+and point the client at ``socks5://localhost:1080``).
+
+**2. The API base is not in the page.** It is compiled into a single 2.25 MB Angular
+bundle — there are no lazy-loaded chunks — as ``Y3_apiBaseUrl``, and endpoints are
+assembled with template literals, so grepping for ``"/api/..."`` string literals
+finds almost nothing. What works::
+
+ grep -oE '[A-Za-z0-9_$]+_apiBaseUrl' main.*.js | head -1
+ grep -oE '\\$\\{VAR\\}[A-Za-z0-9_/{}$.-]{2,60}' main.*.js | sort -u
+
+**3. THE ALL-INDIA SENTINEL IS 99, NOT 0 — and 0 fails silently.** This is the
+expensive one. ``stateId=0&districtId=0`` returns HTTP 200 with
+``Content-Type: application/zip`` and a body that is actually ``%PDF-1.7``: the
+schema document, not data. Every ``reportId`` except 1 then 404s, which reads
+convincingly as "only one report exists". With ``stateId=99&districtId=99`` the same
+URL returns the real 30-70 MB zip and reportIds 2-7 all work. Nothing anywhere says
+99. **Always verify the payload's magic bytes are ``PK``, never trust the
+Content-Type header.**
+
+**4. The district select is disabled when All States is chosen.** Driving the form
+with a browser, selecting into it times out. That is correct behaviour, not a bug.
+
+**5. The schema PDF is two documents, not eight.** ``reportId=1`` across the eight
+years yields four identical copies of one file and four of another (verified by
+sha256): one schema for 2018-19..2021-22, another for 2022-23..2025-26.
+
+THE AUTH FLOW
+=============
+Mobile OTP, three calls::
+
+ GET /api/public/captcha -> {captchaKey, captchaImage} (public)
+ POST /api/v1/auth/mobile/send-otp {mobile, captcha, captchaKey}
+ POST /api/v1/auth/mobile/verify-otp {mobile, otp} -> {accessToken, refreshToken}
+
+The captcha image must be read by a human — this module deliberately provides no
+solver. The OTP goes to the account holder's phone and expires quickly, so build the
+verify call BEFORE asking for the code rather than after.
+
+Thereafter ``Authorization: Bearer ``. There is also
+``/api/v1/auth/login`` {mobile, password, captcha, captchaKey} if a password is set,
+and ``/api/v1/auth/refresh-token`` to extend a session mid-pull.
+
+To drive the SPA instead of the API, inject the session into ``sessionStorage`` under
+the keys ``token``, ``refreshToken``, ``user``, ``sessionStart`` and navigate to the
+hash route ``#/CSVdata``. The app is hash-routed: ``/CSVdata`` as a path is a Tomcat
+404, ``#/CSVdata`` is the page.
+
+TERMS — READ BEFORE REDISTRIBUTING ANYTHING
+===========================================
+The portal's Data Sharing Policy, agreed at download time:
+
+ - data shall **not be redistributed to other parties without prior consent**
+ - **source must be acknowledged in all usages**
+ - no unauthorised re-identification of anonymised data
+
+That first clause constrains open-data deposits: derived aggregates and figures are
+fine, republishing the raw rows is not. The schema also confirms the DSP substitutes
+a ``pseudocode`` for the school's real UDISE code, and keys records on village NAME
+rather than village code — so DSP rows answer "which schools are in village X" and
+cannot be joined to a UDISE code without a separate bridge.
+"""
+from __future__ import annotations
+
+import re
+from typing import Any
+
+from .http_client import make_session
+
+__all__ = [
+ "KYS_BASE", "DSP_BASE", "KYS_ENDPOINTS", "DSP_ENDPOINTS",
+ "ALL_INDIA", "YEAR_IDS", "REPORT_IDS", "SCHEMA_REPORT_ID",
+ "csv_url", "request_otp", "verify_otp", "probe_public",
+]
+
+KYS_BASE = "https://kys.udiseplus.gov.in/web-app/api/"
+DSP_BASE = "https://microdata.udiseplus.gov.in/dsp"
+
+#: The all-India sentinel. NOT 0 — see trap 3 in the module docstring.
+ALL_INDIA = 99
+
+#: Academic year -> yearId. From GET /csv-download/years.
+YEAR_IDS: dict[str, int] = {
+ "2018-19": 5, "2019-20": 6, "2020-21": 7, "2021-22": 8,
+ "2022-23": 9, "2023-24": 10, "2024-25": 11, "2025-26": 12,
+}
+
+#: reportId -> the dataset it returns, and the filename stem the server sends.
+#: Verified by Content-Disposition on ranged requests, 2026-08-15.
+REPORT_IDS: dict[int, str] = {
+ 2: "profile_data_2", # RTE and school management
+ 3: "profile_data_1", # basic profile and location
+ 4: "facility_data", # infrastructure and facilities
+ 5: "teacher_data", # teacher and staff academic
+ 6: "enrolment_data_1", # social category and minority
+ 7: "enrolment_data_2", # age-wise enrolment
+}
+#: reportId=1 is the schema PDF, not data — and only two distinct PDFs exist.
+SCHEMA_REPORT_ID = 1
+
+KYS_ENDPOINTS: dict[str, str] = {
+ "years": "getYears",
+ "states": "getStates/{year_id}",
+ "districts": "getDistricts/{state_id}/{year_id}",
+ "blocks": "getBlocks/{district_id}/{year_id}",
+ "managements": "getManagements",
+ "categories": "getCategories",
+ # Served, which is why KYS school SEARCH is gated: the captcha is real. A
+ # school DETAIL page is reachable at /schooldetail/{udise}/{yearId} if the
+ # code is already known, so KYS is a lookup, never a search.
+ "captcha": "getCaptcha",
+}
+
+DSP_ENDPOINTS: dict[str, str] = {
+ "captcha": "/api/public/captcha",
+ "csv_download": "/csv-download",
+ "csv_years": "/csv-download/years",
+ "login": "/api/v1/auth/login",
+ "send_otp": "/api/v1/auth/mobile/send-otp",
+ "verify_otp": "/api/v1/auth/mobile/verify-otp",
+ "refresh_token": "/api/v1/auth/refresh-token",
+ "logout": "/api/v1/logout",
+ # registration form scaffolding, not data
+ "reg_submit": "/api/v1/registration/submit",
+ "reg_otp": "/api/v1/registration/send-reg-otp",
+ "reg_page_data": "/api/v1/registration/page-data",
+ "reg_countries": "/api/v1/registration/countrys",
+ "reg_states": "/api/v1/registration/states/{year_id}",
+ "reg_districts": "/api/v1/registration/districts/{state_id}/{x}",
+ # admin only; 403 for an ordinary account
+ "admin_users": "/api/admin/users",
+}
+
+
+def csv_url(year: str | int, report_id: int, *, base: str = DSP_BASE,
+ state_id: int = ALL_INDIA, district_id: int = ALL_INDIA) -> str:
+ """The download URL for one dataset.
+
+ ``year`` accepts "2018-19" or the raw yearId. Defaults are the all-India
+ sentinel; passing 0 silently returns the schema PDF instead of data.
+ """
+ yid = YEAR_IDS[year] if isinstance(year, str) else year
+ if report_id not in REPORT_IDS and report_id != SCHEMA_REPORT_ID:
+ raise ValueError(f"reportId {report_id} is not one of "
+ f"{sorted(REPORT_IDS)} (data) or {SCHEMA_REPORT_ID} (schema)")
+ return (f"{base}/csv-download?stateId={state_id}&districtId={district_id}"
+ f"&yearId={yid}&reportId={report_id}")
+
+
+def _captcha(base: str, session: Any, timeout: int) -> tuple[str, bytes]:
+ import base64
+ r = session.get(base + DSP_ENDPOINTS["captcha"], timeout=timeout)
+ d = r.json().get("data", {})
+ img = d.get("captchaImage") or d.get("image") or ""
+ img = re.sub(r"^data:image/\w+;base64,", "", img)
+ return d.get("captchaKey"), base64.b64decode(img) if img else b""
+
+
+def request_otp(mobile: str, *, base: str = DSP_BASE, session: Any = None,
+ timeout: int = 60, solve: Any = None) -> dict[str, Any]:
+ """Fetch a captcha and send an OTP to ``mobile``.
+
+ ``solve`` is a callable taking the PNG bytes and returning the characters. It
+ is REQUIRED and has no default: a human reads the captcha. This module does not
+ ship a solver, and adding one here would defeat a control the portal is
+ entitled to have.
+
+ Returns the server's reply plus the captchaKey used, so a caller can retry the
+ same captcha if the OTP send fails for an unrelated reason.
+ """
+ if solve is None:
+ raise ValueError("request_otp needs solve= str>; "
+ "a human must read the captcha")
+ sess = session or make_session()
+ key, png = _captcha(base, sess, timeout)
+ value = solve(png)
+ r = sess.post(base + DSP_ENDPOINTS["send_otp"],
+ json={"mobile": mobile, "captcha": value, "captchaKey": key},
+ timeout=timeout)
+ out = r.json() if r.content else {}
+ out["captchaKey"] = key
+ return out
+
+
+def verify_otp(mobile: str, otp: str, *, base: str = DSP_BASE,
+ session: Any = None, timeout: int = 60) -> dict[str, Any]:
+ """Exchange the OTP for a bearer token.
+
+ OTPs expire in about a minute, so call this immediately. A common own-goal is
+ shell-quoting the mobile so it arrives empty — the portal then answers
+ ``mobile_invalid_strict`` rather than "expired", and the real OTP is burnt by
+ the time the quoting is fixed.
+ """
+ sess = session or make_session()
+ r = sess.post(base + DSP_ENDPOINTS["verify_otp"],
+ json={"mobile": mobile, "otp": otp}, timeout=timeout)
+ return r.json() if r.content else {}
+
+
+def probe_public(base: str = DSP_BASE, *, session: Any = None,
+ timeout: int = 45) -> dict[str, Any]:
+ """Report which endpoints answer without credentials. Reconnaissance only."""
+ sess = session or make_session()
+ out: dict[str, Any] = {"base": base, "results": {}, "note": None}
+ for path in ("/csv-download/years", "/csv-download", "/api/public/captcha"):
+ try:
+ resp = sess.get(base + path, timeout=timeout)
+ out["results"][path] = getattr(resp, "status_code", None)
+ except Exception as exc:
+ out["results"][path] = f"error: {type(exc).__name__}"
+ codes = list(out["results"].values())
+ if all(isinstance(c, str) and c.startswith("error") for c in codes):
+ out["note"] = ("every request failed — almost certainly EGRESS, not an "
+ "outage. Re-test from an Indian egress or a SOCKS tunnel.")
+ elif out["results"].get("/csv-download") == 401:
+ out["note"] = "401 is expected unauthenticated; authenticate with request_otp()."
+ return out
From 8d4b52944b5ccd972c74116513095348836b6d79 Mon Sep 17 00:00:00 2001
From: skishchampi <996985+skishchampi@users.noreply.github.com>
Date: Sun, 16 Aug 2026 22:34:55 -0400
Subject: [PATCH 3/3] refactor(udise): name the module for its mechanism, and
test the sentinel
The module landed as udise.py, one day before the naming sweep. UDISE+ is a
programme of the Ministry of Education. The name told a developer nothing about
what to implement.
It is now otp_download_portal.py. The mechanism is a bulk download gated behind
a mobile OTP. The docstring names the ministry, the programme, both hosts and
the captcha constraint.
No shim. The module has never been released, so no consumer can import the old
name.
It also had no tests. It has eight now, and they cover the trap that costs an
afternoon: the all-India sentinel is 99, and 0 returns a schema PDF inside a
200 response labelled application/zip.
The changelog now records the module, the sentinel, the India-egress
requirement, the Angular-bundle API base and the OTP flow.
---
CHANGELOG.md | 18 ++++++
commoner_probe/geoserver.py | 2 +-
.../{udise.py => otp_download_portal.py} | 13 ++++-
tests/test_otp_download_portal.py | 57 +++++++++++++++++++
4 files changed, 88 insertions(+), 2 deletions(-)
rename commoner_probe/{udise.py => otp_download_portal.py} (94%)
create mode 100644 tests/test_otp_download_portal.py
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 7bcbe5d..242b075 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -158,6 +158,24 @@ that job silently return a fraction of the layer.
### Added
+- **`commoner_probe.otp_download_portal` — bulk microdata from a portal that
+ gates it behind a mobile OTP.** The Ministry of Education's UDISE+ Data
+ Sharing Portal serves six CSV datasets for each academic year since 2018-19.
+ The module records the whole route, because each step has a trap that returns
+ a plausible wrong answer instead of an error.
+
+ **The expensive one is the all-India sentinel: it is 99, not 0.**
+ `stateId=0&districtId=0` answers HTTP 200 with `Content-Type: application/zip`
+ and a body that is actually `%PDF-1.7` — the schema document, not data. Every
+ `reportId` except 1 then 404s, which reads convincingly as "only one report
+ exists". Always check the payload begins `PK`. Never trust the header.
+
+ Also recorded: the portal times out from a non-Indian connection and answers
+ from ap-south-1; the API base sits in a 2.25 MB Angular bundle as
+ `Y3_apiBaseUrl`, assembled with template literals, so grepping for URL string
+ literals finds almost nothing; the auth flow is captcha, send-OTP, verify-OTP,
+ and the captcha needs a human, so this module ships no solver.
+
- **`commoner_probe.geoserver` — point extraction from a WMS-only GeoServer.**
State spatial-data infrastructures are GeoServer deployments and many publish
WMS while disabling WFS, so there is no vector download. This sweeps a bounding
diff --git a/commoner_probe/geoserver.py b/commoner_probe/geoserver.py
index 5838746..1363ebc 100644
--- a/commoner_probe/geoserver.py
+++ b/commoner_probe/geoserver.py
@@ -60,7 +60,7 @@
import re
import time
from dataclasses import dataclass, field
-from typing import Any, Callable, Iterable, Iterator, Sequence
+from typing import Any, Callable, Iterable, Sequence
from urllib.parse import urlencode
from .http_client import make_session
diff --git a/commoner_probe/udise.py b/commoner_probe/otp_download_portal.py
similarity index 94%
rename from commoner_probe/udise.py
rename to commoner_probe/otp_download_portal.py
index 676d8e2..c1ba5ab 100644
--- a/commoner_probe/udise.py
+++ b/commoner_probe/otp_download_portal.py
@@ -1,5 +1,16 @@
# SPDX-License-Identifier: MIT
-"""UDISE+ — the two national school portals, and how to get the microdata out.
+"""Download bulk microdata from a portal that gates it behind a mobile OTP.
+
+CONTEXT
+=======
+The Ministry of Education operates the source. The programme is UDISE+.
+It is the national school-data system of India.
+Two portals serve it. KYS answers hierarchy queries and holds no microdata.
+The Data Sharing Portal holds the microdata and needs an account.
+The hosts are kys.udiseplus.gov.in and microdata.udiseplus.gov.in.
+The portal serves six CSV datasets for each academic year, from 2018-19.
+A human must read the captcha image. This module ships no solver.
+The OTP reaches the phone of the account holder. It expires quickly.
India's school data sits behind two Ministry of Education portals with completely
different contracts:
diff --git a/tests/test_otp_download_portal.py b/tests/test_otp_download_portal.py
new file mode 100644
index 0000000..f44d0e4
--- /dev/null
+++ b/tests/test_otp_download_portal.py
@@ -0,0 +1,57 @@
+"""The all-India sentinel, and the OTP flow's order of operations.
+
+No network. These cover the two things that cost an afternoon each when they
+are wrong, and nothing else.
+"""
+
+from __future__ import annotations
+
+import pytest
+
+from commoner_probe import otp_download_portal as portal
+
+
+class TestTheAllIndiaSentinel:
+ """`stateId=0` returns HTTP 200 with a PDF body, so the wrong sentinel
+ reads as a working download of a dataset that does not exist."""
+
+ def test_the_sentinel_is_99(self):
+ assert portal.ALL_INDIA == 99
+
+ def test_the_default_url_asks_for_all_india(self):
+ url = portal.csv_url("2023-24", 4)
+ assert "stateId=99" in url and "districtId=99" in url
+
+ def test_the_year_string_resolves_to_its_id(self):
+ assert "yearId=10" in portal.csv_url("2023-24", 4)
+ assert "yearId=5" in portal.csv_url("2018-19", 4)
+
+ def test_a_raw_year_id_passes_through(self):
+ assert "yearId=12" in portal.csv_url(12, 4)
+
+ def test_an_unknown_report_id_is_refused(self):
+ """Every reportId except 1 returns 404 under the wrong sentinel, which
+ reads as 'only one report exists'. The caller must not reach that."""
+ with pytest.raises(ValueError, match="reportId"):
+ portal.csv_url("2023-24", 99)
+
+ def test_the_schema_report_is_accepted_but_named_separately(self):
+ assert "reportId=1" in portal.csv_url("2023-24", portal.SCHEMA_REPORT_ID)
+ assert portal.SCHEMA_REPORT_ID not in portal.REPORT_IDS
+
+
+class TestTheOtpFlow:
+ def test_the_verify_call_is_built_before_the_code_is_asked_for(self):
+ """The OTP expires quickly. A caller that builds the second request
+ after reading the code off a phone has already spent the window."""
+ import inspect
+
+ assert "otp" in inspect.signature(portal.verify_otp).parameters
+ assert "mobile" in inspect.signature(portal.request_otp).parameters
+
+ def test_no_captcha_solver_ships(self):
+ """A human reads the image. Shipping a solver would change what this
+ module is, and the account holder's terms with it."""
+ source = (portal.__file__ and open(portal.__file__).read()) or ""
+ for banned in ("pytesseract", "image_to_string", "solve_captcha"):
+ assert banned not in source