ci: denní kontrola API digiarchivu a AMČR OAI

Plánovaný workflow api_monitor.yml (denně 05:17 UTC + ruční spuštění)
ověřuje kontrakt API, na kterém plugin závisí:

- tests/api_contract.py – stejné dotazy jako plugin, kontrola klíčů,
  typů a tvarů odpovědí (OK / DRIFT / FAIL / UNAVAILABLE)
- tests/api_plugin_live.py – vlastní funkce pluginu (fetch_set,
  load_amcr_data) proti živému API v qgis/qgis:ltr
- tests/api_monitor_report.py – jedno sledovací issue se štítkem
  api-monitor; čistý běh ho zavře, výpadek ho nemění

Neběží na PR, aby výpadek digiarchivu neshodil PR. Popis v AGENTS.md,
OpenSpec změna add-daily-api-monitor.

Připraveno s pomocí AI (Claude), ověřeno proti produkčnímu API.
This commit is contained in:
david-spacil committed 2026-10-02 22:17:01 +02:00
1 parent 7c0401c11b
commit dde76203b2
10 files changed
+2140

No files matched your search

+904
View File
@@ -0,0 +1,904 @@
# -*- coding: utf-8 -*-
"""
API contract test – checks that the live digiarchiv / AMCR OAI API still
answers the way the plugin reads it. Plain requests, no QGIS.
The plugin under test is amcr_viewer/ on this branch; the expectations here
describe what its parsers (amcr_tools.g/g_list, amcr_codelists._facet_name)
actually consume, not the official API documentation.
Status model (per check):
OK – the answer matches the recorded expectation
DRIFT – the answer differs, but the plugin tolerates the new shape
FAIL – the answer differs in a way the plugin does not tolerate
UNAVAILABLE – network error / timeout / HTTP 5xx after retries
Exit code is 1 when any check is FAIL or DRIFT, 0 otherwise (a run where
everything is UNAVAILABLE is green but visible in the summary).
Outage fast-fail: once a host is unreachable after the full retry cycle,
every later request to that host returns UNAVAILABLE immediately (circuit
breaker, no network I/O) – an all-unreachable run finishes in seconds.
Run (from the repository root, outside the repo use uv --no-project so no
uv.lock appears):
uv run -q --no-project --with requests==2.34.2 \\
python tests/api_contract.py
Outputs:
* stdout: a Markdown table of all checks
* results-api_contract.json next to the script (cwd) with one entry per
check: name, status, detail, request URL
* the same table appended to $GITHUB_STEP_SUMMARY when set
Test area (probe 2026-10-02, anonymous = pristupnost A only):
TEST_BBOX (Mikulov, south Moravia) 48.8,16.6,48.9,16.75
akce 185, lokalita 18, samostatny_nalez 2, pian 294
PAGINATION_BBOX (Praha) 49.9,14.3,50.2,14.7 – akce 20 635 records,
paginated with rows=100; only akce is paginated here, the other entities
have few enough records in the small bbox.
Env overrides (for the outage simulation):
AMCR_DA_URL base URL of digiarchiv (default
https://digiarchiv.aiscr.cz)
AMCR_OAI_URL base URL of the AMCR OAI endpoint (default
https://api.aiscr.cz/2.2/oai)
AMCR_TIMEOUT per-request timeout in seconds (default 30)
"""
import json
import os
import re
import sys
import time
import urllib.parse
import xml.etree.ElementTree as ET # nosec B405
import requests
JOB = "api_contract"
DA_URL = os.environ.get("AMCR_DA_URL", "https://digiarchiv.aiscr.cz")
OAI_URL = os.environ.get("AMCR_OAI_URL", "https://api.aiscr.cz/2.2/oai")
TIMEOUT = int(os.environ.get("AMCR_TIMEOUT", "30"))
# Small test area chosen by probe (see module docstring). Filter values are
# taken from live facets of this same bbox in the same run, never from the
# bundled codelists.
TEST_BBOX = "48.8,16.6,48.9,16.75"
PAGINATION_BBOX = "49.9,14.3,50.2,14.7"
# Entities the plugin downloads (typ_dat_vocab in amcr_tools.py).
ENTITIES = ["akce", "lokalita", "samostatny_nalez"]
# OAI sets, mirroring amcr_codelists.slovnicek (name -> OAI set).
OAI_SETS = {
"obdobi": "heslo:obdobi",
"typ_akce": "heslo:akce_typ",
"areal": "heslo:areal",
"kraj": "ruian_kraj",
"organizace": "organizace",
"okres": "ruian_okres",
"katastr": "ruian_katastr",
"pian_presnost": "heslo:pian_presnost",
"typ_lokality": "heslo:lokalita_typ",
"druh_lokality": "heslo:lokalita_druh",
"jistota": "heslo:jistota_urceni",
"lokalita_zachovalost": "heslo:stav_dochovani",
"pristupnost": "heslo:pristupnost",
"nalez_kategorie": "heslo:predmet_druh_kat",
"druh_nalezu": "heslo:predmet_druh",
"specifikace": "heslo:predmet_specifikace",
"nalezove_okolnosti": "heslo:nalezove_okolnosti",
}
# Facet-backed codelists (name -> (entity, facet field)), mirroring
# amcr_codelists.slovnicek.
FACET_SETS = {
"vedouci": ("akce", "f_vedouci"),
"nalezce": ("samostatny_nalez", "f_nalezce"),
}
# Expected facet item shape, as read by amcr_codelists._facet_name:
# list form [str, int] is the current API (Solr 10 / digiarchiv v4.1.0,
# json.nl=arrarr); the object form {"name": str, ...} is the old API
# (pre v4.1.0) which the plugin still tolerates -> DRIFT, not FAIL.
# Any other shape (scalar, empty list item, dict without "name") would
# break the plugin -> FAIL.
FACET_ITEM_FORMS = [
("list", lambda x: isinstance(x, list) and len(x) == 2
and isinstance(x[0], str) and isinstance(x[1], int)),
("object", lambda x: isinstance(x, dict)
and isinstance(x.get("name"), str)),
]
NS = {
"oai": "http://www.openarchives.org/OAI/2.0/",
"dc": "http://purl.org/dc/elements/1.1/",
"oai_dc": "http://www.openarchives.org/OAI/2.0/oai_dc/",
}
SESSION = requests.Session()
SESSION.headers.update({"User-Agent": "amcr-viewer-api-monitor/1.0"})
RESULTS = []
DEPLOYED_VERSION = None
RETRY_BACKOFF = (2, 8) # seconds, after 1st and 2nd attempt
# Circuit breaker (fast outage): netlocs that came back unreachable after
# the full retry cycle. Every later request to such a host returns None
# immediately, without network I/O – an all-unreachable run then takes
# seconds instead of tens of minutes of per-check retries.
DEAD_HOSTS = set()
def _retry_get(url, params=None):
"""GET with retries on network errors, timeouts and HTTP 5xx.
Returns the response, or None when unreachable after all attempts.
HTTP 4xx and error bodies with status 200 are real answers –
the caller judges them by the contract.
Once a host (netloc) is found unreachable after full retries, it is
added to DEAD_HOSTS and every later request to it returns None
without touching the network (see the module docstring).
"""
netloc = urllib.parse.urlparse(url).netloc
if netloc in DEAD_HOSTS:
return None
for attempt in range(3):
try:
resp = SESSION.get(url, params=params, timeout=TIMEOUT)
if resp.status_code < 500:
return resp
except requests.exceptions.RequestException:
pass
if attempt < 2:
time.sleep(RETRY_BACKOFF[attempt])
DEAD_HOSTS.add(netloc)
return None
def record(name, status, detail, url=""):
RESULTS.append({
"name": name,
"status": status,
"detail": detail,
"url": url,
})
print(f" {status:<12} {name} – {detail}")
def get_json(url, params=None):
"""GET + JSON parse with a friendly error, or None when unreachable."""
resp = _retry_get(url, params)
if resp is None:
return None
try:
return resp.json()
except ValueError:
return {"_invalid_json": True, "_status": resp.status_code}
def load_deployed_version():
"""Reads the deployed digiarchiv version from the web bundle.
Missing version is a DRIFT of its own check, never a FAIL.
"""
global DEPLOYED_VERSION
resp = _retry_get(DA_URL + "/home")
if resp is None:
record("deployed-version", "UNAVAILABLE",
f"{DA_URL}/home unreachable")
return False
scripts = re.findall(r'(?:src|href)="([^"]*\.js)"', resp.text)
version = None
for script in scripts:
jresp = _retry_get(DA_URL + "/" + script.lstrip("/"))
if jresp is None:
continue
match = re.search(r'raw:"(v\d[^"]*)"', jresp.text)
if match:
version = match.group(1)
break
if version:
DEPLOYED_VERSION = version
record("deployed-version", "OK", f"{version}")
return True
record("deployed-version", "DRIFT",
"git-describe string raw:\"v…\" not found in the web bundle "
f"({len(scripts)} scripts scanned)")
return False
def check_translations():
url = DA_URL + "/api/assets/i18n/cs.json"
data = get_json(url)
if data is None:
record("i18n cs.json", "UNAVAILABLE", "unreachable")
return
if not isinstance(data, dict) or not data:
record("i18n cs.json", "FAIL",
f"expected a non-empty dict, got {type(data).__name__}",
url)
return
# A few codes the plugin translates via tr_code() in live records
sample = [k for k in data if k.startswith("HES-")]
if not sample:
record("i18n cs.json", "DRIFT",
"no HES-* keys found – tr_code would return codes verbatim",
url)
return
record("i18n cs.json", "OK",
f"{len(data)} keys, {len(sample)} HES-* codes", url)
def _status_of_field(types, good, drift=None):
"""OK/DRIFT/FAIL for a set of observed field types."""
bad = types - good
if not bad:
return "OK"
if drift and bad <= drift:
return "DRIFT"
return "FAIL"
def _check_doc_fields(name, docs, fields, url):
"""Checks per-doc field presence and value types the plugin reads.
fields: {key: (ok_types, drift_types)}
Value normalization: amcr_tools.g() reads doc.get(key) and str()'s it –
lists are read as first item. g_list() iterates the value. So both a
scalar and a list of scalars are consumed; dict values are read with
.get() by dedicated code paths.
"""
for key, (good, drift) in fields.items():
types = set()
missing = 0
for doc in docs:
if key not in doc or doc[key] is None:
missing += 1
else:
v = doc[key]
if isinstance(v, list):
for item in v:
types.add(type(item).__name__)
else:
types.add(type(v).__name__)
if missing == len(docs):
record(f"{name} {key}", "FAIL",
f"missing in all {len(docs)} docs", url)
continue
status = _status_of_field(types, good, drift)
detail = (f"types {sorted(types)}, "
f"{missing}/{len(docs)} docs without the key")
record(f"{name} {key}", status, detail, url)
def fetch_entity_docs(entity, bbox, rows=500):
"""Main query exactly the way the plugin sends it. None = unavailable."""
params = {
"mapa": "true",
"sort": "ident_cely asc",
"entity": entity,
"rows": rows,
"loc_rpt": bbox,
}
url = DA_URL + "/api/search/query"
data = get_json(url, params)
if data is None:
return None, None
if "response" not in data:
return {}, data
return data["response"], data
def check_main_queries():
"""Main query per entity: keys and value types the plugin reads."""
strset = {"str"}
specs = {
"akce": {
"ident_cely": (strset, None),
"loc": (strset, None), # g_list -> list of str
"pristupnost": (strset, None),
"az_okres": (strset, None),
"katastr": (strset, None),
"akce_hlavni_vedouci": (strset, None),
"akce_organizace": (strset, None),
"akce_specifikace_data": (strset, None),
"akce_datum_zahajeni": (strset, None),
"akce_datum_ukonceni": (strset, None),
"akce_hlavni_typ": (strset, None),
"akce_vedlejsi_typ": (strset, None),
"akce_je_nz": ({"bool"}, None),
"akce_projekt": (strset, None),
"az_dj_pian": (strset, None),
"az_chranene_udaje": ({"dict"}, None),
"akce_chranene_udaje": ({"dict"}, None),
"az_dokumentacni_jednotka": ({"dict"}, None),
},
"lokalita": {
"ident_cely": (strset, None),
"loc": (strset, None),
"pristupnost": (strset, None),
"az_okres": (strset, None),
"katastr": (strset, None),
"az_dj_pian": (strset, None),
"az_chranene_udaje": ({"dict"}, None),
"lokalita_chranene_udaje": ({"dict"}, None),
"lokalita_druh": (strset, None),
"lokalita_typ_lokality": (strset, None),
"lokalita_zachovalost": (strset, None),
"az_dokumentacni_jednotka": ({"dict"}, None),
},
"samostatny_nalez": {
"ident_cely": (strset, None),
"loc": (strset, None),
"pristupnost": (strset, None),
"samostatny_nalez_nalezce": (strset, None),
"samostatny_nalez_hloubka": ({"int", "float", "str"}, None),
"samostatny_nalez_okres": (strset, None),
"samostatny_nalez_chranene_udaje": ({"dict"}, None),
"samostatny_nalez_druh_nalezu": (strset, None),
"samostatny_nalez_obdobi": (strset, None),
"samostatny_nalez_specifikace": (strset, None),
"samostatny_nalez_datum_nalezu": (strset, None),
"samostatny_nalez_pocet": ({"str", "int", "float"}, None),
},
}
docs_by_entity = {}
for entity in ENTITIES:
resp, raw = fetch_entity_docs(entity, TEST_BBOX)
url = DA_URL + "/api/search/query"
if resp is None:
record(f"query {entity}", "UNAVAILABLE", "unreachable", url)
continue
if "numFound" not in resp and "docs" not in resp:
record(f"query {entity}", "FAIL",
f"no response block: {json.dumps(raw)[:200]}", url)
continue
num_found = resp.get("numFound")
if not isinstance(num_found, int):
record(f"query {entity} numFound", "FAIL",
f"expected int, got {type(num_found).__name__}", url)
continue
docs = resp.get("docs", [])
if not docs:
record(f"query {entity}", "FAIL",
f"0 docs for the test bbox (numFound={num_found}) – "
"the test area has no records", url)
continue
record(f"query {entity}", "OK",
f"numFound {num_found}, {len(docs)} docs", url)
docs_by_entity[entity] = docs
_check_doc_fields(f"{entity}", docs, specs[entity], url)
return docs_by_entity
def check_numfound_int():
url = DA_URL + "/api/search/query"
for entity in ENTITIES:
resp, _ = fetch_entity_docs(entity, TEST_BBOX, rows=0)
if resp is None:
record(f"numFound {entity}", "UNAVAILABLE", "unreachable", url)
continue
if isinstance(resp.get("numFound"), int):
record(f"numFound {entity}", "OK", f"{resp['numFound']}", url)
else:
record(f"numFound {entity}", "FAIL",
f"expected int, got {type(resp.get('numFound')).__name__}",
url)
def fetch_facets(entity, bbox=None):
"""Facet request exactly the way amcr_codelists.fetch_set sends it."""
params = {
"entity": entity,
"rows": 0,
"noFacets": "false",
"onlyFacets": "true",
}
if bbox:
params["loc_rpt"] = bbox
url = DA_URL + "/api/search/query"
data = get_json(url, params)
if data is None:
return None
try:
return data["facet_counts"]["facet_fields"]
except (KeyError, TypeError):
return {}
def check_facet_sets():
"""Facet-backed codelists vedouci/nalezce: field exists, item shape."""
for name, (entity, field) in FACET_SETS.items():
url = DA_URL + "/api/search/query"
ff = fetch_facets(entity)
if ff is None:
record(f"facet {name}", "UNAVAILABLE", "unreachable", url)
continue
if field not in ff:
record(f"facet {name}", "FAIL",
f"facet field {field} missing from entity {entity}", url)
continue
items = ff[field]
if not isinstance(items, list):
record(f"facet {name}", "FAIL",
f"expected a list of items, got {type(items).__name__}",
url)
continue
if not items:
record(f"facet {name}", "FAIL",
f"facet field {field} came back empty", url)
continue
# classify each item's shape
bad = []
shapes = set()
for item in items:
for shape, test in FACET_ITEM_FORMS:
if test(item):
shapes.add(shape)
break
else:
bad.append(item)
if bad:
record(f"facet {name}", "FAIL",
f"{len(bad)}/{len(items)} items in an unknown shape, "
f"e.g. {json.dumps(bad[0])[:120]}", url)
elif shapes == {"list"}:
record(f"facet {name}", "OK",
f"{len(items)} items, shape [value, count]", url)
elif shapes == {"object"}:
record(f"facet {name}", "DRIFT",
f"{len(items)} items in the OLD object shape "
'{"name": …} – the plugin still tolerates it via '
"_facet_name, but this is a Solr json.nl change; "
"expectation recorded: [value, count]", url)
else:
record(f"facet {name}", "DRIFT",
f"mixed shapes {sorted(shapes)}", url)
def check_oai_sets():
"""Every OAI set in slovnicek: first page + resumptionToken paging."""
url = OAI_URL
for name, oai_set in OAI_SETS.items():
resp = _retry_get(url, params={
"verb": "ListRecords",
"metadataPrefix": "oai_dc",
"set": oai_set,
})
if resp is None:
record(f"oai {name}", "UNAVAILABLE", "unreachable", url)
continue
try:
root = ET.fromstring(resp.content) # nosec B405 B314
except ET.ParseError as e:
record(f"oai {name}", "FAIL", f"XML parse error: {e}", url)
continue
error = root.find(".//oai:error", NS)
if error is not None:
record(f"oai {name}", "FAIL",
f"OAI error {error.get('code')}: "
f"{(error.text or '')[:100]}", url)
continue
records = root.findall(".//oai:record", NS)
if not records:
record(f"oai {name}", "FAIL",
f"set {oai_set} returned no records", url)
continue
# record shape: identifier, titles, dc payload
ok_shape = all(
r.find(".//oai_dc:dc", NS) is not None
and r.find(".//dc:identifier", NS) is not None
for r in records
)
if not ok_shape:
record(f"oai {name}", "FAIL",
"record missing oai_dc:dc or dc:identifier", url)
continue
token = root.find(".//oai:resumptionToken", NS)
token_ok = True
if token is not None and token.text:
# follow one page of the resumption token, the way fetch_set
# does; a broken token means an incomplete codelist
resp2 = _retry_get(url, params={
"verb": "ListRecords",
"resumptionToken": token.text,
})
if resp2 is None:
record(f"oai {name}", "UNAVAILABLE",
"first page OK, token page unreachable", url)
continue
try:
root2 = ET.fromstring(resp2.content) # nosec B405 B314
except ET.ParseError as e:
record(f"oai {name}", "FAIL",
f"token page XML parse error: {e}", url)
continue
recs2 = root2.findall(".//oai:record", NS)
if not recs2:
token_ok = False
time.sleep(0.5) # the plugin pauses between OAI pages
if token_ok:
desc = f"{len(records)} records"
if token is not None and token.text:
desc += ", token page followed"
record(f"oai {name}", "OK", desc, url)
else:
record(f"oai {name}", "FAIL",
"resumptionToken page returned no records", url)
def check_pagination():
"""rows=100 pages over a larger area must not overlap."""
url = DA_URL + "/api/search/query"
seen = []
total = None
page = 0
while True:
params = {
"mapa": "true",
"sort": "ident_cely asc",
"entity": "akce",
"rows": 100,
"loc_rpt": PAGINATION_BBOX,
}
if page > 0:
params["page"] = page
data = get_json(url, params)
if data is None:
record("pagination akce", "UNAVAILABLE", "unreachable", url)
return
if "response" not in data:
record("pagination akce", "FAIL",
f"error body on page {page}", url)
return
resp = data["response"]
if total is None:
total = resp.get("numFound")
if not isinstance(total, int):
record("pagination akce", "FAIL",
"numFound is not an int", url)
return
docs = resp.get("docs", [])
if not docs:
break
seen.extend([d.get("ident_cely") for d in docs])
if len(seen) >= total:
break
page += 1
if page > 220: # safety stop
break
unique = set(seen)
if len(unique) != len(seen):
dupes = len(seen) - len(unique)
record("pagination akce", "FAIL",
f"{dupes} duplicate ids across {page + 1} pages "
f"({len(seen)} ids)", url)
elif len(unique) < total:
record("pagination akce", "FAIL",
f"downloaded {len(unique)} of numFound {total}", url)
else:
record("pagination akce", "OK",
f"{len(unique)} unique ids across {page + 1} pages, "
f"numFound {total}", url)
def check_bbox_restriction():
"""loc_rpt must actually restrict: bbox count << global count."""
url = DA_URL + "/api/search/query"
for entity in ENTITIES:
resp, _ = fetch_entity_docs(entity, TEST_BBOX, rows=0)
if resp is None:
record(f"bbox {entity}", "UNAVAILABLE", "unreachable", url)
continue
global_resp = get_json(url, params={
"mapa": "true", "sort": "ident_cely asc", "entity": entity,
"rows": 0,
})
if global_resp is None or "response" not in global_resp:
record(f"bbox {entity}", "UNAVAILABLE", "global query failed",
url)
continue
n_bbox = resp.get("numFound")
n_all = global_resp["response"].get("numFound")
if not isinstance(n_bbox, int) or not isinstance(n_all, int):
record(f"bbox {entity}", "FAIL", "numFound not int", url)
continue
if n_bbox >= n_all:
record(f"bbox {entity}", "FAIL",
f"loc_rpt did not restrict: {n_bbox} vs {n_all} global",
url)
else:
record(f"bbox {entity}", "OK",
f"{n_bbox} in bbox vs {n_all} global", url)
def check_pian_batch(docs_by_entity):
"""PIAN batch geometry query, exactly the way load_amcr_data sends it."""
url = DA_URL + "/api/search/query"
if "akce" not in docs_by_entity:
record("pian-batch", "UNAVAILABLE",
"depends on the akce query, which is unavailable", url)
return
pian_ids = []
for doc in docs_by_entity["akce"]:
for dj in doc.get("az_dokumentacni_jednotka") or []:
dj_pian = dj.get("dj_pian") or {}
if dj_pian.get("id"):
pian_ids.append(dj_pian["id"])
if not pian_ids:
record("pian-batch", "FAIL",
"no dj_pian ids found in the akce docs", url)
return
batch = pian_ids[:50] # small on purpose
fq = "ident_cely:(" + " OR ".join(batch) + ")"
data = get_json(url, params={
"mapa": "true",
"entity": "pian",
"q": fq,
"rows": len(batch),
"fl": "ident_cely,pian_typ,pian_chranene_udaje,pian_presnost",
})
if data is None:
record("pian-batch", "UNAVAILABLE", "unreachable", url)
return
if "response" not in data:
record("pian-batch", "FAIL",
f"error body: {json.dumps(data)[:200]}", url)
return
docs = data["response"].get("docs", [])
if not docs:
record("pian-batch", "FAIL", "0 docs for a known PIAN id batch", url)
return
with_wkt = 0
for d in docs:
raw = d.get("pian_chranene_udaje")
if isinstance(raw, list) and raw:
raw = raw[0]
jdata = (json.loads(raw) if isinstance(raw, str) else (raw or {}))
if isinstance(jdata, dict) and (
jdata.get("geom_sjtsk_wkt") or jdata.get("geom_wkt")
):
with_wkt += 1
if with_wkt == len(docs):
record("pian-batch", "OK",
f"{len(docs)} PIAN docs, all with WKT geometry", url)
elif with_wkt:
record("pian-batch", "DRIFT",
f"{with_wkt}/{len(docs)} PIAN docs with WKT – "
"records without geometry are skipped by the plugin", url)
else:
record("pian-batch", "FAIL",
"no geom_sjtsk_wkt / geom_wkt in pian_chranene_udaje", url)
def check_filters(docs_by_entity):
"""Every filter key the dialog builds, values from live facets."""
url = DA_URL + "/api/search/query"
# (entity, filter key, facet field it draws its value from)
plan = [
("akce", "f_kraj"), ("akce", "f_okres"), ("akce", "f_katastr"),
("akce", "f_obdobi"), ("akce", "f_areal"),
("akce", "f_pian_presnost"), ("akce", "f_typ_vyzkumu"),
("akce", "f_vedouci"), ("akce", "f_organizace"),
("lokalita", "f_typ_lokality"), ("lokalita", "f_druh_lokality"),
("lokalita", "f_jistota"), ("lokalita", "f_lokalita_zachovalost"),
("samostatny_nalez", "f_kategorie"),
("samostatny_nalez", "f_druh_nalezu"),
("samostatny_nalez", "f_specifikace"),
("samostatny_nalez", "f_nalezove_okolnosti"),
("samostatny_nalez", "f_nalezce"),
]
for entity, key in plan:
ff = fetch_facets(entity, bbox=TEST_BBOX)
if ff is None:
record(f"filter {entity}.{key}", "UNAVAILABLE", "unreachable",
url)
continue
items = ff.get(key) or []
if not items:
record(f"filter {entity}.{key}", "FAIL",
f"no facet values for {key} in the test bbox", url)
continue
item = items[0]
value = item[0] if isinstance(item, list) else item.get("name")
if not value:
record(f"filter {entity}.{key}", "FAIL",
f"facet item for {key} has no value", url)
continue
params = {
"mapa": "true",
"sort": "ident_cely asc",
"entity": entity,
"rows": 1,
"loc_rpt": TEST_BBOX,
key: [f"{value}:or"],
}
data = get_json(url, params)
if data is None:
record(f"filter {entity}.{key}", "UNAVAILABLE", "unreachable",
url)
continue
if "response" not in data:
record(f"filter {entity}.{key}", "FAIL",
f"API error for value {value!r}: "
f"{json.dumps(data)[:150]}", url)
continue
num = data["response"].get("numFound")
record(f"filter {entity}.{key}", "OK",
f"value {value!r} accepted, numFound {num}", url)
def check_date_ranges():
"""Date range filter, sent the way the dialog builds it."""
url = DA_URL + "/api/search/query"
plan = [
("akce", "akce_datum_zahajeni"),
("akce", "akce_datum_ukonceni"),
("samostatny_nalez", "samostatny_nalez_datum_nalezu"),
]
for entity, field in plan:
params = {
"mapa": "true", "sort": "ident_cely asc", "entity": entity,
"rows": 1, "loc_rpt": TEST_BBOX,
field: "1900-01-01,2030-12-31",
}
data = get_json(url, params)
if data is None:
record(f"date {entity}.{field}", "UNAVAILABLE", "unreachable",
url)
continue
if "response" not in data:
record(f"date {entity}.{field}", "FAIL",
f"error body: {json.dumps(data)[:150]}", url)
continue
record(f"date {entity}.{field}", "OK",
f"numFound {data['response'].get('numFound')}", url)
def check_special_params():
"""pristupnost, posevidence, proj_akce – sent as the dialog sends."""
url = DA_URL + "/api/search/query"
plan = [
("akce", {"pristupnost": ["A:or"]}),
("akce", {"posevidence": "true"}),
("akce", {"proj_akce": "true"}),
]
for entity, extra in plan:
key = list(extra)[0]
params = {
"mapa": "true", "sort": "ident_cely asc", "entity": entity,
"rows": 1, "loc_rpt": TEST_BBOX, **extra
}
data = get_json(url, params)
if data is None:
record(f"param {entity}.{key}", "UNAVAILABLE", "unreachable",
url)
continue
if "response" not in data:
record(f"param {entity}.{key}", "FAIL",
f"error body: {json.dumps(data)[:150]}", url)
continue
record(f"param {entity}.{key}", "OK",
f"numFound {data['response'].get('numFound')}", url)
def check_error_answers():
"""Invalid parameter and unknown entity must be an error body, not
an empty result – the plugin reads the absence of 'response'."""
url = DA_URL + "/api/search/query"
data = get_json(url, params={
"entity": "akce", "akce_datum_zahajeni": "notadate"})
if data is None:
record("error invalid-parameter", "UNAVAILABLE", "unreachable", url)
elif isinstance(data, dict) and data.get("error"):
record("error invalid-parameter", "OK",
f"error body: {str(data['error'])[:100]}", url)
elif isinstance(data, dict) and "response" in data:
record("error invalid-parameter", "FAIL",
"invalid date was accepted as a normal response", url)
else:
record("error invalid-parameter", "FAIL",
f"unexpected body: {json.dumps(data)[:150]}", url)
data = get_json(url, params={"entity": "neexistujici_entity", "rows": 1})
if data is None:
record("error unknown-entity", "UNAVAILABLE", "unreachable", url)
elif isinstance(data, dict) and data.get("error"):
record("error unknown-entity", "OK",
f"error body: {str(data['error'])[:100]}", url)
elif isinstance(data, dict) and "response" in data:
record("error unknown-entity", "FAIL",
"unknown entity was accepted as a normal response", url)
else:
record("error unknown-entity", "FAIL",
f"unexpected body: {json.dumps(data)[:150]}", url)
def check_login_error_path():
"""login_to_api with wrong credentials: session None, error 'auth'."""
url = DA_URL + "/api/user/login"
# Deliberately wrong, obviously fake credentials – nothing secret.
wrong_user = "test@example.invalid"
wrong_login_value = "neutron-failure-horse-battery"
try:
resp = SESSION.post(
url,
json={"user": wrong_user, "pwd": wrong_login_value},
timeout=TIMEOUT,
)
except requests.exceptions.RequestException as e:
record("login wrong-credentials", "UNAVAILABLE",
f"{type(e).__name__}: {e}", url)
return
if resp.status_code >= 500:
record("login wrong-credentials", "UNAVAILABLE",
f"HTTP {resp.status_code}", url)
return
try:
body = resp.json()
except ValueError:
record("login wrong-credentials", "FAIL",
f"non-JSON body (HTTP {resp.status_code})", url)
return
if resp.status_code == 200 and body.get("error"):
record("login wrong-credentials", "OK",
f"error body: {str(body['error'])[:80]}", url)
elif resp.status_code in (401, 403):
record("login wrong-credentials", "OK",
f"HTTP {resp.status_code}", url)
else:
record("login wrong-credentials", "FAIL",
f"wrong credentials accepted (HTTP {resp.status_code}, "
f"body {json.dumps(body)[:120]})", url)
def write_outputs():
"""results JSON + Markdown table to stdout and GITHUB_STEP_SUMMARY."""
path = f"results-{JOB}.json"
with open(path, "w", encoding="utf-8") as f:
json.dump(RESULTS, f, ensure_ascii=False, indent=2)
lines = ["| check | status | detail |", "|---|---|---|"]
for r in RESULTS:
detail = str(r["detail"]).replace("|", "\\|").replace("\n", " ")
lines.append(f"| {r['name']} | {r['status']} | {detail} |")
table = "\n".join(lines)
print()
print(table)
summary = os.environ.get("GITHUB_STEP_SUMMARY")
if summary:
with open(summary, "a", encoding="utf-8") as f:
f.write(table + "\n")
print()
print(f"written: {path}")
bad = [r for r in RESULTS if r["status"] in ("FAIL", "DRIFT")]
print(f"checks: {len(RESULTS)}, FAIL/DRIFT: {len(bad)}")
return 1 if bad else 0
def main():
print(f"API contract test against {DA_URL}")
print(f"test bbox: {TEST_BBOX} (Mikulov)")
load_deployed_version()
check_translations()
docs_by_entity = check_main_queries()
check_numfound_int()
check_facet_sets()
check_oai_sets()
check_pagination()
check_bbox_restriction()
check_pian_batch(docs_by_entity)
check_filters(docs_by_entity)
check_date_ranges()
check_special_params()
check_error_answers()
check_login_error_path()
return write_outputs()
if __name__ == "__main__":
sys.exit(main())
+245
View File
@@ -0,0 +1,245 @@
# -*- coding: utf-8 -*-
"""
Issue reporting for the daily API monitor – turns the result JSONs of both
test jobs into one tracking issue (label api-monitor) via gh.
Cases (design decision 8):
* any FAIL/DRIFT, no open issue yet -> create it
* open issue, changed fingerprint (sorted names of non-OK checks,
stored in an HTML comment in the issue body) -> comment + update body
* open issue, identical fingerprint -> do nothing
* every check OK, open issue -> close it with a comment
* no FAIL/DRIFT but some UNAVAILABLE -> leave the issue as it is
* a job without its result file counts as one FAIL named after the job
Runs only for schedule/workflow_dispatch on the default branch (the
workflow guards it, the script trusts its inputs). Issue text is Czech,
not hard-wrapped (GitHub GFM re-flows anyway).
Dry run: API_MONITOR_DRY_RUN=1 prints the gh commands instead of
executing them, and does not touch GitHub. Existing-issue input for
local testing: API_MONITOR_EXISTING_ISSUE=<number> (simulate an open
issue; the search is skipped).
Usage inside the reporting job:
python3 tests/api_monitor_report.py <results-dir> <run-url>
The deployed digiarchiv version is read from the deployed-version check
in the contract results.
Exit code: 0 always – a reporting problem must not mask the test results
(the job's conclusion is already decided by the test jobs).
"""
import json
import os
import subprocess # nosec B404 - volá jen gh se seznamem argumentů
import sys
LABEL = "api-monitor"
FINGERPRINT_MARK = "<!-- api-monitor-fingerprint"
RESULTS_DIR = sys.argv[1] if len(sys.argv) > 1 else "results"
RUN_URL = sys.argv[2] if len(sys.argv) > 2 else ""
# Filled in main() from the deployed-version check of the contract test
VERSION = ""
DRY = os.environ.get("API_MONITOR_DRY_RUN") == "1"
EXISTING_ISSUE = os.environ.get("API_MONITOR_EXISTING_ISSUE", "")
# Test hook: stands in for the body of the existing issue, so the
# identical-fingerprint case can be exercised without GitHub
EXISTING_BODY = os.environ.get("API_MONITOR_EXISTING_BODY", "")
JOBS = ["api_contract", "plugin_live"]
def gh(args, input_text=None):
"""Runs gh (or prints the command in a dry run)."""
cmd = ["gh"] + args
if DRY:
shown = " ".join(cmd)
if input_text:
shown += f" <<'EOF'\n{input_text}EOF"
print(f"[dry-run] {shown}")
return None
return subprocess.run( # nosec B603 B607
cmd, input=input_text, text=True, capture_output=True, check=False
)
def load_checks():
"""One list of check dicts; a missing result file is a FAIL."""
checks = []
missing = []
for job in JOBS:
path = os.path.join(RESULTS_DIR, f"results-{job}.json")
if not os.path.exists(path):
missing.append(job)
continue
try:
with open(path, encoding="utf-8") as f:
data = json.load(f)
checks.extend(data)
except (OSError, ValueError) as e:
missing.append(job)
print(f"cannot read {path}: {e}")
for job in missing:
# A crashed script or an image pull failure – the run is broken
# even though no check had the chance to fail
checks.append({
"name": f"job {job}",
"status": "FAIL",
"detail": "job did not produce its result file",
"url": "",
})
return checks
def find_open_issue():
"""Number of the open issue with the api-monitor label, or None."""
if EXISTING_ISSUE:
return EXISTING_ISSUE
res = gh(["issue", "list", "--label", LABEL, "--state", "open",
"--json", "number", "--limit", "1"])
if res is None:
return None
if res.returncode != 0:
print(f"issue search failed: {res.stderr}")
return None
try:
issues = json.loads(res.stdout)
except ValueError:
return None
return str(issues[0]["number"]) if issues else None
def fingerprint(checks):
"""Sorted names of the FAIL/DRIFT checks – the issue identity.
UNAVAILABLE is left out on purpose: a flaky endpoint next to a real
break would otherwise change the fingerprint and add a comment on
every run.
"""
return ",".join(sorted(
c["name"] for c in checks if c["status"] in ("FAIL", "DRIFT")))
def issue_body(checks):
"""Czech GFM body with the fingerprint hidden in an HTML comment."""
lines = []
lines.append("Denní kontrola API (`api_monitor.yml`) našla problémy.")
lines.append("")
if VERSION and VERSION != "none":
lines.append(f"Nasazená verze digiarchivu: **{VERSION}**")
lines.append("")
lines.append("| kontrola | stav | detail |")
lines.append("|---|---|---|")
for c in checks:
if c["status"] == "OK":
continue
detail = str(c.get("detail", "")).replace("|", "\\|")
lines.append(f"| {c['name']} | {c['status']} | {detail} |")
if RUN_URL:
lines.append("")
lines.append(f"Běh: {RUN_URL}")
lines.append("")
lines.append("Lokální reprodukce:")
lines.append("")
lines.append("```sh")
lines.append("uv run -q --no-project --with requests==2.34.2 "
"python tests/api_contract.py")
lines.append("")
lines.append("docker run --rm -v \"$PWD:/work:ro\" -w /work "
"--user \"$(id -u):$(id -g)\" -e HOME=/tmp "
"-e AMCR_RESULTS_DIR=/tmp/results "
"qgis/qgis:ltr python3 tests/api_plugin_live.py")
lines.append("```")
lines.append("")
# Fingerprint must stay the last line – it is read back as-is
lines.append(f"{FINGERPRINT_MARK}: {fingerprint(checks)} -->")
return "\n".join(lines)
def read_fingerprint(body):
"""Extracts the fingerprint from an issue body, or None."""
for line in body.splitlines():
line = line.strip()
if line.startswith(FINGERPRINT_MARK):
rest = line[len(FINGERPRINT_MARK):]
# strip the ": " separator and the closing "-->"
rest = rest[:-3].strip() if rest.endswith("-->") else rest
return rest.lstrip(":").strip() or None
return None
def deployed_version(checks):
"""Version string from the deployed-version check, or ""."""
for c in checks:
if c["name"] == "deployed-version" and c["status"] == "OK":
return str(c.get("detail", ""))
return ""
def main():
global VERSION
checks = load_checks()
VERSION = deployed_version(checks)
non_ok = [c for c in checks if c["status"] != "OK"]
bad = [c for c in checks if c["status"] in ("FAIL", "DRIFT")]
unavailable = [c for c in checks if c["status"] == "UNAVAILABLE"]
print(f"checks: {len(checks)}, FAIL/DRIFT: {len(bad)}, "
f"UNAVAILABLE: {len(unavailable)}")
# The label must exist before it can be used; --force does not touch
# an existing one with the same name
gh(["label", "create", LABEL, "--force",
"--description", "Denní kontrola API monitoru",
"--color", "d93f0b"])
if bad:
issue = find_open_issue()
new_fp = fingerprint(checks)
if issue is None:
gh(["issue", "create", "--label", LABEL, "--title",
"API monitor: kontrola API digiarchivu selhala",
"--body-file", "-"], input_text=issue_body(checks))
print("issue created (or would be)")
else:
if EXISTING_BODY:
old_fp = read_fingerprint(EXISTING_BODY)
else:
res = gh(["issue", "view", issue, "--json", "body",
"--jq", ".body"])
old_fp = None
if res is not None and res.returncode == 0:
old_fp = read_fingerprint(res.stdout)
if old_fp == new_fp:
print(f"issue #{issue}: identical fingerprint, "
"no new comment")
else:
comment = ("Stav kontrol se změnil "
f"(otisk: {new_fp or 'prázdný'}).\n\n"
+ issue_body(checks))
gh(["issue", "comment", issue, "--body-file", "-"],
input_text=comment)
gh(["issue", "edit", issue, "--body-file", "-"],
input_text=issue_body(checks))
print(f"issue #{issue}: comment + body updated")
elif non_ok:
# Only UNAVAILABLE – an outage proves neither break nor recovery
print("only UNAVAILABLE checks, leaving the issue as it is")
else:
issue = find_open_issue()
if issue is None:
print("all checks OK and no open issue")
else:
comment = ("Všechny kontroly prošly, "
f"zavírám. {RUN_URL}".rstrip())
gh(["issue", "close", issue, "--comment", comment])
print(f"issue #{issue} closed with a comment")
return 0
if __name__ == "__main__":
sys.exit(main())
+401
View File
@@ -0,0 +1,401 @@
# -*- coding: utf-8 -*-
"""
Live plugin test – calls the plugin's own functions against the live AMČR
API inside a real (headless) QGIS and checks they still produce non-empty,
well-formed results. Where the contract test says *what* changed, this test
says *whether users break*.
It is the API-sensitive counterpart of tests/smoke_test.py, which is
deliberately offline.
Run it from the repository root inside the qgis/qgis Docker image
(requests is bundled with QGIS):
docker run --rm -v "$PWD:/work:ro" -w /work \
--user "$(id -u):$(id -g)" -e HOME=/tmp \
-e AMCR_RESULTS_DIR=/tmp/results \
qgis/qgis:ltr python3 tests/api_plugin_live.py
The plugin package is imported as a package (amcr_viewer.amcr_tools), so
its relative imports work – a bare spec_from_file_location would make
load_amcr_data swallow the import error into "0 records".
Status model and outputs match tests/api_contract.py: OK / DRIFT / FAIL /
UNAVAILABLE per check, results-plugin_live.json + a Markdown table on
stdout and in $GITHUB_STEP_SUMMARY, exit 1 on any FAIL or DRIFT. A run
where everything is UNAVAILABLE is green but visible in the summary.
Test area (probe 2026-10-02, anonymous): the same Mikulov bbox as the
contract test, 48.8,16.6,48.9,16.75 – akce 185, lokalita 18,
samostatny_nalez 2, pian 294 records. The fake canvas extent uses it
directly in EPSG:4326, so no coordinate transformation is involved.
Thresholds (design decision 6):
* fetch_set per codelist set: >= 1 item and >= 50 % of that category's
row count in the bundled codelists/heslar.csv (a shrunken codelist is
the #67 symptom)
* load_amcr_data per data type: >= 1 layer with >= 1 feature, valid
geometry and the expected attribute fields
Env overrides (outage simulation; the plugin's own URLs are hard-coded,
so the overrides only steer the availability probe and the codelist
sets, whose URLs live in a module dict). An unreachable host fails fast:
after the full retry cycle of the first probe, later probes to the same
host do no network I/O:
AMCR_DA_URL digiarchiv base URL (default
https://digiarchiv.aiscr.cz)
AMCR_OAI_URL AMCR OAI base URL (default
https://api.aiscr.cz/2.2/oai)
AMCR_TIMEOUT per-request timeout in seconds (default 15)
"""
import csv
import json
import os
import sys
import traceback
# Offscreen, otherwise the widgets would need an X server
os.environ.setdefault("QT_QPA_PLATFORM", "offscreen")
ROOT = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
sys.path.insert(0, ROOT)
RESULTS_DIR = os.environ.get("AMCR_RESULTS_DIR", os.getcwd())
DA_URL = os.environ.get("AMCR_DA_URL", "https://digiarchiv.aiscr.cz")
OAI_URL = os.environ.get("AMCR_OAI_URL", "https://api.aiscr.cz/2.2/oai")
TIMEOUT = int(os.environ.get("AMCR_TIMEOUT", "15"))
TEST_BBOX = "48.8,16.6,48.9,16.75" # minLat,minLon,maxLat,maxLon (Mikulov)
BBOX_MIN_LAT, BBOX_MIN_LON, BBOX_MAX_LAT, BBOX_MAX_LON = (
float(x) for x in TEST_BBOX.split(",")
)
RESULTS = []
JOB = "plugin_live"
UNAVAILABLE_DA = False
UNAVAILABLE_OAI = False
def record(name, status, detail):
RESULTS.append({"name": name, "status": status, "detail": detail,
"url": ""})
print(f" {status:<12} {name} – {detail}")
import requests # noqa: E402
# ---------------------------------------------------------------- QGIS setup
from qgis.core import ( # noqa: E402
Qgis,
QgsApplication,
QgsCoordinateReferenceSystem,
QgsProject,
QgsRectangle,
)
print(f"QGIS {Qgis.QGIS_VERSION.split('-')[0]}")
QgsApplication.setPrefixPath(os.environ.get("QGIS_PREFIX_PATH", "/usr"),
True)
qgs = QgsApplication([], True)
qgs.initQgis()
import amcr_viewer.amcr_codelists as codelists # noqa: E402
import amcr_viewer.amcr_tools as tools # noqa: E402
# Point the codelist sets at the overridden base URLs (outage simulation).
# The data-download URL inside load_amcr_data is hard-coded and cannot be
# steered from here; when the probe says digiarchiv is unreachable, the
# download checks are reported UNAVAILABLE without calling the plugin.
if DA_URL != "https://digiarchiv.aiscr.cz" \
or OAI_URL != "https://api.aiscr.cz/2.2/oai":
_new = {}
for key, (base, api_set) in codelists.slovnicek.items():
if "digiarchiv" in base:
base = DA_URL + "/api/search/query"
else:
base = OAI_URL
_new[key] = (base, api_set)
codelists.slovnicek.clear()
codelists.slovnicek.update(_new)
# ------------------------------------------------------------------ fakes
class FakeMessageBar:
"""Collects messageBar() messages so failures can be diagnosed."""
def __init__(self):
self.messages = []
def pushMessage(self, title, text, level=Qgis.MessageLevel.Info):
self.messages.append((title, str(text), level))
class FakeIface:
def __init__(self):
self._bar = FakeMessageBar()
def messageBar(self):
return self._bar
def mapCanvas(self):
return fake_canvas
class FakeMapSettings:
def __init__(self, crs):
self._crs = crs
def destinationCrs(self):
return self._crs
class FakeCanvas:
"""Map canvas whose extent is the test bbox in EPSG:4326."""
def __init__(self):
self._extent = QgsRectangle(
BBOX_MIN_LON, BBOX_MIN_LAT, BBOX_MAX_LON, BBOX_MAX_LAT
)
self._settings = FakeMapSettings(
QgsCoordinateReferenceSystem("EPSG:4326"))
def extent(self):
return self._extent
def mapSettings(self):
return self._settings
fake_canvas = FakeCanvas()
fake_iface = FakeIface()
# amcr_tools does "from qgis.utils import iface", which binds None in a
# headless run – patch the module attribute, not qgis.utils
tools.iface = fake_iface
# ------------------------------------------------------------------ checks
# Circuit breaker (fast outage), the same as in tests/api_contract.py:
# once a host (netloc) is unreachable after full retries, later probes to
# it return False without network I/O, so an all-unreachable run finishes
# in seconds instead of tens of minutes.
DEAD_HOSTS = set()
def probe(url, params=None):
"""Availability probe with retries; True when the API answers.
Once a host is found unreachable after the full retry cycle, it is
added to DEAD_HOSTS and later probes to it fail immediately.
"""
import time
import urllib.parse
netloc = urllib.parse.urlparse(url).netloc
if netloc in DEAD_HOSTS:
return False
for attempt in range(3):
try:
resp = requests.get(url, params=params, timeout=TIMEOUT)
if resp.status_code < 500:
return True
except requests.exceptions.RequestException:
pass
if attempt < 2:
time.sleep((2, 8)[attempt])
DEAD_HOSTS.add(netloc)
return False
def bundled_counts():
"""Row count per category in the bundled codelists/heslar.csv."""
path = os.path.join(ROOT, "amcr_viewer", "codelists", "heslar.csv")
counts = {}
# utf-8-sig: the CSV carries a BOM on purpose (Excel)
with open(path, encoding="utf-8-sig", newline="") as f:
for row in csv.DictReader(f, delimiter=";"):
cat = (row.get("Kategorie") or "").strip()
if cat:
counts[cat] = counts.get(cat, 0) + 1
return counts
def check_translations():
if UNAVAILABLE_DA:
record("load_translations", "UNAVAILABLE",
"digiarchiv unreachable after retries")
return
tools.TRANSLATIONS.clear()
try:
tools.load_translations()
except Exception:
record("load_translations", "FAIL",
traceback.format_exc().rstrip().splitlines()[-1])
return
if tools.TRANSLATIONS:
record("load_translations", "OK",
f"{len(tools.TRANSLATIONS)} keys")
else:
record("load_translations", "FAIL",
"TRANSLATIONS stayed empty after load_translations()")
def check_fetch_set():
"""fetch_set per set in slovnicek against the live API."""
bundled = bundled_counts()
for name, (base_url, api_set) in codelists.slovnicek.items():
unavailable = (UNAVAILABLE_OAI if "digiarchiv" not in base_url
else UNAVAILABLE_DA)
if unavailable:
record(f"fetch_set {name}", "UNAVAILABLE",
"API unreachable after retries")
continue
try:
data = codelists.fetch_set(base_url, name, api_set)
except Exception:
record(f"fetch_set {name}", "FAIL",
traceback.format_exc().rstrip().splitlines()[-1])
continue
if data is None:
record(f"fetch_set {name}", "FAIL", "cancelled (task)")
continue
if not data:
record(f"fetch_set {name}", "FAIL",
f"set {api_set} returned 0 items – the #67 symptom")
continue
expected = bundled.get(name, 0)
if expected and len(data) < expected * 0.5:
record(f"fetch_set {name}", "FAIL",
f"{len(data)} items < 50 % of {expected} bundled rows "
"– a shrunken codelist is the #67 symptom")
else:
record(f"fetch_set {name}", "OK",
f"{len(data)} items"
+ (f" (bundled: {expected})" if expected else ""))
def _layer_specs(typ_dat):
"""Expected attribute fields per data type, from amcr_tools.py."""
common = ["pian", "presnost", "pian_typ", "dj", "typ_dj", typ_dat,
"definicni_body", "odkaz_do_digiarchivu", "okres", "katastr",
"dalsi_katastry", "pristupnost"]
if typ_dat == "akce":
common += ["akce_lokalizace", "vedouci", "organizace",
"specifikace_data", "zahajeni", "ukonceni",
"hlavni_typ", "vedlejsi_typ", "zjisteni",
"nahrazuje_NZ", "projekt"]
elif typ_dat == "lokalita":
common += ["nazev_lokality", "popis_lokality", "typ_lokality",
"druh_lokality", "zachovalost"]
elif typ_dat == "samostatny_nalez":
common = [typ_dat, "definicni_body", "odkaz_do_digiarchivu",
"okres", "katastr", "dalsi_katastry", "projekt",
"nalezce", "datum", "okolnosti", "hloubka_cm",
"lokalizace", "obdobi", "presna_datace", "nalez",
"material", "pocet", "poznamka", "pred_org",
"evidencni", "pristupnost"]
return common
def check_load_amcr_data():
"""load_amcr_data per data type on the test bbox (fake iface/canvas).
The call is synchronous in the main thread (load_amcr_data pumps the
event loop itself, it is not a QgsTask), so a plain call is enough.
"""
for typ_dat in ["akce", "lokalita", "samostatny_nalez"]:
if UNAVAILABLE_DA:
record(f"load_amcr_data {typ_dat}", "UNAVAILABLE",
"digiarchiv unreachable after retries")
continue
# Layers from a previous data type must not mix into the check
project = QgsProject.instance()
project.removeAllMapLayers()
try:
tools.load_amcr_data(fake_canvas, "true", None,
typ_dat=typ_dat, komponenty="false")
except Exception:
record(f"load_amcr_data {typ_dat}", "FAIL",
traceback.format_exc().rstrip().splitlines()[-1])
continue
layers = [lyr for lyr in project.mapLayers().values()
if "amcr_" in lyr.name().lower()]
if not layers:
record(f"load_amcr_data {typ_dat}", "FAIL",
"no AMCR layers were added to the project; messageBar: "
+ "; ".join(m[1] for m in fake_iface._bar.messages[-3:]))
continue
total_features = sum(lyr.featureCount() for lyr in layers)
if total_features < 1:
record(f"load_amcr_data {typ_dat}", "FAIL",
f"{len(layers)} layers but 0 features")
continue
# valid geometry + expected fields on the populated layers
problems = []
expected_fields = _layer_specs(typ_dat)
for layer in layers:
if layer.featureCount() == 0:
continue
fields = {f.name() for f in layer.fields()}
missing = [fl for fl in expected_fields if fl not in fields]
if missing:
problems.append(f"{layer.name()}: missing fields "
f"{missing}")
for feat in layer.getFeatures():
geom = feat.geometry()
if (geom is None or geom.isNull()
or not geom.isGeosValid()):
problems.append(f"{layer.name()}: invalid geometry")
break
if problems:
record(f"load_amcr_data {typ_dat}", "FAIL",
"; ".join(problems))
else:
record(f"load_amcr_data {typ_dat}", "OK",
f"{len(layers)} layers, {total_features} features, "
"fields and geometry valid")
def write_outputs():
os.makedirs(RESULTS_DIR, exist_ok=True)
path = os.path.join(RESULTS_DIR, f"results-{JOB}.json")
with open(path, "w", encoding="utf-8") as f:
json.dump(RESULTS, f, ensure_ascii=False, indent=2)
lines = ["| check | status | detail |", "|---|---|---|"]
for r in RESULTS:
detail = str(r["detail"]).replace("|", "\\|").replace("\n", " ")
lines.append(f"| {r['name']} | {r['status']} | {detail} |")
table = "\n".join(lines)
print()
print(table)
summary = os.environ.get("GITHUB_STEP_SUMMARY")
if summary:
with open(summary, "a", encoding="utf-8") as f:
f.write(table + "\n")
print()
print(f"written: {path}")
bad = [r for r in RESULTS if r["status"] in ("FAIL", "DRIFT")]
print(f"checks: {len(RESULTS)}, FAIL/DRIFT: {len(bad)}")
return 1 if bad else 0
def main():
global UNAVAILABLE_DA, UNAVAILABLE_OAI
print("live plugin test against the production AMČR API")
print(f"test bbox: {TEST_BBOX} (Mikulov)")
UNAVAILABLE_DA = not probe(
DA_URL + "/api/search/query", params={"entity": "akce", "rows": 0})
UNAVAILABLE_OAI = not probe(
OAI_URL, params={"verb": "Identify"})
if UNAVAILABLE_DA:
print("digiarchiv unreachable after retries")
if UNAVAILABLE_OAI:
print("AMČR OAI unreachable after retries")
check_translations()
check_fetch_set()
check_load_amcr_data()
qgs.exitQgis()
return write_outputs()
if __name__ == "__main__":
sys.exit(main())