telecomkz_scraper/tools/reclassify_hotspots.py
Iliyas Kyrykbayev 5cb347a44b TelecomKz Analytics Mapper: audit fixes, real event keys, 32 mapped screens
Rebuilt the capture and mapping pipeline after an audit found the simulator's
data could not be trusted:

* Hotspot coordinates never matched the screenshots. Capture now scrolls the
  page over CDP and pastes each frame at the measured scrollY, so image pixels
  and DOM coordinates share one grid by construction.
* Metrics were synthesised (1200 + n*410) and presented as analytics. Numbers
  are now attached only when the catalog has a matching row; metrics.json
  carries a `source` label and the UI says "no data" instead of showing zeros.
* Event interception hooked a connector bridge that never fires. The app posts
  to api.amplitude.com using the legacy form-urlencoded v1 API; the hook now
  reads event_type off the wire. 36 keys are verified as `observed`.
* All device access moved into tools/telecom_cdp.py: dynamic WebView socket
  discovery (the PID was hardcoded), id-matched CDP, measured native geometry.
* Editor edits can now be saved to disk; API failures no longer report success
  from a stale result file; screenId is no longer interpolated into a shell.

Screens went from 7 (with fabricated markup) to 32, all verified: image height
equals map height, no out-of-bounds hotspots, no dead links.

The id_card screenshot has been manually redacted - it showed a national ID.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-24 18:16:46 +05:00

141 lines
4.9 KiB
Python

"""
Re-apply the event-key rules and the metrics catalog to an already-captured app map.
Useful when EVENT_RULES changes or after a fresh ClickHouse pull: it fixes keys and
metrics without going near the phone. Hotspots whose key was observed on the wire
(keyConfidence "observed") or set by hand ("manual") are never touched.
py tools/reclassify_hotspots.py # all screens
py tools/reclassify_hotspots.py --dry-run
py tools/reclassify_hotspots.py --screen main_dashboard
"""
import argparse
import sys
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
import telecom_cdp as T
from capture_screen import classify
PROTECTED = {"observed", "manual", "dom-attribute"}
def reclassify(screen_filter=None, dry_run=False):
app_map = T.load_app_map()
catalog = T.load_metrics_catalog()
source = T.metrics_source()
changed_keys = 0
rechecked = 0
for screen in app_map.get("screens", []):
if screen_filter and screen["id"] != screen_filter:
continue
for hs in screen.get("hotspots", []):
rechecked += 1
if hs.get("keyConfidence") in PROTECTED:
T.attach_metrics(hs, catalog, source)
continue
if hs.get("source") == "native-uiautomator":
T.attach_metrics(hs, catalog, source)
continue
if "source" not in hs:
# Legacy row from before provenance was tracked. Its key is of unknown
# origin, so flag it for review rather than overwrite it with another
# caption-derived guess - several would collapse onto the same key.
hs.setdefault("keyConfidence", "guessed")
hs.setdefault("source", "legacy")
T.attach_metrics(hs, catalog, source)
continue
key, name_ru, target, confidence = classify(hs.get("label") or "")
if key != hs.get("eventKey"):
print(
" "
+ screen["id"]
+ ": "
+ str(hs.get("eventKey"))
+ " -> "
+ key
+ " ("
+ (hs.get("label") or "")[:45]
+ ")"
)
changed_keys += 1
if not dry_run:
hs["eventKey"] = key
hs["eventNameRu"] = name_ru
hs["keyConfidence"] = confidence
if target and not hs.get("targetScreenId"):
hs["targetScreenId"] = target
if not dry_run:
T.attach_metrics(hs, catalog, source)
# Two hotspots on one screen must not share an event key, and a targetScreenId
# pointing at a screen that was never captured is a dead link in the simulator.
from capture_screen import slugify_event_key, unique_key
known_ids = {sc["id"] for sc in app_map.get("screens", [])}
deduped = dangling = 0
for screen in app_map.get("screens", []):
if screen_filter and screen["id"] != screen_filter:
continue
used = set()
for hs in screen.get("hotspots", []):
target = hs.get("targetScreenId")
if target and target not in known_ids:
print(" " + screen["id"] + ": dropping dead link -> " + target)
hs.pop("targetScreenId", None)
dangling += 1
original = hs.get("eventKey") or ""
if hs.get("keyConfidence") == "observed":
used.add(original) # never rename a verified key
continue
fixed = unique_key(original or slugify_event_key(hs.get("label") or ""), used)
if fixed != original:
print(" " + screen["id"] + ": duplicate " + original + " -> " + fixed)
hs["eventKey"] = fixed
deduped += 1
print("\nDeduplicated keys: " + str(deduped) + ", dead links removed: " + str(dangling))
if not dry_run:
app_map["metricsSource"] = source
T.save_app_map(app_map)
measured = sum(
1
for s in app_map.get("screens", [])
for h in s.get("hotspots", [])
if h.get("metrics")
)
print(
"\nHotspots checked: "
+ str(rechecked)
+ ", keys changed: "
+ str(changed_keys)
+ ", with metrics: "
+ str(measured)
+ " (source: "
+ source
+ ")"
)
if dry_run:
print("Dry run - nothing was written.")
def main():
parser = argparse.ArgumentParser(description="Re-apply event rules and metrics to the app map.")
parser.add_argument("--screen", default=None)
parser.add_argument("--dry-run", action="store_true")
args = parser.parse_args()
reclassify(args.screen, args.dry_run)
return 0
if __name__ == "__main__":
sys.exit(main())