telecomkz_scraper/tools/telecom_cdp.py
Iliyas Kyrykbayev 5cb347a44b TelecomKz Analytics Mapper: audit fixes, real event keys, 32 mapped screens
Rebuilt the capture and mapping pipeline after an audit found the simulator's
data could not be trusted:

* Hotspot coordinates never matched the screenshots. Capture now scrolls the
  page over CDP and pastes each frame at the measured scrollY, so image pixels
  and DOM coordinates share one grid by construction.
* Metrics were synthesised (1200 + n*410) and presented as analytics. Numbers
  are now attached only when the catalog has a matching row; metrics.json
  carries a `source` label and the UI says "no data" instead of showing zeros.
* Event interception hooked a connector bridge that never fires. The app posts
  to api.amplitude.com using the legacy form-urlencoded v1 API; the hook now
  reads event_type off the wire. 36 keys are verified as `observed`.
* All device access moved into tools/telecom_cdp.py: dynamic WebView socket
  discovery (the PID was hardcoded), id-matched CDP, measured native geometry.
* Editor edits can now be saved to disk; API failures no longer report success
  from a stale result file; screenId is no longer interpolated into a shell.

Screens went from 7 (with fabricated markup) to 32, all verified: image height
equals map height, no out-of-bounds hotspots, no dead links.

The id_card screenshot has been manually redacted - it showed a national ID.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-08-24 18:16:46 +05:00

665 lines
22 KiB
Python

"""
Shared TelecomKz automation core.
Everything the capture/crawl/live tools need in one place:
* ADB discovery + device selection
* dynamic WebView DevTools socket discovery (no hardcoded PID)
* a CDP client that matches responses to request ids (CDP events interleave
with command replies, so a bare recv() is not the answer to your command)
* native layout probing via uiautomator (real WebView / bottom-nav bounds)
* the single coordinate convention used by the whole project
Coordinate convention
---------------------
All hotspot rects live in "stitched screen space": the same pixel grid as the
stitched long screenshot produced by capture_screen.py, which is
[ native top bar : rows 0 .. webview_top ]
[ full WebView content: webview_top .. webview_top + H ]
[ native bottom nav : the last nav_h rows ]
so a DOM element at document offset (left, top + scrollY) maps to
x = round(left * scale)
y = webview_top + round((top + scrollY) * scale) with scale = screen_w / innerWidth
That identity is what makes the overlay line up. The content band height must be
measured from the device, never assumed.
"""
import json
import re
import subprocess
import sys
import xml.etree.ElementTree as ET
from pathlib import Path
if sys.platform == "win32":
try:
sys.stdout.reconfigure(encoding="utf-8")
sys.stderr.reconfigure(encoding="utf-8")
except Exception:
pass
APP_PACKAGE = "kz.telecom.app"
WEBVIEW_HOST = "customer.telecom.kz"
CDP_PORT = 9222
# Project root = parent of tools/. Every path is resolved against it so the
# tools work regardless of the caller's working directory.
PROJECT_ROOT = Path(__file__).resolve().parent.parent
SCREENS_DIR = PROJECT_ROOT / "public" / "assets" / "screens"
DATA_DIR = PROJECT_ROOT / "public" / "data"
APP_MAP_PATH = DATA_DIR / "telecomkz_app_map.json"
METRICS_PATH = DATA_DIR / "metrics.json"
ADB_CANDIDATES = [
r"C:\Users\user\AppData\Local\Android\Sdk\platform-tools\adb.exe",
"adb",
]
_ADB_BIN = None
class DeviceError(RuntimeError):
"""Raised when the phone / WebView is not reachable."""
# --------------------------------------------------------------------------- adb
def find_adb():
global _ADB_BIN
if _ADB_BIN:
return _ADB_BIN
for candidate in ADB_CANDIDATES:
try:
res = subprocess.run([candidate, "version"], capture_output=True, text=True)
if res.returncode == 0:
_ADB_BIN = candidate
return _ADB_BIN
except Exception:
continue
raise DeviceError("adb.exe not found. Install Android platform-tools or add adb to PATH.")
def adb(*args, binary=False, timeout=90, device=None):
cmd = [find_adb()]
if device:
cmd += ["-s", device]
cmd += list(args)
if binary:
return subprocess.run(cmd, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout)
return subprocess.run(
cmd, capture_output=True, text=True, encoding="utf-8", errors="replace", timeout=timeout
)
def get_device():
"""Return the serial of the connected device, or raise with an actionable message."""
res = adb("devices")
ready, other = [], []
for line in res.stdout.strip().splitlines()[1:]:
parts = line.strip().split("\t")
if len(parts) < 2:
continue
if parts[1] == "device":
ready.append(parts[0])
else:
other.append(parts[0] + " (" + parts[1] + ")")
if not ready:
detail = " Devices seen but not ready: " + ", ".join(other) + "." if other else ""
raise DeviceError(
"No Android device is ready over ADB. Connect the phone by USB, enable USB debugging "
"and confirm the authorisation prompt." + detail
)
return ready[0]
def screen_size(device):
res = adb("shell", "wm", "size", device=device)
m = re.search(r"(\d+)x(\d+)", res.stdout)
if not m:
raise DeviceError("Could not read the device screen size (adb shell wm size).")
return int(m.group(1)), int(m.group(2))
def app_is_foreground(device):
res = adb("shell", "dumpsys", "window", device=device)
return APP_PACKAGE in res.stdout.split("mCurrentFocus=")[-1][:200]
def screencap(device):
"""Raw PNG bytes of the current screen."""
res = adb("exec-out", "screencap", "-p", binary=True, device=device)
if res.returncode != 0 or len(res.stdout) < 1000:
raise DeviceError("adb exec-out screencap returned no image.")
return res.stdout
# ------------------------------------------------------------------ devtools socket
def find_webview_socket(device):
"""
Locate the app's WebView debugging socket. The socket name carries the app PID,
which changes on every app start, so it is always discovered, never hardcoded.
"""
res = adb("shell", "pidof", APP_PACKAGE, device=device)
pid = res.stdout.strip().split(" ")[0] if res.stdout.strip() else ""
probe = adb("shell", "cat", "/proc/net/unix", device=device)
if pid and ("webview_devtools_remote_" + pid) in probe.stdout:
return "webview_devtools_remote_" + pid
names = sorted(set(re.findall(r"webview_devtools_remote_\S+", probe.stdout)))
if names:
return names[-1]
raise DeviceError(
"No WebView debugging socket found. Open " + APP_PACKAGE + " on the phone and make sure "
"WebView debugging is enabled for this build."
)
def forward_devtools(device, port=CDP_PORT):
socket = find_webview_socket(device)
res = adb("forward", "tcp:" + str(port), "localabstract:" + socket, device=device)
if res.returncode != 0:
raise DeviceError("adb forward failed: " + res.stderr.strip())
return port
def list_targets(port=CDP_PORT):
import urllib.request
with urllib.request.urlopen("http://localhost:" + str(port) + "/json/list", timeout=5) as resp:
return json.loads(resp.read().decode("utf-8"))
def pick_page_target(targets):
"""The app page, not a service worker and not an ad iframe."""
pages = [t for t in targets if t.get("type") == "page" and t.get("webSocketDebuggerUrl")]
if not pages:
raise DeviceError("The WebView exposes no debuggable page target.")
for t in pages:
if WEBVIEW_HOST in (t.get("url") or ""):
return t
return pages[0]
def screen_is_locked(device):
res = adb("shell", "dumpsys", "window", device=device)
return "mDreamingLockscreen=true" in res.stdout
def require_awake(device):
"""
Refuse to capture while the phone is locked.
A screen that times out mid-session still answers screencap and uiautomator, so a
capture happily records the lock screen and writes it over a real screen's data.
Failing loudly is the only safe behaviour.
"""
if screen_is_locked(device):
raise DeviceError(
"The phone is locked (screen timed out). Unlock it and run the capture again. "
"Tip: Settings > Display > Screen timeout, or `adb shell svc power stayon usb`."
)
def bring_app_to_front(device):
"""
Android freezes cached processes, and a frozen WebView stops answering on its
DevTools socket - the port forward succeeds and every request then times out.
So make sure the app is actually foregrounded before talking to it.
"""
if app_is_foreground(device):
return False
adb(
"shell",
"monkey",
"-p",
APP_PACKAGE,
"-c",
"android.intent.category.LAUNCHER",
"1",
device=device,
)
import time
for _ in range(10):
time.sleep(0.5)
if app_is_foreground(device):
return True
raise DeviceError(
"Could not bring " + APP_PACKAGE + " to the foreground. Open the app on the phone manually."
)
def connect(device=None, port=CDP_PORT):
"""Full handshake: device -> foreground -> adb forward -> page target."""
device = device or get_device()
require_awake(device)
bring_app_to_front(device)
forward_devtools(device, port)
try:
targets = list_targets(port)
except Exception as exc:
raise DeviceError(
"The WebView debugger did not answer on port "
+ str(port)
+ " ("
+ type(exc).__name__
+ "). Make sure the TelecomKz screen is open and visible on the phone."
) from exc
return device, pick_page_target(targets)
# ------------------------------------------------------------------------- cdp client
class CdpSession:
"""
Minimal synchronous CDP client.
Command replies are matched by id and CDP events are buffered, so a command
issued after Page.enable / Network.enable still gets its own answer back
instead of swallowing an unrelated event.
"""
def __init__(self, ws_url, timeout=30):
import websockets.sync.client as ws_client
self._ws = ws_client.connect(ws_url, max_size=256 * 1024 * 1024, open_timeout=timeout)
self._next_id = 0
self.events = []
self.timeout = timeout
self.closed = False
def __enter__(self):
return self
def __exit__(self, *exc):
self.close()
def close(self):
try:
self._ws.close()
except Exception:
pass
def send(self, method, **params):
"""
Send one command and return its result.
A page target can vanish mid-run - navigating to another WebView destroys it
and the socket closes. That surfaces as a websockets ConnectionClosed, which
callers should not have to know about, so it is translated into DeviceError:
one exception type means a long-running caller can stop cleanly and still
keep whatever it has already collected.
"""
self._next_id += 1
msg_id = self._next_id
try:
self._ws.send(json.dumps({"id": msg_id, "method": method, "params": params}))
while True:
msg = json.loads(self._ws.recv(timeout=self.timeout))
if msg.get("id") == msg_id:
if "error" in msg:
raise DeviceError("CDP " + method + " failed: " + json.dumps(msg["error"]))
return msg.get("result", {})
if "method" in msg:
self.events.append(msg)
except DeviceError:
raise
except Exception as exc:
self.closed = True
raise DeviceError(
"CDP connection lost during " + method + " (" + type(exc).__name__ + "). "
"The WebView page was probably replaced or closed."
) from exc
def drain_events(self, idle=0.05, max_messages=20000):
"""
Read whatever the page has pushed since the last command, then stop.
`idle` must stay above zero: websockets' sync recv() treats a falsy timeout
as "block forever", so recv(timeout=0) hangs instead of returning empty.
"""
for _ in range(max_messages):
try:
raw = self._ws.recv(timeout=idle)
except Exception:
break
try:
msg = json.loads(raw)
except Exception:
continue
if "method" in msg:
self.events.append(msg)
return self.events
def evaluate(self, expression, await_promise=False):
res = self.send(
"Runtime.evaluate",
expression=expression,
returnByValue=True,
awaitPromise=await_promise,
)
if res.get("exceptionDetails"):
desc = res["exceptionDetails"].get("exception", {}).get("description")
raise DeviceError("JS evaluation failed: " + str(desc or res["exceptionDetails"]))
return res.get("result", {}).get("value")
def open_session(device=None, port=CDP_PORT):
"""Convenience: connect and return (device, target, CdpSession)."""
device, target = connect(device, port)
return device, target, CdpSession(target["webSocketDebuggerUrl"])
# --------------------------------------------------------------- native layout probe
def _parse_bounds(text):
m = re.match(r"\[(-?\d+),(-?\d+)\]\[(-?\d+),(-?\d+)\]", text or "")
if not m:
return None
x1, y1, x2, y2 = (int(g) for g in m.groups())
return {"x": x1, "y": y1, "width": x2 - x1, "height": y2 - y1}
def dump_ui_xml(device):
adb("shell", "uiautomator", "dump", "/sdcard/tk_ui_dump.xml", device=device, timeout=120)
res = adb("shell", "cat", "/sdcard/tk_ui_dump.xml", device=device)
xml = res.stdout.strip()
if not xml.startswith("<?xml") and not xml.startswith("<hierarchy"):
return None
try:
return ET.fromstring(xml)
except ET.ParseError:
return None
def probe_native_layout(device):
"""
Read the real geometry of the native shell instead of assuming it.
Returns webviewTop / webviewBottom / bottomNavTop plus the native clickable
chrome (toolbar buttons, bottom tabs) with true bounds and accessibility labels.
"""
screen_w, screen_h = screen_size(device)
layout = {
"screenWidth": screen_w,
"screenHeight": screen_h,
"webviewTop": None,
"webviewBottom": None,
"bottomNavTop": None,
"nativeElements": [],
"source": "uiautomator",
}
root = dump_ui_xml(device)
if root is None:
layout.update(
{
"webviewTop": round(screen_h * 0.109),
"webviewBottom": round(screen_h * 0.914),
"bottomNavTop": round(screen_h * 0.914),
"source": "fallback",
}
)
return layout
webviews, nav_bounds, natives = [], None, []
def walk(node):
nonlocal nav_bounds
cls = node.attrib.get("class", "")
rid = node.attrib.get("resource-id", "")
rect = _parse_bounds(node.attrib.get("bounds", ""))
if rect and rect["width"] > 0 and rect["height"] > 0:
if cls == "android.webkit.WebView":
webviews.append(rect)
if rid.endswith(":id/bottomNavigation"):
nav_bounds = rect
# The currently selected bottom tab reports clickable="false", so match
# the navigation item ids too or the active tab silently disappears.
is_nav_item = ":id/action_" in rid
if node.attrib.get("clickable") == "true" or is_nav_item:
natives.append(
{
"resourceId": rid,
"text": (node.attrib.get("text") or "").strip(),
"contentDesc": (node.attrib.get("content-desc") or "").strip(),
"className": cls,
"rect": rect,
}
)
for child in node:
walk(child)
walk(root)
if webviews:
outer = max(webviews, key=lambda r: r["width"] * r["height"])
layout["webviewTop"] = outer["y"]
layout["webviewBottom"] = outer["y"] + outer["height"]
if nav_bounds:
layout["bottomNavTop"] = nav_bounds["y"]
if layout["webviewTop"] is None:
layout["webviewTop"] = round(screen_h * 0.109)
layout["webviewBottom"] = round(screen_h * 0.914)
layout["source"] = "fallback"
if layout["bottomNavTop"] is None:
layout["bottomNavTop"] = layout["webviewBottom"]
# Keep only native chrome: nodes outside the WebView band. Anything inside it is
# a WebView proxy node, and the DOM gives far better data for those.
top, bottom = layout["webviewTop"], layout["webviewBottom"]
layout["nativeElements"] = [
el
for el in natives
if el["rect"]["y"] + el["rect"]["height"] <= top + 4 or el["rect"]["y"] >= bottom - 4
]
return layout
# ------------------------------------------------------------------------ app map I/O
def collect_native_elements(device, min_size=24):
"""
Every clickable native control on screen, with a label gathered from its own
text/content-desc or its descendants'.
probe_native_layout() deliberately returns only the chrome outside the WebView
band, because inside it the DOM is a far better source. That is wrong for screens
with no WebView at all (Музыка, Чаты and the other native tabs), which is what
this is for.
"""
root = dump_ui_xml(device)
if root is None:
return []
screen_w, screen_h = screen_size(device)
def label_of(node):
parts = []
def walk(n):
for attr in ("text", "content-desc"):
value = (n.attrib.get(attr) or "").strip()
if value and value not in parts:
parts.append(value)
for child in n:
walk(child)
walk(node)
return " ".join(parts)[:80]
elements = []
seen = set()
def visit(node):
rect = _parse_bounds(node.attrib.get("bounds", ""))
clickable = node.attrib.get("clickable") == "true"
rid = node.attrib.get("resource-id", "")
if (
rect
and (clickable or ":id/action_" in rid)
and rect["width"] >= min_size
and rect["height"] >= min_size
# Skip full-screen containers that happen to be clickable.
and not (rect["width"] >= screen_w * 0.98 and rect["height"] >= screen_h * 0.9)
):
key = (rect["x"], rect["y"], rect["width"], rect["height"])
if key not in seen:
seen.add(key)
elements.append(
{
"resourceId": rid,
"text": (node.attrib.get("text") or "").strip(),
"contentDesc": (node.attrib.get("content-desc") or "").strip(),
"className": node.attrib.get("class", ""),
"label": label_of(node),
"rect": rect,
}
)
for child in node:
visit(child)
visit(root)
return elements
def load_json(path, default):
try:
with open(path, "r", encoding="utf-8") as f:
return json.load(f)
except Exception:
return default
def load_app_map():
return load_json(
APP_MAP_PATH,
{
"project": "TelecomKz Mobile Analytics",
"version": "2.0.0",
"defaultScreenId": "main_dashboard",
"screens": [],
},
)
def save_app_map(app_map):
"""Atomic write: a crash mid-write must not leave a truncated map behind."""
APP_MAP_PATH.parent.mkdir(parents=True, exist_ok=True)
tmp = APP_MAP_PATH.with_suffix(".json.tmp")
with open(tmp, "w", encoding="utf-8") as f:
json.dump(app_map, f, ensure_ascii=False, indent=2)
tmp.replace(APP_MAP_PATH)
def load_metrics_catalog():
"""eventKey -> metrics row, from whatever fetch_clickhouse.py last wrote."""
raw = load_json(METRICS_PATH, [])
rows = raw.get("events", []) if isinstance(raw, dict) else raw
return {r["eventKey"]: r for r in rows if isinstance(r, dict) and r.get("eventKey")}
def metrics_source():
raw = load_json(METRICS_PATH, [])
if isinstance(raw, dict):
return raw.get("source", "demo")
return "demo"
def attach_metrics(hotspot, catalog, source):
"""
Attach real metrics if the catalog knows this event key, and leave them out
entirely if it does not. A hotspot with no measurement must never carry an
invented number: the UI renders a explicit "no data" state for a missing
metrics block, which is honest; a synthesised number is not.
"""
row = catalog.get(hotspot.get("eventKey"))
if not row:
hotspot.pop("metrics", None)
hotspot["metricsSource"] = "none"
return hotspot
hotspot["metrics"] = {
"totalEvents": row.get("totalEvents", 0),
"uniqueUsers": row.get("uniqueUsers", 0),
"avgEventsPerUser": row.get("avgEventsPerUser"),
"shareOfClicks": row.get("shareOfClicks"),
"lastUpdated": row.get("lastUpdated"),
}
hotspot["metricsSource"] = source
if not hotspot.get("eventNameRu") and row.get("eventNameRu"):
hotspot["eventNameRu"] = row["eventNameRu"]
return hotspot
# Confidence levels that represent real knowledge rather than a caption guess.
# These survive a re-capture; a rule-derived key is disposable and does not.
TRUSTED_CONFIDENCE = ("observed", "manual", "dom-attribute")
def _merge_trusted_keys(new_hotspots, old_hotspots):
"""
Carry verified event keys from the previous capture onto the new one.
A re-capture rebuilds hotspots from scratch, so without this every key that was
confirmed on the wire (keyConfidence "observed") or typed in by an analyst would
be silently replaced by a caption guess - throwing away the only information in
the map that was expensive to obtain.
Matching is by label, which is the element's own caption and is stable across
captures of the same screen.
"""
trusted = {}
for old in old_hotspots or []:
if old.get("keyConfidence") in TRUSTED_CONFIDENCE:
label = (old.get("label") or "").strip().lower()
if label:
trusted[label] = old
carried = 0
for new in new_hotspots:
old = trusted.get((new.get("label") or "").strip().lower())
if not old:
continue
new["eventKey"] = old["eventKey"]
new["eventNameRu"] = old.get("eventNameRu") or new.get("eventNameRu")
new["keyConfidence"] = old["keyConfidence"]
if old.get("targetScreenId") and not new.get("targetScreenId"):
new["targetScreenId"] = old["targetScreenId"]
carried += 1
return carried
def upsert_screen(app_map, screen):
for i, existing in enumerate(app_map.get("screens", [])):
if existing["id"] == screen["id"]:
# Preserve the human-authored name/category across re-captures.
screen["name"] = existing.get("name") or screen["name"]
screen["category"] = existing.get("category") or screen["category"]
screen["carriedKeys"] = _merge_trusted_keys(
screen.get("hotspots", []), existing.get("hotspots", [])
)
app_map["screens"][i] = screen
return app_map
app_map.setdefault("screens", []).append(screen)
return app_map
SCREEN_ID_RE = re.compile(r"^[A-Za-z0-9_-]{1,64}$")
def validate_screen_id(screen_id):
if not SCREEN_ID_RE.match(screen_id or ""):
raise ValueError("Invalid screenId: " + repr(screen_id))
return screen_id