#!/usr/bin/env python3
"""Accessibility snapshot of the box desktop (AT-SPI).

Prints one line per interactive/visible element with its role, name and the
screen-pixel centre the computer tool can click:

    [12] button "登入" @ (640,412) 96x36  win="蝦皮購物 - Chromium"

Usage: box-a11y [--json] [--limit N] [--budget SECONDS]
Exit 0 with an empty list when nothing is exposed; exit 2 when AT-SPI itself is
unreachable (the desktop or registry is not up).
"""
import argparse
import json
import os
import sys
import time
import warnings

warnings.filterwarnings("ignore", category=DeprecationWarning)

# Always listed when on screen, even without a name (the agent can still
# target them by position/value).
INTERACTIVE = {
    "push button", "toggle button", "radio button", "check box", "link",
    "menu item", "check menu item", "radio menu item", "tab", "page tab",
    "list item", "tree item", "combo box", "spin button", "slider",
    "entry", "password text",
}
# Listed only when they carry a name or text (headings, labels, cells...).
NAMED = {
    "heading", "label", "static", "text", "image", "table cell",
    "column header", "row header", "icon", "menu", "tool tip", "alert",
    "dialog", "frame",
}
ROLE_SHORT = {
    "push button": "button", "toggle button": "toggle", "radio button": "radio",
    "check box": "checkbox", "menu item": "menuitem", "check menu item": "menuitem",
    "radio menu item": "menuitem", "page tab": "tab", "list item": "listitem",
    "tree item": "treeitem", "combo box": "combobox", "spin button": "spinner",
    "password text": "password", "entry": "textbox", "text": "text",
    "static": "text", "document frame": "document", "table cell": "cell",
    "column header": "header", "row header": "header", "internal frame": "frame",
}


def load_dbus_env():
    if os.environ.get("DBUS_SESSION_BUS_ADDRESS"):
        return
    try:
        with open("/tmp/lazyboy/dbus.env", encoding="utf-8") as fh:
            for line in fh:
                key, _, value = line.strip().partition("=")
                if key == "DBUS_SESSION_BUS_ADDRESS" and value:
                    os.environ[key] = value
    except OSError:
        pass


def main():
    ap = argparse.ArgumentParser()
    ap.add_argument("--json", action="store_true")
    ap.add_argument("--limit", type=int, default=300)
    ap.add_argument("--budget", type=float, default=6.0)
    ap.add_argument("--max-depth", type=int, default=40)
    args = ap.parse_args()

    load_dbus_env()
    os.environ.setdefault("DISPLAY", ":1")
    try:
        import gi
        gi.require_version("Atspi", "2.0")
        from gi.repository import Atspi
    except Exception as exc:  # noqa: BLE001
        print(f"AT-SPI bindings unavailable: {exc}", file=sys.stderr)
        return 2

    try:
        Atspi.init()
        desktop = Atspi.get_desktop(0)
        app_count = desktop.get_child_count()
    except Exception as exc:  # noqa: BLE001
        print(f"AT-SPI registry unreachable: {exc}", file=sys.stderr)
        return 2

    screen_w = int(os.environ.get("LAZYBOY_SCREEN_W", "0") or 0)
    screen_h = int(os.environ.get("LAZYBOY_SCREEN_H", "0") or 0)
    deadline = time.monotonic() + args.budget
    out = []
    truncated = False
    seen_apps = []

    def showing(node):
        try:
            st = node.get_state_set()
            return st.contains(Atspi.StateType.SHOWING) and st.contains(Atspi.StateType.VISIBLE)
        except Exception:  # noqa: BLE001
            return False

    def extents(node):
        try:
            comp = node.get_component() if hasattr(node, "get_component") else node.get_component_iface()
            if comp is None:
                return None
            r = comp.get_extents(Atspi.CoordType.SCREEN)
            return int(r.x), int(r.y), int(r.width), int(r.height)
        except Exception:  # noqa: BLE001
            return None

    def onscreen(box):
        x, y, w, h = box
        if w <= 0 or h <= 0:
            return False
        if screen_w and (x >= screen_w or x + w <= 0):
            return False
        if screen_h and (y >= screen_h or y + h <= 0):
            return False
        return True

    def walk(node, depth, win_title):
        nonlocal truncated
        if truncated or time.monotonic() > deadline:
            truncated = True
            return
        if depth > args.max_depth or len(out) >= args.limit:
            truncated = truncated or len(out) >= args.limit
            return
        try:
            role = node.get_role_name()
        except Exception:  # noqa: BLE001
            return
        if depth > 1 and not showing(node):
            return
        box = extents(node)
        if depth > 1 and (box is None or not onscreen(box)):
            return
        name = ""
        try:
            name = (node.get_name() or "").strip()
        except Exception:  # noqa: BLE001
            pass
        if role in ("frame", "dialog", "window") and name:
            win_title = name
        try:
            st = node.get_state_set()
            focused = st.contains(Atspi.StateType.FOCUSED)
            editable = st.contains(Atspi.StateType.EDITABLE)
            checked = st.contains(Atspi.StateType.CHECKED)
            selected = st.contains(Atspi.StateType.SELECTED)
            disabled = not st.contains(Atspi.StateType.ENABLED)
        except Exception:  # noqa: BLE001
            focused = editable = checked = selected = disabled = False
        value = ""
        if role in ("entry", "password text", "text", "spin button", "slider") or editable:
            try:
                txt = node.get_text() if hasattr(node, "get_text") else node.get_text_iface()
                if txt is not None and role != "password text":
                    value = (txt.get_text(0, min(txt.get_character_count(), 80)) or "").strip()
            except Exception:  # noqa: BLE001
                pass
        printable = depth > 1 and box is not None and (
            role in INTERACTIVE
            or (role in NAMED and (name or value))
            or (editable and role == "text")
        )
        if printable:
            x, y, w, h = box
            out.append({
                "role": ROLE_SHORT.get(role, role.replace(" ", "")),
                "name": name[:120],
                "value": value[:80],
                "x": x, "y": y, "w": w, "h": h,
                "cx": x + w // 2, "cy": y + h // 2,
                "focused": focused, "checked": checked, "selected": selected,
                "disabled": disabled, "editable": editable,
                "window": win_title,
            })
        try:
            count = node.get_child_count()
        except Exception:  # noqa: BLE001
            return
        for i in range(min(count, 2000)):
            if truncated:
                return
            try:
                child = node.get_child_at_index(i)
            except Exception:  # noqa: BLE001
                continue
            if child is None:
                continue
            walk(child, depth + 1, win_title)

    for i in range(app_count):
        try:
            app = desktop.get_child_at_index(i)
            if app is None:
                continue
            app_name = app.get_name() or ""
        except Exception:  # noqa: BLE001
            continue
        if app_name in ("xfce4-panel", "wrapper-2.0", "xfdesktop", "xfwm4", "box-a11y", "python3"):
            # The desktop shell is visible in the screenshot already; keep the
            # list for the windows the agent actually works in.
            continue
        seen_apps.append(app_name)
        walk(app, 0, app_name)

    # Chromium mirrors some nodes (omnibox rows, nested list items); one entry
    # per role/name/box is enough for targeting. Then top-to-bottom,
    # left-to-right reading order.
    unique = {}
    for e in out:
        unique.setdefault((e["role"], e["name"], e["cx"], e["cy"], e["w"], e["h"]), e)
    window_order = {}
    for e in unique.values():
        window_order.setdefault(e["window"], len(window_order))
    out = sorted(unique.values(), key=lambda e: (window_order[e["window"]], e["cy"] // 12, e["cx"]))
    for i, e in enumerate(out):
        e["id"] = i
    if args.json:
        json.dump({"apps": seen_apps, "elements": out, "truncated": truncated}, sys.stdout, ensure_ascii=False)
        print()
        return 0
    lines = []
    current_window = None
    for e in out:
        if len(window_order) > 1 and e["window"] != current_window:
            current_window = e["window"]
            lines.append(f"## {current_window}")
        flags = []
        if e["focused"]:
            flags.append("focused")
        if e["checked"]:
            flags.append("checked")
        if e["selected"]:
            flags.append("selected")
        if e["disabled"]:
            flags.append("disabled")
        label = f'"{e["name"]}"' if e["name"] else ""
        if e["value"]:
            label += f' value="{e["value"]}"'
        parts = [f"[{e['id']}]", e["role"]]
        if label:
            parts.append(label)
        parts.append(f"@ ({e['cx']},{e['cy']}) {e['w']}x{e['h']}")
        if flags:
            parts.append("(" + ", ".join(flags) + ")")
        lines.append(" ".join(parts))
    if not lines:
        lines.append("(no accessible elements exposed; apps: " + ", ".join(seen_apps) + ")")
    if truncated:
        lines.append(f"... truncated at {len(out)} elements; scroll or focus a window and snapshot again")
    print("\n".join(lines))
    return 0


if __name__ == "__main__":
    sys.exit(main())
