You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Playwright提取Google按钮元素失败,请求代码问题排查

问题:Playwright DOM快照提取交互元素遗漏可点击按钮

我有一段基于Python Playwright的代码,用于从网页DOM树中提取可输入、可点击的交互元素。多数情况运行正常,但会遗漏部分元素——比如Google登录页的「Next」按钮被标记为不可点击。以下是相关代码、日志及DOM结构,请求排查问题。

完整代码

from playwright.sync_api import sync_playwright

VOID_ELEMENTS = {
    "area",
    "base",
    "br",
    "col",
    "embed",
    "hr",
    "img",
    "input",
    "link",
    "meta",
    "param",
    "source",
    "track",
    "wbr",
}
READABLE_ATTRIBUTES = {
    "title",
    "alt",
    "href",
    "placeholder",
    "label",
    "value",
    "caption",
    "summary",
    "aria-label",
    "aria-describedby",
    "datetime",
    "download",
    "selected",
    "checked",
    "type",
}
UNCLICKABLE_ELEMENTS = {"html", "head", "body"}
CLICKABLE_ELEMENTS = {"a", "button", "img", "details", "summary"}
INPUT_ELEMENTS = {"input", "textarea", "select", "option"}


class DOMNode:
    def __init__(self, i, nodes, strings):
        self._on_screen = None
        self.parent = None
        self.children = []
        self.llm_id = None
        ### Only some nodes have these, default None to differentiate between None and False
        self.bounds = None
        self.center = None
        self.inputValue = None
        self.inputChecked = None
        self.isClickable = None
        self.optionSelected = None
        self.parentId = (
            nodes["parentIndex"][i] if nodes["parentIndex"][i] >= 0 else None
        )
        self.nodeType = strings[nodes["nodeType"][i]]
        self.nodeName = strings[nodes["nodeName"][i]].lower()
        self.nodeValue = (
            strings[nodes["nodeValue"][i]].strip()
            if nodes["nodeValue"][i] >= 0
            else None
        )
        self.backendNodeId = nodes["backendNodeId"][i]

        self.attributes = {}
        attrs = nodes["attributes"][i]
        for att1, att2 in zip(attrs[::2], attrs[1::2]):
            self.attributes[strings[att1]] = strings[att2][:100]  # cut off long URLs

        self.readable_attributes = {
            k: v for k, v in self.attributes.items() if k in READABLE_ATTRIBUTES
        }

    def __repr__(self, indent=0) -> str:
        if self.nodeName == "#text":
            return " " * indent + (self.nodeValue or "")

        attr_str = " ".join([f'{k}="{v}"' for k, v in self.readable_attributes.items()])
        attr_str = " " + attr_str if attr_str else ""
        open_tag = f"<{self.nodeName}{attr_str}>"
        close_tag = f"</{self.nodeName}>"

        if len(self.children) == 0:
            return (" " * indent + open_tag) + (
                close_tag if self.nodeName not in VOID_ELEMENTS else ""
            )

        # special case for elements with only one text child -> one-line element
        if len(self.children) == 1 and self.children[0].nodeName == "#text":
            return (" " * indent + open_tag) + self.children[0].__repr__() + close_tag

        children_repr = "\n".join(
            [child.__repr__(indent + 2) for child in self.children]
        )
        return (
            (" " * indent + open_tag)
            + "\n"
            + children_repr
            + "\n"
            + (" " * indent + close_tag)
        )

    def on_screen(self, screen_bounds):
        if len(self.children) > 0:
            return any([child.on_screen(screen_bounds) for child in self.children])

        if (
            self.bounds is None
            or len(self.bounds) != 4
            or self.bounds[2] * self.bounds[3] == 0
        ):
            return False

        x, y, w, h = self.bounds
        win_upper_bound, win_left_bound, win_width, win_height = screen_bounds
        win_right_bound = win_left_bound + win_width
        win_lower_bound = win_upper_bound + win_height
        return (
            x < win_right_bound
            and x + w > win_left_bound
            and y < win_lower_bound
            and y + h > win_upper_bound
        )


class Globot:
    def __init__(self, headless=False):
        playwright = sync_playwright().start()
        self.browser = playwright.chromium.launch(headless=headless)
        self.context = self.browser.new_context()
        self.page = self.context.new_page()

    def go_to_page(self, url):
        self.page.goto(url=url if "://" in url else "https://" + url)
        self.client = self.page.context.new_cdp_session(self.page)
        self.page.wait_for_load_state("domcontentloaded")

    def crawl(self) -> tuple[dict[int, DOMNode], dict[int, DOMNode]]:
        dom = self.client.send(
            "DOMSnapshot.captureSnapshot",
            {"computedStyles": [], "includeDOMRects": True, "includePaintOrder": True},
        )

        dom_strings = dom["strings"]
        document = dom["documents"][0]
        dom_layout = document["layout"]
        dom_nodes = document["nodes"]

        screen_bounds = dom_layout["bounds"][0]
        # For some reason `window.devicePixelRatio` this gives the wrong answer sometimes
        device_pixel_ratio = screen_bounds[2] / self.page.evaluate(
            "window.screen.width"
        )

        nodes = []
        root = None

        # Takes much longer naively
        nodeIndex_flipped = {v: k for k, v in enumerate(dom_layout["nodeIndex"])}
        inputValue_flipped = {
            v: k for k, v in enumerate(dom_nodes["inputValue"]["index"])
        }
        for i in range(len(dom_nodes["parentIndex"])):
            node = DOMNode(i, dom_nodes, dom_strings)
            if i == 0:
                root = node

            if i in nodeIndex_flipped:
                bounds = dom_layout["bounds"][nodeIndex_flipped[i]]
                bounds = [int(b / device_pixel_ratio) for b in bounds]
                node.bounds = bounds
                node.center = (
                    int(bounds[0] + bounds[2] / 2),
                    int(bounds[1] + bounds[3] / 2),
                )

            if i in dom_nodes["isClickable"]["index"]:
                node.isClickable = True

            if i in inputValue_flipped:
                v = dom_nodes["inputValue"]["value"][inputValue_flipped[i]]
                node.inputValue = dom_strings[v] if v >= 0 else ""
                # node.string_attributes['value'] = node.inputValue

            if i in dom_nodes["inputChecked"]["index"]:
                node.inputChecked = True

            if i in dom_nodes["optionSelected"]["index"]:
                node.optionSelected = True

            nodes.append(node)

        # Switch node ids to node pointers
        for node in nodes:
            if node.parentId is not None:
                node.parent = nodes[node.parentId]
                node.parent.children.append(node)

        count = 0
        input_elements = {}
        clickable_elements = {}

        def find_interactive_elements(node):
            nonlocal count
            clickable = (
                node.nodeName in CLICKABLE_ELEMENTS
                and node.isClickable
                and node.center is not None
            )
            inputable = node.nodeName in INPUT_ELEMENTS or node.inputValue is not None

            # Special case for select and option elements
            select_or_option = node.nodeName == "select" or node.nodeName == "option"
            visible = node.on_screen(
                root.bounds
            ) and "visibility: hidden" not in node.attributes.get("style", "")

            if node.nodeName == "button":
                print(f"Node: {node.nodeName}")
                print(f"  Attributes: {node.attributes}")
                print(f"  Bounds: {node.bounds}")
                print(f"  Clickable: {clickable}")
                print(f"  Inputable: {inputable}")
                print(f"  Visible: {visible}")
                print(f"  Center: {node.center}")

            if visible and (clickable or inputable) or select_or_option:
                if clickable:
                    clickable_elements[count] = node
                if inputable or select_or_option:
                    input_elements[count] = node
                node.llm_id = count
                count += 1

            for child in node.children:
                find_interactive_elements(child)

        find_interactive_elements(root)

        return input_elements, clickable_elements

问题复现代码片段

from pprint import pprint

bot = Globot()
bot.go_to_page(
    "https://accounts.google.com/v3/signin/identifier?authuser=0&continue=https%3A%2F%2Fwww.google.com%2F&ec=GAlAmgQ&hl=en&flowName=GlifWebSignIn&flowEntry=AddSession&dsh=S1040273122%3A1718390580872851&ddm=0"
)
inputs, clickables = bot.crawl()

s = ""
for i in inputs.keys() | clickables.keys():
    inputable = False
    clickable = False
    if i in inputs:
        node = inputs[i]
        inputable = True
    if i in clickables:
        node = clickables[i]
        clickable = True

    s += f"&lt;node id={i} clickable={clickable} inputable={inputable}&gt;\n"
    s += node.__repr__(indent=2)
    s += "\n&lt;/node&gt;\n"
html_description = s
pprint(html_description)

Next按钮相关日志

Node: button
  Attributes: {'class': 'VfPpkd-LgbsSe VfPpkd-LgbsSe-OWXEXe-k8QpJ VfPpkd-LgbsSe-OWXEXe-dgl2Hf nCP5yc AjY5Oe DuMIQc LQeN7 BqKG', 'jscontroller': 'soHxf', 'jsaction': 'click:cOuCgd; mousedown:UX7yZ; mouseup:lbsD7e; mouseenter:tfO1Yc; mouseleave:JywGue; touchstart:p6p2', 'data-idom-class': 'nCP5yc AjY5Oe DuMIQc LQeN7 BqKGqe Jskylb TrZEUc lw1w4b', 'jsname': 'LgbsSe', 'type': 'button'}
  Bounds: [965, 453, 78, 40]
  Clickable: None
  Inputable: False
  Visible: True
  Center: (1004, 473)

Next按钮原始HTML

<button class="VfPpkd-LgbsSe VfPpkd-LgbsSe-OWXEXe-k8QpJ VfPpkd-LgbsSe-OWXEXe-dgl2Hf nCP5yc AjY5Oe DuMIQc LQeN7 BqKGqe Jskylb TrZEUc lw1w4b" jscontroller="soHxf" jsaction="click:cOuCgd; mousedown:UX7yZ; mouseup:lbsD7e; mouseenter:tfO1Yc; mouseleave:JywGue; touchstart:p6p2H; touchmove:FwuNnf; touchend:yfqBxc; touchcancel:JMtRjd; focus:AHmuwe; blur:O22p3e; contextmenu:mg9Pef;mlnRJb:fLiPzd;" data-idom-class="nCP5yc AjY5Oe DuMIQc LQeN7 BqKGqe Jskylb TrZEUc lw1w4b" jsname="LgbsSe" type="button"><div class="VfPpkd-Jh9lGc"></div><div class="VfPpkd-J1Ukfc-LhBDec"></div><div class="VfPpkd-RLmnJb"></div><span jsname="V67aGc" class="VfPpkd-vQzf8d">Next</span></button>

页面截图

Google登录页Next按钮截图


问题分析与解决

从日志可见,Next按钮的isClickable属性为None,导致clickable判断不成立。核心问题在于DOMSnapshot的isClickable字段仅标记原生可点击元素,而Google按钮通过jsaction绑定点击事件,未被DOMSnapshot识别。

修复方案:

  1. 修正isClickable默认值:将DOMNode类中isClickable的默认值从None改为False,避免逻辑判断异常。
  2. 扩展可点击判断逻辑:检测元素的jsaction属性,若包含click:前缀则标记为可点击。
  3. 优化页面等待逻辑:确保页面完全加载后再捕获快照,避免元素状态未初始化。
具体代码修改:
  1. DOMNode类初始化修改:
class DOMNode:
    def __init__(self, i, nodes, strings):
        # ... 其他代码 ...
        self.isClickable = False  # 替换原None默认值
        # ... 其他代码 ...
  1. crawl方法中补充jsaction检测:
    在if i in dom_nodes["isClickable"]["index"]:代码块后添加:
# 检测通过jsaction绑定的点击事件
if "jsaction" in node.attributes and "click:" in node.attributes["jsaction"]:
    node.isClickable = True
  1. 优化find_interactive_elements中的clickable判断:
clickable = (
    node.nodeName in CLICKABLE_ELEMENTS
    and (node.isClickable or ("jsaction" in node.attributes and "click:" in node.attributes["jsaction"]))
    and node.center is not None
)
  1. 优化页面等待逻辑:
    修改go_to_page方法,确保页面完全加载:
def go_to_page(self, url):
    self.page.goto(url=url if "://" in url else "https://" + url)
    self.client = self.page.context.new_cdp_session(self.page)
    self.page.wait_for_load_state("load")  # 等待页面完全加载
    self.page.wait_for_selector('button:has-text("Next")')  # 等待目标按钮出现

上述修改后,Next按钮会被正确识别为可点击元素。


内容的提问来源于stack exchange,提问作者Benjamin Geoffrey

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.22 15:32:01