重开仓库内容:改推五个技能(browser-harness / humanizer / humanizer-zh / product-planning / session-mechanism)

按授权清空原有内容后重新提交(原 oil-ui-pro 一并移出,可从历史恢复)。
browser-harness 剔除 .venv 等运行环境;根 .gitignore 补记 .venv/ 与 node_modules/。
This commit is contained in:
admin committed 2026-10-07 16:06:02 +08:00
1 parent 9a152a1952
commit 777f7fe5d0
240 files changed
+52518 -4409

No files matched your search

@@ -0,0 +1,518 @@
"""Browser control via CDP.
Core helpers live here. Agent-editable helpers live in
BH_AGENT_WORKSPACE/agent_helpers.py.
"""
import base64, importlib.util, json, math, os, sys, time, urllib.request
from pathlib import Path
from urllib.parse import urlparse
from . import _ipc as ipc
from . import paths
CORE_DIR = Path(__file__).resolve().parent
REPO_ROOT = CORE_DIR.parent.parent
AGENT_WORKSPACE = paths.workspace_dir()
def _load_env():
paths = [REPO_ROOT / ".env", AGENT_WORKSPACE / ".env"]
for p in paths:
if not p.exists():
continue
_load_env_file(p)
def _load_env_file(p):
for line in p.read_text(encoding="utf-8-sig", errors="replace").splitlines():
line = line.strip()
if not line or line.startswith("#") or "=" not in line:
continue
k, v = line.split("=", 1)
os.environ.setdefault(k.strip(), v.strip().strip('"').strip("'"))
_load_env()
NAME = os.environ.get("BU_NAME", "default")
SOCK = ipc.sock_addr(NAME)
INTERNAL = ("chrome://", "chrome-untrusted://", "devtools://", "chrome-extension://", "about:")
def _send(req):
c, token = ipc.connect(NAME, timeout=5.0)
try:
r = ipc.request(c, token, req)
finally:
c.close()
if "error" in r: raise RuntimeError(r["error"])
return r
def cdp(method, session_id=None, **params):
"""Raw CDP. cdp('Page.navigate', url='...'), cdp('DOM.getDocument', depth=-1)."""
return _send({"method": method, "params": params, "session_id": session_id}).get("result", {})
def drain_events(): return _send({"meta": "drain_events"})["events"]
def _js_snippet(expression, limit=160):
snippet = expression.strip().replace("\n", "\\n")
return snippet[:limit - 3] + "..." if len(snippet) > limit else snippet
def _js_exception_description(result, details):
desc = result.get("description")
exc = details.get("exception") if details else None
if not desc and isinstance(exc, dict):
desc = exc.get("description")
if desc is None and "value" in exc:
desc = str(exc["value"])
if desc is None:
desc = exc.get("className")
if not desc and details:
desc = details.get("text")
return desc or "JavaScript evaluation failed"
def _decode_unserializable_js_value(value):
if value == "NaN":
return math.nan
if value == "Infinity":
return math.inf
if value == "-Infinity":
return -math.inf
if value == "-0":
return -0.0
if value.endswith("n"):
return int(value[:-1])
return value
def _runtime_value(response, expression):
result = response.get("result", {})
details = response.get("exceptionDetails")
if details or result.get("subtype") == "error":
desc = _js_exception_description(result, details)
if details:
line = details.get("lineNumber")
col = details.get("columnNumber")
loc = f" at line {line}, column {col}" if line is not None and col is not None else ""
else:
loc = ""
raise RuntimeError(f"JavaScript evaluation failed{loc}: {desc}; expression: {_js_snippet(expression)}")
if "value" in result:
return result["value"]
if "unserializableValue" in result:
return _decode_unserializable_js_value(result["unserializableValue"])
return None
def _runtime_evaluate(expression, session_id=None, await_promise=False):
try:
r = cdp("Runtime.evaluate", session_id=session_id, expression=expression, returnByValue=True, awaitPromise=await_promise)
except TimeoutError as e:
raise RuntimeError(f"Runtime.evaluate timed out; expression: {_js_snippet(expression)}") from e
return _runtime_value(r, expression)
def _wrap_js_function(expression):
return f"(function(){{{expression}}})()"
def _is_illegal_return_error(exc):
return "Illegal return statement" in str(exc)
# --- navigation / page ---
def goto_url(url):
r = cdp("Page.navigate", url=url)
if os.environ.get("BH_DOMAIN_SKILLS") != "1":
return r
d = (AGENT_WORKSPACE / "domain-skills" / (urlparse(url).hostname or "").removeprefix("www.").split(".")[0])
return {**r, "domain_skills": sorted(p.name for p in d.rglob("*.md"))[:10]} if d.is_dir() else r
def page_info():
"""{url, title, w, h, sx, sy, pw, ph} — viewport + scroll + page size.
If a native dialog (alert/confirm/prompt/beforeunload) is open, returns
{dialog: {type, message, ...}} instead — the page's JS thread is frozen
until the dialog is handled (see interaction-skills/dialogs.md)."""
dialog = _send({"meta": "pending_dialog"}).get("dialog")
if dialog:
return {"dialog": dialog}
expression = "JSON.stringify({url:location.href,title:document.title,w:innerWidth,h:innerHeight,sx:scrollX,sy:scrollY,pw:document.documentElement.scrollWidth,ph:document.documentElement.scrollHeight})"
return json.loads(_runtime_evaluate(expression))
# --- input ---
_debug_click_counter = 0
def click_at_xy(x, y, button="left", clicks=1):
if os.environ.get("BH_DEBUG_CLICKS"):
global _debug_click_counter
try:
from PIL import Image, ImageDraw
dpr = js("window.devicePixelRatio") or 1
path = capture_screenshot(str(ipc._TMP / f"debug_click_{_debug_click_counter}.png"))
img = Image.open(path)
draw = ImageDraw.Draw(img)
px, py = int(x * dpr), int(y * dpr)
r = int(15 * dpr)
draw.ellipse([px - r, py - r, px + r, py + r], outline="red", width=int(3 * dpr))
draw.line([px - r - int(5 * dpr), py, px + r + int(5 * dpr), py], fill="red", width=int(2 * dpr))
draw.line([px, py - r - int(5 * dpr), px, py + r + int(5 * dpr)], fill="red", width=int(2 * dpr))
img.save(path)
print(f"[debug_click] saved {path} (x={x}, y={y}, dpr={dpr})")
except Exception as e:
print(f"[debug_click] overlay failed: {e}")
_debug_click_counter += 1
cdp("Input.dispatchMouseEvent", type="mousePressed", x=x, y=y, button=button, clickCount=clicks)
cdp("Input.dispatchMouseEvent", type="mouseReleased", x=x, y=y, button=button, clickCount=clicks)
def type_text(text):
cdp("Input.insertText", text=text)
def fill_input(selector, text, clear_first=True, timeout=0.0):
"""Fill a framework-managed input (React controlled, Vue v-model, Ember tracked).
type_text() uses Input.insertText which bypasses framework event listeners and leaves
submit buttons disabled. This helper focuses the element, clears it, types via real
key events, then fires synthetic input+change events so the framework sees the update.
Raises RuntimeError if the element is not found. Pass timeout>0 to wait for
late-rendered elements (e.g. after a route change) before typing.
"""
if timeout > 0:
if not wait_for_element(selector, timeout=timeout):
raise RuntimeError(f"fill_input: element not found: {selector!r}")
focused = js(
f"(()=>{{const e=document.querySelector({json.dumps(selector)});"
f"if(!e)return false;e.focus();return true;}})()"
)
if not focused:
raise RuntimeError(f"fill_input: element not found: {selector!r}")
if clear_first:
# Dispatch select-all directly — NOT via press_key, which always emits a
# `char` event for single-char keys. With Ctrl/Cmd held, that `char`
# makes Chrome treat the input as a printable "a" instead of firing the
# select-all shortcut, leaving the field uncleared.
mods = 4 if sys.platform == "darwin" else 2 # Cmd on macOS, Ctrl elsewhere
select_all = {"key": "a", "code": "KeyA", "modifiers": mods,
"windowsVirtualKeyCode": 65, "nativeVirtualKeyCode": 65}
cdp("Input.dispatchKeyEvent", type="rawKeyDown", **select_all)
cdp("Input.dispatchKeyEvent", type="keyUp", **select_all)
press_key("Backspace")
for ch in text:
press_key(ch)
js(
f"(()=>{{const e=document.querySelector({json.dumps(selector)});"
f"if(!e)return;"
f"e.dispatchEvent(new Event('input',{{bubbles:true}}));"
f"e.dispatchEvent(new Event('change',{{bubbles:true}}));}})();"
)
_KEYS = { # key → (windowsVirtualKeyCode, code, text)
"Enter": (13, "Enter", "\r"), "Tab": (9, "Tab", "\t"), "Backspace": (8, "Backspace", ""),
"Escape": (27, "Escape", ""), "Delete": (46, "Delete", ""), " ": (32, "Space", " "),
"ArrowLeft": (37, "ArrowLeft", ""), "ArrowUp": (38, "ArrowUp", ""),
"ArrowRight": (39, "ArrowRight", ""), "ArrowDown": (40, "ArrowDown", ""),
"Home": (36, "Home", ""), "End": (35, "End", ""),
"PageUp": (33, "PageUp", ""), "PageDown": (34, "PageDown", ""),
}
def press_key(key, modifiers=0):
"""Modifiers bitfield: 1=Alt, 2=Ctrl, 4=Meta(Cmd), 8=Shift.
Special keys (Enter, Tab, Arrow*, Backspace, etc.) carry their virtual key codes
so listeners checking e.keyCode / e.key all fire."""
vk, code, text = _KEYS.get(key, (ord(key[0]) if len(key) == 1 else 0, key, key if len(key) == 1 else ""))
base = {"key": key, "code": code, "modifiers": modifiers, "windowsVirtualKeyCode": vk, "nativeVirtualKeyCode": vk}
shortcut_modifiers = modifiers & (1 | 2 | 4) # Alt/Ctrl/Meta turn single keys into shortcuts.
printable_char = len(key) == 1 and bool(text) and not shortcut_modifiers
cdp("Input.dispatchKeyEvent", type="keyDown", **base, **({} if printable_char or not text else {"text": text}))
if printable_char:
cdp("Input.dispatchKeyEvent", type="char", text=text, **{k: v for k, v in base.items() if k != "text"})
cdp("Input.dispatchKeyEvent", type="keyUp", **base)
def scroll(x, y, dy=-300, dx=0):
cdp("Input.dispatchMouseEvent", type="mouseWheel", x=x, y=y, deltaX=dx, deltaY=dy)
# --- visual ---
def capture_screenshot(path=None, full=False, max_dim=None):
"""Save a PNG of the current viewport. Set max_dim=1800 on a 2× display to
keep the file under the 2000px-per-side limit some image-aware LLMs enforce."""
path = path or str(ipc._TMP / "shot.png")
r = cdp("Page.captureScreenshot", format="png", captureBeyondViewport=full)
open(path, "wb").write(base64.b64decode(r["data"]))
if max_dim:
from PIL import Image
img = Image.open(path)
if max(img.size) > max_dim:
img.thumbnail((max_dim, max_dim))
img.save(path)
return path
# --- tabs ---
def _is_agent_startup_placeholder(title, url):
url = str(url or "")
return str(title or "").startswith("Starting agent ") and (
url in ("", "about:blank") or url.startswith("about:blank#")
)
def list_tabs(include_chrome=True):
out = []
for t in cdp("Target.getTargets")["targetInfos"]:
if t["type"] != "page": continue
url = t.get("url", "")
if _is_agent_startup_placeholder(t.get("title", ""), url): continue
if not include_chrome and url.startswith(INTERNAL): continue
out.append({
"targetId": t["targetId"],
"target_id": t["targetId"],
"title": t.get("title", ""),
"url": url,
})
return out
def current_tab():
r = _send({"meta": "current_tab"})
return {
"targetId": r["targetId"],
"target_id": r["targetId"],
"url": r["url"],
"title": r["title"],
}
def _mark_tab():
"""Prepend horse emoji to tab title so the user can see which tab the agent controls."""
try: cdp("Runtime.evaluate", expression="if(!document.title.startsWith('\U0001F434'))document.title='\U0001F434 '+document.title")
except Exception: pass
def switch_tab(target):
# Accept either a raw targetId string or the dict returned by current_tab() / list_tabs(),
# so `switch_tab(current_tab())` works without a manual ["targetId"] dance.
target_id = (target.get("targetId") or target.get("target_id")) if isinstance(target, dict) else target
# Unmark old tab. Horse emoji is a surrogate pair in JS UTF-16 strings (2 code units),
# plus the trailing space = 3 code units, so slice(3) cleanly removes the prefix.
try: cdp("Runtime.evaluate", expression="if(document.title.startsWith('\U0001F434 '))document.title=document.title.slice(3)")
except Exception: pass
cdp("Target.activateTarget", targetId=target_id)
sid = cdp("Target.attachToTarget", targetId=target_id, flatten=True)["sessionId"]
_send({"meta": "set_session", "session_id": sid, "target_id": target_id})
_mark_tab()
return sid
def new_tab(url="about:blank"):
# Always create blank, then goto: passing url to createTarget races with
# attach, so the brief about:blank is "complete" by the time the caller
# polls and wait_for_load() returns before navigation actually starts.
if url != "about:blank":
try:
cur = current_tab()
cur_url = cur.get("url") or ""
# Reuse attached tab when it's blank
if (
cur_url in ("", "about:blank")
or cur_url.startswith("about:blank#")
or cur_url.startswith(("chrome://newtab", "chrome://new-tab-page", "edge://newtab", "about:newtab"))
):
goto_url(url)
return cur.get("targetId") or cur.get("target_id")
except Exception:
pass
tid = cdp("Target.createTarget", url="about:blank")["targetId"]
switch_tab(tid)
if url != "about:blank":
goto_url(url)
return tid
def close_tab(target=None):
"""Close a tab. If `target` is omitted, closes the currently attached tab.
Accepts a raw targetId string or a dict from list_tabs()/current_tab()."""
target_id = (target.get("targetId") or target.get("target_id")) if isinstance(target, dict) else target
if target_id is None:
target_id = current_tab()["targetId"]
cdp("Target.closeTarget", targetId=target_id)
def ensure_real_tab():
"""Switch to a real user tab if current is chrome:// / internal / stale."""
tabs = list_tabs(include_chrome=False)
if not tabs:
return None
try:
cur = current_tab()
if cur["url"] and not cur["url"].startswith(INTERNAL):
return cur
except Exception:
pass
switch_tab(tabs[0]["targetId"])
return tabs[0]
def iframe_target(url_substr):
"""First iframe target whose URL contains `url_substr`. Use with js(..., target_id=...)."""
for t in cdp("Target.getTargets")["targetInfos"]:
if t["type"] == "iframe" and url_substr in t.get("url", ""):
return t["targetId"]
return None
# --- utility ---
def wait(seconds=1.0):
time.sleep(seconds)
def wait_for_load(timeout=15.0):
"""Poll document.readyState == 'complete' or timeout."""
deadline = time.time() + timeout
while time.time() < deadline:
if js("document.readyState") == "complete": return True
time.sleep(0.3)
return False
def wait_for_element(selector, timeout=10.0, visible=False):
"""Poll until querySelector(selector) exists in the DOM, or timeout.
wait_for_load() misses SPAs — the document is 'complete' before the framework renders.
Use this after actions that trigger async rendering (route changes, data fetches).
Set visible=True to also require the element to be non-hidden and in-layout.
Returns True if found, False on timeout.
"""
if visible:
# checkVisibility walks the ancestor chain and respects display:none /
# visibility:hidden / opacity:0 on parents, which a getComputedStyle
# check on the element alone misses (it returns the descendant's own
# style, not the inherited "is this rendered" state). Falls back to
# the per-element CSS check on older Chrome that lacks checkVisibility.
check = (
f"(()=>{{const e=document.querySelector({json.dumps(selector)});"
f"if(!e)return false;"
f"if(typeof e.checkVisibility==='function')"
f"return e.checkVisibility({{checkOpacity:true,checkVisibilityCSS:true}});"
f"const s=getComputedStyle(e);"
f"return s.display!=='none'&&s.visibility!=='hidden'&&s.opacity!=='0'}})()"
)
else:
check = f"!!document.querySelector({json.dumps(selector)})"
deadline = time.time() + timeout
while time.time() < deadline:
if js(check): return True
time.sleep(0.3)
return False
def wait_for_network_idle(timeout=10.0, idle_ms=500):
"""Wait until all in-flight requests finish and no Network.* events arrive for idle_ms ms.
Useful after form submits, SPA route transitions, and any action that triggers
XHR/fetch without a visible DOM change. Builds on drain_events() — no daemon changes.
Returns True if idle window reached, False on timeout.
Events are filtered to the active session — a previously-attached background
tab (e.g. a polling/SSE page the agent switched away from) keeps emitting
Network events into the daemon's global event buffer; without this filter
they would poison the idle check on the current tab.
"""
deadline = time.time() + timeout
last_activity = time.time()
inflight = set()
active_session = _send({"meta": "session"}).get("session_id")
while time.time() < deadline:
for e in drain_events():
if e.get("session_id") != active_session:
continue
method = e.get("method", "")
params = e.get("params", {})
if method == "Network.requestWillBeSent":
inflight.add(params.get("requestId"))
last_activity = time.time()
elif method in ("Network.loadingFinished", "Network.loadingFailed"):
inflight.discard(params.get("requestId"))
last_activity = time.time()
elif method.startswith("Network."):
last_activity = time.time()
if not inflight and (time.time() - last_activity) * 1000 >= idle_ms:
return True
time.sleep(0.1)
return False
def js(expression, target_id=None):
"""Run JS in the attached tab (default) or inside an iframe target (via iframe_target()).
Expressions are evaluated as-is first. If Chrome reports an illegal top-level
`return`, the snippet is retried inside a function wrapper, so both
`document.title` and `const x = 1; return x` work without mis-wrapping nested
functions that contain their own returns.
"""
sid = cdp("Target.attachToTarget", targetId=target_id, flatten=True)["sessionId"] if target_id else None
try:
return _runtime_evaluate(expression, session_id=sid, await_promise=True)
except RuntimeError as e:
if _is_illegal_return_error(e):
return _runtime_evaluate(_wrap_js_function(expression), session_id=sid, await_promise=True)
raise
_KC = {"Enter": 13, "Tab": 9, "Escape": 27, "Backspace": 8, " ": 32, "ArrowLeft": 37, "ArrowUp": 38, "ArrowRight": 39, "ArrowDown": 40}
def dispatch_key(selector, key="Enter", event="keypress"):
"""Dispatch a DOM KeyboardEvent on the matched element.
Use this when a site reacts to synthetic DOM key events on an element more reliably
than to raw CDP input events.
"""
kc = _KC.get(key, ord(key) if len(key) == 1 else 0)
js(
f"(()=>{{const e=document.querySelector({json.dumps(selector)});if(e){{e.focus();e.dispatchEvent(new KeyboardEvent({json.dumps(event)},{{key:{json.dumps(key)},code:{json.dumps(key)},keyCode:{kc},which:{kc},bubbles:true}}));}}}})()"
)
def upload_file(selector, path):
"""Set files on a file input via CDP DOM.setFileInputFiles. `path` is an absolute filepath (use tempfile.mkstemp if needed)."""
doc = cdp("DOM.getDocument", depth=-1)
nid = cdp("DOM.querySelector", nodeId=doc["root"]["nodeId"], selector=selector)["nodeId"]
if not nid: raise RuntimeError(f"no element for {selector}")
cdp("DOM.setFileInputFiles", files=[path] if isinstance(path, str) else list(path), nodeId=nid)
def http_get(url, headers=None, timeout=20.0):
"""Pure HTTP — no browser. Use for static pages / APIs. Wrap in ThreadPoolExecutor for bulk.
When BROWSER_USE_API_KEY is set, routes through the fetch-use proxy (handles bot
detection, residential proxies, retries). Falls back to local urllib otherwise."""
if os.environ.get("BROWSER_USE_API_KEY"):
try:
from fetch_use import fetch_sync
return fetch_sync(url, headers=headers, timeout_ms=int(timeout * 1000)).text
except ImportError:
pass
import gzip
h = {"User-Agent": "Mozilla/5.0", "Accept-Encoding": "gzip"}
if headers: h.update(headers)
with urllib.request.urlopen(urllib.request.Request(url, headers=h), timeout=timeout) as r:
data = r.read()
if r.headers.get("Content-Encoding") == "gzip": data = gzip.decompress(data)
return data.decode()
# Imported at the bottom so recorder's own `from . import helpers` sees a
# fully-defined module. Exposes the recording helpers via `from .helpers import *`.
from .recorder import start_recording, stop_recording, recording_dir
def _load_agent_helpers():
p = AGENT_WORKSPACE / "agent_helpers.py"
if not p.exists():
return
spec = importlib.util.spec_from_file_location("browser_harness_agent_helpers", p)
if not spec or not spec.loader:
return
module = importlib.util.module_from_spec(spec)
spec.loader.exec_module(module)
for name, value in vars(module).items():
if name.startswith("_"):
continue
globals()[name] = value
_load_agent_helpers()