{"id":"0e8ffd1a-a74f-406e-b523-e2fac9a924e2","entityType":"agent","slug":"clawhub-kd-oauth-desktop-control-for-macos","name":"MacOS Desktop Control","canonicalUrl":"https://www.xpersona.co/agent/clawhub-kd-oauth-desktop-control-for-macos","canonicalPath":"/agent/clawhub-kd-oauth-desktop-control-for-macos","generatedAt":"2026-10-10T10:43:38.389Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":null},"description":"Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows. Skill: MacOS Desktop Control Owner: kd-oauth Summary: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows. Tags: latest:1.1.2 Version history: v1.1.2 | 2026-05-13T12:03:35.667Z | user desktop-control-for-macos 1.1.2 - Updated documentation to clarify using AI semantic understanding as the default for target location, with fallback to OCR or","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.6K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s171c10bn0byt7ma9sxfcx56p9853enj:desktop-control-for-macos","sourceUrl":"https://clawhub.ai/kd-oauth/desktop-control-for-macos","homepage":"https://clawhub.ai/kd-oauth/skills/desktop-control-for-macos","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/kd-oauth/desktop-control-for-macos","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/kd-oauth/skills/desktop-control-for-macos","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":64,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows. Skill: MacOS Desktop Control O"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":null},"stars":null,"forks":null,"downloads":1562,"packageName":null,"latestVersion":"1.1.2","tractionLabel":"1.6K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T08:21:36.387Z","lastCrawledAt":"2026-10-10T08:21:36.387Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T08:21:36.387Z","lastVerifiedAt":null,"highlights":[{"version":"1.1.2","createdAt":"2026-05-13T12:03:35.667Z","changelog":"desktop-control-for-macos 1.1.2 - Updated documentation to clarify using AI semantic understanding as the default for target location, with fallback to OCR or OpenCV when necessary. - No file or implementation changes; the update is documentation-only. - Maintains all previous features and workflow conventions.","fileCount":14,"zipByteSize":19250},{"version":"1.1.1","createdAt":"2026-05-06T12:19:01.676Z","changelog":"No changes detected in this version. - Version 1.1.1 includes no file modifications compared to the previous version. - No feature, documentation, or implementation changes recorded.","fileCount":13,"zipByteSize":17434},{"version":"1.1.0","createdAt":"2026-04-30T08:29:05.121Z","changelog":"No code or functionality changes detected in this version. - Documentation expanded: Added a new section describing \"Locate by semantic understanding\" for visual target location via AI image analysis. - Clarified usage scenarios where AI-based target detection is preferable to traditional OCR or template matching. - No changes to scripts, APIs, or requirements; all updates are to documentation only.","fileCount":13,"zipByteSize":17622},{"version":"1.0.13","createdAt":"2026-04-22T07:02:52.516Z","changelog":"Version 1.0.13 - Introduces automatic (\"lazy\") initialization of calibration: the skill now auto-generates the screen calibration file on first use if it does not exist, removing the need for manual pre-initialization. - Calibration file at /tmp/macos_desktop_control/calibration.json is reused if already present. - Scripts that require calibration (e.g., capture_screen.py, mouse.py, locate_text_ocr.py, locate_image_opencv.py) will trigger auto-initialization when needed. - Slightly expands usage documentation and clarifies default behaviors for new initialization logic. - No code changes detected; update is documentation-only.","fileCount":13,"zipByteSize":17006},{"version":"1.0.12","createdAt":"2026-04-22T06:13:25.404Z","changelog":"Remove the description related to non-Retina displays.","fileCount":13,"zipByteSize":15567},{"version":"1.0.11","createdAt":"2026-04-21T12:21:16.520Z","changelog":"- Added a prominent introductory note highlighting Chinese compatibility for text input/recognition. - All other technical features and documentation remain unchanged.","fileCount":13,"zipByteSize":15624},{"version":"1.0.10","createdAt":"2026-04-21T12:19:29.406Z","changelog":"keyboard改成只复制粘贴，用来兼容中文输入","fileCount":13,"zipByteSize":15501},{"version":"1.0.9","createdAt":"2026-04-21T10:43:45.260Z","changelog":"- Added skill metadata block (`name` and `description`) to SKILL.md. - Fixed appleScript app frontmost.","fileCount":13,"zipByteSize":15422}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s171c10bn0byt7ma9sxfcx56p9853enj:desktop-control-for-macos","setupComplexity":"medium","setupSteps":["Python environment detected. Create a strict virtual environment (`python -m venv .venv`) before installing dependencies to prevent system-level package conflicts.","Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T10:43:38.385Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-kd-oauth-desktop-control-for-macos/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":null},"readme":"Skill: MacOS Desktop Control\n\nOwner: kd-oauth\n\nSummary: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n\nTags: latest:1.1.2\n\nVersion history:\n\nv1.1.2 | 2026-05-13T12:03:35.667Z | user\n\ndesktop-control-for-macos 1.1.2\n\n- Updated documentation to clarify using AI semantic understanding as the default for target location, with fallback to OCR or OpenCV when necessary.\n- No file or implementation changes; the update is documentation-only.\n- Maintains all previous features and workflow conventions.\n\nv1.1.1 | 2026-05-06T12:19:01.676Z | user\n\nNo changes detected in this version.\n\n- Version 1.1.1 includes no file modifications compared to the previous version.\n- No feature, documentation, or implementation changes recorded.\n\nv1.1.0 | 2026-04-30T08:29:05.121Z | user\n\nNo code or functionality changes detected in this version.\n\n- Documentation expanded: Added a new section describing \"Locate by semantic understanding\" for visual target location via AI image analysis.\n- Clarified usage scenarios where AI-based target detection is preferable to traditional OCR or template matching.\n- No changes to scripts, APIs, or requirements; all updates are to documentation only.\n\nv1.0.13 | 2026-04-22T07:02:52.516Z | user\n\nVersion 1.0.13\n\n- Introduces automatic (\"lazy\") initialization of calibration: the skill now auto-generates the screen calibration file on first use if it does not exist, removing the need for manual pre-initialization.\n- Calibration file at /tmp/macos_desktop_control/calibration.json is reused if already present.\n- Scripts that require calibration (e.g., capture_screen.py, mouse.py, locate_text_ocr.py, locate_image_opencv.py) will trigger auto-initialization when needed.\n- Slightly expands usage documentation and clarifies default behaviors for new initialization logic.\n- No code changes detected; update is documentation-only.\n\nv1.0.12 | 2026-04-22T06:13:25.404Z | user\n\nRemove the description related to non-Retina displays.\n\nv1.0.11 | 2026-04-21T12:21:16.520Z | user\n\n- Added a prominent introductory note highlighting Chinese compatibility for text input/recognition.\n- All other technical features and documentation remain unchanged.\n\nv1.0.10 | 2026-04-21T12:19:29.406Z | user\n\nkeyboard改成只复制粘贴，用来兼容中文输入\n\nv1.0.9 | 2026-04-21T10:43:45.260Z | user\n\n- Added skill metadata block (`name` and `description`) to SKILL.md.\n- Fixed appleScript app frontmost.\n\nv1.0.8 | 2026-04-21T10:23:01.377Z | user\n\nAdded compatibility support for Chinese text input.\n\nv1.0.7 | 2026-04-21T08:39:39.166Z | user\n\n### Changelog for version 1.0.7\n\n- Updated documentation to clarify features by adding icon bullets to each feature set for improved readability.\n- No code or logic changes; only the SKILL.md file was modified.\n- All features and usage remain unchanged.\n\nv1.0.6 | 2026-04-21T08:35:14.683Z | user\n\nNo functional changes in this release. Documentation (SKILL.md) has been updated:\n\n- Added a new \"Features\" section summarizing capabilities for easier overview\n- Reorganized documentation to clarify groups of actions (app/window, visual control, keyboard, mouse, safety)\n- No changes to scripts, APIs, or functionality\n\nv1.0.5 | 2026-04-21T08:30:37.591Z | user\n\n- Added image region cropping support with new script `scripts/crop_image.py`\n- Documented usage for cropping logical screen regions to extract thumbnails, buttons, dialogs, etc.\n- Updated directory structure and recommended flow to include the crop image functionality\n\nv1.0.4 | 2026-04-21T03:56:20.582Z | user\n\n**AppleScript app and window control added for semantic macOS automation.**\n\n- Added scripts/applescript_app.py and scripts/applescript_window.py for app-level and window-level control using AppleScript.\n- Expanded documentation to clarify skill boundaries, explaining when to use AppleScript versus screen-based automation.\n- Listed and documented new actions: open, activate, check if running, and get frontmost app for apps; get title, count, and list windows.\n- Directory layout and usage flow updated to include new AppleScript capabilities.\n\nv1.0.2 | 2026-04-20T11:57:54.976Z | user\n\nVersion 1.0.1\n\n- Added meta information file (_meta.json) for the skill.\n- Updated documentation: OCR now uses Apple Vision via PyObjC, removing the need for separate Tesseract installation.\n- Clarified in SKILL.md that OCR relies on Apple Vision, and updated related usage notes.\n- No code changes; this version only updates metadata and documentation for improved clarity and ease of use.\n\nv1.0.1 | 2026-04-20T10:34:41.453Z | user\n\nEnhance mouse control functionality and add keyboard control.\n\nv1.0.0 | 2026-04-19T16:15:52.748Z | user\n\ninit\n\nArchive index:\n\nArchive v1.1.2: 14 files, 19250 bytes\n\nFiles: _meta.json (144b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (2017b), scripts/capture_screen.py (949b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (2114b), scripts/locate_text_ocr.py (6310b), scripts/mouse.py (5142b), skill-card.md (2120b), SKILL.md (17208b)\n\nFile v1.1.2:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate targets by AI semantic understanding as the default first choice\n- ✅ Fall back to OCR when the target is best identified by text\n- ✅ Fall back to OpenCV image matching when the target has a stable reusable template\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by AI semantic understanding first, then fall back to OCR or OpenCV when needed\n5. Execute mouse and keyboard actions with Python\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nInitialization behavior in the current version:\n- the skill auto-initializes on first use when the calibration file does not exist\n- it does not re-run mapping on every invocation\n- if `/tmp/macos_desktop_control/calibration.json` already exists, the existing calibration is reused\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\nYou can still run initialization manually:\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nBut in normal use, the skill now performs lazy initialization automatically on first use if the calibration file is missing.\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\nCurrent lazy-init behavior:\n- `capture_screen.py`\n- `mouse.py`\n- `locate_text_ocr.py`\n- `locate_image_opencv.py`\n\nThese scripts first check whether `/tmp/macos_desktop_control/calibration.json` exists.\nIf not, they auto-generate it once and then continue.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nUse AI semantic understanding as the default first-choice locator.\nUse OCR or OpenCV only when AI is unsuitable, unavailable, or cannot produce a reliable target.\n\nThere are three supported strategies.\n\n### Locate by semantic understanding\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"Confirm\"\n```\n\nYou can also constrain OCR to a specific screen region when the same text may appear in multiple places:\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"Chats\" \\\n  --x1 0 --y1 120 --x2 520 --y2 1107\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\nWhen a region is provided, the search runs only inside that rectangle, but the returned coordinates are still in full-screen logical coordinates.\n\nWhen exact matching methods are unnecessary or brittle, use AI-based image understanding first.\n\nThis approach is the default when the target has one or more of these properties: its appearance is not fixed, there is no reusable template, the decision depends on surrounding visual context, or the target can only be described semantically.\n\nTypical examples:\n- there are multiple clickable regions on screen and the caller must determine which one is the intended target\n- the target has no stable text label or icon template, but it can be described in natural language\n- the workflow requires understanding relationships between visual elements, such as who is speaking or what a region means in context\n- the target is easier to describe than to match exactly\n\nPreferred flow:\n- capture a screenshot\n- crop to the smallest reliable region when possible\n- run AI recognition on that region first\n- require structured output from AI before acting\n- if AI cannot produce a reliable target, fall back to OCR or OpenCV\n- map coordinates, then execute the action\n\nAI output contract:\n- require a compact JSON object\n- required fields: `found`, `x`, `y`, `confidence`, `reason`\n- `found` must be boolean\n- `x` and `y` must be logical screen coordinates when `found=true`\n- `confidence` should use a small fixed set such as `high`, `medium`, `low`\n- `reason` should briefly explain why the target was selected or why no reliable target was found\n\nRecommended JSON shape:\n\n```json\n{\n  \"found\": true,\n  \"x\": 742,\n  \"y\": 681,\n  \"confidence\": \"high\",\n  \"reason\": \"Located the send button in the lower-right input area\"\n}\n```\n\nSafety rule for AI-driven clicks:\n- if `found=false`, do not click\n- if `confidence=low`, prefer fallback or verification before clicking\n\nThis skill is responsible for screenshot capture and coordinate conversion. The caller interprets the recognition result and decides the next action.\n\n### Locate by OCR text\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"I am OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf 'I am OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"WeChat\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/WeChat.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"WeChat\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"WeChat\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"WeChat\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"WeChat\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"WeChat\"\npython scripts/applescript_window.py --action title --app \"WeChat\" --json-pretty\n```\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\nUse this as a fallback when AI is not appropriate and the target has a stable reusable visual template.\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + desktop vision + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nDefault visual targeting order in this skill:\n1. Try AI semantic understanding first\n2. If AI cannot produce a reliable target, fall back to OCR for text-driven targets\n3. If OCR is not suitable, fall back to OpenCV template matching for stable visual templates\n\nWhen the same text may appear in multiple places, do not search the full screen by default.\nConstrain OCR to the intended region first, then click using the returned full-screen logical coordinates.\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. AI-based vision tries to find the target first\n4. OCR or OpenCV is used only as fallback when needed\n5. mouse or keyboard automation performs the action\n6. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"WeChat\"\npython scripts/applescript_window.py --action title --app \"WeChat\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\n# first try AI semantic understanding with a bounded screenshot region when possible\n# if AI cannot produce a reliable target, fall back to OCR or OpenCV\npython scripts/locate_text_ocr.py --text \"Confirm\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- Treat AI semantic understanding as the default visual locator, not merely a last-resort add-on.\n- Require structured AI output with `found`, `x`, `y`, `confidence`, and `reason` before acting on AI-located targets.\n- Use OCR and OpenCV as fallback tools when AI cannot reliably identify the target.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.1.2:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.1.2\",\n  \"publishedAt\": 1778673815667\n}\n\nFile v1.1.2:skill-card.md\n\n## Description:\n\nGeneric macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[kd-oauth](https://clawhub.ai/user/kd-oauth)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and automation agents use this skill to control a macOS desktop session through app/window inspection, screenshot capture, OCR or image matching, and mouse or keyboard actions.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can control the screen, mouse, keyboard, clipboard, and AppleScript in the active macOS session.\n\nMitigation: Use it only in a dedicated, low-sensitivity macOS session with explicit Screen Recording and Accessibility permissions.\n\nRisk: Pasted text or generated screenshots may expose sensitive information.\n\nMitigation: Avoid pasting secrets, clear or protect generated screenshots, and verify target fields before sending externally visible input.\n\nRisk: Untrusted app names can trigger AppleScript interpolation risk until the issue is fixed.\n\nMitigation: Use only trusted app names or bundle paths and review the AppleScript command path before installation.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/kd-oauth/skills/desktop-control-for-macos)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell command examples; scripts may emit JSON coordinates, file paths, and image files.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires macOS Screen Recording and Accessibility permissions; OCR uses Apple Vision through PyObjC.]\n\n## Skill Version(s):\n\n1.1.2 (source: server release evidence and target metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v1.1.2:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.1.1: 13 files, 17434 bytes\n\nFiles: _meta.json (144b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (2017b), scripts/capture_screen.py (949b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (2114b), scripts/locate_text_ocr.py (6310b), scripts/mouse.py (5142b), SKILL.md (15145b)\n\nFile v1.1.1:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nInitialization behavior in the current version:\n- the skill auto-initializes on first use when the calibration file does not exist\n- it does not re-run mapping on every invocation\n- if `/tmp/macos_desktop_control/calibration.json` already exists, the existing calibration is reused\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\nYou can still run initialization manually:\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nBut in normal use, the skill now performs lazy initialization automatically on first use if the calibration file is missing.\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\nCurrent lazy-init behavior:\n- `capture_screen.py`\n- `mouse.py`\n- `locate_text_ocr.py`\n- `locate_image_opencv.py`\n\nThese scripts first check whether `/tmp/macos_desktop_control/calibration.json` exists.\nIf not, they auto-generate it once and then continue.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"Confirm\"\n```\n\nYou can also constrain OCR to a specific screen region when the same text may appear in multiple places:\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"Chats\" \\\n  --x1 0 --y1 120 --x2 520 --y2 1107\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\nWhen a region is provided, the search runs only inside that rectangle, but the returned coordinates are still in full-screen logical coordinates.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n### Locate by semantic understanding\n\nWhen exact matching methods such as OpenCV template matching or OCR text lookup cannot reliably cover a target, use AI-based image understanding.\n\nThis approach is a good fit when the target has one or more of these properties: its appearance is not fixed, there is no reusable template, the decision depends on surrounding visual context, or the target can only be described semantically.\n\nTypical examples:\n- there are multiple clickable regions on screen and the caller must determine which one is the intended target\n- the target has no stable text label or icon template, but it can be described in natural language\n- the workflow requires understanding relationships between visual elements, such as who is speaking or what a region means in context\n\nThe operational flow is the same as the two strategies above: capture a screenshot, crop to a bounded region if needed, run AI recognition, map coordinates, then execute the action. This skill is responsible for screenshot capture and coordinate conversion. The caller interprets the recognition result and decides the next action.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"I am OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf 'I am OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"WeChat\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/WeChat.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"WeChat\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"WeChat\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"WeChat\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"WeChat\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"WeChat\"\npython scripts/applescript_window.py --action title --app \"WeChat\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nWhen the same text may appear in multiple places, do not search the full screen by default.\nConstrain OCR to the intended region first, then click using the returned full-screen logical coordinates.\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"WeChat\"\npython scripts/applescript_window.py --action title --app \"WeChat\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"Confirm\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.1.1:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.1.1\",\n  \"publishedAt\": 1778069941676\n}\n\nFile v1.1.1:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.1.0: 13 files, 17622 bytes\n\nFiles: _meta.json (144b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (2017b), scripts/capture_screen.py (949b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (2114b), scripts/locate_text_ocr.py (6310b), scripts/mouse.py (5142b), SKILL.md (14825b)\n\nFile v1.1.0:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nInitialization behavior in the current version:\n- the skill auto-initializes on first use when the calibration file does not exist\n- it does not re-run mapping on every invocation\n- if `/tmp/macos_desktop_control/calibration.json` already exists, the existing calibration is reused\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\nYou can still run initialization manually:\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nBut in normal use, the skill now performs lazy initialization automatically on first use if the calibration file is missing.\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\nCurrent lazy-init behavior:\n- `capture_screen.py`\n- `mouse.py`\n- `locate_text_ocr.py`\n- `locate_image_opencv.py`\n\nThese scripts first check whether `/tmp/macos_desktop_control/calibration.json` exists.\nIf not, they auto-generate it once and then continue.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nYou can also constrain OCR to a specific screen region when the same text may appear in multiple places:\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"会话\" \\\n  --x1 0 --y1 120 --x2 520 --y2 1107\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\nWhen a region is provided, the search runs only inside that rectangle, but the returned coordinates are still in full-screen logical coordinates.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n### Locate by semantic understanding\n\n精确匹配（OpenCV 模板匹配、OCR 文字定位）覆盖不了的目标，可以用 AI 图像理解来处理。\n\n适合交给 AI 判断的场景通常具有以下特征：形态不固定、没有现成模板、需要结合画面上下文、或者只能用语义词汇描述。\n\n典型场景举例：\n- 画面中有多个可点击区域，需要判断哪个是当前目标\n- 目标没有固定文字或图标模板，但可以用语义描述\n- 需要理解画面中各元素之间的关系（谁在说话、某块区域的含义）\n\n操作流程与上述两种方式一致：截图 → crop 限定区域（如需要）→ AI 识别 → 坐标映射 → 动作执行。本 skill 负责截图和坐标转换，识别结果由调用者解读并决定后续动作。\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"我是OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf '我是OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nWhen the same text may appear in multiple places, do not search the full screen by default.\nConstrain OCR to the intended region first, then click using the returned full-screen logical coordinates.\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.1.0\",\n  \"publishedAt\": 1777537745121\n}\n\nFile v1.1.0:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.13: 13 files, 17006 bytes\n\nFiles: _meta.json (145b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (2017b), scripts/capture_screen.py (949b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (2114b), scripts/locate_text_ocr.py (6310b), scripts/mouse.py (5142b), SKILL.md (14018b)\n\nFile v1.0.13:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nInitialization behavior in the current version:\n- the skill auto-initializes on first use when the calibration file does not exist\n- it does not re-run mapping on every invocation\n- if `/tmp/macos_desktop_control/calibration.json` already exists, the existing calibration is reused\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\nYou can still run initialization manually:\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nBut in normal use, the skill now performs lazy initialization automatically on first use if the calibration file is missing.\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\nCurrent lazy-init behavior:\n- `capture_screen.py`\n- `mouse.py`\n- `locate_text_ocr.py`\n- `locate_image_opencv.py`\n\nThese scripts first check whether `/tmp/macos_desktop_control/calibration.json` exists.\nIf not, they auto-generate it once and then continue.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nYou can also constrain OCR to a specific screen region when the same text may appear in multiple places:\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"会话\" \\\n  --x1 0 --y1 120 --x2 520 --y2 1107\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\nWhen a region is provided, the search runs only inside that rectangle, but the returned coordinates are still in full-screen logical coordinates.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"我是OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf '我是OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nWhen the same text may appear in multiple places, do not search the full screen by default.\nConstrain OCR to the intended region first, then click using the returned full-screen logical coordinates.\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.13:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.13\",\n  \"publishedAt\": 1776841372516\n}\n\nFile v1.0.13:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.12: 13 files, 15567 bytes\n\nFiles: _meta.json (145b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (1175b), scripts/capture_screen.py (785b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (1950b), scripts/locate_text_ocr.py (4547b), scripts/mouse.py (4964b), SKILL.md (12692b)\n\nFile v1.0.12:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"我是OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf '我是OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.12:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.12\",\n  \"publishedAt\": 1776838405404\n}\n\nFile v1.0.12:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.11: 13 files, 15624 bytes\n\nFiles: _meta.json (145b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (1175b), scripts/capture_screen.py (785b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (1950b), scripts/locate_text_ocr.py (4547b), scripts/mouse.py (4964b), SKILL.md (12842b)\n\nFile v1.0.11:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\nThe first version focuses on Retina displays. Other cases such as non-Retina displays, scaled displays, and multi-monitor setups can be added later.\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"我是OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf '我是OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.11:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.11\",\n  \"publishedAt\": 1776774076520\n}\n\nFile v1.0.11:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.10: 13 files, 15501 bytes\n\nFiles: _meta.json (145b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (1175b), scripts/capture_screen.py (785b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (3189b), scripts/locate_image_opencv.py (1950b), scripts/locate_text_ocr.py (4547b), scripts/mouse.py (4964b), SKILL.md (12739b)\n\nFile v1.0.10:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\nThe first version focuses on Retina displays. Other cases such as non-Retina displays, scaled displays, and multi-monitor setups can be added later.\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to paste text or trigger shortcuts.\n\nImportant practical note:\n- this skill uses clipboard paste for all text entry by default, including English\n- this avoids input-method issues with Chinese, English, and mixed-language text\n- do not use simulated typing for text entry in this skill\n\n### Paste text\n\n```bash\npython scripts/keyboard.py --action paste --text \"我是OpenClaw\"\n```\n\n### Paste from stdin\n\n```bash\nprintf '我是OpenClaw' | python scripts/keyboard.py --action paste --stdin\n```\n\nDefault input rule for this skill:\n- use clipboard paste for all text input by default, including English\n- click the verified input field first, then paste with `command v`\n- do not use simulated typing for text entry in this skill\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard, preferably via `python scripts/keyboard.py --action paste`\n2. click the verified input field\n3. let the script send `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.10:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.10\",\n  \"publishedAt\": 1776773969406\n}\n\nFile v1.0.10:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.9: 13 files, 15422 bytes\n\nFiles: _meta.json (144b), requirements.txt (114b), scripts/applescript_app.py (3233b), scripts/applescript_window.py (2823b), scripts/calibration.py (1175b), scripts/capture_screen.py (785b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (2861b), scripts/locate_image_opencv.py (1950b), scripts/locate_text_ocr.py (4547b), scripts/mouse.py (4964b), SKILL.md (12632b)\n\nFile v1.0.9:SKILL.md\n\n---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\nThe first version focuses on Retina displays. Other cases such as non-Retina displays, scaled displays, and multi-monitor setups can be added later.\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to type text or trigger shortcuts.\n\n### Type text\n\n```bash\npython scripts/keyboard.py --action write --text \"hello\"\n```\n\nImportant practical note:\n- `write` simulates keystroke-by-keystroke text entry, not clipboard paste\n- this can be affected by the current input method, candidate bar state, full-width or half-width mode, and whether the target field really has focus\n- for Chinese text, mixed-language text, or any text where fidelity matters, prefer clipboard paste over `write`\n- reserve `write` for short plain English text, simple fields, or cases where paste is not appropriate\n\n### Type from stdin\n\n```bash\nprintf 'hello world' | python scripts/keyboard.py --action write --stdin\n```\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard\n2. click the verified input field\n3. use `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.9:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.9\",\n  \"publishedAt\": 1776768225260\n}\n\nFile v1.0.9:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.8: 13 files, 15343 bytes\n\nFiles: _meta.json (144b), requirements.txt (114b), scripts/applescript_app.py (3078b), scripts/applescript_window.py (2823b), scripts/calibration.py (1175b), scripts/capture_screen.py (785b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (2861b), scripts/locate_image_opencv.py (1950b), scripts/locate_text_ocr.py (4547b), scripts/mouse.py (4964b), SKILL.md (12448b)\n\nFile v1.0.8:SKILL.md\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\nThe first version focuses on Retina displays. Other cases such as non-Retina displays, scaled displays, and multi-monitor setups can be added later.\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to type text or trigger shortcuts.\n\n### Type text\n\n```bash\npython scripts/keyboard.py --action write --text \"hello\"\n```\n\nImportant practical note:\n- `write` simulates keystroke-by-keystroke text entry, not clipboard paste\n- this can be affected by the current input method, candidate bar state, full-width or half-width mode, and whether the target field really has focus\n- for Chinese text, mixed-language text, or any text where fidelity matters, prefer clipboard paste over `write`\n- reserve `write` for short plain English text, simple fields, or cases where paste is not appropriate\n\n### Type from stdin\n\n```bash\nprintf 'hello world' | python scripts/keyboard.py --action write --stdin\n```\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\nRecommended paste workflow when text fidelity matters:\n1. copy the exact text into the clipboard\n2. click the verified input field\n3. use `command v` to paste\n4. verify visually before pressing enter if sending would be externally visible\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.8:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.8\",\n  \"publishedAt\": 1776766981377\n}\n\nFile v1.0.8:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0\n\nArchive v1.0.7: 13 files, 15016 bytes\n\nFiles: _meta.json (144b), requirements.txt (114b), scripts/applescript_app.py (3078b), scripts/applescript_window.py (2823b), scripts/calibration.py (1175b), scripts/capture_screen.py (785b), scripts/crop_image.py (1269b), scripts/init_coordinate_mapping.py (1435b), scripts/keyboard.py (2861b), scripts/locate_image_opencv.py (1950b), scripts/locate_text_ocr.py (4547b), scripts/mouse.py (4964b), SKILL.md (11740b)\n\nFile v1.0.7:SKILL.md\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate text by OCR on screenshots\n- ✅ Locate templates by OpenCV image matching\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by OCR or OpenCV image matching\n5. Execute mouse and keyboard actions with Python\n\nThe first version focuses on Retina displays. Other cases such as non-Retina displays, scaled displays, and multi-monitor setups can be added later.\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one point is commonly equal to two pixels.\n\nThis skill writes the coordinate mapping result to a JSON file, so later steps can reuse it without recalculating.\n\nDefault calibration file:\n\n```bash\n/tmp/macos_desktop_control/calibration.json\n```\n\n## Directory layout\n\n```text\nmacos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py\n```\n\n## Requirements\n\nInstall Python dependencies:\n\n```bash\npip install -r requirements.txt\n```\n\nOCR uses Apple Vision through PyObjC, so no separate Tesseract install is required.\n\nOn macOS, grant the terminal or runtime app these permissions:\n\n- Screen Recording\n- Accessibility\n\n## 1. Initialize coordinate mapping\n\nThe first version handles Retina screens by comparing screenshot pixel size with the logical screen size used by `pyautogui`.\n\n```bash\npython scripts/init_coordinate_mapping.py\n```\n\nExample output:\n\n```json\n{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}\n```\n\nLater scripts read this file automatically.\n\n## 2. Capture screen\n\nCapture the current screen and resize the image into the logical coordinate system used by `pyautogui.position()` and `pyautogui.click()`.\n\nThis skill's default convention is:\n- default screenshot is logical\n- default recognition result coordinates are logical\n- default mouse action coordinates are logical\n- default crop operations should use a logical screenshot\n- only use calibration conversion when a workflow explicitly mixes logical screenshots with raw pixel screenshots\n\n```bash\npython scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png\n```\n\nCore idea:\n\n```python\nimport pyautogui\n\nimg = pyautogui.screenshot()\nscreen_w, screen_h = pyautogui.size()\n\n# Resize screenshot to the coordinate system used by pyautogui.position() / click().\nimg = img.resize((screen_w, screen_h))\nimg.save(\"screen_logical.png\")\n```\n\n## 3. Crop image regions\n\nWhen a higher-level skill already knows a target rectangle, crop it directly instead of re-opening previews or re-running visual search.\n\nBy default, crop from a logical screenshot so the crop rectangle stays in the same coordinate system as recognition and mouse targeting.\nOnly crop from a raw Retina or pixel screenshot when there is a specific reason to preserve raw pixels, and in that case convert coordinates first using calibration data.\n\n```bash\npython scripts/crop_image.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --x1 400 --y1 300 --x2 700 --y2 650 \\\n  --output /tmp/macos_desktop_control/crop.png\n```\n\nUse this for:\n- extracting a detected chat image thumbnail\n- saving a button or dialog region for later analysis\n- debugging screenshot-to-action pipelines\n\n## 4. Locate targets\n\nThere are two supported strategies.\n\n### Locate by OCR text\n\n```bash\npython scripts/locate_text_ocr.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --text \"确定\"\n```\n\nThe script prints the center point of the best matched Apple Vision OCR box.\n\n### Locate by OpenCV image matching\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n  --threshold 0.8\n```\n\nThe script prints the center point of the matched template.\n\n## 5. Mouse actions\n\nUse Python and `pyautogui` to control the mouse in logical screen coordinates.\n\n### Single click\n\n```bash\npython scripts/mouse.py --action click --x 500 --y 300\n```\n\n### Move only\n\n```bash\npython scripts/mouse.py --action move --x 500 --y 300 --duration 0.2\n```\n\n### Double click\n\n```bash\npython scripts/mouse.py --action double-click --x 500 --y 300\n```\n\n### Right click\n\n```bash\npython scripts/mouse.py --action right-click --x 500 --y 300\n```\n\n### Drag\n\n```bash\npython scripts/mouse.py --action drag --x 500 --y 300 --to-x 800 --to-y 500 --duration 0.3\n```\n\n### Read current mouse position\n\n```bash\npython scripts/mouse.py --action position\n```\n\nYou can also pipe the result from a locate script:\n\n```bash\npython scripts/locate_image_opencv.py \\\n  --image /tmp/macos_desktop_control/screen_logical.png \\\n  --template ./target_button.png \\\n| python scripts/mouse.py --stdin --action click\n```\n\nStdin accepts either `x y` text or JSON like `{\"x\": 500, \"y\": 300}`.\n\n## 6. Keyboard actions\n\nUse Python and `pyautogui` to type text or trigger shortcuts.\n\n### Type text\n\n```bash\npython scripts/keyboard.py --action write --text \"hello\"\n```\n\n### Type from stdin\n\n```bash\nprintf 'hello world' | python scripts/keyboard.py --action write --stdin\n```\n\n### Press one key\n\n```bash\npython scripts/keyboard.py --action press --key enter\n```\n\n### Press a hotkey\n\n```bash\npython scripts/keyboard.py --action hotkey --keys command v\n```\n\n### Hold and release keys\n\n```bash\npython scripts/keyboard.py --action key-down --key shift\npython scripts/keyboard.py --action key-up --key shift\n```\n\n## 7. AppleScript app control\n\nUse AppleScript when the task is semantic macOS control rather than visual targeting.\n\nGood fits:\n- open or activate an app\n- check whether an app is running\n- read the current frontmost app\n\n### Open by app name\n\n```bash\npython scripts/applescript_app.py --action open --app \"微信\"\n```\n\n### Open by bundle path\n\n```bash\npython scripts/applescript_app.py --action open --path \"/Applications/微信.app\"\n```\n\n### Activate an app\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\n```\n\n### Check whether an app is running\n\n```bash\npython scripts/applescript_app.py --action is-running --app \"微信\"\n```\n\n### Get the current frontmost app\n\n```bash\npython scripts/applescript_app.py --action frontmost-app\npython scripts/applescript_app.py --action frontmost-app --json-pretty\n```\n\n## 8. AppleScript window inspection\n\nUse AppleScript window inspection when you need app-level UI state without relying on OCR.\n\nGood fits:\n- read the front window title\n- count windows for a process\n- list window titles for a process\n\n### Read the front window title\n\n```bash\npython scripts/applescript_window.py --action title --app \"微信\"\n```\n\n### Count windows\n\n```bash\npython scripts/applescript_window.py --action count --app \"微信\"\n```\n\n### List window titles\n\n```bash\npython scripts/applescript_window.py --action list --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\" --json-pretty\n```\n\n## 9. When to use AppleScript vs desktop vision\n\nPrefer AppleScript for:\n- opening or activating apps\n- reading window titles\n- checking the frontmost app\n- simple app and process state queries\n\nDo not add AppleScript UI scripting here for button clicks or deep accessibility-tree automation. That path is intentionally excluded from this skill.\n\nPrefer screenshot + OCR/OpenCV + pyautogui for:\n- buttons or labels that only exist visually\n- apps with weak or unstable accessibility hierarchies\n- targets inside custom-drawn UIs such as chat rows, images, or canvas content\n- direct manipulation such as clicking, dragging, and typing into app surfaces\n\nA practical sequence is often:\n1. AppleScript activates the app\n2. AppleScript reads window or process state\n3. screenshot-based vision finds the target\n4. mouse or keyboard automation performs the action\n5. AppleScript or a fresh screenshot verifies the result\n\n## 10. Recommended flow\n\n```bash\npython scripts/applescript_app.py --action activate --app \"微信\"\npython scripts/applescript_window.py --action title --app \"微信\"\npython scripts/init_coordinate_mapping.py\npython scripts/capture_screen.py\npython scripts/locate_text_ocr.py --text \"确定\"\npython scripts/mouse.py --action click --x 500 --y 300\npython scripts/keyboard.py --action press --key enter\n```\n\n## Notes\n\n- Version 1 assumes a Retina display and single primary screen.\n- Treat logical screenshots as the default working surface for this skill.\n- Treat recognition output coordinates as logical unless a script explicitly says otherwise.\n- Treat mouse and keyboard targeting as logical by default.\n- Treat crop rectangles as logical by default, and prefer cropping from a logical screenshot.\n- If another skill mixes logical screenshots with raw Retina or pixel screenshots, use calibration conversion deliberately. Do not assume logical bounds match raw pixel bounds 1:1.\n- Keep this skill focused on generic desktop primitives. App-specific UI semantics, business rules, and event pipelines should stay in the higher-level app skill.\n- All click, drag, move, and typing actions use Python / `pyautogui`.\n- AppleScript support in this skill is limited to app control and window inspection.\n- For safety, keep `pyautogui.FAILSAFE = True`; moving the mouse to the top-left corner aborts automation.\n\nFile v1.0.7:_meta.json\n\n{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.0.7\",\n  \"publishedAt\": 1776760779166\n}\n\nFile v1.0.7:requirements.txt\n\npyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0","readmeExcerpt":"Skill: MacOS Desktop Control Owner: kd-oauth Summary: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows. Tags: latest:1.1.2 Version history: v1.1.2 | 2026-05-13T12:03:35.667Z | user desktop-control-for-macos 1.1.2 - Updated documentation to clarify using AI semantic understanding as the default for target location, with fallback to OCR or ","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"/tmp/macos_desktop_control/calibration.json"},{"language":"text","snippet":"macos-desktop-control/\n  SKILL.md\n  requirements.txt\n  scripts/\n    calibration.py\n    init_coordinate_mapping.py\n    capture_screen.py\n    crop_image.py\n    locate_text_ocr.py\n    locate_image_opencv.py\n    mouse.py\n    keyboard.py\n    applescript_app.py\n    applescript_window.py"},{"language":"bash","snippet":"pip install -r requirements.txt"},{"language":"bash","snippet":"python scripts/init_coordinate_mapping.py"},{"language":"json","snippet":"{\n  \"screen_width_points\": 1512,\n  \"screen_height_points\": 982,\n  \"screenshot_width_pixels\": 3024,\n  \"screenshot_height_pixels\": 1964,\n  \"scale_x\": 2.0,\n  \"scale_y\": 2.0,\n  \"mode\": \"retina\"\n}"},{"language":"bash","snippet":"python scripts/capture_screen.py --output /tmp/macos_desktop_control/screen_logical.png"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: desktop-control-for-macos\ndescription: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n---\n\n# 写在前面\n\n特别做了中文兼容，包括文字输入/识别等，中文用户放心使用～\n\n# macos-desktop-control\n\nThis skill controls the macOS desktop through a small, explicit pipeline with a clear split between semantic app control and visual UI control:\n\n## Features\n\n### 🖥️ App and window control\n\n- ✅ Activate an app by name or bundle path\n- ✅ Check whether an app is running\n- ✅ Read the current frontmost app\n- ✅ Read front window title, count windows, and list window titles\n\n### 📸 Screenshot and image operations\n\n- ✅ Capture the current screen as a logical-resolution screenshot\n- ✅ Initialize screenshot-to-click calibration for macOS Retina displays\n- ✅ Crop a known rectangular region from an image\n- ✅ Reuse calibration data when a workflow must mix logical and raw screenshots\n\n### 🎯 Visual target location\n\n- ✅ Locate targets by AI semantic understanding as the default first choice\n- ✅ Fall back to OCR when the target is best identified by text\n- ✅ Fall back to OpenCV image matching when the target has a stable reusable template\n- ✅ Constrain later actions to coordinates derived from a screenshot\n\n### ⌨️ Mouse and keyboard control\n\n- ✅ Move the mouse in logical screen coordinates\n- ✅ Left click, right click, double click, and drag\n- ✅ Read current mouse position\n- ✅ Type text, paste via higher-level workflows, press keys, and send hotkeys\n- ✅ Hold and release keys explicitly when needed\n\n### 🛡️ Safety and scope\n\n- ✅ Use logical coordinates as the default working convention\n- ✅ Keep app-specific UI semantics out of this skill\n- ✅ Keep AppleScript usage limited to app and window semantics, not deep UI scripting\n- ✅ Keep `pyautogui.FAILSAFE = True` so moving to the top-left corner aborts automation\n\n1. Use AppleScript for app and window semantics\n2. Initialize coordinate mapping\n3. Capture the screen\n4. Locate targets by AI semantic understanding first, then fall back to OCR or OpenCV when needed\n5. Execute mouse and keyboard actions with Python\n\n## Design boundary\n\nThis skill intentionally does not include AppleScript UI scripting.\n\nUse AppleScript for:\n- opening or activating apps\n- reading frontmost app state\n- reading window titles and counts\n\nUse screenshot-guided OCR/OpenCV plus `pyautogui` for:\n- clicking UI targets\n- typing into custom-drawn interfaces\n- interacting with chat rows, images, canvases, or other visually defined targets\n\nThis boundary keeps the skill predictable. AppleScript is used where semantic macOS state is strong, and `pyautogui` is used where direct UI manipulation is more reliable.\n\n## Why initialization is needed\n\nOn macOS, screenshot coordinates and click coordinates may use different coordinate systems.\n\n- `screencapture` images usually use pixel coordinates.\n- Mouse automation tools often use macOS screen coordinates, also called point coordinates.\n- On Retina displays, one po"},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7fgbaj4zpms7gsnh0gn2kwtd853tne\",\n  \"slug\": \"desktop-control-for-macos\",\n  \"version\": \"1.1.2\",\n  \"publishedAt\": 1778673815667\n}"},{"path":"skill-card.md","content":"## Description:\n\nGeneric macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[kd-oauth](https://clawhub.ai/user/kd-oauth)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and automation agents use this skill to control a macOS desktop session through app/window inspection, screenshot capture, OCR or image matching, and mouse or keyboard actions.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: The skill can control the screen, mouse, keyboard, clipboard, and AppleScript in the active macOS session.\n\nMitigation: Use it only in a dedicated, low-sensitivity macOS session with explicit Screen Recording and Accessibility permissions.\n\nRisk: Pasted text or generated screenshots may expose sensitive information.\n\nMitigation: Avoid pasting secrets, clear or protect generated screenshots, and verify target fields before sending externally visible input.\n\nRisk: Untrusted app names can trigger AppleScript interpolation risk until the issue is fixed.\n\nMitigation: Use only trusted app names or bundle paths and review the AppleScript command path before installation.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/kd-oauth/skills/desktop-control-for-macos)\n\n## Skill Output:\n\n**Output Type(s):** [text, markdown, code, shell commands, configuration, guidance]\n\n**Output Format:** [Markdown guidance with shell command examples; scripts may emit JSON coordinates, file paths, and image files.]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Requires macOS Screen Recording and Accessibility permissions; OCR uses Apple Vision through PyObjC.]\n\n## Skill Version(s):\n\n1.1.2 (source: server release evidence and target metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"requirements.txt","content":"pyautogui>=0.9.54\nPillow>=10.0.0\nopencv-python>=4.8.0\npyobjc-framework-Vision>=10.0\npyobjc-framework-Quartz>=10.0"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows. Skill: MacOS Desktop Control Owner: kd-oauth Summary: Generic macOS desktop control using AppleScript for app and window semantics plus screenshot, OCR, mouse, and keyboard workflows. Tags: latest:1.1.2 Version history: v1.1.2 | 2026-05-13T12:03:35.667Z | user desktop-control-for-macos 1.1.2 - Updated documentation to clarify using AI semantic understanding as the default for target location, with fallback to OCR or","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1377,"uniquenessScore":48,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T08:21:36.387Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T10:43:38.389Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}