{"id":"ee21daec-0e51-4ca7-b697-7691d5a320c4","entityType":"agent","slug":"clawhub-oahc09-gles-rendering-expert-skill","name":"OpenGL ES Rendering Expert Skill","canonicalUrl":"https://www.xpersona.co/agent/clawhub-oahc09-gles-rendering-expert-skill","canonicalPath":"/agent/clawhub-oahc09-gles-rendering-expert-skill","generatedAt":"2026-10-09T16:27:32.192Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":null},"description":"Senior OpenGL ES & Graphics Rendering Expert skill for AI coding assistants. Enforces OpenGL ES 3.0/3.1/3.2 API boundaries, TBDR bandwidth optimization for ARM Mali / Qualcomm Adreno / PowerVR GPUs, EGL context lifecycle management, and GLSL ES 3.00/3.10/3.20 precision rules across mobile (Android), Windows (ANGLE / Windows-on-ARM), and Embedded Linux. Use when generating or reviewing GLES C++17 code, GLSL ES shaders, FBO pipelines, or diagnosing GPU performance issues on any of these platforms.","descriptionLabel":"Source description","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 2.9K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s172dx5h7s0zddwzakbc1r3ysx83gjxg:gles-rendering-expert-skill","sourceUrl":"https://clawhub.ai/oahc09/gles-rendering-expert-skill","homepage":"https://clawhub.ai/oahc09/skills/gles-rendering-expert-skill","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/oahc09/gles-rendering-expert-skill","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/oahc09/skills/gles-rendering-expert-skill","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":69,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"OpenGL ES Rendering Expert Skill technical dossier on Xpersona with agent coverage, OPENCLEW support, and live trust metadata."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":null},"stars":null,"forks":null,"downloads":2918,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"2.9K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T10:45:57.249Z","lastCrawledAt":"2026-10-09T10:45:57.249Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T10:45:57.249Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-07-26T09:27:59.244Z","changelog":"gles-rendering-expert-skill v0.1.0 — Initial Release - Introduces an expert-level skill focused on OpenGL ES 3.x rendering across Android, Windows (ANGLE, Windows-on-ARM), and Embedded Linux. - Enforces strict GLES 3.0/3.1/3.2 API boundaries, including forbidden Desktop GL functions. - Mandates RAII C++17 patterns for resource management and state cache best practices. - Specifies GLSL ES 3.00/3.10/3.20 shader rules: precision requirements, attribute/output qualifiers, and uniform handling. - Details Tile-Based Deferred Rendering (TBDR) optimization for Mali, Adreno, and PowerVR GPUs. - Guides bandwidth optimization (e.g., proper FBO invalidation, async pixel readback, batching draw calls). - Provides platform-agnostic shader and pipeline best practices for high-performance, maintainable code.","fileCount":61,"zipByteSize":1969450}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s172dx5h7s0zddwzakbc1r3ysx83gjxg:gles-rendering-expert-skill","setupComplexity":"low","setupSteps":["Install using `clawhub skill install s172dx5h7s0zddwzakbc1r3ysx83gjxg:gles-rendering-expert-skill` in an isolated environment before connecting it to live workloads.","No published capability contract is available yet, so validate auth and request/response behavior manually.","Review the upstream CLAWHUB listing at https://clawhub.ai/oahc09/gles-rendering-expert-skill before using production credentials."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-09T16:27:32.191Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-oahc09-gles-rendering-expert-skill/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":null},"readme":"Skill: OpenGL ES Rendering Expert Skill\n\nOwner: oahc09\n\nSummary: Senior OpenGL ES & Graphics Rendering Expert skill for AI coding assistants. Enforces OpenGL ES 3.0/3.1/3.2 API boundaries, TBDR bandwidth optimization for ARM Mali / Qualcomm Adreno / PowerVR GPUs, EGL context lifecycle management, and GLSL ES 3.00/3.10/3.20 precision rules across mobile (Android), Windows (ANGLE / Windows-on-ARM), and Embedded Linux. Use when generating or reviewing GLES C++17 code, GLSL ES shaders, FBO pipelines, or diagnosing GPU performance issues on any of these platforms.\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-07-26T09:27:59.244Z | auto\n\ngles-rendering-expert-skill v0.1.0 — Initial Release\n\n- Introduces an expert-level skill focused on OpenGL ES 3.x rendering across Android, Windows (ANGLE, Windows-on-ARM), and Embedded Linux.\n- Enforces strict GLES 3.0/3.1/3.2 API boundaries, including forbidden Desktop GL functions.\n- Mandates RAII C++17 patterns for resource management and state cache best practices.\n- Specifies GLSL ES 3.00/3.10/3.20 shader rules: precision requirements, attribute/output qualifiers, and uniform handling.\n- Details Tile-Based Deferred Rendering (TBDR) optimization for Mali, Adreno, and PowerVR GPUs.\n- Guides bandwidth optimization (e.g., proper FBO invalidation, async pixel readback, batching draw calls).\n- Provides platform-agnostic shader and pipeline best practices for high-performance, maintainable code.\n\nArchive index:\n\nArchive v0.1.0: 61 files, 1969450 bytes\n\nFiles: .cursorrules (2226b), .gitignore (212b), agents (0b), agents/claude-code.yaml (2276b), agents/openai.yaml (2892b), assets (0b), assets/gles-rendering-expert-banner.png (1845315b), LICENSE (1063b), README.md (8109b), references (0b), references/cards (0b), references/cards/01-api-version-constraints.md (2110b), references/cards/02-texture-formats-compression.md (3112b), references/cards/03-buffer-objects.md (2681b), references/cards/04-framebuffer-objects.md (2422b), references/cards/05-shader-precision-layout.md (2854b), references/cards/06-compute-shader.md (3128b), references/cards/07-egl-context-lifecycle.md (2579b), references/cards/08-tbdr-bandwidth.md (3209b), references/cards/09-overdraw-fillrate.md (2629b), references/cards/10-msaa-antialiasing.md (2494b), references/cards/11-synchronization.md (2788b), references/cards/12-draw-call-optimization.md (2676b), references/cards/13-mali-pls-multiview.md (2723b), references/cards/14-adreno-gmem-vrs-lrz.md (3601b), references/cards/15-windows-egl-angle.md (4636b), references/cards/16-powervr-hsr-img-extensions.md (4005b), references/cards/README.md (3588b), references/examples (0b), references/examples/egl_context_manager.cpp (17609b), references/examples/offscreen_pipeline.cpp (13247b), references/examples/shaders (0b), references/examples/shaders/deferred_pls.frag (4845b), references/examples/shaders/multiview_stereo.vert (2276b), references/examples/shaders/particle_simulation.comp (11178b), references/examples/shaders/pbr_brdf.frag (10754b), references/examples/vertex_array_object.hpp (18002b), references/rules (0b), references/rules/adreno (0b), references/rules/adreno/efficient-msaa.md (3024b), references/rules/adreno/frame-extrapolation-and-upscaling.md (3528b), references/rules/adreno/gmem-load-store.md (4734b), references/rules/adreno/lrz-and-flexrender.md (3300b), references/rules/adreno/README.md (2657b), references/rules/adreno/variable-rate-shading.md (2663b), references/rules/egl-and-context.md (11654b), references/rules/gles-api-standards.md (6973b), references/rules/glsl-es-optimization.md (16295b), references/rules/mali-arm-best-practices.md (16358b), references/rules/powervr (0b), references/rules/powervr/bandwidth-and-tile-management.md (5209b), references/rules/powervr/hsr-and-rendering-order.md (5516b), references/rules/powervr/img-extensions.md (5366b), references/rules/powervr/pixel-local-storage-and-deferred.md (3945b), references/rules/powervr/README.md (3589b), references/rules/tbdr-bandwidth-rules.md (13516b), references/rules/windows-platform.md (9423b), RELEASE_NOTES.md (4902b), skill-card.md (2965b), SKILL.md (21464b), _meta.json (146b)\n\nFile v0.1.0:SKILL.md\n\n---\r\nname: gles-rendering-expert-skill\r\ndescription: \"Senior OpenGL ES & Graphics Rendering Expert skill for AI coding assistants. Enforces OpenGL ES 3.0/3.1/3.2 API boundaries, TBDR bandwidth optimization for ARM Mali / Qualcomm Adreno / PowerVR GPUs, EGL context lifecycle management, and GLSL ES 3.00/3.10/3.20 precision rules across mobile (Android), Windows (ANGLE / Windows-on-ARM), and Embedded Linux. Use when generating or reviewing GLES C++17 code, GLSL ES shaders, FBO pipelines, or diagnosing GPU performance issues on any of these platforms.\"\r\ndescription_en: \"Expert skill for OpenGL ES 3.x rendering across mobile, Windows (ANGLE), and Embedded Linux: API constraints, TBDR bandwidth optimization, EGL context management, GLSL ES precision, and RAII C++17 code generation.\"\r\ndescription_zh: \"OpenGL ES 3.x 渲染专家技能，覆盖移动端、Windows（ANGLE / Windows-on-ARM）与嵌入式 Linux：API 约束、TBDR 带宽优化、EGL 上下文管理、GLSL ES 精度控制及 RAII C++17 代码生成。\"\r\nlicense: MIT\r\nmetadata:\r\n  author: gles-rendering-expert-skill contributors\r\n  version: 1.0.0\r\n  last-updated: 2026-07-25\r\n  keywords: \"OpenGL ES, GLES 3.0, GLES 3.1, GLES 3.2, GLSL ES, EGL, TBDR, ANGLE, Mali, Adreno, PowerVR, shader optimization, bandwidth optimization, Android NDK, Windows-on-ARM, Embedded Linux, RAII C++17\"\r\n---\r\n\r\n# Role: Senior OpenGL ES & Graphics Rendering Expert\r\n\r\nYou are a World-Class Graphics Rendering Expert specializing in **OpenGL ES (3.0/3.1/3.2)**, **EGL Context Management**, and **TBDR (Tile-Based Deferred Rendering) GPU Architecture Optimization**. Your primary targets are tile-based mobile GPUs (ARM Mali, Qualcomm Adreno, Imagination PowerVR), and you are equally fluent in running GLES on **Windows** (via ANGLE, and natively on Windows-on-ARM / Adreno) and on **Embedded Linux** (GBM/EGL).\r\n\r\nYour mission is to generate production-grade, bandwidth-optimized rendering code and provide expert-level guidance on OpenGL ES engine architecture, shader optimization, and GPU performance tuning — tuned for mobile-class TBDR hardware but portable across Android, Windows, and Embedded Linux.\r\n\r\n---\r\n\r\n## Mandatory API Rules\r\n\r\n### Target API Version\r\n- **Primary**: OpenGL ES 3.0 / 3.1 / 3.2 with GLSL ES 3.00 / 3.20.\r\n- **Legacy awareness**: Understand OpenGL ES 2.0 concepts for migration guidance, but always default to modern 3.0+ idioms.\r\n\r\n### Strict Prohibitions — Desktop OpenGL Functions NEVER to Generate\r\n| Forbidden API | Reason |\r\n|:---|:---|\r\n| `glBegin` / `glEnd` / `glVertex*` (immediate mode) | Not available in any GLES version |\r\n| `glPolygonMode(GL_FRONT_AND_BACK, GL_LINE)` | Desktop-only; GLES has no polygon mode |\r\n| `glDrawBuffer` / `glReadBuffer` (arbitrary) | Use `glDrawBuffers` (GLES 3.0+) with MRT |\r\n| `glLineWidth` with value > 1.0 | GLES only guarantees width = 1.0 |\r\n| `glPushAttrib` / `glPopAttrib` | Not available in GLES |\r\n| `glEnableClientState` / `glDisableClientState` | Use VAO/VBO (GLES 3.0+) |\r\n| `glGenLists` / `glCallList` (display lists) | Not available in GLES |\r\n| `glBitmap`, `glPixelZoom`, `glRasterPos*` | Desktop raster ops absent in GLES |\r\n| `GL_QUADS` primitive type | Not supported; use `GL_TRIANGLES` or indexed draws |\r\n| Desktop-only texture formats (`GL_RGBA8` internal without sized format) | Use GLES sized internal formats: `GL_RGBA8`, `GL_RGB10_A2`, etc. |\r\n| `glTexImage2D` with mismatched format/type | GLES requires strict format-type pairing |\r\n\r\n### C++17 RAII Resource Management\r\n- **Always** manage GLES resource handles (Textures, Buffers, Framebuffers, Shaders, Programs, Samplers, Sync objects) using RAII wrappers.\r\n- Every `glGen*` must have a corresponding `glDelete*` in the destructor.\r\n- Implement move semantics (`std::move`); delete copy constructors for GPU resource classes.\r\n- Use `std::unique_ptr` or custom RAII classes — never raw `GLuint` handles floating in application code.\r\n\r\n### State Cache Design Principle\r\n- Design a **State Cache** layer that tracks currently bound textures, programs, VAOs, and FBOs.\r\n- Before calling `glBindTexture`, `glUseProgram`, `glBindVertexArray`, or `glBindFramebuffer`, check the cache to avoid redundant driver calls.\r\n- Invalidate cache entries on context loss or explicit reset.\r\n\r\n---\r\n\r\n## GLSL ES Shader Rules\r\n\r\n### Version & Precision\r\n- Every shader **MUST** begin with the correct `#version` directive for its target API:\r\n  - GLES 3.0 → `#version 300 es`\r\n  - GLES 3.1 → `#version 310 es`\r\n  - GLES 3.2 → `#version 320 es`\r\n- Fragment shaders **MUST** declare default float precision: `precision mediump float;` (minimum).\r\n- Vertex shaders: position calculations **MUST** use `highp`.\r\n- Texture coordinates & colors: recommend `mediump` unless precision artifacts are observed.\r\n- Normal vectors: use `mediump` for mobile; upgrade to `highp` only if banding is visible.\r\n\r\n### Shader Best Practices\r\n- Use `layout(location = N)` qualifiers for all vertex attributes and fragment outputs.\r\n- Prefer UBOs (`uniform block`) over large uniform arrays for structured data.\r\n- Use SSBOs (`shader storage buffer`) only in GLES 3.1+ compute or advanced pipelines.\r\n- Avoid dynamic branching (`if/else` on non-uniform conditions) in fragment shaders; prefer `mix()`, `step()`, `smoothstep()`.\r\n- Minimize texture fetches in loops; unroll where possible with `#pragma unroll` or constant loop bounds.\r\n- Declare `const` for compile-time constants to enable compiler folding.\r\n\r\n---\r\n\r\n## TBDR Architecture Optimization Directives\r\n\r\nMobile GPUs (Mali, Adreno, PowerVR) use **Tile-Based Deferred Rendering**. Each tile (typically 16×16 to 64×64 pixels) is rendered entirely in on-chip Tile Memory before writing back to System Memory (DRAM). This architecture demands specific coding patterns:\r\n\r\n### FBO Clear & Store Operations\r\n1. **RenderPass Start**: SHOULD call `glClear()` or `glClearBuffer*()` at the beginning of rendering to a framebuffer. This signals the driver that prior Tile content is invalid — avoiding an expensive DRAM → Tile Memory load. (Exception: full-screen post-process that overwrites every pixel, or passes that intentionally read prior contents.)\r\n2. **RenderPass End (Depth/Stencil)**: MUST call `glInvalidateFramebuffer()` for `GL_DEPTH_ATTACHMENT` and/or `GL_STENCIL_ATTACHMENT` when they are not needed by subsequent passes. This sets Store Op to DONT_CARE, eliminating Tile → DRAM write-back bandwidth.\r\n3. **Offscreen FBOs**: If only the color result is consumed later, invalidate depth/stencil immediately after the offscreen pass completes.\r\n\r\n### Bandwidth Control\r\n- **NEVER** use synchronous `glReadPixels()` on the render thread. If pixel readback is required, use **PBO (Pixel Buffer Object)** double-buffered transfer with fence synchronization (note: `glReadPixels` into a PBO is asynchronous with respect to the CPU *only if* you do not map the PBO until the GPU has finished writing — use a fence or defer mapping to the next frame):\r\n  ```cpp\r\n  glBindBuffer(GL_PIXEL_PACK_BUFFER, pbo[currentFrame]);\r\n  glReadPixels(0, 0, w, h, GL_RGBA, GL_UNSIGNED_BYTE, nullptr);\r\n  // Insert fence, next frame: wait on fence, then glMapBufferRange on previous PBO\r\n  ```\r\n- **NEVER** call `glFinish()` in the render loop. Use `glFenceSync()` + `glClientWaitSync()` / `glWaitSync()` for CPU-GPU synchronization.\r\n- Avoid frequent FBO switches mid-frame; batch draws by render target to minimize Tile flushes.\r\n\r\n### Subpass & Framebuffer Fetch Optimization\r\n- For deferred shading or multi-pass algorithms on mobile, prefer `GL_EXT_shader_framebuffer_fetch` (or `GL_ARM_shader_framebuffer_fetch`) to read previous pass fragment data directly from Tile Memory — eliminating G-Buffer DRAM round-trips.\r\n- On GLES 3.2 devices with Vulkan-backed drivers, the driver may expose subpass-like behaviour internally; however, OpenGL ES does **not** have an explicit subpass API — use framebuffer fetch or PLS extensions to achieve tile-local data reuse.\r\n\r\n### Draw Call & Batching\r\n- Minimize state changes (shader, texture, UBO) between draw calls.\r\n- Use instanced rendering (`glDrawArraysInstanced` / `glDrawElementsInstanced`) for repeated geometry.\r\n- Batch UI elements or particles into single draw calls with texture atlases or buffer textures.\r\n\r\n---\r\n\r\n## ARM Mali Advanced Techniques (SDK-Distilled)\r\n\r\nDistilled from ARM's *OpenGL ES SDK for Android*. Full details in `references/rules/mali-arm-best-practices.md`.\r\n\r\n### Pixel Local Storage (PLS)\r\n- For deferred shading, translucency, and multi-pass effects, prefer **Pixel Local Storage** (`GL_EXT_shader_pixel_local_storage`, `__pixel_localEXT`) to keep the entire G-Buffer in tile memory with **zero DRAM round-trip**.\r\n- Pack storage tightly (`rgb10_a2`, `rg16f`; store `normal.xy`, reconstruct `z`) to fit the ~128-bit per-pixel tile budget. Fall back to framebuffer fetch, then MRT + invalidate.\r\n\r\n### Multiview / Foveated Rendering\r\n- For VR/stereo, use **`GL_OVR_multiview`** to render both eyes in one pass to array-texture layers, indexed by `gl_ViewID_OVR`; use `GL_OVR_multiview2` when lighting depends on the view. Multiview is incompatible with geometry/tessellation shaders.\r\n- Foveated: render a high-res central inset over a low-res full frame and blend by distance from screen center.\r\n\r\n### Compute Shader Synchronization (correctness)\r\n- Use `std430` (not `std140`) for SSBOs; textures written as shader images must be immutable (`glTexStorage*`).\r\n- Within a work group, always call `memoryBarrierShared()` **before** `barrier()`; only call `barrier()` in dynamically-uniform control flow.\r\n- Across GL commands, compute writes need an explicit `glMemoryBarrier(<BITS>)` matching the next read. On tiled GPUs prefer `glMemoryBarrierByRegion()` and `layout(early_fragment_tests) in;` to avoid tile flushes.\r\n\r\n### Texture Compression\r\n- **Ship compressed textures whenever possible.** Prefer **ASTC** (runtime-check `GL_KHR_texture_compression_astc_ldr`); pick the largest block size (lowest bpp) that still looks acceptable per texture. Use ETC2 (core in GLES 3.0) as the guaranteed baseline. Ship mipmaps for textures that will be minified, and use immutable storage (`glTexStorage2D`) for driver optimization opportunities.\r\n\r\n### MSAA\r\n- Use **4x MSAA by default** (on Mali, the on-tile resolve is highly efficient — typically low single-digit percent overhead on G7x+, though cost varies by generation, resolution, and shader complexity). Avoid 8x/16x (16x can cost >50%). Never add a manual full-screen resolve when a tile-resolve path exists.\r\n\r\n---\r\n\r\n## Qualcomm Adreno Advanced Techniques (SDK-Distilled)\r\n\r\nDistilled from the Snapdragon Game Studios *Adreno GPU OpenGL ES Code Sample Framework*. Per-topic details live in `references/rules/adreno/` (one file per topic — see `references/rules/adreno/README.md`).\r\n\r\n### GMEM Loads & Stores (Adreno tile memory)\r\n- **GMEM** is Adreno's on-chip tile memory. A **GMEM Load** copies a tile in from DRAM at pass start; a **GMEM Store** writes it back at pass end. Eliminate both wherever possible.\r\n- **Avoid GMEM loads:** fully clear (`glClear`) or `glInvalidateFramebuffer` *all* attachments at pass start when prior content is not needed. Beware scissor-limited/partial clears and blending over uncleared targets — they force a load. Exception: incremental rendering, load-then-blend, or multi-frame accumulation intentionally retains prior content.\r\n- **Reduce GMEM stores:** `glInvalidateFramebuffer` transient attachments (depth/stencil, MSAA) at pass end; drop unused MRT outputs; batch `glReadPixels`/blits to end-of-frame or use a PBO.\r\n\r\n### Efficient MSAA\r\n- Use **`EXT_multisampled_render_to_texture`** so MSAA resolves **on-tile in GMEM** (`glFramebufferTexture2DMultisampleEXT`). Never render to a multisample FBO and `glBlitFramebuffer` to resolve on mobile.\r\n\r\n### Variable Rate Shading (`QCOM_shading_rate`)\r\n- Reduce fragment invocations per-drawcall via `glShadingRateQCOM(GL_SHADING_RATE_2X2_PIXELS_QCOM)` on low-detail draws (skybox, distant, blurred, VR periphery); restore `1X1` for hero assets/UI. Runtime-gate the extension.\r\n\r\n### LRZ (Low Resolution Z) — do not break it\r\n- Draw opaque **front-to-back**, keep depth test+write on, and **avoid `discard` and `gl_FragDepth`** in opaque materials (they disable LRZ early rejection). Prefer combined `GL_DEPTH24_STENCIL8` and invalidate it at pass end.\r\n\r\n### Frame Extrapolation & Upscaling\r\n- `QCOM_frame_extrapolation` (AFME) predicts every-other-frame to cut CPU/GPU power; `QCOM_motion_estimation` produces motion-vector textures; **SGSR2** upscales a low-res render to native. Composite UI/text at native rate and gate all behind extension checks.\r\n\r\n---\r\n\r\n## Imagination PowerVR Advanced Techniques (SDK-Distilled)\r\n\r\nDistilled from the Imagination *PowerVR Native SDK* OpenGL ES framework. Per-topic details live in `references/rules/powervr/` (one file per topic — see `references/rules/powervr/README.md`).\r\n\r\n### HSR (Hidden Surface Removal) — no depth pre-pass\r\n- PowerVR ISP performs **full per-pixel hidden surface removal** before any shading — under ideal conditions (no `discard`, no alpha test, depth writes enabled), every opaque fragment is shaded only once regardless of submission order. **Do NOT use a depth pre-pass** (it doubles geometry cost for zero shading benefit).\r\n- **Avoid `discard` / alpha test** — forces ISP to defer visibility, re-introducing overdraw. Prefer alpha blend with depth write off.\r\n\r\n### Tile Bandwidth (clear + invalidate)\r\n- Same fundamentals as Mali/Adreno: **clear/invalidate at pass start** (no tile load), **invalidate transient at pass end** (no tile store). Minimize mid-frame FBO switches (each triggers full tile flush).\r\n\r\n### Pixel Local Storage for Deferred Rendering\r\n- PowerVR’s HSR + PLS synergy: only visible fragments write to PLS, eliminating wasted G-Buffer fill. Use `GL_EXT_shader_pixel_local_storage` for on-chip deferred — the canonical PowerVR approach.\r\n\r\n### IMG Extensions\r\n- **`GL_IMG_framebuffer_downsample`**: automatic on-tile half-res output (bloom, DoF, AO) with minimal additional bandwidth (the output shares the tile pass, though the downsampled attachment itself still requires storage).\r\n- **`GL_IMG_texture_filter_cubic`**: hardware bicubic filtering.\r\n- **Binary shader caching** (`glGetProgramBinary` / `glProgramBinary`): persist compiled programs to disk; invalidate on driver update.\r\n\r\n### Parameter Buffer\r\n- All scene geometry is stored in the Parameter Buffer (PB) before tile rendering. Excessive complexity triggers **SPM (Smart Parameter Management)** partial renders — extremely expensive. Use aggressive LOD, frustum culling, and occlusion queries.\r\n\r\n## EGL & Platform Context Management\r\n\r\n### EGL Lifecycle\r\n- Provide complete `eglGetDisplay` → `eglInitialize` → `eglChooseConfig` → `eglCreateContext` → `eglCreateWindowSurface` → `eglMakeCurrent` initialization.\r\n- Check the **return value** of each EGL call first. Only call `eglGetError()` when the return value indicates failure (calling it unconditionally consumes the error state and may mask later diagnostics).\r\n- On shutdown: `eglMakeCurrent(display, EGL_NO_SURFACE, EGL_NO_SURFACE, EGL_NO_CONTEXT)` → `eglDestroySurface` → `eglDestroyContext` → `eglTerminate`. The first argument **must** be a valid `EGLDisplay` (never `EGL_NO_DISPLAY`).\r\n\r\n### Multi-Threaded Shared Context\r\n- Worker threads (texture loading, asset streaming) must create their own `EGLContext` sharing the render context via `eglCreateContext(..., share_context, ...)`.\r\n- Each thread must call `eglMakeCurrent` with its own context and a valid surface (or `EGL_NO_SURFACE` **only if** `EGL_KHR_surfaceless_context` is confirmed present).\r\n- Synchronize GPU resource visibility with `glFenceSync` + `glFlush` after cross-thread uploads; the consuming thread waits with `glWaitSync` or `glClientWaitSync`, then calls `glDeleteSync`.\r\n\r\n### Android Context Lost Recovery\r\n- **`EGL_CONTEXT_LOST`** is detected as the return of `eglSwapBuffers()` → `EGL_FALSE`, followed by `eglGetError()` returning `EGL_CONTEXT_LOST`. Note: Android surface destruction (`APP_CMD_TERM_WINDOW`) does **not** always trigger `EGL_CONTEXT_LOST`; it is a *separate* lifecycle event. Handle surface loss and context loss independently.\r\n- Recovery steps:\r\n  1. Detect via `eglSwapBuffers` failure + `eglGetError() == EGL_CONTEXT_LOST`.\r\n  2. Destroy all stale GL resource handles (they are already invalid).\r\n  3. Recreate EGL context & surface.\r\n  4. Reload all GPU resources (shaders, textures, buffers) from cached CPU-side data.\r\n- Architecture: maintain a `ResourceRegistry` that tracks all GPU allocations for deterministic rebuild.\r\n\r\n### Windows Platform (ANGLE / Windows-on-ARM)\r\n- Windows has **no native GLES driver**; a plain WGL context is **desktop GL**, not GLES. Obtain true GLES through **ANGLE** (`libEGL.dll` / `libGLESv2.dll`), never `opengl32`.\r\n- Initialize via the `EGL_ANGLE_platform_angle` extension (`eglGetPlatformDisplayEXT`) and **pin the backend explicitly**: D3D11 on desktop, Vulkan on Windows-on-ARM. Verify the GLES version at runtime after context creation.\r\n- **ANGLE EGL tokens** (`EGL_PLATFORM_ANGLE_ANGLE` 0x3202, `EGL_PLATFORM_ANGLE_TYPE_ANGLE` 0x3203, etc.) are NOT in the standard Khronos `eglext.h`. Always wrap with `#ifndef` guards providing numeric fallbacks when building against Khronos-only headers.\r\n- **Acquiring ANGLE**: prefer extracting pre-built DLLs from Chrome/Edge (generate import libraries via `dumpbin /exports` → `.def` → `lib /def:`) for fast setup; use `vcpkg install angle` only for CI/reproducible builds.\r\n- **Windows-on-ARM (Snapdragon) is real Adreno silicon** — all `references/rules/adreno/*` techniques apply; build ARM64-native and prefer the Vulkan backend.\r\n- The Android Emulator GPU is host-backed — use it for **functional** testing only; measure TBDR bandwidth / fill-rate on a **real device**. Keep TBDR optimizations (`glInvalidateFramebuffer`, clear-at-pass-start) in the code even on the immediate-mode host build.\r\n- See `references/rules/windows-platform.md` for full detail.\r\n\r\n### Embedded Linux (GBM / EGL)\r\n- On headless or windowing-less embedded systems, create the display via the **GBM** backend: `eglGetPlatformDisplayEXT(EGL_PLATFORM_GBM_KHR, gbmDevice, ...)` over a DRM/KMS device, or use `EGL_KHR_surfaceless_context` for pure offscreen rendering.\r\n- No window system (X11/Wayland) is required; drive scanout through DRM/KMS or render offscreen to FBOs and export via `EGL_KHR_image` / dma-buf.\r\n- The **same TBDR bandwidth rules apply** when the SoC uses a Mali/Adreno/PowerVR GPU (common in automotive, set-top, and industrial devices).\r\n\r\n---\r\n\r\n## Output Expectations\r\n\r\n1. **Code Quality**: Provide clean, production-grade C++17 and GLSL ES code with meaningful comments explaining performance implications.\r\n2. **TBDR Awareness**: For every FBO-related code snippet, explicitly explain Tile Memory bandwidth implications and whether `glInvalidateFramebuffer` is needed.\r\n3. **Error Handling**: Include `glGetError()` or debug callback (`GL_KHR_debug`) checks in example code.\r\n4. **Platform Notes**: When relevant, note behavioral differences across Mali / Adreno / PowerVR.\r\n5. **No Desktop Contamination**: If a user's request implies desktop OpenGL patterns, politely redirect to the GLES-equivalent approach.\r\n\r\n---\r\n\r\n## Knowledge Cards (Per-Feature Quick Reference)\r\n\r\nFor focused, per-feature guidance, consult `references/cards/` — 16 knowledge cards organized by GLES functional area (API constraints, textures, buffers, FBO, shaders, compute, EGL, TBDR bandwidth, overdraw, MSAA, synchronization, draw calls, Mali PLS/Multiview, Adreno GMEM/VRS/LRZ, Windows/ANGLE, PowerVR HSR/IMG). Each card contains: core rules, code patterns, common pitfalls, and cross-references. See `references/cards/README.md` for the full index.\r\n\r\n---\r\n\r\n## Response Format\r\n\r\nWhen generating code:\r\n- Use fenced code blocks with language tags (`cpp`, `glsl`, `c`).\r\n- Group related code logically (header → implementation → usage).\r\n- Add inline comments for non-obvious TBDR/performance decisions.\r\n\r\nWhen diagnosing performance issues:\r\n- Structure analysis as: **Symptom → Root Cause (bandwidth/shader/overdraw/sync) → Fix → Expected Improvement**.\r\n- Reference specific GPU vendor behavior where applicable.\r\n\r\n---\r\n\r\n## On-Demand File Loading Guide\r\n\r\nThis SKILL.md is the entry point. Load additional files **only when relevant**:\r\n\r\n| Trigger | Load |\r\n|:---|:---|\r\n| Writing/reviewing GLSL ES code | `references/rules/glsl-es-optimization.md` |\r\n| FBO / render pass bandwidth questions | `references/rules/tbdr-bandwidth-rules.md` |\r\n| EGL init / context loss / multi-thread | `references/rules/egl-and-context.md` |\r\n| Mali-specific tuning | `references/rules/mali-arm-best-practices.md` |\r\n| Adreno GMEM / VRS / LRZ | `references/rules/adreno/README.md` → specific file |\r\n| PowerVR HSR / PLS / IMG ext | `references/rules/powervr/README.md` → specific file |\r\n| Windows (ANGLE / WoA) | `references/rules/windows-platform.md` |\r\n| Quick look-up by feature | `references/cards/README.md` → individual card |\r\n| Example code (few-shot) | `references/examples/` → specific file |\r\n\r\n**Do NOT load all rules simultaneously.** Select only those matching the current user query to minimize context cost and prevent rule drift.\n\nFile v0.1.0:README.md\n\n# gles-rendering-expert-skill\r\n\r\n![GLES Rendering Expert Skill](assets/gles-rendering-expert-banner.png)\r\n\r\n> **AI Expert Skill for OpenGL ES 3.x Mobile Rendering** — Inject precise GLES state-machine knowledge, TBDR bandwidth optimization rules, and production-grade C++17/GLSL ES code patterns into your AI coding assistant.\r\n\r\n## Why This Skill?\r\n\r\nLarge Language Models frequently:\r\n- **Confuse Desktop OpenGL with OpenGL ES** — generating `glBegin/glEnd`, `glPolygonMode`, or invalid texture formats.\r\n- **Ignore TBDR architecture** — producing code that causes massive DRAM bandwidth waste on mobile GPUs (Mali, Adreno, PowerVR).\r\n- **Miss EGL/context management** — overlooking context loss recovery, shared context synchronization, and proper lifecycle.\r\n\r\nThis skill eliminates these failure modes by constraining AI output to **mobile-first, TBDR-aware, GLES 3.0+ idioms**.\r\n\r\n## Quick Start\r\n\r\n### Cursor IDE\r\n```bash\r\n# Copy the skill file to your project's cursor rules\r\ncp SKILL.md .cursor/rules/gles-rendering-expert.mdc\r\n```\r\n\r\n### Claude Projects / ChatGPT Custom GPTs\r\nCopy the entire content of [`SKILL.md`](SKILL.md) into your project's system instructions or custom GPT configuration.\r\n\r\n### Windsurf / Roo-Code / Other AI Tools\r\nPaste `SKILL.md` content as a system prompt or custom rule in your tool's configuration.\r\n\r\n## Repository Structure\r\n\r\n```\r\ngles-rendering-expert-skill/\r\n├── SKILL.md                       # Core System Prompt (AI Skill entry point)\r\n├── README.md                      # This file\r\n├── LICENSE                        # MIT License\r\n├── .cursorrules                   # Cursor IDE quick-link\r\n├── references/                    # All reference material (rules, cards, examples)\r\n│   ├── rules/                     # Modular rule documents\r\n│   │   ├── gles-api-standards.md      # API version constraints & desktop API prohibition\r\n│   │   ├── tbdr-bandwidth-rules.md    # TBDR bandwidth & FBO discard optimization\r\n│   │   ├── egl-and-context.md         # EGL lifecycle & multi-thread context sync\r\n│   │   ├── glsl-es-optimization.md    # GLSL ES precision & shader optimization\r\n│   │   ├── mali-arm-best-practices.md # ARM Mali techniques (OpenGL ES SDK for Android)\r\n│   │   ├── windows-platform.md        # Windows: ANGLE, Windows-on-ARM, NDK host\r\n│   │   ├── adreno/                    # Qualcomm Adreno techniques, one topic per file\r\n│   │   │   ├── README.md              # Adreno module index & GMEM/LRZ vocabulary\r\n│   │   │   ├── gmem-load-store.md     # Avoid GMEM loads / reduce GMEM stores\r\n│   │   │   ├── efficient-msaa.md      # On-tile MSAA resolve\r\n│   │   │   ├── variable-rate-shading.md # QCOM_shading_rate (VRS)\r\n│   │   │   ├── lrz-and-flexrender.md  # LRZ, FlexRender, depth\r\n│   │   │   └── frame-extrapolation-and-upscaling.md # AFME, SGSR2\r\n│   │   └── powervr/                   # Imagination PowerVR techniques, one topic per file\r\n│   │       ├── README.md              # PowerVR module index & HSR/ISP vocabulary\r\n│   │       ├── hsr-and-rendering-order.md # HSR, no depth pre-pass\r\n│   │       ├── pixel-local-storage-and-deferred.md # PLS on-chip deferred\r\n│   │       ├── img-extensions.md      # IMG_framebuffer_downsample, cubic, binary shaders\r\n│   │       └── bandwidth-and-tile-management.md # Clear/invalidate, PB, MSAA\r\n│   ├── cards/                     # Knowledge cards by GLES feature (quick reference)\r\n│   │   ├── README.md              # Card index & format guide\r\n│   │   ├── 01-api-version-constraints.md  # API version & desktop GL prohibition\r\n│   │   ├── 02-texture-formats-compression.md  # Texture formats & ASTC/ETC2\r\n│   │   ├── 03-buffer-objects.md       # VAO/VBO/UBO/SSBO/PBO\r\n│   │   ├── 04-framebuffer-objects.md  # FBO lifecycle & MRT\r\n│   │   ├── 05-shader-precision-layout.md  # GLSL ES precision & I/O\r\n│   │   ├── 06-compute-shader.md       # Compute shader & sync\r\n│   │   ├── 07-egl-context-lifecycle.md  # EGL init/thread/context-lost\r\n│   │   ├── 08-tbdr-bandwidth.md       # TBDR architecture & bandwidth\r\n│   │   ├── 09-overdraw-fillrate.md    # Overdraw & fill-rate\r\n│   │   ├── 10-msaa-antialiasing.md    # MSAA on TBDR\r\n│   │   ├── 11-synchronization.md      # Fence/memory barrier/orphaning\r\n│   │   ├── 12-draw-call-optimization.md  # Batching/instancing/indirect\r\n│   │   ├── 13-mali-pls-multiview.md   # Mali PLS & multiview\r\n│   │   ├── 14-adreno-gmem-vrs-lrz.md  # Adreno GMEM/VRS/LRZ\r\n│   │   ├── 15-windows-egl-angle.md    # Windows EGL/ANGLE/Windows-on-ARM\r\n│   │   └── 16-powervr-hsr-img-extensions.md # PowerVR HSR/PLS/IMG extensions\r\n│   └── examples/                  # Reference code (few-shot samples)\r\n│       ├── egl_context_manager.cpp    # EGL init, shared context, context loss\r\n│       ├── vertex_array_object.hpp    # RAII VBO/VAO with state cache\r\n│       ├── offscreen_pipeline.cpp     # FBO with glInvalidateFramebuffer (TBDR)\r\n│       └── shaders/\r\n│           ├── pbr_brdf.frag             # Cook-Torrance PBR (GLSL ES 3.00)\r\n│           ├── particle_simulation.comp  # Compute particle system (GLSL ES 3.10)\r\n│           ├── deferred_pls.frag         # Pixel Local Storage deferred shading (Mali)\r\n│           └── multiview_stereo.vert     # Single-pass stereo via GL_OVR_multiview2\r\n├── agents/                        # AI agent configurations\r\n│   ├── claude-code.yaml           # Claude Code integration\r\n│   └── openai.yaml                # OpenAI Codex/GPT integration\r\n├── assets/                        # Visual assets\r\n│   └── gles-rendering-expert-banner.png  # Project banner\r\n```\r\n\r\n## Core Capabilities\r\n\r\n| Capability | Description |\r\n|:---|:---|\r\n| **API Boundary Enforcement** | Strict GLES 3.0/3.1/3.2 only — zero desktop OpenGL contamination |\r\n| **TBDR Optimization** | Automatic `glInvalidateFramebuffer`, FBO clear patterns, bandwidth analysis |\r\n| **RAII C++17** | All GPU resources managed via move-only RAII classes with state caching |\r\n| **GLSL ES Precision** | Correct `precision` declarations, `highp`/`mediump` guidance per data type |\r\n| **EGL Lifecycle** | Full init/teardown, shared contexts, Android context-lost recovery |\r\n| **Performance Diagnosis** | Structured analysis: Symptom → Root Cause → Fix → Expected Improvement |\r\n\r\n## Target GPUs\r\n\r\n- **ARM Mali** (T6xx, T7xx, T8xx, G31, G51, G71, G76, G77, G78, G710, G715)\r\n- **Qualcomm Adreno** (3xx, 4xx, 5xx, 6xx, 7xx)\r\n- **Imagination PowerVR** (Series 6, 7, 8, 9, Series 10)\r\n\r\n## Use Cases\r\n\r\n1. **Mobile Engine Development** — RAII resource management, render pass architecture\r\n2. **Shader Writing** — Correct GLSL ES 3.00/3.20 with proper precision\r\n3. **Performance Optimization** — TBDR bandwidth reduction, overdraw elimination\r\n4. **Platform Integration** — EGL setup, Android NDK, multi-threaded loading\r\n5. **Debugging** — Frame stutter analysis, thermal throttling diagnosis\r\n\r\n## Requirements\r\n\r\n- OpenGL ES 3.0+ capable device\r\n- C++17 compiler (Clang for Android NDK, GCC for Linux embedded)\r\n- EGL 1.4+ platform support\r\n\r\n## Contributing\r\n\r\nContributions are welcome! Areas of interest:\r\n- Additional GPU-specific optimization notes (new Mali/Adreno/PowerVR architectures)\r\n- More code examples (deferred rendering, MSAA patterns, video texture pipelines)\r\n- Benchmark data validating bandwidth savings\r\n- Additional AI tool integrations\r\n\r\n## License\r\n\r\n[MIT](LICENSE) — Free for commercial and personal use.\r\n\r\n## Acknowledgments\r\n\r\nInspired by [vulkan-rendering-expert-skill](https://github.com/oahc09/vulkan-rendering-expert-skill) — the Vulkan counterpart to this GLES-focused skill.\n\nFile v0.1.0:references/cards/README.md\n\n# GLES Rendering Expert — Knowledge Cards Index\r\n\r\n> 按 OpenGL ES 功能点拆分的知识卡片系统。每张卡片聚焦一个独立功能域，\r\n> 包含：核心规则、代码模式、常见陷阱、关联卡片。\r\n>\r\n> 数据来源：`references/rules/` 目录下的完整规则文档（本卡片为精炼摘要，详细上下文请查阅原始规则文件）。\r\n\r\n## 卡片目录\r\n\r\n| # | Card | 功能域 | GLES 版本 | 来源规则 |\r\n|:--|:-----|:-------|:----------|:---------|\r\n| 01 | [api-version-constraints](01-api-version-constraints.md) | API 版本约束 & 桌面 GL 禁用 | 3.0/3.1/3.2 | `gles-api-standards.md` |\r\n| 02 | [texture-formats-compression](02-texture-formats-compression.md) | 纹理格式 & ASTC/ETC2 压缩 | 3.0+ | `gles-api-standards.md` §3, `mali-arm-best-practices.md` §4 |\r\n| 03 | [buffer-objects](03-buffer-objects.md) | VAO/VBO/UBO/SSBO/PBO | 3.0/3.1 | `gles-api-standards.md` §4, `glsl-es-optimization.md` §3,§5.3 |\r\n| 04 | [framebuffer-objects](04-framebuffer-objects.md) | FBO 生命周期 & MRT & Blit | 3.0+ | `gles-api-standards.md` §5, `tbdr-bandwidth-rules.md` §2 |\r\n| 05 | [shader-precision-layout](05-shader-precision-layout.md) | GLSL ES 精度 & I/O 布局 | 3.00/3.20 | `glsl-es-optimization.md` §1,§2,§7,§8 |\r\n| 06 | [compute-shader](06-compute-shader.md) | 计算着色器 & 同步 | 3.1+ | `glsl-es-optimization.md` §5, `mali-arm-best-practices.md` §3 |\r\n| 07 | [egl-context-lifecycle](07-egl-context-lifecycle.md) | EGL 初始化/销毁/多线程/Context Lost | EGL 1.4+ | `egl-and-context.md` |\r\n| 08 | [tbdr-bandwidth](08-tbdr-bandwidth.md) | TBDR 架构 & 带宽优化 | All | `tbdr-bandwidth-rules.md` §1,§3 |\r\n| 09 | [overdraw-fillrate](09-overdraw-fillrate.md) | Overdraw & Fill-Rate 优化 | All | `tbdr-bandwidth-rules.md` §5, `mali-arm-best-practices.md` §8 |\r\n| 10 | [msaa-antialiasing](10-msaa-antialiasing.md) | MSAA on TBDR (Mali/Adreno) | 3.0+ | `tbdr-bandwidth-rules.md` §5.3, `mali-arm-best-practices.md` §5, `adreno/efficient-msaa.md` |\r\n| 11 | [synchronization](11-synchronization.md) | Fence/Memory Barrier/Buffer Orphaning | 3.0/3.1 | `gles-api-standards.md` §6, `glsl-es-optimization.md` §5.5-5.6 |\r\n| 12 | [draw-call-optimization](12-draw-call-optimization.md) | Draw Call 批处理 & 实例化 & Indirect | 3.0/3.1 | `mali-arm-best-practices.md` §7, `gles-api-standards.md` §4 |\r\n| 13 | [mali-pls-multiview](13-mali-pls-multiview.md) | Mali PLS & Multiview/Foveated | 3.0+ ext | `mali-arm-best-practices.md` §1,§2 |\r\n| 14 | [adreno-gmem-vrs-lrz](14-adreno-gmem-vrs-lrz.md) | Adreno GMEM/VRS/LRZ/FlexRender | 3.0+ ext | `adreno/*.md` |\r\n| 15 | [windows-egl-angle](15-windows-egl-angle.md) | Windows 平台 EGL/ANGLE/Windows-on-ARM | 3.0/3.1 via ANGLE | `windows-platform.md` |\r\n| 16 | [powervr-hsr-img-extensions](16-powervr-hsr-img-extensions.md) | PowerVR HSR/PLS/IMG 扩展/Tile 带宽 | 3.0+ ext | `powervr/*.md` |\r\n\r\n## 卡片格式说明\r\n\r\n每张卡片遵循统一结构：\r\n\r\n```\r\n# [标题]\r\n> Category | GLES Version | Source\r\n\r\n## 核心规则        ← 必须遵守的硬性规则（生成代码时强制执行）\r\n## 代码模式        ← 正确用法的典型代码片段\r\n## 常见陷阱        ← 高频错误 & 其后果\r\n## 关联卡片        ← 交叉引用\r\n```\r\n\r\n## 使用方式\r\n\r\n- **代码生成时**：根据涉及的功能域加载对应卡片的核心规则作为约束。\r\n- **代码审查时**：对照卡片的\"常见陷阱\"逐条检查。\r\n- **性能诊断时**：从 `08-tbdr-bandwidth` 和 `09-overdraw-fillrate` 入手定位瓶颈。\n\nFile v0.1.0:references/rules/adreno/README.md\n\n# Qualcomm Adreno GPU Best Practices (Distilled)\r\n\r\n> **Source of truth:** Snapdragon Game Studios / Qualcomm\r\n> [*Adreno GPU OpenGL ES Code Sample Framework*](https://github.com/SnapdragonGameStudios/adreno-gpu-opengl-es-code-sample-framework)\r\n> and the Qualcomm *Adreno GPU on Mobile: Best Practices* documentation.\r\n>\r\n> This module distills the vendor-recommended Adreno techniques into small,\r\n> focused rule files (one topic per file) so they stay easy to consume during\r\n> code generation and review. Everything here targets **Qualcomm Adreno** GPUs,\r\n> which use a **tiled / binning** architecture; most rules also help other\r\n> tile-based mobile GPUs (ARM Mali, Imagination PowerVR).\r\n\r\n## Vocabulary: Adreno vs. the generic TBDR terms\r\n\r\n| Adreno term | Meaning | Generic equivalent |\r\n|:---|:---|:---|\r\n| **GMEM** (Graphics Memory) | Fast on-chip tile memory | Tile memory / on-chip framebuffer |\r\n| **GMEM Load** | Copy a tile from system memory *into* GMEM at render-pass start | Tile \"load\" / unresolve |\r\n| **GMEM Store** | Write a GMEM tile *back* to system memory at render-pass end | Tile \"store\" / resolve |\r\n| **Binning / FlexRender** | Pass that sorts primitives into tile bins | Tiling / deferred binning |\r\n| **LRZ** (Low Resolution Z) | Early coarse depth rejection of hidden fragments | Hidden-surface removal / early-Z |\r\n\r\n## Rule files in this module\r\n\r\n| File | Topic | Maps to sample |\r\n|:---|:---|:---|\r\n| [`gmem-load-store.md`](gmem-load-store.md) | Avoid GMEM loads, reduce GMEM stores | `avoid_gmem_loads`, `reduce_gmem_stores` |\r\n| [`efficient-msaa.md`](efficient-msaa.md) | On-tile MSAA resolve without a blit | `msaa` |\r\n| [`variable-rate-shading.md`](variable-rate-shading.md) | `QCOM_shading_rate` (VRS) | `shading_rate` |\r\n| [`lrz-and-flexrender.md`](lrz-and-flexrender.md) | LRZ, FlexRender/binning, depth choices | `hello_gltf` scenes, general arch |\r\n| [`frame-extrapolation-and-upscaling.md`](frame-extrapolation-and-upscaling.md) | AFME, motion estimation, SGSR2 | `amfe_power_saving`, `motion_estimation`, `sgsr2` |\r\n\r\n## Golden rules (one-line summary)\r\n\r\n1. **Clear or invalidate every attachment at render-pass start** → kills GMEM loads.\r\n2. **Invalidate transient attachments (depth/stencil/MSAA) at render-pass end** → kills GMEM stores.\r\n3. **Resolve MSAA on-tile** via `EXT_multisampled_render_to_texture`, never a manual blit.\r\n4. **Never break LRZ**: draw opaque front-to-back, avoid `discard` / fragment depth writes where possible.\r\n5. **Spend fewer fragment invocations**: use `QCOM_shading_rate` on low-detail draws and temporal upscaling (SGSR2) instead of shading every pixel every frame.\n\nFile v0.1.0:references/rules/powervr/README.md\n\n# Imagination PowerVR GPU Best Practices (Distilled)\r\n\r\n> **Source of truth:** Imagination Technologies\r\n> [*PowerVR Native SDK — OpenGL ES Framework*](https://github.com/powervr-graphics/Native_SDK/tree/master/framework/PVRUtils/OpenGLES)\r\n> and the official *PowerVR Performance Recommendations* / *Introduction to PowerVR for Developers* documentation.\r\n>\r\n> This module distills PowerVR-specific GLES techniques into focused rule files.\r\n> PowerVR uses a **Tile-Based Deferred Rendering (TBDR)** architecture with **full\r\n> Hidden Surface Removal (HSR)** — a unique hardware pass that eliminates ALL\r\n> invisible fragments before shading. This fundamentally changes rendering strategy\r\n> compared to both desktop GPUs and even other mobile TBDR GPUs (Mali, Adreno).\r\n\r\n## Vocabulary: PowerVR-specific terms\r\n\r\n| PowerVR term | Meaning | Generic equivalent |\r\n|:---|:---|:---|\r\n| **HSR** (Hidden Surface Removal) | Hardware pass that processes ALL primitives in a tile, determines visibility, then shades ONLY visible fragments | Early-Z / LRZ (partial equivalent — HSR is more thorough) |\r\n| **ISP** (Image Synthesis Processor) | Fixed-function unit performing HSR + depth/stencil tests before fragment shading | Rasterizer + early-Z |\r\n| **USC** (Unified Shading Cluster) | Shader processor units | Shader cores / ALUs |\r\n| **Parameter Buffer (PB)** | Off-chip memory storing transformed geometry for tiling | Bin buffer / tiling buffer |\r\n| **SPM** (Smart Parameter Management) | Hardware manages PB overflow by partial renders | Binning overflow handling |\r\n| **Tile Memory** | Fast on-chip framebuffer per tile (~32×32 pixels) | GMEM (Adreno) / Tile buffer (Mali) |\r\n| **PVRTC** | PowerVR Texture Compression (2bpp / 4bpp); legacy, prefer ASTC/ETC2 on modern HW | — |\r\n| **PVRScope** | PowerVR's GPU profiling tool | Streamline (Mali) / Snapdragon Profiler (Adreno) |\r\n\r\n## Rule files in this module\r\n\r\n| File | Topic | Maps to SDK example/doc |\r\n|:---|:---|:---|\r\n| [`hsr-and-rendering-order.md`](hsr-and-rendering-order.md) | HSR, no depth pre-pass, alpha test vs blend, draw order | Performance Recommendations §HSR, forum guidance |\r\n| [`pixel-local-storage-and-deferred.md`](pixel-local-storage-and-deferred.md) | PLS for on-chip deferred rendering | `DeferredShading` example |\r\n| [`img-extensions.md`](img-extensions.md) | IMG_framebuffer_downsample, IMG_texture_filter_cubic, binary shaders | `IMGFramebufferDownsample`, `IMGTextureFilterCubic`, `BinaryShaders` |\r\n| [`bandwidth-and-tile-management.md`](bandwidth-and-tile-management.md) | Clear/invalidate, transient stores, parameter buffer, MSAA | `PostProcessing`, Performance Recommendations |\r\n\r\n## Golden rules (one-line summary)\r\n\r\n1. **Do NOT use a depth pre-pass** — PowerVR HSR already eliminates 100% of hidden opaque fragments; a depth pre-pass doubles geometry cost with zero shading benefit.\r\n2. **Avoid `discard` / alpha test where possible** — it delays HSR, forcing the ISP to defer visibility decisions until after shading; prefer alpha blend for semi-transparent cutouts.\r\n3. **Clear or invalidate every attachment at render-pass start** — prevents tile memory load from DRAM.\r\n4. **Invalidate transient attachments (depth/stencil) at pass end** — prevents unnecessary tile memory store to DRAM.\r\n5. **Use `GL_EXT_shader_pixel_local_storage` for deferred rendering** — keeps G-Buffer on-chip in tile memory, zero DRAM round-trips.\r\n6. **Keep geometry complexity bounded** — excessive vertices overflow the Parameter Buffer, triggering partial renders (SPM) that kill performance.\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn759mntake5xe2rz3cvmka7c582er3a\",\n  \"slug\": \"gles-rendering-expert-skill\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1785058079244\n}\n\nFile v0.1.0:references/cards/01-api-version-constraints.md\n\n# API 版本约束 & 桌面 OpenGL 禁用\r\n\r\n> **Category**: API Standards | **GLES Version**: 3.0 / 3.1 / 3.2 | **Source**: `references/rules/gles-api-standards.md`\r\n\r\n## 核心规则\r\n\r\n1. **默认目标 OpenGL ES 3.0 + GLSL ES 3.00**；仅在明确需要时使用 3.1/3.2 特性。\r\n2. **严禁生成任何桌面 OpenGL API**：`glBegin/glEnd`、`glVertex*`、`glMatrixMode`、`glLight`、`glFog`、`glPushAttrib`、Display Lists、`GL_QUADS`、`glPolygonMode`、`glDrawPixels`、Evaluators 等。\r\n3. 版本-特性对应：\r\n   - **3.0**: VAO, UBO, MRT, ETC2, Transform Feedback, Instancing, PBO\r\n   - **3.1**: Compute Shader, SSBO, Image Load/Store, Indirect Draw, Separate Shader Objects\r\n   - **3.2**: Geometry Shader, Tessellation, ASTC, Debug Output, Blend Equation Advanced\r\n4. 使用扩展前必须：查询可用性 → `eglGetProcAddress` 加载函数指针 → 提供 fallback。\r\n5. 图元类型仅限：`GL_TRIANGLES`、`GL_TRIANGLE_STRIP`、`GL_TRIANGLE_FAN`、`GL_POINTS`、`GL_LINES`、`GL_LINE_STRIP`、`GL_LINE_LOOP`。\r\n\r\n## 代码模式\r\n\r\n```cpp\r\n// ✅ 正确的 GLES 3.0 绘制\r\nglBindVertexArray(vao);\r\nglDrawElements(GL_TRIANGLES, indexCount, GL_UNSIGNED_SHORT, nullptr);\r\n\r\n// ✅ 扩展检查\r\nconst char* exts = (const char*)glGetString(GL_EXTENSIONS);\r\nbool hasASTC = strstr(exts, \"GL_KHR_texture_compression_astc_ldr\") != nullptr;\r\n```\r\n\r\n## 常见陷阱\r\n\r\n| 陷阱 | 后果 | 修正 |\r\n|:-----|:-----|:-----|\r\n| 使用 `glBegin/glEnd` 立即模式 | 编译失败（GLES 无此 API） | VBO + VAO + `glDrawArrays` |\r\n| `GL_QUADS` 图元 | 无效枚举 | 拆分为三角形 |\r\n| `glPolygonMode(GL_LINE)` 线框 | 不支持 | 重心坐标 shader 或 Geometry Shader (3.2) |\r\n| `glLineWidth(>1.0)` | 仅保证 1.0 | 用几何体模拟宽线 |\r\n| 未检查扩展直接使用 | 运行时崩溃 | 查询 + fallback |\r\n\r\n## 关联卡片\r\n\r\n- [05-shader-precision-layout](05-shader-precision-layout.md) — GLSL ES 版本对应\r\n- [02-texture-formats-compression](02-texture-formats-compression.md) — 格式严格配对\r\n- [06-compute-shader](06-compute-shader.md) — 3.1+ 特性\n\nFile v0.1.0:references/cards/02-texture-formats-compression.md\n\n# 纹理格式 & ASTC/ETC2 压缩\r\n\r\n> **Category**: Texture | **GLES Version**: 3.0+ | **Source**: `references/rules/gles-api-standards.md` §3, `references/rules/mali-arm-best-practices.md` §4\r\n\r\n## 核心规则\r\n\r\n1. **始终使用压缩纹理**（ASTC 优先，ETC2 为 GLES 3.0 保底）。\r\n2. GLES 要求 `glTexImage2D` 的 internalformat/format/type **严格配对**，不可随意组合。\r\n3. 优先使用 **`glTexStorage2D`（不可变存储）** 而非 `glTexImage2D`——驱动优化更好，且 shader image 必须用不可变纹理。\r\n4. **`glTexStorage2D` 的 `levels` 参数必须满足 `levels >= 1 + floor(log2(max(w,h)))`**——这是完整 mipmap chain 的层数。若 levels 小于此值，后续调用 `glGenerateMipmap` 会触发 `GL_INVALID_OPERATION`（不可变纹理不允许新增层级）。\r\n5. **始终生成/附带 mipmaps**——减少纹理缓存 miss、带宽和走样。\r\n6. ASTC 使用前必须 **运行时检查** `GL_KHR_texture_compression_astc_ldr`。\r\n7. 选择 **满足视觉要求的最大 ASTC block（最低 bpp）**，不要一刀切 4x4。\r\n\r\n## 代码模式\r\n\r\n```cpp\r\n// ✅ 不可变纹理 + mipmap（levels 必须覆盖完整 mip chain）\r\nGLsizei mipLevels = 1 + static_cast<GLsizei>(std::floor(std::log2(\r\n    static_cast<float>(std::max(w, h)))));\r\nglBindTexture(GL_TEXTURE_2D, tex);\r\nglTexStorage2D(GL_TEXTURE_2D, mipLevels, GL_RGBA8, w, h);\r\nglTexSubImage2D(GL_TEXTURE_2D, 0, 0, 0, w, h, GL_RGBA, GL_UNSIGNED_BYTE, data);\r\nglGenerateMipmap(GL_TEXTURE_2D);  // OK: levels 已预分配\r\nglTexParameteri(GL_TEXTURE_2D, GL_TEXTURE_MIN_FILTER, GL_LINEAR_MIPMAP_LINEAR);\r\n\r\n// ✅ ASTC 上传\r\nglCompressedTexImage2D(GL_TEXTURE_2D, 0, GL_COMPRESSED_RGBA_ASTC_6x6_KHR,\r\n                       w, h, 0, dataSize, astcData);\r\n```\r\n\r\n**ASTC Block 选择表：**\r\n\r\n| Block | bpp | 适用场景 |\r\n|:------|:----|:---------|\r\n| 4×4 | 8.00 | UI、主角 albedo（最高质量） |\r\n| 6×6 | 3.56 | 通用 albedo / 平衡 |\r\n| 8×8 | 2.00 | 大面积 diffuse / 低频细节 |\r\n| 12×12 | 0.89 | 天空盒 / 背景（最低成本） |\r\n\r\n## 常见陷阱\r\n\r\n| 陷阱 | 后果 | 修正 |\r\n|:-----|:-----|:-----|\r\n| `glTexStorage2D` 的 levels < 完整 mip chain | `glGenerateMipmap` 触发 `GL_INVALID_OPERATION` | `levels = 1 + floor(log2(max(w,h)))` |\r\n| `glTexImage2D` 传 `GL_RGBA8` 作 internalformat (GLES 2.0 兼容模式) | INVALID_ENUM | 用 `GL_RGBA` 或改用 `glTexStorage2D` |\r\n| 未检查 ASTC 扩展直接上传 | INVALID_ENUM / 黑屏 | 运行时查询，fallback 到 ETC2 |\r\n| 法线贴图用 ETC2 | 通道耦合导致质量差 | ASTC uncorrelated 模式或 RG 双通道 |\r\n| 无 mipmap 的 minify 采样 | 闪烁走样 + 带宽浪费 | 生成 mipmap + `GL_LINEAR_MIPMAP_LINEAR` |\r\n| 用 `glTexImage2D` 创建 shader image 纹理 | 未定义行为 | 必须 `glTexStorage*` 不可变 |\r\n\r\n## 关联卡片\r\n\r\n- [01-api-version-constraints](01-api-version-constraints.md) — 扩展检查流程\r\n- [03-buffer-objects](03-buffer-objects.md) — PBO 异步纹理上传\r\n- [08-tbdr-bandwidth](08-tbdr-bandwidth.md) — 纹理带宽是 TBDR 主要开销\n\nFile v0.1.0:references/cards/03-buffer-objects.md\n\n# Buffer Objects — VAO / VBO / UBO / SSBO / PBO\r\n\r\n> **Category**: Buffer | **GLES Version**: 3.0 / 3.1 | **Source**: `references/rules/gles-api-standards.md` §4, `references/rules/glsl-es-optimization.md` §3, §5.3\r\n\r\n## 核心规则\r\n\r\n1. **始终使用 VAO**（GLES 3.0+ 默认 VAO name 0 已废弃）；先绑 VAO 再设顶点属性。\r\n2. VBO usage hint 如实设置：`GL_STATIC_DRAW`（一次上传多次绘制）、`GL_DYNAMIC_DRAW`（每帧更新）、`GL_STREAM_DRAW`（每帧更新且只绘一次）。\r\n3. 全量更新用 `glMapBufferRange` + `GL_MAP_WRITE_BIT | GL_MAP_INVALIDATE_BUFFER_BIT`（orphan 旧存储，避免同步阻塞）。\r\n4. **UBO** 用 `std140` 布局，按更新频率分组（per-frame vs per-material），绑定到显式 binding point。\r\n5. **SSBO**（3.1+）用 **`std430`** 布局（紧凑打包，不像 std140 把标量数组 pad 到 vec4）；最小保证 128 MiB；支持 unsized trailing array。\r\n6. **PBO** 用于异步纹理上传/像素回读，避免 CPU-GPU 同步阻塞。\r\n7. 映射指针 **不得跨帧持有** 而不 unmap。\r\n\r\n## 代码模式\r\n\r\n```cpp\r\n// ✅ VAO + VBO 标准流程\r\nglGenVertexArrays(1, &vao);\r\nglBindVertexArray(vao);\r\nglBindBuffer(GL_ARRAY_BUFFER, vbo);\r\nglBufferData(GL_ARRAY_BUFFER, size, data, GL_STATIC_DRAW);\r\nglEnableVertexAttribArray(0);\r\nglVertexAttribPointer(0, 3, GL_FLOAT, GL_FALSE, stride, (void*)offset);\r\nglBindVertexArray(0);\r\n\r\n// ✅ UBO std140 (binding set from host via glUniformBlockBinding in ES 3.00;\r\n//    layout(binding=N) requires ES 3.10+)\r\nlayout(std140) uniform Matrices {\r\n    highp mat4 u_Model;   // offset 0\r\n    highp mat4 u_View;    // offset 16\r\n    highp mat4 u_Proj;    // offset 32\r\n};\r\n\r\n// ✅ SSBO std430 (requires GLES 3.1 / #version 310 es)\r\nlayout(std430, binding = 1) buffer Particles {\r\n    Particle data[];      // unsized array, query .length()\r\n};\r\n```\r\n\r\n## 常见陷阱\r\n\r\n| 陷阱 | 后果 | 修正 |\r\n|:-----|:-----|:-----|\r\n| 不绑 VAO 直接设 attrib | INVALID_OPERATION (GLES 3.0+) | 先 `glBindVertexArray` |\r\n| 每帧 `glBufferData` 更新 UBO | 重新分配存储，驱动 stall | `glBufferSubData` 或 `glMapBufferRange` |\r\n| SSBO 用 `std140` | 标量数组 4× 内存浪费 | 改用 `std430` |\r\n| 映射指针跨帧不 unmap | 未定义行为 / 内存泄漏 | 每帧结束前 `glUnmapBuffer` |\r\n| PBO 回读后立即 map 同一 PBO | CPU 阻塞等待 GPU | 双 PBO ping-pong |\r\n\r\n## 关联卡片\r\n\r\n- [06-compute-shader](06-compute-shader.md) — SSBO 在 compute 中的使用\r\n- [11-synchronization](11-synchronization.md) — Buffer orphaning & fence\r\n- [12-draw-call-optimization](12-draw-call-optimization.md) — Instancing + VBO\n\nFile v0.1.0:references/cards/04-framebuffer-objects.md\n\n# Framebuffer Objects — 生命周期 & MRT & Blit\r\n\r\n> **Category**: FBO | **GLES Version**: 3.0+ | **Source**: `references/rules/gles-api-standards.md` §5, `references/rules/tbdr-bandwidth-rules.md` §2\r\n\r\n## 核心规则\r\n\r\n1. **Render Pass 开始应 `glClear` 或 `glInvalidateFramebuffer` 所有 attachment**——否则 TBDR GPU 被迫从 DRAM 加载旧 tile 内容（GMEM Load）。例外：需要保留旧内容的场景（增量渲染、Load-then-blend、多帧累积）可省略。\r\n2. **Render Pass 结束必须 `glInvalidateFramebuffer` 不再使用的 attachment**（尤其 depth/stencil）——跳过 DRAM 写回（GMEM Store）。\r\n3. 创建后验证 `glCheckFramebufferStatus() == GL_FRAMEBUFFER_COMPLETE`（debug 构建）。\r\n4. MRT：`glDrawBuffers(n, bufs)` 启用多目标；fragment 输出用 `layout(location = N) out vec4`。\r\n5. `glBlitFramebuffer` 在 TBDR 上触发 tile resolve——**能直接渲染到目标就不要 blit**。\r\n6. 最小化 FBO 切换次数：每次切换 = 一次 Store + 一次 Load。按 target 分组批处理 draw call。\r\n\r\n## 代码模式\r\n\r\n```cpp\r\n// ✅ 完整 FBO 生命周期\r\nglBindFramebuffer(GL_FRAMEBUFFER, fbo);\r\nglViewport(0, 0, w, h);\r\nglClear(GL_COLOR_BUFFER_BIT | GL_DEPTH_BUFFER_BIT);  // 避免 GMEM Load\r\n\r\nDrawScene();\r\n\r\n// depth/stencil 后续不再读取 → invalidate\r\nGLenum discards[] = { GL_DEPTH_ATTACHMENT, GL_STENCIL_ATTACHMENT };\r\nglInvalidateFramebuffer(GL_FRAMEBUFFER, 2, discards);  // 避免 GMEM Store\r\n\r\nglBindFramebuffer(GL_FRAMEBUFFER, 0);  // 切到下一个 target\r\n```\r\n\r\n## 常见陷阱\r\n\r\n| 陷阱 | 后果 | 修正 |\r\n|:-----|:-----|:-----|\r\n| 绑定 FBO 后不 clear 直接 draw | 驱动执行 DRAM→Tile Load（全屏读） | 加 `glClear` 或 `glInvalidateFramebuffer` |\r\n| Scissor 限制下 clear | 只清部分区域 → 仍需 Load 保留其余 | 全 clear 或 invalidate |\r\n| Depth/stencil 不 invalidate | 每帧多一次全屏 DRAM 写 | pass 结束 invalidate |\r\n| 频繁 FBO ping-pong | N 次 Store + N 次 Load | 按 target 分组，减少切换 |\r\n| 用 blit 做 MSAA resolve | 额外 DRAM 往返 | `EXT_multisampled_render_to_texture` on-tile resolve |\r\n\r\n## 关联卡片\r\n\r\n- [08-tbdr-bandwidth](08-tbdr-bandwidth.md) — TBDR 带宽模型\r\n- [10-msaa-antialiasing](10-msaa-antialiasing.md) — MSAA resolve 策略\r\n- [14-adreno-gmem-vrs-lrz](14-adreno-gmem-vrs-lrz.md) — Adreno GMEM Load/Store 细节\n\nFile v0.1.0:references/cards/05-shader-precision-layout.md\n\n# GLSL ES 精度控制 & Shader I/O 布局\r\n\r\n> **Category**: Shader | **GLES Version**: GLSL ES 3.00 / 3.20 | **Source**: `references/rules/glsl-es-optimization.md` §1, §2, §7, §8\r\n\r\n## 核心规则\r\n\r\n1. **每个 shader 第一行必须是 `#version 300 es`（或 `310 es` / `320 es`）**，之前不得有空行或注释。\r\n2. **Fragment shader 必须显式声明精度**：`precision mediump float;`（无默认值，缺失则编译失败）。\r\n3. 精度选择：\r\n   - `highp`：顶点位置、深度、MVP 矩阵、时间累加器\r\n   - `mediump`：纹理坐标、颜色、法线、光照方向\r\n   - `lowp`：仅 8-bit 颜色输出（极少使用）\r\n4. **始终使用 `layout(location = N)`** 绑定 attribute 和 fragment output；禁止依赖 `glGetAttribLocation`。\r\n5. 使用 `in`/`out`（GLES 3.0+）；**禁止** `attribute`/`varying`/`gl_FragColor`/`texture2D()`。\r\n6. Fragment 优化：避免 divergent branch（用 `mix`/`step`/`smoothstep`）；循环用常量上界；尽量用内置函数。\r\n7. 可移到 vertex shader 或 CPU 的计算不要留在 fragment shader（per-pixel 成本最高）。\r\n\r\n## 代码模式\r\n\r\n```glsl\r\n// ✅ Vertex Shader\r\n#version 300 es\r\nprecision highp float;\r\nlayout(location = 0) in vec3 a_Position;\r\nlayout(location = 1) in vec3 a_Normal;\r\nlayout(location = 2) in vec2 a_TexCoord;\r\nlayout(std140) uniform Matrices {  // binding via glUniformBlockBinding (ES 3.00)\r\n    highp mat4 u_MVP;\r\n    highp mat3 u_NormalMat;\r\n};\r\nout mediump vec3 v_Normal;\r\nout mediump vec2 v_UV;\r\nvoid main() {\r\n    v_Normal = normalize(u_NormalMat * a_Normal);\r\n    v_UV = a_TexCoord;\r\n    gl_Position = u_MVP * vec4(a_Position, 1.0);\r\n}\r\n\r\n// ✅ Fragment Shader\r\n#version 300 es\r\nprecision mediump float;\r\nlayout(location = 0) out vec4 o_Color;\r\nin mediump vec3 v_Normal;\r\nin mediump vec2 v_UV;\r\nuniform sampler2D u_Albedo;\r\nvoid main() {\r\n    o_Color = texture(u_Albedo, v_UV) * vec4(v_Normal * 0.5 + 0.5, 1.0);\r\n}\r\n```\r\n\r\n## 常见陷阱\r\n\r\n| 陷阱 | 后果 | 修正 |\r\n|:-----|:-----|:-----|\r\n| Fragment 不声明 `precision` | 编译错误 | 加 `precision mediump float;` |\r\n| 顶点位置用 `mediump` | 大坐标抖动 | 位置/矩阵一律 `highp` |\r\n| 大纹理 UV 用 `mediump` | 接缝/闪烁 | 4096+ 纹理 UV 升 `highp` |\r\n| 使用 `gl_FragColor` | GLES 3.0 编译失败 | `layout(location=0) out vec4` |\r\n| `texture2D()` | GLES 3.0 编译失败 | `texture()` |\r\n| Divergent `if` 选纹理 | SIMD 两路都执行 | `mix(texA, texB, selector)` |\r\n| 时间 uniform 用 `mediump` | 溢出/动画卡顿 | `highp` + `mod()` 回绕 |\r\n\r\n## 关联卡片\r\n\r\n- [01-api-version-constraints](01-api-version-constraints.md) — 版本-特性对应\r\n- [06-compute-shader](06-compute-shader.md) — Compute shader 精度 (默认 highp)\r\n- [09-overdraw-fillrate](09-overdraw-fillrate.md) — Fragment 成本优化\n\nFile v0.1.0:references/cards/06-compute-shader.md\n\n# Compute Shader — 工作组织 & 同步\r\n\r\n> **Category**: Compute | **GLES Version**: 3.1+ (GLSL ES 3.10) | **Source**: `references/rules/glsl-es-optimization.md` §5, `references/rules/mali-arm-best-practices.md` §3\r\n\r\n## 核心规则\r\n\r\n1. `#version 310 es` + `layout(local_size_x, local_size_y, local_size_z) in;`；local_size 选 warp 倍数（Adreno 32, Mali 16）；最低保证 128 invocations/workgroup。\r\n2. **SSBO 用 `std430`**（紧凑）；最小 128 MiB；支持 unsized array + `.length()`。\r\n3. **Shader image 纹理必须不可变**（`glTexStorage*`）；format qualifier 必须匹配 `glBindImageTexture` 的 format 参数；用 `layout(binding=N)` 而非 `glUniform1i`。\r\n4. **`shared` 内存** ≥16 KiB，未初始化、非持久——用于 workgroup 内数据复用以减少带宽。\r\n5. **同步正确性（关键）**：\r\n   - Workgroup 内：`memoryBarrierShared()` **必须在** `barrier()` **之前**；`barrier()` 只能在 dynamically-uniform 控制流中调用。\r\n   - 跨 GL 命令：`glMemoryBarrier(<BITS>)` 描述下一步如何读取数据（如 `GL_VERTEX_ATTRIB_ARRAY_BARRIER_BIT`）。\r\n6. **TBDR 友好**：fragment image load/store 用 `glMemoryBarrierByRegion()`（避免全 tile flush）；加 `layout(early_fragment_tests) in;` 恢复 early-Z。\r\n7. Compute 相对 GL 其余部分 **异步执行**——dispatch 后不自动同步。\r\n\r\n## 代码模式\r\n\r\n```glsl\r\n#version 310 es\r\nlayout(local_size_x = 64) in;\r\nlayout(std430, binding = 0) buffer ParticleBuf {\r\n    Particle particles[];\r\n};\r\nuniform uint u_Count;\r\nuniform float u_Dt;\r\n\r\nshared float s_MaxSpeed;  // workgroup-local reduction\r\n\r\nvoid main() {\r\n    uint id = gl_GlobalInvocationID.x;\r\n    if (id >= u_Count) return;\r\n    particles[id].pos += particles[id].vel * u_Dt;\r\n\r\n    // Workgroup reduction example\r\n    s_MaxSpeed = 0.0;\r\n    memoryBarrierShared();  // ← MUST be before barrier()\r\n    barrier();\r\n    // ... atomicMax into s_MaxSpeed ...\r\n}\r\n```\r\n\r\n```cpp\r\n// Host: dispatch → barrier → draw from same SSBO\r\nglDispatchCompute(groups, 1, 1);\r\nglMemoryBarrier(GL_VERTEX_ATTRIB_ARRAY_BARRIER_BIT);\r\nglDrawElements(GL_TRIANGLES, count, GL_UNSIGNED_SHORT, nullptr);\r\n```\r\n\r\n## 常见陷阱\r\n\r\n| 陷阱 | 后果 | 修正 |\r\n|:-----|:-----|:-----|\r\n| `barrier()` 前不调 `memoryBarrierShared()` | 读到 stale shared 数据 | 先 memory barrier 再 execution barrier |\r\n| `barrier()` 在 divergent branch 中 | 死锁 | 仅 uniform 分支或所有线程都到达 |\r\n| Dispatch 后直接 draw 同一 SSBO | 读到旧数据 | `glMemoryBarrier(GL_VERTEX_ATTRIB_ARRAY_BARRIER_BIT)` |\r\n| 用 `glTexImage2D` 创建 image 纹理 | 未定义行为 | `glTexStorage2D` 不可变 |\r\n| 全屏 fragment image 用 `glMemoryBarrier` | 全 tile flush 到 DRAM | `glMemoryBarrierByRegion()` |\r\n| local_size 非 warp 倍数 | 线程浪费 | Adreno→32, Mali→16 的倍数 |\r\n\r\n## 关联卡片\r\n\r\n- [03-buffer-objects](03-buffer-objects.md) — SSBO std430 布局\r\n- [11-synchronization](11-synchronization.md) — Memory barrier 全表\r\n- [09-overdraw-fillrate](09-overdraw-fillrate.md) — early_fragment_tests","readmeExcerpt":"Skill: OpenGL ES Rendering Expert Skill Owner: oahc09 Summary: Senior OpenGL ES & Graphics Rendering Expert skill for AI coding assistants. Enforces OpenGL ES 3.0/3.1/3.2 API boundaries, TBDR bandwidth optimization for ARM Mali / Qualcomm Adreno / PowerVR GPUs, EGL context lifecycle management, and GLSL ES 3.00/3.10/3.20 precision rules across mobile (Android), Windows (ANGLE / Windows-on-ARM), and Embedded Linux. Us","codeSnippets":[],"executableExamples":[],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\r\nname: gles-rendering-expert-skill\r\ndescription: \"Senior OpenGL ES & Graphics Rendering Expert skill for AI coding assistants. Enforces OpenGL ES 3.0/3.1/3.2 API boundaries, TBDR bandwidth optimization for ARM Mali / Qualcomm Adreno / PowerVR GPUs, EGL context lifecycle management, and GLSL ES 3.00/3.10/3.20 precision rules across mobile (Android), Windows (ANGLE / Windows-on-ARM), and Embedded Linux. Use when generating or reviewing GLES C++17 code, GLSL ES shaders, FBO pipelines, or diagnosing GPU performance issues on any of these platforms.\"\r\ndescription_en: \"Expert skill for OpenGL ES 3.x rendering across mobile, Windows (ANGLE), and Embedded Linux: API constraints, TBDR bandwidth optimization, EGL context management, GLSL ES precision, and RAII C++17 code generation.\"\r\ndescription_zh: \"OpenGL ES 3.x 渲染专家技能，覆盖移动端、Windows（ANGLE / Windows-on-ARM）与嵌入式 Linux：API 约束、TBDR 带宽优化、EGL 上下文管理、GLSL ES 精度控制及 RAII C++17 代码生成。\"\r\nlicense: MIT\r\nmetadata:\r\n  author: gles-rendering-expert-skill contributors\r\n  version: 1.0.0\r\n  last-updated: 2026-07-25\r\n  keywords: \"OpenGL ES, GLES 3.0, GLES 3.1, GLES 3.2, GLSL ES, EGL, TBDR, ANGLE, Mali, Adreno, PowerVR, shader optimization, bandwidth optimization, Android NDK, Windows-on-ARM, Embedded Linux, RAII C++17\"\r\n---\r\n\r\n# Role: Senior OpenGL ES & Graphics Rendering Expert\r\n\r\nYou are a World-Class Graphics Rendering Expert specializing in **OpenGL ES (3.0/3.1/3.2)**, **EGL Context Management**, and **TBDR (Tile-Based Deferred Rendering) GPU Architecture Optimization**. Your primary targets are tile-based mobile GPUs (ARM Mali, Qualcomm Adreno, Imagination PowerVR), and you are equally fluent in running GLES on **Windows** (via ANGLE, and natively on Windows-on-ARM / Adreno) and on **Embedded Linux** (GBM/EGL).\r\n\r\nYour mission is to generate production-grade, bandwidth-optimized rendering code and provide expert-level guidance on OpenGL ES engine architecture, shader optimization, and GPU performance tuning — tuned for mobile-class TBDR hardware but portable across Android, Windows, and Embedded Linux.\r\n\r\n---\r\n\r\n## Mandatory API Rules\r\n\r\n### Target API Version\r\n- **Primary**: OpenGL ES 3.0 / 3.1 / 3.2 with GLSL ES 3.00 / 3.20.\r\n- **Legacy awareness**: Understand OpenGL ES 2.0 concepts for migration guidance, but always default to modern 3.0+ idioms.\r\n\r\n### Strict Prohibitions — Desktop OpenGL Functions NEVER to Generate\r\n| Forbidden API | Reason |\r\n|:---|:---|\r\n| `glBegin` / `glEnd` / `glVertex*` (immediate mode) | Not available in any GLES version |\r\n| `glPolygonMode(GL_FRONT_AND_BACK, GL_LINE)` | Desktop-only; GLES has no polygon mode |\r\n| `glDrawBuffer` / `glReadBuffer` (arbitrary) | Use `glDrawBuffers` (GLES 3.0+) with MRT |\r\n| `glLineWidth` with value > 1.0 | GLES only guarantees width = 1.0 |\r\n| `glPushAttrib` / `glPopAttrib` | Not available in GLES |\r\n| `glEnableClientState` / `glDisableClientState` | Use VAO/VBO (GLES 3.0+) |\r\n| `glGenLists` / `glCallList` (display lists) | Not available in GLES |\r\n| `glBitm"},{"path":"README.md","content":"# gles-rendering-expert-skill\r\n\r\n![GLES Rendering Expert Skill](assets/gles-rendering-expert-banner.png)\r\n\r\n> **AI Expert Skill for OpenGL ES 3.x Mobile Rendering** — Inject precise GLES state-machine knowledge, TBDR bandwidth optimization rules, and production-grade C++17/GLSL ES code patterns into your AI coding assistant.\r\n\r\n## Why This Skill?\r\n\r\nLarge Language Models frequently:\r\n- **Confuse Desktop OpenGL with OpenGL ES** — generating `glBegin/glEnd`, `glPolygonMode`, or invalid texture formats.\r\n- **Ignore TBDR architecture** — producing code that causes massive DRAM bandwidth waste on mobile GPUs (Mali, Adreno, PowerVR).\r\n- **Miss EGL/context management** — overlooking context loss recovery, shared context synchronization, and proper lifecycle.\r\n\r\nThis skill eliminates these failure modes by constraining AI output to **mobile-first, TBDR-aware, GLES 3.0+ idioms**.\r\n\r\n## Quick Start\r\n\r\n### Cursor IDE\r\n```bash\r\n# Copy the skill file to your project's cursor rules\r\ncp SKILL.md .cursor/rules/gles-rendering-expert.mdc\r\n```\r\n\r\n### Claude Projects / ChatGPT Custom GPTs\r\nCopy the entire content of [`SKILL.md`](SKILL.md) into your project's system instructions or custom GPT configuration.\r\n\r\n### Windsurf / Roo-Code / Other AI Tools\r\nPaste `SKILL.md` content as a system prompt or custom rule in your tool's configuration.\r\n\r\n## Repository Structure\r\n\r\n```\r\ngles-rendering-expert-skill/\r\n├── SKILL.md                       # Core System Prompt (AI Skill entry point)\r\n├── README.md                      # This file\r\n├── LICENSE                        # MIT License\r\n├── .cursorrules                   # Cursor IDE quick-link\r\n├── references/                    # All reference material (rules, cards, examples)\r\n│   ├── rules/                     # Modular rule documents\r\n│   │   ├── gles-api-standards.md      # API version constraints & desktop API prohibition\r\n│   │   ├── tbdr-bandwidth-rules.md    # TBDR bandwidth & FBO discard optimization\r\n│   │   ├── egl-and-context.md         # EGL lifecycle & multi-thread context sync\r\n│   │   ├── glsl-es-optimization.md    # GLSL ES precision & shader optimization\r\n│   │   ├── mali-arm-best-practices.md # ARM Mali techniques (OpenGL ES SDK for Android)\r\n│   │   ├── windows-platform.md        # Windows: ANGLE, Windows-on-ARM, NDK host\r\n│   │   ├── adreno/                    # Qualcomm Adreno techniques, one topic per file\r\n│   │   │   ├── README.md              # Adreno module index & GMEM/LRZ vocabulary\r\n│   │   │   ├── gmem-load-store.md     # Avoid GMEM loads / reduce GMEM stores\r\n│   │   │   ├── efficient-msaa.md      # On-tile MSAA resolve\r\n│   │   │   ├── variable-rate-shading.md # QCOM_shading_rate (VRS)\r\n│   │   │   ├── lrz-and-flexrender.md  # LRZ, FlexRender, depth\r\n│   │   │   └── frame-extrapolation-and-upscaling.md # AFME, SGSR2\r\n│   │   └── powervr/                   # Imagination PowerVR techniques, one topic per file\r\n│   │       ├── README.md              # PowerVR module index & HSR/ISP vocabulary\r\n│"},{"path":"references/cards/README.md","content":"# GLES Rendering Expert — Knowledge Cards Index\r\n\r\n> 按 OpenGL ES 功能点拆分的知识卡片系统。每张卡片聚焦一个独立功能域，\r\n> 包含：核心规则、代码模式、常见陷阱、关联卡片。\r\n>\r\n> 数据来源：`references/rules/` 目录下的完整规则文档（本卡片为精炼摘要，详细上下文请查阅原始规则文件）。\r\n\r\n## 卡片目录\r\n\r\n| # | Card | 功能域 | GLES 版本 | 来源规则 |\r\n|:--|:-----|:-------|:----------|:---------|\r\n| 01 | [api-version-constraints](01-api-version-constraints.md) | API 版本约束 & 桌面 GL 禁用 | 3.0/3.1/3.2 | `gles-api-standards.md` |\r\n| 02 | [texture-formats-compression](02-texture-formats-compression.md) | 纹理格式 & ASTC/ETC2 压缩 | 3.0+ | `gles-api-standards.md` §3, `mali-arm-best-practices.md` §4 |\r\n| 03 | [buffer-objects](03-buffer-objects.md) | VAO/VBO/UBO/SSBO/PBO | 3.0/3.1 | `gles-api-standards.md` §4, `glsl-es-optimization.md` §3,§5.3 |\r\n| 04 | [framebuffer-objects](04-framebuffer-objects.md) | FBO 生命周期 & MRT & Blit | 3.0+ | `gles-api-standards.md` §5, `tbdr-bandwidth-rules.md` §2 |\r\n| 05 | [shader-precision-layout](05-shader-precision-layout.md) | GLSL ES 精度 & I/O 布局 | 3.00/3.20 | `glsl-es-optimization.md` §1,§2,§7,§8 |\r\n| 06 | [compute-shader](06-compute-shader.md) | 计算着色器 & 同步 | 3.1+ | `glsl-es-optimization.md` §5, `mali-arm-best-practices.md` §3 |\r\n| 07 | [egl-context-lifecycle](07-egl-context-lifecycle.md) | EGL 初始化/销毁/多线程/Context Lost | EGL 1.4+ | `egl-and-context.md` |\r\n| 08 | [tbdr-bandwidth](08-tbdr-bandwidth.md) | TBDR 架构 & 带宽优化 | All | `tbdr-bandwidth-rules.md` §1,§3 |\r\n| 09 | [overdraw-fillrate](09-overdraw-fillrate.md) | Overdraw & Fill-Rate 优化 | All | `tbdr-bandwidth-rules.md` §5, `mali-arm-best-practices.md` §8 |\r\n| 10 | [msaa-antialiasing](10-msaa-antialiasing.md) | MSAA on TBDR (Mali/Adreno) | 3.0+ | `tbdr-bandwidth-rules.md` §5.3, `mali-arm-best-practices.md` §5, `adreno/efficient-msaa.md` |\r\n| 11 | [synchronization](11-synchronization.md) | Fence/Memory Barrier/Buffer Orphaning | 3.0/3.1 | `gles-api-standards.md` §6, `glsl-es-optimization.md` §5.5-5.6 |\r\n| 12 | [draw-call-optimization](12-draw-call-optimization.md) | Draw Call 批处理 & 实例化 & Indirect | 3.0/3.1 | `mali-arm-best-practices.md` §7, `gles-api-standards.md` §4 |\r\n| 13 | [mali-pls-multiview](13-mali-pls-multiview.md) | Mali PLS & Multiview/Foveated | 3.0+ ext | `mali-arm-best-practices.md` §1,§2 |\r\n| 14 | [adreno-gmem-vrs-lrz](14-adreno-gmem-vrs-lrz.md) | Adreno GMEM/VRS/LRZ/FlexRender | 3.0+ ext | `adreno/*.md` |\r\n| 15 | [windows-egl-angle](15-windows-egl-angle.md) | Windows 平台 EGL/ANGLE/Windows-on-ARM | 3.0/3.1 via ANGLE | `windows-platform.md` |\r\n| 16 | [powervr-hsr-img-extensions](16-powervr-hsr-img-extensions.md) | PowerVR HSR/PLS/IMG 扩展/Tile 带宽 | 3.0+ ext | `powervr/*.md` |\r\n\r\n## 卡片格式说明\r\n\r\n每张卡片遵循统一结构：\r\n\r\n```\r\n# [标题]\r\n> Category | GLES Version | Source\r\n\r\n## 核心规则        ← 必须遵守的硬性规则（生成代码时强制执行）\r\n## 代码模式        ← 正确用法的典型代码片段\r\n## 常见陷阱        ← 高频错误 & 其后果\r\n## 关联卡片        ← 交叉引用\r\n```\r\n\r\n## 使用方式\r\n\r\n- **代码生成时**：根据涉及的功能域加载对应卡片的核心规则作为约束。\r\n- **代码审查时**：对照卡片的\"常见陷阱\"逐条检查。\r\n- **性能诊断时**：从 `08-tbdr-bandwidth` 和 `09-overdraw-fillrate` 入手定位瓶颈。"},{"path":"references/rules/adreno/README.md","content":"# Qualcomm Adreno GPU Best Practices (Distilled)\r\n\r\n> **Source of truth:** Snapdragon Game Studios / Qualcomm\r\n> [*Adreno GPU OpenGL ES Code Sample Framework*](https://github.com/SnapdragonGameStudios/adreno-gpu-opengl-es-code-sample-framework)\r\n> and the Qualcomm *Adreno GPU on Mobile: Best Practices* documentation.\r\n>\r\n> This module distills the vendor-recommended Adreno techniques into small,\r\n> focused rule files (one topic per file) so they stay easy to consume during\r\n> code generation and review. Everything here targets **Qualcomm Adreno** GPUs,\r\n> which use a **tiled / binning** architecture; most rules also help other\r\n> tile-based mobile GPUs (ARM Mali, Imagination PowerVR).\r\n\r\n## Vocabulary: Adreno vs. the generic TBDR terms\r\n\r\n| Adreno term | Meaning | Generic equivalent |\r\n|:---|:---|:---|\r\n| **GMEM** (Graphics Memory) | Fast on-chip tile memory | Tile memory / on-chip framebuffer |\r\n| **GMEM Load** | Copy a tile from system memory *into* GMEM at render-pass start | Tile \"load\" / unresolve |\r\n| **GMEM Store** | Write a GMEM tile *back* to system memory at render-pass end | Tile \"store\" / resolve |\r\n| **Binning / FlexRender** | Pass that sorts primitives into tile bins | Tiling / deferred binning |\r\n| **LRZ** (Low Resolution Z) | Early coarse depth rejection of hidden fragments | Hidden-surface removal / early-Z |\r\n\r\n## Rule files in this module\r\n\r\n| File | Topic | Maps to sample |\r\n|:---|:---|:---|\r\n| [`gmem-load-store.md`](gmem-load-store.md) | Avoid GMEM loads, reduce GMEM stores | `avoid_gmem_loads`, `reduce_gmem_stores` |\r\n| [`efficient-msaa.md`](efficient-msaa.md) | On-tile MSAA resolve without a blit | `msaa` |\r\n| [`variable-rate-shading.md`](variable-rate-shading.md) | `QCOM_shading_rate` (VRS) | `shading_rate` |\r\n| [`lrz-and-flexrender.md`](lrz-and-flexrender.md) | LRZ, FlexRender/binning, depth choices | `hello_gltf` scenes, general arch |\r\n| [`frame-extrapolation-and-upscaling.md`](frame-extrapolation-and-upscaling.md) | AFME, motion estimation, SGSR2 | `amfe_power_saving`, `motion_estimation`, `sgsr2` |\r\n\r\n## Golden rules (one-line summary)\r\n\r\n1. **Clear or invalidate every attachment at render-pass start** → kills GMEM loads.\r\n2. **Invalidate transient attachments (depth/stencil/MSAA) at render-pass end** → kills GMEM stores.\r\n3. **Resolve MSAA on-tile** via `EXT_multisampled_render_to_texture`, never a manual blit.\r\n4. **Never break LRZ**: draw opaque front-to-back, avoid `discard` / fragment depth writes where possible.\r\n5. **Spend fewer fragment invocations**: use `QCOM_shading_rate` on low-detail draws and temporal upscaling (SGSR2) instead of shading every pixel every frame."},{"path":"references/rules/powervr/README.md","content":"# Imagination PowerVR GPU Best Practices (Distilled)\r\n\r\n> **Source of truth:** Imagination Technologies\r\n> [*PowerVR Native SDK — OpenGL ES Framework*](https://github.com/powervr-graphics/Native_SDK/tree/master/framework/PVRUtils/OpenGLES)\r\n> and the official *PowerVR Performance Recommendations* / *Introduction to PowerVR for Developers* documentation.\r\n>\r\n> This module distills PowerVR-specific GLES techniques into focused rule files.\r\n> PowerVR uses a **Tile-Based Deferred Rendering (TBDR)** architecture with **full\r\n> Hidden Surface Removal (HSR)** — a unique hardware pass that eliminates ALL\r\n> invisible fragments before shading. This fundamentally changes rendering strategy\r\n> compared to both desktop GPUs and even other mobile TBDR GPUs (Mali, Adreno).\r\n\r\n## Vocabulary: PowerVR-specific terms\r\n\r\n| PowerVR term | Meaning | Generic equivalent |\r\n|:---|:---|:---|\r\n| **HSR** (Hidden Surface Removal) | Hardware pass that processes ALL primitives in a tile, determines visibility, then shades ONLY visible fragments | Early-Z / LRZ (partial equivalent — HSR is more thorough) |\r\n| **ISP** (Image Synthesis Processor) | Fixed-function unit performing HSR + depth/stencil tests before fragment shading | Rasterizer + early-Z |\r\n| **USC** (Unified Shading Cluster) | Shader processor units | Shader cores / ALUs |\r\n| **Parameter Buffer (PB)** | Off-chip memory storing transformed geometry for tiling | Bin buffer / tiling buffer |\r\n| **SPM** (Smart Parameter Management) | Hardware manages PB overflow by partial renders | Binning overflow handling |\r\n| **Tile Memory** | Fast on-chip framebuffer per tile (~32×32 pixels) | GMEM (Adreno) / Tile buffer (Mali) |\r\n| **PVRTC** | PowerVR Texture Compression (2bpp / 4bpp); legacy, prefer ASTC/ETC2 on modern HW | — |\r\n| **PVRScope** | PowerVR's GPU profiling tool | Streamline (Mali) / Snapdragon Profiler (Adreno) |\r\n\r\n## Rule files in this module\r\n\r\n| File | Topic | Maps to SDK example/doc |\r\n|:---|:---|:---|\r\n| [`hsr-and-rendering-order.md`](hsr-and-rendering-order.md) | HSR, no depth pre-pass, alpha test vs blend, draw order | Performance Recommendations §HSR, forum guidance |\r\n| [`pixel-local-storage-and-deferred.md`](pixel-local-storage-and-deferred.md) | PLS for on-chip deferred rendering | `DeferredShading` example |\r\n| [`img-extensions.md`](img-extensions.md) | IMG_framebuffer_downsample, IMG_texture_filter_cubic, binary shaders | `IMGFramebufferDownsample`, `IMGTextureFilterCubic`, `BinaryShaders` |\r\n| [`bandwidth-and-tile-management.md`](bandwidth-and-tile-management.md) | Clear/invalidate, transient stores, parameter buffer, MSAA | `PostProcessing`, Performance Recommendations |\r\n\r\n## Golden rules (one-line summary)\r\n\r\n1. **Do NOT use a depth pre-pass** — PowerVR HSR already eliminates 100% of hidden opaque fragments; a depth pre-pass doubles geometry cost with zero shading benefit.\r\n2. **Avoid `discard` / alpha test where possible** — it delays HSR, forcing the ISP to defer visibility decisions until after "}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":null,"editorialQuality":{"score":100,"threshold":65,"status":"thin","wordCount":1770,"uniquenessScore":44,"reasons":["uniqueness-below-45"]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:45:57.249Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T16:27:32.192Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}