{"id":"9ccf17e0-6ffb-4fac-bfe5-002ce1a8ba66","entityType":"agent","slug":"clawhub-zack-dev-cm-agentic-codex-dev","name":"Agentic Codex Dev Reviewer","canonicalUrl":"https://www.xpersona.co/agent/clawhub-zack-dev-cm-agentic-codex-dev","canonicalPath":"/agent/clawhub-zack-dev-cm-agentic-codex-dev","generatedAt":"2026-10-10T21:39:01.921Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":null},"description":"Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple... Skill: Agentic Codex Dev Reviewer Owner: zack-dev-cm Summary: Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple... Tags: agentic-development:0.3.5, clawhub:0.3.6, codex:0.3.6, github:0.3.6, latest:0.3.6, multi-agent:0.3.4, review:0.3.6 Version history: v0.3.6 | 2026-05-15T15:35:08.460Z | user Republish public p","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 1.3K downloads reported by the source. Last updated 10/10/2026.","installCommand":"clawhub skill install s170hyv1nagjajq3y5c6kpzx3s84jp29:agentic-codex-dev","sourceUrl":"https://clawhub.ai/zack-dev-cm/agentic-codex-dev","homepage":"https://clawhub.ai/zack-dev-cm/skills/agentic-codex-dev","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/zack-dev-cm/agentic-codex-dev","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/zack-dev-cm/skills/agentic-codex-dev","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":63,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":null},"stars":null,"forks":null,"downloads":1333,"packageName":null,"latestVersion":"0.3.6","tractionLabel":"1.3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":null},"lastUpdatedAt":"2026-10-10T16:47:13.453Z","lastCrawledAt":"2026-10-10T16:47:13.453Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-11T16:47:13.453Z","lastVerifiedAt":null,"highlights":[{"version":"0.3.6","createdAt":"2026-05-15T15:35:08.460Z","changelog":"Republish public package as instruction-only delivery readiness review.","fileCount":4,"zipByteSize":2632},{"version":"0.3.5","createdAt":"2026-05-15T08:09:47.556Z","changelog":"Reword public-surface gate language so strict install-gate checks return PASS.","fileCount":8,"zipByteSize":17657},{"version":"0.3.4","createdAt":"2026-05-14T07:35:08.991Z","changelog":"Add Skill Import Project Mode for adapting upstream skills with staging, audit, and active-project impact gates.","fileCount":8,"zipByteSize":17651},{"version":"0.3.3","createdAt":"2026-04-30T02:35:43.636Z","changelog":"Reviewed source and refreshed public ClawHub skill surface.","fileCount":8,"zipByteSize":16636},{"version":"0.3.2","createdAt":"2026-04-25T17:47:19.125Z","changelog":"Add the ClawHub skill key and clarify runtime requirement metadata for Git, ClawHub CLI, and Python alternatives.","fileCount":8,"zipByteSize":16616},{"version":"0.3.1","createdAt":"2026-04-22T08:22:38.320Z","changelog":"Disable model auto-invocation for clean public release","fileCount":8,"zipByteSize":16546},{"version":"0.3.0","createdAt":"2026-04-22T08:18:23.431Z","changelog":"Add repo-owned real-run evidence and deployment ledger","fileCount":8,"zipByteSize":16494},{"version":"0.2.2","createdAt":"2026-04-22T08:09:20.988Z","changelog":"Require explicit invocation for public multi-agent workflow","fileCount":8,"zipByteSize":16471}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s170hyv1nagjajq3y5c6kpzx3s84jp29:agentic-codex-dev","setupComplexity":"low","setupSteps":["Setup complexity is classified as HIGH. You must provision dedicated cloud infrastructure or an isolated VM. Do not run this directly on your local workstation.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T21:39:01.917Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-zack-dev-cm-agentic-codex-dev/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":null},"readme":"Skill: Agentic Codex Dev Reviewer\n\nOwner: zack-dev-cm\n\nSummary: Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple...\n\nTags: agentic-development:0.3.5, clawhub:0.3.6, codex:0.3.6, github:0.3.6, latest:0.3.6, multi-agent:0.3.4, review:0.3.6\n\nVersion history:\n\nv0.3.6 | 2026-05-15T15:35:08.460Z | user\n\nRepublish public package as instruction-only delivery readiness review.\n\nv0.3.5 | 2026-05-15T08:09:47.556Z | user\n\nReword public-surface gate language so strict install-gate checks return PASS.\n\nv0.3.4 | 2026-05-14T07:35:08.991Z | user\n\nAdd Skill Import Project Mode for adapting upstream skills with staging, audit, and active-project impact gates.\n\nv0.3.3 | 2026-04-30T02:35:43.636Z | user\n\nReviewed source and refreshed public ClawHub skill surface.\n\nv0.3.2 | 2026-04-25T17:47:19.125Z | user\n\nAdd the ClawHub skill key and clarify runtime requirement metadata for Git, ClawHub CLI, and Python alternatives.\n\nv0.3.1 | 2026-04-22T08:22:38.320Z | user\n\nDisable model auto-invocation for clean public release\n\nv0.3.0 | 2026-04-22T08:18:23.431Z | user\n\nAdd repo-owned real-run evidence and deployment ledger\n\nv0.2.2 | 2026-04-22T08:09:20.988Z | user\n\nRequire explicit invocation for public multi-agent workflow\n\nv0.2.1 | 2026-04-22T08:06:25.992Z | user\n\nDeclare runtime requirements and credential boundary for multi-agent release\n\nv0.2.0 | 2026-04-22T07:54:14.209Z | user\n\nAdd multi-agent system design, model policy, ledgers, evals, and anti-bleed tests\n\nv0.1.2 | 2026-04-22T07:07:03.361Z | user\n\nMove source to dedicated repository\n\nv0.1.1 | 2026-04-22T06:26:09.398Z | user\n\nFix publish command reference\n\nv0.1.0 | 2026-04-22T06:20:03.001Z | user\n\nInitial public release\n\nArchive index:\n\nArchive v0.3.6: 4 files, 2632 bytes\n\nFiles: agents/openai.yaml (318b), skill-card.md (1778b), SKILL.md (1851b), _meta.json (136b)\n\nFile v0.3.6:SKILL.md\n\n---\nname: agentic-codex-dev\ndescription: Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, implementation review, public-surface checks, or release evidence review without running remote-changing commands.\n---\n\n# Agentic Codex Dev\n\nUse this skill as a text-only review layer for agentic development work. It helps turn a request, diff, repository note, or release checklist into a scoped plan and a conservative readiness verdict.\n\n## Review Workflow\n\n1. Restate the requested outcome and the smallest safe scope.\n2. Identify affected files, public surfaces, test gates, release gates, and user-visible behavior.\n3. Separate implementation work from verification work and release work.\n4. Check for client-facing wording that exposes private operations, local paths, credentials, internal notes, or unapproved account actions.\n5. Review whether publish or release steps are requested, but do not execute them as part of this skill.\n6. Return a clear verdict: `ready`, `ready_with_notes`, `blocked`, or `do_not_ship`.\n\n## Boundaries\n\n- Do not request, print, store, or infer credentials.\n- Do not stage, commit, push, publish, delete, hide, or transfer anything.\n- Do not assume logged-in GitHub, ClawHub, browser, or other account authority.\n- Do not create persistent project memory, ledgers, or reports unless the user separately asks for a file artifact.\n- Keep public-surface advice focused on wording, scope, tests, and release evidence.\n\n## Output Shape\n\nReturn:\n\n- `Scope`: what is being reviewed.\n- `Findings`: concrete issues ordered by severity.\n- `Public surface`: wording or packaging risks.\n- `Verification`: tests or checks that should pass before release.\n- `Verdict`: one of `ready`, `ready_with_notes`, `blocked`, or `do_not_ship`.\n\nFile v0.3.6:_meta.json\n\n{\n  \"ownerId\": \"kn7dhjt1k1f111whp13fmrqwnh81tn1v\",\n  \"slug\": \"agentic-codex-dev\",\n  \"version\": \"0.3.6\",\n  \"publishedAt\": 1778859308460\n}\n\nFile v0.3.6:skill-card.md\n\n## Description:\n\nReview agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, implementation review, public-surface checks, or release evidence review without running remote-changing commands.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zack-dev-cm](https://clawhub.ai/user/zack-dev-cm)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and release reviewers use this skill to review Codex, GitHub, and ClawHub delivery plans, diffs, release checklists, public-facing wording, verification evidence, and readiness before publication or remote-changing work.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Readiness verdicts may be mistaken for authorization to publish, push, or run account-changing commands.\n\nMitigation: Treat verdicts as advisory and require normal human approval, test gates, and release controls before remote-changing actions.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zack-dev-cm/skills/agentic-codex-dev)\n\n## Skill Output:\n\n**Output Type(s):** [Analysis, Guidance, Markdown]\n\n**Output Format:** [Markdown sections with findings, public-surface review, verification checks, and a readiness verdict]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Verdict is one of ready, ready_with_notes, blocked, or do_not_ship.]\n\n## Skill Version(s):\n\n0.3.6 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment.\n\nFile v0.3.6:agents/openai.yaml\n\ninterface:\n  display_name: \"Agentic Codex Dev Reviewer\"\n  short_description: \"Review scoped Codex/GitHub/ClawHub delivery readiness.\"\n  default_prompt: \"Use $agentic-codex-dev to review this agentic development plan, diff, or release checklist for scope, public surface, verification evidence, and release readiness.\"\n\nArchive v0.3.5: 8 files, 17657 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3728b), references/publish-checklist.md (2089b), references/source-review.md (7182b), references/system-design.md (3616b), SKILL.md (16515b), _meta.json (136b)\n\nFile v0.3.5:SKILL.md\n\n---\nname: agentic-codex-dev\ndescription: Use when planning, implementing, reviewing, coordinating, or publishing agentic software development work with Codex, GitHub, and OpenClaw/ClawHub. Provides a production-grade multi-agent operating loop with role roster, model policy, task ledger, memory ledger, report artifacts, verification gates, and anti-bleed public-surface review.\nversion: 0.3.5\nuser-invocable: true\ndisable-model-invocation: true\nmetadata: {\"openclaw\":{\"homepage\":\"https://github.com/zack-dev-cm/agentic-codex-dev-skill\",\"skillKey\":\"agentic-codex-dev\",\"requires\":{\"bins\":[\"git\",\"clawhub\"],\"anyBins\":[\"python3\",\"python\"]},\"install\":[{\"kind\":\"node\",\"label\":\"Install ClawHub CLI\",\"package\":\"clawhub\",\"bins\":[\"clawhub\"]}],\"tags\":[\"codex\",\"github\",\"clawhub\",\"agentic-development\"]}}\n---\n\n# Agentic Codex Dev\n\nOperate Codex like a disciplined software team: clear goal, explicit roles, scoped ownership, evidence, tests, review, report.\n\n## When to Use\n\nUse this skill for:\n\n- coding tasks where Codex should inspect, modify, test, and report on a GitHub repo\n- turning a rough product or bug request into scoped implementation work\n- setting up repo-local `AGENTS.md`, `.codex/agents/`, or skill instructions\n- reviewing agent-generated code for correctness, tests, security, and public-surface leaks\n- preparing a GitHub repo or ClawHub skill for open-source publication\n- coordinating explicit parallel/subagent work with role ownership and integration control\n\nDo not use it for one-line answers, pure brainstorming, or tasks that only need a command output.\n\n## Runtime Requirements\n\nClawHub requirement metadata for this skill declares `git`, `python3`, and `clawhub`, following the ClawHub skill metadata format at <https://github.com/openclaw/clawhub/blob/main/docs/skill-format.md>.\n\n- Local plan, review, and implementation modes may work with the tools already available in the host.\n- Verification and publish modes expect the declared binaries plus optional Python modules such as `antirot` and `codex_harness`.\n- This skill should not request, print, or store credentials. GitHub and ClawHub publishing must use existing local authenticated CLI sessions, or the user must authenticate manually outside the prompt.\n- Do not run `git push`, `clawhub publish`, or other remote-changing commands unless the user asked for publish or remote update work.\n\n## Core Loop\n\n1. Restate the goal and name the verification step before editing.\n2. Read the repo map: `AGENTS.md`, README, package config, tests, and the files closest to the task.\n3. Define concrete success criteria that would let a reviewer say \"done\".\n4. Make the narrowest defensible change. Match local style. Avoid speculative abstractions.\n5. Run the highest-signal local check. Add a focused smoke test when behavior changed.\n6. Review the diff for bugs, regressions, secrets, private paths, and public-surface bleed.\n7. Report what changed, how it was verified, and any residual risk.\n\nIf the task is unclear, stop early and name the ambiguity. Prefer one precise question over guessing.\n\n## Operating Rules\n\n- Treat repository files as the source of truth. If knowledge matters later, put it in repo docs.\n- Keep `AGENTS.md` short. Use it as an index to durable docs, not a giant prompt.\n- Prefer boring, inspectable code over opaque magic. Agents compound what they can read.\n- Touch only files required for the goal. Mention unrelated problems; do not fix them unless asked.\n- Use structured APIs, tests, and parsers where available. Avoid fragile string tricks.\n- Convert repeated review feedback into checks, docs, or templates.\n- Keep logs and long command output out of the main narrative; summarize the signal.\n- Avoid asking an agent to read undeclared secret files or sync credentials as part of a skill.\n\n## Scope Modes\n\nPick the mode that fits the risk:\n\n- **Patch**: one bug or one focused feature. Read close code, edit, test, review.\n- **Plan**: ambiguous or multi-file work. Write a short acceptance plan before editing.\n- **Review**: findings first, with emphasis on correctness, regressions, security, tests, and leaks as summarized in [source review](references/source-review.md).\n- **Harness**: improve repo legibility: docs, CI, local scripts, custom agents, or audit gates.\n- **Evolve**: metric-driven optimization. One variable per experiment, fixed budget, log keep/discard.\n- **Publish**: GitHub/ClawHub release readiness, metadata, license, docs, and verification.\n- **Skill Import**: adapt upstream skills or agent workflows into local ClawHub/Codex/Claude artifacts with staging, audit, and active-project impact gates before install.\n- **Multi-Agent**: explicit role roster, task ledger, isolation plan, review gates, memory update, and final report.\n\nPrefer Patch unless the task shows it needs more structure. Use Multi-Agent only when the user explicitly asks for subagents, delegation, or parallel agent work.\n\n## Skill Import Project Mode\n\nUse this mode when adapting upstream skill repositories, `SKILL.md` files, Claude commands, agent personas, or workflow packs into local Codex, Claude Code, ClawHub, or OpenClaw artifacts.\n\nDefault stance: import ideas, not trust. Do not install, enable, publish, or run upstream skill code until provenance, public-surface safety, and active-project impact are understood.\n\nWorkflow:\n\n1. Inventory the source:\n   - source path or URL,\n   - commit or release tag when available,\n   - skill names,\n   - commands,\n   - personas,\n   - hooks,\n   - scripts,\n   - references,\n   - licenses.\n2. Classify each candidate:\n   - `PORT`: narrow, safe, runtime-neutral enough to adapt directly,\n   - `REWRITE`: useful workflow pattern but runtime mechanics, trigger breadth, scripts, privacy boundaries, or proof criteria must change,\n   - `REJECT`: prompt override, credential access, hidden install, destructive behavior, unreviewable executable content, scraping, auto-posting, or claims risk.\n3. Stage in an isolated repo-local workspace. Do not install into global Codex, Claude, or OpenClaw paths during evaluation.\n4. Assign owners in the task ledger:\n   - explorer for source inventory,\n   - architect for runtime boundary decisions,\n   - implementer for rewritten skill text,\n   - reviewer for public-surface and install-risk review,\n   - memory curator for stable decisions and rejected patterns.\n5. Convert to a portable core plus adapters:\n   - canonical `SKILL.md` core,\n   - Codex notes for custom agents, sandbox, subagent boundaries, and verification,\n   - Claude Code notes for skills, commands, agents, and hook cautions,\n   - ClawHub/OpenClaw notes for metadata, required binaries, install destination, and public registry risk.\n6. Run gates before install or publish:\n   - static skill safety scan,\n   - duplicate-name and precedence check,\n   - public-surface review for private paths, local URLs, credential-shaped strings, copied paid/community content, and stale claims,\n   - focused fixture or dry-run prompt,\n   - report entry naming accepted, rewritten, and rejected candidates.\n7. Install or publish only after explicit user approval for the exact artifact, destination, and command.\n\nStop conditions:\n\n- source provenance is unclear,\n- candidate requires secrets or private exports,\n- destination would shadow an active skill without explicit replace approval,\n- hooks or scripts mutate workspaces outside declared ownership,\n- no credible verification path exists.\n\n## System Design\n\nFor non-trivial or multi-agent work, set up a control plane before coding:\n\n- **Orchestrator**: the main Codex thread owns requirements, task split, agent selection, integration, final review, and user communication.\n- **Role agents**: subagents are optional workers with declared purpose, model, reasoning effort, sandbox, file ownership, and output schema.\n- **Artifacts**: use repo-local ledgers so work survives context loss and can be reviewed without private chat history.\n- **Isolation**: prefer branches or worktrees per writer when multiple agents edit. If one checkout is shared, assign disjoint file ownership.\n- **Gates**: no task is done until its acceptance criteria, verification command, diff review, public-surface scan, and report entry are complete.\n\nWhen this structure is overkill, keep a solo Patch flow and still preserve the same verification discipline.\n\n## Task, Memory, and Report Ledgers\n\nCreate or update these artifacts when work is multi-agent, multi-turn, risky, or intended for publication:\n\n- `docs/agentic/tasks.md`: task id, owner role, goal, owned files, status, acceptance criteria, verification, result, blocker.\n- `docs/agentic/memory.md`: stable repo facts, architecture decisions, commands that actually work, hazards, rejected approaches, last-verified date. Do not store secrets, tokens, private paths, or raw logs.\n- `docs/agentic/reports/<date>-<slug>.md`: final objective, source links, task outcomes, changed files, tests, review findings, unresolved risks, release or PR status.\n\nIf the target repo already has equivalent docs, use the local convention instead of inventing new paths.\n\n## Role Roster\n\nUse this roster as the default multi-agent team. The parent thread stays responsible for coordination and final judgment.\n\n| Role | Default model | Reasoning | Scope | Required output |\n| --- | --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` for critical design/release, `high` otherwise | Owns task split, integration, report | plan, assignments, final decision |\n| Analyst | `gpt-5.4` | `high` | Turns vague request into requirements and risks | assumptions, open questions, acceptance criteria |\n| Architect | `gpt-5.4` | `xhigh` | System design, boundaries, dependency choices | design note, rejected options, invariants |\n| Planner | `gpt-5.4` | `high` | Breaks design into ordered tasks | task ledger rows with owners and gates |\n| Explorer | `gpt-5.4-mini` or `gpt-5.3-codex-spark` | `medium` | Read-only code mapping and evidence gathering | files, symbols, execution path, uncertainty |\n| Implementer | `gpt-5.4` for risky code, `gpt-5.3-codex-spark` for bounded edits | `high` or `medium` | Writes only owned files | patch summary, tests, residual risks |\n| Reviewer | `gpt-5.4` | `xhigh` | Correctness, security, regressions, tests, public surface | findings first, file/line evidence, verdict |\n| QA/CI Analyst | `gpt-5.4` | `high` | Reproduction, failing checks, browser or CLI evidence | exact command, observed failure, fix owner |\n| Memory Curator | `gpt-5.4-mini` | `medium` | Updates durable docs after decisions land | memory entries, stale entries removed |\n\n## Subagents\n\nOnly use subagents when the user explicitly asks for subagents, delegation, or parallel agent work.\n\nGood delegation targets:\n\n- read-heavy codebase mapping\n- independent test or CI-log analysis\n- independent review categories such as security, test gaps, or docs correctness\n- disjoint implementation slices with clearly separate file ownership\n\nBad delegation targets:\n\n- the immediate blocker for your next local step\n- tightly coupled edits in the same files\n- vague \"go improve the code\" work\n- recursive fan-out with no cap\n\nWhen delegating, give each agent a bounded task, a clear output shape, and explicit ownership. Keep the main thread focused on requirements, decisions, integration, and final review. Keep `agents.max_depth = 1` unless the user explicitly accepts recursive delegation risk; this matches the Codex subagent configuration surface documented at <https://developers.openai.com/codex/subagents>.\n\nDelegation prompt shape:\n\n```text\nRole: reviewer\nModel: gpt-5.4\nReasoning: xhigh\nOwnership: read-only review of <files or branch>\nTask: find correctness, security, regression, test, and public-surface risks.\nOutput: findings first with file/line evidence, then open questions, then verdict.\nDo not edit files. Do not inspect secrets. Do not broaden scope.\n```\n\n## Model Policy\n\n- Use `gpt-5.4` with `xhigh` reasoning for architecture, security review, release decisions, and ambiguous multi-agent coordination; Codex custom-agent examples document `gpt-5.4` reviewer roles at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4` with `high` reasoning for implementation where correctness or cross-module behavior matters; model selection follows the Codex custom-agent configuration surface at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4-mini` or `gpt-5.3-codex-spark` for read-only exploration, docs checks, and bounded cleanup where speed matters and the output will be reviewed; both model families appear in Codex custom-agent examples at <https://developers.openai.com/codex/subagents>.\n- Do not use a budget model for final architecture, security, or publish verdicts.\n- Use extra compute selectively: best-of-N, independent reviewer passes, or verifier checks only when the decision is expensive to reverse; optillm documents inference-time scaling techniques at <https://github.com/algorithmicsuperintelligence/optillm>.\n\n## Implementation Discipline\n\nBefore editing:\n\n- inspect the existing patterns\n- identify the likely tests or smoke command\n- check dirty git state and avoid touching unrelated user changes\n- state the planned edit in one or two sentences\n\nWhile editing:\n\n- keep the diff surgical\n- add tests when behavior, contracts, or public output changes\n- avoid new dependencies unless they clearly reduce risk or complexity\n- keep comments rare and useful\n\nAfter editing:\n\n- run the named verification\n- inspect the diff, not just test output\n- update docs only when user-facing behavior or workflow changed\n- do not call work published until the public surface is clean\n\n## Review Checklist\n\nReview every non-trivial result for:\n\n- Does every changed line trace to the stated goal?\n- Are edge cases covered by tests or a clear smoke path?\n- Did the change preserve existing public APIs and CLI behavior?\n- Did docs/examples drift from actual behavior?\n- Did any secret-like string, local path, private URL, copied dashboard, or stale release note enter the repo?\n- Did the final diff remove avoidable complexity from the first draft, as recommended in [source review](references/source-review.md)?\n\n## Consistency and Effectiveness Gates\n\nFor multi-agent work, verify the process itself:\n\n- Every task has an owner, owned files, acceptance criteria, verification command, and result.\n- Every subagent output is mapped to a task or explicitly discarded with a reason.\n- No writer agent edits outside its assigned ownership without parent approval.\n- At least one reviewer pass is read-only and independent of the implementer.\n- The final report names changed files, commands run, failed checks, source links, residual risk, and release status.\n- Memory updates contain stable facts only; do not store raw chat, secrets, local credentials, or transient logs.\n- If a metric-driven change is attempted, record baseline, candidate, verifier, result, and keep/discard decision.\n\n## Real Example Eval\n\nFor a serious workflow eval, run this skill against a real repo task and archive the result in the report ledger. A valid eval has:\n\n- baseline repo state and user goal\n- role roster used, including model and reasoning choices\n- task ledger rows with owners and file boundaries\n- at least one implementation or review task with verification output\n- public-release check for private-path examples, local-only URLs, secret-shaped placeholders, and stale claims\n- final report with changed files, tests, residual risks, and follow-up blockers\n\nUse [example run](references/example-run.md) as the minimum acceptance shape.\n\n## GitHub and ClawHub Publish Gate\n\nBefore publishing:\n\n- README or skill summary says what it does, when to use it, and what it does not do.\n- License is compatible with the target surface. ClawHub publishes skills under MIT-0.\n- `SKILL.md` has frontmatter `name`, `description`, and `version`.\n- The skill folder contains only text-based files needed at runtime.\n- No hidden install scripts, credential readers, service restarts, or local machine assumptions.\n- Public repo has security, contribution, support, CI, and release/audit checks when applicable.\n- Run the repo's public-surface gate before pushing or publishing to a registry.\n\nFor this skill's source analysis, read `references/source-review.md` and `references/comparison-matrix.md`.\nFor multi-agent artifacts and templates, read `references/system-design.md`.\nFor release commands and manual checks, read `references/publish-checklist.md`.\n\nFile v0.3.5:_meta.json\n\n{\n  \"ownerId\": \"kn7dhjt1k1f111whp13fmrqwnh81tn1v\",\n  \"slug\": \"agentic-codex-dev\",\n  \"version\": \"0.3.5\",\n  \"publishedAt\": 1778832587556\n}\n\nFile v0.3.5:references/comparison-matrix.md\n\n# Comparison Matrix\n\nReviewed on 2026-04-22. Use this matrix when judging whether the skill is strong enough for multi-agent software development rather than a generic coding checklist.\n\n| Source | Strong pattern | Risk if copied blindly | Skill response |\n| --- | --- | --- | --- |\n| OpenAI Codex subagents | Custom roles with model, reasoning, sandbox, and developer instructions | More agents can add cost and coordination failure | Subagents require explicit user intent, role ownership, and parent integration |\n| OpenAI Codex concepts | Context isolation and parallel investigation | Delegation can hide the critical path | Parent keeps immediate blockers local and uses agents for side work |\n| OpenAI harness engineering | Repo-local memory, docs, tests, and legible state | Prose policy can drift without checks | Task, memory, report ledgers plus public-surface tests |\n| openai/symphony | Isolated implementation runs | Workflow engine complexity may exceed need | Use branches/worktrees for writers; do not require a daemon |\n| karpathy/autoresearch | Baseline, budget, one metric, keep/discard/crash log | Research framing may not fit product work | Apply evaluator-first discipline only when optimizing behavior |\n| forrestchang/andrej-karpathy-skills | Assumptions, tradeoffs, surgical diffs | Too little structure for multi-agent runs | Keep the concise style, add role roster and ledgers only for serious work |\n| openevolve | Reproducible evaluator-first evolution, Pareto tradeoffs, cascade checks | Claims can outrun evidence | Require baseline, verifier, result log, and keep/discard decision |\n| optillm | Inference-time scaling, verifier passes, multi-agent reasoning | Expensive compute can become default theater | Reserve `gpt-5.4`/`xhigh` and independent reviewers for hard decisions |\n| agent-orchestrator | Worktree isolation, CI/review feedback routed to owner | Autonomous fleets need heavy operations | Adopt ownership and feedback routing without requiring its platform |\n| gstack | Clear role taxonomy and dispatch tiers | Full role stack for every task slows direct fixes | Scope modes decide when to use Patch vs Multi-Agent |\n| paperclip | Goals, budgets, heartbeats, org chart, audit trail | Continuous autonomy can run away | Add owner, budget/risk thinking, stop conditions, and reports |\n| openclaw | Local-first assistant, skills, sandbox and channel safety | Broad host access is unnecessary for a methodology skill | Keep bundle instruction-only and public-surface clean |\n| rdudov/agents | Analyst, architect, planner, developer, reviewer boundaries | Rigid phase gates can create process drag | Use roles when the task is ambiguous, risky, or parallel |\n\n## Previous Version Weak Points\n\nVersion 0.1.2 was useful as a publish-clean patch loop, but it was not enough for the stated goal of multi-agent software development:\n\n- no explicit role roster with model and reasoning policy\n- no default use of `gpt-5.4`/`xhigh` for architecture, review, and release decisions\n- no system design for orchestration, isolation, ledgers, or reports\n- no task ledger, memory ledger, or report artifact\n- no real-run eval shape\n- no process consistency checks that tie agent outputs to tasks and verification\n- no test ratchet for role/model/system-design coverage\n- bleed scan omitted Python files\n\n## Target Bar\n\nA release is acceptable only when a reviewer can see:\n\n- the role roster and model policy in `SKILL.md`\n- source-by-source comparison in this file\n- system design templates in `references/system-design.md`\n- real-run acceptance shape in `references/example-run.md`\n- automated tests that fail if the public skill regresses to a generic patch loop\n- anti-bleed coverage that includes Python test and helper files\n\nFile v0.3.5:references/example-run.md\n\n# Example Run\n\nThis is the minimum shape for a real eval of the skill. It is intentionally concrete enough to review and general enough to run on any public repository.\n\n## Scenario\n\nGoal: upgrade an instruction-only Codex skill from a generic patch workflow into a multi-agent software-development operating loop.\n\nAcceptance:\n\n- `SKILL.md` includes system design, role roster, model policy, task ledger, memory ledger, report ledger, consistency gates, and real-run eval guidance.\n- Source comparison covers all cited public projects.\n- Custom agent configs declare models and reasoning effort.\n- Public-surface tests fail if role/model/system-design coverage is removed.\n- Anti-bleed scan includes Python files.\n- Local verification passes, with any CI or publish blocker named explicitly.\n\n## Role Roster Used\n\n| Role | Model | Reasoning | Purpose |\n| --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` | own requirements, source synthesis, implementation, final report |\n| Architect | `gpt-5.4` | `xhigh` | validate system design and role boundaries |\n| Implementer | `gpt-5.4` | `high` | edit skill, references, docs, tests, metadata |\n| Reviewer | `gpt-5.4` | `xhigh` | check correctness, public surface, test coverage, publish risk |\n| Memory Curator | `gpt-5.4-mini` | `medium` | update durable docs after verification |\n\n## Task Ledger Sample\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | `SKILL.md`, `references/system-design.md` | done | system design and role roster are explicit | public-surface tests | pending until local run | none |\n| T2 | implementer | `tests/test_public_surface.py` | done | Python files are scanned for bleed | unit tests | pending until local run | none |\n| T3 | reviewer | public repo surface | review | no private paths, tokens, local URLs, stale version claims | anti-bleed and AntiRot | pending until local run | CI workflow depends on token scope |\n\n## Memory Ledger Sample\n\nStable facts:\n\n- Runtime skill files are text-only and publish through ClawHub.\n- `.clawhubignore` excludes repo harness files from the public skill bundle.\n- GitHub Actions workflow creation may require token scope outside the normal repo push permission.\n\nDecisions:\n\n- Keep the skill instruction-only. Add templates and tests rather than a daemon.\n- Use `gpt-5.4`/`xhigh` for architecture and reviewer roles; use faster models only for reviewed, read-only work.\n\nHazards:\n\n- Do not claim CI enforcement unless the workflow exists in the remote repository.\n- Do not let public examples include private paths, local URLs, tokens, or copied chat logs.\n\n## Report Sample\n\nObjective: release version 0.3.1 with multi-agent operating structure, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned real-run evidence.\n\nSources: OpenAI Codex subagents, OpenAI harness engineering, optillm, openevolve, autoresearch, symphony, paperclip, gstack, OpenClaw, Andrej Karpathy skills, agent-orchestrator, rdudov agents.\n\nChanged files:\n\n- `SKILL.md`\n- `references/source-review.md`\n- `references/comparison-matrix.md`\n- `references/system-design.md`\n- `references/example-run.md`\n- `.codex/agents/*.toml`\n- `tests/test_public_surface.py`\n\nVerification:\n\n- `python3 -m unittest discover -s tests`\n- `python3 -m antirot.cli lint SKILL.md --strict`\n- `python3 -m codex_harness audit . --strict --min-score 90`\n- `clawhub inspect agentic-codex-dev --files` after publish\n\nResidual risks:\n\n- CI cannot be considered enforced until the remote repository has a workflow.\n- Any future source review must re-check the public URLs because repository behavior can change.\n\nFile v0.3.5:references/publish-checklist.md\n\n# Publish Checklist\n\nUse this when preparing the skill for GitHub and ClawHub.\n\n## Local Review\n\n1. Confirm `SKILL.md` frontmatter:\n   - `name: agentic-codex-dev`\n   - `description: ...`\n   - `version: 0.3.1`\n2. Confirm the folder name is the intended ClawHub slug: `agentic-codex-dev`.\n3. Confirm every file is text-based and needed:\n   - `SKILL.md`\n   - `agents/openai.yaml`\n   - `references/source-review.md`\n   - `references/comparison-matrix.md`\n   - `references/system-design.md`\n   - `references/example-run.md`\n   - `references/publish-checklist.md`\n4. Search for private paths, local URLs, tokens, and copied private notes.\n5. Run the repository gate:\n\n```bash\npython3 -m unittest discover -s tests\npython3 -m antirot.cli lint SKILL.md --strict\npython3 -m codex_harness audit . --strict --min-score 90\n```\n\n## GitHub Publish\n\nFrom the repository root:\n\n```bash\ngit status --short\ngit add .\ngit commit -m \"Upgrade agentic Codex development skill\"\ngit push\n```\n\nIf the worktree contains unrelated user changes, stage only the files above.\n\nDo not claim GitHub CI enforcement unless `.github/workflows/ci.yml` exists on the remote default branch. If push tokens cannot create workflows, record that as a release blocker or create the workflow through the GitHub UI before claiming enforcement.\n\n## ClawHub Publish\n\nThe ClawHub CLI must be installed and logged in:\n\n```bash\nclawhub whoami\n```\n\nPublish the skill:\n\n```bash\nclawhub publish . --version 0.3.1\n```\n\nAfter publishing:\n\n```bash\nclawhub inspect agentic-codex-dev --files\n```\n\nCheck that the listing shows the expected files, summary, version, and homepage. Remember that ClawHub publishes skills under MIT-0.\n\n## Manual Acceptance\n\nThe skill is publish-ready when:\n\n- a reviewer can understand the runtime behavior by reading `SKILL.md` alone\n- the source review explains why each major rule exists\n- no command in the skill installs software, reads credentials, restarts services, or changes global agent state\n- the repository tests and audit gate pass\n- the ClawHub listing, if published, points back to the GitHub source\n\nFile v0.3.5:references/source-review.md\n\n# Source Review\n\nReviewed on 2026-04-22. This file distills the cited public projects into operating rules for a general Codex software-development skill. It keeps transferable engineering structure and avoids copying project-specific product machinery.\n\n## Core Synthesis\n\nThe strongest pattern is not \"spawn more agents.\" The strongest pattern is a controlled software-development system:\n\n1. Make the target repo legible through `AGENTS.md`, docs, tests, and visible logs.\n2. Convert the user goal into acceptance criteria, task ownership, and verification commands.\n3. Assign roles only when a role changes the quality bar or parallelism.\n4. Isolate writers through branches, worktrees, or disjoint file ownership.\n5. Use strong models for design, review, and release decisions.\n6. Preserve durable state in repo-local task, memory, and report ledgers.\n7. Keep only changes that pass tests, review, and public-surface gates.\n\n## Source Notes\n\n### OpenAI Codex Subagents\n\nSources: <https://developers.openai.com/codex/subagents>, <https://developers.openai.com/codex/concepts/subagents>\n\n- Keep: custom agents should declare role, model, reasoning effort, sandbox, and developer instructions.\n- Keep: strong reviewer examples use `gpt-5.4`; docs examples also use `gpt-5.4-mini` and `gpt-5.3-codex-spark` for read-only or bounded tasks.\n- Keep: subagents help most with context isolation, read-heavy exploration, independent review, tests, and bounded implementation slices.\n- Keep: depth and thread caps matter because recursive delegation can increase cost and unpredictability.\n- Avoid: implicit fan-out. The parent thread must own task split, integration, and final judgment.\n\n### OpenAI Harness Engineering\n\nSource: <https://openai.com/index/harness-engineering/>\n\n- Keep: repo knowledge should be versioned and inspectable by agents.\n- Keep: `AGENTS.md` should be a map, with durable detail moved into docs and tests.\n- Keep: legibility is an engineering feature: state, logs, commands, and metrics should be available without private chat context.\n- Keep: taste and architecture need mechanical enforcement through tests, lint rules, boundaries, and release gates.\n- Avoid: prose-only policy when a repeated issue can become a check.\n\n### openai/symphony\n\nSource: <https://github.com/openai/symphony>\n\n- Keep: project work should become isolated autonomous implementation runs.\n- Keep: the operator manages work and evidence, not agent chatter.\n- Keep: runtime status, proof of work, retries, and handoff states should be visible.\n- Avoid: shared mutable workspaces for parallel writers unless ownership is explicit.\n\n### karpathy/autoresearch\n\nSource: <https://github.com/karpathy/autoresearch>\n\n- Keep: constrain research with one editable surface, a fixed budget, and one primary metric.\n- Keep: run a baseline first.\n- Keep: log every experiment as keep, discard, or crash.\n- Keep: equal metric results should prefer simpler code and fewer moving parts.\n- Avoid: letting exploratory logs flood the main context; store logs separately and summarize the signal.\n\n### forrestchang/andrej-karpathy-skills\n\nSource: <https://github.com/forrestchang/andrej-karpathy-skills>\n\n- Keep: state assumptions, tradeoffs, and success criteria before coding.\n- Keep: use minimum necessary code, local style, and surgical diffs.\n- Keep: simple tasks do not need the full process, but non-trivial tasks need rigor.\n- Avoid: speculative flexibility, adjacent refactors, and hidden confusion.\n\n### algorithmicsuperintelligence/openevolve\n\nSource: <https://github.com/algorithmicsuperintelligence/openevolve>\n\n- Keep: evaluator-first development. The evaluator defines truth.\n- Keep: reproducibility, seeded runs, and component isolation for experiments.\n- Keep: multi-objective scoring when correctness, performance, complexity, and memory all matter.\n- Keep: cascade evaluation to reject bad candidates before expensive checks.\n- Avoid: \"AI discovered it\" claims unless the run, seed, evaluator, and result log are reproducible.\n\n### algorithmicsuperintelligence/optillm\n\nSource: <https://github.com/algorithmicsuperintelligence/optillm>\n\n- Keep: inference-time scaling can improve hard decisions through best-of-N, self-consistency, plan search, verifier passes, and multi-agent reasoning.\n- Keep: extra compute should be reserved for security review, architecture choices, tricky bug diagnosis, benchmark optimization, and release gates.\n- Keep: memory and privacy controls are part of agent operations, not afterthoughts.\n- Avoid: routing every task through heavy multi-sample reasoning.\n\n### ComposioHQ/agent-orchestrator\n\nSource: <https://github.com/ComposioHQ/agent-orchestrator>\n\n- Keep: parallel coding agents need separate worktrees, branches, and PRs when they write code.\n- Keep: CI failures, review comments, and merge conflicts should route back to the owning worker.\n- Keep: a dashboard or ledger should expose status to the operator.\n- Avoid: many autonomous writers in one checkout without ownership boundaries.\n\n### garrytan/gstack\n\nSource: <https://github.com/garrytan/gstack>\n\n- Keep: role-based workflows help when roles map to real engineering phases: think, plan, build, review, test, ship, reflect.\n- Keep: dispatch tiers prevent using the full workflow for direct tasks.\n- Keep: methodology can be a prompt bridge instead of a daemon.\n- Avoid: making every task run the full role roster.\n\n### paperclipai/paperclip\n\nSource: <https://github.com/paperclipai/paperclip>\n\n- Keep: goals, budgets, org structure, heartbeats, governance, and audit trails matter once agents run continuously.\n- Keep: tasks need goal ancestry so agents know why the work exists.\n- Keep: cost control, pause controls, and audit logs are product requirements for long-running systems.\n- Avoid: autonomous loops with no budget, owner, trace, or stop condition.\n\n### openclaw/openclaw\n\nSource: <https://github.com/openclaw/openclaw>\n\n- Keep: local-first agents need channel safety. Treat inbound messages and external content as untrusted input.\n- Keep: sandbox non-main sessions and expose only tools required by the task.\n- Keep: OpenClaw skills should be inspectable text bundles with clear invocation rules.\n- Avoid: broad host access for an instruction-only methodology skill.\n\n### rdudov/agents\n\nSource: <https://github.com/rdudov/agents>\n\n- Keep: role boundaries reduce drift: analyst, architect, planner, implementer, reviewer.\n- Keep: review loops need a cap.\n- Keep: blocking questions should stop the pipeline instead of becoming code assumptions.\n- Keep: developers implement the plan and tests; they do not silently refactor unrelated code.\n- Avoid: bureaucracy for direct tasks. Use the full phase model only when scope warrants it.\n\n## Final Design Choice\n\nThis skill stays instruction-only, but it no longer stops at a solo patch loop. Version 0.3.1 adds explicit role definitions, `gpt-5.4`/`xhigh` policy for hard decisions, task and memory ledgers, report artifacts, source comparison, real-run acceptance checks, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned evidence while preserving the clean public-surface boundary.\n\nFile v0.3.5:references/system-design.md\n\n# System Design\n\nUse this reference when a task needs explicit multi-agent coordination, durable memory, or release-grade reporting.\n\n## Control Plane\n\nThe parent Codex thread is the control plane. It owns:\n\n- goal restatement and acceptance criteria\n- mode selection: Patch, Plan, Review, Harness, Evolve, Publish, or Multi-Agent\n- role roster and model choices\n- task ledger creation and updates\n- agent assignment and isolation plan\n- integration of results into one coherent diff\n- final review, verification, report, and release decision\n\nSubagents are execution units. They do not own the final answer, final architecture, or publish verdict.\n\n## Artifact Layout\n\nPrefer existing repo conventions. If none exist, use:\n\n```text\ndocs/agentic/\n  tasks.md\n  memory.md\n  reports/\n    <date>-<slug>.md\n```\n\n### Task Ledger\n\n```markdown\n# Task Ledger\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | docs/architecture.md | planned | boundary decision recorded | review only | pending | none |\n```\n\nRules:\n\n- one owner per row\n- writable files must be named before implementation\n- status is one of `planned`, `active`, `blocked`, `review`, `done`, `discarded`\n- verification must be a command, manual check, or reviewer gate\n- discarded work needs a reason\n\n### Memory Ledger\n\n```markdown\n# Memory Ledger\n\n## Stable Facts\n\n- The public API entry point is `<symbol>`; last verified on `<date>` with `<command>`.\n\n## Decisions\n\n- Use `<approach>` because `<reason>`. Rejected `<alternative>` because `<reason>`.\n\n## Hazards\n\n- Do not touch `<surface>` without running `<check>`.\n```\n\nRules:\n\n- store stable facts, decisions, commands, and hazards\n- do not store secrets, private endpoints, local machine paths, raw logs, or copied chat\n- include last-verified dates for facts that can decay\n- remove stale facts instead of appending contradictions\n\n### Report\n\n```markdown\n# Agentic Report: <goal>\n\n## Objective\n\n## Sources\n\n## Tasks\n\n## Changed Files\n\n## Verification\n\n## Review Findings\n\n## Memory Updates\n\n## Residual Risks\n\n## Release Status\n```\n\nRules:\n\n- every completed task has a report entry\n- every failed command is named with the reason it failed or the follow-up owner\n- final status is one of `not ready`, `ready for PR`, `ready to publish`, or `published`\n\n## Assignment Template\n\n```text\nRole: <architect|explorer|implementer|reviewer|qa|memory-curator>\nModel: <model id>\nReasoning: <effort>\nSandbox: <read-only|workspace-write>\nOwned files: <paths or read-only surface>\nTask: <one bounded objective>\nAcceptance: <observable done condition>\nVerification: <command or check>\nOutput: <required sections>\nConstraints: do not inspect secrets; do not broaden scope; do not edit outside ownership.\n```\n\n## Isolation Policy\n\n- Read-only agents may share a checkout.\n- Writer agents should use separate branches or worktrees when parallel edits are possible.\n- If writers share a checkout, file ownership must be disjoint and recorded in the task ledger.\n- CI failures and review comments go back to the owner of the task that introduced the change.\n- The parent resolves conflicts and merges, then runs final verification.\n\n## Stop Conditions\n\nStop and ask the user, or downgrade to a plan-only result, when:\n\n- acceptance criteria cannot be stated\n- required secrets, accounts, paid resources, or private dashboards are unavailable\n- two agents need to edit the same files without a safe order\n- tests cannot run and no credible manual verifier exists\n- security or public-surface scan finds unresolved leaks\n\nFile v0.3.5:agents/openai.yaml\n\ninterface:\n  display_name: \"Agentic Codex Dev\"\n  short_description: \"Codex multi-agent software development system\"\n  default_prompt: \"Use $agentic-codex-dev to plan, implement, verify, review, report, and publish an agentic software development change with explicit roles and a clean GitHub and ClawHub surface.\"\n\npolicy:\n  allow_implicit_invocation: false\n\nArchive v0.3.4: 8 files, 17651 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3728b), references/publish-checklist.md (2089b), references/source-review.md (7182b), references/system-design.md (3616b), SKILL.md (16494b), _meta.json (136b)\n\nFile v0.3.4:SKILL.md\n\n---\nname: agentic-codex-dev\ndescription: Use when planning, implementing, reviewing, coordinating, or publishing agentic software development work with Codex, GitHub, and OpenClaw/ClawHub. Provides a production-grade multi-agent operating loop with role roster, model policy, task ledger, memory ledger, report artifacts, verification gates, and anti-bleed public-surface review.\nversion: 0.3.4\nuser-invocable: true\ndisable-model-invocation: true\nmetadata: {\"openclaw\":{\"homepage\":\"https://github.com/zack-dev-cm/agentic-codex-dev-skill\",\"skillKey\":\"agentic-codex-dev\",\"requires\":{\"bins\":[\"git\",\"clawhub\"],\"anyBins\":[\"python3\",\"python\"]},\"install\":[{\"kind\":\"node\",\"label\":\"Install ClawHub CLI\",\"package\":\"clawhub\",\"bins\":[\"clawhub\"]}],\"tags\":[\"codex\",\"github\",\"clawhub\",\"agentic-development\"]}}\n---\n\n# Agentic Codex Dev\n\nOperate Codex like a disciplined software team: clear goal, explicit roles, scoped ownership, evidence, tests, review, report.\n\n## When to Use\n\nUse this skill for:\n\n- coding tasks where Codex should inspect, modify, test, and report on a GitHub repo\n- turning a rough product or bug request into scoped implementation work\n- setting up repo-local `AGENTS.md`, `.codex/agents/`, or skill instructions\n- reviewing agent-generated code for correctness, tests, security, and public-surface leaks\n- preparing a GitHub repo or ClawHub skill for open-source publication\n- coordinating explicit parallel/subagent work with role ownership and integration control\n\nDo not use it for one-line answers, pure brainstorming, or tasks that only need a command output.\n\n## Runtime Requirements\n\nClawHub requirement metadata for this skill declares `git`, `python3`, and `clawhub`, following the ClawHub skill metadata format at <https://github.com/openclaw/clawhub/blob/main/docs/skill-format.md>.\n\n- Local plan, review, and implementation modes may work with the tools already available in the host.\n- Verification and publish modes expect the declared binaries plus optional Python modules such as `antirot` and `codex_harness`.\n- This skill should not request, print, or store credentials. GitHub and ClawHub publishing must use existing local authenticated CLI sessions, or the user must authenticate manually outside the prompt.\n- Do not run `git push`, `clawhub publish`, or other remote-changing commands unless the user asked for publish or remote update work.\n\n## Core Loop\n\n1. Restate the goal and name the verification step before editing.\n2. Read the repo map: `AGENTS.md`, README, package config, tests, and the files closest to the task.\n3. Define concrete success criteria that would let a reviewer say \"done\".\n4. Make the narrowest defensible change. Match local style. Avoid speculative abstractions.\n5. Run the highest-signal local check. Add a focused smoke test when behavior changed.\n6. Review the diff for bugs, regressions, secrets, private paths, and public-surface bleed.\n7. Report what changed, how it was verified, and any residual risk.\n\nIf the task is unclear, stop early and name the ambiguity. Prefer one precise question over guessing.\n\n## Operating Rules\n\n- Treat repository files as the source of truth. If knowledge matters later, put it in repo docs.\n- Keep `AGENTS.md` short. Use it as an index to durable docs, not a giant prompt.\n- Prefer boring, inspectable code over opaque magic. Agents compound what they can read.\n- Touch only files required for the goal. Mention unrelated problems; do not fix them unless asked.\n- Use structured APIs, tests, and parsers where available. Avoid fragile string tricks.\n- Convert repeated review feedback into checks, docs, or templates.\n- Keep logs and long command output out of the main narrative; summarize the signal.\n- Avoid asking an agent to read undeclared secret files or sync credentials as part of a skill.\n\n## Scope Modes\n\nPick the mode that fits the risk:\n\n- **Patch**: one bug or one focused feature. Read close code, edit, test, review.\n- **Plan**: ambiguous or multi-file work. Write a short acceptance plan before editing.\n- **Review**: findings first, with emphasis on correctness, regressions, security, tests, and leaks as summarized in [source review](references/source-review.md).\n- **Harness**: improve repo legibility: docs, CI, local scripts, custom agents, or audit gates.\n- **Evolve**: metric-driven optimization. One variable per experiment, fixed budget, log keep/discard.\n- **Publish**: GitHub/ClawHub release readiness, metadata, license, docs, and verification.\n- **Skill Import**: adapt upstream skills or agent workflows into local ClawHub/Codex/Claude artifacts with staging, audit, and active-project impact gates before install.\n- **Multi-Agent**: explicit role roster, task ledger, isolation plan, review gates, memory update, and final report.\n\nPrefer Patch unless the task shows it needs more structure. Use Multi-Agent only when the user explicitly asks for subagents, delegation, or parallel agent work.\n\n## Skill Import Project Mode\n\nUse this mode when adapting upstream skill repositories, `SKILL.md` files, Claude commands, agent personas, or workflow packs into local Codex, Claude Code, ClawHub, or OpenClaw artifacts.\n\nDefault stance: import ideas, not trust. Do not install, enable, publish, or run upstream skill code until provenance, public-surface safety, and active-project impact are understood.\n\nWorkflow:\n\n1. Inventory the source:\n   - source path or URL,\n   - commit or release tag when available,\n   - skill names,\n   - commands,\n   - personas,\n   - hooks,\n   - scripts,\n   - references,\n   - licenses.\n2. Classify each candidate:\n   - `PORT`: narrow, safe, runtime-neutral enough to adapt directly,\n   - `REWRITE`: useful workflow pattern but runtime mechanics, trigger breadth, scripts, privacy boundaries, or proof criteria must change,\n   - `REJECT`: prompt override, credential access, hidden install, destructive behavior, unreviewable executable content, scraping, auto-posting, or claims risk.\n3. Stage in an isolated repo-local workspace. Do not install into global Codex, Claude, or OpenClaw paths during evaluation.\n4. Assign owners in the task ledger:\n   - explorer for source inventory,\n   - architect for runtime boundary decisions,\n   - implementer for rewritten skill text,\n   - reviewer for public-surface and install-risk review,\n   - memory curator for stable decisions and rejected patterns.\n5. Convert to a portable core plus adapters:\n   - canonical `SKILL.md` core,\n   - Codex notes for custom agents, sandbox, subagent boundaries, and verification,\n   - Claude Code notes for skills, commands, agents, and hook cautions,\n   - ClawHub/OpenClaw notes for metadata, required binaries, install destination, and public registry risk.\n6. Run gates before install or publish:\n   - static skill safety scan,\n   - duplicate-name and precedence check,\n   - public-surface scan for private paths, local URLs, tokens, copied paid/community content, and stale claims,\n   - focused fixture or dry-run prompt,\n   - report entry naming accepted, rewritten, and rejected candidates.\n7. Install or publish only after explicit user approval for the exact artifact, destination, and command.\n\nStop conditions:\n\n- source provenance is unclear,\n- candidate requires secrets or private exports,\n- destination would shadow an active skill without explicit replace approval,\n- hooks or scripts mutate workspaces outside declared ownership,\n- no credible verification path exists.\n\n## System Design\n\nFor non-trivial or multi-agent work, set up a control plane before coding:\n\n- **Orchestrator**: the main Codex thread owns requirements, task split, agent selection, integration, final review, and user communication.\n- **Role agents**: subagents are optional workers with declared purpose, model, reasoning effort, sandbox, file ownership, and output schema.\n- **Artifacts**: use repo-local ledgers so work survives context loss and can be reviewed without private chat history.\n- **Isolation**: prefer branches or worktrees per writer when multiple agents edit. If one checkout is shared, assign disjoint file ownership.\n- **Gates**: no task is done until its acceptance criteria, verification command, diff review, public-surface scan, and report entry are complete.\n\nWhen this structure is overkill, keep a solo Patch flow and still preserve the same verification discipline.\n\n## Task, Memory, and Report Ledgers\n\nCreate or update these artifacts when work is multi-agent, multi-turn, risky, or intended for publication:\n\n- `docs/agentic/tasks.md`: task id, owner role, goal, owned files, status, acceptance criteria, verification, result, blocker.\n- `docs/agentic/memory.md`: stable repo facts, architecture decisions, commands that actually work, hazards, rejected approaches, last-verified date. Do not store secrets, tokens, private paths, or raw logs.\n- `docs/agentic/reports/<date>-<slug>.md`: final objective, source links, task outcomes, changed files, tests, review findings, unresolved risks, release or PR status.\n\nIf the target repo already has equivalent docs, use the local convention instead of inventing new paths.\n\n## Role Roster\n\nUse this roster as the default multi-agent team. The parent thread stays responsible for coordination and final judgment.\n\n| Role | Default model | Reasoning | Scope | Required output |\n| --- | --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` for critical design/release, `high` otherwise | Owns task split, integration, report | plan, assignments, final decision |\n| Analyst | `gpt-5.4` | `high` | Turns vague request into requirements and risks | assumptions, open questions, acceptance criteria |\n| Architect | `gpt-5.4` | `xhigh` | System design, boundaries, dependency choices | design note, rejected options, invariants |\n| Planner | `gpt-5.4` | `high` | Breaks design into ordered tasks | task ledger rows with owners and gates |\n| Explorer | `gpt-5.4-mini` or `gpt-5.3-codex-spark` | `medium` | Read-only code mapping and evidence gathering | files, symbols, execution path, uncertainty |\n| Implementer | `gpt-5.4` for risky code, `gpt-5.3-codex-spark` for bounded edits | `high` or `medium` | Writes only owned files | patch summary, tests, residual risks |\n| Reviewer | `gpt-5.4` | `xhigh` | Correctness, security, regressions, tests, public surface | findings first, file/line evidence, verdict |\n| QA/CI Analyst | `gpt-5.4` | `high` | Reproduction, failing checks, browser or CLI evidence | exact command, observed failure, fix owner |\n| Memory Curator | `gpt-5.4-mini` | `medium` | Updates durable docs after decisions land | memory entries, stale entries removed |\n\n## Subagents\n\nOnly use subagents when the user explicitly asks for subagents, delegation, or parallel agent work.\n\nGood delegation targets:\n\n- read-heavy codebase mapping\n- independent test or CI-log analysis\n- independent review categories such as security, test gaps, or docs correctness\n- disjoint implementation slices with clearly separate file ownership\n\nBad delegation targets:\n\n- the immediate blocker for your next local step\n- tightly coupled edits in the same files\n- vague \"go improve the code\" work\n- recursive fan-out with no cap\n\nWhen delegating, give each agent a bounded task, a clear output shape, and explicit ownership. Keep the main thread focused on requirements, decisions, integration, and final review. Keep `agents.max_depth = 1` unless the user explicitly accepts recursive delegation risk; this matches the Codex subagent configuration surface documented at <https://developers.openai.com/codex/subagents>.\n\nDelegation prompt shape:\n\n```text\nRole: reviewer\nModel: gpt-5.4\nReasoning: xhigh\nOwnership: read-only review of <files or branch>\nTask: find correctness, security, regression, test, and public-surface risks.\nOutput: findings first with file/line evidence, then open questions, then verdict.\nDo not edit files. Do not inspect secrets. Do not broaden scope.\n```\n\n## Model Policy\n\n- Use `gpt-5.4` with `xhigh` reasoning for architecture, security review, release decisions, and ambiguous multi-agent coordination; Codex custom-agent examples document `gpt-5.4` reviewer roles at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4` with `high` reasoning for implementation where correctness or cross-module behavior matters; model selection follows the Codex custom-agent configuration surface at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4-mini` or `gpt-5.3-codex-spark` for read-only exploration, docs checks, and bounded cleanup where speed matters and the output will be reviewed; both model families appear in Codex custom-agent examples at <https://developers.openai.com/codex/subagents>.\n- Do not use a budget model for final architecture, security, or publish verdicts.\n- Use extra compute selectively: best-of-N, independent reviewer passes, or verifier checks only when the decision is expensive to reverse; optillm documents inference-time scaling techniques at <https://github.com/algorithmicsuperintelligence/optillm>.\n\n## Implementation Discipline\n\nBefore editing:\n\n- inspect the existing patterns\n- identify the likely tests or smoke command\n- check dirty git state and avoid touching unrelated user changes\n- state the planned edit in one or two sentences\n\nWhile editing:\n\n- keep the diff surgical\n- add tests when behavior, contracts, or public output changes\n- avoid new dependencies unless they clearly reduce risk or complexity\n- keep comments rare and useful\n\nAfter editing:\n\n- run the named verification\n- inspect the diff, not just test output\n- update docs only when user-facing behavior or workflow changed\n- do not call work published until the public surface is clean\n\n## Review Checklist\n\nReview every non-trivial result for:\n\n- Does every changed line trace to the stated goal?\n- Are edge cases covered by tests or a clear smoke path?\n- Did the change preserve existing public APIs and CLI behavior?\n- Did docs/examples drift from actual behavior?\n- Did any secret-like string, local path, private URL, copied dashboard, or stale release note enter the repo?\n- Did the final diff remove avoidable complexity from the first draft, as recommended in [source review](references/source-review.md)?\n\n## Consistency and Effectiveness Gates\n\nFor multi-agent work, verify the process itself:\n\n- Every task has an owner, owned files, acceptance criteria, verification command, and result.\n- Every subagent output is mapped to a task or explicitly discarded with a reason.\n- No writer agent edits outside its assigned ownership without parent approval.\n- At least one reviewer pass is read-only and independent of the implementer.\n- The final report names changed files, commands run, failed checks, source links, residual risk, and release status.\n- Memory updates contain stable facts only; do not store raw chat, secrets, local credentials, or transient logs.\n- If a metric-driven change is attempted, record baseline, candidate, verifier, result, and keep/discard decision.\n\n## Real Example Eval\n\nFor a serious workflow eval, run this skill against a real repo task and archive the result in the report ledger. A valid eval has:\n\n- baseline repo state and user goal\n- role roster used, including model and reasoning choices\n- task ledger rows with owners and file boundaries\n- at least one implementation or review task with verification output\n- public-release check for private-path examples, local-only URLs, secret-shaped placeholders, and stale claims\n- final report with changed files, tests, residual risks, and follow-up blockers\n\nUse [example run](references/example-run.md) as the minimum acceptance shape.\n\n## GitHub and ClawHub Publish Gate\n\nBefore publishing:\n\n- README or skill summary says what it does, when to use it, and what it does not do.\n- License is compatible with the target surface. ClawHub publishes skills under MIT-0.\n- `SKILL.md` has frontmatter `name`, `description`, and `version`.\n- The skill folder contains only text-based files needed at runtime.\n- No hidden install scripts, credential readers, service restarts, or local machine assumptions.\n- Public repo has security, contribution, support, CI, and release/audit checks when applicable.\n- Run the repo's public-surface gate before pushing or publishing to a registry.\n\nFor this skill's source analysis, read `references/source-review.md` and `references/comparison-matrix.md`.\nFor multi-agent artifacts and templates, read `references/system-design.md`.\nFor release commands and manual checks, read `references/publish-checklist.md`.\n\nFile v0.3.4:_meta.json\n\n{\n  \"ownerId\": \"kn7dhjt1k1f111whp13fmrqwnh81tn1v\",\n  \"slug\": \"agentic-codex-dev\",\n  \"version\": \"0.3.4\",\n  \"publishedAt\": 1778744108991\n}\n\nFile v0.3.4:references/comparison-matrix.md\n\n# Comparison Matrix\n\nReviewed on 2026-04-22. Use this matrix when judging whether the skill is strong enough for multi-agent software development rather than a generic coding checklist.\n\n| Source | Strong pattern | Risk if copied blindly | Skill response |\n| --- | --- | --- | --- |\n| OpenAI Codex subagents | Custom roles with model, reasoning, sandbox, and developer instructions | More agents can add cost and coordination failure | Subagents require explicit user intent, role ownership, and parent integration |\n| OpenAI Codex concepts | Context isolation and parallel investigation | Delegation can hide the critical path | Parent keeps immediate blockers local and uses agents for side work |\n| OpenAI harness engineering | Repo-local memory, docs, tests, and legible state | Prose policy can drift without checks | Task, memory, report ledgers plus public-surface tests |\n| openai/symphony | Isolated implementation runs | Workflow engine complexity may exceed need | Use branches/worktrees for writers; do not require a daemon |\n| karpathy/autoresearch | Baseline, budget, one metric, keep/discard/crash log | Research framing may not fit product work | Apply evaluator-first discipline only when optimizing behavior |\n| forrestchang/andrej-karpathy-skills | Assumptions, tradeoffs, surgical diffs | Too little structure for multi-agent runs | Keep the concise style, add role roster and ledgers only for serious work |\n| openevolve | Reproducible evaluator-first evolution, Pareto tradeoffs, cascade checks | Claims can outrun evidence | Require baseline, verifier, result log, and keep/discard decision |\n| optillm | Inference-time scaling, verifier passes, multi-agent reasoning | Expensive compute can become default theater | Reserve `gpt-5.4`/`xhigh` and independent reviewers for hard decisions |\n| agent-orchestrator | Worktree isolation, CI/review feedback routed to owner | Autonomous fleets need heavy operations | Adopt ownership and feedback routing without requiring its platform |\n| gstack | Clear role taxonomy and dispatch tiers | Full role stack for every task slows direct fixes | Scope modes decide when to use Patch vs Multi-Agent |\n| paperclip | Goals, budgets, heartbeats, org chart, audit trail | Continuous autonomy can run away | Add owner, budget/risk thinking, stop conditions, and reports |\n| openclaw | Local-first assistant, skills, sandbox and channel safety | Broad host access is unnecessary for a methodology skill | Keep bundle instruction-only and public-surface clean |\n| rdudov/agents | Analyst, architect, planner, developer, reviewer boundaries | Rigid phase gates can create process drag | Use roles when the task is ambiguous, risky, or parallel |\n\n## Previous Version Weak Points\n\nVersion 0.1.2 was useful as a publish-clean patch loop, but it was not enough for the stated goal of multi-agent software development:\n\n- no explicit role roster with model and reasoning policy\n- no default use of `gpt-5.4`/`xhigh` for architecture, review, and release decisions\n- no system design for orchestration, isolation, ledgers, or reports\n- no task ledger, memory ledger, or report artifact\n- no real-run eval shape\n- no process consistency checks that tie agent outputs to tasks and verification\n- no test ratchet for role/model/system-design coverage\n- bleed scan omitted Python files\n\n## Target Bar\n\nA release is acceptable only when a reviewer can see:\n\n- the role roster and model policy in `SKILL.md`\n- source-by-source comparison in this file\n- system design templates in `references/system-design.md`\n- real-run acceptance shape in `references/example-run.md`\n- automated tests that fail if the public skill regresses to a generic patch loop\n- anti-bleed coverage that includes Python test and helper files\n\nFile v0.3.4:references/example-run.md\n\n# Example Run\n\nThis is the minimum shape for a real eval of the skill. It is intentionally concrete enough to review and general enough to run on any public repository.\n\n## Scenario\n\nGoal: upgrade an instruction-only Codex skill from a generic patch workflow into a multi-agent software-development operating loop.\n\nAcceptance:\n\n- `SKILL.md` includes system design, role roster, model policy, task ledger, memory ledger, report ledger, consistency gates, and real-run eval guidance.\n- Source comparison covers all cited public projects.\n- Custom agent configs declare models and reasoning effort.\n- Public-surface tests fail if role/model/system-design coverage is removed.\n- Anti-bleed scan includes Python files.\n- Local verification passes, with any CI or publish blocker named explicitly.\n\n## Role Roster Used\n\n| Role | Model | Reasoning | Purpose |\n| --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` | own requirements, source synthesis, implementation, final report |\n| Architect | `gpt-5.4` | `xhigh` | validate system design and role boundaries |\n| Implementer | `gpt-5.4` | `high` | edit skill, references, docs, tests, metadata |\n| Reviewer | `gpt-5.4` | `xhigh` | check correctness, public surface, test coverage, publish risk |\n| Memory Curator | `gpt-5.4-mini` | `medium` | update durable docs after verification |\n\n## Task Ledger Sample\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | `SKILL.md`, `references/system-design.md` | done | system design and role roster are explicit | public-surface tests | pending until local run | none |\n| T2 | implementer | `tests/test_public_surface.py` | done | Python files are scanned for bleed | unit tests | pending until local run | none |\n| T3 | reviewer | public repo surface | review | no private paths, tokens, local URLs, stale version claims | anti-bleed and AntiRot | pending until local run | CI workflow depends on token scope |\n\n## Memory Ledger Sample\n\nStable facts:\n\n- Runtime skill files are text-only and publish through ClawHub.\n- `.clawhubignore` excludes repo harness files from the public skill bundle.\n- GitHub Actions workflow creation may require token scope outside the normal repo push permission.\n\nDecisions:\n\n- Keep the skill instruction-only. Add templates and tests rather than a daemon.\n- Use `gpt-5.4`/`xhigh` for architecture and reviewer roles; use faster models only for reviewed, read-only work.\n\nHazards:\n\n- Do not claim CI enforcement unless the workflow exists in the remote repository.\n- Do not let public examples include private paths, local URLs, tokens, or copied chat logs.\n\n## Report Sample\n\nObjective: release version 0.3.1 with multi-agent operating structure, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned real-run evidence.\n\nSources: OpenAI Codex subagents, OpenAI harness engineering, optillm, openevolve, autoresearch, symphony, paperclip, gstack, OpenClaw, Andrej Karpathy skills, agent-orchestrator, rdudov agents.\n\nChanged files:\n\n- `SKILL.md`\n- `references/source-review.md`\n- `references/comparison-matrix.md`\n- `references/system-design.md`\n- `references/example-run.md`\n- `.codex/agents/*.toml`\n- `tests/test_public_surface.py`\n\nVerification:\n\n- `python3 -m unittest discover -s tests`\n- `python3 -m antirot.cli lint SKILL.md --strict`\n- `python3 -m codex_harness audit . --strict --min-score 90`\n- `clawhub inspect agentic-codex-dev --files` after publish\n\nResidual risks:\n\n- CI cannot be considered enforced until the remote repository has a workflow.\n- Any future source review must re-check the public URLs because repository behavior can change.\n\nFile v0.3.4:references/publish-checklist.md\n\n# Publish Checklist\n\nUse this when preparing the skill for GitHub and ClawHub.\n\n## Local Review\n\n1. Confirm `SKILL.md` frontmatter:\n   - `name: agentic-codex-dev`\n   - `description: ...`\n   - `version: 0.3.1`\n2. Confirm the folder name is the intended ClawHub slug: `agentic-codex-dev`.\n3. Confirm every file is text-based and needed:\n   - `SKILL.md`\n   - `agents/openai.yaml`\n   - `references/source-review.md`\n   - `references/comparison-matrix.md`\n   - `references/system-design.md`\n   - `references/example-run.md`\n   - `references/publish-checklist.md`\n4. Search for private paths, local URLs, tokens, and copied private notes.\n5. Run the repository gate:\n\n```bash\npython3 -m unittest discover -s tests\npython3 -m antirot.cli lint SKILL.md --strict\npython3 -m codex_harness audit . --strict --min-score 90\n```\n\n## GitHub Publish\n\nFrom the repository root:\n\n```bash\ngit status --short\ngit add .\ngit commit -m \"Upgrade agentic Codex development skill\"\ngit push\n```\n\nIf the worktree contains unrelated user changes, stage only the files above.\n\nDo not claim GitHub CI enforcement unless `.github/workflows/ci.yml` exists on the remote default branch. If push tokens cannot create workflows, record that as a release blocker or create the workflow through the GitHub UI before claiming enforcement.\n\n## ClawHub Publish\n\nThe ClawHub CLI must be installed and logged in:\n\n```bash\nclawhub whoami\n```\n\nPublish the skill:\n\n```bash\nclawhub publish . --version 0.3.1\n```\n\nAfter publishing:\n\n```bash\nclawhub inspect agentic-codex-dev --files\n```\n\nCheck that the listing shows the expected files, summary, version, and homepage. Remember that ClawHub publishes skills under MIT-0.\n\n## Manual Acceptance\n\nThe skill is publish-ready when:\n\n- a reviewer can understand the runtime behavior by reading `SKILL.md` alone\n- the source review explains why each major rule exists\n- no command in the skill installs software, reads credentials, restarts services, or changes global agent state\n- the repository tests and audit gate pass\n- the ClawHub listing, if published, points back to the GitHub source\n\nFile v0.3.4:references/source-review.md\n\n# Source Review\n\nReviewed on 2026-04-22. This file distills the cited public projects into operating rules for a general Codex software-development skill. It keeps transferable engineering structure and avoids copying project-specific product machinery.\n\n## Core Synthesis\n\nThe strongest pattern is not \"spawn more agents.\" The strongest pattern is a controlled software-development system:\n\n1. Make the target repo legible through `AGENTS.md`, docs, tests, and visible logs.\n2. Convert the user goal into acceptance criteria, task ownership, and verification commands.\n3. Assign roles only when a role changes the quality bar or parallelism.\n4. Isolate writers through branches, worktrees, or disjoint file ownership.\n5. Use strong models for design, review, and release decisions.\n6. Preserve durable state in repo-local task, memory, and report ledgers.\n7. Keep only changes that pass tests, review, and public-surface gates.\n\n## Source Notes\n\n### OpenAI Codex Subagents\n\nSources: <https://developers.openai.com/codex/subagents>, <https://developers.openai.com/codex/concepts/subagents>\n\n- Keep: custom agents should declare role, model, reasoning effort, sandbox, and developer instructions.\n- Keep: strong reviewer examples use `gpt-5.4`; docs examples also use `gpt-5.4-mini` and `gpt-5.3-codex-spark` for read-only or bounded tasks.\n- Keep: subagents help most with context isolation, read-heavy exploration, independent review, tests, and bounded implementation slices.\n- Keep: depth and thread caps matter because recursive delegation can increase cost and unpredictability.\n- Avoid: implicit fan-out. The parent thread must own task split, integration, and final judgment.\n\n### OpenAI Harness Engineering\n\nSource: <https://openai.com/index/harness-engineering/>\n\n- Keep: repo knowledge should be versioned and inspectable by agents.\n- Keep: `AGENTS.md` should be a map, with durable detail moved into docs and tests.\n- Keep: legibility is an engineering feature: state, logs, commands, and metrics should be available without private chat context.\n- Keep: taste and architecture need mechanical enforcement through tests, lint rules, boundaries, and release gates.\n- Avoid: prose-only policy when a repeated issue can become a check.\n\n### openai/symphony\n\nSource: <https://github.com/openai/symphony>\n\n- Keep: project work should become isolated autonomous implementation runs.\n- Keep: the operator manages work and evidence, not agent chatter.\n- Keep: runtime status, proof of work, retries, and handoff states should be visible.\n- Avoid: shared mutable workspaces for parallel writers unless ownership is explicit.\n\n### karpathy/autoresearch\n\nSource: <https://github.com/karpathy/autoresearch>\n\n- Keep: constrain research with one editable surface, a fixed budget, and one primary metric.\n- Keep: run a baseline first.\n- Keep: log every experiment as keep, discard, or crash.\n- Keep: equal metric results should prefer simpler code and fewer moving parts.\n- Avoid: letting exploratory logs flood the main context; store logs separately and summarize the signal.\n\n### forrestchang/andrej-karpathy-skills\n\nSource: <https://github.com/forrestchang/andrej-karpathy-skills>\n\n- Keep: state assumptions, tradeoffs, and success criteria before coding.\n- Keep: use minimum necessary code, local style, and surgical diffs.\n- Keep: simple tasks do not need the full process, but non-trivial tasks need rigor.\n- Avoid: speculative flexibility, adjacent refactors, and hidden confusion.\n\n### algorithmicsuperintelligence/openevolve\n\nSource: <https://github.com/algorithmicsuperintelligence/openevolve>\n\n- Keep: evaluator-first development. The evaluator defines truth.\n- Keep: reproducibility, seeded runs, and component isolation for experiments.\n- Keep: multi-objective scoring when correctness, performance, complexity, and memory all matter.\n- Keep: cascade evaluation to reject bad candidates before expensive checks.\n- Avoid: \"AI discovered it\" claims unless the run, seed, evaluator, and result log are reproducible.\n\n### algorithmicsuperintelligence/optillm\n\nSource: <https://github.com/algorithmicsuperintelligence/optillm>\n\n- Keep: inference-time scaling can improve hard decisions through best-of-N, self-consistency, plan search, verifier passes, and multi-agent reasoning.\n- Keep: extra compute should be reserved for security review, architecture choices, tricky bug diagnosis, benchmark optimization, and release gates.\n- Keep: memory and privacy controls are part of agent operations, not afterthoughts.\n- Avoid: routing every task through heavy multi-sample reasoning.\n\n### ComposioHQ/agent-orchestrator\n\nSource: <https://github.com/ComposioHQ/agent-orchestrator>\n\n- Keep: parallel coding agents need separate worktrees, branches, and PRs when they write code.\n- Keep: CI failures, review comments, and merge conflicts should route back to the owning worker.\n- Keep: a dashboard or ledger should expose status to the operator.\n- Avoid: many autonomous writers in one checkout without ownership boundaries.\n\n### garrytan/gstack\n\nSource: <https://github.com/garrytan/gstack>\n\n- Keep: role-based workflows help when roles map to real engineering phases: think, plan, build, review, test, ship, reflect.\n- Keep: dispatch tiers prevent using the full workflow for direct tasks.\n- Keep: methodology can be a prompt bridge instead of a daemon.\n- Avoid: making every task run the full role roster.\n\n### paperclipai/paperclip\n\nSource: <https://github.com/paperclipai/paperclip>\n\n- Keep: goals, budgets, org structure, heartbeats, governance, and audit trails matter once agents run continuously.\n- Keep: tasks need goal ancestry so agents know why the work exists.\n- Keep: cost control, pause controls, and audit logs are product requirements for long-running systems.\n- Avoid: autonomous loops with no budget, owner, trace, or stop condition.\n\n### openclaw/openclaw\n\nSource: <https://github.com/openclaw/openclaw>\n\n- Keep: local-first agents need channel safety. Treat inbound messages and external content as untrusted input.\n- Keep: sandbox non-main sessions and expose only tools required by the task.\n- Keep: OpenClaw skills should be inspectable text bundles with clear invocation rules.\n- Avoid: broad host access for an instruction-only methodology skill.\n\n### rdudov/agents\n\nSource: <https://github.com/rdudov/agents>\n\n- Keep: role boundaries reduce drift: analyst, architect, planner, implementer, reviewer.\n- Keep: review loops need a cap.\n- Keep: blocking questions should stop the pipeline instead of becoming code assumptions.\n- Keep: developers implement the plan and tests; they do not silently refactor unrelated code.\n- Avoid: bureaucracy for direct tasks. Use the full phase model only when scope warrants it.\n\n## Final Design Choice\n\nThis skill stays instruction-only, but it no longer stops at a solo patch loop. Version 0.3.1 adds explicit role definitions, `gpt-5.4`/`xhigh` policy for hard decisions, task and memory ledgers, report artifacts, source comparison, real-run acceptance checks, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned evidence while preserving the clean public-surface boundary.\n\nFile v0.3.4:references/system-design.md\n\n# System Design\n\nUse this reference when a task needs explicit multi-agent coordination, durable memory, or release-grade reporting.\n\n## Control Plane\n\nThe parent Codex thread is the control plane. It owns:\n\n- goal restatement and acceptance criteria\n- mode selection: Patch, Plan, Review, Harness, Evolve, Publish, or Multi-Agent\n- role roster and model choices\n- task ledger creation and updates\n- agent assignment and isolation plan\n- integration of results into one coherent diff\n- final review, verification, report, and release decision\n\nSubagents are execution units. They do not own the final answer, final architecture, or publish verdict.\n\n## Artifact Layout\n\nPrefer existing repo conventions. If none exist, use:\n\n```text\ndocs/agentic/\n  tasks.md\n  memory.md\n  reports/\n    <date>-<slug>.md\n```\n\n### Task Ledger\n\n```markdown\n# Task Ledger\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | docs/architecture.md | planned | boundary decision recorded | review only | pending | none |\n```\n\nRules:\n\n- one owner per row\n- writable files must be named before implementation\n- status is one of `planned`, `active`, `blocked`, `review`, `done`, `discarded`\n- verification must be a command, manual check, or reviewer gate\n- discarded work needs a reason\n\n### Memory Ledger\n\n```markdown\n# Memory Ledger\n\n## Stable Facts\n\n- The public API entry point is `<symbol>`; last verified on `<date>` with `<command>`.\n\n## Decisions\n\n- Use `<approach>` because `<reason>`. Rejected `<alternative>` because `<reason>`.\n\n## Hazards\n\n- Do not touch `<surface>` without running `<check>`.\n```\n\nRules:\n\n- store stable facts, decisions, commands, and hazards\n- do not store secrets, private endpoints, local machine paths, raw logs, or copied chat\n- include last-verified dates for facts that can decay\n- remove stale facts instead of appending contradictions\n\n### Report\n\n```markdown\n# Agentic Report: <goal>\n\n## Objective\n\n## Sources\n\n## Tasks\n\n## Changed Files\n\n## Verification\n\n## Review Findings\n\n## Memory Updates\n\n## Residual Risks\n\n## Release Status\n```\n\nRules:\n\n- every completed task has a report entry\n- every failed command is named with the reason it failed or the follow-up owner\n- final status is one of `not ready`, `ready for PR`, `ready to publish`, or `published`\n\n## Assignment Template\n\n```text\nRole: <architect|explorer|implementer|reviewer|qa|memory-curator>\nModel: <model id>\nReasoning: <effort>\nSandbox: <read-only|workspace-write>\nOwned files: <paths or read-only surface>\nTask: <one bounded objective>\nAcceptance: <observable done condition>\nVerification: <command or check>\nOutput: <required sections>\nConstraints: do not inspect secrets; do not broaden scope; do not edit outside ownership.\n```\n\n## Isolation Policy\n\n- Read-only agents may share a checkout.\n- Writer agents should use separate branches or worktrees when parallel edits are possible.\n- If writers share a checkout, file ownership must be disjoint and recorded in the task ledger.\n- CI failures and review comments go back to the owner of the task that introduced the change.\n- The parent resolves conflicts and merges, then runs final verification.\n\n## Stop Conditions\n\nStop and ask the user, or downgrade to a plan-only result, when:\n\n- acceptance criteria cannot be stated\n- required secrets, accounts, paid resources, or private dashboards are unavailable\n- two agents need to edit the same files without a safe order\n- tests cannot run and no credible manual verifier exists\n- security or public-surface scan finds unresolved leaks\n\nFile v0.3.4:agents/openai.yaml\n\ninterface:\n  display_name: \"Agentic Codex Dev\"\n  short_description: \"Codex multi-agent software development system\"\n  default_prompt: \"Use $agentic-codex-dev to plan, implement, verify, review, report, and publish an agentic software development change with explicit roles and a clean GitHub and ClawHub surface.\"\n\npolicy:\n  allow_implicit_invocation: false\n\nArchive v0.3.3: 8 files, 16636 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3728b), references/publish-checklist.md (2089b), references/source-review.md (7182b), references/system-design.md (3616b), SKILL.md (13830b), _meta.json (136b)\n\nFile v0.3.3:SKILL.md\n\n---\nname: agentic-codex-dev\ndescription: Use when planning, implementing, reviewing, coordinating, or publishing agentic software development work with Codex, GitHub, and OpenClaw/ClawHub. Provides a production-grade multi-agent operating loop with role roster, model policy, task ledger, memory ledger, report artifacts, verification gates, and anti-bleed public-surface review.\nversion: 0.3.3\nuser-invocable: true\ndisable-model-invocation: true\nmetadata: {\"openclaw\":{\"homepage\":\"https://github.com/zack-dev-cm/agentic-codex-dev-skill\",\"skillKey\":\"agentic-codex-dev\",\"requires\":{\"bins\":[\"git\",\"clawhub\"],\"anyBins\":[\"python3\",\"python\"]},\"install\":[{\"kind\":\"node\",\"label\":\"Install ClawHub CLI\",\"package\":\"clawhub\",\"bins\":[\"clawhub\"]}],\"tags\":[\"codex\",\"github\",\"clawhub\",\"agentic-development\"]}}\n---\n\n# Agentic Codex Dev\n\nOperate Codex like a disciplined software team: clear goal, explicit roles, scoped ownership, evidence, tests, review, report.\n\n## When to Use\n\nUse this skill for:\n\n- coding tasks where Codex should inspect, modify, test, and report on a GitHub repo\n- turning a rough product or bug request into scoped implementation work\n- setting up repo-local `AGENTS.md`, `.codex/agents/`, or skill instructions\n- reviewing agent-generated code for correctness, tests, security, and public-surface leaks\n- preparing a GitHub repo or ClawHub skill for open-source publication\n- coordinating explicit parallel/subagent work with role ownership and integration control\n\nDo not use it for one-line answers, pure brainstorming, or tasks that only need a command output.\n\n## Runtime Requirements\n\nClawHub requirement metadata for this skill declares `git`, `python3`, and `clawhub`, following the ClawHub skill metadata format at <https://github.com/openclaw/clawhub/blob/main/docs/skill-format.md>.\n\n- Local plan, review, and implementation modes may work with the tools already available in the host.\n- Verification and publish modes expect the declared binaries plus optional Python modules such as `antirot` and `codex_harness`.\n- This skill should not request, print, or store credentials. GitHub and ClawHub publishing must use existing local authenticated CLI sessions, or the user must authenticate manually outside the prompt.\n- Do not run `git push`, `clawhub publish`, or other remote-changing commands unless the user asked for publish or remote update work.\n\n## Core Loop\n\n1. Restate the goal and name the verification step before editing.\n2. Read the repo map: `AGENTS.md`, README, package config, tests, and the files closest to the task.\n3. Define concrete success criteria that would let a reviewer say \"done\".\n4. Make the narrowest defensible change. Match local style. Avoid speculative abstractions.\n5. Run the highest-signal local check. Add a focused smoke test when behavior changed.\n6. Review the diff for bugs, regressions, secrets, private paths, and public-surface bleed.\n7. Report what changed, how it was verified, and any residual risk.\n\nIf the task is unclear, stop early and name the ambiguity. Prefer one precise question over guessing.\n\n## Operating Rules\n\n- Treat repository files as the source of truth. If knowledge matters later, put it in repo docs.\n- Keep `AGENTS.md` short. Use it as an index to durable docs, not a giant prompt.\n- Prefer boring, inspectable code over opaque magic. Agents compound what they can read.\n- Touch only files required for the goal. Mention unrelated problems; do not fix them unless asked.\n- Use structured APIs, tests, and parsers where available. Avoid fragile string tricks.\n- Convert repeated review feedback into checks, docs, or templates.\n- Keep logs and long command output out of the main narrative; summarize the signal.\n- Avoid asking an agent to read undeclared secret files or sync credentials as part of a skill.\n\n## Scope Modes\n\nPick the mode that fits the risk:\n\n- **Patch**: one bug or one focused feature. Read close code, edit, test, review.\n- **Plan**: ambiguous or multi-file work. Write a short acceptance plan before editing.\n- **Review**: findings first, with emphasis on correctness, regressions, security, tests, and leaks as summarized in [source review](references/source-review.md).\n- **Harness**: improve repo legibility: docs, CI, local scripts, custom agents, or audit gates.\n- **Evolve**: metric-driven optimization. One variable per experiment, fixed budget, log keep/discard.\n- **Publish**: GitHub/ClawHub release readiness, metadata, license, docs, and verification.\n- **Multi-Agent**: explicit role roster, task ledger, isolation plan, review gates, memory update, and final report.\n\nPrefer Patch unless the task shows it needs more structure. Use Multi-Agent only when the user explicitly asks for subagents, delegation, or parallel agent work.\n\n## System Design\n\nFor non-trivial or multi-agent work, set up a control plane before coding:\n\n- **Orchestrator**: the main Codex thread owns requirements, task split, agent selection, integration, final review, and user communication.\n- **Role agents**: subagents are optional workers with declared purpose, model, reasoning effort, sandbox, file ownership, and output schema.\n- **Artifacts**: use repo-local ledgers so work survives context loss and can be reviewed without private chat history.\n- **Isolation**: prefer branches or worktrees per writer when multiple agents edit. If one checkout is shared, assign disjoint file ownership.\n- **Gates**: no task is done until its acceptance criteria, verification command, diff review, public-surface scan, and report entry are complete.\n\nWhen this structure is overkill, keep a solo Patch flow and still preserve the same verification discipline.\n\n## Task, Memory, and Report Ledgers\n\nCreate or update these artifacts when work is multi-agent, multi-turn, risky, or intended for publication:\n\n- `docs/agentic/tasks.md`: task id, owner role, goal, owned files, status, acceptance criteria, verification, result, blocker.\n- `docs/agentic/memory.md`: stable repo facts, architecture decisions, commands that actually work, hazards, rejected approaches, last-verified date. Do not store secrets, tokens, private paths, or raw logs.\n- `docs/agentic/reports/<date>-<slug>.md`: final objective, source links, task outcomes, changed files, tests, review findings, unresolved risks, release or PR status.\n\nIf the target repo already has equivalent docs, use the local convention instead of inventing new paths.\n\n## Role Roster\n\nUse this roster as the default multi-agent team. The parent thread stays responsible for coordination and final judgment.\n\n| Role | Default model | Reasoning | Scope | Required output |\n| --- | --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` for critical design/release, `high` otherwise | Owns task split, integration, report | plan, assignments, final decision |\n| Analyst | `gpt-5.4` | `high` | Turns vague request into requirements and risks | assumptions, open questions, acceptance criteria |\n| Architect | `gpt-5.4` | `xhigh` | System design, boundaries, dependency choices | design note, rejected options, invariants |\n| Planner | `gpt-5.4` | `high` | Breaks design into ordered tasks | task ledger rows with owners and gates |\n| Explorer | `gpt-5.4-mini` or `gpt-5.3-codex-spark` | `medium` | Read-only code mapping and evidence gathering | files, symbols, execution path, uncertainty |\n| Implementer | `gpt-5.4` for risky code, `gpt-5.3-codex-spark` for bounded edits | `high` or `medium` | Writes only owned files | patch summary, tests, residual risks |\n| Reviewer | `gpt-5.4` | `xhigh` | Correctness, security, regressions, tests, public surface | findings first, file/line evidence, verdict |\n| QA/CI Analyst | `gpt-5.4` | `high` | Reproduction, failing checks, browser or CLI evidence | exact command, observed failure, fix owner |\n| Memory Curator | `gpt-5.4-mini` | `medium` | Updates durable docs after decisions land | memory entries, stale entries removed |\n\n## Subagents\n\nOnly use subagents when the user explicitly asks for subagents, delegation, or parallel agent work.\n\nGood delegation targets:\n\n- read-heavy codebase mapping\n- independent test or CI-log analysis\n- independent review categories such as security, test gaps, or docs correctness\n- disjoint implementation slices with clearly separate file ownership\n\nBad delegation targets:\n\n- the immediate blocker for your next local step\n- tightly coupled edits in the same files\n- vague \"go improve the code\" work\n- recursive fan-out with no cap\n\nWhen delegating, give each agent a bounded task, a clear output shape, and explicit ownership. Keep the main thread focused on requirements, decisions, integration, and final review. Keep `agents.max_depth = 1` unless the user explicitly accepts recursive delegation risk; this matches the Codex subagent configuration surface documented at <https://developers.openai.com/codex/subagents>.\n\nDelegation prompt shape:\n\n```text\nRole: reviewer\nModel: gpt-5.4\nReasoning: xhigh\nOwnership: read-only review of <files or branch>\nTask: find correctness, security, regression, test, and public-surface risks.\nOutput: findings first with file/line evidence, then open questions, then verdict.\nDo not edit files. Do not inspect secrets. Do not broaden scope.\n```\n\n## Model Policy\n\n- Use `gpt-5.4` with `xhigh` reasoning for architecture, security review, release decisions, and ambiguous multi-agent coordination; Codex custom-agent examples document `gpt-5.4` reviewer roles at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4` with `high` reasoning for implementation where correctness or cross-module behavior matters; model selection follows the Codex custom-agent configuration surface at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4-mini` or `gpt-5.3-codex-spark` for read-only exploration, docs checks, and bounded cleanup where speed matters and the output will be reviewed; both model families appear in Codex custom-agent examples at <https://developers.openai.com/codex/subagents>.\n- Do not use a budget model for final architecture, security, or publish verdicts.\n- Use extra compute selectively: best-of-N, independent reviewer passes, or verifier checks only when the decision is expensive to reverse; optillm documents inference-time scaling techniques at <https://github.com/algorithmicsuperintelligence/optillm>.\n\n## Implementation Discipline\n\nBefore editing:\n\n- inspect the existing patterns\n- identify the likely tests or smoke command\n- check dirty git state and avoid touching unrelated user changes\n- state the planned edit in one or two sentences\n\nWhile editing:\n\n- keep the diff surgical\n- add tests when behavior, contracts, or public output changes\n- avoid new dependencies unless they clearly reduce risk or complexity\n- keep comments rare and useful\n\nAfter editing:\n\n- run the named verification\n- inspect the diff, not just test output\n- update docs only when user-facing behavior or workflow changed\n- do not call work published until the public surface is clean\n\n## Review Checklist\n\nReview every non-trivial result for:\n\n- Does every changed line trace to the stated goal?\n- Are edge cases covered by tests or a clear smoke path?\n- Did the change preserve existing public APIs and CLI behavior?\n- Did docs/examples drift from actual behavior?\n- Did any secret-like string, local path, private URL, copied dashboard, or stale release note enter the repo?\n- Did the final diff remove avoidable complexity from the first draft, as recommended in [source review](references/source-review.md)?\n\n## Consistency and Effectiveness Gates\n\nFor multi-agent work, verify the process itself:\n\n- Every task has an owner, owned files, acceptance criteria, verification command, and result.\n- Every subagent output is mapped to a task or explicitly discarded with a reason.\n- No writer agent edits outside its assigned ownership without parent approval.\n- At least one reviewer pass is read-only and independent of the implementer.\n- The final report names changed files, commands run, failed checks, source links, residual risk, and release status.\n- Memory updates contain stable facts only; do not store raw chat, secrets, local credentials, or transient logs.\n- If a metric-driven change is attempted, record baseline, candidate, verifier, result, and keep/discard decision.\n\n## Real Example Eval\n\nFor a serious workflow eval, run this skill against a real repo task and archive the result in the report ledger. A valid eval has:\n\n- baseline repo state and user goal\n- role roster used, including model and reasoning choices\n- task ledger rows with owners and file boundaries\n- at least one implementation or review task with verification output\n- public-release check for private-path examples, local-only URLs, secret-shaped placeholders, and stale claims\n- final report with changed files, tests, residual risks, and follow-up blockers\n\nUse [example run](references/example-run.md) as the minimum acceptance shape.\n\n## GitHub and ClawHub Publish Gate\n\nBefore publishing:\n\n- README or skill summary says what it does, when to use it, and what it does not do.\n- License is compatible with the target surface. ClawHub publishes skills under MIT-0.\n- `SKILL.md` has frontmatter `name`, `description`, and `version`.\n- The skill folder contains only text-based files needed at runtime.\n- No hidden install scripts, credential readers, service restarts, or local machine assumptions.\n- Public repo has security, contribution, support, CI, and release/audit checks when applicable.\n- Run the repo's public-surface gate before pushing or publishing to a registry.\n\nFor this skill's source analysis, read `references/source-review.md` and `references/comparison-matrix.md`.\nFor multi-agent artifacts and templates, read `references/system-design.md`.\nFor release commands and manual checks, read `references/publish-checklist.md`.\n\nFile v0.3.3:_meta.json\n\n{\n  \"ownerId\": \"kn7dhjt1k1f111whp13fmrqwnh81tn1v\",\n  \"slug\": \"agentic-codex-dev\",\n  \"version\": \"0.3.3\",\n  \"publishedAt\": 1777516543636\n}\n\nFile v0.3.3:references/comparison-matrix.md\n\n# Comparison Matrix\n\nReviewed on 2026-04-22. Use this matrix when judging whether the skill is strong enough for multi-agent software development rather than a generic coding checklist.\n\n| Source | Strong pattern | Risk if copied blindly | Skill response |\n| --- | --- | --- | --- |\n| OpenAI Codex subagents | Custom roles with model, reasoning, sandbox, and developer instructions | More agents can add cost and coordination failure | Subagents require explicit user intent, role ownership, and parent integration |\n| OpenAI Codex concepts | Context isolation and parallel investigation | Delegation can hide the critical path | Parent keeps immediate blockers local and uses agents for side work |\n| OpenAI harness engineering | Repo-local memory, docs, tests, and legible state | Prose policy can drift without checks | Task, memory, report ledgers plus public-surface tests |\n| openai/symphony | Isolated implementation runs | Workflow engine complexity may exceed need | Use branches/worktrees for writers; do not require a daemon |\n| karpathy/autoresearch | Baseline, budget, one metric, keep/discard/crash log | Research framing may not fit product work | Apply evaluator-first discipline only when optimizing behavior |\n| forrestchang/andrej-karpathy-skills | Assumptions, tradeoffs, surgical diffs | Too little structure for multi-agent runs | Keep the concise style, add role roster and ledgers only for serious work |\n| openevolve | Reproducible evaluator-first evolution, Pareto tradeoffs, cascade checks | Claims can outrun evidence | Require baseline, verifier, result log, and keep/discard decision |\n| optillm | Inference-time scaling, verifier passes, multi-agent reasoning | Expensive compute can become default theater | Reserve `gpt-5.4`/`xhigh` and independent reviewers for hard decisions |\n| agent-orchestrator | Worktree isolation, CI/review feedback routed to owner | Autonomous fleets need heavy operations | Adopt ownership and feedback routing without requiring its platform |\n| gstack | Clear role taxonomy and dispatch tiers | Full role stack for every task slows direct fixes | Scope modes decide when to use Patch vs Multi-Agent |\n| paperclip | Goals, budgets, heartbeats, org chart, audit trail | Continuous autonomy can run away | Add owner, budget/risk thinking, stop conditions, and reports |\n| openclaw | Local-first assistant, skills, sandbox and channel safety | Broad host access is unnecessary for a methodology skill | Keep bundle instruction-only and public-surface clean |\n| rdudov/agents | Analyst, architect, planner, developer, reviewer boundaries | Rigid phase gates can create process drag | Use roles when the task is ambiguous, risky, or parallel |\n\n## Previous Version Weak Points\n\nVersion 0.1.2 was useful as a publish-clean patch loop, but it was not enough for the stated goal of multi-agent software development:\n\n- no explicit role roster with model and reasoning policy\n- no default use of `gpt-5.4`/`xhigh` for architecture, review, and release decisions\n- no system design for orchestration, isolation, ledgers, or reports\n- no task ledger, memory ledger, or report artifact\n- no real-run eval shape\n- no process consistency checks that tie agent outputs to tasks and verification\n- no test ratchet for role/model/system-design coverage\n- bleed scan omitted Python files\n\n## Target Bar\n\nA release is acceptable only when a reviewer can see:\n\n- the role roster and model policy in `SKILL.md`\n- source-by-source comparison in this file\n- system design templates in `references/system-design.md`\n- real-run acceptance shape in `references/example-run.md`\n- automated tests that fail if the public skill regresses to a generic patch loop\n- anti-bleed coverage that includes Python test and helper files\n\nFile v0.3.3:references/example-run.md\n\n# Example Run\n\nThis is the minimum shape for a real eval of the skill. It is intentionally concrete enough to review and general enough to run on any public repository.\n\n## Scenario\n\nGoal: upgrade an instruction-only Codex skill from a generic patch workflow into a multi-agent software-development operating loop.\n\nAcceptance:\n\n- `SKILL.md` includes system design, role roster, model policy, task ledger, memory ledger, report ledger, consistency gates, and real-run eval guidance.\n- Source comparison covers all cited public projects.\n- Custom agent configs declare models and reasoning effort.\n- Public-surface tests fail if role/model/system-design coverage is removed.\n- Anti-bleed scan includes Python files.\n- Local verification passes, with any CI or publish blocker named explicitly.\n\n## Role Roster Used\n\n| Role | Model | Reasoning | Purpose |\n| --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` | own requirements, source synthesis, implementation, final report |\n| Architect | `gpt-5.4` | `xhigh` | validate system design and role boundaries |\n| Implementer | `gpt-5.4` | `high` | edit skill, references, docs, tests, metadata |\n| Reviewer | `gpt-5.4` | `xhigh` | check correctness, public surface, test coverage, publish risk |\n| Memory Curator | `gpt-5.4-mini` | `medium` | update durable docs after verification |\n\n## Task Ledger Sample\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | `SKILL.md`, `references/system-design.md` | done | system design and role roster are explicit | public-surface tests | pending until local run | none |\n| T2 | implementer | `tests/test_public_surface.py` | done | Python files are scanned for bleed | unit tests | pending until local run | none |\n| T3 | reviewer | public repo surface | review | no private paths, tokens, local URLs, stale version claims | anti-bleed and AntiRot | pending until local run | CI workflow depends on token scope |\n\n## Memory Ledger Sample\n\nStable facts:\n\n- Runtime skill files are text-only and publish through ClawHub.\n- `.clawhubignore` excludes repo harness files from the public skill bundle.\n- GitHub Actions workflow creation may require token scope outside the normal repo push permission.\n\nDecisions:\n\n- Keep the skill instruction-only. Add templates and tests rather than a daemon.\n- Use `gpt-5.4`/`xhigh` for architecture and reviewer roles; use faster models only for reviewed, read-only work.\n\nHazards:\n\n- Do not claim CI enforcement unless the workflow exists in the remote repository.\n- Do not let public examples include private paths, local URLs, tokens, or copied chat logs.\n\n## Report Sample\n\nObjective: release version 0.3.1 with multi-agent operating structure, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned real-run evidence.\n\nSources: OpenAI Codex subagents, OpenAI harness engineering, optillm, openevolve, autoresearch, symphony, paperclip, gstack, OpenClaw, Andrej Karpathy skills, agent-orchestrator, rdudov agents.\n\nChanged files:\n\n- `SKILL.md`\n- `references/source-review.md`\n- `references/comparison-matrix.md`\n- `references/system-design.md`\n- `references/example-run.md`\n- `.codex/agents/*.toml`\n- `tests/test_public_surface.py`\n\nVerification:\n\n- `python3 -m unittest discover -s tests`\n- `python3 -m antirot.cli lint SKILL.md --strict`\n- `python3 -m codex_harness audit . --strict --min-score 90`\n- `clawhub inspect agentic-codex-dev --files` after publish\n\nResidual risks:\n\n- CI cannot be considered enforced until the remote repository has a workflow.\n- Any future source review must re-check the public URLs because repository behavior can change.\n\nFile v0.3.3:references/publish-checklist.md\n\n# Publish Checklist\n\nUse this when preparing the skill for GitHub and ClawHub.\n\n## Local Review\n\n1. Confirm `SKILL.md` frontmatter:\n   - `name: agentic-codex-dev`\n   - `description: ...`\n   - `version: 0.3.1`\n2. Confirm the folder name is the intended ClawHub slug: `agentic-codex-dev`.\n3. Confirm every file is text-based and needed:\n   - `SKILL.md`\n   - `agents/openai.yaml`\n   - `references/source-review.md`\n   - `references/comparison-matrix.md`\n   - `references/system-design.md`\n   - `references/example-run.md`\n   - `references/publish-checklist.md`\n4. Search for private paths, local URLs, tokens, and copied private notes.\n5. Run the repository gate:\n\n```bash\npython3 -m unittest discover -s tests\npython3 -m antirot.cli lint SKILL.md --strict\npython3 -m codex_harness audit . --strict --min-score 90\n```\n\n## GitHub Publish\n\nFrom the repository root:\n\n```bash\ngit status --short\ngit add .\ngit commit -m \"Upgrade agentic Codex development skill\"\ngit push\n```\n\nIf the worktree contains unrelated user changes, stage only the files above.\n\nDo not claim GitHub CI enforcement unless `.github/workflows/ci.yml` exists on the remote default branch. If push tokens cannot create workflows, record that as a release blocker or create the workflow through the GitHub UI before claiming enforcement.\n\n## ClawHub Publish\n\nThe ClawHub CLI must be installed and logged in:\n\n```bash\nclawhub whoami\n```\n\nPublish the skill:\n\n```bash\nclawhub publish . --version 0.3.1\n```\n\nAfter publishing:\n\n```bash\nclawhub inspect agentic-codex-dev --files\n```\n\nCheck that the listing shows the expected files, summary, version, and homepage. Remember that ClawHub publishes skills under MIT-0.\n\n## Manual Acceptance\n\nThe skill is publish-ready when:\n\n- a reviewer can understand the runtime behavior by reading `SKILL.md` alone\n- the source review explains why each major rule exists\n- no command in the skill installs software, reads credentials, restarts services, or changes global agent state\n- the repository tests and audit gate pass\n- the ClawHub listing, if published, points back to the GitHub source\n\nFile v0.3.3:references/source-review.md\n\n# Source Review\n\nReviewed on 2026-04-22. This file distills the cited public projects into operating rules for a general Codex software-development skill. It keeps transferable engineering structure and avoids copying project-specific product machinery.\n\n## Core Synthesis\n\nThe strongest pattern is not \"spawn more agents.\" The strongest pattern is a controlled software-development system:\n\n1. Make the target repo legible through `AGENTS.md`, docs, tests, and visible logs.\n2. Convert the user goal into acceptance criteria, task ownership, and verification commands.\n3. Assign roles only when a role changes the quality bar or parallelism.\n4. Isolate writers through branches, worktrees, or disjoint file ownership.\n5. Use strong models for design, review, and release decisions.\n6. Preserve durable state in repo-local task, memory, and report ledgers.\n7. Keep only changes that pass tests, review, and public-surface gates.\n\n## Source Notes\n\n### OpenAI Codex Subagents\n\nSources: <https://developers.openai.com/codex/subagents>, <https://developers.openai.com/codex/concepts/subagents>\n\n- Keep: custom agents should declare role, model, reasoning effort, sandbox, and developer instructions.\n- Keep: strong reviewer examples use `gpt-5.4`; docs examples also use `gpt-5.4-mini` and `gpt-5.3-codex-spark` for read-only or bounded tasks.\n- Keep: subagents help most with context isolation, read-heavy exploration, independent review, tests, and bounded implementation slices.\n- Keep: depth and thread caps matter because recursive delegation can increase cost and unpredictability.\n- Avoid: implicit fan-out. The parent thread must own task split, integration, and final judgment.\n\n### OpenAI Harness Engineering\n\nSource: <https://openai.com/index/harness-engineering/>\n\n- Keep: repo knowledge should be versioned and inspectable by agents.\n- Keep: `AGENTS.md` should be a map, with durable detail moved into docs and tests.\n- Keep: legibility is an engineering feature: state, logs, commands, and metrics should be available without private chat context.\n- Keep: taste and architecture need mechanical enforcement through tests, lint rules, boundaries, and release gates.\n- Avoid: prose-only policy when a repeated issue can become a check.\n\n### openai/symphony\n\nSource: <https://github.com/openai/symphony>\n\n- Keep: project work should become isolated autonomous implementation runs.\n- Keep: the operator manages work and evidence, not agent chatter.\n- Keep: runtime status, proof of work, retries, and handoff states should be visible.\n- Avoid: shared mutable workspaces for parallel writers unless ownership is explicit.\n\n### karpathy/autoresearch\n\nSource: <https://github.com/karpathy/autoresearch>\n\n- Keep: constrain research with one editable surface, a fixed budget, and one primary metric.\n- Keep: run a baseline first.\n- Keep: log every experiment as keep, discard, or crash.\n- Keep: equal metric results should prefer simpler code and fewer moving parts.\n- Avoid: letting exploratory logs flood the main context; store logs separately and summarize the signal.\n\n### forrestchang/andrej-karpathy-skills\n\nSource: <https://github.com/forrestchang/andrej-karpathy-skills>\n\n- Keep: state assumptions, tradeoffs, and success criteria before coding.\n- Keep: use minimum necessary code, local style, and surgical diffs.\n- Keep: simple tasks do not need the full process, but non-trivial tasks need rigor.\n- Avoid: speculative flexibility, adjacent refactors, and hidden confusion.\n\n### algorithmicsuperintelligence/openevolve\n\nSource: <https://github.com/algorithmicsuperintelligence/openevolve>\n\n- Keep: evaluator-first development. The evaluator defines truth.\n- Keep: reproducibility, seeded runs, and component isolation for experiments.\n- Keep: multi-objective scoring when correctness, performance, complexity, and memory all matter.\n- Keep: cascade evaluation to reject bad candidates before expensive checks.\n- Avoid: \"AI discovered it\" claims unless the run, seed, evaluator, and result log are reproducible.\n\n### algorithmicsuperintelligence/optillm\n\nSource: <https://github.com/algorithmicsuperintelligence/optillm>\n\n- Keep: inference-time scaling can improve hard decisions through best-of-N, self-consistency, plan search, verifier passes, and multi-agent reasoning.\n- Keep: extra compute should be reserved for security review, architecture choices, tricky bug diagnosis, benchmark optimization, and release gates.\n- Keep: memory and privacy controls are part of agent operations, not afterthoughts.\n- Avoid: routing every task through heavy multi-sample reasoning.\n\n### ComposioHQ/agent-orchestrator\n\nSource: <https://github.com/ComposioHQ/agent-orchestrator>\n\n- Keep: parallel coding agents need separate worktrees, branches, and PRs when they write code.\n- Keep: CI failures, review comments, and merge conflicts should route back to the owning worker.\n- Keep: a dashboard or ledger should expose status to the operator.\n- Avoid: many autonomous writers in one checkout without ownership boundaries.\n\n### garrytan/gstack\n\nSource: <https://github.com/garrytan/gstack>\n\n- Keep: role-based workflows help when roles map to real engineering phases: think, plan, build, review, test, ship, reflect.\n- Keep: dispatch tiers prevent using the full workflow for direct tasks.\n- Keep: methodology can be a prompt bridge instead of a daemon.\n- Avoid: making every task run the full role roster.\n\n### paperclipai/paperclip\n\nSource: <https://github.com/paperclipai/paperclip>\n\n- Keep: goals, budgets, org structure, heartbeats, governance, and audit trails matter once agents run continuously.\n- Keep: tasks need goal ancestry so agents know why the work exists.\n- Keep: cost control, pause controls, and audit logs are product requirements for long-running systems.\n- Avoid: autonomous loops with no budget, owner, trace, or stop condition.\n\n### openclaw/openclaw\n\nSource: <https://github.com/openclaw/openclaw>\n\n- Keep: local-first agents need channel safety. Treat inbound messages and external content as untrusted input.\n- Keep: sandbox non-main sessions and expose only tools required by the task.\n- Keep: OpenClaw skills should be inspectable text bundles with clear invocation rules.\n- Avoid: broad host access for an instruction-only methodology skill.\n\n### rdudov/agents\n\nSource: <https://github.com/rdudov/agents>\n\n- Keep: role boundaries reduce drift: analyst, architect, planner, implementer, reviewer.\n- Keep: review loops need a cap.\n- Keep: blocking questions should stop the pipeline instead of becoming code assumptions.\n- Keep: developers implement the plan and tests; they do not silently refactor unrelated code.\n- Avoid: bureaucracy for direct tasks. Use the full phase model only when scope warrants it.\n\n## Final Design Choice\n\nThis skill stays instruction-only, but it no longer stops at a solo patch loop. Version 0.3.1 adds explicit role definitions, `gpt-5.4`/`xhigh` policy for hard decisions, task and memory ledgers, report artifacts, source comparison, real-run acceptance checks, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned evidence while preserving the clean public-surface boundary.\n\nFile v0.3.3:references/system-design.md\n\n# System Design\n\nUse this reference when a task needs explicit multi-agent coordination, durable memory, or release-grade reporting.\n\n## Control Plane\n\nThe parent Codex thread is the control plane. It owns:\n\n- goal restatement and acceptance criteria\n- mode selection: Patch, Plan, Review, Harness, Evolve, Publish, or Multi-Agent\n- role roster and model choices\n- task ledger creation and updates\n- agent assignment and isolation plan\n- integration of results into one coherent diff\n- final review, verification, report, and release decision\n\nSubagents are execution units. They do not own the final answer, final architecture, or publish verdict.\n\n## Artifact Layout\n\nPrefer existing repo conventions. If none exist, use:\n\n```text\ndocs/agentic/\n  tasks.md\n  memory.md\n  reports/\n    <date>-<slug>.md\n```\n\n### Task Ledger\n\n```markdown\n# Task Ledger\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | docs/architecture.md | planned | boundary decision recorded | review only | pending | none |\n```\n\nRules:\n\n- one owner per row\n- writable files must be named before implementation\n- status is one of `planned`, `active`, `blocked`, `review`, `done`, `discarded`\n- verification must be a command, manual check, or reviewer gate\n- discarded work needs a reason\n\n### Memory Ledger\n\n```markdown\n# Memory Ledger\n\n## Stable Facts\n\n- The public API entry point is `<symbol>`; last verified on `<date>` with `<command>`.\n\n## Decisions\n\n- Use `<approach>` because `<reason>`. Rejected `<alternative>` because `<reason>`.\n\n## Hazards\n\n- Do not touch `<surface>` without running `<check>`.\n```\n\nRules:\n\n- store stable facts, decisions, commands, and hazards\n- do not store secrets, private endpoints, local machine paths, raw logs, or copied chat\n- include last-verified dates for facts that can decay\n- remove stale facts instead of appending contradictions\n\n### Report\n\n```markdown\n# Agentic Report: <goal>\n\n## Objective\n\n## Sources\n\n## Tasks\n\n## Changed Files\n\n## Verification\n\n## Review Findings\n\n## Memory Updates\n\n## Residual Risks\n\n## Release Status\n```\n\nRules:\n\n- every completed task has a report entry\n- every failed command is named with the reason it failed or the follow-up owner\n- final status is one of `not ready`, `ready for PR`, `ready to publish`, or `published`\n\n## Assignment Template\n\n```text\nRole: <architect|explorer|implementer|reviewer|qa|memory-curator>\nModel: <model id>\nReasoning: <effort>\nSandbox: <read-only|workspace-write>\nOwned files: <paths or read-only surface>\nTask: <one bounded objective>\nAcceptance: <observable done condition>\nVerification: <command or check>\nOutput: <required sections>\nConstraints: do not inspect secrets; do not broaden scope; do not edit outside ownership.\n```\n\n## Isolation Policy\n\n- Read-only agents may share a checkout.\n- Writer agents should use separate branches or worktrees when parallel edits are possible.\n- If writers share a checkout, file ownership must be disjoint and recorded in the task ledger.\n- CI failures and review comments go back to the owner of the task that introduced the change.\n- The parent resolves conflicts and merges, then runs final verification.\n\n## Stop Conditions\n\nStop and ask the user, or downgrade to a plan-only result, when:\n\n- acceptance criteria cannot be stated\n- required secrets, accounts, paid resources, or private dashboards are unavailable\n- two agents need to edit the same files without a safe order\n- tests cannot run and no credible manual verifier exists\n- security or public-surface scan finds unresolved leaks\n\nFile v0.3.3:agents/openai.yaml\n\ninterface:\n  display_name: \"Agentic Codex Dev\"\n  short_description: \"Codex multi-agent software development system\"\n  default_prompt: \"Use $agentic-codex-dev to plan, implement, verify, review, report, and publish an agentic software development change with explicit roles and a clean GitHub and ClawHub surface.\"\n\npolicy:\n  allow_implicit_invocation: false\n\nArchive v0.3.2: 8 files, 16616 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3728b), references/publish-checklist.md (2089b), references/source-review.md (7182b), references/system-design.md (3616b), SKILL.md (13796b), _meta.json (136b)\n\nFile v0.3.2:SKILL.md\n\n---\nname: agentic-codex-dev\ndescription: Use when planning, implementing, reviewing, coordinating, or publishing agentic software development work with Codex, GitHub, and OpenClaw/ClawHub. Provides a production-grade multi-agent operating loop with role roster, model policy, task ledger, memory ledger, report artifacts, verification gates, and anti-bleed public-surface review.\nversion: 0.3.2\nuser-invocable: true\ndisable-model-invocation: true\nmetadata: {\"openclaw\":{\"homepage\":\"https://github.com/zack-dev-cm/agentic-codex-dev-skill\",\"skillKey\":\"agentic-codex-dev\",\"requires\":{\"bins\":[\"git\",\"clawhub\"],\"anyBins\":[\"python3\",\"python\"]},\"install\":[{\"kind\":\"node\",\"label\":\"Install ClawHub CLI\",\"package\":\"clawhub\",\"bins\":[\"clawhub\"]}],\"tags\":[\"codex\",\"github\",\"clawhub\",\"agentic-development\"]}}\n---\n\n# Agentic Codex Dev\n\nOperate Codex like a disciplined software team: clear goal, explicit roles, scoped ownership, evidence, tests, review, report.\n\n## When to Use\n\nUse this skill for:\n\n- coding tasks where Codex should inspect, modify, test, and report on a GitHub repo\n- turning a rough product or bug request into scoped implementation work\n- setting up repo-local `AGENTS.md`, `.codex/agents/`, or skill instructions\n- reviewing agent-generated code for correctness, tests, security, and public-surface leaks\n- preparing a GitHub repo or ClawHub skill for open-source publication\n- coordinating explicit parallel/subagent work with role ownership and integration control\n\nDo not use it for one-line answers, pure brainstorming, or tasks that only need a command output.\n\n## Runtime Requirements\n\nClawHub requirement metadata for this skill declares `git`, `python3`, and `clawhub`, following the ClawHub skill metadata format at <https://github.com/openclaw/clawhub/blob/main/docs/skill-format.md>.\n\n- Local plan, review, and implementation modes may work with the tools already available in the host.\n- Verification and publish modes expect the declared binaries plus optional Python modules such as `antirot` and `codex_harness`.\n- This skill should not request, print, or store credentials. GitHub and ClawHub publishing must use existing local authenticated CLI sessions, or the user must authenticate manually outside the prompt.\n- Do not run `git push`, `clawhub publish`, or other remote-changing commands unless the user asked for publish or remote update work.\n\n## Core Loop\n\n1. Restate the goal and name the verification step before editing.\n2. Read the repo map: `AGENTS.md`, README, package config, tests, and the files closest to the task.\n3. Define concrete success criteria that would let a reviewer say \"done\".\n4. Make the narrowest defensible change. Match local style. Avoid speculative abstractions.\n5. Run the highest-signal local check. Add a focused smoke test when behavior changed.\n6. Review the diff for bugs, regressions, secrets, private paths, and public-surface bleed.\n7. Report what changed, how it was verified, and any residual risk.\n\nIf the task is unclear, stop early and name the ambiguity. Prefer one precise question over guessing.\n\n## Operating Rules\n\n- Treat repository files as the source of truth. If knowledge matters later, put it in repo docs.\n- Keep `AGENTS.md` short. Use it as an index to durable docs, not a giant prompt.\n- Prefer boring, inspectable code over opaque magic. Agents compound what they can read.\n- Touch only files required for the goal. Mention unrelated problems; do not fix them unless asked.\n- Use structured APIs, tests, and parsers where available. Avoid fragile string tricks.\n- Convert repeated review feedback into checks, docs, or templates.\n- Keep logs and long command output out of the main narrative; summarize the signal.\n- Avoid asking an agent to read undeclared secret files or sync credentials as part of a skill.\n\n## Scope Modes\n\nPick the mode that fits the risk:\n\n- **Patch**: one bug or one focused feature. Read close code, edit, test, review.\n- **Plan**: ambiguous or multi-file work. Write a short acceptance plan before editing.\n- **Review**: findings first, with emphasis on correctness, regressions, security, tests, and leaks as summarized in [source review](references/source-review.md).\n- **Harness**: improve repo legibility: docs, CI, local scripts, custom agents, or audit gates.\n- **Evolve**: metric-driven optimization. One variable per experiment, fixed budget, log keep/discard.\n- **Publish**: GitHub/ClawHub release readiness, metadata, license, docs, and verification.\n- **Multi-Agent**: explicit role roster, task ledger, isolation plan, review gates, memory update, and final report.\n\nPrefer Patch unless the task shows it needs more structure. Use Multi-Agent only when the user explicitly asks for subagents, delegation, or parallel agent work.\n\n## System Design\n\nFor non-trivial or multi-agent work, set up a control plane before coding:\n\n- **Orchestrator**: the main Codex thread owns requirements, task split, agent selection, integration, final review, and user communication.\n- **Role agents**: subagents are optional workers with declared purpose, model, reasoning effort, sandbox, file ownership, and output schema.\n- **Artifacts**: use repo-local ledgers so work survives context loss and can be reviewed without private chat history.\n- **Isolation**: prefer branches or worktrees per writer when multiple agents edit. If one checkout is shared, assign disjoint file ownership.\n- **Gates**: no task is done until its acceptance criteria, verification command, diff review, public-surface scan, and report entry are complete.\n\nWhen this structure is overkill, keep a solo Patch flow and still preserve the same verification discipline.\n\n## Task, Memory, and Report Ledgers\n\nCreate or update these artifacts when work is multi-agent, multi-turn, risky, or intended for publication:\n\n- `docs/agentic/tasks.md`: task id, owner role, goal, owned files, status, acceptance criteria, verification, result, blocker.\n- `docs/agentic/memory.md`: stable repo facts, architecture decisions, commands that actually work, hazards, rejected approaches, last-verified date. Do not store secrets, tokens, private paths, or raw logs.\n- `docs/agentic/reports/<date>-<slug>.md`: final objective, source links, task outcomes, changed files, tests, review findings, unresolved risks, release or PR status.\n\nIf the target repo already has equivalent docs, use the local convention instead of inventing new paths.\n\n## Role Roster\n\nUse this roster as the default multi-agent team. The parent thread stays responsible for coordination and final judgment.\n\n| Role | Default model | Reasoning | Scope | Required output |\n| --- | --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` for critical design/release, `high` otherwise | Owns task split, integration, report | plan, assignments, final decision |\n| Analyst | `gpt-5.4` | `high` | Turns vague request into requirements and risks | assumptions, open questions, acceptance criteria |\n| Architect | `gpt-5.4` | `xhigh` | System design, boundaries, dependency choices | design note, rejected options, invariants |\n| Planner | `gpt-5.4` | `high` | Breaks design into ordered tasks | task ledger rows with owners and gates |\n| Explorer | `gpt-5.4-mini` or `gpt-5.3-codex-spark` | `medium` | Read-only code mapping and evidence gathering | files, symbols, execution path, uncertainty |\n| Implementer | `gpt-5.4` for risky code, `gpt-5.3-codex-spark` for bounded edits | `high` or `medium` | Writes only owned files | patch summary, tests, residual risks |\n| Reviewer | `gpt-5.4` | `xhigh` | Correctness, security, regressions, tests, public surface | findings first, file/line evidence, verdict |\n| QA/CI Analyst | `gpt-5.4` | `high` | Reproduction, failing checks, browser or CLI evidence | exact command, observed failure, fix owner |\n| Memory Curator | `gpt-5.4-mini` | `medium` | Updates durable docs after decisions land | memory entries, stale entries removed |\n\n## Subagents\n\nOnly use subagents when the user explicitly asks for subagents, delegation, or parallel agent work.\n\nGood delegation targets:\n\n- read-heavy codebase mapping\n- independent test or CI-log analysis\n- independent review categories such as security, test gaps, or docs correctness\n- disjoint implementation slices with clearly separate file ownership\n\nBad delegation targets:\n\n- the immediate blocker for your next local step\n- tightly coupled edits in the same files\n- vague \"go improve the code\" work\n- recursive fan-out with no cap\n\nWhen delegating, give each agent a bounded task, a clear output shape, and explicit ownership. Keep the main thread focused on requirements, decisions, integration, and final review. Keep `agents.max_depth = 1` unless the user explicitly accepts recursive delegation risk; this matches the Codex subagent configuration surface documented at <https://developers.openai.com/codex/subagents>.\n\nDelegation prompt shape:\n\n```text\nRole: reviewer\nModel: gpt-5.4\nReasoning: xhigh\nOwnership: read-only review of <files or branch>\nTask: find correctness, security, regression, test, and public-surface risks.\nOutput: findings first with file/line evidence, then open questions, then verdict.\nDo not edit files. Do not inspect secrets. Do not broaden scope.\n```\n\n## Model Policy\n\n- Use `gpt-5.4` with `xhigh` reasoning for architecture, security review, release decisions, and ambiguous multi-agent coordination; Codex custom-agent examples document `gpt-5.4` reviewer roles at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4` with `high` reasoning for implementation where correctness or cross-module behavior matters; model selection follows the Codex custom-agent configuration surface at <https://developers.openai.com/codex/subagents>.\n- Use `gpt-5.4-mini` or `gpt-5.3-codex-spark` for read-only exploration, docs checks, and bounded cleanup where speed matters and the output will be reviewed; both model families appear in Codex custom-agent examples at <https://developers.openai.com/codex/subagents>.\n- Do not use a budget model for final architecture, security, or publish verdicts.\n- Use extra compute selectively: best-of-N, independent reviewer passes, or verifier checks only when the decision is expensive to reverse; optillm documents inference-time scaling techniques at <https://github.com/algorithmicsuperintelligence/optillm>.\n\n## Implementation Discipline\n\nBefore editing:\n\n- inspect the existing patterns\n- identify the likely tests or smoke command\n- check dirty git state and avoid touching unrelated user changes\n- state the planned edit in one or two sentences\n\nWhile editing:\n\n- keep the diff surgical\n- add tests when behavior, contracts, or public output changes\n- avoid new dependencies unless they clearly reduce risk or complexity\n- keep comments rare and useful\n\nAfter editing:\n\n- run the named verification\n- inspect the diff, not just test output\n- update docs only when user-facing behavior or workflow changed\n- do not call work published until the public surface is clean\n\n## Review Checklist\n\nReview every non-trivial result for:\n\n- Does every changed line trace to the stated goal?\n- Are edge cases covered by tests or a clear smoke path?\n- Did the change preserve existing public APIs and CLI behavior?\n- Did docs/examples drift from actual behavior?\n- Did any secret-like string, local path, private URL, copied dashboard, or stale release note enter the repo?\n- Did the final diff remove avoidable complexity from the first draft, as recommended in [source review](references/source-review.md)?\n\n## Consistency and Effectiveness Gates\n\nFor multi-agent work, verify the process itself:\n\n- Every task has an owner, owned files, acceptance criteria, verification command, and result.\n- Every subagent output is mapped to a task or explicitly discarded with a reason.\n- No writer agent edits outside its assigned ownership without parent approval.\n- At least one reviewer pass is read-only and independent of the implementer.\n- The final report names changed files, commands run, failed checks, source links, residual risk, and release status.\n- Memory updates contain stable facts only; do not store raw chat, secrets, local credentials, or transient logs.\n- If a metric-driven change is attempted, record baseline, candidate, verifier, result, and keep/discard decision.\n\n## Real Example Eval\n\nFor a serious workflow eval, run this skill against a real repo task and archive the result in the report ledger. A valid eval has:\n\n- baseline repo state and user goal\n- role roster used, including model and reasoning choices\n- task ledger rows with owners and file boundaries\n- at least one implementation or review task with verification output\n- public-surface scan for private paths, local URLs, tokens, and stale claims\n- final report with changed files, tests, residual risks, and follow-up blockers\n\nUse [example run](references/example-run.md) as the minimum acceptance shape.\n\n## GitHub and ClawHub Publish Gate\n\nBefore publishing:\n\n- README or skill summary says what it does, when to use it, and what it does not do.\n- License is compatible with the target surface. ClawHub publishes skills under MIT-0.\n- `SKILL.md` has frontmatter `name`, `description`, and `version`.\n- The skill folder contains only text-based files needed at runtime.\n- No hidden install scripts, credential readers, service restarts, or local machine assumptions.\n- Public repo has security, contribution, support, CI, and release/audit checks when applicable.\n- Run the repo's public-surface gate before pushing or publishing to a registry.\n\nFor this skill's source analysis, read `references/source-review.md` and `references/comparison-matrix.md`.\nFor multi-agent artifacts and templates, read `references/system-design.md`.\nFor release commands and manual checks, read `references/publish-checklist.md`.\n\nFile v0.3.2:_meta.json\n\n{\n  \"ownerId\": \"kn7dhjt1k1f111whp13fmrqwnh81tn1v\",\n  \"slug\": \"agentic-codex-dev\",\n  \"version\": \"0.3.2\",\n  \"publishedAt\": 1777139239125\n}\n\nFile v0.3.2:references/comparison-matrix.md\n\n# Comparison Matrix\n\nReviewed on 2026-04-22. Use this matrix when judging whether the skill is strong enough for multi-agent software development rather than a generic coding checklist.\n\n| Source | Strong pattern | Risk if copied blindly | Skill response |\n| --- | --- | --- | --- |\n| OpenAI Codex subagents | Custom roles with model, reasoning, sandbox, and developer instructions | More agents can add cost and coordination failure | Subagents require explicit user intent, role ownership, and parent integration |\n| OpenAI Codex concepts | Context isolation and parallel investigation | Delegation can hide the critical path | Parent keeps immediate blockers local and uses agents for side work |\n| OpenAI harness engineering | Repo-local memory, docs, tests, and legible state | Prose policy can drift without checks | Task, memory, report ledgers plus public-surface tests |\n| openai/symphony | Isolated implementation runs | Workflow engine complexity may exceed need | Use branches/worktrees for writers; do not require a daemon |\n| karpathy/autoresearch | Baseline, budget, one metric, keep/discard/crash log | Research framing may not fit product work | Apply evaluator-first discipline only when optimizing behavior |\n| forrestchang/andrej-karpathy-skills | Assumptions, tradeoffs, surgical diffs | Too little structure for multi-agent runs | Keep the concise style, add role roster and ledgers only for serious work |\n| openevolve | Reproducible evaluator-first evolution, Pareto tradeoffs, cascade checks | Claims can outrun evidence | Require baseline, verifier, result log, and keep/discard decision |\n| optillm | Inference-time scaling, verifier passes, multi-agent reasoning | Expensive compute can become default theater | Reserve `gpt-5.4`/`xhigh` and independent reviewers for hard decisions |\n| agent-orchestrator | Worktree isolation, CI/review feedback routed to owner | Autonomous fleets need heavy operations | Adopt ownership and feedback routing without requiring its platform |\n| gstack | Clear role taxonomy and dispatch tiers | Full role stack for every task slows direct fixes | Scope modes decide when to use Patch vs Multi-Agent |\n| paperclip | Goals, budgets, heartbeats, org chart, audit trail | Continuous autonomy can run away | Add owner, budget/risk thinking, stop conditions, and reports |\n| openclaw | Local-first assistant, skills, sandbox and channel safety | Broad host access is unnecessary for a methodology skill | Keep bundle instruction-only and public-surface clean |\n| rdudov/agents | Analyst, architect, planner, developer, reviewer boundaries | Rigid phase gates can create process drag | Use roles when the task is ambiguous, risky, or parallel |\n\n## Previous Version Weak Points\n\nVersion 0.1.2 was useful as a publish-clean patch loop, but it was not enough for the stated goal of multi-agent software development:\n\n- no explicit role roster with model and reasoning policy\n- no default use of `gpt-5.4`/`xhigh` for architecture, review, and release decisions\n- no system design for orchestration, isolation, ledgers, or reports\n- no task ledger, memory ledger, or report artifact\n- no real-run eval shape\n- no process consistency checks that tie agent outputs to tasks and verification\n- no test ratchet for role/model/system-design coverage\n- bleed scan omitted Python files\n\n## Target Bar\n\nA release is acceptable only when a reviewer can see:\n\n- the role roster and model policy in `SKILL.md`\n- source-by-source comparison in this file\n- system design templates in `references/system-design.md`\n- real-run acceptance shape in `references/example-run.md`\n- automated tests that fail if the public skill regresses to a generic patch loop\n- anti-bleed coverage that includes Python test and helper files\n\nFile v0.3.2:references/example-run.md\n\n# Example Run\n\nThis is the minimum shape for a real eval of the skill. It is intentionally concrete enough to review and general enough to run on any public repository.\n\n## Scenario\n\nGoal: upgrade an instruction-only Codex skill from a generic patch workflow into a multi-agent software-development operating loop.\n\nAcceptance:\n\n- `SKILL.md` includes system design, role roster, model policy, task ledger, memory ledger, report ledger, consistency gates, and real-run eval guidance.\n- Source comparison covers all cited public projects.\n- Custom agent configs declare models and reasoning effort.\n- Public-surface tests fail if role/model/system-design coverage is removed.\n- Anti-bleed scan includes Python files.\n- Local verification passes, with any CI or publish blocker named explicitly.\n\n## Role Roster Used\n\n| Role | Model | Reasoning | Purpose |\n| --- | --- | --- | --- |\n| Orchestrator | `gpt-5.4` | `xhigh` | own requirements, source synthesis, implementation, final report |\n| Architect | `gpt-5.4` | `xhigh` | validate system design and role boundaries |\n| Implementer | `gpt-5.4` | `high` | edit skill, references, docs, tests, metadata |\n| Reviewer | `gpt-5.4` | `xhigh` | check correctness, public surface, test coverage, publish risk |\n| Memory Curator | `gpt-5.4-mini` | `medium` | update durable docs after verification |\n\n## Task Ledger Sample\n\n| ID | Owner | Files | Status | Acceptance | Verification | Result | Blocker |\n| --- | --- | --- | --- | --- | --- | --- | --- |\n| T1 | architect | `SKILL.md`, `references/system-design.md` | done | system design and role roster are explicit | public-surface tests | pending until local run | none |\n| T2 | implementer | `tests/test_public_surface.py` | done | Python files are scanned for bleed | unit tests | pending until local run | none |\n| T3 | reviewer | public repo surface | review | no private paths, tokens, local URLs, stale version claims | anti-bleed and AntiRot | pending until local run | CI workflow depends on token scope |\n\n## Memory Ledger Sample\n\nStable facts:\n\n- Runtime skill files are text-only and publish through ClawHub.\n- `.clawhubignore` excludes repo harness files from the public skill bundle.\n- GitHub Actions workflow creation may require token scope outside the normal repo push permission.\n\nDecisions:\n\n- Keep the skill instruction-only. Add templates and tests rather than a daemon.\n- Use `gpt-5.4`/`xhigh` for architecture and reviewer roles; use faster models only for reviewed, read-only work.\n\nHazards:\n\n- Do not claim CI enforcement unless the workflow exists in the remote repository.\n- Do not let public examples include private paths, local URLs, tokens, or copied chat logs.\n\n## Report Sample\n\nObjective: release version 0.3.1 with multi-agent operating structure, declared runtime requirements, explicit invocation, disabled model auto-invocation, and repo-owned real-run evidence.\n\nSources: OpenAI Codex subagents, OpenAI harness engineering, optillm, openevolve, autoresearch, symphony, paperclip, gstack, OpenClaw, Andrej Karpathy skills, agent-orchestrator, rdudov agents.\n\nChanged files:\n\n- `SKILL.md`\n- `references/source-review.md`\n- `references/comparison-matrix.md`\n- `references/system-design.md`\n- `references/example-run.md`\n- `.codex/agents/*.toml`\n- `tests/test_public_surface.py`\n\nVerification:\n\n- `python3 -m unittest discover -s tests`\n- `python3 -m antirot.cli lint SKILL.md --strict`\n- `python3 -m codex_harness audit . --strict --min-score 90`\n- `clawhub inspect agentic-codex-dev --files` after publish\n\nResidual risks:\n\n- CI cannot be considered enforced until the remote repository has a workflow.\n- Any future source review must re-check the public URLs because repository behavior can change.\n\nFile v0.3.2:references/publish-checklist.md\n\n# Publish Checklist\n\nUse this when preparing the skill for GitHub and ClawHub.\n\n## Local Review\n\n1. Confirm `SKILL.md` frontmatter:\n   - `name: agentic-codex-dev`\n   - `description: ...`\n   - `version: 0.3.1`\n2. Confirm the folder name is the intended ClawHub slug: `agentic-codex-dev`.\n3. Confirm every file is text-based and needed:\n   - `SKILL.md`\n   - `agents/openai.yaml`\n   - `references/source-review.md`\n   - `references/comparison-matrix.md`\n   - `references/system-design.md`\n   - `references/example-run.md`\n   - `references/publish-checklist.md`\n4. Search for private paths, local URLs, tokens, and copied private notes.\n5. Run the repository gate:\n\n```bash\npython3 -m unittest discover -s tests\npython3 -m antirot.cli lint SKILL.md --strict\npython3 -m codex_harness audit . --strict --min-score 90\n```\n\n## GitHub Publish\n\nFrom the repository root:\n\n```bash\ngit status --short\ngit add .\ngit commit -m \"Upgrade agentic Codex development skill\"\ngit push\n```\n\nIf the worktree contains unrelated user changes, stage only the files above.\n\nDo not claim GitHub CI enforcement unless `.github/workflows/ci.yml` exists on the remote default branch. If push tokens cannot create workflows, record that as a release blocker or create the workflow through the GitHub UI before claiming enforcement.\n\n## ClawHub Publish\n\nThe ClawHub CLI must be installed and logged in:\n\n```bash\nclawhub whoami\n```\n\nPublish the skill:\n\n```bash\nclawhub publish . --version 0.3.1\n```\n\nAfter publishing:\n\n```bash\nclawhub inspect agentic-codex-dev --files\n```\n\nCheck that the listing shows the expected files, summary, version, and homepage. Remember that ClawHub publishes skills under MIT-0.\n\n## Manual Acceptance\n\nThe skill is publish-ready when:\n\n- a reviewer can understand the runtime behavior by reading `SKILL.md` alone\n- the source review explains why each major rule exists\n- no command in the skill installs software, reads credentials, restarts services, or changes global agent state\n- the repository tests and audit gate pass\n- the ClawHub listing, if published, points back to the GitHub source\n\nFile v0.3.2:references/source-review.md\n\n# Source Review\n\nReviewed on 2026-04-22. This file distills the cited public projects into operating rules for a general Codex software-development skill. It keeps transferable engineering structure and avoids copying project-specific product machinery.\n\n## Core Synthesis\n\nThe strongest pattern is not \"spawn more agents.\" The strongest pattern is a controlled software-development system:\n\n1. Make the target repo legible through `AGENTS.md`, docs, tests, and visible logs.\n2. Convert the user goal into acceptance criteria, task ownership, and verification commands.\n3. Assign roles only when a role changes the quality bar or parallelism.\n4. Isolate writers through branches, worktrees, or disjoint file ownership.\n5. Use strong models for design, review, and release decisions.\n6. Preserve durable state in repo-local task, memory, and report ledgers.\n7. Keep only changes that pass tests, review, and public-surface gates.\n\n## Source Notes\n\n### OpenAI Codex Subagents\n\nSources: <https://developers.openai.com/codex/subagents>, <https://developers.openai.com/codex/concepts/subagents>\n\n- Keep: custom agents should declare role, model, reasoning effort, sandbox, and developer instructions.\n- Keep: strong reviewer examples use `gpt-5.4`; docs examples also use `gpt-5.4-mini` and `gpt-5.3-codex-spark` for read-only or bounded tasks.\n- Keep: subagents help most with context isolation, read-heavy exploration, independent review, tests, and bounded implementation slices.\n- Keep: depth and thread caps matter because recursive delegation can increase cost and unpredictability.\n- Avoid: implicit fan-out. The parent thread must own task split, integration, and final judgment.\n\n### OpenAI Harness Engineering\n\nSource: <https://openai.com/index/harness-engineering/>\n\n- Keep: repo knowledge should be versioned and inspectable by agents.\n- Keep: `AGENTS.md` should be a map, with durable detail moved into docs and tests.\n- Keep: legibility is an engineering feature: state, logs, commands, and metrics should be available without private chat context.\n- Keep: taste and architecture need mechanical enforcement through tests, lint rules, boundaries, and release gates.\n- Avoid: prose-only policy when a repeated issue can become a check.\n\n### openai/symphony\n\nSource: <https://github.com/openai/symphony>\n\n- Keep: project work should become isolated autonomous implementation runs.\n- Keep: the operator manages work and evidence, not agent chatter.\n- Keep: runtime status, proof of work, retries, and handoff states should be visible.\n- Avoid: shared mutable workspaces for parallel writers unless ownership is explicit.\n\n### karpathy/autoresearch\n\nSource: <https://github.com/karpathy/autoresearch>\n\n- Keep: constrain research with one editable surface, a fixed budget, and one primary metric.\n- Keep: run a baseline first.\n- Keep: log every experiment as keep, discard, or crash.\n- Keep: equal metric results should prefer simpler code and fewer moving parts.\n- Avoid: letting explorato\n\nArchive v0.3.1: 8 files, 16546 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3728b), references/publish-checklist.md (2089b), references/source-review.md (7182b), references/system-design.md (3616b), SKILL.md (13647b), _meta.json (136b)\n\nArchive v0.3.0: 8 files, 16494 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3696b), references/publish-checklist.md (2089b), references/source-review.md (7150b), references/system-design.md (3616b), SKILL.md (13595b), _meta.json (136b)\n\nArchive v0.2.2: 8 files, 16471 bytes\n\nFiles: agents/openai.yaml (358b), references/comparison-matrix.md (3759b), references/example-run.md (3666b), references/publish-checklist.md (2089b), references/source-review.md (7129b), references/system-design.md (3616b), SKILL.md (13595b), _meta.json (136b)\n\nArchive v0.2.1: 8 files, 16454 bytes\n\nFiles: agents/openai.yaml (357b), references/comparison-matrix.md (3759b), references/example-run.md (3644b), references/publish-checklist.md (2089b), references/source-review.md (7108b), references/system-design.md (3616b), SKILL.md (13595b), _meta.json (136b)\n\nArchive v0.2.0: 8 files, 16098 bytes\n\nFiles: agents/openai.yaml (357b), references/comparison-matrix.md (3759b), references/example-run.md (3610b), references/publish-checklist.md (2089b), references/source-review.md (7077b), references/system-design.md (3616b), SKILL.md (12770b), _meta.json (136b)","readmeExcerpt":"Skill: Agentic Codex Dev Reviewer Owner: zack-dev-cm Summary: Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple... Tags: agentic-development:0.3.5, clawhub:0.3.6, codex:0.3.6, github:0.3.6, latest:0.3.6, multi-agent:0.3.4, review:0.3.6 Version history: v0.3.6 | 2026-05-15T15:35:08.460Z | user Republish public p","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"Role: reviewer\nModel: gpt-5.4\nReasoning: xhigh\nOwnership: read-only review of <files or branch>\nTask: find correctness, security, regression, test, and public-surface risks.\nOutput: findings first with file/line evidence, then open questions, then verdict.\nDo not edit files. Do not inspect secrets. Do not broaden scope."},{"language":"bash","snippet":"python3 -m unittest discover -s tests\npython3 -m antirot.cli lint SKILL.md --strict\npython3 -m codex_harness audit . --strict --min-score 90"},{"language":"bash","snippet":"git status --short\ngit add .\ngit commit -m \"Upgrade agentic Codex development skill\"\ngit push"},{"language":"bash","snippet":"clawhub whoami"},{"language":"bash","snippet":"clawhub publish . --version 0.3.1"},{"language":"bash","snippet":"clawhub inspect agentic-codex-dev --files"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"SKILL.md","content":"---\nname: agentic-codex-dev\ndescription: Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, implementation review, public-surface checks, or release evidence review without running remote-changing commands.\n---\n\n# Agentic Codex Dev\n\nUse this skill as a text-only review layer for agentic development work. It helps turn a request, diff, repository note, or release checklist into a scoped plan and a conservative readiness verdict.\n\n## Review Workflow\n\n1. Restate the requested outcome and the smallest safe scope.\n2. Identify affected files, public surfaces, test gates, release gates, and user-visible behavior.\n3. Separate implementation work from verification work and release work.\n4. Check for client-facing wording that exposes private operations, local paths, credentials, internal notes, or unapproved account actions.\n5. Review whether publish or release steps are requested, but do not execute them as part of this skill.\n6. Return a clear verdict: `ready`, `ready_with_notes`, `blocked`, or `do_not_ship`.\n\n## Boundaries\n\n- Do not request, print, store, or infer credentials.\n- Do not stage, commit, push, publish, delete, hide, or transfer anything.\n- Do not assume logged-in GitHub, ClawHub, browser, or other account authority.\n- Do not create persistent project memory, ledgers, or reports unless the user separately asks for a file artifact.\n- Keep public-surface advice focused on wording, scope, tests, and release evidence.\n\n## Output Shape\n\nReturn:\n\n- `Scope`: what is being reviewed.\n- `Findings`: concrete issues ordered by severity.\n- `Public surface`: wording or packaging risks.\n- `Verification`: tests or checks that should pass before release.\n- `Verdict`: one of `ready`, `ready_with_notes`, `blocked`, or `do_not_ship`."},{"path":"_meta.json","content":"{\n  \"ownerId\": \"kn7dhjt1k1f111whp13fmrqwnh81tn1v\",\n  \"slug\": \"agentic-codex-dev\",\n  \"version\": \"0.3.6\",\n  \"publishedAt\": 1778859308460\n}"},{"path":"skill-card.md","content":"## Description:\n\nReview agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, implementation review, public-surface checks, or release evidence review without running remote-changing commands.\n\nThis skill is ready for commercial/non-commercial use.\n\n## Publisher:\n\n[zack-dev-cm](https://clawhub.ai/user/zack-dev-cm)\n\n### License/Terms of Use:\n\nMIT-0\n\n## Use Case:\n\nDevelopers and release reviewers use this skill to review Codex, GitHub, and ClawHub delivery plans, diffs, release checklists, public-facing wording, verification evidence, and readiness before publication or remote-changing work.\n\n### Deployment Geography for Use:\n\nGlobal\n\n## Known Risks and Mitigations:\n\nRisk: Readiness verdicts may be mistaken for authorization to publish, push, or run account-changing commands.\n\nMitigation: Treat verdicts as advisory and require normal human approval, test gates, and release controls before remote-changing actions.\n\n## Reference(s):\n\n- [ClawHub skill page](https://clawhub.ai/zack-dev-cm/skills/agentic-codex-dev)\n\n## Skill Output:\n\n**Output Type(s):** [Analysis, Guidance, Markdown]\n\n**Output Format:** [Markdown sections with findings, public-surface review, verification checks, and a readiness verdict]\n\n**Output Parameters:** [1D]\n\n**Other Properties Related to Output:** [Verdict is one of ready, ready_with_notes, blocked, or do_not_ship.]\n\n## Skill Version(s):\n\n0.3.6 (source: server release metadata)\n\n## Ethical Considerations:\n\nUsers should evaluate whether this skill is appropriate for their environment, review any generated or modified files before relying on them, and apply their organization's safety, security, and compliance requirements before deployment."},{"path":"agents/openai.yaml","content":"interface:\n  display_name: \"Agentic Codex Dev Reviewer\"\n  short_description: \"Review scoped Codex/GitHub/ClawHub delivery readiness.\"\n  default_prompt: \"Use $agentic-codex-dev to review this agentic development plan, diff, or release checklist for scope, public surface, verification evidence, and release readiness.\""}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple... Skill: Agentic Codex Dev Reviewer Owner: zack-dev-cm Summary: Review agentic software-development plans and release readiness for Codex, GitHub, and ClawHub work. Use when a user asks for scoped delivery planning, imple... Tags: agentic-development:0.3.5, clawhub:0.3.6, codex:0.3.6, github:0.3.6, latest:0.3.6, multi-agent:0.3.4, review:0.3.6 Version history: v0.3.6 | 2026-05-15T15:35:08.460Z | user Republish public p","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1024,"uniquenessScore":49,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-10T16:47:13.453Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T21:39:01.921Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}