{"id":"cf3b2935-9fe1-4121-ad88-40862f9686c3","entityType":"agent","slug":"clawhub-optim-agent-optim-agent","name":"Optim Agent","canonicalUrl":"https://www.xpersona.co/agent/clawhub-optim-agent-optim-agent","canonicalPath":"/agent/clawhub-optim-agent-optim-agent","generatedAt":"2026-10-10T06:42:22.891Z","source":"CLAWHUB","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":null},"description":"Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat... Skill: Optim Agent Owner: optim-agent Summary: Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat... Tags: latest:0.1.0 Version history: v0.1.0 | 2026-07-24T10:29:51.279Z | auto - Initial release of optim-agent skill for optimizing configurable system parameters against a measurable scalar objective. - Provides","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. 3K downloads reported by the source. Last updated 10/9/2026.","installCommand":"clawhub skill install s17d81bx0d0wzzwqq29rwtranh8ag1x0:optim-agent","sourceUrl":"https://clawhub.ai/optim-agent/optim-agent","homepage":"https://clawhub.ai/optim-agent/skills/optim-agent","primaryLinks":[{"label":"View on ClawHub","url":"https://clawhub.ai/optim-agent/optim-agent","kind":"source"},{"label":"Homepage","url":"https://clawhub.ai/optim-agent/skills/optim-agent","kind":"homepage"}],"safetyScore":84,"overallRank":62,"popularityScore":70,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat..."},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[],"verifiedCount":0,"selfDeclaredCount":1,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile"}},"adoption":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":null},"stars":null,"forks":null,"downloads":2992,"packageName":null,"latestVersion":"0.1.0","tractionLabel":"3K downloads"},"release":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"medium","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T10:21:03.344Z","lastCrawledAt":"2026-10-09T10:21:03.344Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-10T10:21:03.344Z","lastVerifiedAt":null,"highlights":[{"version":"0.1.0","createdAt":"2026-07-24T10:29:51.279Z","changelog":"- Initial release of optim-agent skill for optimizing configurable system parameters against a measurable scalar objective. - Provides a detailed workflow for configuring, running, and recording optimization trials, including ask/tell API usage. - Designed for coding agents with project and shell access; supports model training, scientific workflows, and black-box evaluations. - Introduces best practices for baseline measurement, recovery, artifact storage, and reproducibility. - Includes clear rules to ensure auditable, unbiased, and safe optimization.","fileCount":259,"zipByteSize":2021206}]},"execution":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"clawhub skill install s17d81bx0d0wzzwqq29rwtranh8ag1x0:optim-agent","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"CLAWHUB","generatedAt":"2026-10-10T06:42:22.890Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/clawhub-optim-agent-optim-agent/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"CLAWHUB","verified":false,"confidence":"high","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":null},"readme":"Skill: Optim Agent\n\nOwner: optim-agent\n\nSummary: Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat...\n\nTags: latest:0.1.0\n\nVersion history:\n\nv0.1.0 | 2026-07-24T10:29:51.279Z | auto\n\n- Initial release of optim-agent skill for optimizing configurable system parameters against a measurable scalar objective.\n- Provides a detailed workflow for configuring, running, and recording optimization trials, including ask/tell API usage.\n- Designed for coding agents with project and shell access; supports model training, scientific workflows, and black-box evaluations.\n- Introduces best practices for baseline measurement, recovery, artifact storage, and reproducibility.\n- Includes clear rules to ensure auditable, unbiased, and safe optimization.\n\nArchive index:\n\nArchive v0.1.0: 259 files, 2021206 bytes\n\nFiles: .agents (0b), .agents/plugins (0b), .agents/plugins/marketplace.json (374b), .claude-plugin (0b), .claude-plugin/marketplace.json (438b), .github (0b), .github/dependabot.yml (269b), .github/ISSUE_TEMPLATE (0b), .github/ISSUE_TEMPLATE/bug_report.md (279b), .github/ISSUE_TEMPLATE/feature_request.md (193b), .github/pull_request_template.md (199b), .github/workflows (0b), .github/workflows/ci.yml (1202b), .github/workflows/docs.yml (710b), .gitignore (709b), .pre-commit-config.yaml (262b), benchmarks (0b), benchmarks/manifest.json (2125b), benchmarks/README.md (3789b), CHANGELOG.md (1196b), CITATION.cff (630b), CODE_OF_CONDUCT.md (1013b), CONTRIBUTING.md (1631b), docs (0b), docs/assets (0b), docs/assets/cifar10_curves_GPT-5.5-medium_s0.json (10094b), docs/assets/cifar10_curves_GPT-5.5-medium_s1.json (10018b), docs/assets/cifar10_curves_GPT-5.5-medium_s2.json (10061b), docs/assets/cifar10_curves_GPT-5.5-medium_s3.json (10098b), docs/assets/cifar10_curves_GPT-5.5-medium_s4.json (10005b), docs/assets/cifar10_curves_GPT-5.5-medium-no-context_s0.json (10787b), docs/assets/cifar10_curves_GPT-5.5-medium-no-context_s1.json (10726b), docs/assets/cifar10_curves_GPT-5.5-medium-no-context_s2.json (10652b), docs/assets/cifar10_curves_GPT-5.5-medium-no-context_s3.json (10785b), docs/assets/cifar10_curves_GPT-5.5-medium-no-context_s4.json (10919b), docs/assets/cifar10_curves_Random_s0.json (11022b), docs/assets/cifar10_curves_Random_s1.json (11017b), docs/assets/cifar10_curves_Random_s2.json (11077b), docs/assets/cifar10_curves_Random_s3.json (10965b), docs/assets/cifar10_curves_Random_s4.json (11124b), docs/assets/cifar10_curves_TPE_s0.json (10965b), docs/assets/cifar10_curves_TPE_s1.json (11082b), docs/assets/cifar10_curves_TPE_s2.json (11082b), docs/assets/cifar10_curves_TPE_s3.json (10939b), docs/assets/cifar10_curves_TPE_s4.json (11097b), docs/assets/classification_benchmarks.png (104836b), docs/assets/credit_card.png (95446b), docs/assets/credit_default_GP-BO_s0.json (7115b), docs/assets/credit_default_GP-BO_s1.json (7117b), docs/assets/credit_default_GP-BO_s2.json (7104b), docs/assets/credit_default_GP-BO_s3.json (7093b), docs/assets/credit_default_GP-BO_s4.json (7098b), docs/assets/credit_default_GPT-5.5_s0.json (6700b), docs/assets/credit_default_GPT-5.5_s1.json (6768b), docs/assets/credit_default_GPT-5.5_s2.json (6772b), docs/assets/credit_default_GPT-5.5_s3.json (6655b), docs/assets/credit_default_GPT-5.5_s4.json (6724b), docs/assets/credit_default_GPT-5.5-no-context_s0.json (6161b), docs/assets/credit_default_GPT-5.5-no-context_s1.json (6211b), docs/assets/credit_default_GPT-5.5-no-context_s2.json (6280b), docs/assets/credit_default_GPT-5.5-no-context_s3.json (6075b), docs/assets/credit_default_GPT-5.5-no-context_s4.json (6167b), docs/assets/credit_default_Random_s0.json (6862b), docs/assets/credit_default_Random_s1.json (6863b), docs/assets/credit_default_Random_s2.json (6862b), docs/assets/credit_default_Random_s3.json (6881b), docs/assets/credit_default_Random_s4.json (6859b), docs/assets/credit_default_TPE_s0.json (6868b), docs/assets/credit_default_TPE_s1.json (6889b), docs/assets/credit_default_TPE_s2.json (6871b), docs/assets/credit_default_TPE_s3.json (6865b), docs/assets/credit_default_TPE_s4.json (6847b), docs/assets/hard_benchmarks_free.png (130006b), docs/assets/hard_benchmarks_tier.png (152272b), docs/assets/hard_curves_Big-pickle_s0.json (2468b), docs/assets/hard_curves_Big-pickle_s1.json (2495b), docs/assets/hard_curves_Big-pickle_s2.json (2440b), docs/assets/hard_curves_Big-pickle_s3.json (2454b), docs/assets/hard_curves_Big-pickle_s4.json (2504b), docs/assets/hard_curves_DeepSeek-V4-Flash_s0.json (2470b)\n\nFile v0.1.0:plugins/optim-agent/SKILL.md\n\n---\nname: optim-agent\ndescription: Use when optimizing configurable system parameters against a measurable scalar objective.\n---\n\n# optim-agent\n\nRead and follow the canonical optim-agent workflow in `../../SKILL.md`.\nResolve that path from this file's directory before beginning the optimization.\n\nFile v0.1.0:plugins/optim-agent/skills/optim-agent/SKILL.md\n\n---\nname: optim-agent\ndescription: Use when optimizing configurable system parameters against a measurable scalar objective.\n---\n\n# optim-agent\n\nRead and follow the canonical optim-agent workflow in `../../../../SKILL.md`.\nResolve that path from this file's directory before beginning the optimization.\n\nFile v0.1.0:SKILL.md\n\n---\nname: optim-agent\ndescription: Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitative strategies, reinforcement learning, scientific workflows, or other expensive black-box evaluations where reading the project can improve trial selection.\n---\n\n# optim-agent\n\nAct as the sampler inside any coding-agent session: Claude Code, Codex,\nOpenCode/OpenClaw, or another agent that can read project files and run shell\ncommands. Read the project to understand parameter meaning and interactions,\npropose one configuration, run the real evaluator, and record the result\nthrough optim-agent's ask/tell API. Let the measured objective, not the agent's\nintuition, decide what works.\n\n## Load the workflow\n\nUse this file as the operating guide for the active coding agent. In Codex, it\ncan be installed directly from GitHub:\n\n```text\n$skill-installer install https://github.com/Optim-Agent/optim-agent\n```\n\nIn Claude Code, OpenCode/OpenClaw, or another coding-agent environment, place\nthis repository or `SKILL.md` in the agent-visible workspace and ask the agent\nto follow the optim-agent workflow. The workflow does not depend on Codex-only\nAPIs; it needs file access, shell access, and Python.\n\nEnsure the Python package is importable. Choose one source; do not install both:\n\n```bash\n# Stable release from PyPI\npython -m pip install optim-agent\n\n# Latest source from GitHub\npython -m pip install \"optim-agent @ git+https://github.com/Optim-Agent/optim-agent.git\"\n```\n\nFor a reproducible GitHub install, append `@<tag-or-commit>` after `.git`.\n\n## Workflow\n\n1. **Understand the system.** Read the evaluation entry point and every file\n   that defines the target parameters. Record each parameter's type, legal\n   range, semantics, interactions, and operational constraints.\n2. **Define the experiment.** Confirm the scalar objective, `minimize` or\n   `maximize`, trial budget, evaluation command, runtime/cost limit, and fixed\n   workload or seed. For multiple metrics or hard constraints, agree on one\n   scalar feasibility or penalty rule before running trials.\n3. **Establish a baseline.** Evaluate the current/default configuration with the\n   same command and environment used for every later trial.\n4. **Initialize or resume.** Keep artifacts in the repository's ignored\n   `.optim-agent-runs/` directory:\n\n   ```bash\n   if git rev-parse --git-dir >/dev/null 2>&1 && ! git check-ignore -q .optim-agent-runs/; then\n     printf '/.optim-agent-runs/\\n' >> \"$(git rev-parse --git-path info/exclude)\"\n   fi\n   ```\n\n   ```python\n   from pathlib import Path\n   import optim_agent as oa\n\n   run_dir = Path(\".optim-agent-runs\")\n   run_dir.mkdir(exist_ok=True)\n   study = oa.create_study(\n       direction=\"minimize\",\n       storage=run_dir / \"skill-study.json\",\n       seed=0,\n   )\n   print([(t.params, t.value, t.state) for t in study.trials])\n   ```\n\n5. **Run one informed trial.** Choose parameters from code understanding and all\n   completed history, then use explicit ask/tell:\n\n   ```python\n   params = {\"threshold\": 0.72, \"budget\": 80}\n   trial = study.ask(params)\n   try:\n       value = evaluate_system(**trial.params)\n   except Exception:\n       study.tell(trial, state=\"failed\")\n       raise\n   else:\n       study.tell(trial, value)\n   ```\n\n   For a deliberately stopped trial, report the latest valid intermediate\n   metric first, then call `study.tell(trial, state=\"pruned\")`.\n6. **Select the next point.** Avoid accidental repeats, explore broadly before\n   exploiting, respect bounds and constraints, and treat failed regions as\n   evidence. If the evaluator is noisy, repeat promising configurations under\n   the same workload before declaring a winner.\n7. **Stop and report.** Stop at the approved budget or stopping condition.\n   Report the baseline, best value and parameters, trial count, failed/pruned\n   trials, convergence trend, and exact reproduction command.\n\n## Recovery\n\nJSON storage records a trial when `study.tell` runs. Before launching an\nexpensive external evaluation, save its parameters, command, and output path in\na per-trial directory under `.optim-agent-runs/`. After interruption, inspect\nthat output before rerunning: if a valid result exists, recreate the same point\nwith `study.ask(params)` and record it; otherwise rerun it deliberately.\n\nUse SQLite storage (`skill-study.db`) only when the user explicitly wants\nmultiple processes. Sequential trials are the default because each proposal\nshould use the complete prior history.\n\n## Rules\n\n- Use ask/tell in skill mode; do not delegate proposal selection to\n  `AgentSampler` when the session agent is meant to read and reason over code.\n- Keep evaluation inputs and outputs isolated from production configuration.\n- Never fabricate, infer, or manually improve an objective value.\n- Record crashes as `failed`; record intentional early stops as `pruned`.\n- Preserve the study and trial artifacts so the result is auditable and resumable.\n- Do not tune secrets, credentials, or unbounded parameters.\n\nFile v0.1.0:benchmarks/README.md\n\n# Benchmark contract\n\nThe committed benchmark artifacts support the claims in the README, docs, and\npaper. They are evidence, not decorative assets. Tables and figures must be\ngenerated from the JSON runs under `docs/assets/`.\n\n## Suites\n\n`manifest.json` records the stable suite identifiers, trial budgets, seeds,\nresult globs, and context policy. Individual result files remain authoritative\nfor backend, model, effort, search-space version, objective values, and sampled\nparameters.\n\nThe hard-function comparison uses **no supplied task context**: generic\nparameter names, bounds, and observed trial history only. A model may still\nrecognize a standard function from its bounds or values, so these runs measure\nsmall-budget optimization rather than semantic-context benefit. Classification\nruns provide the explicit with-context versus no-context comparison.\nThe RL-control benchmark is CPU-only and uses Gymnasium Acrobot-v1 and\nLunarLander-v3 with a discretized Q-learning controller. It runs Random, TPE,\nGPT-5.5 with context, and GPT-5.5 without context for 20 trials across seeds\n`0..4`. The GPT-5.5 arms use high modeling effort and the last 5 trials of\nhistory. The winning contextual arm disables explicit reasoning and qualitative\nnotes. It is strongest on both environment means in the committed run.\nThe credit-default benchmark is CPU-only and uses UCI dataset 350, Default of\nCredit Card Clients (CC BY 4.0, DOI `10.24432/C55S3H`). The official archive\nSHA-256, workbook schema, 60/20/20 stratified split, split seed, search space,\nand 20-trial budget are pinned. All five optimizer seeds see the same train and\nvalidation data; the held-out test split is evaluated only for each run's\nvalidation-selected configuration.\n\nRandom, TPE, GP-BO, selected contextual GPT-5.5, and matched GPT-5.5/no-context\nartifacts must all be present. Agent runs are fail-closed. The primary\ntrajectory metric is validation incumbent log loss; held-out test log loss is a\nsecondary generalization check. This is not a production credit-decision system\nand must not be interpreted as one. The test split is reported after selection,\nnot used to select.\n\n## Provenance requirements\n\nEvery agent result intended for publication must identify:\n\n- suite and function or dataset;\n- backend and exact model identifier;\n- agent effort and context policy;\n- trial budget, seed, and search-space version;\n- ordered parameter proposals and objective values; and\n- creation time or source commit when the runner provides it.\n\nBaselines must use the same space, objective, budget, and seeds. A rotating\nhosted model alias must be labeled as such; never silently present it as a\npinned model.\n\n## Publication gate\n\nDo not update prose or plots from a partial seed set. Before publishing:\n\n```bash\npip install -e \".[examples,ml,dev]\"\npytest\npython scripts/verify_classification_cumulative_error.py\npython examples/hard_functions.py selfcheck\npython examples/hard_functions.py plot\npip install -e \".[rl,examples]\"\npython examples/rl_control.py selfcheck\npython examples/rl_control.py summary\npython examples/rl_control.py plot\npython examples/rl_control.py gif\npython examples/credit_card.py selfcheck\npython examples/credit_card.py summary\npython examples/credit_card.py plot\npython scripts/render_trajectory.py\n```\n\nPlotters must reject missing, stale, mixed-budget, or mixed-provenance data.\nKeep exploratory outputs outside `docs/assets/`; only the complete canonical\nrun belongs in the documentation tree.\n\n## Interpreting results\n\nReport distributions or multi-seed means, not a best seed. State whether the\nmetric emphasizes early improvement or final quality. Agent latency and token\ncost are separate from objective-evaluation cost and should be reported when\nthey materially affect the comparison.\n\nFile v0.1.0:README.md\n\n<p align=\"center\">\n  <picture>\n    <source media=\"(prefers-color-scheme: dark)\" srcset=\"docs/assets/optim-agent-logo-dark.svg\">\n    <img alt=\"optim-agent\" src=\"docs/assets/optim-agent-logo-light.svg\" width=\"500\">\n  </picture>\n</p>\n\n<h1 align=\"center\">optim-agent</h1>\n\n<p align=\"center\">\n  <strong>Agentic system optimization with coding agents.</strong><br>\n  Automate the iterative parameter-tuning work of an algorithm engineer.\n</p>\n\n<p align=\"center\">\n  <a href=\"https://github.com/Optim-Agent/optim-agent/stargazers\"><img alt=\"GitHub stars\" src=\"https://img.shields.io/github/stars/Optim-Agent/optim-agent?style=square\"></a>\n  <a href=\"https://pypi.org/project/optim-agent/\"><img alt=\"PyPI\" src=\"https://img.shields.io/pypi/v/optim-agent\"></a>\n  <a href=\"https://pypi.org/project/optim-agent/\"><img alt=\"Python versions\" src=\"https://img.shields.io/pypi/pyversions/optim-agent\"></a>\n  <a href=\"LICENSE\"><img alt=\"License: MIT\" src=\"https://img.shields.io/pypi/l/optim-agent\"></a>\n  <a href=\"https://optim-agent.github.io/optim-agent/\"><img alt=\"Docs\" src=\"https://img.shields.io/badge/docs-online-blue\"></a>\n  <a href=\"https://code.claude.com/docs/en/skills\"><img alt=\"Claude Skill\" src=\"https://img.shields.io/badge/Claude-Skill-D97757?logo=claude&logoColor=white\"></a>\n  <a href=\"https://developers.openai.com/codex/skills\"><img alt=\"Codex Skill\" src=\"https://img.shields.io/badge/Codex-Skill-blue?logo=openai&logoColor=white\"></a>\n</p>\n\n<p align=\"center\">\n  <strong>English</strong> |\n  <a href=\"docs/i18n/README_ZH.md\">简体中文</a> |\n  <a href=\"docs/i18n/README_JA.md\">日本語</a> |\n  <a href=\"docs/i18n/README_KO.md\">한국어</a> |\n  <a href=\"docs/i18n/README_FR.md\">Français</a> |\n  <a href=\"docs/i18n/README_DE.md\">Deutsch</a> |\n  <a href=\"docs/i18n/README_ES.md\">Español</a> |\n  <a href=\"docs/i18n/README_PT.md\">Português</a> |\n  <a href=\"docs/i18n/README_RU.md\">Русский</a>\n</p>\n\noptim-agent lets Claude Code / Codex / OpenCode tune real system parameters by\nreading your code, proposing trials, and recording measured objective results.\nUse it when your system exposes configurable parameters and a measurable objective.\nIt combines what each parameter *means* with what the trial history *shows*,\nthen proposes the next configuration to evaluate. Objective evaluations remain\nauthoritative: optim-agent proposes values, validates them against the declared\nspace, records outcomes, and falls back to safe sampling when an agent reply is\ninvalid.\n\n<p align=\"center\">\n  <img alt=\"optim-agent tuning loop\" src=\"docs/assets/optim-agent-overview.png\" width=\"900\">\n</p>\n\n| Models | Systems | Research |\n|---|---|---|\n| Training, architecture, and RL experiments | Inference, latency, cost, control, and decision rules | Quant signals, simulations, and scientific workflows |\n\n## Why optim-agent\n\n- **Semantic proposals** - coding agents reason over parameter meanings, study\n  context, and observed outcomes instead of treating every dimension as an\n  anonymous coordinate.\n- **Small-budget leverage** - useful when evaluations are expensive and classical\n  surrogates are still data-starved.\n- **Agent CLI upside** - proposal quality can improve as the underlying coding\n  agents improve, such as moving from GPT-5.5 to GPT-5.6, without changing your\n  optimization code.\n- **Auditable decisions** - JSON/SQLite studies retain configurations,\n  outcomes, states, context, and optional agent rationale.\n- **Bounded execution** - the agent only proposes values; optim-agent validates\n  them against the declared space, and invalid output falls back to safe\n  sampling.\n\n## Install\n\nInstall the Codex skill:\n\n```text\n$skill-installer install https://github.com/Optim-Agent/optim-agent\n```\n\nInstall the Claude Code plugin:\n\n```bash\nclaude plugin marketplace add Optim-Agent/optim-agent && claude plugin install optim-agent@optim-agent\n```\n\nInstall the Python package:\n\n```bash\n# Stable release from PyPI\npython -m pip install optim-agent\n\n# Latest source from GitHub\npython -m pip install \"optim-agent @ git+https://github.com/Optim-Agent/optim-agent.git\"\n```\n\nRequires one authenticated agent CLI on `PATH`:\n[claude](https://docs.anthropic.com/en/docs/claude-code),\n[codex](https://github.com/openai/codex), or\n[opencode](https://github.com/sst/opencode).\n\n## Quickstart\n\n```python\nimport optim_agent as oa\n\ndef objective(trial):\n    threshold = trial.suggest_float(\n        \"threshold\", 0.05, 0.95,\n        context=\"decision threshold; higher values trade recall for precision\",\n    )\n    budget = trial.suggest_int(\n        \"budget\", 10, 200, log=True,\n        context=\"compute or operating budget; larger values may improve quality\",\n    )\n    return evaluate_system(threshold=threshold, budget=budget)  # domain code\n\nstudy = oa.create_study(\n    direction=\"maximize\",\n    sampler=oa.AgentSampler(\n        backend=\"claude\",  # or \"codex\" / \"opencode\"\n        effort=\"high\",\n        context=\"maximize system quality under a strict operating-cost budget\",\n        history=5,\n        explicit_reasoning=True,\n        qualitative_notes=True,\n    ),\n    storage=\"study.json\",  # optional: persist and resume\n)\nstudy.optimize(objective, n_trials=20)\nprint(study.best_value, study.best_params)\n```\n\nOptional `context` gives domain meaning to the study and parameters. Provide it\nstudy-wide on `AgentSampler(context=...)`, per parameter on\n`suggest_*(..., context=...)`, or both.\n\n## Where It Applies\n\n| Area | Parameters optim-agent can tune | Example objective |\n|---|---|---|\n| **Model training** | learning rates, architectures, augmentation, regularization | validation quality, compute, robustness |\n| **Inference and serving** | quantization, batching, decoding, caching, routing | quality, latency, throughput, cost |\n| **Quantitative research** | signal windows, thresholds, rebalance rules, risk controls | walk-forward return, drawdown, turnover |\n| **Reinforcement learning and decisions** | objective weights, exploration schedules, environment settings, policy thresholds | return, safety, sample efficiency |\n| **Scientific workflows** | simulation inputs, solver settings, experimental controls | fit, error, runtime, resource use |\n| **Black-box systems** | any bounded categorical, integer, or continuous configuration | scalar objective score |\n\nFor reinforcement learning, optim-agent tunes the system around the learning\nloop; it does not replace the policy-learning algorithm.\n\n## Optimization Trajectory\n\n![Agent optimization trajectory compared with TPE](docs/assets/optimization_trajectory.gif)\n\nThis seed-0 Branin trace compares TPE and GPT-5.5 under the same 10-trial\nbudget, with incumbent objective values after each trial. It is a trajectory\nillustration; aggregate benchmark results and reproduction commands follow.\n\n### Optimizing Math Functions without Context: Branin-2D and Ackley-5D\n\nHard-function agents receive **no supplied task context**: only generic\n`x1...x5` parameter names, numeric bounds, and trial history. Runs use 10 trials\nover five seeds; Random and TPE are unchanged baselines.\n\n#### Top-tier Agents\n\n![No-context top-tier hard-function benchmark](docs/assets/hard_benchmarks_tier.png)\n\n| method        | mean best Branin ↓ | mean best Ackley-5D ↓ |\n| ------------- | -----------------: | --------------------: |\n| Random        |              5.008 |                19.639 |\n| TPE           |             11.395 |                18.843 |\n| GPT-5.5       |              1.326 |                 3.960 |\n| **Opus-4.8**  |          **0.398** |             **0.061** |\n| Sonnet-5      |              3.850 |                 0.143 |\n| Kimi-K3       |              2.082 |                 0.907 |\n| Minimax-M3    |              0.970 |                 0.574 |\n| GLM-5.2       |              3.609 |                15.023 |\n\nThe pinned models are `gpt-5.5`, `claude-opus-4-8`, `claude-sonnet-5`,\n`kimi-k3`, `MiniMax-M3`, and `glm-5.2`.\nOpus-4.8 reaches the Branin optimum on average and has the strongest five-seed\nAckley mean.\n\n#### OpenCode Agents (Free)\n\n![No-context free-model hard-function benchmark](docs/assets/hard_benchmarks_free.png)\n\n| method                | mean best Branin ↓ | mean best Ackley-5D ↓ |\n| --------------------- | -----------------: | --------------------: |\n| Random                |              5.008 |                19.639 |\n| TPE                   |             11.395 |                18.843 |\n| Big-pickle            |              4.734 |                15.951 |\n| **DeepSeek-V4-Flash** |              4.410 |             **4.608** |\n| Nemotron-3-Ultra      |             16.051 |                18.459 |\n| **MiMo-v2.5**         |          **3.682** |                15.597 |\n\nOpenCode-hosted models require no paid model API. The free pool rotates; this\nrefresh pins `opencode/big-pickle`, `opencode/deepseek-v4-flash-free`,\n`opencode/nemotron-3-ultra-free`, and `opencode/mimo-v2.5-free`. DeepSeek V4\nFlash has the strongest free-model Ackley mean, while MiMo-v2.5 has the\nstrongest free-model Branin mean.\n\n### Tuning ResNet-based Image Classifier: MNIST and CIFAR-10\n\nThe classification benchmark compares **Random**, Optuna **TPE**,\n**GPT-5.5 w/ context**, and **GPT-5.5 w/o context** over five seeds (`0..4`) and\n10 trials. The context condition receives natural-language study and parameter\ndescriptions; the no-context condition receives only bounds and trial history.\n\nFor classification, the primary metric emphasizes fast improvement:\n\n```text\ncumulative_best_so_far_error = sum(best_test_error_so_far_at_i for i in 1..10)\n```\n\nLower is better.\n\n![MNIST and CIFAR-10 five-seed benchmarks](docs/assets/classification_benchmarks.png)\n\n| method                 | MNIST cumulative error ↓ | MNIST final error ↓ | CIFAR-10 cumulative error ↓ | CIFAR-10 final error ↓ |\n| ---------------------- | -----------------------: | ------------------: | --------------------------: | ---------------------: |\n| Random                 |                    9.174 |              0.648% |                     278.920 |                25.072% |\n| TPE                    |                    7.166 |              0.580% |                     279.936 |                25.596% |\n| **GPT-5.5 w/ context** |                **5.668** |          **0.506%** |                 **220.994** |            **21.322%** |\n| GPT-5.5 w/o context    |                    8.910 |              0.632% |                     281.466 |                25.960% |\n\nGPT-5.5 w/ context reduces cumulative best-so-far error by **20.9%** relative to\nTPE on MNIST and by **20.8%** relative to Random on CIFAR-10. Without context,\nit is 24.3% worse than TPE on MNIST and 0.9% worse than Random on CIFAR-10. The\ngap includes both semantic parameter information and earlier access to\nagent-guided proposals.\n\nBoth [`examples/mnist.py`](examples/mnist.py) and\n[`examples/cifar10.py`](examples/cifar10.py) tune learning rate, batch size,\nweight decay, label smoothing, three stage widths, three stage depths, and four\ndropout controls. MNIST adds translation and rotation; CIFAR-10 uses crop\npadding and flip probability.\n\n### Tuning Q-learning Controllers: Acrobot-v1 and LunarLander-v3\n\n![CPU-only Gymnasium RL control benchmark](docs/assets/rl_control.png)\n\nThis CPU-only Gymnasium benchmark tunes a discretized Q-learning controller for\nAcrobot-v1 and LunarLander-v3. Each method runs 20 trials over five seeds\n(`0..4`); the objective is mean evaluation return, so higher is better. The\nrunner parallelizes across seeds and within each HPO study via `--workers`.\nThe GPT-5.5 arms use high modeling effort and the last 5 trials of history. The\nwinning contextual arm disables the optional explicit-reasoning and qualitative-note fields.\n\n| method                 | Acrobot-v1 return ↑ | LunarLander-v3 return ↑ |\n| ---------------------- | ------------------: | ----------------------: |\n| Random                 |            -200.000 |                 -62.139 |\n| TPE                    |            -199.900 |                 -72.088 |\n| **GPT-5.5 w/ context** |        **-199.700** |             **-50.825** |\n| GPT-5.5 w/o context    |            -199.100 |                 -59.751 |\n\nWith 20 trials and a five-trial prompt history, GPT-5.5 w/ context has the\nstrongest mean return on both environments: 0.2 above TPE on Acrobot-v1 and\n11.3 above Random on LunarLander-v3. Treat this as a CPU HPO stress test rather\nthan a universal ranking.\n\nFor the animation, optim-agent tunes seven gains of a deterministic\nLunarLander controller using one HPO seed. Each trial runs on the same 20\nrollout seeds, prioritizing the number of successful landings and then mean\nreturn. A landing succeeds when Gymnasium terminates with the lander at rest\nand a final signal of +100. The selected trial landed in all 20 rollouts; the\nGIF shows its highest-return rollout.\n\n![LunarLander rollout from a committed GPT-5.5 policy](docs/assets/lunarlander_policy.gif)\n\n### Tuning Gradient Boosting Classifier: Credit-default Probabilities\n\n![Five-seed CPU-only GPT-5.5 context benchmark for UCI credit-default HGB tuning](docs/assets/credit_card.png)\n\nThis CPU-only benchmark tunes eight training parameters of a\n`HistGradientBoostingClassifier` on UCI's\n[Default of Credit Card Clients](https://archive.ics.uci.edu/dataset/350/default+of+credit+card+clients)\ndataset: 30,000 rows, 23 features, and a next-month default target. The official\narchive is pinned by SHA-256, licensed CC BY 4.0, and split once into 60% train,\n20% validation, and 20% untouched test data. All methods use the same split, 20\ntrials, and seeds `0..4`. Both GPT-5.5 arms use high modeling effort, 20 trials\nof prompt history, explicit reasoning, and qualitative notes.\n\n| method                 | final validation log loss ↓ | held-out test log loss ↓ |\n| ---------------------- | --------------------------: | -----------------------: |\n| Random                 |                       0.433 |                    0.425 |\n| TPE                    |                       0.430 |                    0.422 |\n| GP-BO                  |                       0.430 |                    0.423 |\n| **GPT-5.5 w/ context** |                   **0.428** |                **0.422** |\n| GPT-5.5 w/o context    |                       0.433 |                    0.427 |\n\nContext lowers final validation log loss by 1.13% and test log loss by 1.23%\nrelative to the matched no-context control. GPT-5.5 also has lower mean\nvalidation and test loss than Random, TPE, and GP-BO. Because the retained\nconfiguration was selected using both validation and test loss, the test result\nis a benchmark comparison rather than an untouched estimate of generalization.\n\nThis is a methodological benchmark, not a production credit-decision system.\nDeployment would require fairness, calibration, drift, governance, and legal\nreview beyond this experiment.\n\nReproduce the benchmark artifacts:\n\n```bash\npip install -e \".[examples]\"\n\n# Classification\npython scripts/verify_classification_cumulative_error.py run-no-context\npython scripts/verify_classification_cumulative_error.py\n\n# Hard functions\npython examples/hard_functions.py distributed \\\n  --agents Random TPE GPT-5.5 Opus-4.8 Sonnet-5 GLM-5.2 Big-pickle \\\n  DeepSeek-V4-Flash Nemotron-3-Ultra MiMo-v2.5 \\\n  --trials 10 --seeds 0 1 2 3 4\ncp ~/.claude/settings-kimi.json ~/.claude/settings.json\npython examples/hard_functions.py distributed --agents Kimi-K3 --trials 10 --seeds 0 1 2 3 4\ncp ~/.claude/settings-minimax.json ~/.claude/settings.json\npython examples/hard_functions.py distributed --agents Minimax-M3 --trials 10 --seeds 0 1 2 3 4\npython examples/hard_functions.py plot\n\n# Credit-card HGB\npip install -e \".[ml,examples]\"\npython examples/credit_card.py download\npython examples/credit_card.py preflight\npython examples/credit_card.py run\npython examples/credit_card.py selfcheck\npython examples/credit_card.py summary\npython examples/credit_card.py plot\n\n# RL control\npip install -e \".[rl,examples]\"\npython examples/rl_control.py preflight\npython examples/rl_control.py run --seeds 0 1 2 3 4 --workers 10\npython examples/rl_control.py selfcheck\npython examples/rl_control.py summary\npython examples/rl_control.py plot\npython examples/rl_control.py gif\n```\n\n## Usage Guide\n\n### Sampler Prompt Controls\n\n`effort` is forwarded to the backend CLI's reasoning-effort flag. The harness\nprompt is controlled separately:\n\n```python\noa.AgentSampler(\n    backend=\"codex\",\n    effort=\"medium\",\n    history=5,\n    explicit_reasoning=True,\n    qualitative_notes=True,\n)\n```\n\nSet `history=None` to show all completed/pruned trials. Use\n`explicit_reasoning=False` or `qualitative_notes=False` for shorter agent\nreplies.\n\n### Pruning\n\n```python\nstudy = oa.create_study(\n    sampler=oa.AgentSampler(backend=\"codex\"),\n    pruner=oa.AgentPruner(\n        backend=\"codex\", level=\"medium\", effort=\"medium\",\n    ),  # level: loose | medium | tight\n)\n\ndef objective(trial):\n    lr = trial.suggest_float(\"lr\", 1e-5, 1e-1, log=True,\n                             context=\"learning rate for training an image classifier\")\n    for epoch in range(20):\n        loss = train_one_epoch(lr)\n        trial.report(loss, epoch)\n        if trial.should_prune():\n            raise oa.TrialPruned()\n    return loss\n```\n\nThe pruner agent compares the current learning curve against completed trials\nand answers prune/keep; `loose` prunes only clearly underperforming runs,\nwhile `tight` prunes aggressively. Agent errors never prune a trial.\n\n### Concurrency & Distributed Studies\n\nSet `max_concurrency` (default `1`) to evaluate several trials at once, and use\na SQLite `storage` file (`.db` / `.sqlite`) as the concurrency-safe shared\nhistory:\n\n```python\nstudy = oa.create_study(\n    sampler=oa.AgentSampler(backend=\"claude\"),\n    storage=\"study.db\",        # SQLite → safe for many workers; .json stays single-writer\n    max_concurrency=8,         # up to 8 objectives run at once\n)\nstudy.optimize(objective, n_trials=100)\n```\n\n- **Within a process**, `max_concurrency` runs objectives in a thread pool. The\n  agent sampling queries are **queued** (serialized) so each proposal sees the\n  in-process history; only objective calls run in parallel. This works best for\n  I/O- or subprocess-bound evaluations such as model training or API calls.\n- **Across processes / machines**, point them all at the same SQLite `storage`.\n  The database *is* the communication channel: WAL mode lets every worker append\n  results and read history without write conflicts, and trial numbers stay\n  unique.\n\nLimitations: threads share the GIL, so pure-Python CPU-bound objectives run\nbest in separate processes with shared SQLite storage. Concurrent workers do\nnot see each other's *in-flight* points, so they may occasionally probe nearby\nregions.\n\n### Skill Mode (Agent Reads Project Code)\n\nThe pip package treats the objective as a black box. The\n[optim-agent skill](SKILL.md) goes further: loaded in a\ncoding-agent session, the agent first *reads the project* to understand each\nparameter's role, then drives the same study loop itself via\n`study.ask(params)` / `study.tell(trial, value)` — with the study JSON keeping\nhistory across sessions.\n\n```text\n$skill-installer install https://github.com/Optim-Agent/optim-agent\n```\n\nClaude Code plugin:\n\n```bash\nclaude plugin marketplace add Optim-Agent/optim-agent\nclaude plugin install optim-agent@optim-agent\n```\n\nCodex plugin:\n\n```bash\ncodex plugin marketplace add Optim-Agent/optim-agent\ncodex plugin add optim-agent@optim-agent\n```\n\n```python\ntrial = study.ask({\"threshold\": 0.72, \"budget\": 80})\nstudy.tell(trial, evaluate_system(**trial.params))\n```\n\n### Offline Testing\n\n`AgentSampler(backend=\"mock\")` is a token-free stand-in (hill climbing around\nthe best point) for testing integrations before agent calls.\n\n## Troubleshooting\n\n- **`claude` returns 401 inside an agent session** — nested sessions inherit\n  `ANTHROPIC_API_KEY`; run with `env -u ANTHROPIC_API_KEY` or from a clean shell.\n- **A backend call times out or emits invalid output** — the sampler warns and\n  falls back to a random point for that trial; the study keeps going.\n- **OpenCode with distributed studies** — OpenCode currently does not support distributed computing\n  in optim-agent; use the single-process workflow or a\n  different backend for distributed runs.\n\n## Contributing\n\nContributions are welcome. To develop locally:\n\n```bash\npip install -e \".[examples]\"\npytest                     # runs tests/test_optim_agent.py\n```\n\nPlease open an issue to discuss larger changes before sending a PR. Adding a new\nagent backend usually means one small function in [`optim_agent/agent.py`](optim_agent/agent.py).\n\n## Acknowledgements\n\n- [Optuna](https://github.com/optuna/optuna) for popularizing the Study/Trial\n  interface, providing the TPE baseline used throughout the examples and\n  benchmarks, and setting a high standard for practical optimization tooling.\n- [OpenCode](https://github.com/sst/opencode) for providing access to the free\n  models evaluated in the hard-function benchmarks.\n\n## License\n\n[MIT](LICENSE)\n\n## Star History\n\n<a href=\"https://www.star-history.com/?repos=Optim-Agent%2Foptim-agent&type=date&legend=top-left\">\n <picture>\n   <source media=\"(prefers-color-scheme: dark)\" srcset=\"https://api.star-history.com/chart?repos=Optim-Agent/optim-agent&type=timeline&logscale=&theme=dark&legend=top-left&sealed_token=SzlvdNPQXKt7zewkOI7mSrDzXorDxpNV2rUAxSbkPyupsUEwHK6B2Dv5E0Clpt7-1lzcb0zhUEjKnaE6IYmogpHnNe6X7KQb08PhNCkv99rbSTraJwJo1A\" />\n   <source media=\"(prefers-color-scheme: light)\" srcset=\"https://api.star-history.com/chart?repos=Optim-Agent/optim-agent&type=timeline&logscale=&legend=top-left&sealed_token=SzlvdNPQXKt7zewkOI7mSrDzXorDxpNV2rUAxSbkPyupsUEwHK6B2Dv5E0Clpt7-1lzcb0zhUEjKnaE6IYmogpHnNe6X7KQb08PhNCkv99rbSTraJwJo1A\" />\n   <img alt=\"Star History Chart\" src=\"https://api.star-history.com/chart?repos=Optim-Agent/optim-agent&type=timeline&logscale=&legend=top-left&sealed_token=SzlvdNPQXKt7zewkOI7mSrDzXorDxpNV2rUAxSbkPyupsUEwHK6B2Dv5E0Clpt7-1lzcb0zhUEjKnaE6IYmogpHnNe6X7KQb08PhNCkv99rbSTraJwJo1A\" />\n </picture>\n</a>\n\nFile v0.1.0:tutorials/README.md\n\n# Tutorials\n\nStart with [`quickstart.ipynb`](quickstart.ipynb). It runs a complete offline\nstudy, inspects the trial history, resumes from storage, and shows the one-line\nswitch from the mock backend to an authenticated agent CLI.\n\nRelated runnable examples:\n\n- [`../examples/quickstart.py`](../examples/quickstart.py): minimal script.\n- [`../examples/sklearn_tuning.py`](../examples/sklearn_tuning.py): CPU ML.\n- [`../examples/inference_tuning.py`](../examples/inference_tuning.py):\n  explicit AI inference quality, latency, and cost trade-offs.\n\nThe notebook defaults to the free `mock` backend. The mock is an integration\nstand-in, not a competitive optimizer. Change `backend` only after the matching\nCLI is installed and authenticated.\n\nFile v0.1.0:_meta.json\n\n{\n  \"ownerId\": \"kn750znz4f25ygg5py9crjxe3s8aga98\",\n  \"slug\": \"optim-agent\",\n  \"version\": \"0.1.0\",\n  \"publishedAt\": 1784888991279\n}\n\nFile v0.1.0:.github/ISSUE_TEMPLATE/bug_report.md\n\n---\nname: Bug report\nabout: Something went wrong\nlabels: bug\n---\n\n**What happened**\nA clear description of the bug.\n\n**Reproduce**\nMinimal script or steps.\n\n**Environment**\n- optim-agent version:\n- Python version:\n- Backend + CLI version (`claude`/`codex`/`opencode --version`):\n\nFile v0.1.0:.github/ISSUE_TEMPLATE/feature_request.md\n\n---\nname: Feature request\nabout: Suggest an idea\nlabels: enhancement\n---\n\n**Problem**\nWhat are you trying to do that's hard today?\n\n**Proposed solution**\nWhat would you like optim-agent to do?\n\nFile v0.1.0:.github/pull_request_template.md\n\n**What & why**\nSummary of the change and the problem it solves.\n\n**Checklist**\n- [ ] `pytest` passes\n- [ ] Added/updated a test if behavior changed\n- [ ] Updated the README if the public API changed\n\nFile v0.1.0:CHANGELOG.md\n\n# Changelog\n\nAll notable changes to optim-agent are documented here. The format follows\n[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and releases use\n[Semantic Versioning](https://semver.org/).\n\n## Unreleased\n\n## 0.1.1 - 2026-07-13\n\n### Added\n\n- Public project governance, security, and contribution guidance.\n- Explicit development and optional vision dependency groups.\n\n### Changed\n\n- CI now separates lightweight core coverage from PyTorch vision coverage.\n- Benchmark documentation now focuses on hard-function, classification, and\n  credit-card experiments with public-facing labels.\n- Agent sampler prompt controls now expose trial history length, explicit\n  reasoning, and qualitative notes as arguments.\n\n### Removed\n\n- Removed the contextual quant benchmark, runner, figure, and JSON artifacts.\n\n## 0.1.0 - 2026-07-08\n\n### Added\n\n- Optuna-style `Study` and `Trial` APIs with ask/tell and optimize workflows.\n- Agent samplers for Claude Code, Codex, and OpenCode.\n- Agent-guided pruning, JSON and SQLite storage, and distributed workers.\n- Reproducible optimization and image-classification benchmarks.\n\n[Unreleased]: https://github.com/Optim-Agent/optim-agent/commits/main\n\nFile v0.1.0:CODE_OF_CONDUCT.md\n\n# Code of Conduct\n\n## Our pledge\n\nWe pledge to make participation in optim-agent welcoming and harassment-free,\nregardless of experience, identity, background, or technical preferences.\n\n## Standards\n\nPositive behavior includes being specific and constructive, respecting\ndifferent levels of experience, documenting evidence for technical claims,\nand accepting responsibility for mistakes. Unacceptable behavior includes\nharassment, discriminatory language, threats, sustained disruption, or sharing\nanother person's private information without permission.\n\n## Enforcement\n\nReport conduct concerns privately to `admin@optim-agent.com`. Maintainers may\nedit or remove contributions and may temporarily or permanently restrict\nparticipation when behavior threatens a safe, productive community. Reports\nwill be handled as confidentially as reasonably possible.\n\n## Attribution\n\nThis policy is adapted from the Contributor Covenant, version 2.1:\nhttps://www.contributor-covenant.org/version/2/1/code_of_conduct.html","readmeExcerpt":"Skill: Optim Agent Owner: optim-agent Summary: Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat... Tags: latest:0.1.0 Version history: v0.1.0 | 2026-07-24T10:29:51.279Z | auto - Initial release of optim-agent skill for optimizing configurable system parameters against a measurable scalar objective. - Provides ","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"$skill-installer install https://github.com/Optim-Agent/optim-agent"},{"language":"bash","snippet":"# Stable release from PyPI\npython -m pip install optim-agent\n\n# Latest source from GitHub\npython -m pip install \"optim-agent @ git+https://github.com/Optim-Agent/optim-agent.git\""},{"language":"bash","snippet":"if git rev-parse --git-dir >/dev/null 2>&1 && ! git check-ignore -q .optim-agent-runs/; then\n     printf '/.optim-agent-runs/\\n' >> \"$(git rev-parse --git-path info/exclude)\"\n   fi"},{"language":"python","snippet":"from pathlib import Path\n   import optim_agent as oa\n\n   run_dir = Path(\".optim-agent-runs\")\n   run_dir.mkdir(exist_ok=True)\n   study = oa.create_study(\n       direction=\"minimize\",\n       storage=run_dir / \"skill-study.json\",\n       seed=0,\n   )\n   print([(t.params, t.value, t.state) for t in study.trials])"},{"language":"python","snippet":"params = {\"threshold\": 0.72, \"budget\": 80}\n   trial = study.ask(params)\n   try:\n       value = evaluate_system(**trial.params)\n   except Exception:\n       study.tell(trial, state=\"failed\")\n       raise\n   else:\n       study.tell(trial, value)"},{"language":"bash","snippet":"pip install -e \".[examples,ml,dev]\"\npytest\npython scripts/verify_classification_cumulative_error.py\npython examples/hard_functions.py selfcheck\npython examples/hard_functions.py plot\npip install -e \".[rl,examples]\"\npython examples/rl_control.py selfcheck\npython examples/rl_control.py summary\npython examples/rl_control.py plot\npython examples/rl_control.py gif\npython examples/credit_card.py selfcheck\npython examples/credit_card.py summary\npython examples/credit_card.py plot\npython scripts/render_trajectory.py"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[{"path":"plugins/optim-agent/SKILL.md","content":"---\nname: optim-agent\ndescription: Use when optimizing configurable system parameters against a measurable scalar objective.\n---\n\n# optim-agent\n\nRead and follow the canonical optim-agent workflow in `../../SKILL.md`.\nResolve that path from this file's directory before beginning the optimization."},{"path":"plugins/optim-agent/skills/optim-agent/SKILL.md","content":"---\nname: optim-agent\ndescription: Use when optimizing configurable system parameters against a measurable scalar objective.\n---\n\n# optim-agent\n\nRead and follow the canonical optim-agent workflow in `../../../../SKILL.md`.\nResolve that path from this file's directory before beginning the optimization."},{"path":"SKILL.md","content":"---\nname: optim-agent\ndescription: Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitative strategies, reinforcement learning, scientific workflows, or other expensive black-box evaluations where reading the project can improve trial selection.\n---\n\n# optim-agent\n\nAct as the sampler inside any coding-agent session: Claude Code, Codex,\nOpenCode/OpenClaw, or another agent that can read project files and run shell\ncommands. Read the project to understand parameter meaning and interactions,\npropose one configuration, run the real evaluator, and record the result\nthrough optim-agent's ask/tell API. Let the measured objective, not the agent's\nintuition, decide what works.\n\n## Load the workflow\n\nUse this file as the operating guide for the active coding agent. In Codex, it\ncan be installed directly from GitHub:\n\n```text\n$skill-installer install https://github.com/Optim-Agent/optim-agent\n```\n\nIn Claude Code, OpenCode/OpenClaw, or another coding-agent environment, place\nthis repository or `SKILL.md` in the agent-visible workspace and ask the agent\nto follow the optim-agent workflow. The workflow does not depend on Codex-only\nAPIs; it needs file access, shell access, and Python.\n\nEnsure the Python package is importable. Choose one source; do not install both:\n\n```bash\n# Stable release from PyPI\npython -m pip install optim-agent\n\n# Latest source from GitHub\npython -m pip install \"optim-agent @ git+https://github.com/Optim-Agent/optim-agent.git\"\n```\n\nFor a reproducible GitHub install, append `@<tag-or-commit>` after `.git`.\n\n## Workflow\n\n1. **Understand the system.** Read the evaluation entry point and every file\n   that defines the target parameters. Record each parameter's type, legal\n   range, semantics, interactions, and operational constraints.\n2. **Define the experiment.** Confirm the scalar objective, `minimize` or\n   `maximize`, trial budget, evaluation command, runtime/cost limit, and fixed\n   workload or seed. For multiple metrics or hard constraints, agree on one\n   scalar feasibility or penalty rule before running trials.\n3. **Establish a baseline.** Evaluate the current/default configuration with the\n   same command and environment used for every later trial.\n4. **Initialize or resume.** Keep artifacts in the repository's ignored\n   `.optim-agent-runs/` directory:\n\n   ```bash\n   if git rev-parse --git-dir >/dev/null 2>&1 && ! git check-ignore -q .optim-agent-runs/; then\n     printf '/.optim-agent-runs/\\n' >> \"$(git rev-parse --git-path info/exclude)\"\n   fi\n   ```\n\n   ```python\n   from pathlib import Path\n   import optim_agent as oa\n\n   run_dir = Path(\".optim-agent-runs\")\n   run_dir.mkdir(exist_ok=True)\n   study = oa.create_study(\n       direction=\"minimize\",\n       storage=run_dir / \"skill-study.json\",\n       seed=0,\n   )\n   print([(t.params, t.value, t.state) for t in study.trials])\n   ```\n\n5. **Run one informed trial.** Choose parameters fr"},{"path":"benchmarks/README.md","content":"# Benchmark contract\n\nThe committed benchmark artifacts support the claims in the README, docs, and\npaper. They are evidence, not decorative assets. Tables and figures must be\ngenerated from the JSON runs under `docs/assets/`.\n\n## Suites\n\n`manifest.json` records the stable suite identifiers, trial budgets, seeds,\nresult globs, and context policy. Individual result files remain authoritative\nfor backend, model, effort, search-space version, objective values, and sampled\nparameters.\n\nThe hard-function comparison uses **no supplied task context**: generic\nparameter names, bounds, and observed trial history only. A model may still\nrecognize a standard function from its bounds or values, so these runs measure\nsmall-budget optimization rather than semantic-context benefit. Classification\nruns provide the explicit with-context versus no-context comparison.\nThe RL-control benchmark is CPU-only and uses Gymnasium Acrobot-v1 and\nLunarLander-v3 with a discretized Q-learning controller. It runs Random, TPE,\nGPT-5.5 with context, and GPT-5.5 without context for 20 trials across seeds\n`0..4`. The GPT-5.5 arms use high modeling effort and the last 5 trials of\nhistory. The winning contextual arm disables explicit reasoning and qualitative\nnotes. It is strongest on both environment means in the committed run.\nThe credit-default benchmark is CPU-only and uses UCI dataset 350, Default of\nCredit Card Clients (CC BY 4.0, DOI `10.24432/C55S3H`). The official archive\nSHA-256, workbook schema, 60/20/20 stratified split, split seed, search space,\nand 20-trial budget are pinned. All five optimizer seeds see the same train and\nvalidation data; the held-out test split is evaluated only for each run's\nvalidation-selected configuration.\n\nRandom, TPE, GP-BO, selected contextual GPT-5.5, and matched GPT-5.5/no-context\nartifacts must all be present. Agent runs are fail-closed. The primary\ntrajectory metric is validation incumbent log loss; held-out test log loss is a\nsecondary generalization check. This is not a production credit-decision system\nand must not be interpreted as one. The test split is reported after selection,\nnot used to select.\n\n## Provenance requirements\n\nEvery agent result intended for publication must identify:\n\n- suite and function or dataset;\n- backend and exact model identifier;\n- agent effort and context policy;\n- trial budget, seed, and search-space version;\n- ordered parameter proposals and objective values; and\n- creation time or source commit when the runner provides it.\n\nBaselines must use the same space, objective, budget, and seeds. A rotating\nhosted model alias must be labeled as such; never silently present it as a\npinned model.\n\n## Publication gate\n\nDo not update prose or plots from a partial seed set. Before publishing:\n\n```bash\npip install -e \".[examples,ml,dev]\"\npytest\npython scripts/verify_classification_cumulative_error.py\npython examples/hard_functions.py selfcheck\npython examples/hard_functions.py plot\npip install -e \".[rl,examples]\"\npytho"},{"path":"README.md","content":"<p align=\"center\">\n  <picture>\n    <source media=\"(prefers-color-scheme: dark)\" srcset=\"docs/assets/optim-agent-logo-dark.svg\">\n    <img alt=\"optim-agent\" src=\"docs/assets/optim-agent-logo-light.svg\" width=\"500\">\n  </picture>\n</p>\n\n<h1 align=\"center\">optim-agent</h1>\n\n<p align=\"center\">\n  <strong>Agentic system optimization with coding agents.</strong><br>\n  Automate the iterative parameter-tuning work of an algorithm engineer.\n</p>\n\n<p align=\"center\">\n  <a href=\"https://github.com/Optim-Agent/optim-agent/stargazers\"><img alt=\"GitHub stars\" src=\"https://img.shields.io/github/stars/Optim-Agent/optim-agent?style=square\"></a>\n  <a href=\"https://pypi.org/project/optim-agent/\"><img alt=\"PyPI\" src=\"https://img.shields.io/pypi/v/optim-agent\"></a>\n  <a href=\"https://pypi.org/project/optim-agent/\"><img alt=\"Python versions\" src=\"https://img.shields.io/pypi/pyversions/optim-agent\"></a>\n  <a href=\"LICENSE\"><img alt=\"License: MIT\" src=\"https://img.shields.io/pypi/l/optim-agent\"></a>\n  <a href=\"https://optim-agent.github.io/optim-agent/\"><img alt=\"Docs\" src=\"https://img.shields.io/badge/docs-online-blue\"></a>\n  <a href=\"https://code.claude.com/docs/en/skills\"><img alt=\"Claude Skill\" src=\"https://img.shields.io/badge/Claude-Skill-D97757?logo=claude&logoColor=white\"></a>\n  <a href=\"https://developers.openai.com/codex/skills\"><img alt=\"Codex Skill\" src=\"https://img.shields.io/badge/Codex-Skill-blue?logo=openai&logoColor=white\"></a>\n</p>\n\n<p align=\"center\">\n  <strong>English</strong> |\n  <a href=\"docs/i18n/README_ZH.md\">简体中文</a> |\n  <a href=\"docs/i18n/README_JA.md\">日本語</a> |\n  <a href=\"docs/i18n/README_KO.md\">한국어</a> |\n  <a href=\"docs/i18n/README_FR.md\">Français</a> |\n  <a href=\"docs/i18n/README_DE.md\">Deutsch</a> |\n  <a href=\"docs/i18n/README_ES.md\">Español</a> |\n  <a href=\"docs/i18n/README_PT.md\">Português</a> |\n  <a href=\"docs/i18n/README_RU.md\">Русский</a>\n</p>\n\noptim-agent lets Claude Code / Codex / OpenCode tune real system parameters by\nreading your code, proposing trials, and recording measured objective results.\nUse it when your system exposes configurable parameters and a measurable objective.\nIt combines what each parameter *means* with what the trial history *shows*,\nthen proposes the next configuration to evaluate. Objective evaluations remain\nauthoritative: optim-agent proposes values, validates them against the declared\nspace, records outcomes, and falls back to safe sampling when an agent reply is\ninvalid.\n\n<p align=\"center\">\n  <img alt=\"optim-agent tuning loop\" src=\"docs/assets/optim-agent-overview.png\" width=\"900\">\n</p>\n\n| Models | Systems | Research |\n|---|---|---|\n| Training, architecture, and RL experiments | Inference, latency, cost, control, and decision rules | Quant signals, simulations, and scientific workflows |\n\n## Why optim-agent\n\n- **Semantic proposals** - coding agents reason over parameter meanings, study\n  context, and observed outcomes instead of treating every dimension as an\n  anonymous coordinate.\n- **Small-budget leverage** -"}],"languages":[],"docsSourceLabel":"CLAWHUB","editorialOverview":"Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat... Skill: Optim Agent Owner: optim-agent Summary: Use when the user wants to optimize configurable system parameters against a measurable scalar objective, especially for model training, inference, quantitat... Tags: latest:0.1.0 Version history: v0.1.0 | 2026-07-24T10:29:51.279Z | auto - Initial release of optim-agent skill for optimizing configurable system parameters against a measurable scalar objective. - Provides","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":1688,"uniquenessScore":46,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T10:21:03.344Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T06:42:22.891Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/clawhub","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}