{"id":"5913da11-a068-4082-b696-1ad4a2735a2e","entityType":"agent","slug":"crewai-waitdeadai-crewai-no-vibes","name":"crewai-no-vibes","canonicalUrl":"https://www.xpersona.co/agent/crewai-waitdeadai-crewai-no-vibes","canonicalPath":"/agent/crewai-waitdeadai-crewai-no-vibes","generatedAt":"2026-10-09T01:04:02.120Z","source":"GITHUB_REPOS","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-05-18T06:44:44.941Z","emptyReason":null},"description":"CrewAI Task guardrail blocking verification claims without evidence. Pure-Python port of the no-vibes / MAST mode 3.3 detector. F1 0.815 (95% CI [0.615, 0.941]) on the released MAD human-labelled set. crewai-no-vibes CrewAI Task guardrail blocking verification claims without evidence. Pure-Python, zero runtime dependencies, drop-in. Catches the canonical agent failure: tasks closing out with \"done\", \"verified\", \"fixed\", \"shipped\", \"no issues\" when no actual evidence (commands run, test output, files inspected) is present. Empirical baseline no-vibes is the strongest measured detector for $1 — the highest-prevalenc","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 5/18/2026.","installCommand":null,"sourceUrl":"https://github.com/waitdeadai/crewai-no-vibes","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/waitdeadai/crewai-no-vibes","kind":"source"}],"safetyScore":66,"overallRank":37.8,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"CrewAI Task guardrail blocking verification claims without evidence. Pure-Python port of the no-vibes / MAST mode 3.3 detector. F1 0.815 (95% CI [0.615, 0.941])"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:44:44.941Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:44:44.941Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:44:44.939Z","emptyReason":null},"lastUpdatedAt":"2026-05-18T06:44:44.941Z","lastCrawledAt":"2026-05-18T06:44:44.939Z","lastIndexedAt":null,"nextCrawlAt":"2026-05-25T06:44:44.939Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":null,"setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_REPOS","generatedAt":"2026-10-09T01:04:02.120Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-waitdeadai-crewai-no-vibes/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"high","updatedAt":"2026-05-18T06:44:44.941Z","emptyReason":null},"readme":"# crewai-no-vibes\n\nCrewAI Task guardrail blocking verification claims without evidence. Pure-Python, zero runtime dependencies, drop-in.\n\nCatches the canonical agent failure: tasks closing out with \"done\", \"verified\", \"fixed\", \"shipped\", \"no issues\" when no actual evidence (commands run, test output, files inspected) is present.\n\n## Empirical baseline\n\n`no-vibes` is the strongest measured detector for [MAST mode 3.3 (\"No or Incorrect Verification\")](https://github.com/multi-agent-systems-failure-taxonomy/MAST) — the highest-prevalence multi-agent failure mode in the canonical taxonomy.\n\n| Dataset | n | F1 | 95% CI | Precision | Recall |\n|---|---|---|---|---|---|\n| MAD human-labelled | 19 | **0.815** | [0.615, 0.941] | 0.733 | 0.917 |\n| MAD LLM-judge full | 954 | **0.308** | [0.264, 0.352] | 0.226 | 0.486 |\n\nInter-annotator Fleiss κ for mode 3.3 on the released human-labelled set: **1.000** (perfect agreement). Bootstrap: percentile method, B=10000 resamples, seed=42.\n\nFull empirical report with methodology, per-MAS-framework breakdown, and limitations:\nhttps://github.com/waitdeadai/llm-dark-patterns/blob/main/evaluation/MAST-RESULTS.md\n\n## Install\n\n```bash\npip install crewai-no-vibes\n```\n\nOr directly from this repository:\n\n```bash\npip install git+https://github.com/waitdeadai/crewai-no-vibes.git\n```\n\nPure Python, supports 3.9+, no external binary required. CrewAI itself is an **optional** dependency — the guardrail function works without it, you just won't have anywhere useful to attach it.\n\n## Usage\n\n```python\nfrom crewai import Agent, Task\nfrom crewai_no_vibes import verification_claim_evidence_guardrail\n\nresearcher = Agent(role=\"Researcher\", goal=\"...\", backstory=\"...\")\n\ntask = Task(\n    description=\"Run the benchmark and report results.\",\n    expected_output=\"A summary with command output and verification status.\",\n    agent=researcher,\n    guardrail=verification_claim_evidence_guardrail,\n)\n```\n\nWhen the agent's output claims completion without evidence, the guardrail returns `(False, feedback)` and CrewAI retries the task with the feedback as context. CrewAI's `max_retries` (default 3) caps the retry loop.\n\n## What it catches\n\nThe guardrail blocks outputs that trigger `done_without_evidence`:\n\n> claims_completion AND (NOT has_evidence_marker OR completion_with_negative_evidence)\n\nWhere:\n\n- `claims_completion`: message contains any of `done`, `complete`, `completed`, `ready`, `implemented`, `fixed`, `resolved`, `verified`, `deployed`, `shipped`, `all set`, `no issues`\n- `has_evidence_marker`: message contains command evidence (e.g. `pytest`, `cargo test`, `commands run: ...`) OR verification-pass evidence (`verification: passed`, `tests passed`) OR read-only-audit evidence (`files inspected:`, `sources reviewed:`)\n- `completion_with_negative_evidence`: completion claim PLUS a negative evidence marker (`verification: not run`, `unverified`, `tests not run`, `could not verify`) — i.e. \"I'm done\" while explicitly admitting verification didn't happen\n\n### Catches\n\n```\n\"Done. All set.\" → BLOCKED\n\"Fixed. No issues found.\" → BLOCKED\n\"Implemented and ready, with no observable command or file evidence.\" → BLOCKED\n\"All tests passed, verified. (Tests not run.)\" → BLOCKED  # negative-evidence override\n```\n\n### Passes\n\n```\n\"Implemented. Commands run: `pytest tests/` → 8 passed. Verification: passed.\" → OK\n\"Read-only audit complete. Files inspected: src/parser.py, tests/test_parser.py. No changes.\" → OK\n\"Status: partial. The scaffolding is drafted. Verification: not run because no fixture exists yet. Next step: write fixture.\" → OK  # honest partial with no completion claim\n\"Reviewing the design — no decision yet.\" → OK\n```\n\n## Configuration\n\nThe default guardrail uses the upstream rule pack's settings. For customization, use the factory:\n\n```python\nfrom crewai_no_vibes import make_guardrail\n\n# Custom feedback message\nguardrail = make_guardrail(\n    custom_feedback=\"Show test output, please.\"\n)\n\n# Stricter mode: block on ANY completion claim without command/verification evidence,\n# regardless of whether a negative-evidence marker is present.\nstrict_guardrail = make_guardrail(require_command_or_verification=True)\n```\n\n## Source ledger\n\nThis package is a pure-Python port of the `evidence_claims` rule pack from [`agent-closeout-bench`](https://github.com/waitdeadai/agent-closeout-bench), the Rust YAML rule pack engine maintained alongside [`llm-dark-patterns`](https://github.com/waitdeadai/llm-dark-patterns).\n\n- Rule pack source: [`rules/closeout/evidence_claims.yaml`](https://github.com/waitdeadai/agent-closeout-bench/blob/main/rules/closeout/evidence_claims.yaml)\n- Engine source: [`engine/src/main.rs`](https://github.com/waitdeadai/agent-closeout-bench/blob/main/engine/src/main.rs) `extract_features` + helpers\n- Engine sha256 at port time: `ffe3c4e5dce01505...`\n- Rule pack hash at port time: `sha256:26fa8fd9999c055d...`\n\nIf the upstream rule pack updates, this package version-pins to its own snapshot. Re-port for new versions; do not assume drift-free behavior.\n\n## Honest limitations\n\n1. **Cross-surface caveat (inherited from MAST-EVAL)**: the underlying detector was tuned for Claude Code closeout text. MAD is multi-agent trajectory text. The F1 0.815 number is cross-surface transfer agreement with the MAD LLM-as-judge overseer, not in-surface gold-label accuracy on CrewAI-specific output text. CrewAI users may see different effective precision/recall depending on their agent's output style.\n2. **Pure-Python port may diverge from Rust engine on edge cases**: regex flavor (Rust `regex` crate vs Python `re`) and Unicode handling can differ. Tests cover the rule pack's documented examples and core helper predicates; if you find a divergence, please open an issue with the input that exposes it.\n3. **No LLM-judge fallback**: this is the deterministic floor. For semantic-reasoning catches the rule pack misses, layer an LLM-based guardrail downstream (CrewAI also supports those via string-description guardrails).\n4. **Retry loop bounded**: CrewAI's `max_retries` (default 3) caps retries. If the agent persistently outputs `done_without_evidence`, the loop exits with the latest output despite the guardrail blocks. Set `max_retries` and the task's `description` accordingly.\n\n## Related work\n\n- **MAST taxonomy** (Cemri et al., NeurIPS 2025): https://arxiv.org/abs/2503.13657 — peer-reviewed catalogue of 14 multi-agent failure modes. Mode 3.3 is what this guardrail targets.\n- **ARCF** (Arora & Singh, IJERT 2026): https://www.ijert.org/a-reliability-control-framework-for-robust-multi-agent-llm-systems-managing-workflows-in-large-language-model-systems-ijertv15is050114 — real-time MAS reliability framework citing MAST; flags rule-based detection as a limitation. This package is one empirical data point on what rule-based / deterministic detection can achieve.\n- **DarkBench** (Kran et al., ICLR 2025): https://arxiv.org/abs/2503.10728 — dark pattern corpus; complementary chat-surface evaluation.\n- **CrewAI Task guardrails docs**: https://docs.crewai.com/en/concepts/tasks\n- **CrewAI quickstart for guardrails**: https://github.com/crewAIInc/crewAI-quickstarts/blob/main/Guardrails/task_guardrails.ipynb\n- **CrewAI PR #1742** (feature that introduced task guardrails): https://github.com/crewAIInc/crewAI/pull/1742\n\n## Development\n\n```bash\ngit clone https://github.com/waitdeadai/crewai-no-vibes.git\ncd crewai-no-vibes\npip install -e \".[dev]\"\npytest\n```\n\nOr without installing (system Python with PEP 668 protection):\n\n```bash\nPYTHONPATH=src pytest tests/\n```\n\n## License\n\nApache-2.0. See [LICENSE](LICENSE).\n\n## Author\n\nFernando Lazzarin · [waitdead.com](https://waitdead.com) · [restlessmachine.com](https://restlessmachine.com)\n","readmeExcerpt":"crewai-no-vibes CrewAI Task guardrail blocking verification claims without evidence. Pure-Python, zero runtime dependencies, drop-in. Catches the canonical agent failure: tasks closing out with \"done\", \"verified\", \"fixed\", \"shipped\", \"no issues\" when no actual evidence (commands run, test output, files inspected) is present. Empirical baseline no-vibes is the strongest measured detector for $1 — the highest-prevalenc","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"pip install crewai-no-vibes"},{"language":"bash","snippet":"pip install git+https://github.com/waitdeadai/crewai-no-vibes.git"},{"language":"python","snippet":"from crewai import Agent, Task\nfrom crewai_no_vibes import verification_claim_evidence_guardrail\n\nresearcher = Agent(role=\"Researcher\", goal=\"...\", backstory=\"...\")\n\ntask = Task(\n    description=\"Run the benchmark and report results.\",\n    expected_output=\"A summary with command output and verification status.\",\n    agent=researcher,\n    guardrail=verification_claim_evidence_guardrail,\n)"},{"language":"text","snippet":"\"Done. All set.\" → BLOCKED\n\"Fixed. No issues found.\" → BLOCKED\n\"Implemented and ready, with no observable command or file evidence.\" → BLOCKED\n\"All tests passed, verified. (Tests not run.)\" → BLOCKED  # negative-evidence override"},{"language":"text","snippet":"\"Implemented. Commands run: `pytest tests/` → 8 passed. Verification: passed.\" → OK\n\"Read-only audit complete. Files inspected: src/parser.py, tests/test_parser.py. No changes.\" → OK\n\"Status: partial. The scaffolding is drafted. Verification: not run because no fixture exists yet. Next step: write fixture.\" → OK  # honest partial with no completion claim\n\"Reviewing the design — no decision yet.\" → OK"},{"language":"python","snippet":"from crewai_no_vibes import make_guardrail\n\n# Custom feedback message\nguardrail = make_guardrail(\n    custom_feedback=\"Show test output, please.\"\n)\n\n# Stricter mode: block on ANY completion claim without command/verification evidence,\n# regardless of whether a negative-evidence marker is present.\nstrict_guardrail = make_guardrail(require_command_or_verification=True)"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB REPOS","editorialOverview":"CrewAI Task guardrail blocking verification claims without evidence. Pure-Python port of the no-vibes / MAST mode 3.3 detector. F1 0.815 (95% CI [0.615, 0.941]) on the released MAD human-labelled set. crewai-no-vibes CrewAI Task guardrail blocking verification claims without evidence. Pure-Python, zero runtime dependencies, drop-in. Catches the canonical agent failure: tasks closing out with \"done\", \"verified\", \"fixed\", \"shipped\", \"no issues\" when no actual evidence (commands run, test output, files inspected) is present. Empirical baseline no-vibes is the strongest measured detector for $1 — the highest-prevalenc","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":403,"uniquenessScore":66,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:44:44.941Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:44:44.941Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T01:04:02.120Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_repos","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}