{"id":"c28fa377-670d-48ce-b6ba-3d15423e8086","entityType":"agent","slug":"crewai-rxbass-multi-agent-customer-support-buildathon","name":"Multi-Agent-Customer-Support-Buildathon","canonicalUrl":"https://www.xpersona.co/agent/crewai-rxbass-multi-agent-customer-support-buildathon","canonicalPath":"/agent/crewai-rxbass-multi-agent-customer-support-buildathon","generatedAt":"2026-10-09T20:22:59.635Z","source":"GITHUB_REPOS","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T12:48:04.630Z","emptyReason":null},"description":"Customer support system made of three agents that run one after another using the CrewAI framework, with a Streamlit user interface. When a query or task comes in, the first agent answers it directly, the second agent searches the web and answers, and the third agent saves everything to a text file and returns both answers to the user in the UI 🛎️ Crew Desk Support — Self-Checking Multi-Agent Customer Support **One agent answers. One agent verifies. One agent reconciles and records.** Crew Desk Support is a three-agent customer-support system built with **CrewAI + Streamlit**. A user submits a support query; the crew processes it **sequentially**: 1. **Assistant** answers from the LLM's own knowledge. 2. **Web Search Assistant** searches the live web and p","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 10/9/2026.","installCommand":null,"sourceUrl":"https://github.com/rxbass/Multi-Agent-Customer-Support-Buildathon","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/rxbass/Multi-Agent-Customer-Support-Buildathon","kind":"source"}],"safetyScore":66,"overallRank":29.6,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Customer support system made of three agents that run one after another using the CrewAI framework, with a Streamlit user interface. When a query or task comes "},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T12:48:04.630Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-10-09T12:48:04.630Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-10-09T12:48:04.597Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T12:48:04.630Z","lastCrawledAt":"2026-10-09T12:48:04.597Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-16T12:48:04.597Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":null,"setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_REPOS","generatedAt":"2026-10-09T20:22:59.635Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-rxbass-multi-agent-customer-support-buildathon/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"high","updatedAt":"2026-10-09T12:48:04.630Z","emptyReason":null},"readme":"# 🛎️ Crew Desk Support — Self-Checking Multi-Agent Customer Support\n\n> **One agent answers. One agent verifies. One agent reconciles and records.**\n\nCrew Desk Support is a three-agent customer-support system built with **CrewAI + Streamlit**.\nA user submits a support query; the crew processes it **sequentially**:\n\n1. **Assistant** answers from the LLM's own knowledge.\n2. **Web Search Assistant** searches the live web and produces a web-grounded answer with sources.\n3. **Entry Agent (+ Reconciler)** compares both answers, identifies contradictions, produces a resolved response (or refuses when it can't verify), and saves the complete interaction to `answers.txt`.\n\nThe goal is simple: make a **second source of truth visible**, and use the third\nagent to **reconcile the model's answer against live web evidence** before\npresenting a result to the user.\n\nThe implementation follows the buildathon constraints exactly: **exactly three\nagents, sequential execution, a Streamlit UI, environment-based API keys, and a\nsingle `app.py`.**\n\n---\n\n## 1. Why the Third Agent Exists\n\nAn LLM answering from memory can be confidently wrong — it may state a policy or a\nfigure that has since changed. That's the classic support failure: **a wrong answer\ndelivered with total confidence.**\n\nCrew Desk Support doesn't claim to *know* when the model is wrong. It does something\nnarrower and defensible: it **cross-checks the model's answer against live web\nevidence and surfaces any contradiction** between the two. When the web evidence\nconflicts with the model's answer, the third agent flags it, resolves to the\nbetter-supported answer, and — when it can't verify confidently — **says so instead\nof guessing.**\n\n**Worked example (illustrative):**\n\n```\nUser:\n  \"What is the current refund window for Product X?\"\n\nAgent 1 (Assistant, from memory):\n  \"Customers can request a refund within 30 days.\"\n\nAgent 2 (Web Search, from live sources):\n  \"Current documentation states refunds are available within 14 days.\"\n  sources: [https://example.com/refund-policy]\n\nAgent 3 (Entry Agent + Reconciler):\n  contradiction_found:  true\n  severity:             material\n  resolved_answer:      \"The current policy is a 14-day refund window. The model's\n                         direct answer (30 days) conflicts with the live source.\"\n  confidence:           0.82\n  → written to answers.txt\n```\n\nThat single exchange is the whole point of the third agent: without it, the user\ngets the confident-but-stale 30-day answer.\n\n---\n\n## 2. Architecture & Data Flow\n\n```\nUser Query (Streamlit input)\n        │\n        ▼\n┌──────────────────────────┐\n│ Input guardrails         │  small talk → instant reply · length cap\n│ (functions, not an agent)│  · prompt-injection heuristics · PII masking\n│                          │  · OpenAI moderation (free)\n└────────────┬─────────────┘\n             ▼\n┌──────────────────────────┐\n│ Agent 1 — Assistant      │  no tools\n│  answers from memory     │  → direct_answer\n└────────────┬─────────────┘\n             │  context\n             ▼\n┌──────────────────────────┐\n│ Agent 2 — Web Search     │  SerperDevTool (this agent only)\n│  live web + sources      │  → web_answer + source URLs\n└────────────┬─────────────┘\n             │  context = [task1, task2]\n             ▼\n┌──────────────────────────┐\n│ Agent 3 — Entry Agent    │  file-writer tool\n│  + Reconciler            │  • compare both answers\n│                          │  • detect contradiction + severity\n│                          │  • resolve, or refuse if unverified\n│                          │  • write answers.txt\n└────────────┬─────────────┘\n             ▼\n┌──────────────────────────┐\n│ Output moderation        │  each answer checked before display\n└────────────┬─────────────┘\n             ▼\n       Streamlit UI  (direct · web + sources · resolved)\n```\n\n**The mechanism the evaluator will look for** — in a sequential crew, each task\nreceives prior outputs through its `context`. The reconciler declares both earlier\ntasks as context, which is the only way it can compare and persist both answers:\n\n```python\ntask3 = Task(\n    description=\"Compare the direct answer and the web answer, detect any \"\n                \"contradiction, resolve to the best-supported answer (or refuse \"\n                \"if it cannot be verified), then save the record to answers.txt.\",\n    agent=entry_agent,\n    context=[task1, task2],          # ← receives query + both prior answers\n    output_pydantic=ReconciledAnswer,\n)\n```\n\nThe web-search tool is assigned to **Agent 2 only** — so Agent 1 answers purely from\nmemory (the fallible source being checked), and no extra tool schemas inflate the\nother agents' token cost.\n\n---\n\n## 3. The Three Agents\n\n| # | Agent | Tools | Produces |\n|---|-------|-------|----------|\n| 1 | **Assistant** | none | `direct_answer` — from model knowledge alone |\n| 2 | **Web Search Assistant** | `SerperDevTool` | `web_answer` + preserved source URLs |\n| 3 | **Entry Agent + Reconciler** | file-writer | `ReconciledAnswer`; writes `answers.txt` |\n\n**Agent 2 is explicitly instructed to preserve real source URLs** from the search\nresults and never to invent them — the grounding claim depends on this.\n\n---\n\n## 4. Structured Verdict (Agent 3 output)\n\nAgent 3's task uses `output_pydantic`, so its verdict is structured JSON — the UI\nstays robust and the output is machine-checkable.\n\n```python\nfrom pydantic import BaseModel\nfrom typing import Literal\n\nclass ReconciledAnswer(BaseModel):\n    query: str\n    direct_answer: str = \"\"                               # Agent 1's answer, filled by app.py\n    web_answer: str = \"\"                                  # Agent 2's answer, filled by app.py\n    sources: list[str]                                    # Agent 2's real URLs\n    contradiction_found: bool                             # web conflicts with model?\n    contradiction_severity: Literal[\"none\", \"minor\", \"material\"]\n    contradiction_detail: str                             # what differed (\"\" if none)\n    resolved_answer: str                                  # best-supported answer, OR the\n                                                          # refusal string when unverified\n    confidence: float                                     # agent-ASSESSED, 0.0–1.0\n```\n\n**Why two fields are filled by the app, not the model.** `direct_answer` and\n`web_answer` are the prior answers *verbatim*. Making Agent 3 reproduce them — once\nas tool arguments and again in the JSON — cost ~300 output tokens at the median and\n~1,800 at the worst, and since output tokens are generated serially that was the\nsingle reason the reconciler dominated latency (and sometimes truncated its own\nJSON mid-copy). Now the file tool pulls both answers straight from the earlier task\noutputs, and `app.py` fills these two fields from `result.tasks_output[i].raw`. The\nrecord and the verdict are byte-exact rather than an LLM transcription, and every\njudgement field — `contradiction_found`, `contradiction_severity`,\n`contradiction_detail`, `resolved_answer`, `confidence` — is still entirely the\nagent's.\n\n**`confidence` is agent-assessed, not objective reliability.** It drives a defined,\nreproducible rule:\n\n```\nconfidence < 0.60  →  resolved_answer = \"I couldn't verify this confidently.\"\n```\n\nThat threshold is what makes the closed-fail behavior reproducible rather than a\nmatter of the model's mood.\n\n---\n\n## 5. Guards & Reliability\n\n| Guard | Prevents | How |\n|-------|----------|-----|\n| `max_iter` (e.g. 3) | Web agent looping on an elusive query, draining credits | Set per agent |\n| `max_rpm` | Rate-limit crash mid-run | Set per agent |\n| Tool scoping | Token bloat + Agent 1 \"cheating\" by searching | Serper on Agent 2 only |\n| File-writer **tool** | Unreliable \"please write the file\" instructions | Agent 3 writes via a real tool, not a prompt request. If a run ends without the tool having been called, the app persists Agent 3's structured output itself and the UI footer says \"saved by app fallback\" |\n| Graceful degradation | A failed web search crashing the app | Fall back to `direct_answer`, and say so |\n| Confidence threshold | Confident wrong answers | `< 0.60` → explicit refusal |\n| `output_pydantic` | UI breaking on messy LLM text | Structured verdict |\n| No verbatim re-typing | Agent 3 dominating latency and truncating its own JSON while copying both answers | File tool pulls the prior answers; `app.py` fills the two provenance fields |\n| Progressive display | A blank spinner for the whole run | Agent 1's answer is shown as soon as it lands, then Agent 2's, then the reconciled verdict |\n| Input guardrails | Prompt injection, PII leakage, abusive queries | Regex heuristics + PII masking + OpenAI moderation, before the crew runs |\n| Output moderation | Unsafe text reaching the customer | Every answer moderated before display; flagged text withheld |\n| Per-agent timeout | A stalled agent hanging the UI | `max_execution_time=90` → clean error |\n| Retry cap | CrewAI's default `max_retry_limit=2` re-running a timed-out task, so one stalled agent costs 3× the timeout | `max_retry_limit=1` → worst case 180 s, not 450 s |\n| `max_rpm=30`, not 10 | CrewAI's RPM controller does a blind `time.sleep(60)` when the limit is hit, and it is consulted on every LLM call *and* tool step — a low cap silently added a minute to ordinary queries | Set high enough that a single query never trips it, while still bounding a runaway loop |\n\n<details>\n<summary><b>Guardrails in detail</b> (click to expand)</summary>\n\nAll guardrails are plain functions in `app.py` wrapped around the crew by\n`guarded_run()` — the crew is still exactly three agents. Both the UI and\nevaluation mode go through the same entry point.\n\n| Stage | What it does | What the user sees |\n|-------|--------------|--------------------|\n| **Small talk** | \"hi\", \"thanks\", \"bye\", \"how are you\"… answered with a canned greeting | 👋 instant reply, crew not run, nothing billed |\n| **Length cap** | Rejects queries over 1,000 characters | 🛡️ blocked, reason shown, crew not run |\n| **Prompt-injection heuristics** | Regex for instruction-override (\"ignore previous instructions\"), system-prompt extraction, role/chat-format spoofing (`### system:`), raw model tokens (`<\\|im_start\\|>`), and tool hijacking (\"use the save tool to write … to x.txt\") | 🛡️ blocked, matched phrase shown, crew not run |\n| **PII masking** | Emails, Luhn-valid card numbers, SSNs, IBANs and phone numbers are replaced with `[EMAIL]`, `[CARD]`, … *before* the query reaches the LLM, Serper, or `answers.txt`. Order IDs like `#A-4471` are kept. | 🛡️ info note listing what was masked; query still runs |\n| **Input moderation** | OpenAI `omni-moderation-latest` (free endpoint) on the masked query | 🛡️ blocked with the categories (e.g. harassment, violence) |\n| **Output moderation** | Same endpoint on the direct, web and resolved answers | Flagged text replaced by `[Withheld by output moderation: …]`; a 🛡️ note says which answer |\n\nHonest limits: injection detection is heuristic (a determined attacker can\nrephrase); PII regexes cover common formats, not every national ID; moderation\n**fails open** — if the endpoint is unreachable the check is skipped and a 🛡️\nwarning says so, rather than the app refusing to work. **Deliberate tradeoff:**\nAgent 3 writes `answers.txt` via its tool *before* output moderation runs, so the\nrecord preserves the raw agent decision (PII-masked, since masking happens before\nthe crew). Moderating first would mean the app, not the agent, writes the file —\nthe spec prefers the agent-tool path. `answers.txt` is an audit record, not a\ncustomer-facing sanitized transcript.\n\n</details>\n\n---\n\n## 6. Setup & Run (step by step)\n\nTested on Python 3.13; any Python 3.10+ should work. Commands are shown for\nmacOS/Linux (bash) and Windows (PowerShell).\n\n### Step 1 — Clone the repository\n\n```bash\ngit clone <this-repo-url>\ncd Multi-Agent-Customer-Support-Buildathon\n```\n\n### Step 2 — Create and activate a virtual environment\n\n```bash\n# macOS / Linux\npython -m venv venv\nsource venv/bin/activate\n```\n\n```powershell\n# Windows (PowerShell)\npython -m venv venv\n.\\venv\\Scripts\\Activate.ps1\n```\n\nYour prompt should now start with `(venv)`.\n\n### Step 3 — Install the pinned dependencies\n\n```bash\npip install -r requirements.txt\n```\n\n`requirements.txt` pins `crewai` and `crewai-tools` to a matching pair\n(`1.15.22`) — install them together, not separately.\n\n### Step 4 — Provide the API keys (environment variables only)\n\nYou need two keys: an **OpenAI** key (all three agents) and a **Serper** key\n(Agent 2's web search — free tier at [serper.dev](https://serper.dev)).\n\nEither export them in the shell:\n\n```bash\n# macOS / Linux\nexport OPENAI_API_KEY=\"sk-...\"\nexport SERPER_API_KEY=\"...\"\n```\n\n```powershell\n# Windows (PowerShell)\n$env:OPENAI_API_KEY=\"sk-...\"\n$env:SERPER_API_KEY=\"...\"\n```\n\n…or copy the template and fill in a local `.env` file (gitignored, loaded into\n`os.environ` on startup):\n\n```bash\ncp .env.example .env      # Windows: Copy-Item .env.example .env\n# then edit .env\n```\n\nIf either key is missing, the app stops with a message naming the missing\nvariable — it never falls back to a hard-coded key.\n\n| Var | For |\n|-----|-----|\n| `OPENAI_API_KEY` | All three agents (`gpt-4o-mini`) |\n| `SERPER_API_KEY` | Agent 2's web search |\n| `CREWDESK_RECONCILER_MODEL` | *(optional)* Agent 3's model, default `gpt-4o-mini`. Set to `gpt-4o` for stronger contradiction judgement (note: entry-tier OpenAI accounts cap `gpt-4o` at 30k TPM) |\n\n### Step 5 — Run the app\n\n```bash\nstreamlit run app.py\n```\n\nStreamlit opens `http://localhost:8501`. Type a support question (e.g.\n*\"What is the latest stable version of Python?\"*) and click **Ask**. The crew runs\nAssistant → Web Search → Entry Agent, typically 10–30 s in total, and the answers\nappear **as they land**: the direct answer within a few seconds (labelled *not yet\nverified*), then the web answer, then the reconciled verdict with a confidence bar\nand the two answers side by side. Greetings like \"hi\" are answered instantly\nwithout running the crew.\n\nEarlier questions from the same session stay on the page under **Earlier in this\nsession**, collapsed and tagged with their outcome — display only; each question is\nstill an independent crew run. Every run appends a record to `answers.txt` in the\nrepo root, and the terminal shows the verbose agent trace.\n\n### Step 6 — Run the evaluation (optional)\n\n1. With the app running, open the **sidebar** (arrow at top-left).\n2. Tick **Run evaluation**.\n3. The app loads `golden_set.json` (16 labelled queries: 6 catch, 6 control,\n   4 refusal), runs the full crew on each item, and reports **catch rate**,\n   **false-positive rate**, refusal accuracy, and a one-line latency/cost figure.\n\nThe evaluation runs 16 crew executions, so expect several minutes and roughly\n16× the cost of a single query. Section 7 shows the numbers from one such run;\nyour run will differ somewhat — this is an LLM system, not a unit test.\n\n<details>\n<summary><b>Troubleshooting</b> (click to expand)</summary>\n\n| Symptom | Fix |\n|---------|-----|\n| `Missing environment variable(s): OPENAI_API_KEY …` | Set the key (Step 4) and restart Streamlit |\n| `ImportError: cannot import name 'SerperDevTool'` | You have a mismatched `crewai-tools`; re-run `pip install -r requirements.txt` |\n| Web answer says \"Sources: none\" | Serper returned nothing usable; the Entry Agent will lower confidence or refuse |\n| Terminal shows `429 … tokens per min (TPM)` | OpenAI rate limit; wait a minute or raise the eval pause. Most likely if you set Agent 3 to `gpt-4o` on an entry-tier account |\n\n</details>\n\n---\n\n## 7. Optional: Evaluation Mode\n\nThe assignment doesn't require evaluation — but to show the reconciler actually\nearns its place, `app.py` includes an **optional evaluation mode** (a sidebar\ntoggle). It runs the crew over a small labelled set (`golden_set.json`) and reports\ntwo headline numbers:\n\n| Metric | Question it answers |\n|--------|---------------------|\n| **Catch rate** | Of the stale-answer cases the crew *should* flag, how many did it flag as a material contradiction? |\n| **False-positive rate** | On cases where the model was already correct, how often did it wrongly cry contradiction? Reported twice: *any* contradiction (the golden-set definition) and `material` only (what the system is for). |\n\n*(Latency and token cost per run are shown as a single line for context.)*\n\nThe golden set mixes three case types on purpose — **catches** (model likely stale,\nweb corrects it), **controls** (model already correct, expect no contradiction), and\n**refusals** (unanswerable without private data, expect the closed-fail response) —\nso the catch rate is reported *alongside* its false-positive rate. A two-sided\nnumber is the honest one.\n\n### Results — 2026-09-20, all agents `gpt-4o-mini`\n\n> **Note:** these numbers were measured before the reconciler latency fix (the two\n> provenance fields are no longer retyped by Agent 3) and before the timeout/retry\n> caps were tightened. The verdict behaviour is unchanged in kind, but a re-run\n> would show lower latency and fewer stalls. Re-running evaluation mode regenerates\n> `eval_results.json` and these figures.\n\nRaw per-item output is in [`eval_results.json`](eval_results.json); its `summary`\nblock is computed by the same `summarize()` the sidebar uses, so **the numbers\nbelow are exactly what the app's evaluation mode shows** — nothing is adjusted by\nhand. 16 cases: 6 catch, 6 control, 4 refusal. **14 runs produced a verdict; 2\ntimed out** at the per-agent cap (150 s in that run; since lowered to 90 s with a\nsingle retry after this was found to be a latency problem) and are excluded from\nevery rate.\n\n**Provenance, stated plainly:** 15 rows come from one full 16-item run. The original\n`catch-02` (\"current monthly price of ChatGPT Plus\") turned out not to be a stale\ncase — the model's remembered price ($20) was still current, so nothing could be\ncaught. It was replaced with \"What is the latest Ubuntu LTS release?\" (a fact that\nchanges on a fixed two-year cadence) and that one item was run separately the same\nday. The replaced row is kept in `eval_results.json` under\n`_about.replaced_row_original`, and the replacement is noted on the item in\n`golden_set.json`.\n\n| Metric | Result | Reading |\n|--------|--------|---------|\n| **Material-contradiction catch rate** | **6 / 6 = 100 %** | Python, Ubuntu LTS, iPhone, Node LTS, OpenAI limits, CrewAI version — all flagged `material` at confidence ≥ 0.9. |\n| **False-positive rate, any contradiction** (golden-set definition) | **2 / 5 = 40 %** | Both `minor` (password reset, strong-password tips): the web answer added platform-specific detail and the reconciler called it a discrepancy. |\n| **False-positive rate, `material`** | **0 / 5** | No material contradiction on the scored control cases. |\n| **Refusals correct** | **3 / 3 completed = 100 %** (1 timed out) | All three scored confidence 0.3 → refusal string shown. `refusal-03` also reported a `material` contradiction between two generic answers — an over-flag hidden behind the refusal. |\n| Timed-out runs | 2 / 16 | `control-05` (\"What is a VPN\"), `refusal-04`. Both stalled in Agent 3; **cause not diagnosed** (traces were not inspected). The per-agent cap turned each into a clean error instead of a hang. |\n| **Avg latency / cost per completed run** | **21.7 s · ~9,300 tokens · ≈ $0.002** | n = 14 completed verdicts; timed-out runs excluded. `gpt-4o-mini` list prices. |\n\nPer item:\n\n| id | category | outcome | severity | confidence | time |\n|----|----------|---------|----------|------------|------|\n| catch-01 | catch | caught | material | 1.0 | 14.4 s |\n| catch-02 † | catch | caught | material | 1.0 | 14.5 s |\n| catch-03 | catch | caught | material | 1.0 | 16.5 s |\n| catch-04 | catch | caught | material | 1.0 | 16.6 s |\n| catch-05 | catch | caught | material | 0.9 | 31.6 s |\n| catch-06 | catch | caught | material | 0.9 | 19.2 s |\n| control-01 | control | false positive (minor) | minor | 0.8 | 32.2 s |\n| control-02 | control | clean | none | 1.0 | 23.1 s |\n| control-03 | control | clean | none | 1.0 | 14.6 s |\n| control-04 | control | false positive (minor) | minor | 0.9 | 27.3 s |\n| control-05 | control | timed out | — | — | 180 s |\n| control-06 | control | clean | none | 1.0 | 49.0 s |\n| refusal-01 | refusal | refused | none | 0.3 | 13.8 s |\n| refusal-02 | refusal | refused | none | 0.3 | 11.7 s |\n| refusal-03 | refusal | refused (but flagged material) | material | 0.3 | 19.2 s |\n| refusal-04 | refusal | timed out | — | — | 180 s |\n\n† replacement item, run separately the same day (see provenance above).\n\nWhat this run says: in this evaluation set the reconciler caught all six cases\nwhere the model's answer was stale, and produced no `material` false positive on\nthe scored control cases — but `gpt-4o-mini` is better at spotting factual\nconflicts than at distinguishing them from harmless differences in specificity\n(two `minor` over-flags on controls, one `material` over-flag on a refusal case).\nAgent 3 is also the latency bottleneck: 2 of 16 runs reached the execution limit.\nIn an **informal 5-case spot check**, `gpt-4o` as Agent 3\n(`CREWDESK_RECONCILER_MODEL=gpt-4o`) did not produce the minor over-flags; that\nsample is far too small to treat as a benchmark, and `gpt-4o` carries a low\nper-minute token cap on entry-tier OpenAI accounts. A single 16-item run is a\ndirection, not a benchmark.\n\n---\n\n## 8. Repository\n\n```\napp.py            # the complete application — 3 sequential agents, one file\ngolden_set.json   # labelled queries for optional evaluation mode (data, not code)\neval_results.json # measured evaluation rows + summary (the numbers in section 7)\nrequirements.txt  # pinned dependencies (crewai + crewai-tools as a matching pair)\n.streamlit/config.toml  # UI theme (light, brand colours) — config, not code\nanswers.txt       # generated — query + both answers + verdict, per run (gitignored)\n.env.example      # environment-variable template (real keys in .env, gitignored)\nREADME.md         # this file\n```\n\nAll application code lives in `app.py`, per the spec. `golden_set.json` is evaluation\n*data*, and evaluation runs as a mode *inside* `app.py` — there is no separate code\nfile.\n\n---\n\n## 9. Design Decisions\n\n- **Agent 3 is the Entry Agent *and* a reconciler.** The spec's Entry Agent is\n  usually a five-line file-writer. Keeping it a single agent but having it reconcile\n  the two prior answers adds real depth without a fourth agent — the spec is\n  followed to the letter.\n- **The claim is precise.** The system doesn't \"know\" the model is wrong; it\n  cross-checks against live web evidence and surfaces contradictions. That's what it\n  actually does, and what it can defend.\n- **Agent 3 writes via a file tool, not an instruction.** \"Please write the file\" is\n  unreliable; a real tool guarantees the record is saved.\n- **Serper on Agent 2 only.** Agent 1 must answer from memory to be the source under\n  check; tool scoping also cuts token cost.\n- **Confidence is agent-assessed with a fixed 0.60 threshold** — so \"refuse when\n  unsure\" is reproducible, not vibes.\n- **Agent 2 is isolated from Agent 1's answer.** In a CrewAI sequential crew, a\n  task with no `context` declared receives *every* prior output. Left at the\n  default, the web agent saw the model's answer before searching and simply\n  restated it — so Task 2 sets `context=[]`, and only Task 3 gets both answers.\n  Without this the comparison is meaningless.\n- **`gpt-4o-mini` for all three agents.** Early runs showed the reconciler\n  reading \"more detailed\" as \"contradicts\" and scoring private-data queries above\n  the refusal threshold; that was fixed in the Task 3 prompt (a contradiction is\n  *incompatible facts only*; confidence measures whether the customer's question\n  was answered). `CREWDESK_RECONCILER_MODEL=gpt-4o` is available for Agent 3 if\n  judgement quality needs it (informal 5-case spot check only), at the cost of a\n  low per-minute token cap on entry-tier accounts.\n- **No agent memory, by design.** CrewAI memory stays off (`Crew(memory=False)`,\n  the default) and a fresh crew is built per query, so each query is an\n  independent trial. If Agent 1 remembered a previous run's web-corrected answer,\n  it would stop being the fallible source under check and the contradiction would\n  vanish — the evaluation would measure nothing. The only persistence is the\n  `answers.txt` audit record and Streamlit session state for re-rendering the UI,\n  neither of which is fed back into the agents. The **\"Earlier in this session\"**\n  transcript on the page is display-only: each question is still an independent\n  crew run that receives nothing but that question.\n- **The reconciler never retypes what it was given.** The measured cost of the two\n  carried-forward fields was ~300 output tokens per query (worst case ~1,800), all\n  of it serial generation. The tool now pulls both prior answers from the task\n  outputs and the app fills those two schema fields, so Agent 3 only writes its\n  own judgement. Faster, and byte-exact instead of an LLM copy.\n- **Partial answers stream to the UI.** The direct answer appears a few seconds in,\n  labelled *not yet verified*, then the web answer, then the reconciled verdict —\n  so the wait is visible progress rather than a blank spinner.\n- **`output_pydantic` everywhere it matters** — structured verdict, robust UI,\n  scorable output.\n- **`max_iter` / `max_rpm` caps** — cheap insurance against the classic multi-agent\n  failure of a looping search agent.\n- **Guardrails are functions, not an agent.** Injection heuristics, PII masking and\n  moderation wrap the crew in `guarded_run()`; adding a \"safety agent\" would break\n  the three-agent constraint and cost an LLM call per query — the moderation\n  endpoint is free and the regexes are instant.\n- **One sequential crew, exactly three agents, one `app.py`.** No Flows, no fourth\n  agent, no second code file.\n\n---\n\n## 10. Limitations\n\n- Agent 1's knowledge reflects the model's training cut-off; the crew's value is in\n  catching where that's stale — but it can only catch what Agent 2's search surfaces.\n- Contradiction detection is only as good as the web answer; a weak search yields a\n  weak `web_answer`, which the confidence/refusal path is there to handle.\n- `confidence` is the agent's self-assessment, not a calibrated probability — hence\n  the fixed threshold and the honest label.\n- Guardrails are heuristic + moderation-API, not a security boundary: injection\n  regexes can be evaded by rephrasing, PII masking covers common formats only, and\n  moderation fails open when the endpoint is unavailable (the UI says so).\n- The golden set is small and hand-labelled; results show direction and consistency,\n  not a precise percentage. Labelling rationale lives in `golden_set.json` so every\n  number is traceable.\n\n---\n\n*Built for the Weekly Buildathon — Multi-Agent Customer Support (CrewAI + Streamlit).*\n","readmeExcerpt":"🛎️ Crew Desk Support — Self-Checking Multi-Agent Customer Support **One agent answers. One agent verifies. One agent reconciles and records.** Crew Desk Support is a three-agent customer-support system built with **CrewAI + Streamlit**. A user submits a support query; the crew processes it **sequentially**: 1. **Assistant** answers from the LLM's own knowledge. 2. **Web Search Assistant** searches the live web and p","codeSnippets":[],"executableExamples":[{"language":"text","snippet":"User:\n  \"What is the current refund window for Product X?\"\n\nAgent 1 (Assistant, from memory):\n  \"Customers can request a refund within 30 days.\"\n\nAgent 2 (Web Search, from live sources):\n  \"Current documentation states refunds are available within 14 days.\"\n  sources: [https://example.com/refund-policy]\n\nAgent 3 (Entry Agent + Reconciler):\n  contradiction_found:  true\n  severity:             material\n  resolved_answer:      \"The current policy is a 14-day refund window. The model's\n                         direct answer (30 days) conflicts with the live source.\"\n  confidence:           0.82\n  → written to answers.txt"},{"language":"text","snippet":"User Query (Streamlit input)\n        │\n        ▼\n┌──────────────────────────┐\n│ Input guardrails         │  small talk → instant reply · length cap\n│ (functions, not an agent)│  · prompt-injection heuristics · PII masking\n│                          │  · OpenAI moderation (free)\n└────────────┬─────────────┘\n             ▼\n┌──────────────────────────┐\n│ Agent 1 — Assistant      │  no tools\n│  answers from memory     │  → direct_answer\n└────────────┬─────────────┘\n             │  context\n             ▼\n┌──────────────────────────┐\n│ Agent 2 — Web Search     │  SerperDevTool (this agent only)\n│  live web + sources      │  → web_answer + source URLs\n└────────────┬─────────────┘\n             │  context = [task1, task2]\n             ▼\n┌──────────────────────────┐\n│ Agent 3 — Entry Agent    │  file-writer tool\n│  + Reconciler            │  • compare both answers\n│                          │  • detect contradiction + severity\n│                          │  • resolve, or refuse if unverified\n│                          │  • write answers.txt\n└────────────┬─────────────┘\n             ▼\n┌──────────────────────────┐\n│ Output moderation        │  each answer checked before display\n└────────────┬─────────────┘\n             ▼\n       Streamlit UI  (direct · web + sources · resolved)"},{"language":"python","snippet":"task3 = Task(\n    description=\"Compare the direct answer and the web answer, detect any \"\n                \"contradiction, resolve to the best-supported answer (or refuse \"\n                \"if it cannot be verified), then save the record to answers.txt.\",\n    agent=entry_agent,\n    context=[task1, task2],          # ← receives query + both prior answers\n    output_pydantic=ReconciledAnswer,\n)"},{"language":"python","snippet":"from pydantic import BaseModel\nfrom typing import Literal\n\nclass ReconciledAnswer(BaseModel):\n    query: str\n    direct_answer: str = \"\"                               # Agent 1's answer, filled by app.py\n    web_answer: str = \"\"                                  # Agent 2's answer, filled by app.py\n    sources: list[str]                                    # Agent 2's real URLs\n    contradiction_found: bool                             # web conflicts with model?\n    contradiction_severity: Literal[\"none\", \"minor\", \"material\"]\n    contradiction_detail: str                             # what differed (\"\" if none)\n    resolved_answer: str                                  # best-supported answer, OR the\n                                                          # refusal string when unverified\n    confidence: float                                     # agent-ASSESSED, 0.0–1.0"},{"language":"text","snippet":"confidence < 0.60  →  resolved_answer = \"I couldn't verify this confidently.\""},{"language":"bash","snippet":"git clone <this-repo-url>\ncd Multi-Agent-Customer-Support-Buildathon"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB REPOS","editorialOverview":"Customer support system made of three agents that run one after another using the CrewAI framework, with a Streamlit user interface. When a query or task comes in, the first agent answers it directly, the second agent searches the web and answers, and the third agent saves everything to a text file and returns both answers to the user in the UI 🛎️ Crew Desk Support — Self-Checking Multi-Agent Customer Support **One agent answers. One agent verifies. One agent reconciles and records.** Crew Desk Support is a three-agent customer-support system built with **CrewAI + Streamlit**. A user submits a support query; the crew processes it **sequentially**: 1. **Assistant** answers from the LLM's own knowledge. 2. **Web Search Assistant** searches the live web and p","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":452,"uniquenessScore":57,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T12:48:04.630Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T12:48:04.630Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T20:22:59.635Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_repos","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}