{"id":"40bec3fb-a97d-4cbd-a64e-acc0529291b4","entityType":"agent","slug":"crewai-steosumit-customer-claim-agent","name":"customer-claim-agent","canonicalUrl":"https://www.xpersona.co/agent/crewai-steosumit-customer-claim-agent","canonicalPath":"/agent/crewai-steosumit-customer-claim-agent","generatedAt":"2026-10-10T02:02:18.022Z","source":"GITHUB_REPOS","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-10-09T18:14:39.280Z","emptyReason":null},"description":"Agent processes user made claims with pictures and detects and classifies them as contradictory, true, or lack of information. Made as a part of Hackerrank Orchestrate Hackathon. Powered by CrewAI and Ollama cloud Gemma3 and Gemma4 models Multi-Modal Evidence Review — HackerRank Orchestrate Submission A multi-agent system that verifies visual evidence for damage claims across **cars**, **laptops**, and **packages**. Built with CrewAI Flows and a dual-model architecture using local Ollama instances. Read $1 for the full task spec, input/output schema, and allowed values. --- Quick Start Produces output.csv in the repo root. --- Setup Prerequisites - Py","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 10/9/2026.","installCommand":null,"sourceUrl":"https://github.com/Steosumit/customer-claim-agent","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/Steosumit/customer-claim-agent","kind":"source"}],"safetyScore":66,"overallRank":18.6,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Agent processes user made claims with pictures and detects and classifies them as contradictory, true, or lack of information. Made as a part of Hackerrank Orch"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-10-09T18:14:39.280Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-10-09T18:14:39.280Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-10-09T18:14:39.273Z","emptyReason":null},"lastUpdatedAt":"2026-10-09T18:14:39.280Z","lastCrawledAt":"2026-10-09T18:14:39.273Z","lastIndexedAt":null,"nextCrawlAt":"2026-10-16T18:14:39.273Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":null,"setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_REPOS","generatedAt":"2026-10-10T02:02:18.021Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-steosumit-customer-claim-agent/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB REPOS","verified":false,"confidence":"high","updatedAt":"2026-10-09T18:14:39.280Z","emptyReason":null},"readme":"# Multi-Modal Evidence Review — HackerRank Orchestrate Submission\n\nA multi-agent system that verifies visual evidence for damage claims across **cars**, **laptops**, and **packages**. Built with CrewAI Flows and a dual-model architecture using local Ollama instances.\n\n![architecture diagram](architecture.png)\n\nRead [`problem_statement.md`](./problem_statement.md) for the full task spec, input/output schema, and allowed values.\n\n---\n\n## Quick Start\n\n```bash\ngit clone <repo-url>\ncd hackerrank-orchestrate-june26\n\n# Create virtual environment\npython3.12 -m venv venv\nsource venv/bin/activate\n\n# Install dependencies\npip install crewai crewai-tools pyyaml python-dotenv requests\n\n# Configure .env (see Setup section below)\ncp .env.example .env  # or create manually\n\n# Run the system\npython code/main.py\n```\n\nProduces `output.csv` in the repo root.\n\n---\n\n## Setup\n\n### Prerequisites\n\n- Python 3.12 (Python 3.14 is incompatible with CrewAI dependencies)\n- [Ollama](https://ollama.ai) running locally\n- ~8 GB RAM for running both models concurrently\n\n### Ollama Configuration\n\nTwo Ollama instances run in parallel on different ports:\n\n| Instance | Port | Model | Purpose |\n|----------|------|-------|---------|\n| Primary | `11434` | `gemma4:31b-cloud` | Vision analysis (images) |\n| Secondary | `11435` | `gemma3:27b-cloud` | Text processing (extraction, synthesis) |\n\nStart the second instance in a separate terminal:\n\n```bash\nOLLAMA_HOST=127.0.0.1:11435 ollama serve\n```\n\nPull the required models:\n\n```bash\nollama pull gemma4:31b-cloud\nollama pull gemma3:27b-cloud\n```\n\n### Environment Variables (`.env`)\n\n```env\n# Ollama endpoints\nOLLAMA_MODEL_GEMMA4_URL=http://127.0.0.1:11434\nOLLAMA_MODEL_GEMMA3_URL=http://127.0.0.1:11435\n\n# Text agents — Gemma3 (port 11435)\nMODEL_CLAIM_EXTRACTOR=ollama/gemma3:27b-cloud\nMODEL_SYNTHESIZER=ollama/gemma3:27b-cloud\n\n# Vision agents — Gemma4 (port 11434)\nMODEL_VISUAL_JUDGE=ollama/gemma4:31b-cloud\nMODEL_QUALITY_JUDGE=ollama/gemma4:31b-cloud\n\n# Cross-check agent — Gemma3 (different model for jury diversity)\nMODEL_AUTHENTICITY_JUDGE=ollama/gemma3:27b-cloud\n```\n\n### Running\n\n```bash\nsource venv/bin/activate\npython code/main.py\n```\n\nFor evaluation on sample data:\n\n```bash\npython code/evaluation/main.py --max-rows 20\n```\n\n---\n\n## Approach\n\n### Problem\n\nGiven a claim conversation, submitted images, user history, and evidence requirements, determine whether the images support, contradict, or fail to prove the damage claim.\n\n### Architecture: Parallel Jury with Synthesis\n\nThe system uses a **3-phase pipeline** orchestrated by CrewAI Flows:\n\n```\nPhase 1: Claim Extraction (Gemma3 — text-only)\n    Parse conversation → issue_type, object_part, claim_description\n                        │\n         ┌──────────────┼──────────────┐\n         ▼              ▼              ▼\n   ┌───────────┐  ┌───────────┐  ┌───────────────┐\n   │  Visual   │  │  Quality  │  │ Authenticity  │\n   │  Judge    │  │  Judge    │  │    Judge      │\n   │ (Gemma4)  │  │ (Gemma4)  │  │  (Gemma3)     │\n   │           │  │           │  │               │\n   │ Analyze   │  │ Check     │  │ Cross-check   │\n   │ images    │  │ image     │  │ for           │\n   │ for       │  │ quality   │  │ manipulation  │\n   │ damage    │  │ & validity│  │ & mismatches  │\n   └─────┬─────┘  └─────┬─────┘  └───────┬───────┘\n         │              │                 │\n         └──────────────┼─────────────────┘\n                        │  and_() gate (waits for all 3)\n                        ▼\nPhase 3: Synthesis (Gemma3 — text-only)\n    Merge all jury outputs → final verdict\n    Aggregate risk_flags from all judges\n```\n\n### Why Two Models?\n\n| Model | Role | Rationale |\n|-------|------|-----------|\n| **Gemma4 31b** | Visual judges | Vision-capable, analyzes submitted images for damage evidence |\n| **Gemma3 27b** | Extraction, synthesis, authenticity cross-check | Text-only, provides independent \"second opinion\" to reduce correlated errors |\n\nThe **authenticity judge** uses a different model than the visual judges. This adds **jury diversity** — it cross-checks the vision model's findings independently, reducing the chance of systematic bias from a single model.\n\n### Key Design Decisions\n\n1. **Direct Ollama API for vision** — CrewAI's Ollama provider routes through an OpenAI-compatible endpoint that doesn't support image input. Vision agents call Ollama's native `/api/chat` endpoint directly with base64-encoded images.\n\n2. **`_JudgeCollector` pattern** — CrewAI Flows clones Pydantic state per-method during async execution. A thread-safe module-level collector (`flow_id` → `{key: value}`) bypasses this state isolation so the synthesis step can read outputs from parallel vision judges.\n\n3. **Magic-byte image filtering** — Many `.jpg` files in the dataset are actually MP4/RIFF video containers that Ollama rejects. `_is_valid_image()` checks for JPEG (`ff d8`) or PNG (`89 50 4e 47`) file headers before sending.\n\n4. **Retry with exponential backoff** — Transient Ollama errors are retried up to 2 times with exponential backoff.\n\n5. **Prompt injection defense** — The authenticity judge detects instruction-like text in claims (e.g., \"ignore all previous instructions\") and flags `text_instruction_present` in risk_flags.\n\n### Evaluation Results\n\nTested on `dataset/sample_claims.csv` (20 labeled rows):\n\n| Field | Accuracy |\n|-------|----------|\n| object_part | 85% |\n| evidence_standard_met | 75% |\n| claim_status | 65% |\n| issue_type | 50% |\n| valid_image | 45% |\n| severity | 40% |\n| **Overall** | **60%** |\n\nRuntime: ~550s for 20 rows (~27.5s/row).\n\n---\n\n## Repository Layout\n\n```\n.\n├── AGENTS.md                         # AI coding tool rules + transcript logging\n├── problem_statement.md              # Full task description and I/O schema\n├── README.md                         # You are here\n├── .env                              # Model endpoints and API keys (not committed)\n├── code/\n│   ├── main.py                       # Batch runner entry point\n│   ├── config.py                     # Loads .env + YAML, resolves model URLs\n│   ├── config/\n│   │   ├── agents.yaml               # Agent definitions (role, goal, model)\n│   │   └── tasks.yaml                # Task definitions with structured prompts\n│   ├── data/\n│   │   ├── loader.py                 # CSV loading, ClaimRow/ClaimContext dataclasses\n│   │   └── writer.py                 # output.csv writer (exact 14-column schema)\n│   ├── crews/\n│   │   └── jury_crew.py              # ClaimReviewFlow + helpers + run_pipeline()\n│   └── evaluation/\n│       ├── main.py                   # Evaluation entry point\n│       ├── eval_results.json         # Per-field accuracy metrics\n│       └── report.md                 # Human-readable eval report\n├── output.csv                        # Generated predictions (44 rows)\n└── dataset/\n    ├── sample_claims.csv             # Labeled examples for dev/eval\n    ├── claims.csv                    # Input-only rows (final submission)\n    ├── user_history.csv              # Historical claim data\n    ├── evidence_requirements.csv     # Minimum evidence rules\n    └── images/\n        ├── sample/                   # Images for sample_claims.csv\n        └── test/                     # Images for claims.csv\n```\n\n---\n\n## Submission Result\n\nRanked #604 worldwide \n","readmeExcerpt":"Multi-Modal Evidence Review — HackerRank Orchestrate Submission A multi-agent system that verifies visual evidence for damage claims across **cars**, **laptops**, and **packages**. Built with CrewAI Flows and a dual-model architecture using local Ollama instances. Read $1 for the full task spec, input/output schema, and allowed values. --- Quick Start Produces output.csv in the repo root. --- Setup Prerequisites - Py","codeSnippets":[],"executableExamples":[{"language":"bash","snippet":"git clone <repo-url>\ncd hackerrank-orchestrate-june26\n\n# Create virtual environment\npython3.12 -m venv venv\nsource venv/bin/activate\n\n# Install dependencies\npip install crewai crewai-tools pyyaml python-dotenv requests\n\n# Configure .env (see Setup section below)\ncp .env.example .env  # or create manually\n\n# Run the system\npython code/main.py"},{"language":"bash","snippet":"OLLAMA_HOST=127.0.0.1:11435 ollama serve"},{"language":"bash","snippet":"ollama pull gemma4:31b-cloud\nollama pull gemma3:27b-cloud"},{"language":"env","snippet":"# Ollama endpoints\nOLLAMA_MODEL_GEMMA4_URL=http://127.0.0.1:11434\nOLLAMA_MODEL_GEMMA3_URL=http://127.0.0.1:11435\n\n# Text agents — Gemma3 (port 11435)\nMODEL_CLAIM_EXTRACTOR=ollama/gemma3:27b-cloud\nMODEL_SYNTHESIZER=ollama/gemma3:27b-cloud\n\n# Vision agents — Gemma4 (port 11434)\nMODEL_VISUAL_JUDGE=ollama/gemma4:31b-cloud\nMODEL_QUALITY_JUDGE=ollama/gemma4:31b-cloud\n\n# Cross-check agent — Gemma3 (different model for jury diversity)\nMODEL_AUTHENTICITY_JUDGE=ollama/gemma3:27b-cloud"},{"language":"bash","snippet":"source venv/bin/activate\npython code/main.py"},{"language":"bash","snippet":"python code/evaluation/main.py --max-rows 20"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB REPOS","editorialOverview":"Agent processes user made claims with pictures and detects and classifies them as contradictory, true, or lack of information. Made as a part of Hackerrank Orchestrate Hackathon. Powered by CrewAI and Ollama cloud Gemma3 and Gemma4 models Multi-Modal Evidence Review — HackerRank Orchestrate Submission A multi-agent system that verifies visual evidence for damage claims across **cars**, **laptops**, and **packages**. Built with CrewAI Flows and a dual-model architecture using local Ollama instances. Read $1 for the full task spec, input/output schema, and allowed values. --- Quick Start Produces output.csv in the repo root. --- Setup Prerequisites - Py","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":408,"uniquenessScore":67,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-10-09T18:14:39.280Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-10-09T18:14:39.280Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-10T02:02:18.022Z","emptyReason":null},"items":[{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-10-09T19:11:12.944Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_repos","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}