{"id":"a5137a10-a0b1-41cf-b1b4-9542c9d75f95","entityType":"agent","slug":"crewai-mrunmayeenaik-document-analyzer-crewai","name":"Document-Analyzer-CrewAI","canonicalUrl":"https://www.xpersona.co/agent/crewai-mrunmayeenaik-document-analyzer-crewai","canonicalPath":"/agent/crewai-mrunmayeenaik-document-analyzer-crewai","generatedAt":"2026-10-09T03:31:35.838Z","source":"GITHUB_OPENCLEW","claimStatus":"UNCLAIMED","verificationTier":"NONE","summary":{"evidence":{"source":"editorial-content","verified":true,"confidence":"high","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":null},"description":"Multi-agent system that reads financial PDFs and returns structured investment insights via REST API Financial Document Analyzer — Multi-Agent AI Pipeline Upload any financial PDF (annual report, investor deck, earnings statement) and get structured investment insights returned via a REST API. Built with CrewAI multi-agent orchestration — a financial analyst agent reads the document, reasons over it, and optionally searches the web for market context before producing a structured report. --- Tech stack | Layer | Too","descriptionLabel":"Technical summary","evidenceSummary":"Capability contract not published. No trust telemetry is available yet. Last updated 5/18/2026.","installCommand":"git clone https://github.com/MrunmayeeNaik/Document-Analyzer-CrewAI.git","sourceUrl":"https://github.com/MrunmayeeNaik/Document-Analyzer-CrewAI","homepage":null,"primaryLinks":[{"label":"View Source","url":"https://github.com/MrunmayeeNaik/Document-Analyzer-CrewAI","kind":"source"}],"safetyScore":66,"overallRank":31.6,"popularityScore":0,"trustScore":null,"claimedByName":null,"isOwner":false,"seoDescription":"Multi-agent system that reads financial PDFs and returns structured investment insights via REST API Financial Document Analyzer — Multi-Agent AI Pipeline Uploa"},"coverage":{"evidence":{"source":"public-profile","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":null},"protocols":[{"protocol":"OPENCLEW","label":"OpenClaw","status":"self-declared","notes":"Declared in the public agent profile."}],"capabilities":[{"label":"crewai","status":"self-declared"},{"label":"multi-agent","status":"self-declared"}],"verifiedCount":0,"selfDeclaredCount":3,"capabilityMatrix":{"rows":[{"key":"OPENCLEW","type":"protocol","support":"unknown","confidenceSource":"profile","notes":"Listed on profile"},{"key":"crewai","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"},{"key":"multi-agent","type":"capability","support":"supported","confidenceSource":"profile","notes":"Declared in agent profile metadata"}],"flattenedTokens":"protocol:OPENCLEW|unknown|profile capability:crewai|supported|profile capability:multi-agent|supported|profile"}},"adoption":{"evidence":{"source":"no-adoption-signals","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":"No source adoption metrics were available."},"stars":0,"forks":0,"downloads":null,"packageName":null,"latestVersion":null,"tractionLabel":null},"release":{"evidence":{"source":"agent-index","verified":false,"confidence":"medium","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":null},"lastUpdatedAt":"2026-05-18T06:45:00.168Z","lastCrawledAt":"2026-05-18T06:45:00.168Z","lastIndexedAt":null,"nextCrawlAt":"2026-05-25T06:45:00.168Z","lastVerifiedAt":null,"highlights":[]},"execution":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No published capability contract is available yet."},"installCommand":"git clone https://github.com/MrunmayeeNaik/Document-Analyzer-CrewAI.git","setupComplexity":"low","setupSteps":["Setup complexity is LOW. This package is likely designed for quick installation with minimal external side-effects.","Final validation: Expose the agent to a mock request payload inside a sandbox and trace the network egress before allowing access to real customer data."],"contract":{"contractStatus":"missing","authModes":[],"requires":[],"forbidden":[],"supportsMcp":false,"supportsA2a":false,"supportsStreaming":false,"inputSchemaRef":null,"outputSchemaRef":null,"dataRegion":null,"contractUpdatedAt":null,"sourceUpdatedAt":null,"freshnessSeconds":null},"invocationGuide":{"preferredApi":{"snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/trust"},"curlExamples":["curl -s \"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/snapshot\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/contract\"","curl -s \"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/trust\""],"jsonRequestTemplate":{"query":"summarize this repo","constraints":{"maxLatencyMs":2000,"protocolPreference":["OPENCLEW"]}},"jsonResponseTemplate":{"ok":true,"result":{"summary":"...","confidence":0.9},"meta":{"source":"GITHUB_OPENCLEW","generatedAt":"2026-10-09T03:31:35.837Z"}},"retryPolicy":{"maxAttempts":3,"backoffMs":[500,1500,3500],"retryableConditions":["HTTP_429","HTTP_503","NETWORK_TIMEOUT"]}},"endpoints":{"dossierUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/dossier","snapshotUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/snapshot","contractUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/contract","trustUrl":"https://www.xpersona.co/api/v1/agents/crewai-mrunmayeenaik-document-analyzer-crewai/trust"}},"reliability":{"evidence":{"source":"runtime-metrics","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No trust, reliability, or runtime telemetry is available."},"trust":{"status":"unavailable","handshakeStatus":"UNKNOWN","verificationFreshnessHours":null,"reputationScore":null,"p95LatencyMs":null,"successRate30d":null,"fallbackRate":null,"attempts30d":null,"trustUpdatedAt":null,"trustConfidence":"unknown","sourceUpdatedAt":null,"freshnessSeconds":null},"decisionGuardrails":{"doNotUseIf":["Contract metadata is missing or unavailable for deterministic execution."],"safeUseWhen":[],"riskFlags":["missing_or_unavailable_contract","trust_data_unavailable","schema_references_missing"],"operationalConfidence":"low"},"executionMetrics":{"observedLatencyMsP50":null,"observedLatencyMsP95":null,"estimatedCostUsd":null,"uptime30d":null,"rateLimitRpm":null,"rateLimitBurst":null,"lastVerifiedAt":null,"verificationSource":null},"runtimeMetrics":{"successRate":null,"avgLatencyMs":null,"avgCostUsd":null,"hallucinationRate":null,"retryRate":null,"disputeRate":null,"p50Latency":null,"p95Latency":null,"lastUpdated":null}},"benchmarks":{"evidence":{"source":"no-benchmark-data","verified":false,"confidence":"low","updatedAt":null,"emptyReason":"No benchmark suites or observed failure patterns are available."},"suites":[],"failurePatterns":[]},"artifacts":{"evidence":{"source":"GITHUB OPENCLEW","verified":false,"confidence":"high","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":null},"readme":"\n# Financial Document Analyzer — Multi-Agent AI Pipeline\n\nUpload any financial PDF (annual report, investor deck, earnings statement) \nand get structured investment insights returned via a REST API.\n\nBuilt with CrewAI multi-agent orchestration — a financial analyst agent \nreads the document, reasons over it, and optionally searches the web \nfor market context before producing a structured report.\n\n---\n\n## Tech stack\n\n| Layer | Tool |\n|---|---|\n| Agent orchestration | CrewAI |\n| LLM backend | OpenAI-compatible (Groq) |\n| PDF extraction | Custom tool (PyMuPDF) |\n| Web search | Serper API |\n| API layer | FastAPI |\n| Storage | SQLite via SQLAlchemy |\n| Testing | Postman |\n\n---\n\n## What it produces\n\nFor any uploaded financial PDF, the system returns:\n1. Document summary\n2. Key financial metrics and trends\n3. Analysis relevant to your specific query\n4. Risks and uncertainties\n5. High-level recommendations (with disclaimer)\n\nPrevious analyses are stored and retrievable via `GET /history`.\n\n---\n\n### 1. Bugs Found and How They Were Fixed\n\n- **LLM ignored the uploaded PDF and returned a generic answer**\n  - **Symptom**: `analysis` field said things like “without access to the specific financial document…” even though a PDF was uploaded.\n  - **Cause**: The `analyze_financial_document` task in `task.py` only referenced `{query}` and never mentioned `{file_path}`, so the `financial_analyst` agent was not clearly instructed to use the PDF reader tool with the actual path.\n  - **Fix**: Updated the task description to explicitly include `{file_path}` and to instruct the agent to call the `financial_document_reader` tool with that path before analyzing.\n\n- **HTTP 500 with LLM error `Request too large / rate_limit_exceeded`**\n  - **Symptom**: `POST /analyze` returned a 500 with a nested 413‑style error from Groq: “Request too large for model … Limit 12000, Requested 13632”.\n  - **Cause**: `FinancialDocumentTool` in `tools.py` always returned the full text of the PDF, which made some prompts exceed the model’s token‑per‑minute limits for large documents.\n  - **Fix**: Added truncation in `FinancialDocumentTool._run` so only the first N characters (default 20,000, configurable via `MAX_PDF_CHARS`) are passed to the model, with a note appended when truncation occurs.\n\n- **No records visible in `/history`**\n  - **Symptom**: `GET /history` always returned an empty list.\n  - **Causes**:\n    - Earlier `/analyze` calls failed (see above), so the DB write in `main.py` never executed.\n    - The SQLite URL (`sqlite:///./analysis.db`) is relative, so starting the server from the wrong working directory can create or query the wrong DB file.\n  - **Fix**: After fixing the LLM errors and ensuring the server is started from the project root, successful `/analyze` calls now insert rows into `Analysis`, and `/history` returns data.\n\n- **Incorrect install command in README**\n  - **Symptom**: README instructed `pip install -r requirement.txt`.\n  - **Cause**: The actual dependency file in the repo is `requirements.txt`.\n  - **Fix**: Updated the installation instructions to use `requirements.txt`.\n\n---\n\n### 2. Setup Instructions\n\n- **Requirements**\n  - Python 3.10+ recommended\n  - `pip` for installing dependencies\n\n- **Install dependencies**\n\n```sh\npip install -r requirements.txt\n```\n\n- **Environment variables**\n\nCreate a `.env` file in the project root with at least:\n\n- **`OPENAI_API_KEY`**: Groq/OpenAI‑compatible key used by CrewAI’s `LLM` (backed by the Groq endpoint).\n- **`OPENAI_BASE_URL`**: Base URL for the Groq OpenAI‑compatible API (already set to `https://api.groq.com/openai/v1` in this project).\n- **`SERPER_API_KEY`**: API key for the Serper search tool used by the agents.\n- **Optional**: `MAX_PDF_CHARS` – maximum number of characters of PDF text passed to the LLM (defaults to `20000`).\n\n> **Security note**: Keep `.env` out of version control; do not share your API keys.\n\n- **Run the API**\n\nFrom the project root (`financial-document-analyzer-debug`):\n\n```sh\nuvicorn main:app --host 0.0.0.0 --port 8000 --reload\n```\n\nThe API will be available at `http://localhost:8000`.\n\n---\n\n### 3. Usage Instructions\n\n- **1) Start the server**\n  - Run the `uvicorn` command above from the project root so that `analysis.db` is created and used in the correct directory.\n\n- **2) Analyze a financial PDF**\n  - Endpoint: `POST /analyze`\n  - Use a tool like Postman, curl, or a frontend form with:\n    - A file field named `file` (PDF).\n    - An optional form field `query` (defaults to: “Analyze this financial document for investment insights”).\n  - The backend will:\n    - Save the uploaded PDF to `data/financial_document_<uuid>.pdf`.\n    - Run the CrewAI pipeline (financial analyst agent + tools).\n    - Store the result in SQLite (`analysis.db`, table `analyses`).\n    - Return the structured analysis in the response.\n\n- **3) View analysis history**\n  - Endpoint: `GET /history`\n  - Returns basic metadata for all stored analyses (ID, file name, query, timestamp).\n\n- **4) Inspect the database (optional)**\n  - A SQLite DB file named `analysis.db` is created in the project root.\n  - You can open it with any SQLite browser to inspect the `analyses` table.\n\n---\n\n### 4. API Documentation\n\n#### `GET /`\n\n- **Description**: Health check.\n- **Response**:\n  - `200 OK` – JSON:\n    - `message`: `\"Financial Document Analyzer API is running\"`\n\n#### `POST /analyze`\n\n- **Description**: Analyze an uploaded financial PDF and persist the result.\n\n- **Request**\n  - **Content-Type**: `multipart/form-data`\n  - **Fields**:\n    - **`file`** (required): The PDF to analyze (`UploadFile`).\n    - **`query`** (optional, `Form` string):\n      - Default: `\"Analyze this financial document for investment insights\"`.\n      - Used as the user’s prompt for the analysis.\n\n- **Processing steps**\n  - Save the uploaded file to `data/financial_document_<uuid>.pdf`.\n  - Call `run_crew(query, file_path)`:\n    - Creates a `Crew` with the `financial_analyst` agent and `analyze_financial_document` task.\n    - The task:\n      - Uses `financial_document_reader` to read and clean the PDF text (truncated to avoid token limits).\n      - Optionally uses the Serper search tool for market context.\n      - Produces a structured analysis with:\n        1. Document Summary  \n        2. Key Financial Metrics and Trends  \n        3. Analysis Relevant to the User's Query  \n        4. Risks and Uncertainties  \n        5. High-level, non-personalized recommendations and a disclaimer\n  - Store the result in SQLite via the `Analysis` model.\n  - Delete the temporary PDF from `data/` in the `finally` block.\n\n- **Successful Response**\n  - **Status**: `200 OK`\n  - **Body** (JSON):\n    - `status`: `\"success\"`\n    - `query`: The final query string used.\n    - `analysis`: The full textual analysis produced by the Crew.\n    - `file_processed`: Original filename of the uploaded PDF.\n\n- **Error Responses**\n  - **`400/422`**: Validation errors (e.g., missing file field) handled by FastAPI automatically.\n  - **`500`**: Internal errors, wrapped as:\n    - `{\"detail\": \"Error processing financial document: <message>\"}`  \n    - Examples:\n      - File not found / invalid PDF path.\n      - LLM API issues (e.g., network or credential problems).\n      - Unexpected exceptions during crew execution.\n\n#### `GET /history`\n\n- **Description**: Return a list of previously stored analyses (metadata only).\n\n- **Response**\n  - **Status**: `200 OK`\n  - **Body** (JSON array):\n\n```json\n[\n  {\n    \"id\": \"uuid-string\",\n    \"file_name\": \"example.pdf\",\n    \"query\": \"Analyze this financial document for investment insights\",\n    \"created_at\": \"2026-02-26T12:34:56.789000\"\n  }\n]\n```\n\n> The full analysis text is stored in the database but not returned by `/history` to keep the payload small. You can extend the API with a `/history/{id}` endpoint if you need to retrieve full analyses by ID.\n\n---\n\n### 5. High-Level Architecture\n\n- **FastAPI (`main.py`)**\n  - Defines the HTTP endpoints.\n  - Orchestrates file upload, crew execution, and DB persistence.\n\n- **CrewAI Agents and Tasks (`agents.py`, `task.py`)**\n  - `financial_analyst`: Main agent that reads the PDF and generates the structured report.\n  - Tasks describe what to do with the uploaded document and the user query.\n\n- **Tools (`tools.py`)**\n  - `financial_document_reader`: Reads and cleans PDF content, now with length limiting to avoid LLM token issues.\n  - `search_tool`: Serper‑based web search for financial and market context.\n\n- **Database (`database.py`)**\n  - SQLite via SQLAlchemy.\n  - `Analysis` model stores file name, query, analysis text, and creation timestamp.\n","readmeExcerpt":"Financial Document Analyzer — Multi-Agent AI Pipeline Upload any financial PDF (annual report, investor deck, earnings statement) and get structured investment insights returned via a REST API. Built with CrewAI multi-agent orchestration — a financial analyst agent reads the document, reasons over it, and optionally searches the web for market context before producing a structured report. --- Tech stack | Layer | Too","codeSnippets":[],"executableExamples":[{"language":"sh","snippet":"pip install -r requirements.txt"},{"language":"sh","snippet":"uvicorn main:app --host 0.0.0.0 --port 8000 --reload"},{"language":"json","snippet":"[\n  {\n    \"id\": \"uuid-string\",\n    \"file_name\": \"example.pdf\",\n    \"query\": \"Analyze this financial document for investment insights\",\n    \"created_at\": \"2026-02-26T12:34:56.789000\"\n  }\n]"}],"parameters":null,"dependencies":[],"permissions":[],"extractedFiles":[],"languages":["python"],"docsSourceLabel":"GITHUB OPENCLEW","editorialOverview":"Multi-agent system that reads financial PDFs and returns structured investment insights via REST API Financial Document Analyzer — Multi-Agent AI Pipeline Upload any financial PDF (annual report, investor deck, earnings statement) and get structured investment insights returned via a REST API. Built with CrewAI multi-agent orchestration — a financial analyst agent reads the document, reasons over it, and optionally searches the web for market context before producing a structured report. --- Tech stack | Layer | Too","editorialQuality":{"score":100,"threshold":65,"status":"ready","wordCount":386,"uniquenessScore":66,"reasons":[]}},"media":{"evidence":{"source":"no-media","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":"No screenshots, media assets, or demo links are available."},"primaryImageUrl":null,"mediaAssetCount":0,"assets":[],"demoUrl":null},"ownerResources":{"evidence":{"source":"unclaimed","verified":false,"confidence":"low","updatedAt":"2026-05-18T06:45:00.168Z","emptyReason":"This page has not been claimed by the agent owner."},"hasCustomPage":false,"customPageUpdatedAt":null,"customLinks":[],"structuredLinks":{"docsUrl":null,"demoUrl":null,"supportUrl":null,"pricingUrl":null,"statusUrl":null},"customPage":null},"relatedAgents":{"evidence":{"source":"protocol-neighbors","verified":false,"confidence":"medium","updatedAt":"2026-10-09T03:31:35.838Z","emptyReason":null},"items":[{"id":"b917f68a-ebff-438e-84f8-3f4b2494c0bc","entityType":"agent","canonicalPath":"/agent/activepieces-activepieces","slug":"activepieces-activepieces","name":"activepieces","description":"AI Agents & MCPs & AI Workflow Automation • (~400 MCP servers for AI agents) • AI Automation / AI Agent with MCPs • AI Workflows & AI Agents • MCPs for AI Agents","url":"https://github.com/activepieces/activepieces","homepage":"https://www.activepieces.com","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-15T02:22:12.426Z","createdAt":"2026-02-25T03:38:12.412Z","downloads":null},{"id":"5cb26759-3a39-483f-94cf-276a98c13bb8","entityType":"agent","canonicalPath":"/agent/cherryhq-cherry-studio","slug":"cherryhq-cherry-studio","name":"cherry-studio","description":"AI productivity studio with smart chat, autonomous agents, and 300+ assistants. Unified access to frontier LLMs","url":"https://github.com/CherryHQ/cherry-studio","homepage":"https://cherry-ai.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-11T14:38:40.986Z","createdAt":"2026-02-25T03:38:19.379Z","downloads":null},{"id":"8ebccd8e-3863-4187-8355-c3f14e1f9edf","entityType":"agent","canonicalPath":"/agent/iofficeai-aionui","slug":"iofficeai-aionui","name":"AionUi","description":"Free, local, open-source 24/7 Cowork app and OpenClaw for Gemini CLI, Claude Code, Codex, OpenCode, Qwen Code, Goose CLI, Auggie, and more | 🌟 Star if you like it!","url":"https://github.com/iOfficeAI/AionUi","homepage":"https://www.aionui.com","source":"GITHUB_REPOS","protocols":["MCP","OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-04-10T18:48:31.762Z","createdAt":"2026-02-25T03:38:16.584Z","downloads":null},{"id":"6f6582d0-5d76-4f0f-b81d-86520247950b","entityType":"agent","canonicalPath":"/agent/copilotkit-copilotkit","slug":"copilotkit-copilotkit","name":"CopilotKit","description":"The Frontend for Agents & Generative UI. React + Angular","url":"https://github.com/CopilotKit/CopilotKit","homepage":"https://docs.copilotkit.ai","source":"GITHUB_REPOS","protocols":["OPENCLAW"],"capabilities":[],"safetyScore":100,"overallRank":70,"updatedAt":"2026-03-25T09:50:57.846Z","createdAt":"2026-02-25T03:39:14.617Z","downloads":null}],"links":{"hub":"/agent","source":"/agent/source/github_openclew","protocols":[{"label":"OpenClaw","href":"/agent/protocol/openclew"}]}}}