{"id":"5708b679-d6a4-4357-a32d-cc53dfc72a76","slug":"gitlab-public-projects-saritprofile-agent-eval-framework","name":"agent-eval-framework","description":"Layered evaluation framework for autonomous LLM coding agents, MCP tool-use & skills — built on promptfoo. Deterministic safety assertions, adversarial gates, judge-trust (κ) & statistical significance.","canonicalUrl":"https://www.xpersona.co/agent/gitlab-public-projects-saritprofile-agent-eval-framework","sourceUrl":"https://gitlab.com/SaritProfile/agent-eval-framework","homepage":"https://gitlab.com/SaritProfile/agent-eval-framework","source":"GITLAB_PUBLIC_PROJECTS","vendor":{"slug":"gitlab","label":"Gitlab","url":"https://gitlab.com/SaritProfile/agent-eval-framework"},"protocols":["MCP"],"capabilities":["ai-agents","evaluation","llm","mcp","promptfoo","qa","testing"],"trustScore":null,"trustConfidence":"unknown","artifactCount":0,"benchmarkCount":0,"lastRelease":null,"freshnessAt":"2026-10-09T01:21:01.074Z","freshnessLabel":"Oct 9, 2026","securityReviewed":true,"openapiReady":false,"stats":[{"label":"Trust score","value":"Unknown"},{"label":"Compatibility","value":"MCP"},{"label":"Freshness","value":"Oct 9, 2026"},{"label":"Vendor","value":"Gitlab"},{"label":"Artifacts","value":"0"},{"label":"Benchmarks","value":"0"},{"label":"Last release","value":"Unpublished"}],"factsPreview":[{"factKey":"vendor","category":"vendor","label":"Vendor","value":"Gitlab","href":"https://gitlab.com/SaritProfile/agent-eval-framework","sourceUrl":"https://gitlab.com/SaritProfile/agent-eval-framework","sourceType":"profile","confidence":"medium","observedAt":"2026-10-09T01:21:01.074Z","isPublic":true},{"factKey":"protocols","category":"compatibility","label":"Protocol compatibility","value":"MCP","href":"https://www.xpersona.co/api/v1/agents/gitlab-public-projects-saritprofile-agent-eval-framework/contract","sourceUrl":"https://www.xpersona.co/api/v1/agents/gitlab-public-projects-saritprofile-agent-eval-framework/contract","sourceType":"contract","confidence":"medium","observedAt":"2026-10-09T01:21:01.074Z","isPublic":true},{"factKey":"handshake_status","category":"security","label":"Handshake status","value":"UNKNOWN","href":"https://www.xpersona.co/api/v1/agents/gitlab-public-projects-saritprofile-agent-eval-framework/trust","sourceUrl":"https://www.xpersona.co/api/v1/agents/gitlab-public-projects-saritprofile-agent-eval-framework/trust","sourceType":"trust","confidence":"medium","observedAt":null,"isPublic":true}],"highlights":["Trust evidence available"],"agentCard":{"name":"agent-eval-framework","description":"Layered evaluation framework for autonomous LLM coding agents, MCP tool-use & skills — built on promptfoo. Deterministic safety assertions, adversarial gates, judge-trust (κ) & statistical significance.","source":"GITLAB_PUBLIC_PROJECTS","sourceId":"gitlab:saritprofile/agent-eval-framework","homepage":"https://gitlab.com/SaritProfile/agent-eval-framework","repository":"https://gitlab.com/SaritProfile/agent-eval-framework","documentation":"https://www.xpersona.co/agent/gitlab-public-projects-saritprofile-agent-eval-framework","protocols":["MCP"],"capabilities":["ai-agents","evaluation","llm","mcp","promptfoo","qa","testing"]}}