{
  "project_id": "agentevals-dev/agentevals",
  "repo": "agentevals-dev/agentevals",
  "name": "agentevals",
  "github_url": "https://github.com/agentevals-dev/agentevals",
  "homepage_url": "https://aevals.ai/",
  "language": "Python",
  "license": "Apache-2.0",
  "project_kind": "project",
  "category": [
    "llm_eval"
  ],
  "tags": [
    "agentevals",
    "agents",
    "evals",
    "evaluation",
    "genai",
    "llm",
    "llm-as-judge",
    "opentelemetry",
    "otel",
    "tracing"
  ],
  "description": "agentevals is a framework-agnostic evaluations solution based on OpenTelemetry traces",
  "overview": "Use agentevals-dev/agentevals when the user needs a llm eval project with docker, kubernetes, library-only deployment options.",
  "alternatives": [
    {
      "repo": "comet-ml/opik",
      "reason": "Same llm eval intent with observability overlap."
    },
    {
      "repo": "Arize-ai/phoenix",
      "reason": "Same llm eval intent with observability overlap."
    },
    {
      "repo": "lmnr-ai/lmnr",
      "reason": "Same llm eval intent with observability overlap."
    }
  ],
  "related": [
    {
      "repo": "promptfoo/promptfoo",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "langfuse/langfuse",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "ai-twinkle/Eval",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "EricLBuehler/mistral.rs",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "coze-dev/coze-loop",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "dataelement/bisheng",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "modelscope/evalscope",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and docker deployment."
    },
    {
      "repo": "Purewhiter/mobilegym",
      "reason": "Related to agentevals-dev/agentevals through llm eval category and library_only deployment."
    }
  ],
  "dependencies": [
    "LLM provider"
  ],
  "deployments": [
    "docker",
    "kubernetes",
    "library_only",
    "local",
    "cloud"
  ],
  "difficulty": "beginner",
  "cloudflare_ready": false,
  "use_cases": [
    "evaluate LLM outputs",
    "benchmark prompts and agents",
    "track model quality"
  ],
  "not_good_for": [
    "edge-only Cloudflare Workers deployment without adaptation",
    "users expecting a complete hosted product"
  ],
  "classification": {
    "category": {
      "confidence": "high",
      "evidence": [
        "Matched \"eval\" in metadata.",
        "Matched \"evaluation\" in metadata."
      ]
    },
    "deployment": {
      "confidence": "high",
      "evidence": [
        "Found Docker configuration file.",
        "Matched \"kubernetes\" in repository content.",
        "Matched \"helm chart\" in repository content.",
        "Matched \"pip install\" in repository content.",
        "Local usage is assumed for open source repositories unless contradicted."
      ]
    },
    "difficulty": {
      "confidence": "medium",
      "evidence": [
        "Repository has under 10k stars, so complexity is treated conservatively."
      ]
    },
    "cloudflare_ready": {
      "confidence": "high",
      "evidence": [
        "Runtime blocker: python, postgres.",
        "No Cloudflare deployment signal detected.",
        "No wrangler.toml found in inspected repository files."
      ]
    }
  },
  "quality_signals": {
    "stars": 162,
    "recent_commits": 26,
    "contributors": 17,
    "issue_response_time_hours": null,
    "release_frequency_180d": 25
  },
  "quality_signal_confidence": {
    "stars_30d_delta": "snapshot",
    "stars30d_window_days": 36,
    "commits_30d": "complete",
    "releases_180d": "complete",
    "contributors_90d": "complete"
  },
  "quality_score": 25,
  "agent_score": 69,
  "score": 86,
  "agent_score_breakdown": {
    "documentation": 90,
    "maintenance": 49,
    "deployment": 100,
    "popularity": 44,
    "community": 62
  },
  "git_top_score": 86,
  "git_top_score_breakdown": {
    "community": 100,
    "maintenance": 65,
    "documentation": 92,
    "stability": 100,
    "adoption": 63,
    "agent_readability": 100
  },
  "verification": {
    "schema_version": "git-top.verification.v1",
    "overall_status": "partial",
    "observed_at": "2026-09-22T20:01:00.470Z",
    "signals": {
      "activity": {
        "status": "observed",
        "source": "github_metadata",
        "evidence": [
          "project.pushed_at",
          "metrics.recent_push_days"
        ]
      },
      "license": {
        "status": "observed",
        "source": "github_metadata",
        "evidence": [
          "project.license"
        ]
      },
      "documentation": {
        "status": "observed",
        "source": "github_metadata",
        "evidence": [
          "project.description"
        ]
      },
      "runtime": {
        "status": "inferred",
        "source": "git_top_heuristic",
        "evidence": [
          "agent_card.deployment",
          "agent_card.cloudflare_ready"
        ]
      },
      "install": {
        "status": "not_run",
        "source": "git_top_runtime",
        "evidence": [
          "installation/build probe is not enabled"
        ]
      },
      "security": {
        "status": "not_run",
        "source": "git_top_runtime",
        "evidence": [
          "dependency/security scan is not enabled"
        ]
      },
      "mcp": {
        "status": "not_applicable",
        "source": "git_top_runtime",
        "evidence": [
          "project is not classified as mcp_server"
        ]
      }
    },
    "coverage": {
      "observed": 3,
      "inferred": 1,
      "not_run": 2,
      "unknown": 0
    },
    "next_steps": [
      "Run an installation/build probe before production adoption.",
      "Run a dependency and security scan before treating the score as a trust claim."
    ]
  },
  "summary": {
    "tl_dr": "Use agentevals-dev/agentevals when the user needs a llm eval project with docker, kubernetes, library-only deployment options.",
    "purpose": "Use agentevals-dev/agentevals when the user needs a llm eval project with docker, kubernetes, library-only deployment options.",
    "install": "Run with Docker using the repository's container instructions.",
    "inputs": [
      "prompts",
      "model outputs",
      "test cases"
    ],
    "outputs": [
      "scores",
      "benchmarks",
      "eval reports"
    ],
    "good_for": [
      "evaluate LLM outputs",
      "benchmark prompts and agents",
      "track model quality",
      "model evaluation",
      "benchmarking",
      "regression testing"
    ],
    "not_good_for": [
      "edge-only Cloudflare Workers deployment without adaptation",
      "users expecting a complete hosted product",
      "production inference serving",
      "end-user chat apps"
    ],
    "deployment": [
      "docker",
      "kubernetes",
      "library_only",
      "local",
      "cloud"
    ],
    "alternatives": [
      {
        "repo": "comet-ml/opik",
        "reason": "Same llm eval intent with observability overlap."
      },
      {
        "repo": "Arize-ai/phoenix",
        "reason": "Same llm eval intent with observability overlap."
      },
      {
        "repo": "lmnr-ai/lmnr",
        "reason": "Same llm eval intent with observability overlap."
      }
    ]
  },
  "evidence": {
    "classification": {
      "category": {
        "confidence": "high",
        "evidence": [
          "Matched \"eval\" in metadata.",
          "Matched \"evaluation\" in metadata."
        ]
      },
      "deployment": {
        "confidence": "high",
        "evidence": [
          "Found Docker configuration file.",
          "Matched \"kubernetes\" in repository content.",
          "Matched \"helm chart\" in repository content.",
          "Matched \"pip install\" in repository content.",
          "Local usage is assumed for open source repositories unless contradicted."
        ]
      },
      "difficulty": {
        "confidence": "medium",
        "evidence": [
          "Repository has under 10k stars, so complexity is treated conservatively."
        ]
      },
      "cloudflare_ready": {
        "confidence": "high",
        "evidence": [
          "Runtime blocker: python, postgres.",
          "No Cloudflare deployment signal detected.",
          "No wrangler.toml found in inspected repository files."
        ]
      }
    },
    "quality_signal_confidence": {
      "stars_30d_delta": "snapshot",
      "stars30d_window_days": 36,
      "commits_30d": "complete",
      "releases_180d": "complete",
      "contributors_90d": "complete"
    },
    "verification": {
      "schema_version": "git-top.verification.v1",
      "overall_status": "partial",
      "observed_at": "2026-09-22T20:01:00.470Z",
      "signals": {
        "activity": {
          "status": "observed",
          "source": "github_metadata",
          "evidence": [
            "project.pushed_at",
            "metrics.recent_push_days"
          ]
        },
        "license": {
          "status": "observed",
          "source": "github_metadata",
          "evidence": [
            "project.license"
          ]
        },
        "documentation": {
          "status": "observed",
          "source": "github_metadata",
          "evidence": [
            "project.description"
          ]
        },
        "runtime": {
          "status": "inferred",
          "source": "git_top_heuristic",
          "evidence": [
            "agent_card.deployment",
            "agent_card.cloudflare_ready"
          ]
        },
        "install": {
          "status": "not_run",
          "source": "git_top_runtime",
          "evidence": [
            "installation/build probe is not enabled"
          ]
        },
        "security": {
          "status": "not_run",
          "source": "git_top_runtime",
          "evidence": [
            "dependency/security scan is not enabled"
          ]
        },
        "mcp": {
          "status": "not_applicable",
          "source": "git_top_runtime",
          "evidence": [
            "project is not classified as mcp_server"
          ]
        }
      },
      "coverage": {
        "observed": 3,
        "inferred": 1,
        "not_run": 2,
        "unknown": 0
      },
      "next_steps": [
        "Run an installation/build probe before production adoption.",
        "Run a dependency and security scan before treating the score as a trust claim."
      ]
    },
    "source_fields": [
      "project.description",
      "project.topics",
      "project.language",
      "project.license",
      "agent_card.summary_for_agent",
      "agent_card.use_cases",
      "agent_card.deployment",
      "agent_card.classification",
      "metrics",
      "verification.signals"
    ],
    "caveats": [
      "edge-only Cloudflare Workers deployment without adaptation",
      "users expecting a complete hosted product",
      "Independent installation, security, or runtime probes have not been run for this project."
    ],
    "confidence_reason": "Classification evidence and quality signals are strong enough for shortlist reasoning when metadata is current.",
    "last_verified_at": "2026-09-22T20:01:00.470Z"
  },
  "caveats": [
    "edge-only Cloudflare Workers deployment without adaptation",
    "users expecting a complete hosted product",
    "Independent installation, security, or runtime probes have not been run for this project."
  ],
  "confidence_reason": "Classification evidence and quality signals are strong enough for shortlist reasoning when metadata is current.",
  "source_fields": [
    "project.description",
    "project.topics",
    "project.language",
    "project.license",
    "agent_card.summary_for_agent",
    "agent_card.use_cases",
    "agent_card.deployment",
    "agent_card.classification",
    "metrics",
    "verification.signals"
  ],
  "last_verified_at": "2026-09-22T20:01:00.470Z",
  "knowledge": {
    "project": {
      "id": "agentevals-dev/agentevals",
      "owner": "agentevals-dev",
      "name": "agentevals",
      "full_name": "agentevals-dev/agentevals",
      "github_url": "https://github.com/agentevals-dev/agentevals",
      "homepage_url": "https://aevals.ai/",
      "description": "agentevals is a framework-agnostic evaluations solution based on OpenTelemetry traces",
      "language": "Python",
      "topics": [
        "agentevals",
        "agents",
        "evals",
        "evaluation",
        "genai",
        "llm",
        "llm-as-judge",
        "opentelemetry",
        "otel",
        "tracing"
      ],
      "license": "Apache-2.0",
      "stars": 162,
      "forks": 26,
      "open_issues": 28,
      "default_branch": "main",
      "created_at": "2026-02-24T17:54:56Z",
      "updated_at": "2026-09-22T09:36:53Z",
      "pushed_at": "2026-09-22T09:36:56Z",
      "synced_at": "2026-09-22T20:01:00.470Z"
    },
    "agent_card": {
      "project_id": "agentevals-dev/agentevals",
      "project_kind": "project",
      "category": "llm_eval",
      "difficulty": "beginner",
      "deployment": [
        "docker",
        "kubernetes",
        "library_only",
        "local",
        "cloud"
      ],
      "cloudflare_ready": false,
      "use_cases": [
        "evaluate LLM outputs",
        "benchmark prompts and agents",
        "track model quality"
      ],
      "not_good_for": [
        "edge-only Cloudflare Workers deployment without adaptation",
        "users expecting a complete hosted product"
      ],
      "alternatives": [
        {
          "project_id": "comet-ml/opik",
          "reason": "Same llm eval intent with observability overlap."
        },
        {
          "project_id": "Arize-ai/phoenix",
          "reason": "Same llm eval intent with observability overlap."
        },
        {
          "project_id": "lmnr-ai/lmnr",
          "reason": "Same llm eval intent with observability overlap."
        }
      ],
      "summary_for_agent": "Use agentevals-dev/agentevals when the user needs a llm eval project with docker, kubernetes, library-only deployment options.",
      "classification": {
        "category": {
          "confidence": "high",
          "evidence": [
            "Matched \"eval\" in metadata.",
            "Matched \"evaluation\" in metadata."
          ]
        },
        "deployment": {
          "confidence": "high",
          "evidence": [
            "Found Docker configuration file.",
            "Matched \"kubernetes\" in repository content.",
            "Matched \"helm chart\" in repository content.",
            "Matched \"pip install\" in repository content.",
            "Local usage is assumed for open source repositories unless contradicted."
          ]
        },
        "difficulty": {
          "confidence": "medium",
          "evidence": [
            "Repository has under 10k stars, so complexity is treated conservatively."
          ]
        },
        "cloudflare_ready": {
          "confidence": "high",
          "evidence": [
            "Runtime blocker: python, postgres.",
            "No Cloudflare deployment signal detected.",
            "No wrangler.toml found in inspected repository files."
          ]
        }
      },
      "schema_version": "v1",
      "generated_at": "2026-09-22T20:01:00.470Z"
    },
    "metrics": {
      "project_id": "agentevals-dev/agentevals",
      "stars_30d_delta": 8,
      "commits_30d": 26,
      "releases_180d": 25,
      "contributors_90d": 17,
      "issue_first_response_median_hours": null,
      "recent_push_days": 0,
      "git_score": 25,
      "maintenance_score": 49,
      "signal_confidence": {
        "stars_30d_delta": "snapshot",
        "stars30d_window_days": 36,
        "commits_30d": "complete",
        "releases_180d": "complete",
        "contributors_90d": "complete"
      },
      "calculated_at": "2026-09-22T20:01:00.470Z"
    }
  },
  "resolved_from": {
    "requested_id": "agentevals-dev/agentevals",
    "resolved_id": "agentevals-dev/agentevals",
    "resolution": "direct"
  },
  "metadata": {
    "source": "d1",
    "reason": "d1_query",
    "project_count": 1213,
    "generated_at": "2026-09-26T18:26:02.703Z",
    "snapshot_id": "d1:1213:2026-09-26T18:01:08.000Z",
    "latest_synced_at": "2026-09-26T18:01:08.000Z",
    "schema_version": "git-top.knowledge.v1",
    "loaded_project_limit": 2000,
    "truncated": false
  }
}