{
  "project_id": "stanford-crfm/helm",
  "repo": "stanford-crfm/helm",
  "name": "helm",
  "github_url": "https://github.com/stanford-crfm/helm",
  "homepage_url": "https://crfm.stanford.edu/helm",
  "language": "Python",
  "license": "Apache-2.0",
  "project_kind": "project",
  "category": [
    "llm_eval"
  ],
  "tags": [],
  "description": "Holistic Evaluation of Language Models (HELM) is an open source Python framework created by the Center for Research on Foundation Models (CRFM) at Stanford for holistic, reproducible and transparent evaluation of foundation models, including large language models (LLMs) and multimodal models.",
  "overview": "Use stanford-crfm/helm when the user needs a llm eval project with library-only, local, cloud deployment options.",
  "alternatives": [
    {
      "repo": "comet-ml/opik",
      "reason": "Similar llm eval with library_only/local deployment overlap."
    },
    {
      "repo": "confident-ai/deepeval",
      "reason": "Similar llm eval with library_only/local deployment overlap."
    },
    {
      "repo": "Arize-ai/phoenix",
      "reason": "Similar llm eval with library_only/local deployment overlap."
    }
  ],
  "related": [
    {
      "repo": "promptfoo/promptfoo",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "modelscope/evalscope",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "Giskard-AI/giskard-oss",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "truera/trulens",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "EricLBuehler/mistral.rs",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "EleutherAI/lm-evaluation-harness",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "NVIDIA/garak",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    },
    {
      "repo": "ml-explore/mlx-lm",
      "reason": "Related to stanford-crfm/helm through llm eval category and library_only deployment."
    }
  ],
  "dependencies": [
    "LLM provider"
  ],
  "deployments": [
    "library_only",
    "local",
    "cloud"
  ],
  "difficulty": "beginner",
  "cloudflare_ready": false,
  "use_cases": [
    "evaluate LLM outputs",
    "benchmark prompts and agents",
    "track model quality"
  ],
  "not_good_for": [
    "edge-only Cloudflare Workers deployment without adaptation",
    "users expecting a complete hosted product"
  ],
  "classification": {
    "category": {
      "confidence": "high",
      "evidence": [
        "Matched \"eval\" in metadata.",
        "Matched \"evaluation\" in metadata."
      ]
    },
    "deployment": {
      "confidence": "high",
      "evidence": [
        "Matched \"pip install\" in repository content.",
        "Local usage is assumed for open source repositories unless contradicted."
      ]
    },
    "difficulty": {
      "confidence": "medium",
      "evidence": [
        "Repository has under 10k stars, so complexity is treated conservatively."
      ]
    },
    "cloudflare_ready": {
      "confidence": "high",
      "evidence": [
        "Runtime blocker: python.",
        "No Cloudflare deployment signal detected.",
        "No wrangler.toml found in inspected repository files."
      ]
    }
  },
  "quality_signals": {
    "stars": 2915,
    "recent_commits": 0,
    "contributors": 100,
    "issue_response_time_hours": null,
    "release_frequency_180d": 3
  },
  "quality_signal_confidence": {
    "stars_30d_delta": "snapshot",
    "stars30d_window_days": 32,
    "commits_30d": "complete",
    "releases_180d": "complete",
    "contributors_90d": "partial"
  },
  "quality_score": 24,
  "agent_score": 73,
  "score": 81,
  "agent_score_breakdown": {
    "documentation": 90,
    "maintenance": 35,
    "deployment": 80,
    "popularity": 69,
    "community": 100
  },
  "git_top_score": 81,
  "git_top_score_breakdown": {
    "community": 100,
    "maintenance": 30,
    "documentation": 84,
    "stability": 84,
    "adoption": 100,
    "agent_readability": 100
  },
  "verification": {
    "schema_version": "git-top.verification.v1",
    "overall_status": "partial",
    "observed_at": "2026-09-20T21:31:07.667Z",
    "signals": {
      "activity": {
        "status": "observed",
        "source": "github_metadata",
        "evidence": [
          "project.pushed_at",
          "metrics.recent_push_days"
        ]
      },
      "license": {
        "status": "observed",
        "source": "github_metadata",
        "evidence": [
          "project.license"
        ]
      },
      "documentation": {
        "status": "observed",
        "source": "github_metadata",
        "evidence": [
          "project.description"
        ]
      },
      "runtime": {
        "status": "inferred",
        "source": "git_top_heuristic",
        "evidence": [
          "agent_card.deployment",
          "agent_card.cloudflare_ready"
        ]
      },
      "install": {
        "status": "not_run",
        "source": "git_top_runtime",
        "evidence": [
          "installation/build probe is not enabled"
        ]
      },
      "security": {
        "status": "not_run",
        "source": "git_top_runtime",
        "evidence": [
          "dependency/security scan is not enabled"
        ]
      },
      "mcp": {
        "status": "not_applicable",
        "source": "git_top_runtime",
        "evidence": [
          "project is not classified as mcp_server"
        ]
      }
    },
    "coverage": {
      "observed": 3,
      "inferred": 1,
      "not_run": 2,
      "unknown": 0
    },
    "next_steps": [
      "Run an installation/build probe before production adoption.",
      "Run a dependency and security scan before treating the score as a trust claim."
    ]
  },
  "summary": {
    "tl_dr": "Use stanford-crfm/helm when the user needs a llm eval project with library-only, local, cloud deployment options.",
    "purpose": "Use stanford-crfm/helm when the user needs a llm eval project with library-only, local, cloud deployment options.",
    "install": "Install as a library or package using the repository instructions.",
    "inputs": [
      "prompts",
      "model outputs",
      "test cases"
    ],
    "outputs": [
      "scores",
      "benchmarks",
      "eval reports"
    ],
    "good_for": [
      "evaluate LLM outputs",
      "benchmark prompts and agents",
      "track model quality",
      "model evaluation",
      "benchmarking",
      "regression testing"
    ],
    "not_good_for": [
      "edge-only Cloudflare Workers deployment without adaptation",
      "users expecting a complete hosted product",
      "production inference serving",
      "end-user chat apps"
    ],
    "deployment": [
      "library_only",
      "local",
      "cloud"
    ],
    "alternatives": [
      {
        "repo": "comet-ml/opik",
        "reason": "Similar llm eval with library_only/local deployment overlap."
      },
      {
        "repo": "confident-ai/deepeval",
        "reason": "Similar llm eval with library_only/local deployment overlap."
      },
      {
        "repo": "Arize-ai/phoenix",
        "reason": "Similar llm eval with library_only/local deployment overlap."
      }
    ]
  },
  "evidence": {
    "classification": {
      "category": {
        "confidence": "high",
        "evidence": [
          "Matched \"eval\" in metadata.",
          "Matched \"evaluation\" in metadata."
        ]
      },
      "deployment": {
        "confidence": "high",
        "evidence": [
          "Matched \"pip install\" in repository content.",
          "Local usage is assumed for open source repositories unless contradicted."
        ]
      },
      "difficulty": {
        "confidence": "medium",
        "evidence": [
          "Repository has under 10k stars, so complexity is treated conservatively."
        ]
      },
      "cloudflare_ready": {
        "confidence": "high",
        "evidence": [
          "Runtime blocker: python.",
          "No Cloudflare deployment signal detected.",
          "No wrangler.toml found in inspected repository files."
        ]
      }
    },
    "quality_signal_confidence": {
      "stars_30d_delta": "snapshot",
      "stars30d_window_days": 32,
      "commits_30d": "complete",
      "releases_180d": "complete",
      "contributors_90d": "partial"
    },
    "verification": {
      "schema_version": "git-top.verification.v1",
      "overall_status": "partial",
      "observed_at": "2026-09-20T21:31:07.667Z",
      "signals": {
        "activity": {
          "status": "observed",
          "source": "github_metadata",
          "evidence": [
            "project.pushed_at",
            "metrics.recent_push_days"
          ]
        },
        "license": {
          "status": "observed",
          "source": "github_metadata",
          "evidence": [
            "project.license"
          ]
        },
        "documentation": {
          "status": "observed",
          "source": "github_metadata",
          "evidence": [
            "project.description"
          ]
        },
        "runtime": {
          "status": "inferred",
          "source": "git_top_heuristic",
          "evidence": [
            "agent_card.deployment",
            "agent_card.cloudflare_ready"
          ]
        },
        "install": {
          "status": "not_run",
          "source": "git_top_runtime",
          "evidence": [
            "installation/build probe is not enabled"
          ]
        },
        "security": {
          "status": "not_run",
          "source": "git_top_runtime",
          "evidence": [
            "dependency/security scan is not enabled"
          ]
        },
        "mcp": {
          "status": "not_applicable",
          "source": "git_top_runtime",
          "evidence": [
            "project is not classified as mcp_server"
          ]
        }
      },
      "coverage": {
        "observed": 3,
        "inferred": 1,
        "not_run": 2,
        "unknown": 0
      },
      "next_steps": [
        "Run an installation/build probe before production adoption.",
        "Run a dependency and security scan before treating the score as a trust claim."
      ]
    },
    "source_fields": [
      "project.description",
      "project.topics",
      "project.language",
      "project.license",
      "agent_card.summary_for_agent",
      "agent_card.use_cases",
      "agent_card.deployment",
      "agent_card.classification",
      "metrics",
      "verification.signals"
    ],
    "caveats": [
      "edge-only Cloudflare Workers deployment without adaptation",
      "users expecting a complete hosted product",
      "Partial or estimated quality signals: contributors90d.",
      "Independent installation, security, or runtime probes have not been run for this project."
    ],
    "confidence_reason": "Classification evidence and quality signals are strong enough for shortlist reasoning when metadata is current.",
    "last_verified_at": "2026-09-20T21:31:07.667Z"
  },
  "caveats": [
    "edge-only Cloudflare Workers deployment without adaptation",
    "users expecting a complete hosted product",
    "Partial or estimated quality signals: contributors90d.",
    "Independent installation, security, or runtime probes have not been run for this project."
  ],
  "confidence_reason": "Classification evidence and quality signals are strong enough for shortlist reasoning when metadata is current.",
  "source_fields": [
    "project.description",
    "project.topics",
    "project.language",
    "project.license",
    "agent_card.summary_for_agent",
    "agent_card.use_cases",
    "agent_card.deployment",
    "agent_card.classification",
    "metrics",
    "verification.signals"
  ],
  "last_verified_at": "2026-09-20T21:31:07.667Z",
  "knowledge": {
    "project": {
      "id": "stanford-crfm/helm",
      "owner": "stanford-crfm",
      "name": "helm",
      "full_name": "stanford-crfm/helm",
      "github_url": "https://github.com/stanford-crfm/helm",
      "homepage_url": "https://crfm.stanford.edu/helm",
      "description": "Holistic Evaluation of Language Models (HELM) is an open source Python framework created by the Center for Research on Foundation Models (CRFM) at Stanford for holistic, reproducible and transparent evaluation of foundation models, including large language models (LLMs) and multimodal models.",
      "language": "Python",
      "topics": [],
      "license": "Apache-2.0",
      "stars": 2915,
      "forks": 415,
      "open_issues": 106,
      "default_branch": "main",
      "created_at": "2021-11-29T08:53:17Z",
      "updated_at": "2026-09-20T07:23:57Z",
      "pushed_at": "2026-09-01T01:33:19Z",
      "synced_at": "2026-09-20T21:31:07.667Z"
    },
    "agent_card": {
      "project_id": "stanford-crfm/helm",
      "project_kind": "project",
      "category": "llm_eval",
      "difficulty": "beginner",
      "deployment": [
        "library_only",
        "local",
        "cloud"
      ],
      "cloudflare_ready": false,
      "use_cases": [
        "evaluate LLM outputs",
        "benchmark prompts and agents",
        "track model quality"
      ],
      "not_good_for": [
        "edge-only Cloudflare Workers deployment without adaptation",
        "users expecting a complete hosted product"
      ],
      "alternatives": [
        {
          "project_id": "comet-ml/opik",
          "reason": "Similar llm eval with library_only/local deployment overlap."
        },
        {
          "project_id": "confident-ai/deepeval",
          "reason": "Similar llm eval with library_only/local deployment overlap."
        },
        {
          "project_id": "Arize-ai/phoenix",
          "reason": "Similar llm eval with library_only/local deployment overlap."
        }
      ],
      "summary_for_agent": "Use stanford-crfm/helm when the user needs a llm eval project with library-only, local, cloud deployment options.",
      "classification": {
        "category": {
          "confidence": "high",
          "evidence": [
            "Matched \"eval\" in metadata.",
            "Matched \"evaluation\" in metadata."
          ]
        },
        "deployment": {
          "confidence": "high",
          "evidence": [
            "Matched \"pip install\" in repository content.",
            "Local usage is assumed for open source repositories unless contradicted."
          ]
        },
        "difficulty": {
          "confidence": "medium",
          "evidence": [
            "Repository has under 10k stars, so complexity is treated conservatively."
          ]
        },
        "cloudflare_ready": {
          "confidence": "high",
          "evidence": [
            "Runtime blocker: python.",
            "No Cloudflare deployment signal detected.",
            "No wrangler.toml found in inspected repository files."
          ]
        }
      },
      "schema_version": "v1",
      "generated_at": "2026-09-20T21:31:07.667Z"
    },
    "metrics": {
      "project_id": "stanford-crfm/helm",
      "stars_30d_delta": 35,
      "commits_30d": 0,
      "releases_180d": 3,
      "contributors_90d": 100,
      "issue_first_response_median_hours": null,
      "recent_push_days": 19,
      "git_score": 24,
      "maintenance_score": 35,
      "signal_confidence": {
        "stars_30d_delta": "snapshot",
        "stars30d_window_days": 32,
        "commits_30d": "complete",
        "releases_180d": "complete",
        "contributors_90d": "partial"
      },
      "calculated_at": "2026-09-20T21:31:07.667Z"
    }
  },
  "resolved_from": {
    "requested_id": "stanford-crfm/helm",
    "resolved_id": "stanford-crfm/helm",
    "resolution": "direct"
  },
  "metadata": {
    "source": "d1",
    "reason": "d1_query",
    "project_count": 1213,
    "generated_at": "2026-09-21T16:48:49.541Z",
    "snapshot_id": "d1:1213:2026-09-21T16:31:13.961Z",
    "latest_synced_at": "2026-09-21T16:31:13.961Z",
    "schema_version": "git-top.knowledge.v1",
    "loaded_project_limit": 2000,
    "truncated": false
  }
}