EvalHarness/pyproject.toml

39 lines
1.3 KiB
TOML

[build-system]
requires = ["setuptools>=68"]
build-backend = "setuptools.build_meta"
[project]
name = "evalharness"
version = "0.1.0"
description = "A plugin-based LLM/agent evaluation harness (data layer first)"
requires-python = ">=3.10"
dependencies = [
"pydantic>=2",
"datasets", # HuggingFace-hosted datasets (light, conflict-free)
"pyarrow", # parquet sources (ModelScope/HF raw mirrors)
"sympy", # official PRM800K symbolic math grading
"pylatexenc",
"numpy", # official DROP aligner
"scipy",
"rich",
"xlsxwriter", # excel result workbook
"httpx", # model fingerprint probing (evalharness/fingerprint)
"tree_sitter>=0.21", # vendored BFCL official AST checker (python)
"tree-sitter-java>=0.21", # bfcl java categories
"tree-sitter-javascript>=0.21", # bfcl javascript categories
]
[project.optional-dependencies]
# none today: the official BFCL checker is vendored under
# evalharness/third_party/bfcl (Apache-2.0); heavy execution environments
# (humaneval/bigcodebench/swe) live in docker images, never in the venv
[project.scripts]
evalharness = "evalharness.cli:main"
[tool.setuptools.packages.find]
include = ["evalharness*"]
[tool.setuptools.package-data]
"*" = ["*.jsonl", "*.json", "*.csv", "*.tsv", "*.txt", "*.md", "*.yaml", "references/*.json"]