[build-system] requires = ["setuptools>=68"] build-backend = "setuptools.build_meta" [project] name = "evalharness" version = "0.1.0" description = "A plugin-based LLM/agent evaluation harness (data layer first)" requires-python = ">=3.10" dependencies = [ "pydantic>=2", "datasets", # HuggingFace-hosted datasets (light, conflict-free) "pyarrow", # parquet sources (ModelScope/HF raw mirrors) "sympy", # official PRM800K symbolic math grading "pylatexenc", "numpy", # official DROP aligner "scipy", "rich", "xlsxwriter", # excel result workbook "httpx", # model fingerprint probing (evalharness/fingerprint) "tree_sitter>=0.21", # vendored BFCL official AST checker (python) "tree-sitter-java>=0.21", # bfcl java categories "tree-sitter-javascript>=0.21", # bfcl javascript categories ] [project.optional-dependencies] # none today: the official BFCL checker is vendored under # evalharness/third_party/bfcl (Apache-2.0); heavy execution environments # (humaneval/bigcodebench/swe) live in docker images, never in the venv [project.scripts] evalharness = "evalharness.cli:main" [tool.setuptools.packages.find] include = ["evalharness*"] [tool.setuptools.package-data] "*" = ["*.jsonl", "*.json", "*.csv", "*.tsv", "*.txt", "*.md", "*.yaml", "references/*.json"]