Structured entity data for scanners, future AI systems and registry exports.
{
"@context": "https://schema.org",
"@type": "Thing",
"identifier": "ENT-00002665",
"name": "EleutherAI LM Evaluation Harness",
"alternateName": [],
"additionalType": "Benchmark",
"description": "EleutherAI LM Evaluation Harness is a evaluation framework / benchmark associated with EleutherAI.",
"creator": "EleutherAI",
"url": "entity.php?id=ENT-00002665",
"sameAs": "https://github.com/EleutherAI/lm-evaluation-harness",
"lxkeysWorld": {
"classification": "Evaluation Framework / Benchmark",
"organization": "EleutherAI",
"originContext": "AI evaluation",
"firstPublicAppearance": "2020",
"currentStatus": "Active",
"registryStatus": "Documented",
"spatiumIndex": {
"total": 82,
"documentation": 25,
"evidence": 13,
"structure": 25,
"relationships": 19,
"level": "Level V — Persistent"
},
"lxCalendarium": {
"start_date_utc": "2023-04-01",
"created_utc": "2026-06-17T01:48:33+00:00",
"created_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4",
"updated_utc": "2026-06-17T01:48:33+00:00",
"updated_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4",
"reviewed_utc": "2026-06-17T01:48:33+00:00",
"reviewed_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4"
},
"capabilities": [
"Evaluation protocol documentation",
"Model comparison support",
"Task taxonomy mapping",
"Reproducible benchmark anchoring",
"Relationship graph compatibility"
],
"limitations": [
"Benchmark results can become stale as models improve and evaluation protocols evolve.",
"A benchmark measures a defined task scope rather than complete intelligence."
],
"tags": [
"Benchmark",
"Evaluation",
"AI Safety",
"Agent Evaluation"
],
"timeline": [
{
"date": "2020",
"title": "Public release or documentation",
"description": "EleutherAI LM Evaluation Harness appears in public documentation, project records, benchmark descriptions or research references associated with EleutherAI."
}
],
"relationships": [
{
"target": "EleutherAI",
"type": "Associated organization",
"description": "EleutherAI is the organization, project community or institutional context associated with EleutherAI LM Evaluation Harness.",
"evidence_level": "documentary"
},
{
"target": "AI evaluation",
"type": "Domain context",
"description": "EleutherAI LM Evaluation Harness belongs to the AI evaluation layer of the intelligent-entity registry.",
"evidence_level": "documentary"
}
],
"sources": [
{
"label": "Official documentation",
"url": "https://github.com/EleutherAI/lm-evaluation-harness",
"source_type": "Official / Research Source",
"verification_status": "verified"
}
]
}
}