Structured entry data for human tools, AI systems and machine clients.
{
"@context": [
"https://schema.org",
{
"lxw": "https://lxkeys.world/schema/"
}
],
"@type": "Thing",
"identifier": "ENT-00002665",
"name": "EleutherAI LM Evaluation Harness",
"alternateName": [],
"additionalType": {
"category": "Data and Evaluation",
"type": "Benchmark",
"subtype": "",
"lxkeysEntity": false
},
"description": "EleutherAI LM Evaluation Harness is a AI benchmark associated with EleutherAI, classified in LXKeys.world as AI Benchmark / Evaluation.",
"creator": "EleutherAI",
"url": "https://lxkeys.world/entry.php?id=ENT-00002665&lang=en",
"sameAs": "https://github.com/EleutherAI/lm-evaluation-harness",
"image": "",
"lxkeysWorld": {
"worldId": "ENT-00002665",
"kind": "Data and Evaluation",
"type": "Benchmark",
"subtype": "",
"classification": "AI Benchmark / Evaluation",
"organization": "EleutherAI",
"originContext": "AI evaluation",
"firstPublicAppearance": "2020",
"currentStatus": "Active",
"documentationStatus": "Documented",
"documentationIndex": {
"total": 82,
"documentation": 25,
"evidence": 13,
"structure": 25,
"relationships": 19,
"level": "Level V Persistent"
},
"lxCalendarium": {
"start_date_utc": "2023-04-01",
"created_utc": "2026-06-17T01:48:33+00:00",
"created_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4",
"updated_utc": "2026-06-17T01:48:33+00:00",
"updated_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4"
},
"facts": [],
"capabilities": [
"Evaluation protocol documentation",
"Model comparison support",
"Task taxonomy mapping",
"Reproducible benchmark anchoring",
"Relationship graph compatibility"
],
"limitations": [
"Benchmark results can become stale as models improve and evaluation protocols evolve.",
"A benchmark measures a defined task scope rather than complete intelligence."
],
"tags": [
"Benchmark",
"Evaluation",
"AI Safety",
"Agent Evaluation"
],
"timeline": [
{
"date": "2020",
"title": "Initial benchmark release",
"description": "EleutherAI LM Evaluation Harness entered the documented public record in 2020. This event is retained at the precision supported by the Entry’s reviewed source history.",
"source_url": "https://github.com/EleutherAI/lm-evaluation-harness",
"verification_status": "source_backed_curated_baseline"
}
],
"relationships": [
{
"target": "EleutherAI",
"type": "Associated organization",
"description": "EleutherAI is the organization, project community or institutional context associated with EleutherAI LM Evaluation Harness.",
"evidence_level": "documentary",
"target_id": "ENT-00000726"
},
{
"target": "AI evaluation",
"type": "Domain context",
"description": "EleutherAI LM Evaluation Harness belongs to the AI evaluation layer of the intelligent-entity registry.",
"evidence_level": "documentary"
}
],
"sources": [
{
"label": "Official documentation",
"url": "https://github.com/EleutherAI/lm-evaluation-harness",
"source_type": "Official / Research Source",
"verification_status": "verified",
"authority": "TRUSTED_PRIMARY",
"role": "code_repository",
"update_enabled": false,
"authority_basis": "curated-corpus-refresh-2026-09-08"
}
],
"canonical": [],
"imageMeta": []
}
}