{
    "schema": "lxkeys.world.entry@2.2.1",
    "language": "en",
    "entry": {
        "id": "ENT-00001968",
        "status": "published",
        "name": "MLE-bench",
        "aliases": [],
        "entity_type": "Benchmark",
        "classification": "AI Benchmark / Evaluation",
        "creator": "OpenAI",
        "organization": "OpenAI",
        "origin_context": "Public AI and technical record",
        "first_public_appearance": "2024",
        "current_status": "Active",
        "official_website": "https://github.com/openai/mle-bench",
        "image": "",
        "short_description": "MLE-bench is a AI benchmark associated with OpenAI, classified in LXKeys.world as AI Benchmark / Evaluation.",
        "public_description": "MLE-bench is a documented AI dataset or benchmark associated with OpenAI. The record identifies its role in evaluation, training, measurement or comparison of intelligent systems, with emphasis on the source context and the type of capability it helps assess.",
        "technical_description": "Structured LXKeys.world registry record for MLE-bench. Entity type: Benchmark; classification: AI Benchmark / Evaluation; creator/organization context: OpenAI. Canonical source anchor: https://github.com/openai/mle-bench. The record tracks source authority, first public appearance, timeline, capabilities, limitations, registry status and graph relationships. Automatic refresh is limited to sources explicitly classified as OFFICIAL and enabled for updates; documentary and research references remain non-authoritative unless reviewed.",
        "capabilities": [
            "Documented source context",
            "Public reference anchor",
            "Structured classification"
        ],
        "limitations": [
            "Benchmark scores depend on the exact dataset version, prompt or evaluation protocol, scoring implementation and contamination controls.",
            "Leaderboard performance should not be treated as a complete measure of real-world capability or safety.",
            "Benchmark relevance can decline as models, data and evaluation practices evolve."
        ],
        "timeline": [
            {
                "date": "2024",
                "title": "Initial benchmark release",
                "description": "MLE-bench entered the documented public record in 2024. This event is retained at the precision supported by the Entry’s reviewed source history.",
                "source_url": "https://github.com/openai/mle-bench",
                "verification_status": "source_backed_curated_baseline"
            }
        ],
        "sources": [
            {
                "label": "MLE-bench repository",
                "url": "https://github.com/openai/mle-bench",
                "source_type": "Official Repository",
                "verification_status": "verified",
                "authority": "TRUSTED_PRIMARY",
                "role": "code_repository",
                "update_enabled": false,
                "authority_basis": "curated-corpus-refresh-2026-09-08"
            }
        ],
        "relationships": [
            {
                "target": "OpenAI",
                "type": "Published by",
                "description": "MLE-bench is associated with OpenAI through its documented source context.",
                "evidence_level": "documentary",
                "target_id": "ENT-00000048"
            }
        ],
        "tags": [
            "Documented Entity",
            "AI System",
            "Registry Candidate"
        ],
        "registry_status": "Documented",
        "created_at": "2026-06-16T23:59:29+00:00",
        "updated_at": "2026-09-08T04:55:00+00:00",
        "dypclt_created": "D-0 Y-2 P-3 C-3 L-21 T-3",
        "dypclt_updated": "D-0 Y-2 P-3 C-3 L-21 T-3",
        "temporal_index": {
            "start_date_utc": "2023-04-01",
            "created_utc": "2026-06-16T23:59:29+00:00",
            "created_dypclt": "D-0 Y-2 P-3 C-3 L-21 T-3",
            "updated_utc": "2026-06-16T23:59:29+00:00",
            "updated_dypclt": "D-0 Y-2 P-3 C-3 L-21 T-3"
        },
        "documentation_status": "Documented",
        "machine_readable_purpose": "Machine-readable registry record for MLE-bench: entity type, classification, organization, source, public appearance, capability profile and graph relationships.",
        "schema_ready": true,
        "first_public_appearance_precision": "year",
        "first_public_appearance_verification": "preserved_from_source_record",
        "image_status": "missing_official_image_candidate",
        "reviewed_at": "2026-09-08T04:55:00+00:00",
        "quality_status": "curated_official_first_baseline",
        "source_policy": "official_first_secondary_context_only",
        "quality_review": {
            "reviewed_at": "2026-09-08T04:55:00+00:00",
            "official_sources": 0,
            "trusted_primary_sources": 1,
            "secondary_sources": 0,
            "date_status": "preserved_from_source_record",
            "image_status": "missing",
            "manual_followup_required": true,
            "review_scope": "official-first structural curation; records flagged for follow-up are not claimed as individually exhaustive fact-checks"
        },
        "dypclt_reviewed": "D-0 Y-2 P-4 C-4 L-33 T-6",
        "world_id": "ENT-00001968",
        "kind": "Data and Evaluation",
        "type": "Benchmark",
        "subtype": "",
        "facts": [],
        "canonical": [],
        "image_meta": [],
        "source_state": [],
        "i18n": {
            "en": [],
            "fr": []
        }
    },
    "computed": {
        "category": "Data and Evaluation",
        "type": "Benchmark",
        "subtype": "",
        "documentation_index": {
            "total": 74,
            "documentation": 25,
            "evidence": 13,
            "structure": 25,
            "relationships": 11,
            "level": "Level IV Connected"
        },
        "is_lxkeys_entity": false,
        "graph_endpoint": "graph.php?center=ENT-00001968&depth=1"
    }
}