{
    "schema": "lxkeys.world.entry@2.2.1",
    "language": "en",
    "entry": {
        "id": "ENT-00002961",
        "status": "published",
        "name": "Jigsaw Toxic Comment Classification",
        "aliases": [],
        "entity_type": "Dataset",
        "classification": "Language / Training Dataset",
        "creator": "Jigsaw / Conversation AI",
        "organization": "Jigsaw / Conversation AI",
        "origin_context": "Language model evaluation and NLP data",
        "first_public_appearance": "2018",
        "current_status": "Active",
        "official_website": "https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge",
        "image": "",
        "short_description": "Jigsaw Toxic Comment Classification is a dataset associated with Jigsaw / Conversation AI, classified in LXKeys.world as Language / Training Dataset.",
        "public_description": "Jigsaw Toxic Comment Classification is a documented language dataset or benchmark. It is recorded as toxicity classification dataset, supporting evaluation or training for question answering, reasoning, dialogue, classification, translation or multilingual language understanding.",
        "technical_description": "Structured LXKeys.world registry record for Jigsaw Toxic Comment Classification. Entity type: Dataset; classification: Language / Training Dataset; creator/organization context: Jigsaw / Conversation AI. Canonical source anchor: https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge. The record tracks source authority, first public appearance, timeline, capabilities, limitations, registry status and graph relationships. Automatic refresh is limited to sources explicitly classified as OFFICIAL and enabled for updates; documentary and research references remain non-authoritative unless reviewed.",
        "capabilities": [
            "Dataset reference",
            "Benchmark or training-data context",
            "Task-level documentation",
            "Model evaluation support",
            "Source-based traceability"
        ],
        "limitations": [
            "Dataset coverage, licensing, annotation quality and benchmark relevance depend on the source version and use context.",
            "Performance claims should be assessed through models evaluated on the dataset rather than inferred from the dataset alone."
        ],
        "timeline": [
            {
                "date": "2018",
                "title": "Initial dataset release",
                "description": "Jigsaw Toxic Comment Classification entered the documented public record in 2018. This event is retained at the precision supported by the Entry’s reviewed source history.",
                "source_url": "https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge",
                "verification_status": "source_backed_official"
            }
        ],
        "sources": [
            {
                "label": "Official or reference source",
                "url": "https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge",
                "source_type": "Primary / Reference Source",
                "verification_status": "verified",
                "authority": "OFFICIAL",
                "role": "primary",
                "update_enabled": true,
                "authority_basis": "curated-corpus-refresh-2026-09-08"
            }
        ],
        "relationships": [
            {
                "target": "Jigsaw / Conversation AI",
                "type": "Associated organization",
                "description": "Jigsaw / Conversation AI is the organization, project community or institutional context associated with Jigsaw Toxic Comment Classification.",
                "evidence_level": "documentary"
            },
            {
                "target": "Language model evaluation and NLP data",
                "type": "Domain context",
                "description": "Jigsaw Toxic Comment Classification belongs to the Language model evaluation and NLP data layer of the intelligent-entity registry.",
                "evidence_level": "documentary"
            }
        ],
        "tags": [
            "Dataset",
            "NLP",
            "Benchmark",
            "Language Understanding"
        ],
        "registry_status": "Documented",
        "created_at": "2026-06-17T01:48:33+00:00",
        "updated_at": "2026-09-08T04:55:00+00:00",
        "dypclt_created": "D-0 Y-2 P-3 C-3 L-22 T-4",
        "dypclt_updated": "D-0 Y-2 P-3 C-3 L-22 T-4",
        "temporal_index": {
            "start_date_utc": "2023-04-01",
            "created_utc": "2026-06-17T01:48:33+00:00",
            "created_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4",
            "updated_utc": "2026-06-17T01:48:33+00:00",
            "updated_dypclt": "D-0 Y-2 P-3 C-3 L-22 T-4"
        },
        "machine_readable_purpose": "Machine-readable record for Jigsaw Toxic Comment Classification: classification, source anchor, timeline, limitations, tags and relationship mapping for intelligent-entity discovery.",
        "schema_ready": true,
        "first_public_appearance_precision": "year",
        "first_public_appearance_verification": "preserved_from_source_record",
        "image_status": "missing_official_image_candidate",
        "reviewed_at": "2026-09-08T04:55:00+00:00",
        "quality_status": "curated_official_first_baseline",
        "source_policy": "official_first_secondary_context_only",
        "quality_review": {
            "reviewed_at": "2026-09-08T04:55:00+00:00",
            "official_sources": 1,
            "trusted_primary_sources": 0,
            "secondary_sources": 0,
            "date_status": "preserved_from_source_record",
            "image_status": "missing",
            "manual_followup_required": false,
            "review_scope": "official-first structural curation; records flagged for follow-up are not claimed as individually exhaustive fact-checks"
        },
        "dypclt_reviewed": "D-0 Y-2 P-4 C-4 L-33 T-6",
        "world_id": "ENT-00002961",
        "kind": "Data and Evaluation",
        "type": "Dataset",
        "subtype": "",
        "documentation_status": "Documented",
        "facts": [],
        "canonical": [],
        "image_meta": [],
        "source_state": [],
        "i18n": {
            "en": [],
            "fr": []
        }
    },
    "computed": {
        "category": "Data and Evaluation",
        "type": "Dataset",
        "subtype": "",
        "documentation_index": {
            "total": 82,
            "documentation": 25,
            "evidence": 13,
            "structure": 25,
            "relationships": 19,
            "level": "Level V Persistent"
        },
        "is_lxkeys_entity": false,
        "graph_endpoint": "graph.php?center=ENT-00002961&depth=1"
    }
}