{
  "files": [
    {
      "path": "predictive/lending_fairness.csv",
      "bytes": 56513,
      "sha256": "6e74356aa7b6b94964da0df0b30f069ac0d0d02125db5805257ce0a10a4a489c",
      "pathway": "predictive",
      "tier": "A2",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "Synthetic credit decisions with model scores",
      "label": "y_true",
      "prediction": "y_pred",
      "score": "y_prob",
      "protected": [
        "gender",
        "race",
        "age"
      ],
      "planted": "Measured: Black applicants approved 6.0 points less often than White (42.0 vs 48.0 percent), Hispanic 4.5 points; women 9.2 points less than men (41.0 vs 50.2).",
      "rows": 1000,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race",
        "income",
        "debt_to_income",
        "credit_score",
        "employment_years",
        "loan_amount",
        "y_true",
        "y_pred",
        "y_prob"
      ]
    },
    {
      "path": "predictive/hiring_fairness.csv",
      "bytes": 52131,
      "sha256": "6f4a42b61c0eb7ad48a9677dffbb6b4397eff87ccb4d8a9db102243d8a7088d2",
      "pathway": "predictive",
      "tier": "A2",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "Synthetic hiring decisions with model scores, includes non-binary",
      "label": "y_true",
      "prediction": "y_pred",
      "score": "y_prob",
      "protected": [
        "gender",
        "ethnicity",
        "age"
      ],
      "planted": "Measured: small gaps only. At equal interview and technical scores women are selected 2.0 points less often (not significant); non-binary n = 50. Useful to check that a tool does not over-flag.",
      "rows": 1000,
      "columns": [
        "candidate_id",
        "age",
        "gender",
        "ethnicity",
        "education_level",
        "years_experience",
        "interview_score",
        "technical_score",
        "referral",
        "y_true",
        "y_pred",
        "y_prob"
      ]
    },
    {
      "path": "predictive/healthcare_fairness.csv",
      "bytes": 57619,
      "sha256": "32fd1f4a032dddd56d7504f2ec4fd69dc5ae39a0c0bf9456de3d8608670698d4",
      "pathway": "predictive",
      "tier": "A2",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "Synthetic clinical risk scores",
      "label": "y_true",
      "prediction": "y_pred",
      "score": "y_prob",
      "protected": [
        "sex",
        "race",
        "age"
      ],
      "planted": "Measured: at similar clinical profiles Black patients get risk scores 4.2 points lower. Probabilities understate risk for every group by 9 to 17 points, so the miscalibration is global, not group-specific.",
      "rows": 1000,
      "columns": [
        "patient_id",
        "age",
        "sex",
        "race",
        "bmi",
        "blood_pressure",
        "chronic_conditions",
        "prior_visits",
        "insurance_type",
        "risk_score",
        "y_true",
        "y_pred",
        "y_prob"
      ]
    },
    {
      "path": "predictive/recruitment_fairness_dataset.csv",
      "bytes": 1278441,
      "sha256": "446e1dff99ca7665412a7d6fcbcd17aa108e3d169b04d4f0cb61d50622781e0e",
      "pathway": "predictive",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "Synthetic recruitment funnel, 9 planted biases, ground truth included",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "religion",
        "disability_status",
        "sexual_orientation",
        "national_origin",
        "marital_status",
        "veteran_status"
      ],
      "planted": "Nine mechanisms, flagged per row in _bias_flags (B1 to B9). All personal details are invented.",
      "rows": 2700,
      "columns": [
        "applicant_id",
        "first_name",
        "last_name",
        "email",
        "phone",
        "date_of_birth",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "street_address",
        "city",
        "state",
        "zip_code",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "n_prior_companies",
        "employment_gap_months",
        "education_level",
        "university",
        "university_tier",
        "gpa",
        "skills",
        "skill_match_score",
        "certifications_count",
        "extracurriculars",
        "photo_url",
        "photo_attractiveness_score",
        "video_interview_score",
        "linkedin_url",
        "application_date",
        "source",
        "true_qualification_score",
        "model_score",
        "invite_decision",
        "_bias_flags"
      ]
    },
    {
      "path": "predictive/adult_test_with_predictions.csv",
      "bytes": 1753770,
      "sha256": "304a3c8ee6fc3e52771715d6157b7b408b93c7ab5a42fe3fa16bd33b49766987",
      "pathway": "predictive",
      "tier": "A2",
      "licence": "CC BY 4.0",
      "origin": "UCI Adult, Becker and Kohavi (1996), doi:10.24432/C5XW20; predictions by Validant",
      "origin_url": "https://archive.ics.uci.edu/dataset/2/adult",
      "title": "UCI Adult test split with a pinned logistic regression's predictions",
      "label": "y_true",
      "prediction": "y_pred",
      "score": "y_prob",
      "protected": [
        "sex",
        "race",
        "age"
      ],
      "planted": "Nothing planted: a real 1994 census extract. The model never sees sex or race.",
      "rows": 15060,
      "columns": [
        "age",
        "workclass",
        "fnlwgt",
        "education",
        "education_num",
        "marital_status",
        "occupation",
        "relationship",
        "race",
        "sex",
        "capital_gain",
        "capital_loss",
        "hours_per_week",
        "native_country",
        "y_true",
        "y_prob",
        "y_pred"
      ]
    },
    {
      "path": "predictive/adult_logreg_model.json",
      "bytes": 4236,
      "sha256": "6ebd59bf8e455889268c43c28f8876890f870d56e42f4e4b7f8192a7470a0076",
      "pathway": "predictive",
      "tier": "A3",
      "licence": "CC BY 4.0",
      "origin": "Validant, trained on UCI Adult",
      "origin_url": "https://archive.ics.uci.edu/dataset/2/adult",
      "title": "The same model, every weight: rebuild predict() in ten lines",
      "protected": [],
      "planted": "n/a"
    },
    {
      "path": "scripts/get_compas.py",
      "bytes": 2482,
      "sha256": "2b98e40af859bc22d95138bf038810f7ce8ce8d6d7c7ea0b2d138f092158a267",
      "pathway": "predictive",
      "tier": "A2",
      "licence": "MIT (script); data is not re-hosted",
      "origin": "ProPublica, compas-analysis",
      "origin_url": "https://github.com/propublica/compas-analysis",
      "title": "Download COMPAS from ProPublica and add prediction columns",
      "protected": [
        "race",
        "sex",
        "age"
      ],
      "planted": "Nothing planted: the real Northpointe decile scores."
    },
    {
      "path": "synthetic/clean.csv",
      "bytes": 118288,
      "sha256": "a1bd6f577d67b76bc7456cb4adf2ffeeb7a340acc723066bc770bc2a3ef88bef",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Control: no bias planted",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "None. A correct detector reports nothing here.",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/gender_penalty.csv",
      "bytes": 118797,
      "sha256": "bc0a91743c0a7da0ac852f6d6f27b7531fef54359265af25393d3434996a3494",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Gender penalty in tech roles",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "Half of female invites in tech roles removed (B1).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/race_penalty.csv",
      "bytes": 118976,
      "sha256": "6aebab36bfe188077de3da8eedfdff2c5b67c6619b3c4ee8400190d753510592",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Race penalty",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "Half of Black and Hispanic invites removed (B2).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/age_cliff.csv",
      "bytes": 118493,
      "sha256": "d34b7859d7f607a7ed17eb55bd713a52819605459499682decc6d5ad3d2daa0b",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Age cliff",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "Applicants under 25 never invited (B3).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/disability_penalty.csv",
      "bytes": 118557,
      "sha256": "94b95cd77859d1f8714de754f7fcd334ca1ba62685124dd9440266eeb20297fe",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Disability penalty",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "80 percent of invites removed for any reported disability (B7).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/zip_redlining.csv",
      "bytes": 118937,
      "sha256": "b3ef176c870734de327d63f5cfb24068ea42b194d681ec10890d52c256609f98",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "ZIP code redlining (a proxy)",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "Minority-majority ZIP invites halved; race itself untouched (B4).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/surname_proxy.csv",
      "bytes": 115107,
      "sha256": "7f9580a9e3110759cf000c73c81ba0977b042d02b9c521aaade2834cd095bcbb",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Surname proxy, race column removed",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "Surnames typical of Black and Hispanic applicants lose half their invites; race_ethnicity is not in the file (B5).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "synthetic/photo_laundering.csv",
      "bytes": 118903,
      "sha256": "a5205d150dbc7867a3b625e3663750df33292b63d42a3941795a4043851e9004",
      "pathway": "synthetic",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (generated)",
      "origin_url": "",
      "title": "Photo score laundering",
      "label": "true_qualification_score",
      "prediction": "invite_decision",
      "score": "model_score",
      "protected": [
        "gender",
        "race_ethnicity",
        "age",
        "disability_status"
      ],
      "planted": "A photo score correlated with race drives half the decision (B9).",
      "rows": 500,
      "columns": [
        "applicant_id",
        "age",
        "gender",
        "race_ethnicity",
        "national_origin",
        "religion",
        "marital_status",
        "sexual_orientation",
        "veteran_status",
        "disability_status",
        "primary_language",
        "zip_minority_majority",
        "role_applied",
        "years_experience",
        "education_level",
        "university_tier",
        "skill_match_score",
        "photo_attractiveness_score",
        "last_name",
        "first_name",
        "application_date",
        "source",
        "invite_decision",
        "true_qualification_score",
        "_bias_flags",
        "model_score"
      ]
    },
    {
      "path": "llm/generative_support_fairness.csv",
      "bytes": 39958,
      "sha256": "e3e86121d222a4a52cb35f435df7206499af9203c24d854d93a68556fa83bd65",
      "pathway": "llm",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "Support-bot prompts and replies by customer gender and ethnicity",
      "protected": [
        "gender",
        "ethnicity"
      ],
      "planted": "Tone and refusal differ by group in the replies.",
      "rows": 210,
      "columns": [
        "prompt",
        "response",
        "gender",
        "ethnicity"
      ]
    },
    {
      "path": "llm/discrim_eval_qwen25_7b.csv",
      "bytes": 445404,
      "sha256": "cae9945e172fefee2c484c8e9f437814c8aa1e0000fac5b8db2e1b5556cc1103",
      "pathway": "llm",
      "tier": "A2",
      "licence": "CC BY 4.0",
      "origin": "Prompts: Anthropic discrim-eval (Tamkin et al. 2023); answers and logprobs: Validant run of qwen2.5:7b",
      "origin_url": "https://huggingface.co/datasets/Anthropic/discrim-eval",
      "title": "Real yes/no decisions with log probabilities from a pinned open model",
      "protected": [
        "age",
        "gender",
        "race"
      ],
      "planted": "Nothing planted: measured behaviour of the model.",
      "rows": 3240,
      "columns": [
        "decision_question_id",
        "age",
        "gender",
        "race",
        "fill_type",
        "answer",
        "p_yes_raw",
        "p_no_raw",
        "p_yes",
        "model",
        "model_digest"
      ]
    },
    {
      "path": "agent/agent_traces_support.csv",
      "bytes": 37917,
      "sha256": "6ecd2d323ffc7555d10b96e3b94deba27c490dd605ee123694931a0e14007a84",
      "pathway": "agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "Benefits-triage agent, one row per episode",
      "protected": [
        "gender"
      ],
      "planted": "Tool choice and routing differ by gender.",
      "rows": 240,
      "columns": [
        "trace_id",
        "gender",
        "tool",
        "route",
        "steps",
        "outcome",
        "timestamp",
        "memory_reads",
        "memory_writes",
        "memory_read_text",
        "memory_write_text"
      ]
    },
    {
      "path": "agent/agent_traces_otel.jsonl",
      "bytes": 497933,
      "sha256": "5c7dd211f12ca6d1c39340644d4d3bc1b6b3dfbcfcbb58024879790a3ca6218f",
      "pathway": "agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "The same agent as raw OpenTelemetry GenAI spans",
      "protected": [
        "gender"
      ],
      "planted": "As above.",
      "rows": 2130
    },
    {
      "path": "agent/agent_traces_langfuse.jsonl",
      "bytes": 254764,
      "sha256": "d3bf2ef9f64ac092906a364d4a8528b396fd3fa03232fd2316d3ce69ff1b962a",
      "pathway": "agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant (synthetic)",
      "title": "The same agent as a Langfuse export",
      "protected": [
        "gender"
      ],
      "planted": "As above.",
      "rows": 240
    },
    {
      "path": "agent/loan_agent_persona_runs.csv",
      "bytes": 138188,
      "sha256": "bf3ae4f22d98a8c6961e131305b9111cf796db9244e09533cd5d657c55dc934c",
      "pathway": "agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant run of qwen2.5:7b",
      "title": "A real tool-calling loan agent, counterfactual personas, one row per episode",
      "protected": [
        "persona_gender",
        "persona_origin"
      ],
      "planted": "Nothing planted: measured behaviour of the agent.",
      "rows": 780,
      "columns": [
        "trace_id",
        "group",
        "persona_gender",
        "persona_origin",
        "persona_name",
        "profile_id",
        "repeat",
        "seed",
        "tool",
        "tool_calls",
        "steps",
        "outcome",
        "timestamp",
        "model",
        "model_digest"
      ]
    },
    {
      "path": "agent/loan_agent_otel.jsonl",
      "bytes": 415854,
      "sha256": "3ab291bd8c5e2726119a6612146453ec16e88309cb163c1f4ec047b69feb4ceb",
      "pathway": "agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant run of qwen2.5:7b",
      "title": "The same runs as OpenTelemetry GenAI spans",
      "protected": [
        "persona_gender",
        "persona_origin"
      ],
      "planted": "Nothing planted.",
      "rows": 1550
    },
    {
      "path": "scripts/loan_agent_persona_harness.py",
      "bytes": 12051,
      "sha256": "1f85937085533a72fb2464c4fcdf9652bc4df2d99521510283b0068987e5387f",
      "pathway": "agent",
      "tier": "A1",
      "licence": "MIT",
      "origin": "Validant",
      "title": "The harness that produced the runs: point it at your own agent",
      "protected": [],
      "planted": "n/a"
    },
    {
      "path": "multi_agent/loan_committee_runs.csv",
      "bytes": 118483,
      "sha256": "57e1d55e27e4f2d0b36ed54bf1879972cdb0d1186dc50bfc975f3775dfc8c492",
      "pathway": "multi_agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant run of qwen2.5:7b",
      "title": "Two-agent loan committee: screener, then reviewer, per-agent outputs",
      "protected": [
        "persona_gender",
        "persona_origin"
      ],
      "planted": "Nothing planted: measured behaviour of the system.",
      "rows": 780,
      "columns": [
        "sample_id",
        "group",
        "persona_gender",
        "persona_origin",
        "persona_name",
        "profile_id",
        "repeat",
        "seed",
        "screener_risk",
        "reviewer_alone_risk",
        "reviewer_alone_approve",
        "reviewer_after_risk",
        "reviewer_after_approve",
        "model",
        "model_digest"
      ]
    },
    {
      "path": "multi_agent/loan_committee_harness_trace.json",
      "bytes": 49444,
      "sha256": "81c55b698ef557651fe4bd94e129d4d4fcdaad1016facd26d099441c197984a6",
      "pathway": "multi_agent",
      "tier": "A1",
      "licence": "CC BY 4.0 (Validant)",
      "origin": "Validant run of qwen2.5:7b",
      "title": "The same run as a MultiAgentRunHarness trace",
      "protected": [
        "persona_gender",
        "persona_origin"
      ],
      "planted": "Nothing planted."
    }
  ]
}
