{
  "schema_version": 1,
  "id": "ext_race_day_context",
  "question_id": "s5_pacing_vs_difficult_day",
  "title": "Was it my pacing or a difficult race day?",
  "status": "ready",
  "as_of": "2026-09-12T14:13:07Z",
  "n": 555437,
  "input_export_id": "private-20260912-0934",
  "input_as_of": "2026-09-12T13:37:16Z",
  "live_json_as_of": "2026-09-12T04:11:06Z",
  "input_asset_sha256": "cb4e0ab7432010de29755ec5d75ca3b370e64e481eda5b20c44b359f39552204",
  "input_manifest_sha256": "a6f534d40bac291e0a1bc2eeb70633660acd00a710a090320876736de9c30f4d",
  "analysis_script_sha256": "4a26017aa2c2b2796e8d5f3498ea5e0a2f3c1e691a7332ef20cdbc256d4cdd42",
  "analysis_version": 2,
  "engine": "DuckDB 1.4.4",
  "corpus": {
    "n_records": 4462379,
    "n_cities": 34,
    "n_race_years": 256
  },
  "cohort": {
    "deduplicated": 4462379,
    "complete": 3830211,
    "increasing": 3604993,
    "timing_eligible": 3595426,
    "eligible": 3517336,
    "source_quality_excluded": 78090,
    "raw": 4462379,
    "duplicates_removed": 0,
    "missing_or_unparsed": 632168,
    "non_increasing": 225218,
    "outside_quality_bounds": 9567
  },
  "methodology_prose": [
    "For every city and race year, calculate the median and 10th\u201390th percentiles of percentage finish change versus each linked runner\u2019s recent recorded best. Require 100 such finishes per edition. Compare an individual\u2019s percentage change with that edition median to describe their position relative to the field.",
    "This contemporaneous edition reference is retrospective, includes the runner when eligible, and may shift with selection and fitness changes. It is not a weather correction or a prediction available before race day.",
    "For explicitly audited canonical-ID releases, validate unique matching CORE/FULL ID sets and matching recorded edition/name labels, then join by record ID with matching finish and all nine section durations. Legacy exports retain the one-to-one edition, trimmed lowercase name and full-timing join because their IDs are incompatible. Durations are compared to milliseconds. Keep only supplied non-ambiguous runner identities without conflicting recorded gender, inferred birth years spanning more than two years, or duplicate editions. These are candidate cross-race identities, not independently verified people; unlinked runners are absent. The linkage audit records which join was used.",
    "Recompute the benchmark as the fastest eligible finish in the two strictly earlier calendar years. The current race and every other race in its calendar year are excluded. This avoids guessing within-year chronology and prevents current-outcome leakage. It is a recent recorded best, not a fitness measurement, an expected finish, or a lifetime personal best.",
    "Performance change is 100 \u00d7 (current finish / recent recorded best \u2212 1). Negative is faster. Opening change compares 0\u201310 km pace with that earlier best\u2019s full-marathon pace. Faster opening: more than 2% faster; similar: within 2%; slower: more than 2% slower. The \u00b12% and \u00b15% cutoffs are predefined descriptions, not physiological thresholds.",
    "Use complete, strictly increasing elapsed checkpoints at 5, 10, 15, 20, 25, 30, 35, 40 and 42.195 km. Clock strings must parse as H:MM:SS or M:SS. No missing splits are interpolated.",
    "Remove exact duplicate race records, ignoring database IDs, ingestion timestamps and source URLs. Retain finishes from 90 minutes to 12 hours with every section between 2 and 20 minutes per km. These quality filters can exclude genuine unusual performances; the analysis describes this eligible cohort, not every entrant.",
    "Full source records are available in public GitHub Releases. These chart tables require at least 100 eligible observations per cell for estimate reliability. Counts refer to finishes, linked pairs or event observations as specified in that answer.",
    "These are observational results. Fitness changes, intentions, training, selection into the dataset and unmeasured conditions can explain differences. Outcome percentiles describe variation among performances, not confidence intervals or advice about the best strategy.",
    "Apply the reviewed source-quality edition exclusions for this exact export after the timing checks. Known invalid split grids, incomplete ingestion, unreconciled HOLD editions and a selected top-finisher field do not contribute to the analyses or prior benchmarks. Report source exclusions separately; an already invalid timing row is not counted twice. Missing age or recorded gender alone does not exclude an otherwise eligible finish from the overall cohort. Other sparse editions are not declared incomplete merely from their size."
  ],
  "observational": true,
  "minimum_public_cell": 100,
  "source_quality": {
    "version": 1,
    "release_tag": "private-export-20260912-0934",
    "reviewed_edition_policy": true,
    "editions": [
      {
        "city": "Honolulu",
        "year": 2016,
        "category": "invalid_split_grid",
        "reason": "Producer explicitly force-dropped this incomplete 5 km grid.",
        "raw_records": 20118,
        "deduplicated_records": 20118,
        "timing_eligible_excluded": 0
      },
      {
        "city": "Rotterdam",
        "year": 2012,
        "category": "invalid_split_grid",
        "reason": "Producer explicitly force-dropped this incomplete 5 km grid.",
        "raw_records": 7542,
        "deduplicated_records": 7542,
        "timing_eligible_excluded": 0
      },
      {
        "city": "Eindhoven",
        "year": 2009,
        "category": "invalid_split_grid",
        "reason": "Producer explicitly force-dropped this incomplete 5 km grid.",
        "raw_records": 1407,
        "deduplicated_records": 1407,
        "timing_eligible_excluded": 0
      },
      {
        "city": "Eindhoven",
        "year": 2010,
        "category": "invalid_split_grid",
        "reason": "Producer explicitly force-dropped this incomplete 5 km grid.",
        "raw_records": 1462,
        "deduplicated_records": 1462,
        "timing_eligible_excluded": 0
      },
      {
        "city": "Frankfurt",
        "year": 2019,
        "category": "invalid_split_grid",
        "reason": "Producer explicitly force-dropped this 5 km grid; numerically plausible rows are not reinstated.",
        "raw_records": 12513,
        "deduplicated_records": 12513,
        "timing_eligible_excluded": 10624
      },
      {
        "city": "New York",
        "year": 2008,
        "category": "unreconciled_hold",
        "reason": "Previously partial ingestion has grown to 38,047 records, but source-total reconciliation and explicit completion evidence are absent; keep the prior hold pending review.",
        "raw_records": 38047,
        "deduplicated_records": 38047,
        "timing_eligible_excluded": 37387
      },
      {
        "city": "Chicago",
        "year": 2016,
        "category": "incomplete_ingestion",
        "reason": "Producer handoff identifies this 6,512-record edition as incomplete and on HOLD.",
        "raw_records": 6512,
        "deduplicated_records": 6512,
        "timing_eligible_excluded": 6506
      },
      {
        "city": "Tokyo",
        "year": 2026,
        "category": "selected_field",
        "reason": "Source notes identify top-500 PDFs, not a representative full marathon field.",
        "raw_records": 353,
        "deduplicated_records": 353,
        "timing_eligible_excluded": 353
      },
      {
        "city": "Amsterdam",
        "year": 2025,
        "category": "unreconciled_hold",
        "reason": "Source audit marks HOLD: 23,328 listed versus 23,325 retrieved. This conservative hold does not establish a large bias from the three missing records.",
        "raw_records": 23325,
        "deduplicated_records": 23325,
        "timing_eligible_excluded": 23220
      },
      {
        "city": "Valencia",
        "year": 2017,
        "category": "incomplete_ingestion",
        "reason": "Source audit marks HOLD: 3,373 listed detail pages missing; the observed records also lack 20 km.",
        "raw_records": 12485,
        "deduplicated_records": 12485,
        "timing_eligible_excluded": 0
      },
      {
        "city": "Valencia",
        "year": 2018,
        "category": "incomplete_ingestion",
        "reason": "Producer handoff retains an incomplete/HOLD status; the edition also lacks the observed 20 km checkpoint.",
        "raw_records": 19236,
        "deduplicated_records": 19236,
        "timing_eligible_excluded": 0
      }
    ],
    "policy_sha256": "c08e83c993025dc41c0344be7b5442d74b3fb0588c38949481d5bc0a08f45fa0",
    "script_sha256": "433e8b9566b5afd809f4de1180fab222e63ad40678a33f6c93f08712cf820c77"
  },
  "source_quality_script_sha256": "433e8b9566b5afd809f4de1180fab222e63ad40678a33f6c93f08712cf820c77",
  "observation_unit": "eligible finishes",
  "evidence_scope": "partial comparison",
  "supporting_script_sha256": "7c6959b427fa1b06bb9600bb3377ecb226257d14aa0ec61a71a63f42bc87a3dd",
  "linkage_audit": {
    "feature_rows": 4462379,
    "safe_identity_groups": 2892236,
    "record_join_method": "canonical_record_id",
    "canonical_id_contract_verified": true,
    "candidate_link_rows": 3194070,
    "natural_key_candidate_rows": 0,
    "canonical_id_candidate_rows": 3194070,
    "ambiguous_cross_export_matches": 0,
    "linked_eligible_finishes": 3194070,
    "recent_benchmark_finishes": 555437,
    "consecutive_cross_year_pairs": 583670,
    "linked_with_supplied_date": 3185321
  },
  "narrative_script_sha256": "a9c6755fe727caa6c5a3994bd565e615067ec09c9c6f31e4e076622b810b5dcf"
}
