"""Load county polygons and normalize county identifiers.""" from __future__ import annotations from pathlib import Path import geopandas as gpd DEFAULT_COUNTIES_GEOJSON_URL = "https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json" STATE_FIPS_TO_ABBR = { "01": "AL", "02": "AK", "04": "AZ", "05": "AR", "06": "CA", "08": "CO", "09": "CT", "10": "DE", "11": "DC", "12": "FL", "13": "GA", "15": "HI", "16": "ID", "17": "IL", "18": "IN", "19": "IA", "20": "KS", "21": "KY", "22": "LA", "23": "ME", "24": "MD", "25": "MA", "26": "MI", "27": "MN", "28": "MS", "29": "MO", "30": "MT", "31": "NE", "32": "NV", "33": "NH", "34": "NJ", "35": "NM", "36": "NY", "37": "NC", "38": "ND", "39": "OH", "40": "OK", "41": "OR", "42": "PA", "44": "RI", "45": "SC", "46": "SD", "47": "TN", "48": "TX", "49": "UT", "50": "VT", "51": "VA", "53": "WA", "54": "WV", "55": "WI", "56": "WY", "60": "AS", "66": "GU", "69": "MP", "72": "PR", "78": "VI", } def normalize_fips(value: object, width: int) -> str: """Return a zero-padded FIPS code with the requested width.""" text = str(value).strip() digits = "".join(ch for ch in text if ch.isdigit()) if not digits: return "" return digits.zfill(width)[-width:] def load_counties(counties_geojson: Path) -> gpd.GeoDataFrame: """Load county polygons and normalize fields used downstream.""" if not counties_geojson.exists(): try: print( f"County GeoJSON not found at {counties_geojson}. " f"Attempting download from {DEFAULT_COUNTIES_GEOJSON_URL}..." ) gdf = gpd.read_file(DEFAULT_COUNTIES_GEOJSON_URL) counties_geojson.parent.mkdir(parents=True, exist_ok=True) # Cache the downloaded file for subsequent runs. gdf.to_file(counties_geojson, driver="GeoJSON") print(f"Downloaded and cached county GeoJSON to {counties_geojson}") except Exception as exc: raise FileNotFoundError( f"County GeoJSON not found at {counties_geojson}, and download from " f"{DEFAULT_COUNTIES_GEOJSON_URL} failed. Download the file manually " "and rerun with --counties-geojson pointing to it." ) from exc gdf = gpd.read_file(counties_geojson) if gdf.crs is None: gdf = gdf.set_crs("EPSG:4326") else: gdf = gdf.to_crs("EPSG:4326") feature_id = None if "id" in gdf.columns: feature_id = gdf["id"] elif "GEOID" in gdf.columns: feature_id = gdf["GEOID"] elif "GEOID10" in gdf.columns: feature_id = gdf["GEOID10"] elif "fips" in gdf.columns: feature_id = gdf["fips"] else: raise ValueError("Unable to locate county FIPS identifier column in county polygons.") gdf["county_fips"] = feature_id.map(lambda value: normalize_fips(value, 5)) gdf = gdf[gdf["county_fips"] != ""].copy() if "NAME" in gdf.columns: gdf["county_name"] = gdf["NAME"].fillna("").astype(str).str.strip() elif "name" in gdf.columns: gdf["county_name"] = gdf["name"].fillna("").astype(str).str.strip() else: gdf["county_name"] = gdf["county_fips"].map(lambda value: f"County {value}") gdf["state_fips"] = gdf["county_fips"].str.slice(0, 2) gdf["state"] = gdf["state_fips"].map(lambda code: STATE_FIPS_TO_ABBR.get(code, f"S{code}")) gdf = gdf.sort_values("county_fips").reset_index(drop=True) return gdf