Classify each county by area-weighted Köppen class shares: a county is predominantly its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed (133 of 3,143 counties in the 50 states and DC). - Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and apply_koppen_metric_to_climate_data.py (writes koppenZone plus koppenPrimaryClass/koppenSecondaryClass for Mixed counties). - Move shared helpers into scripts/common/ (county loading, Köppen legend, area-weighted raster shares); fix the 180th-meridian raster window for Aleutians West. - Add check_climate_data.py to validate the app CSV. - Draw Mixed counties in app.js as diagonal stripes of their top two classes, fixed to the ground and following the map at every zoom, with a crossfade only when the stripe size changes. Filtering a class also matches Mixed counties where it is primary or secondary. - Document the rule, display, and pipeline plan in docs/ and update the README and data-source notes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
132 lines
3.6 KiB
Python
132 lines
3.6 KiB
Python
"""Load county polygons and normalize county identifiers."""
|
|
|
|
from __future__ import annotations
|
|
|
|
from pathlib import Path
|
|
|
|
import geopandas as gpd
|
|
|
|
DEFAULT_COUNTIES_GEOJSON_URL = "https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json"
|
|
|
|
STATE_FIPS_TO_ABBR = {
|
|
"01": "AL",
|
|
"02": "AK",
|
|
"04": "AZ",
|
|
"05": "AR",
|
|
"06": "CA",
|
|
"08": "CO",
|
|
"09": "CT",
|
|
"10": "DE",
|
|
"11": "DC",
|
|
"12": "FL",
|
|
"13": "GA",
|
|
"15": "HI",
|
|
"16": "ID",
|
|
"17": "IL",
|
|
"18": "IN",
|
|
"19": "IA",
|
|
"20": "KS",
|
|
"21": "KY",
|
|
"22": "LA",
|
|
"23": "ME",
|
|
"24": "MD",
|
|
"25": "MA",
|
|
"26": "MI",
|
|
"27": "MN",
|
|
"28": "MS",
|
|
"29": "MO",
|
|
"30": "MT",
|
|
"31": "NE",
|
|
"32": "NV",
|
|
"33": "NH",
|
|
"34": "NJ",
|
|
"35": "NM",
|
|
"36": "NY",
|
|
"37": "NC",
|
|
"38": "ND",
|
|
"39": "OH",
|
|
"40": "OK",
|
|
"41": "OR",
|
|
"42": "PA",
|
|
"44": "RI",
|
|
"45": "SC",
|
|
"46": "SD",
|
|
"47": "TN",
|
|
"48": "TX",
|
|
"49": "UT",
|
|
"50": "VT",
|
|
"51": "VA",
|
|
"53": "WA",
|
|
"54": "WV",
|
|
"55": "WI",
|
|
"56": "WY",
|
|
"60": "AS",
|
|
"66": "GU",
|
|
"69": "MP",
|
|
"72": "PR",
|
|
"78": "VI",
|
|
}
|
|
|
|
|
|
def normalize_fips(value: object, width: int) -> str:
|
|
"""Return a zero-padded FIPS code with the requested width."""
|
|
text = str(value).strip()
|
|
digits = "".join(ch for ch in text if ch.isdigit())
|
|
if not digits:
|
|
return ""
|
|
return digits.zfill(width)[-width:]
|
|
|
|
|
|
def load_counties(counties_geojson: Path) -> gpd.GeoDataFrame:
|
|
"""Load county polygons and normalize fields used downstream."""
|
|
if not counties_geojson.exists():
|
|
try:
|
|
print(
|
|
f"County GeoJSON not found at {counties_geojson}. "
|
|
f"Attempting download from {DEFAULT_COUNTIES_GEOJSON_URL}..."
|
|
)
|
|
gdf = gpd.read_file(DEFAULT_COUNTIES_GEOJSON_URL)
|
|
counties_geojson.parent.mkdir(parents=True, exist_ok=True)
|
|
# Cache the downloaded file for subsequent runs.
|
|
gdf.to_file(counties_geojson, driver="GeoJSON")
|
|
print(f"Downloaded and cached county GeoJSON to {counties_geojson}")
|
|
except Exception as exc:
|
|
raise FileNotFoundError(
|
|
f"County GeoJSON not found at {counties_geojson}, and download from "
|
|
f"{DEFAULT_COUNTIES_GEOJSON_URL} failed. Download the file manually "
|
|
"and rerun with --counties-geojson pointing to it."
|
|
) from exc
|
|
|
|
gdf = gpd.read_file(counties_geojson)
|
|
if gdf.crs is None:
|
|
gdf = gdf.set_crs("EPSG:4326")
|
|
else:
|
|
gdf = gdf.to_crs("EPSG:4326")
|
|
|
|
feature_id = None
|
|
if "id" in gdf.columns:
|
|
feature_id = gdf["id"]
|
|
elif "GEOID" in gdf.columns:
|
|
feature_id = gdf["GEOID"]
|
|
elif "GEOID10" in gdf.columns:
|
|
feature_id = gdf["GEOID10"]
|
|
elif "fips" in gdf.columns:
|
|
feature_id = gdf["fips"]
|
|
else:
|
|
raise ValueError("Unable to locate county FIPS identifier column in county polygons.")
|
|
|
|
gdf["county_fips"] = feature_id.map(lambda value: normalize_fips(value, 5))
|
|
gdf = gdf[gdf["county_fips"] != ""].copy()
|
|
|
|
if "NAME" in gdf.columns:
|
|
gdf["county_name"] = gdf["NAME"].fillna("").astype(str).str.strip()
|
|
elif "name" in gdf.columns:
|
|
gdf["county_name"] = gdf["name"].fillna("").astype(str).str.strip()
|
|
else:
|
|
gdf["county_name"] = gdf["county_fips"].map(lambda value: f"County {value}")
|
|
|
|
gdf["state_fips"] = gdf["county_fips"].str.slice(0, 2)
|
|
gdf["state"] = gdf["state_fips"].map(lambda code: STATE_FIPS_TO_ABBR.get(code, f"S{code}"))
|
|
gdf = gdf.sort_values("county_fips").reset_index(drop=True)
|
|
return gdf
|