Complete Köppen-Geiger filter review with Mixed climate class

Classify each county by area-weighted Köppen class shares: a county is
predominantly its top class when that class covers at least 50% of its
land and leads the runner-up by at least 5 percentage points; otherwise
it is Mixed (133 of 3,143 counties in the 50 states and DC).

- Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and
  apply_koppen_metric_to_climate_data.py (writes koppenZone plus
  koppenPrimaryClass/koppenSecondaryClass for Mixed counties).
- Move shared helpers into scripts/common/ (county loading, Köppen
  legend, area-weighted raster shares); fix the 180th-meridian raster
  window for Aleutians West.
- Add check_climate_data.py to validate the app CSV.
- Draw Mixed counties in app.js as diagonal stripes of their top two
  classes, fixed to the ground and following the map at every zoom, with
  a crossfade only when the stripe size changes. Filtering a class also
  matches Mixed counties where it is primary or secondary.
- Document the rule, display, and pipeline plan in docs/ and update the
  README and data-source notes.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-09-14 02:54:25 -04:00
co-authored by Claude Opus 5
parent 92fbbfb2e9
commit 4d2b3e3d44
20 changed files with 6400 additions and 3450 deletions
+30 -199
View File
@@ -31,102 +31,11 @@ import rasterio
import xarray as xr
from affine import Affine
from rasterio.features import geometry_mask
from shapely.geometry.base import BaseGeometry
DEFAULT_COUNTIES_GEOJSON_URL = "https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json"
STATE_FIPS_TO_ABBR = {
"01": "AL",
"02": "AK",
"04": "AZ",
"05": "AR",
"06": "CA",
"08": "CO",
"09": "CT",
"10": "DE",
"11": "DC",
"12": "FL",
"13": "GA",
"15": "HI",
"16": "ID",
"17": "IL",
"18": "IN",
"19": "IA",
"20": "KS",
"21": "KY",
"22": "LA",
"23": "ME",
"24": "MD",
"25": "MA",
"26": "MI",
"27": "MN",
"28": "MS",
"29": "MO",
"30": "MT",
"31": "NE",
"32": "NV",
"33": "NH",
"34": "NJ",
"35": "NM",
"36": "NY",
"37": "NC",
"38": "ND",
"39": "OH",
"40": "OK",
"41": "OR",
"42": "PA",
"44": "RI",
"45": "SC",
"46": "SD",
"47": "TN",
"48": "TX",
"49": "UT",
"50": "VT",
"51": "VA",
"53": "WA",
"54": "WV",
"55": "WI",
"56": "WY",
"60": "AS",
"66": "GU",
"69": "MP",
"72": "PR",
"78": "VI",
}
# Beck et al legend key is expected as text file, but this default handles common codes.
DEFAULT_KOPPEN_CODE_MAP = {
1: "Af",
2: "Am",
3: "Aw",
4: "BWh",
5: "BWk",
6: "BSh",
7: "BSk",
8: "Csa",
9: "Csb",
10: "Csc",
11: "Cwa",
12: "Cwb",
13: "Cwc",
14: "Cfa",
15: "Cfb",
16: "Cfc",
17: "Dsa",
18: "Dsb",
19: "Dsc",
20: "Dsd",
21: "Dwa",
22: "Dwb",
23: "Dwc",
24: "Dwd",
25: "Dfa",
26: "Dfb",
27: "Dfc",
28: "Dfd",
29: "ET",
30: "EF",
}
from common.counties import load_counties, normalize_fips
from common.county_zonal_stats import geometry_window, split_at_antimeridian
from common.koppen_legend import load_koppen_legend
MONTH_NAMES = [
"January",
@@ -144,94 +53,6 @@ MONTH_NAMES = [
]
def _normalize_fips(value: object, width: int) -> str:
"""Return a zero-padded FIPS code with the requested width."""
text = str(value).strip()
digits = "".join(ch for ch in text if ch.isdigit())
if not digits:
return ""
return digits.zfill(width)[-width:]
def _load_counties(counties_geojson: Path) -> gpd.GeoDataFrame:
"""Load county polygons and normalize fields used downstream."""
if not counties_geojson.exists():
try:
print(
f"County GeoJSON not found at {counties_geojson}. "
f"Attempting download from {DEFAULT_COUNTIES_GEOJSON_URL}..."
)
gdf = gpd.read_file(DEFAULT_COUNTIES_GEOJSON_URL)
counties_geojson.parent.mkdir(parents=True, exist_ok=True)
# Cache the downloaded file for subsequent runs.
gdf.to_file(counties_geojson, driver="GeoJSON")
print(f"Downloaded and cached county GeoJSON to {counties_geojson}")
except Exception as exc:
raise FileNotFoundError(
f"County GeoJSON not found at {counties_geojson}, and download from "
f"{DEFAULT_COUNTIES_GEOJSON_URL} failed. Download the file manually "
"and rerun with --counties-geojson pointing to it."
) from exc
gdf = gpd.read_file(counties_geojson)
if gdf.crs is None:
gdf = gdf.set_crs("EPSG:4326")
else:
gdf = gdf.to_crs("EPSG:4326")
feature_id = None
if "id" in gdf.columns:
feature_id = gdf["id"]
elif "GEOID" in gdf.columns:
feature_id = gdf["GEOID"]
elif "GEOID10" in gdf.columns:
feature_id = gdf["GEOID10"]
elif "fips" in gdf.columns:
feature_id = gdf["fips"]
else:
raise ValueError("Unable to locate county FIPS identifier column in county polygons.")
gdf["county_fips"] = feature_id.map(lambda value: _normalize_fips(value, 5))
gdf = gdf[gdf["county_fips"] != ""].copy()
if "NAME" in gdf.columns:
gdf["county_name"] = gdf["NAME"].fillna("").astype(str).str.strip()
elif "name" in gdf.columns:
gdf["county_name"] = gdf["name"].fillna("").astype(str).str.strip()
else:
gdf["county_name"] = gdf["county_fips"].map(lambda value: f"County {value}")
gdf["state_fips"] = gdf["county_fips"].str.slice(0, 2)
gdf["state"] = gdf["state_fips"].map(lambda code: STATE_FIPS_TO_ABBR.get(code, f"S{code}"))
gdf = gdf.sort_values("county_fips").reset_index(drop=True)
return gdf
def _load_koppen_legend(legend_path: Path | None) -> Dict[int, str]:
"""Load Koppen raster codes, using defaults when no legend exists."""
if legend_path is None:
return DEFAULT_KOPPEN_CODE_MAP
mapping: Dict[int, str] = {}
for line in legend_path.read_text(encoding="utf-8").splitlines():
text = line.strip()
if not text or text.startswith("#"):
continue
# Handles patterns like:
# "1: Af ..." or "1 = Af" or "1 Af"
import re
match = re.match(r"^(\d+)\s*[:=]?\s*([A-Za-z]{2,3})\b", text)
if not match:
continue
key = int(match.group(1))
value = match.group(2)
mapping[key] = value
return mapping if mapping else DEFAULT_KOPPEN_CODE_MAP
def _select_data_var(dataset: xr.Dataset, preferred: str) -> str:
"""Choose the best matching climate variable from a dataset."""
if preferred in dataset.data_vars:
@@ -283,7 +104,7 @@ def _load_solar_ghi_csv(solar_ghi_csv: Path, counties: gpd.GeoDataFrame) -> List
solar_by_fips: Dict[str, float] = {}
for row in reader:
county_fips = _normalize_fips(row.get(fips_field, ""), 5)
county_fips = normalize_fips(row.get(fips_field, ""), 5)
raw_value = str(row.get("meanDailyGlobalHorizontalRadiationKwhM2Day", "")).strip()
if not county_fips or not raw_value:
continue
@@ -414,6 +235,26 @@ def _zonal_mean_raster(raster_path: Path, counties: gpd.GeoDataFrame) -> List[fl
return _zonal_mean(values, source.transform, raster_counties)
def _touched_raster_values(source, geometry: BaseGeometry, split_antimeridian: bool) -> np.ndarray:
"""Read every raster cell a geometry touches, using a small window per piece."""
pieces = split_at_antimeridian(geometry) if split_antimeridian else [geometry]
selected: List[np.ndarray] = []
for piece in pieces:
window = geometry_window(source, piece)
values = source.read(1, window=window, masked=True).filled(0)
if source.nodata is not None:
values = np.where(values == source.nodata, 0, values)
mask = geometry_mask(
[piece.__geo_interface__],
out_shape=values.shape,
transform=source.window_transform(window),
invert=True,
all_touched=True,
)
selected.append(values[mask])
return np.concatenate(selected) if selected else np.array([], dtype=np.int64)
def _zonal_majority_class(koppen_raster: Path, counties: gpd.GeoDataFrame, code_map: Dict[int, str]) -> List[str]:
"""Assign each county its most common Koppen-Geiger class."""
classes: List[str] = []
@@ -421,21 +262,11 @@ def _zonal_majority_class(koppen_raster: Path, counties: gpd.GeoDataFrame, code_
raster_counties = counties
if source.crs is not None and counties.crs is not None and counties.crs != source.crs:
raster_counties = counties.to_crs(source.crs)
data = source.read(1, masked=True)
values = np.asarray(data.filled(0))
if source.nodata is not None:
values = np.where(values == source.nodata, 0, values)
# The 180-degree split only makes sense for longitude/latitude rasters.
split_antimeridian = source.crs is None or source.crs.is_geographic
for geometry in raster_counties.geometry:
mask = geometry_mask(
[geometry.__geo_interface__],
out_shape=values.shape,
transform=source.transform,
invert=True,
all_touched=True,
)
selected = values[mask]
selected = _touched_raster_values(source, geometry, split_antimeridian)
selected = selected[selected != 0]
if selected.size == 0:
classes.append("Cfa")
@@ -545,9 +376,9 @@ def build_county_records(
solar_ghi_csv: Path | None,
) -> Dict[str, dict]:
"""Build county climate records consumed by the web app."""
counties = _load_counties(counties_geojson)
counties = load_counties(counties_geojson)
koppen_classes = _zonal_majority_class(koppen_raster, counties, _load_koppen_legend(koppen_legend))
koppen_classes = _zonal_majority_class(koppen_raster, counties, load_koppen_legend(koppen_legend))
monthly_tavg = xr.open_dataset(monthly_tavg_nc, decode_times=True)
monthly_prcp = xr.open_dataset(monthly_prcp_nc, decode_times=True)