Complete Köppen-Geiger filter review with Mixed climate class

Classify each county by area-weighted Köppen class shares: a county is
predominantly its top class when that class covers at least 50% of its
land and leads the runner-up by at least 5 percentage points; otherwise
it is Mixed (133 of 3,143 counties in the 50 states and DC).

- Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and
  apply_koppen_metric_to_climate_data.py (writes koppenZone plus
  koppenPrimaryClass/koppenSecondaryClass for Mixed counties).
- Move shared helpers into scripts/common/ (county loading, Köppen
  legend, area-weighted raster shares); fix the 180th-meridian raster
  window for Aleutians West.
- Add check_climate_data.py to validate the app CSV.
- Draw Mixed counties in app.js as diagonal stripes of their top two
  classes, fixed to the ground and following the map at every zoom, with
  a crossfade only when the stripe size changes. Filtering a class also
  matches Mixed counties where it is primary or secondary.
- Document the rule, display, and pipeline plan in docs/ and update the
  README and data-source notes.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
2026-09-14 02:54:25 -04:00
co-authored by Claude Opus 5
parent 92fbbfb2e9
commit 4d2b3e3d44
20 changed files with 6400 additions and 3450 deletions
+155
View File
@@ -0,0 +1,155 @@
"""Build the county Koppen-Geiger metric file from area-weighted class shares.
Writes data/metrics/koppen.csv. A county is predominantly its top class when
that class covers at least 50% of the county's land and leads the runner-up by
at least 5 percentage points; otherwise it is "Mixed". Counties with no valid
raster cells are left blank. See docs/filter-calculations.md, section 1.
Run:
.venv\\Scripts\\python.exe scripts\\build_county_koppen_metric.py
"""
from __future__ import annotations
import argparse
import csv
from collections import Counter
from pathlib import Path
from typing import Dict, List, Tuple
import geopandas as gpd
import rasterio
from common.counties import load_counties
from common.county_zonal_stats import DEFAULT_SUBCELLS, area_weighted_class_weights
from common.koppen_legend import load_koppen_legend
PROJECT_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_COUNTIES_GEOJSON = PROJECT_ROOT / "data" / "geojson-counties-fips.json"
DEFAULT_KOPPEN_RASTER = PROJECT_ROOT / "data" / "koppen_geiger_tif" / "1991_2020" / "koppen_geiger_0p00833333.tif"
DEFAULT_KOPPEN_LEGEND = PROJECT_ROOT / "data" / "koppen_geiger_tif" / "legend.txt"
DEFAULT_OUT = PROJECT_ROOT / "data" / "metrics" / "koppen.csv"
PREDOMINANT_MIN_SHARE = 0.50
PREDOMINANT_MIN_GAP = 0.05
MIXED_CLASS = "Mixed"
# Keeps shares that sit exactly on a cutoff from failing on floating-point error.
RULE_TOLERANCE = 1e-9
FIELDS = [
"countyFips",
"countyName",
"state",
"koppenZone",
"koppenTopClass",
"koppenTopShare",
"koppenSecondClass",
"koppenSecondShare",
]
def rank_class_shares(weights: Dict[int, float], code_map: Dict[int, str]) -> List[Tuple[str, float]]:
"""Convert per-code area weights to class shares, largest first.
Exact ties go to the smaller raster code, matching the previous build.
"""
total = sum(weights.values())
if total <= 0:
return []
unknown = sorted(code for code in weights if code not in code_map)
if unknown:
raise ValueError(f"Raster codes {unknown} are not in the Koppen legend.")
ranked = sorted(weights.items(), key=lambda item: (-item[1], item[0]))
return [(code_map[code], weight / total) for code, weight in ranked]
def classify(ranked: List[Tuple[str, float]]) -> str:
"""Return the predominant class, "Mixed", or blank when there is no data."""
if not ranked:
return ""
top_share = ranked[0][1]
second_share = ranked[1][1] if len(ranked) > 1 else 0.0
has_majority = top_share >= PREDOMINANT_MIN_SHARE - RULE_TOLERANCE
has_clear_lead = top_share - second_share >= PREDOMINANT_MIN_GAP - RULE_TOLERANCE
return ranked[0][0] if has_majority and has_clear_lead else MIXED_CLASS
def _format_share(ranked: List[Tuple[str, float]], index: int) -> Tuple[str, str]:
"""Return the class and 4-decimal share at a rank, or blanks."""
if index >= len(ranked):
return "", ""
code, share = ranked[index]
return code, f"{share:.4f}"
def build_koppen_records(
counties: gpd.GeoDataFrame,
koppen_raster: Path,
code_map: Dict[int, str],
subcells: int = DEFAULT_SUBCELLS,
) -> List[dict]:
"""Classify every county and return its metric row."""
records: List[dict] = []
with rasterio.open(koppen_raster) as source:
raster_counties = counties
if source.crs is not None and counties.crs is not None and counties.crs != source.crs:
raster_counties = counties.to_crs(source.crs)
geographic = source.crs is None or source.crs.is_geographic
for county, geometry in zip(counties.itertuples(), raster_counties.geometry):
weights = area_weighted_class_weights(source, geometry, geographic=geographic, subcells=subcells)
ranked = rank_class_shares(weights, code_map)
top_class, top_share = _format_share(ranked, 0)
second_class, second_share = _format_share(ranked, 1)
records.append(
{
"countyFips": county.county_fips,
"countyName": county.county_name,
"state": county.state,
"koppenZone": classify(ranked),
"koppenTopClass": top_class,
"koppenTopShare": top_share,
"koppenSecondClass": second_class,
"koppenSecondShare": second_share,
}
)
return records
def write_records(records: List[dict], out_file: Path) -> None:
"""Write the Koppen metric rows."""
out_file.parent.mkdir(parents=True, exist_ok=True)
with out_file.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=FIELDS)
writer.writeheader()
writer.writerows(records)
def parse_args() -> argparse.Namespace:
"""Define and parse command-line options for this builder."""
parser = argparse.ArgumentParser(description="Build the county Koppen-Geiger metric file.")
parser.add_argument("--counties-geojson", type=Path, default=DEFAULT_COUNTIES_GEOJSON, help="County polygon GeoJSON path.")
parser.add_argument("--koppen-raster", type=Path, default=DEFAULT_KOPPEN_RASTER, help="Koppen-Geiger raster TIFF path.")
parser.add_argument("--koppen-legend", type=Path, default=DEFAULT_KOPPEN_LEGEND, help="legend.txt mapping raster codes.")
parser.add_argument("--subcells", type=int, default=DEFAULT_SUBCELLS, help="Sub-cells per raster cell edge.")
parser.add_argument("--out", type=Path, default=DEFAULT_OUT, help="Output metric CSV path.")
return parser.parse_args()
def main() -> None:
args = parse_args()
counties = load_counties(args.counties_geojson)
records = build_koppen_records(counties, args.koppen_raster, load_koppen_legend(args.koppen_legend), args.subcells)
write_records(records, args.out)
outcomes = Counter(
"blank" if not r["koppenZone"] else "Mixed" if r["koppenZone"] == MIXED_CLASS else "predominant" for r in records
)
print(
f"Wrote {len(records)} counties to {args.out}: {outcomes['predominant']} predominant, "
f"{outcomes['Mixed']} Mixed, {outcomes['blank']} blank."
)
if __name__ == "__main__":
main()