Complete Köppen-Geiger filter review with Mixed climate class
Classify each county by area-weighted Köppen class shares: a county is predominantly its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed (133 of 3,143 counties in the 50 states and DC). - Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and apply_koppen_metric_to_climate_data.py (writes koppenZone plus koppenPrimaryClass/koppenSecondaryClass for Mixed counties). - Move shared helpers into scripts/common/ (county loading, Köppen legend, area-weighted raster shares); fix the 180th-meridian raster window for Aleutians West. - Add check_climate_data.py to validate the app CSV. - Draw Mixed counties in app.js as diagonal stripes of their top two classes, fixed to the ground and following the map at every zoom, with a crossfade only when the stripe size changes. Filtering a class also matches Mixed counties where it is primary or secondary. - Document the rule, display, and pipeline plan in docs/ and update the README and data-source notes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,155 @@
|
||||
"""Build the county Koppen-Geiger metric file from area-weighted class shares.
|
||||
|
||||
Writes data/metrics/koppen.csv. A county is predominantly its top class when
|
||||
that class covers at least 50% of the county's land and leads the runner-up by
|
||||
at least 5 percentage points; otherwise it is "Mixed". Counties with no valid
|
||||
raster cells are left blank. See docs/filter-calculations.md, section 1.
|
||||
|
||||
Run:
|
||||
.venv\\Scripts\\python.exe scripts\\build_county_koppen_metric.py
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Tuple
|
||||
|
||||
import geopandas as gpd
|
||||
import rasterio
|
||||
|
||||
from common.counties import load_counties
|
||||
from common.county_zonal_stats import DEFAULT_SUBCELLS, area_weighted_class_weights
|
||||
from common.koppen_legend import load_koppen_legend
|
||||
|
||||
PROJECT_ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_COUNTIES_GEOJSON = PROJECT_ROOT / "data" / "geojson-counties-fips.json"
|
||||
DEFAULT_KOPPEN_RASTER = PROJECT_ROOT / "data" / "koppen_geiger_tif" / "1991_2020" / "koppen_geiger_0p00833333.tif"
|
||||
DEFAULT_KOPPEN_LEGEND = PROJECT_ROOT / "data" / "koppen_geiger_tif" / "legend.txt"
|
||||
DEFAULT_OUT = PROJECT_ROOT / "data" / "metrics" / "koppen.csv"
|
||||
|
||||
PREDOMINANT_MIN_SHARE = 0.50
|
||||
PREDOMINANT_MIN_GAP = 0.05
|
||||
MIXED_CLASS = "Mixed"
|
||||
# Keeps shares that sit exactly on a cutoff from failing on floating-point error.
|
||||
RULE_TOLERANCE = 1e-9
|
||||
|
||||
FIELDS = [
|
||||
"countyFips",
|
||||
"countyName",
|
||||
"state",
|
||||
"koppenZone",
|
||||
"koppenTopClass",
|
||||
"koppenTopShare",
|
||||
"koppenSecondClass",
|
||||
"koppenSecondShare",
|
||||
]
|
||||
|
||||
|
||||
def rank_class_shares(weights: Dict[int, float], code_map: Dict[int, str]) -> List[Tuple[str, float]]:
|
||||
"""Convert per-code area weights to class shares, largest first.
|
||||
|
||||
Exact ties go to the smaller raster code, matching the previous build.
|
||||
"""
|
||||
total = sum(weights.values())
|
||||
if total <= 0:
|
||||
return []
|
||||
unknown = sorted(code for code in weights if code not in code_map)
|
||||
if unknown:
|
||||
raise ValueError(f"Raster codes {unknown} are not in the Koppen legend.")
|
||||
ranked = sorted(weights.items(), key=lambda item: (-item[1], item[0]))
|
||||
return [(code_map[code], weight / total) for code, weight in ranked]
|
||||
|
||||
|
||||
def classify(ranked: List[Tuple[str, float]]) -> str:
|
||||
"""Return the predominant class, "Mixed", or blank when there is no data."""
|
||||
if not ranked:
|
||||
return ""
|
||||
top_share = ranked[0][1]
|
||||
second_share = ranked[1][1] if len(ranked) > 1 else 0.0
|
||||
has_majority = top_share >= PREDOMINANT_MIN_SHARE - RULE_TOLERANCE
|
||||
has_clear_lead = top_share - second_share >= PREDOMINANT_MIN_GAP - RULE_TOLERANCE
|
||||
return ranked[0][0] if has_majority and has_clear_lead else MIXED_CLASS
|
||||
|
||||
|
||||
def _format_share(ranked: List[Tuple[str, float]], index: int) -> Tuple[str, str]:
|
||||
"""Return the class and 4-decimal share at a rank, or blanks."""
|
||||
if index >= len(ranked):
|
||||
return "", ""
|
||||
code, share = ranked[index]
|
||||
return code, f"{share:.4f}"
|
||||
|
||||
|
||||
def build_koppen_records(
|
||||
counties: gpd.GeoDataFrame,
|
||||
koppen_raster: Path,
|
||||
code_map: Dict[int, str],
|
||||
subcells: int = DEFAULT_SUBCELLS,
|
||||
) -> List[dict]:
|
||||
"""Classify every county and return its metric row."""
|
||||
records: List[dict] = []
|
||||
with rasterio.open(koppen_raster) as source:
|
||||
raster_counties = counties
|
||||
if source.crs is not None and counties.crs is not None and counties.crs != source.crs:
|
||||
raster_counties = counties.to_crs(source.crs)
|
||||
geographic = source.crs is None or source.crs.is_geographic
|
||||
|
||||
for county, geometry in zip(counties.itertuples(), raster_counties.geometry):
|
||||
weights = area_weighted_class_weights(source, geometry, geographic=geographic, subcells=subcells)
|
||||
ranked = rank_class_shares(weights, code_map)
|
||||
top_class, top_share = _format_share(ranked, 0)
|
||||
second_class, second_share = _format_share(ranked, 1)
|
||||
records.append(
|
||||
{
|
||||
"countyFips": county.county_fips,
|
||||
"countyName": county.county_name,
|
||||
"state": county.state,
|
||||
"koppenZone": classify(ranked),
|
||||
"koppenTopClass": top_class,
|
||||
"koppenTopShare": top_share,
|
||||
"koppenSecondClass": second_class,
|
||||
"koppenSecondShare": second_share,
|
||||
}
|
||||
)
|
||||
return records
|
||||
|
||||
|
||||
def write_records(records: List[dict], out_file: Path) -> None:
|
||||
"""Write the Koppen metric rows."""
|
||||
out_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
with out_file.open("w", encoding="utf-8", newline="") as csv_file:
|
||||
writer = csv.DictWriter(csv_file, fieldnames=FIELDS)
|
||||
writer.writeheader()
|
||||
writer.writerows(records)
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
"""Define and parse command-line options for this builder."""
|
||||
parser = argparse.ArgumentParser(description="Build the county Koppen-Geiger metric file.")
|
||||
parser.add_argument("--counties-geojson", type=Path, default=DEFAULT_COUNTIES_GEOJSON, help="County polygon GeoJSON path.")
|
||||
parser.add_argument("--koppen-raster", type=Path, default=DEFAULT_KOPPEN_RASTER, help="Koppen-Geiger raster TIFF path.")
|
||||
parser.add_argument("--koppen-legend", type=Path, default=DEFAULT_KOPPEN_LEGEND, help="legend.txt mapping raster codes.")
|
||||
parser.add_argument("--subcells", type=int, default=DEFAULT_SUBCELLS, help="Sub-cells per raster cell edge.")
|
||||
parser.add_argument("--out", type=Path, default=DEFAULT_OUT, help="Output metric CSV path.")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
counties = load_counties(args.counties_geojson)
|
||||
records = build_koppen_records(counties, args.koppen_raster, load_koppen_legend(args.koppen_legend), args.subcells)
|
||||
write_records(records, args.out)
|
||||
|
||||
outcomes = Counter(
|
||||
"blank" if not r["koppenZone"] else "Mixed" if r["koppenZone"] == MIXED_CLASS else "predominant" for r in records
|
||||
)
|
||||
print(
|
||||
f"Wrote {len(records)} counties to {args.out}: {outcomes['predominant']} predominant, "
|
||||
f"{outcomes['Mixed']} Mixed, {outcomes['blank']} blank."
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user