Files
Climate-Mood-Analysis/scripts/build_county_koppen_metric.py
T
KnouandClaude Opus 5 e855d583e3 Document restructuring and the beginnings of Filter 2 changes
Split the pipeline documentation by purpose so each fact has one home:
- docs/pipeline-plan.md keeps the plan, checklist, tracker, and guardrails
- docs/decisions.md holds open decisions and the dated decision log
- docs/reviews/ holds findings and tasks: one file per filter, plus
  00-cross-filter.md for findings that span filters
- scripts/common/README.md holds the shared-helper rules (formerly Phase 2)
- filter-calculations.md now describes calculations only

Filed findings 12-22 from a consistency audit of the app, docs, and scripts.

Filter 1 (Köppen-Geiger): use "Köppen" with the umlaut in all prose, labels,
docstrings, help text, and checker messages (finding 21), and correct the
base build's "majority" docstring (finding 22).

Filter 2 (annual avg temperature): record the adopted definition in
filter-calculations.md §2: equally weighted 1991-2020 monthly normals, per
WMO-No. 1203 and NOAA's 2020 methodology; area-weighted county means; blank
unless all 12 months exist. Code changes for this filter are still pending.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-15 16:13:11 -04:00

156 lines
6.1 KiB
Python

"""Build the county Köppen-Geiger metric file from area-weighted class shares.
Writes data/metrics/koppen.csv. A county is predominantly its top class when
that class covers at least 50% of the county's land and leads the runner-up by
at least 5 percentage points; otherwise it is "Mixed". Counties with no valid
raster cells are left blank. See docs/filter-calculations.md, section 1.
Run:
.venv\\Scripts\\python.exe scripts\\build_county_koppen_metric.py
"""
from __future__ import annotations
import argparse
import csv
from collections import Counter
from pathlib import Path
from typing import Dict, List, Tuple
import geopandas as gpd
import rasterio
from common.counties import load_counties
from common.county_zonal_stats import DEFAULT_SUBCELLS, area_weighted_class_weights
from common.koppen_legend import load_koppen_legend
PROJECT_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_COUNTIES_GEOJSON = PROJECT_ROOT / "data" / "geojson-counties-fips.json"
DEFAULT_KOPPEN_RASTER = PROJECT_ROOT / "data" / "koppen_geiger_tif" / "1991_2020" / "koppen_geiger_0p00833333.tif"
DEFAULT_KOPPEN_LEGEND = PROJECT_ROOT / "data" / "koppen_geiger_tif" / "legend.txt"
DEFAULT_OUT = PROJECT_ROOT / "data" / "metrics" / "koppen.csv"
PREDOMINANT_MIN_SHARE = 0.50
PREDOMINANT_MIN_GAP = 0.05
MIXED_CLASS = "Mixed"
# Keeps shares that sit exactly on a cutoff from failing on floating-point error.
RULE_TOLERANCE = 1e-9
FIELDS = [
"countyFips",
"countyName",
"state",
"koppenZone",
"koppenTopClass",
"koppenTopShare",
"koppenSecondClass",
"koppenSecondShare",
]
def rank_class_shares(weights: Dict[int, float], code_map: Dict[int, str]) -> List[Tuple[str, float]]:
"""Convert per-code area weights to class shares, largest first.
Exact ties go to the smaller raster code, matching the previous build.
"""
total = sum(weights.values())
if total <= 0:
return []
unknown = sorted(code for code in weights if code not in code_map)
if unknown:
raise ValueError(f"Raster codes {unknown} are not in the Köppen legend.")
ranked = sorted(weights.items(), key=lambda item: (-item[1], item[0]))
return [(code_map[code], weight / total) for code, weight in ranked]
def classify(ranked: List[Tuple[str, float]]) -> str:
"""Return the predominant class, "Mixed", or blank when there is no data."""
if not ranked:
return ""
top_share = ranked[0][1]
second_share = ranked[1][1] if len(ranked) > 1 else 0.0
has_majority = top_share >= PREDOMINANT_MIN_SHARE - RULE_TOLERANCE
has_clear_lead = top_share - second_share >= PREDOMINANT_MIN_GAP - RULE_TOLERANCE
return ranked[0][0] if has_majority and has_clear_lead else MIXED_CLASS
def _format_share(ranked: List[Tuple[str, float]], index: int) -> Tuple[str, str]:
"""Return the class and 4-decimal share at a rank, or blanks."""
if index >= len(ranked):
return "", ""
code, share = ranked[index]
return code, f"{share:.4f}"
def build_koppen_records(
counties: gpd.GeoDataFrame,
koppen_raster: Path,
code_map: Dict[int, str],
subcells: int = DEFAULT_SUBCELLS,
) -> List[dict]:
"""Classify every county and return its metric row."""
records: List[dict] = []
with rasterio.open(koppen_raster) as source:
raster_counties = counties
if source.crs is not None and counties.crs is not None and counties.crs != source.crs:
raster_counties = counties.to_crs(source.crs)
geographic = source.crs is None or source.crs.is_geographic
for county, geometry in zip(counties.itertuples(), raster_counties.geometry):
weights = area_weighted_class_weights(source, geometry, geographic=geographic, subcells=subcells)
ranked = rank_class_shares(weights, code_map)
top_class, top_share = _format_share(ranked, 0)
second_class, second_share = _format_share(ranked, 1)
records.append(
{
"countyFips": county.county_fips,
"countyName": county.county_name,
"state": county.state,
"koppenZone": classify(ranked),
"koppenTopClass": top_class,
"koppenTopShare": top_share,
"koppenSecondClass": second_class,
"koppenSecondShare": second_share,
}
)
return records
def write_records(records: List[dict], out_file: Path) -> None:
"""Write the Köppen metric rows."""
out_file.parent.mkdir(parents=True, exist_ok=True)
with out_file.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=FIELDS)
writer.writeheader()
writer.writerows(records)
def parse_args() -> argparse.Namespace:
"""Define and parse command-line options for this builder."""
parser = argparse.ArgumentParser(description="Build the county Köppen-Geiger metric file.")
parser.add_argument("--counties-geojson", type=Path, default=DEFAULT_COUNTIES_GEOJSON, help="County polygon GeoJSON path.")
parser.add_argument("--koppen-raster", type=Path, default=DEFAULT_KOPPEN_RASTER, help="Köppen-Geiger raster TIFF path.")
parser.add_argument("--koppen-legend", type=Path, default=DEFAULT_KOPPEN_LEGEND, help="legend.txt mapping raster codes.")
parser.add_argument("--subcells", type=int, default=DEFAULT_SUBCELLS, help="Sub-cells per raster cell edge.")
parser.add_argument("--out", type=Path, default=DEFAULT_OUT, help="Output metric CSV path.")
return parser.parse_args()
def main() -> None:
args = parse_args()
counties = load_counties(args.counties_geojson)
records = build_koppen_records(counties, args.koppen_raster, load_koppen_legend(args.koppen_legend), args.subcells)
write_records(records, args.out)
outcomes = Counter(
"blank" if not r["koppenZone"] else "Mixed" if r["koppenZone"] == MIXED_CLASS else "predominant" for r in records
)
print(
f"Wrote {len(records)} counties to {args.out}: {outcomes['predominant']} predominant, "
f"{outcomes['Mixed']} Mixed, {outcomes['blank']} blank."
)
if __name__ == "__main__":
main()