Files
Climate-Mood-Analysis/scripts/apply_koppen_metric_to_climate_data.py
T
KnouandClaude Opus 5 4d2b3e3d44 Complete Köppen-Geiger filter review with Mixed climate class
Classify each county by area-weighted Köppen class shares: a county is
predominantly its top class when that class covers at least 50% of its
land and leads the runner-up by at least 5 percentage points; otherwise
it is Mixed (133 of 3,143 counties in the 50 states and DC).

- Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and
  apply_koppen_metric_to_climate_data.py (writes koppenZone plus
  koppenPrimaryClass/koppenSecondaryClass for Mixed counties).
- Move shared helpers into scripts/common/ (county loading, Köppen
  legend, area-weighted raster shares); fix the 180th-meridian raster
  window for Aleutians West.
- Add check_climate_data.py to validate the app CSV.
- Draw Mixed counties in app.js as diagonal stripes of their top two
  classes, fixed to the ground and following the map at every zoom, with
  a crossfade only when the stripe size changes. Filtering a class also
  matches Mixed counties where it is primary or secondary.
- Document the rule, display, and pipeline plan in docs/ and update the
  README and data-source notes.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-14 02:54:25 -04:00

139 lines
6.0 KiB
Python

"""Apply the county Koppen-Geiger metric to the app CSV.
Replaces the koppenZone column of data/climate-data.csv with the values in
data/metrics/koppen.csv and writes koppenPrimaryClass and koppenSecondaryClass:
the two stripe classes the app draws for Mixed counties. Both are blank for
predominant counties. The two columns are added after koppenZone if missing.
Every other column, and the column order, is left unchanged. Use --dry-run to
report the changes without writing.
Run:
.venv\\Scripts\\python.exe scripts\\apply_koppen_metric_to_climate_data.py --dry-run
"""
from __future__ import annotations
import argparse
import csv
from collections import Counter
from pathlib import Path
from typing import Dict, List, Tuple
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_KOPPEN_METRIC = REPO_ROOT / "data" / "metrics" / "koppen.csv"
ZONE_FIELD = "koppenZone"
PRIMARY_FIELD = "koppenPrimaryClass"
SECONDARY_FIELD = "koppenSecondaryClass"
STRIPE_FIELDS = [PRIMARY_FIELD, SECONDARY_FIELD]
MIXED_CLASS = "Mixed"
METRIC_FIELDS = ["countyFips", ZONE_FIELD, "koppenTopClass", "koppenSecondClass"]
# (countyFips, column, old value, new value)
Change = Tuple[str, str, str, str]
def read_csv_rows(path: Path) -> Tuple[List[str], List[dict]]:
"""Read a CSV while preserving the source field order."""
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
reader = csv.DictReader(csv_file)
if reader.fieldnames is None:
raise ValueError(f"{path} has no CSV header.")
return list(reader.fieldnames), list(reader)
def fieldnames_with_stripe_columns(fields: List[str]) -> List[str]:
"""Place the stripe-class columns directly after koppenZone."""
base = [field for field in fields if field not in STRIPE_FIELDS]
insert_at = base.index(ZONE_FIELD) + 1
return base[:insert_at] + STRIPE_FIELDS + base[insert_at:]
def load_koppen_values(koppen_metric: Path) -> Dict[str, Dict[str, str]]:
"""Return koppenZone and the two stripe classes for each county in the metric file."""
fields, rows = read_csv_rows(koppen_metric)
missing_fields = [field for field in METRIC_FIELDS if field not in fields]
if missing_fields:
raise ValueError(f"{koppen_metric} is missing columns {missing_fields}.")
values: Dict[str, Dict[str, str]] = {}
for row in rows:
is_mixed = row[ZONE_FIELD] == MIXED_CLASS
primary = row["koppenTopClass"] if is_mixed else ""
secondary = row["koppenSecondClass"] if is_mixed else ""
if is_mixed and not (primary and secondary):
raise ValueError(f"{row['countyFips']} is Mixed but has no top or second class in {koppen_metric}.")
values[row["countyFips"]] = {ZONE_FIELD: row[ZONE_FIELD], PRIMARY_FIELD: primary, SECONDARY_FIELD: secondary}
return values
def apply_koppen_metric(
climate_data: Path, koppen_metric: Path, out: Path, dry_run: bool = False
) -> Tuple[List[Change], List[str]]:
"""Update the Koppen columns; return the value changes and any columns that were added."""
fields, rows = read_csv_rows(climate_data)
if ZONE_FIELD not in fields:
raise ValueError(f"{climate_data} has no {ZONE_FIELD} column.")
lookup = load_koppen_values(koppen_metric)
missing = [row["countyFips"] for row in rows if row["countyFips"] not in lookup]
if missing:
raise ValueError(f"{len(missing)} counties are missing from {koppen_metric}, e.g. {missing[:5]}")
added_columns = [field for field in STRIPE_FIELDS if field not in fields]
changes: List[Change] = []
for row in rows:
new_values = lookup[row["countyFips"]]
for field in [ZONE_FIELD, *STRIPE_FIELDS]:
old_value = row.get(field) or ""
if old_value != new_values[field]:
changes.append((row["countyFips"], field, old_value, new_values[field]))
row[field] = new_values[field]
if not dry_run:
with out.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=fieldnames_with_stripe_columns(fields))
writer.writeheader()
writer.writerows(rows)
return changes, added_columns
def parse_args() -> argparse.Namespace:
"""Define and parse command-line options for this apply step."""
parser = argparse.ArgumentParser(description="Apply the county Koppen metric to the app CSV.")
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA, help="App climate CSV to update.")
parser.add_argument("--koppen-metric", type=Path, default=DEFAULT_KOPPEN_METRIC, help="Koppen metric CSV.")
parser.add_argument("--out", type=Path, default=None, help="Output path; defaults to updating --climate-data in place.")
parser.add_argument("--dry-run", action="store_true", help="Report changes without writing.")
return parser.parse_args()
def main() -> None:
args = parse_args()
changes, added_columns = apply_koppen_metric(
args.climate_data, args.koppen_metric, args.out or args.climate_data, args.dry_run
)
verb = "would" if args.dry_run else "did"
if added_columns:
print(f"Columns added after {ZONE_FIELD} ({verb} write): {', '.join(added_columns)}")
zone_changes = [change for change in changes if change[1] == ZONE_FIELD]
kinds = Counter(
"to Mixed" if new == MIXED_CLASS else "to blank" if not new else "to another class"
for _, _, _, new in zone_changes
)
stripe_values = sum(1 for _, field, _, new in changes if field in STRIPE_FIELDS and new)
print(f"{len(zone_changes)} {ZONE_FIELD} values {'would change' if args.dry_run else 'changed'}: {dict(kinds)}")
print(f"{stripe_values} stripe-class values {'would be set' if args.dry_run else 'set'}.")
for fips, _, old, new in zone_changes[:10]:
print(f" {fips}: {old or '(blank)'} -> {new or '(blank)'}")
if len(zone_changes) > 10:
print(f" ... and {len(zone_changes) - 10} more")
if args.dry_run:
print("Dry run: nothing was written.")
if __name__ == "__main__":
main()