Files
Climate-Mood-Analysis/scripts/apply_koppen_metric_to_climate_data.py
T
KnouandClaude Opus 5 e855d583e3 Document restructuring and the beginnings of Filter 2 changes
Split the pipeline documentation by purpose so each fact has one home:
- docs/pipeline-plan.md keeps the plan, checklist, tracker, and guardrails
- docs/decisions.md holds open decisions and the dated decision log
- docs/reviews/ holds findings and tasks: one file per filter, plus
  00-cross-filter.md for findings that span filters
- scripts/common/README.md holds the shared-helper rules (formerly Phase 2)
- filter-calculations.md now describes calculations only

Filed findings 12-22 from a consistency audit of the app, docs, and scripts.

Filter 1 (Köppen-Geiger): use "Köppen" with the umlaut in all prose, labels,
docstrings, help text, and checker messages (finding 21), and correct the
base build's "majority" docstring (finding 22).

Filter 2 (annual avg temperature): record the adopted definition in
filter-calculations.md §2: equally weighted 1991-2020 monthly normals, per
WMO-No. 1203 and NOAA's 2020 methodology; area-weighted county means; blank
unless all 12 months exist. Code changes for this filter are still pending.

Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
2026-09-15 16:13:11 -04:00

139 lines
6.1 KiB
Python

"""Apply the county Köppen-Geiger metric to the app CSV.
Replaces the koppenZone column of data/climate-data.csv with the values in
data/metrics/koppen.csv and writes koppenPrimaryClass and koppenSecondaryClass:
the two stripe classes the app draws for Mixed counties. Both are blank for
predominant counties. The two columns are added after koppenZone if missing.
Every other column, and the column order, is left unchanged. Use --dry-run to
report the changes without writing.
Run:
.venv\\Scripts\\python.exe scripts\\apply_koppen_metric_to_climate_data.py --dry-run
"""
from __future__ import annotations
import argparse
import csv
from collections import Counter
from pathlib import Path
from typing import Dict, List, Tuple
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_KOPPEN_METRIC = REPO_ROOT / "data" / "metrics" / "koppen.csv"
ZONE_FIELD = "koppenZone"
PRIMARY_FIELD = "koppenPrimaryClass"
SECONDARY_FIELD = "koppenSecondaryClass"
STRIPE_FIELDS = [PRIMARY_FIELD, SECONDARY_FIELD]
MIXED_CLASS = "Mixed"
METRIC_FIELDS = ["countyFips", ZONE_FIELD, "koppenTopClass", "koppenSecondClass"]
# (countyFips, column, old value, new value)
Change = Tuple[str, str, str, str]
def read_csv_rows(path: Path) -> Tuple[List[str], List[dict]]:
"""Read a CSV while preserving the source field order."""
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
reader = csv.DictReader(csv_file)
if reader.fieldnames is None:
raise ValueError(f"{path} has no CSV header.")
return list(reader.fieldnames), list(reader)
def fieldnames_with_stripe_columns(fields: List[str]) -> List[str]:
"""Place the stripe-class columns directly after koppenZone."""
base = [field for field in fields if field not in STRIPE_FIELDS]
insert_at = base.index(ZONE_FIELD) + 1
return base[:insert_at] + STRIPE_FIELDS + base[insert_at:]
def load_koppen_values(koppen_metric: Path) -> Dict[str, Dict[str, str]]:
"""Return koppenZone and the two stripe classes for each county in the metric file."""
fields, rows = read_csv_rows(koppen_metric)
missing_fields = [field for field in METRIC_FIELDS if field not in fields]
if missing_fields:
raise ValueError(f"{koppen_metric} is missing columns {missing_fields}.")
values: Dict[str, Dict[str, str]] = {}
for row in rows:
is_mixed = row[ZONE_FIELD] == MIXED_CLASS
primary = row["koppenTopClass"] if is_mixed else ""
secondary = row["koppenSecondClass"] if is_mixed else ""
if is_mixed and not (primary and secondary):
raise ValueError(f"{row['countyFips']} is Mixed but has no top or second class in {koppen_metric}.")
values[row["countyFips"]] = {ZONE_FIELD: row[ZONE_FIELD], PRIMARY_FIELD: primary, SECONDARY_FIELD: secondary}
return values
def apply_koppen_metric(
climate_data: Path, koppen_metric: Path, out: Path, dry_run: bool = False
) -> Tuple[List[Change], List[str]]:
"""Update the Köppen columns; return the value changes and any columns that were added."""
fields, rows = read_csv_rows(climate_data)
if ZONE_FIELD not in fields:
raise ValueError(f"{climate_data} has no {ZONE_FIELD} column.")
lookup = load_koppen_values(koppen_metric)
missing = [row["countyFips"] for row in rows if row["countyFips"] not in lookup]
if missing:
raise ValueError(f"{len(missing)} counties are missing from {koppen_metric}, e.g. {missing[:5]}")
added_columns = [field for field in STRIPE_FIELDS if field not in fields]
changes: List[Change] = []
for row in rows:
new_values = lookup[row["countyFips"]]
for field in [ZONE_FIELD, *STRIPE_FIELDS]:
old_value = row.get(field) or ""
if old_value != new_values[field]:
changes.append((row["countyFips"], field, old_value, new_values[field]))
row[field] = new_values[field]
if not dry_run:
with out.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=fieldnames_with_stripe_columns(fields))
writer.writeheader()
writer.writerows(rows)
return changes, added_columns
def parse_args() -> argparse.Namespace:
"""Define and parse command-line options for this apply step."""
parser = argparse.ArgumentParser(description="Apply the county Köppen metric to the app CSV.")
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA, help="App climate CSV to update.")
parser.add_argument("--koppen-metric", type=Path, default=DEFAULT_KOPPEN_METRIC, help="Köppen metric CSV.")
parser.add_argument("--out", type=Path, default=None, help="Output path; defaults to updating --climate-data in place.")
parser.add_argument("--dry-run", action="store_true", help="Report changes without writing.")
return parser.parse_args()
def main() -> None:
args = parse_args()
changes, added_columns = apply_koppen_metric(
args.climate_data, args.koppen_metric, args.out or args.climate_data, args.dry_run
)
verb = "would" if args.dry_run else "did"
if added_columns:
print(f"Columns added after {ZONE_FIELD} ({verb} write): {', '.join(added_columns)}")
zone_changes = [change for change in changes if change[1] == ZONE_FIELD]
kinds = Counter(
"to Mixed" if new == MIXED_CLASS else "to blank" if not new else "to another class"
for _, _, _, new in zone_changes
)
stripe_values = sum(1 for _, field, _, new in changes if field in STRIPE_FIELDS and new)
print(f"{len(zone_changes)} {ZONE_FIELD} values {'would change' if args.dry_run else 'changed'}: {dict(kinds)}")
print(f"{stripe_values} stripe-class values {'would be set' if args.dry_run else 'set'}.")
for fips, _, old, new in zone_changes[:10]:
print(f" {fips}: {old or '(blank)'} -> {new or '(blank)'}")
if len(zone_changes) > 10:
print(f" ... and {len(zone_changes) - 10} more")
if args.dry_run:
print("Dry run: nothing was written.")
if __name__ == "__main__":
main()