Classify each county by area-weighted Köppen class shares: a county is predominantly its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed (133 of 3,143 counties in the 50 states and DC). - Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and apply_koppen_metric_to_climate_data.py (writes koppenZone plus koppenPrimaryClass/koppenSecondaryClass for Mixed counties). - Move shared helpers into scripts/common/ (county loading, Köppen legend, area-weighted raster shares); fix the 180th-meridian raster window for Aleutians West. - Add check_climate_data.py to validate the app CSV. - Draw Mixed counties in app.js as diagonal stripes of their top two classes, fixed to the ground and following the map at every zoom, with a crossfade only when the stripe size changes. Filtering a class also matches Mixed counties where it is primary or secondary. - Document the rule, display, and pipeline plan in docs/ and update the README and data-source notes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
139 lines
6.0 KiB
Python
139 lines
6.0 KiB
Python
"""Apply the county Koppen-Geiger metric to the app CSV.
|
|
|
|
Replaces the koppenZone column of data/climate-data.csv with the values in
|
|
data/metrics/koppen.csv and writes koppenPrimaryClass and koppenSecondaryClass:
|
|
the two stripe classes the app draws for Mixed counties. Both are blank for
|
|
predominant counties. The two columns are added after koppenZone if missing.
|
|
Every other column, and the column order, is left unchanged. Use --dry-run to
|
|
report the changes without writing.
|
|
|
|
Run:
|
|
.venv\\Scripts\\python.exe scripts\\apply_koppen_metric_to_climate_data.py --dry-run
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import csv
|
|
from collections import Counter
|
|
from pathlib import Path
|
|
from typing import Dict, List, Tuple
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
|
|
DEFAULT_KOPPEN_METRIC = REPO_ROOT / "data" / "metrics" / "koppen.csv"
|
|
|
|
ZONE_FIELD = "koppenZone"
|
|
PRIMARY_FIELD = "koppenPrimaryClass"
|
|
SECONDARY_FIELD = "koppenSecondaryClass"
|
|
STRIPE_FIELDS = [PRIMARY_FIELD, SECONDARY_FIELD]
|
|
MIXED_CLASS = "Mixed"
|
|
METRIC_FIELDS = ["countyFips", ZONE_FIELD, "koppenTopClass", "koppenSecondClass"]
|
|
|
|
# (countyFips, column, old value, new value)
|
|
Change = Tuple[str, str, str, str]
|
|
|
|
|
|
def read_csv_rows(path: Path) -> Tuple[List[str], List[dict]]:
|
|
"""Read a CSV while preserving the source field order."""
|
|
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
|
reader = csv.DictReader(csv_file)
|
|
if reader.fieldnames is None:
|
|
raise ValueError(f"{path} has no CSV header.")
|
|
return list(reader.fieldnames), list(reader)
|
|
|
|
|
|
def fieldnames_with_stripe_columns(fields: List[str]) -> List[str]:
|
|
"""Place the stripe-class columns directly after koppenZone."""
|
|
base = [field for field in fields if field not in STRIPE_FIELDS]
|
|
insert_at = base.index(ZONE_FIELD) + 1
|
|
return base[:insert_at] + STRIPE_FIELDS + base[insert_at:]
|
|
|
|
|
|
def load_koppen_values(koppen_metric: Path) -> Dict[str, Dict[str, str]]:
|
|
"""Return koppenZone and the two stripe classes for each county in the metric file."""
|
|
fields, rows = read_csv_rows(koppen_metric)
|
|
missing_fields = [field for field in METRIC_FIELDS if field not in fields]
|
|
if missing_fields:
|
|
raise ValueError(f"{koppen_metric} is missing columns {missing_fields}.")
|
|
|
|
values: Dict[str, Dict[str, str]] = {}
|
|
for row in rows:
|
|
is_mixed = row[ZONE_FIELD] == MIXED_CLASS
|
|
primary = row["koppenTopClass"] if is_mixed else ""
|
|
secondary = row["koppenSecondClass"] if is_mixed else ""
|
|
if is_mixed and not (primary and secondary):
|
|
raise ValueError(f"{row['countyFips']} is Mixed but has no top or second class in {koppen_metric}.")
|
|
values[row["countyFips"]] = {ZONE_FIELD: row[ZONE_FIELD], PRIMARY_FIELD: primary, SECONDARY_FIELD: secondary}
|
|
return values
|
|
|
|
|
|
def apply_koppen_metric(
|
|
climate_data: Path, koppen_metric: Path, out: Path, dry_run: bool = False
|
|
) -> Tuple[List[Change], List[str]]:
|
|
"""Update the Koppen columns; return the value changes and any columns that were added."""
|
|
fields, rows = read_csv_rows(climate_data)
|
|
if ZONE_FIELD not in fields:
|
|
raise ValueError(f"{climate_data} has no {ZONE_FIELD} column.")
|
|
lookup = load_koppen_values(koppen_metric)
|
|
|
|
missing = [row["countyFips"] for row in rows if row["countyFips"] not in lookup]
|
|
if missing:
|
|
raise ValueError(f"{len(missing)} counties are missing from {koppen_metric}, e.g. {missing[:5]}")
|
|
|
|
added_columns = [field for field in STRIPE_FIELDS if field not in fields]
|
|
changes: List[Change] = []
|
|
for row in rows:
|
|
new_values = lookup[row["countyFips"]]
|
|
for field in [ZONE_FIELD, *STRIPE_FIELDS]:
|
|
old_value = row.get(field) or ""
|
|
if old_value != new_values[field]:
|
|
changes.append((row["countyFips"], field, old_value, new_values[field]))
|
|
row[field] = new_values[field]
|
|
|
|
if not dry_run:
|
|
with out.open("w", encoding="utf-8", newline="") as csv_file:
|
|
writer = csv.DictWriter(csv_file, fieldnames=fieldnames_with_stripe_columns(fields))
|
|
writer.writeheader()
|
|
writer.writerows(rows)
|
|
return changes, added_columns
|
|
|
|
|
|
def parse_args() -> argparse.Namespace:
|
|
"""Define and parse command-line options for this apply step."""
|
|
parser = argparse.ArgumentParser(description="Apply the county Koppen metric to the app CSV.")
|
|
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA, help="App climate CSV to update.")
|
|
parser.add_argument("--koppen-metric", type=Path, default=DEFAULT_KOPPEN_METRIC, help="Koppen metric CSV.")
|
|
parser.add_argument("--out", type=Path, default=None, help="Output path; defaults to updating --climate-data in place.")
|
|
parser.add_argument("--dry-run", action="store_true", help="Report changes without writing.")
|
|
return parser.parse_args()
|
|
|
|
|
|
def main() -> None:
|
|
args = parse_args()
|
|
changes, added_columns = apply_koppen_metric(
|
|
args.climate_data, args.koppen_metric, args.out or args.climate_data, args.dry_run
|
|
)
|
|
verb = "would" if args.dry_run else "did"
|
|
|
|
if added_columns:
|
|
print(f"Columns added after {ZONE_FIELD} ({verb} write): {', '.join(added_columns)}")
|
|
zone_changes = [change for change in changes if change[1] == ZONE_FIELD]
|
|
kinds = Counter(
|
|
"to Mixed" if new == MIXED_CLASS else "to blank" if not new else "to another class"
|
|
for _, _, _, new in zone_changes
|
|
)
|
|
stripe_values = sum(1 for _, field, _, new in changes if field in STRIPE_FIELDS and new)
|
|
print(f"{len(zone_changes)} {ZONE_FIELD} values {'would change' if args.dry_run else 'changed'}: {dict(kinds)}")
|
|
print(f"{stripe_values} stripe-class values {'would be set' if args.dry_run else 'set'}.")
|
|
for fips, _, old, new in zone_changes[:10]:
|
|
print(f" {fips}: {old or '(blank)'} -> {new or '(blank)'}")
|
|
if len(zone_changes) > 10:
|
|
print(f" ... and {len(zone_changes) - 10} more")
|
|
if args.dry_run:
|
|
print("Dry run: nothing was written.")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|