Complete Köppen-Geiger filter review with Mixed climate class
Classify each county by area-weighted Köppen class shares: a county is predominantly its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed (133 of 3,143 counties in the 50 states and DC). - Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and apply_koppen_metric_to_climate_data.py (writes koppenZone plus koppenPrimaryClass/koppenSecondaryClass for Mixed counties). - Move shared helpers into scripts/common/ (county loading, Köppen legend, area-weighted raster shares); fix the 180th-meridian raster window for Aleutians West. - Add check_climate_data.py to validate the app CSV. - Draw Mixed counties in app.js as diagonal stripes of their top two classes, fixed to the ground and following the map at every zoom, with a crossfade only when the stripe size changes. Filtering a class also matches Mixed counties where it is primary or secondary. - Document the rule, display, and pipeline plan in docs/ and update the README and data-source notes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,138 @@
|
||||
"""Apply the county Koppen-Geiger metric to the app CSV.
|
||||
|
||||
Replaces the koppenZone column of data/climate-data.csv with the values in
|
||||
data/metrics/koppen.csv and writes koppenPrimaryClass and koppenSecondaryClass:
|
||||
the two stripe classes the app draws for Mixed counties. Both are blank for
|
||||
predominant counties. The two columns are added after koppenZone if missing.
|
||||
Every other column, and the column order, is left unchanged. Use --dry-run to
|
||||
report the changes without writing.
|
||||
|
||||
Run:
|
||||
.venv\\Scripts\\python.exe scripts\\apply_koppen_metric_to_climate_data.py --dry-run
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
from collections import Counter
|
||||
from pathlib import Path
|
||||
from typing import Dict, List, Tuple
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
|
||||
DEFAULT_KOPPEN_METRIC = REPO_ROOT / "data" / "metrics" / "koppen.csv"
|
||||
|
||||
ZONE_FIELD = "koppenZone"
|
||||
PRIMARY_FIELD = "koppenPrimaryClass"
|
||||
SECONDARY_FIELD = "koppenSecondaryClass"
|
||||
STRIPE_FIELDS = [PRIMARY_FIELD, SECONDARY_FIELD]
|
||||
MIXED_CLASS = "Mixed"
|
||||
METRIC_FIELDS = ["countyFips", ZONE_FIELD, "koppenTopClass", "koppenSecondClass"]
|
||||
|
||||
# (countyFips, column, old value, new value)
|
||||
Change = Tuple[str, str, str, str]
|
||||
|
||||
|
||||
def read_csv_rows(path: Path) -> Tuple[List[str], List[dict]]:
|
||||
"""Read a CSV while preserving the source field order."""
|
||||
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
||||
reader = csv.DictReader(csv_file)
|
||||
if reader.fieldnames is None:
|
||||
raise ValueError(f"{path} has no CSV header.")
|
||||
return list(reader.fieldnames), list(reader)
|
||||
|
||||
|
||||
def fieldnames_with_stripe_columns(fields: List[str]) -> List[str]:
|
||||
"""Place the stripe-class columns directly after koppenZone."""
|
||||
base = [field for field in fields if field not in STRIPE_FIELDS]
|
||||
insert_at = base.index(ZONE_FIELD) + 1
|
||||
return base[:insert_at] + STRIPE_FIELDS + base[insert_at:]
|
||||
|
||||
|
||||
def load_koppen_values(koppen_metric: Path) -> Dict[str, Dict[str, str]]:
|
||||
"""Return koppenZone and the two stripe classes for each county in the metric file."""
|
||||
fields, rows = read_csv_rows(koppen_metric)
|
||||
missing_fields = [field for field in METRIC_FIELDS if field not in fields]
|
||||
if missing_fields:
|
||||
raise ValueError(f"{koppen_metric} is missing columns {missing_fields}.")
|
||||
|
||||
values: Dict[str, Dict[str, str]] = {}
|
||||
for row in rows:
|
||||
is_mixed = row[ZONE_FIELD] == MIXED_CLASS
|
||||
primary = row["koppenTopClass"] if is_mixed else ""
|
||||
secondary = row["koppenSecondClass"] if is_mixed else ""
|
||||
if is_mixed and not (primary and secondary):
|
||||
raise ValueError(f"{row['countyFips']} is Mixed but has no top or second class in {koppen_metric}.")
|
||||
values[row["countyFips"]] = {ZONE_FIELD: row[ZONE_FIELD], PRIMARY_FIELD: primary, SECONDARY_FIELD: secondary}
|
||||
return values
|
||||
|
||||
|
||||
def apply_koppen_metric(
|
||||
climate_data: Path, koppen_metric: Path, out: Path, dry_run: bool = False
|
||||
) -> Tuple[List[Change], List[str]]:
|
||||
"""Update the Koppen columns; return the value changes and any columns that were added."""
|
||||
fields, rows = read_csv_rows(climate_data)
|
||||
if ZONE_FIELD not in fields:
|
||||
raise ValueError(f"{climate_data} has no {ZONE_FIELD} column.")
|
||||
lookup = load_koppen_values(koppen_metric)
|
||||
|
||||
missing = [row["countyFips"] for row in rows if row["countyFips"] not in lookup]
|
||||
if missing:
|
||||
raise ValueError(f"{len(missing)} counties are missing from {koppen_metric}, e.g. {missing[:5]}")
|
||||
|
||||
added_columns = [field for field in STRIPE_FIELDS if field not in fields]
|
||||
changes: List[Change] = []
|
||||
for row in rows:
|
||||
new_values = lookup[row["countyFips"]]
|
||||
for field in [ZONE_FIELD, *STRIPE_FIELDS]:
|
||||
old_value = row.get(field) or ""
|
||||
if old_value != new_values[field]:
|
||||
changes.append((row["countyFips"], field, old_value, new_values[field]))
|
||||
row[field] = new_values[field]
|
||||
|
||||
if not dry_run:
|
||||
with out.open("w", encoding="utf-8", newline="") as csv_file:
|
||||
writer = csv.DictWriter(csv_file, fieldnames=fieldnames_with_stripe_columns(fields))
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
return changes, added_columns
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
"""Define and parse command-line options for this apply step."""
|
||||
parser = argparse.ArgumentParser(description="Apply the county Koppen metric to the app CSV.")
|
||||
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA, help="App climate CSV to update.")
|
||||
parser.add_argument("--koppen-metric", type=Path, default=DEFAULT_KOPPEN_METRIC, help="Koppen metric CSV.")
|
||||
parser.add_argument("--out", type=Path, default=None, help="Output path; defaults to updating --climate-data in place.")
|
||||
parser.add_argument("--dry-run", action="store_true", help="Report changes without writing.")
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
args = parse_args()
|
||||
changes, added_columns = apply_koppen_metric(
|
||||
args.climate_data, args.koppen_metric, args.out or args.climate_data, args.dry_run
|
||||
)
|
||||
verb = "would" if args.dry_run else "did"
|
||||
|
||||
if added_columns:
|
||||
print(f"Columns added after {ZONE_FIELD} ({verb} write): {', '.join(added_columns)}")
|
||||
zone_changes = [change for change in changes if change[1] == ZONE_FIELD]
|
||||
kinds = Counter(
|
||||
"to Mixed" if new == MIXED_CLASS else "to blank" if not new else "to another class"
|
||||
for _, _, _, new in zone_changes
|
||||
)
|
||||
stripe_values = sum(1 for _, field, _, new in changes if field in STRIPE_FIELDS and new)
|
||||
print(f"{len(zone_changes)} {ZONE_FIELD} values {'would change' if args.dry_run else 'changed'}: {dict(kinds)}")
|
||||
print(f"{stripe_values} stripe-class values {'would be set' if args.dry_run else 'set'}.")
|
||||
for fips, _, old, new in zone_changes[:10]:
|
||||
print(f" {fips}: {old or '(blank)'} -> {new or '(blank)'}")
|
||||
if len(zone_changes) > 10:
|
||||
print(f" ... and {len(zone_changes) - 10} more")
|
||||
if args.dry_run:
|
||||
print("Dry run: nothing was written.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user