Complete Köppen-Geiger filter review with Mixed climate class
Classify each county by area-weighted Köppen class shares: a county is predominantly its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed (133 of 3,143 counties in the 50 states and DC). - Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and apply_koppen_metric_to_climate_data.py (writes koppenZone plus koppenPrimaryClass/koppenSecondaryClass for Mixed counties). - Move shared helpers into scripts/common/ (county loading, Köppen legend, area-weighted raster shares); fix the 180th-meridian raster window for Aleutians West. - Add check_climate_data.py to validate the app CSV. - Draw Mixed counties in app.js as diagonal stripes of their top two classes, fixed to the ground and following the map at every zoom, with a crossfade only when the stripe size changes. Filtering a class also matches Mixed counties where it is primary or secondary. - Document the rule, display, and pipeline plan in docs/ and update the README and data-source notes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,5 @@
|
||||
"""Shared helpers imported by the county data pipeline scripts.
|
||||
|
||||
Modules here are not run directly. Scripts in the parent folder import them,
|
||||
for example ``from common.counties import load_counties``.
|
||||
"""
|
||||
@@ -0,0 +1,131 @@
|
||||
"""Load county polygons and normalize county identifiers."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import geopandas as gpd
|
||||
|
||||
DEFAULT_COUNTIES_GEOJSON_URL = "https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json"
|
||||
|
||||
STATE_FIPS_TO_ABBR = {
|
||||
"01": "AL",
|
||||
"02": "AK",
|
||||
"04": "AZ",
|
||||
"05": "AR",
|
||||
"06": "CA",
|
||||
"08": "CO",
|
||||
"09": "CT",
|
||||
"10": "DE",
|
||||
"11": "DC",
|
||||
"12": "FL",
|
||||
"13": "GA",
|
||||
"15": "HI",
|
||||
"16": "ID",
|
||||
"17": "IL",
|
||||
"18": "IN",
|
||||
"19": "IA",
|
||||
"20": "KS",
|
||||
"21": "KY",
|
||||
"22": "LA",
|
||||
"23": "ME",
|
||||
"24": "MD",
|
||||
"25": "MA",
|
||||
"26": "MI",
|
||||
"27": "MN",
|
||||
"28": "MS",
|
||||
"29": "MO",
|
||||
"30": "MT",
|
||||
"31": "NE",
|
||||
"32": "NV",
|
||||
"33": "NH",
|
||||
"34": "NJ",
|
||||
"35": "NM",
|
||||
"36": "NY",
|
||||
"37": "NC",
|
||||
"38": "ND",
|
||||
"39": "OH",
|
||||
"40": "OK",
|
||||
"41": "OR",
|
||||
"42": "PA",
|
||||
"44": "RI",
|
||||
"45": "SC",
|
||||
"46": "SD",
|
||||
"47": "TN",
|
||||
"48": "TX",
|
||||
"49": "UT",
|
||||
"50": "VT",
|
||||
"51": "VA",
|
||||
"53": "WA",
|
||||
"54": "WV",
|
||||
"55": "WI",
|
||||
"56": "WY",
|
||||
"60": "AS",
|
||||
"66": "GU",
|
||||
"69": "MP",
|
||||
"72": "PR",
|
||||
"78": "VI",
|
||||
}
|
||||
|
||||
|
||||
def normalize_fips(value: object, width: int) -> str:
|
||||
"""Return a zero-padded FIPS code with the requested width."""
|
||||
text = str(value).strip()
|
||||
digits = "".join(ch for ch in text if ch.isdigit())
|
||||
if not digits:
|
||||
return ""
|
||||
return digits.zfill(width)[-width:]
|
||||
|
||||
|
||||
def load_counties(counties_geojson: Path) -> gpd.GeoDataFrame:
|
||||
"""Load county polygons and normalize fields used downstream."""
|
||||
if not counties_geojson.exists():
|
||||
try:
|
||||
print(
|
||||
f"County GeoJSON not found at {counties_geojson}. "
|
||||
f"Attempting download from {DEFAULT_COUNTIES_GEOJSON_URL}..."
|
||||
)
|
||||
gdf = gpd.read_file(DEFAULT_COUNTIES_GEOJSON_URL)
|
||||
counties_geojson.parent.mkdir(parents=True, exist_ok=True)
|
||||
# Cache the downloaded file for subsequent runs.
|
||||
gdf.to_file(counties_geojson, driver="GeoJSON")
|
||||
print(f"Downloaded and cached county GeoJSON to {counties_geojson}")
|
||||
except Exception as exc:
|
||||
raise FileNotFoundError(
|
||||
f"County GeoJSON not found at {counties_geojson}, and download from "
|
||||
f"{DEFAULT_COUNTIES_GEOJSON_URL} failed. Download the file manually "
|
||||
"and rerun with --counties-geojson pointing to it."
|
||||
) from exc
|
||||
|
||||
gdf = gpd.read_file(counties_geojson)
|
||||
if gdf.crs is None:
|
||||
gdf = gdf.set_crs("EPSG:4326")
|
||||
else:
|
||||
gdf = gdf.to_crs("EPSG:4326")
|
||||
|
||||
feature_id = None
|
||||
if "id" in gdf.columns:
|
||||
feature_id = gdf["id"]
|
||||
elif "GEOID" in gdf.columns:
|
||||
feature_id = gdf["GEOID"]
|
||||
elif "GEOID10" in gdf.columns:
|
||||
feature_id = gdf["GEOID10"]
|
||||
elif "fips" in gdf.columns:
|
||||
feature_id = gdf["fips"]
|
||||
else:
|
||||
raise ValueError("Unable to locate county FIPS identifier column in county polygons.")
|
||||
|
||||
gdf["county_fips"] = feature_id.map(lambda value: normalize_fips(value, 5))
|
||||
gdf = gdf[gdf["county_fips"] != ""].copy()
|
||||
|
||||
if "NAME" in gdf.columns:
|
||||
gdf["county_name"] = gdf["NAME"].fillna("").astype(str).str.strip()
|
||||
elif "name" in gdf.columns:
|
||||
gdf["county_name"] = gdf["name"].fillna("").astype(str).str.strip()
|
||||
else:
|
||||
gdf["county_name"] = gdf["county_fips"].map(lambda value: f"County {value}")
|
||||
|
||||
gdf["state_fips"] = gdf["county_fips"].str.slice(0, 2)
|
||||
gdf["state"] = gdf["state_fips"].map(lambda code: STATE_FIPS_TO_ABBR.get(code, f"S{code}"))
|
||||
gdf = gdf.sort_values("county_fips").reset_index(drop=True)
|
||||
return gdf
|
||||
@@ -0,0 +1,128 @@
|
||||
"""Shared county zonal statistics for raster-based metrics.
|
||||
|
||||
Area weighting estimates how much of each raster cell lies inside a county by
|
||||
rasterizing the county on a finer grid of sub-cells, then scales each cell by
|
||||
the cosine of its latitude so cells count by their true surface area. Counties
|
||||
that cross the 180th meridian are split so each side is read from its own small
|
||||
raster window.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import math
|
||||
from typing import Dict, List
|
||||
|
||||
import numpy as np
|
||||
import shapely
|
||||
from affine import Affine
|
||||
from rasterio.features import rasterize
|
||||
from rasterio.windows import Window, from_bounds
|
||||
from shapely.affinity import translate
|
||||
from shapely.geometry import box, mapping
|
||||
from shapely.geometry.base import BaseGeometry
|
||||
from shapely.ops import unary_union
|
||||
|
||||
DEFAULT_SUBCELLS = 16
|
||||
# Largest sub-cell grid rasterized for one county piece; bigger pieces use a coarser grid.
|
||||
SUBCELL_BUDGET = 80_000_000
|
||||
|
||||
|
||||
def split_at_antimeridian(geometry: BaseGeometry) -> List[BaseGeometry]:
|
||||
"""Split a lon/lat geometry into pieces that each stay on one side of 180 degrees.
|
||||
|
||||
A county such as Aleutians West, AK has islands at both +179 and -179
|
||||
degrees longitude. Its bounding box then spans nearly the whole globe, so
|
||||
each side is returned as its own piece.
|
||||
"""
|
||||
minx, _, maxx, _ = geometry.bounds
|
||||
if maxx - minx <= 180.0:
|
||||
return [geometry]
|
||||
|
||||
positive: List[BaseGeometry] = []
|
||||
negative: List[BaseGeometry] = []
|
||||
for part in getattr(geometry, "geoms", [geometry]):
|
||||
part_minx, _, part_maxx, _ = part.bounds
|
||||
if part_maxx - part_minx > 180.0:
|
||||
# One outline crosses the line: unwrap to 0..360, cut at 180, rewrap.
|
||||
unwrapped = shapely.transform(
|
||||
part,
|
||||
lambda xy: np.column_stack((np.where(xy[:, 0] < 0, xy[:, 0] + 360.0, xy[:, 0]), xy[:, 1])),
|
||||
)
|
||||
positive.append(unwrapped.intersection(box(0.0, -90.0, 180.0, 90.0)))
|
||||
negative.append(translate(unwrapped.intersection(box(180.0, -90.0, 360.0, 90.0)), xoff=-360.0))
|
||||
elif part_minx >= 0:
|
||||
positive.append(part)
|
||||
else:
|
||||
negative.append(part)
|
||||
|
||||
pieces = [unary_union(group) for group in (positive, negative) if group]
|
||||
return [piece for piece in pieces if not piece.is_empty]
|
||||
|
||||
|
||||
def geometry_window(source, geometry: BaseGeometry) -> Window:
|
||||
"""Return the raster window covering a geometry, padded by one cell on each side."""
|
||||
window = from_bounds(*geometry.bounds, transform=source.transform)
|
||||
col_start = math.floor(window.col_off) - 1
|
||||
row_start = math.floor(window.row_off) - 1
|
||||
col_stop = math.ceil(window.col_off + window.width) + 1
|
||||
row_stop = math.ceil(window.row_off + window.height) + 1
|
||||
padded = Window(col_start, row_start, col_stop - col_start, row_stop - row_start)
|
||||
return padded.intersection(Window(0, 0, source.width, source.height))
|
||||
|
||||
|
||||
def subcells_for(shape: tuple[int, int], requested: int) -> int:
|
||||
"""Return the finest sub-cell count, up to the request, that fits the budget."""
|
||||
rows, cols = shape
|
||||
subcells = requested
|
||||
while subcells > 1 and rows * cols * subcells * subcells > SUBCELL_BUDGET:
|
||||
subcells //= 2
|
||||
return subcells
|
||||
|
||||
|
||||
def cell_coverage_fractions(
|
||||
shape: tuple[int, int], transform: Affine, geometry: BaseGeometry, subcells: int
|
||||
) -> np.ndarray:
|
||||
"""Estimate the fraction of each raster cell covered by a geometry."""
|
||||
rows, cols = shape
|
||||
fine = rasterize(
|
||||
[mapping(geometry)],
|
||||
out_shape=(rows * subcells, cols * subcells),
|
||||
transform=transform * Affine.scale(1.0 / subcells),
|
||||
fill=0,
|
||||
default_value=1,
|
||||
dtype="uint8",
|
||||
)
|
||||
return fine.reshape(rows, subcells, cols, subcells).mean(axis=(1, 3))
|
||||
|
||||
|
||||
def area_weighted_class_weights(
|
||||
source,
|
||||
geometry: BaseGeometry,
|
||||
*,
|
||||
geographic: bool = True,
|
||||
subcells: int = DEFAULT_SUBCELLS,
|
||||
) -> Dict[int, float]:
|
||||
"""Return the area inside a geometry covered by each value of a categorical raster.
|
||||
|
||||
Weights are relative surface areas: the fraction of each cell inside the
|
||||
geometry, times cos(latitude) for geographic rasters. Cells equal to 0 or
|
||||
the raster's nodata value are excluded.
|
||||
"""
|
||||
weights: Dict[int, float] = {}
|
||||
pieces = split_at_antimeridian(geometry) if geographic else [geometry]
|
||||
for piece in pieces:
|
||||
window = geometry_window(source, piece)
|
||||
values = source.read(1, window=window, masked=True).filled(0)
|
||||
if source.nodata is not None:
|
||||
values = np.where(values == source.nodata, 0, values)
|
||||
transform = source.window_transform(window)
|
||||
cell_weights = cell_coverage_fractions(values.shape, transform, piece, subcells_for(values.shape, subcells))
|
||||
if geographic:
|
||||
row_lat = transform.f + (np.arange(values.shape[0]) + 0.5) * transform.e
|
||||
cell_weights = cell_weights * np.cos(np.radians(row_lat))[:, None]
|
||||
|
||||
counted = (values != 0) & (cell_weights > 0)
|
||||
for value in np.unique(values[counted]):
|
||||
code = int(value)
|
||||
weights[code] = weights.get(code, 0.0) + float(cell_weights[counted & (values == value)].sum())
|
||||
return weights
|
||||
@@ -0,0 +1,64 @@
|
||||
"""Koppen-Geiger raster codes and the Beck et al. legend loader."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from pathlib import Path
|
||||
from typing import Dict
|
||||
|
||||
# Beck et al legend key is expected as text file, but this default handles common codes.
|
||||
DEFAULT_KOPPEN_CODE_MAP = {
|
||||
1: "Af",
|
||||
2: "Am",
|
||||
3: "Aw",
|
||||
4: "BWh",
|
||||
5: "BWk",
|
||||
6: "BSh",
|
||||
7: "BSk",
|
||||
8: "Csa",
|
||||
9: "Csb",
|
||||
10: "Csc",
|
||||
11: "Cwa",
|
||||
12: "Cwb",
|
||||
13: "Cwc",
|
||||
14: "Cfa",
|
||||
15: "Cfb",
|
||||
16: "Cfc",
|
||||
17: "Dsa",
|
||||
18: "Dsb",
|
||||
19: "Dsc",
|
||||
20: "Dsd",
|
||||
21: "Dwa",
|
||||
22: "Dwb",
|
||||
23: "Dwc",
|
||||
24: "Dwd",
|
||||
25: "Dfa",
|
||||
26: "Dfb",
|
||||
27: "Dfc",
|
||||
28: "Dfd",
|
||||
29: "ET",
|
||||
30: "EF",
|
||||
}
|
||||
|
||||
|
||||
def load_koppen_legend(legend_path: Path | None) -> Dict[int, str]:
|
||||
"""Load Koppen raster codes, using defaults when no legend exists."""
|
||||
if legend_path is None:
|
||||
return DEFAULT_KOPPEN_CODE_MAP
|
||||
|
||||
mapping: Dict[int, str] = {}
|
||||
for line in legend_path.read_text(encoding="utf-8").splitlines():
|
||||
text = line.strip()
|
||||
if not text or text.startswith("#"):
|
||||
continue
|
||||
# Handles patterns like:
|
||||
# "1: Af ..." or "1 = Af" or "1 Af"
|
||||
match = re.match(r"^(\d+)\s*[:=]?\s*([A-Za-z]{2,3})\b", text)
|
||||
if not match:
|
||||
continue
|
||||
|
||||
key = int(match.group(1))
|
||||
value = match.group(2)
|
||||
mapping[key] = value
|
||||
|
||||
return mapping if mapping else DEFAULT_KOPPEN_CODE_MAP
|
||||
Reference in New Issue
Block a user