Complete Köppen-Geiger filter review with Mixed climate class
Classify each county by area-weighted Köppen class shares: a county is predominantly its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed (133 of 3,143 counties in the 50 states and DC). - Add build_county_koppen_metric.py (writes data/metrics/koppen.csv) and apply_koppen_metric_to_climate_data.py (writes koppenZone plus koppenPrimaryClass/koppenSecondaryClass for Mixed counties). - Move shared helpers into scripts/common/ (county loading, Köppen legend, area-weighted raster shares); fix the 180th-meridian raster window for Aleutians West. - Add check_climate_data.py to validate the app CSV. - Draw Mixed counties in app.js as diagonal stripes of their top two classes, fixed to the ground and following the map at every zoom, with a crossfade only when the stripe size changes. Filtering a class also matches Mixed counties where it is primary or secondary. - Document the rule, display, and pipeline plan in docs/ and update the README and data-source notes. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
This commit is contained in:
@@ -0,0 +1,211 @@
|
||||
from __future__ import annotations
|
||||
|
||||
import csv
|
||||
import json
|
||||
import sys
|
||||
import unittest
|
||||
from pathlib import Path
|
||||
from tempfile import TemporaryDirectory
|
||||
|
||||
SCRIPTS_DIR = Path(__file__).resolve().parents[1] / "scripts"
|
||||
sys.path.insert(0, str(SCRIPTS_DIR))
|
||||
|
||||
from check_climate_data import ( # noqa: E402
|
||||
CHECK_BLANKS,
|
||||
CHECK_COLUMNS,
|
||||
CHECK_COUNTIES,
|
||||
CHECK_CROSS,
|
||||
CHECK_FORMAT,
|
||||
CHECK_GEOMETRY,
|
||||
CHECK_SOURCES,
|
||||
CHECK_VALUES,
|
||||
DEFAULT_CLIMATE_CSV,
|
||||
DEFAULT_COUNTIES_GEOJSON,
|
||||
DEFAULT_METRIC_SOURCES,
|
||||
EXPECTED_COLUMNS,
|
||||
run_checks,
|
||||
)
|
||||
|
||||
NOAA_GRID_COLUMNS = (
|
||||
"avgTempF",
|
||||
"avgDiurnalTempRangeF",
|
||||
"annualPrecipIn",
|
||||
"seasonalityIndex",
|
||||
"wettestPrecipMonth",
|
||||
"driestPrecipMonth",
|
||||
"absoluteExtremeDays",
|
||||
"avgSummerSpecificHumidityGKg",
|
||||
"humidHeatDays",
|
||||
"humidHeatSourceFips",
|
||||
)
|
||||
|
||||
|
||||
def make_row(fips: str, state: str, **overrides: str) -> dict[str, str]:
|
||||
row = {
|
||||
"countyFips": fips,
|
||||
"countyName": "Test",
|
||||
"state": state,
|
||||
"koppenZone": "Cfa",
|
||||
"koppenPrimaryClass": "",
|
||||
"koppenSecondaryClass": "",
|
||||
"avgTempF": "60.0",
|
||||
"avgDiurnalTempRangeF": "20.00",
|
||||
"annualPrecipIn": "40.0",
|
||||
"seasonalityIndex": "20",
|
||||
"wettestPrecipMonth": "May",
|
||||
"driestPrecipMonth": "October",
|
||||
"absoluteExtremeDays": "10.0",
|
||||
"meanDailyGlobalHorizontalRadiationKwhM2Day": "4.5",
|
||||
"clearSkyGhiReductionIndex": "0.25",
|
||||
"avgSummerSpecificHumidityGKg": "12.0",
|
||||
"humidHeatDays": "30.0",
|
||||
"humidHeatSourceFips": fips,
|
||||
"humidHeatFipsAdjustment": "",
|
||||
"source": "test",
|
||||
}
|
||||
row.update(overrides)
|
||||
return row
|
||||
|
||||
|
||||
def valid_rows() -> list[dict[str, str]]:
|
||||
alaska = make_row("02013", "AK", koppenZone="Dfc", **{column: "" for column in NOAA_GRID_COLUMNS})
|
||||
return [make_row("01001", "AL"), alaska]
|
||||
|
||||
|
||||
class CheckClimateDataTests(unittest.TestCase):
|
||||
def setUp(self) -> None:
|
||||
self._temp_dir = TemporaryDirectory()
|
||||
self.base = Path(self._temp_dir.name)
|
||||
self.csv_path = self.base / "climate-data.csv"
|
||||
self.geojson_path = self.base / "counties.json"
|
||||
self.sources_path = self.base / "metric_sources.json"
|
||||
self.write_geojson(["01001", "02013"])
|
||||
self.sources_path.write_text(json.dumps({"schemaVersion": 1, "metrics": {}}), encoding="utf-8")
|
||||
|
||||
def tearDown(self) -> None:
|
||||
self._temp_dir.cleanup()
|
||||
|
||||
def write_csv(self, rows: list[dict[str, str]], fieldnames: list[str] | None = None, encoding: str = "utf-8") -> None:
|
||||
with self.csv_path.open("w", encoding=encoding, newline="") as handle:
|
||||
writer = csv.DictWriter(handle, fieldnames=fieldnames or list(EXPECTED_COLUMNS))
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
|
||||
def write_geojson(self, fips_codes: list[str]) -> None:
|
||||
features = [
|
||||
{"type": "Feature", "properties": {"id": fips, "STATE": fips[:2]}, "geometry": None}
|
||||
for fips in fips_codes
|
||||
]
|
||||
self.geojson_path.write_text(json.dumps({"type": "FeatureCollection", "features": features}), encoding="utf-8")
|
||||
|
||||
def run_report(self):
|
||||
return run_checks(self.csv_path, self.geojson_path, self.sources_path)
|
||||
|
||||
def test_valid_file_passes_with_allowed_alaska_blanks(self) -> None:
|
||||
self.write_csv(valid_rows())
|
||||
|
||||
report = self.run_report()
|
||||
|
||||
self.assertTrue(report.ok, report.checks)
|
||||
|
||||
def test_mixed_koppen_county_with_stripe_classes_passes(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[0].update(koppenZone="Mixed", koppenPrimaryClass="Csb", koppenSecondaryClass="Dsb")
|
||||
self.write_csv(rows)
|
||||
|
||||
self.assertTrue(self.run_report().ok)
|
||||
|
||||
def test_mixed_county_without_stripe_classes_is_reported(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[0]["koppenZone"] = "Mixed"
|
||||
self.write_csv(rows)
|
||||
|
||||
problems = self.run_report().checks[CHECK_CROSS]
|
||||
|
||||
self.assertEqual(problems, ["01001: Mixed Koppen county needs koppenPrimaryClass and koppenSecondaryClass"])
|
||||
|
||||
def test_stripe_classes_on_predominant_county_are_reported(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[0].update(koppenPrimaryClass="Csb", koppenSecondaryClass="Dsb")
|
||||
self.write_csv(rows)
|
||||
|
||||
problems = self.run_report().checks[CHECK_CROSS]
|
||||
|
||||
self.assertEqual(problems, ["01001: Koppen stripe classes are set but koppenZone is Cfa, not Mixed"])
|
||||
|
||||
def test_invalid_or_identical_stripe_classes_are_reported(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[0].update(koppenZone="Mixed", koppenPrimaryClass="Csb", koppenSecondaryClass="Csb")
|
||||
rows[1].update(koppenZone="Mixed", koppenPrimaryClass="Xyz", koppenSecondaryClass="Dfc")
|
||||
self.write_csv(rows)
|
||||
|
||||
problems = self.run_report().checks[CHECK_CROSS]
|
||||
|
||||
self.assertEqual(
|
||||
problems,
|
||||
["01001: Koppen stripe classes are both Csb", "02013: invalid Koppen stripe classes 'Xyz'/'Dfc'"],
|
||||
)
|
||||
|
||||
def test_bad_values_and_categories_are_reported(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[0].update(avgTempF="120", koppenZone="cfa", wettestPrecipMonth="Febuary", seasonalityIndex="12.5")
|
||||
self.write_csv(rows)
|
||||
|
||||
problems = self.run_report().checks[CHECK_VALUES]
|
||||
|
||||
self.assertEqual(len(problems), 4, problems)
|
||||
self.assertTrue(any("avgTempF=120 outside" in problem for problem in problems))
|
||||
self.assertTrue(any("'cfa' is not an allowed category" in problem for problem in problems))
|
||||
|
||||
def test_blank_outside_allowed_states_is_reported(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[0]["avgTempF"] = ""
|
||||
self.write_csv(rows)
|
||||
|
||||
problems = self.run_report().checks[CHECK_BLANKS]
|
||||
|
||||
self.assertEqual(problems, ["01001 (AL): avgTempF is blank"])
|
||||
|
||||
def test_county_missing_from_map_is_reported(self) -> None:
|
||||
self.write_csv(valid_rows())
|
||||
self.write_geojson(["01001"])
|
||||
|
||||
problems = self.run_report().checks[CHECK_GEOMETRY]
|
||||
|
||||
self.assertEqual(problems, ["02013 is in the CSV but has no map polygon"])
|
||||
|
||||
def test_duplicate_fips_and_unexpected_column_are_reported(self) -> None:
|
||||
rows = valid_rows()
|
||||
rows[1] = make_row("01001", "AL", extraMetric="1")
|
||||
self.write_csv(rows, fieldnames=list(EXPECTED_COLUMNS) + ["extraMetric"])
|
||||
|
||||
report = self.run_report()
|
||||
|
||||
self.assertIn("duplicate countyFips 01001", report.checks[CHECK_COUNTIES])
|
||||
self.assertTrue(any("unexpected column 'extraMetric'" in problem for problem in report.checks[CHECK_COLUMNS]))
|
||||
|
||||
def test_byte_order_mark_is_reported(self) -> None:
|
||||
self.write_csv(valid_rows(), encoding="utf-8-sig")
|
||||
|
||||
report = self.run_report()
|
||||
|
||||
self.assertEqual(len(report.checks[CHECK_FORMAT]), 1)
|
||||
self.assertEqual(report.checks[CHECK_COLUMNS], [])
|
||||
|
||||
def test_unknown_metric_in_sources_file_is_reported(self) -> None:
|
||||
self.write_csv(valid_rows())
|
||||
self.sources_path.write_text(json.dumps({"metrics": {"notAMetric": {}}}), encoding="utf-8")
|
||||
|
||||
problems = self.run_report().checks[CHECK_SOURCES]
|
||||
|
||||
self.assertEqual(problems, ["metric_sources.json describes unknown metric 'notAMetric'"])
|
||||
|
||||
@unittest.skipUnless(DEFAULT_CLIMATE_CSV.exists(), "project climate CSV not present")
|
||||
def test_project_climate_csv_passes(self) -> None:
|
||||
report = run_checks(DEFAULT_CLIMATE_CSV, DEFAULT_COUNTIES_GEOJSON, DEFAULT_METRIC_SOURCES)
|
||||
|
||||
self.assertTrue(report.ok, report.checks)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
Reference in New Issue
Block a user