Split the pipeline documentation by purpose so each fact has one home: - docs/pipeline-plan.md keeps the plan, checklist, tracker, and guardrails - docs/decisions.md holds open decisions and the dated decision log - docs/reviews/ holds findings and tasks: one file per filter, plus 00-cross-filter.md for findings that span filters - scripts/common/README.md holds the shared-helper rules (formerly Phase 2) - filter-calculations.md now describes calculations only Filed findings 12-22 from a consistency audit of the app, docs, and scripts. Filter 1 (Köppen-Geiger): use "Köppen" with the umlaut in all prose, labels, docstrings, help text, and checker messages (finding 21), and correct the base build's "majority" docstring (finding 22). Filter 2 (annual avg temperature): record the adopted definition in filter-calculations.md §2: equally weighted 1991-2020 monthly normals, per WMO-No. 1203 and NOAA's 2020 methodology; area-weighted county means; blank unless all 12 months exist. Code changes for this filter are still pending. Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
212 lines
7.4 KiB
Python
212 lines
7.4 KiB
Python
from __future__ import annotations
|
|
|
|
import csv
|
|
import json
|
|
import sys
|
|
import unittest
|
|
from pathlib import Path
|
|
from tempfile import TemporaryDirectory
|
|
|
|
SCRIPTS_DIR = Path(__file__).resolve().parents[1] / "scripts"
|
|
sys.path.insert(0, str(SCRIPTS_DIR))
|
|
|
|
from check_climate_data import ( # noqa: E402
|
|
CHECK_BLANKS,
|
|
CHECK_COLUMNS,
|
|
CHECK_COUNTIES,
|
|
CHECK_CROSS,
|
|
CHECK_FORMAT,
|
|
CHECK_GEOMETRY,
|
|
CHECK_SOURCES,
|
|
CHECK_VALUES,
|
|
DEFAULT_CLIMATE_CSV,
|
|
DEFAULT_COUNTIES_GEOJSON,
|
|
DEFAULT_METRIC_SOURCES,
|
|
EXPECTED_COLUMNS,
|
|
run_checks,
|
|
)
|
|
|
|
NOAA_GRID_COLUMNS = (
|
|
"avgTempF",
|
|
"avgDiurnalTempRangeF",
|
|
"annualPrecipIn",
|
|
"seasonalityIndex",
|
|
"wettestPrecipMonth",
|
|
"driestPrecipMonth",
|
|
"absoluteExtremeDays",
|
|
"avgSummerSpecificHumidityGKg",
|
|
"humidHeatDays",
|
|
"humidHeatSourceFips",
|
|
)
|
|
|
|
|
|
def make_row(fips: str, state: str, **overrides: str) -> dict[str, str]:
|
|
row = {
|
|
"countyFips": fips,
|
|
"countyName": "Test",
|
|
"state": state,
|
|
"koppenZone": "Cfa",
|
|
"koppenPrimaryClass": "",
|
|
"koppenSecondaryClass": "",
|
|
"avgTempF": "60.0",
|
|
"avgDiurnalTempRangeF": "20.00",
|
|
"annualPrecipIn": "40.0",
|
|
"seasonalityIndex": "20",
|
|
"wettestPrecipMonth": "May",
|
|
"driestPrecipMonth": "October",
|
|
"absoluteExtremeDays": "10.0",
|
|
"meanDailyGlobalHorizontalRadiationKwhM2Day": "4.5",
|
|
"clearSkyGhiReductionIndex": "0.25",
|
|
"avgSummerSpecificHumidityGKg": "12.0",
|
|
"humidHeatDays": "30.0",
|
|
"humidHeatSourceFips": fips,
|
|
"humidHeatFipsAdjustment": "",
|
|
"source": "test",
|
|
}
|
|
row.update(overrides)
|
|
return row
|
|
|
|
|
|
def valid_rows() -> list[dict[str, str]]:
|
|
alaska = make_row("02013", "AK", koppenZone="Dfc", **{column: "" for column in NOAA_GRID_COLUMNS})
|
|
return [make_row("01001", "AL"), alaska]
|
|
|
|
|
|
class CheckClimateDataTests(unittest.TestCase):
|
|
def setUp(self) -> None:
|
|
self._temp_dir = TemporaryDirectory()
|
|
self.base = Path(self._temp_dir.name)
|
|
self.csv_path = self.base / "climate-data.csv"
|
|
self.geojson_path = self.base / "counties.json"
|
|
self.sources_path = self.base / "metric_sources.json"
|
|
self.write_geojson(["01001", "02013"])
|
|
self.sources_path.write_text(json.dumps({"schemaVersion": 1, "metrics": {}}), encoding="utf-8")
|
|
|
|
def tearDown(self) -> None:
|
|
self._temp_dir.cleanup()
|
|
|
|
def write_csv(self, rows: list[dict[str, str]], fieldnames: list[str] | None = None, encoding: str = "utf-8") -> None:
|
|
with self.csv_path.open("w", encoding=encoding, newline="") as handle:
|
|
writer = csv.DictWriter(handle, fieldnames=fieldnames or list(EXPECTED_COLUMNS))
|
|
writer.writeheader()
|
|
writer.writerows(rows)
|
|
|
|
def write_geojson(self, fips_codes: list[str]) -> None:
|
|
features = [
|
|
{"type": "Feature", "properties": {"id": fips, "STATE": fips[:2]}, "geometry": None}
|
|
for fips in fips_codes
|
|
]
|
|
self.geojson_path.write_text(json.dumps({"type": "FeatureCollection", "features": features}), encoding="utf-8")
|
|
|
|
def run_report(self):
|
|
return run_checks(self.csv_path, self.geojson_path, self.sources_path)
|
|
|
|
def test_valid_file_passes_with_allowed_alaska_blanks(self) -> None:
|
|
self.write_csv(valid_rows())
|
|
|
|
report = self.run_report()
|
|
|
|
self.assertTrue(report.ok, report.checks)
|
|
|
|
def test_mixed_koppen_county_with_stripe_classes_passes(self) -> None:
|
|
rows = valid_rows()
|
|
rows[0].update(koppenZone="Mixed", koppenPrimaryClass="Csb", koppenSecondaryClass="Dsb")
|
|
self.write_csv(rows)
|
|
|
|
self.assertTrue(self.run_report().ok)
|
|
|
|
def test_mixed_county_without_stripe_classes_is_reported(self) -> None:
|
|
rows = valid_rows()
|
|
rows[0]["koppenZone"] = "Mixed"
|
|
self.write_csv(rows)
|
|
|
|
problems = self.run_report().checks[CHECK_CROSS]
|
|
|
|
self.assertEqual(problems, ["01001: Mixed Köppen county needs koppenPrimaryClass and koppenSecondaryClass"])
|
|
|
|
def test_stripe_classes_on_predominant_county_are_reported(self) -> None:
|
|
rows = valid_rows()
|
|
rows[0].update(koppenPrimaryClass="Csb", koppenSecondaryClass="Dsb")
|
|
self.write_csv(rows)
|
|
|
|
problems = self.run_report().checks[CHECK_CROSS]
|
|
|
|
self.assertEqual(problems, ["01001: Köppen stripe classes are set but koppenZone is Cfa, not Mixed"])
|
|
|
|
def test_invalid_or_identical_stripe_classes_are_reported(self) -> None:
|
|
rows = valid_rows()
|
|
rows[0].update(koppenZone="Mixed", koppenPrimaryClass="Csb", koppenSecondaryClass="Csb")
|
|
rows[1].update(koppenZone="Mixed", koppenPrimaryClass="Xyz", koppenSecondaryClass="Dfc")
|
|
self.write_csv(rows)
|
|
|
|
problems = self.run_report().checks[CHECK_CROSS]
|
|
|
|
self.assertEqual(
|
|
problems,
|
|
["01001: Köppen stripe classes are both Csb", "02013: invalid Köppen stripe classes 'Xyz'/'Dfc'"],
|
|
)
|
|
|
|
def test_bad_values_and_categories_are_reported(self) -> None:
|
|
rows = valid_rows()
|
|
rows[0].update(avgTempF="120", koppenZone="cfa", wettestPrecipMonth="Febuary", seasonalityIndex="12.5")
|
|
self.write_csv(rows)
|
|
|
|
problems = self.run_report().checks[CHECK_VALUES]
|
|
|
|
self.assertEqual(len(problems), 4, problems)
|
|
self.assertTrue(any("avgTempF=120 outside" in problem for problem in problems))
|
|
self.assertTrue(any("'cfa' is not an allowed category" in problem for problem in problems))
|
|
|
|
def test_blank_outside_allowed_states_is_reported(self) -> None:
|
|
rows = valid_rows()
|
|
rows[0]["avgTempF"] = ""
|
|
self.write_csv(rows)
|
|
|
|
problems = self.run_report().checks[CHECK_BLANKS]
|
|
|
|
self.assertEqual(problems, ["01001 (AL): avgTempF is blank"])
|
|
|
|
def test_county_missing_from_map_is_reported(self) -> None:
|
|
self.write_csv(valid_rows())
|
|
self.write_geojson(["01001"])
|
|
|
|
problems = self.run_report().checks[CHECK_GEOMETRY]
|
|
|
|
self.assertEqual(problems, ["02013 is in the CSV but has no map polygon"])
|
|
|
|
def test_duplicate_fips_and_unexpected_column_are_reported(self) -> None:
|
|
rows = valid_rows()
|
|
rows[1] = make_row("01001", "AL", extraMetric="1")
|
|
self.write_csv(rows, fieldnames=list(EXPECTED_COLUMNS) + ["extraMetric"])
|
|
|
|
report = self.run_report()
|
|
|
|
self.assertIn("duplicate countyFips 01001", report.checks[CHECK_COUNTIES])
|
|
self.assertTrue(any("unexpected column 'extraMetric'" in problem for problem in report.checks[CHECK_COLUMNS]))
|
|
|
|
def test_byte_order_mark_is_reported(self) -> None:
|
|
self.write_csv(valid_rows(), encoding="utf-8-sig")
|
|
|
|
report = self.run_report()
|
|
|
|
self.assertEqual(len(report.checks[CHECK_FORMAT]), 1)
|
|
self.assertEqual(report.checks[CHECK_COLUMNS], [])
|
|
|
|
def test_unknown_metric_in_sources_file_is_reported(self) -> None:
|
|
self.write_csv(valid_rows())
|
|
self.sources_path.write_text(json.dumps({"metrics": {"notAMetric": {}}}), encoding="utf-8")
|
|
|
|
problems = self.run_report().checks[CHECK_SOURCES]
|
|
|
|
self.assertEqual(problems, ["metric_sources.json describes unknown metric 'notAMetric'"])
|
|
|
|
@unittest.skipUnless(DEFAULT_CLIMATE_CSV.exists(), "project climate CSV not present")
|
|
def test_project_climate_csv_passes(self) -> None:
|
|
report = run_checks(DEFAULT_CLIMATE_CSV, DEFAULT_COUNTIES_GEOJSON, DEFAULT_METRIC_SOURCES)
|
|
|
|
self.assertTrue(report.ok, report.checks)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|