From e855d583e334d4cb612cec72d26a582e9234b57e Mon Sep 17 00:00:00 2001 From: Justin Fisher Date: Tue, 15 Sep 2026 16:13:11 -0400 Subject: [PATCH] Document restructuring and the beginnings of Filter 2 changes MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Split the pipeline documentation by purpose so each fact has one home: - docs/pipeline-plan.md keeps the plan, checklist, tracker, and guardrails - docs/decisions.md holds open decisions and the dated decision log - docs/reviews/ holds findings and tasks: one file per filter, plus 00-cross-filter.md for findings that span filters - scripts/common/README.md holds the shared-helper rules (formerly Phase 2) - filter-calculations.md now describes calculations only Filed findings 12-22 from a consistency audit of the app, docs, and scripts. Filter 1 (Köppen-Geiger): use "Köppen" with the umlaut in all prose, labels, docstrings, help text, and checker messages (finding 21), and correct the base build's "majority" docstring (finding 22). Filter 2 (annual avg temperature): record the adopted definition in filter-calculations.md §2: equally weighted 1991-2020 monthly normals, per WMO-No. 1203 and NOAA's 2020 methodology; area-weighted county means; blank unless all 12 months exist. Code changes for this filter are still pending. Co-Authored-By: Claude Opus 5 --- README.md | 8 +- app.js | 24 +-- docs/decisions.md | 45 +++++ docs/filter-calculations.md | 164 +++++++++--------- docs/koppen-mixed-display-plan.md | 3 +- docs/pipeline-plan.md | 152 ++++------------ docs/reviews/00-cross-filter.md | 121 +++++++++++++ docs/reviews/01-koppen.md | 78 +++++++++ docs/reviews/02-annual-avg-temperature.md | 81 +++++++++ docs/reviews/03-diurnal-temperature-range.md | 23 +++ docs/reviews/04-extreme-temperature-days.md | 40 +++++ docs/reviews/05-heat-index-days.md | 30 ++++ docs/reviews/06-annual-precipitation.md | 27 +++ docs/reviews/07-seasonality-index.md | 24 +++ docs/reviews/08-wettest-month.md | 24 +++ docs/reviews/09-driest-month.md | 25 +++ docs/reviews/10-summer-specific-humidity.md | 24 +++ docs/reviews/11-solar-ghi.md | 41 +++++ docs/reviews/12-clear-sky-ghi-reduction.md | 26 +++ .../apply_koppen_metric_to_climate_data.py | 8 +- scripts/build_county_climate_data.py | 6 +- scripts/build_county_koppen_metric.py | 10 +- scripts/check_climate_data.py | 10 +- scripts/common/README.md | 39 +++++ scripts/common/koppen_legend.py | 4 +- scripts/county_data_sources.md | 6 +- tests/test_check_climate_data.py | 6 +- 27 files changed, 807 insertions(+), 242 deletions(-) create mode 100644 docs/decisions.md create mode 100644 docs/reviews/00-cross-filter.md create mode 100644 docs/reviews/01-koppen.md create mode 100644 docs/reviews/02-annual-avg-temperature.md create mode 100644 docs/reviews/03-diurnal-temperature-range.md create mode 100644 docs/reviews/04-extreme-temperature-days.md create mode 100644 docs/reviews/05-heat-index-days.md create mode 100644 docs/reviews/06-annual-precipitation.md create mode 100644 docs/reviews/07-seasonality-index.md create mode 100644 docs/reviews/08-wettest-month.md create mode 100644 docs/reviews/09-driest-month.md create mode 100644 docs/reviews/10-summer-specific-humidity.md create mode 100644 docs/reviews/11-solar-ghi.md create mode 100644 docs/reviews/12-clear-sky-ghi-reduction.md create mode 100644 scripts/common/README.md diff --git a/README.md b/README.md index dc6c00a..8361d80 100644 --- a/README.md +++ b/README.md @@ -28,7 +28,7 @@ The browser currently loads 3,221 county-level records from | Group | Metrics | | --- | --- | -| Koppen-Geiger Classification | Predominant county climate class (covering at least 50% of the county's land and leading the runner-up by at least 5 points), or Mixed, drawn as stripes of the county's top two classes | +| Köppen-Geiger Classification | Predominant county climate class (covering at least 50% of the county's land and leading the runner-up by at least 5 points), or Mixed, drawn as stripes of the county's top two classes | | Temperature & Extremes | Annual average temperature, diurnal temperature range, annual extreme temperature days, and annual 90 F+ heat-index days | | Precipitation & Moisture | Annual precipitation, precipitation seasonality, wettest month, driest month, and summer specific humidity | | Solar Resource | Mean daily global horizontal radiation (GHI) and clear-sky GHI reduction index | @@ -97,7 +97,7 @@ will not work. | `scripts/` | Climate-data download, aggregation, and update tools | | `scripts/county_data_sources.md` | Detailed metric definitions, data provenance, and pipeline examples | | `scripts/requirements_county_etl.txt` | Python dependencies for the offline data pipeline | -| `docs/` | Filter calculations, the pipeline plan, and design notes | +| `docs/` | Filter calculations, the pipeline plan, decisions, per-filter reviews (`docs/reviews/`), and design notes | | `tests/` | Tests for the Köppen metric, the climate CSV check, NSRDB request/download wrappers, cloud-metric merging, and gridMET heat-index calculations | ## Climate Data Pipeline @@ -106,7 +106,7 @@ The checked-in browser assets are the final outputs needed to run the explorer. The scripts directory contains the larger offline workflow used to derive those outputs from sources including: -- Beck et al. Koppen-Geiger climate classification data +- Beck et al. Köppen-Geiger climate classification data - NOAA NCEI Climate Normals and nClimGrid data - gridMET humidity data - NREL National Solar Radiation Database data @@ -135,7 +135,7 @@ default. | Script | Current role | | --- | --- | -| `build_county_climate_data.py` | Builds the base app CSV from county geometry, Koppen-Geiger data, NOAA temperature/precipitation data, and optional solar inputs. Its older `extremeDays` output is replaced by the current absolute-threshold stage below, and its largest-share `koppenZone` by the Köppen stage. It writes only the base columns, so do not run it over the live CSV. | +| `build_county_climate_data.py` | Builds the base app CSV from county geometry, Köppen-Geiger data, NOAA temperature/precipitation data, and optional solar inputs. Its older `extremeDays` output is replaced by the current absolute-threshold stage below, and its largest-share `koppenZone` by the Köppen stage. It writes only the base columns, so do not run it over the live CSV. | | `build_county_koppen_metric.py` / `apply_koppen_metric_to_climate_data.py` | Builds area-weighted Köppen class shares per county in `data/metrics/koppen.csv`, classifies each county as predominant or Mixed, and writes `koppenZone` plus the two stripe-class columns for Mixed counties. | | `apply_precipitation_month_metrics_to_climate_data.py` | Recomputes and merges the 1991-2020 wettest- and driest-month categories from monthly nClimGrid precipitation. | | `build_county_locally_extreme_data.py` | Downloads or reads cached nClimGrid-Daily county Tmax/Tmin files, calculates county-percentile diagnostics, and calculates the app-facing absolute 95 F / 0 F day counts. | diff --git a/app.js b/app.js index ac6cbd3..cc46df5 100644 --- a/app.js +++ b/app.js @@ -106,7 +106,7 @@ const KOPPEN_CLASS_META = { Mixed: { label: "Mixed Climate", color: "#9aa3b2" } }; -// Counties with no predominant Koppen-Geiger class; see docs/koppen-mixed-display-plan.md. +// Counties with no predominant Köppen-Geiger class; see docs/koppen-mixed-display-plan.md. const KOPPEN_MIXED_CLASS = "Mixed"; // Standard style: share of each stripe pair given to the primary (top) class; the secondary class // gets the rest. @@ -147,17 +147,17 @@ const MONTH_CATEGORY_META = { const METRICS = { koppenZone: { - label: "Koppen-Geiger Climate Class", + label: "Köppen-Geiger Climate Class", type: "categorical", unit: "class", categoryMeta: KOPPEN_CLASS_META, - allOptionLabel: "All Koppen Classes", + allOptionLabel: "All Köppen Classes", source: { description: - "Beck et al. 2023 Koppen-Geiger maps (1991-2020, 1 km). County class shares are area-weighted. A county shows its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed Climate, striped in the colors of its top two classes.", + "Beck et al. 2023 Köppen-Geiger maps (1991-2020, 1 km). County class shares are area-weighted. A county shows its top class when that class covers at least 50% of its land and leads the runner-up by at least 5 percentage points; otherwise it is Mixed Climate, striped in the colors of its top two classes.", links: [ { - label: "Beck et al. 2023 Koppen-Geiger", + label: "Beck et al. 2023 Köppen-Geiger", url: "https://www.nature.com/articles/s41597-023-02549-6" } ] @@ -389,7 +389,7 @@ const COUNTY_DETAIL_METRIC_KEYS = [...DATA_METRIC_KEYS]; const METRIC_GROUPS = [ { key: "koppenGeigerClassification", - label: "Koppen-Geiger Classification", + label: "Köppen-Geiger Classification", metricKeys: ["koppenZone"] }, { @@ -733,7 +733,7 @@ function normalizeAntimeridianFeature(feature) { }); } -// Clean and validate a Koppen-Geiger climate class code. +// Clean and validate a Köppen-Geiger climate class code. function normalizeKoppenCode(code) { if (typeof code !== "string") { return null; @@ -923,7 +923,7 @@ function sanitizeOverrideRecord(rawRecord) { } }); - // Stripe classes for Mixed Koppen counties; blank for predominant counties. + // Stripe classes for Mixed Köppen counties; blank for predominant counties. sanitizedRecord.koppenPrimaryClass = normalizeKoppenCode(rawRecord.koppenPrimaryClass); sanitizedRecord.koppenSecondaryClass = normalizeKoppenCode(rawRecord.koppenSecondaryClass); @@ -1166,7 +1166,7 @@ function getKoppenMixedStripePairWidthPx() { return pixelsPer100Miles / KOPPEN_MIXED_LINE_PAIRS_PER_100_MILES; } -// Return a Mixed county's primary and secondary Koppen classes, or null when it is not striped. +// Return a Mixed county's primary and secondary Köppen classes, or null when it is not striped. function getKoppenStripeClasses(stats) { if (stats?.koppenZone !== KOPPEN_MIXED_CLASS) { return null; @@ -1421,7 +1421,7 @@ function createKoppenStripePatterns() { applyKoppenStripeStyle(getKoppenStripeStyle(map.getZoom())); } -// Pick a county's map fill: a stripe pattern for Mixed Koppen counties, otherwise the value color. +// Pick a county's map fill: a stripe pattern for Mixed Köppen counties, otherwise the value color. function fillForCounty(stats) { if (currentMetricKey === "koppenZone") { const classes = getKoppenStripeClasses(stats); @@ -1442,7 +1442,7 @@ function buildKoppenMixedLegendSwatchBackground() { ); } -// Format a county's metric value, naming the stripe classes for Mixed Koppen counties. +// Format a county's metric value, naming the stripe classes for Mixed Köppen counties. function formatCountyMetricValue(metricKey, stats) { const text = formatMetricValue(metricKey, stats[metricKey]); const classes = metricKey === "koppenZone" ? getKoppenStripeClasses(stats) : null; @@ -1459,7 +1459,7 @@ function shouldFeaturePassFilter(stats) { if (currentCategoricalFilter === "all" || categoryValue === currentCategoricalFilter) { return true; } - // A Mixed Koppen county also matches its primary and secondary classes. + // A Mixed Köppen county also matches its primary and secondary classes. const stripeClasses = currentMetricKey === "koppenZone" ? getKoppenStripeClasses(stats) : null; return Boolean(stripeClasses) && [stripeClasses.primary, stripeClasses.secondary].includes(currentCategoricalFilter); } diff --git a/docs/decisions.md b/docs/decisions.md new file mode 100644 index 0000000..01a223e --- /dev/null +++ b/docs/decisions.md @@ -0,0 +1,45 @@ +# Decisions + +Open decisions and a dated log of decided ones for the pipeline and the +filter-by-filter review. The findings that led to them are in +[reviews/](reviews/); the plan they serve is [pipeline-plan.md](pipeline-plan.md). + +## Open decisions + +| Decision | Options | Needed by | +| --- | --- | --- | +| Committing large intermediates | Commit metric files only, or also source summaries | Phase 3 | +| Alaska and Hawaii coverage | Leave blank with a stated coverage gap, or add a source that covers them (for example Daymet or TerraClimate). NOAA nClimGrid and gridMET cover the contiguous U.S. only, so 29 AK and 5 HI counties are blank in filters 2–10; see finding 12 in [reviews/00-cross-filter.md](reviews/00-cross-filter.md) | Before Phase 3 | + +## Decided + +- **Köppen audit columns (2026-09-12):** `koppen.csv` stores the top class and + share and the runner-up class and share alongside `koppenZone`. +- **Köppen no-data fallback (2026-09-12):** a county with no valid raster cells + is left blank, not assigned `Cfa`. +- **Applying Köppen to the app CSV (2026-09-12):** wait until the app supports + the Mixed class. Done 2026-09-13. +- **Metric files (2026-09-13):** one CSV per metric under `data/metrics/`, + starting with `koppen.csv`. +- **Mixed climate display (2026-09-13):** diagonal stripes of each Mixed + county's top two classes; see + [koppen-mixed-display-plan.md](koppen-mixed-display-plan.md). +- **Puerto Rico (2026-09-13):** off the map and out of every filter. The app + already drops state FIPS 72; the data files keep the rows. +- **Shared helpers (2026-09-13):** `scripts/common/` holds only code used by + more than one data source, one topic per module; code shared within one data + source stays with that source. Duplicates move during their own filter's + review. +- **Annual avg temperature month weighting (2026-09-15):** the annual value is + the equally weighted mean of the 12 monthly normals, not weighted by month + length. This follows WMO-No. 1203 (2017) §4.3.3(a), whose footnote + recommends against day-length weighting, and NOAA's 1991–2020 Normals + methodology (p. 2). Day weighting would have shifted values by only + +0.03 to +0.13 °F. Closes finding 1. +- **Annual avg temperature aggregation (2026-09-15):** area-weighted county + means (sub-cell coverage fraction × cos(latitude)), matching Köppen. Closes + finding 2 for this metric. +- **Annual avg temperature completeness (2026-09-15):** a county is blank + unless all 12 monthly values are present (WMO-No. 1203 §4.3.3). +- **Annual avg temperature in Alaska and Hawaii (2026-09-15):** left blank for + now; tracked as the cross-filter "Alaska and Hawaii coverage" open decision. diff --git a/docs/filter-calculations.md b/docs/filter-calculations.md index 964a021..d2aadb7 100644 --- a/docs/filter-calculations.md +++ b/docs/filter-calculations.md @@ -54,7 +54,7 @@ in `app.js`. | Group | UI label | Data key | Type | Display unit | | --- | --- | --- | --- | --- | -| Climate Classification | Köppen-Geiger Climate Class | `koppenZone` | Categorical | Class | +| Köppen-Geiger Classification | Köppen-Geiger Climate Class | `koppenZone` | Categorical | Class | | Temperature & Extremes | Annual Avg Temperature (Normals) | `avgTempF` | Numeric | °F | | Temperature & Extremes | Diurnal Temperature Range | `avgDiurnalTempRangeF` | Numeric | °F difference | | Temperature & Extremes | Annual Extreme Temperature Days | `absoluteExtremeDays` | Numeric | Days/year | @@ -194,8 +194,20 @@ the Köppen apply step must run after it. **Data key:** `avgTempF` -If the NOAA source is a historical monthly series, the script first forms a -1991--2020 climatology for each calendar month and cell: +The annual value is a 1991--2020 climatological standard normal: the mean of +the 12 monthly normals of NOAA nClimGrid-Monthly average temperature, averaged +over each county's area. The definition was adopted on 2026-09-15. It is not +yet applied to `data/climate-data.csv`, whose values still use the current +method below. + +**Source.** nClimGrid-Monthly is a 1/24° (about 5 km) grid of monthly values +interpolated from GHCN station data, covering the contiguous U.S. from 1895 to +the present. Its average temperature, `tavg`, is the mean of maximum and +minimum temperature, (Tmax + Tmin)/2, not a 24-hour mean. Alaska and Hawaii +are outside the grid, so their counties are blank. + +**Cell normals.** For each grid cell \(i\) and calendar month \(m\), the +normal is the mean over the 1991--2020 years with a valid value: $$ T_{i,m}^{\mathrm{norm}} @@ -204,17 +216,78 @@ T_{i,m}^{\mathrm{norm}} \sum_{y\in V_{i,m}}T_{i,m,y}. $$ -The monthly county value is an unweighted mean of valid touched raster cells: +This is NCEI's own method for its gridded normals, which it describes as "a +simple 30-year average of monthly grids" (Rennie and Palecki, +[*U.S. Monthly Gridded Precipitation and Temperature Climate Normals*](https://www.ncei.noaa.gov/sites/default/files/2022-04/Readme_Monthly_Gridded_Normals.pdf)). +Our copy of nClimGrid is a later version than the June 2021 data NCEI used, so +values can differ slightly from NCEI's published grids. The result is not +NCEI's station-based U.S. Climate Normals product. + +**County monthly values.** Each cell is weighted by the area of the cell that +lies inside the county, \(a_{c,i}\), estimated as for Köppen (§1). Cells +without data, such as ocean, are excluded: + +$$ +T_{c,m} += +\frac{\sum_{i}a_{c,i}\,T_{i,m}^{\mathrm{norm}}}{\sum_{i}a_{c,i}}. +$$ + +**Annual value.** Every month has equal weight. The value is defined only when +all 12 monthly values \(T_{c,m}\) exist; otherwise the county is blank: + +$$ +T_c(^\circ\mathrm{F}) += +\left( +\frac{1}{12}\sum_{m=1}^{12}T_{c,m} +\right)\frac{9}{5}+32. +$$ + +Values are kept at full precision until the stored value is rounded to +0.1 °F. + +**Rationale.** The method follows the WMO rules for annual normals, which +NOAA also applies to its 1991--2020 Normals: + +- *Equal month weights.* For a mean, WMO-No. 1203 §4.3.3(a) defines the annual + normal as "the mean of the monthly normals", and its footnote says weighting + months by their number of days "is not recommended for internationally + exchanged products" + ([WMO Guidelines on the Calculation of Climate Normals, 2017](https://library.wmo.int/records/item/55797-wmo-guidelines-on-the-calculation-of-climate-normals)). + NOAA's methodology states that "each month is treated equally in + calculating seasonal and annual averages; they are not weighted by the length + of month" + ([Normals Calculation Methodology 2020](https://www.ncei.noaa.gov/data/normals-annualseasonal/1991-2020/doc/Normals_Calculation_Methodology_2020.pdf), + p. 2). Day-length weighting would have raised values by only 0.03 to + 0.13 °F at five sample points. +- *From monthly normals.* WMO §4.3.3: annual normals "should be calculated + from the monthly normals, and not from the individual annual values." +- *Completeness.* WMO §4.3.3: "If the monthly normal for any of the constituent + months of the period of interest is missing, then the multimonth normal + should also be considered as missing." +- *Area weighting.* Equal weights for every touched cell give full weight to + cells that lie mostly outside the county, which matters most for small, + narrow, and coastal counties. Area weighting matches the Köppen shares (§1). + +Implementation: pending; see +[reviews/02-annual-avg-temperature.md](reviews/02-annual-avg-temperature.md). + +### Current method + +Until the new values are applied, the cell normals are computed as above, but +the two later steps differ. Each county month is the unweighted mean of every +valid cell the county polygon touches: $$ T_{c,m} = \frac{1}{|G_{c,m}|} -\sum_{i\in G_{c,m}}T_{i,m}^{\mathrm{norm}}. +\sum_{i\in G_{c,m}}T_{i,m}^{\mathrm{norm}}, $$ -The annual value is the equally weighted mean of the available monthly values, -converted from Celsius to Fahrenheit: +and the annual value averages whichever of the \(M_c\) monthly values are +available, normally 12: $$ T_c(^\circ\mathrm{F}) @@ -224,8 +297,6 @@ T_c(^\circ\mathrm{F}) \right)\frac{9}{5}+32. $$ -Normally \(M_c=12\). The stored value is rounded to 0.1 °F. - Implementation: `scripts/build_county_climate_data.py:124--166`, `scripts/build_county_climate_data.py:206--221`, and `scripts/build_county_climate_data.py:478--514`. @@ -548,73 +619,10 @@ Implementation: `scripts/summarize_nsrdb_county_polygon_cloud_archives.py:138--218` and `scripts/summarize_nsrdb_county_polygon_cloud_archives.py:222--304`. -## Calculation review findings +## Review findings -1. **Annual temperature weights months equally.** February has the same weight - as January or July. If the intended label means an average across all days, - monthly normals should instead be weighted by the number of days in each - month. - -2. **Base NOAA aggregation is not area-weighted.** Every touched raster cell - receives equal weight, including cells that intersect only a small portion - of a county. This can matter most for small or narrow counties and along - coastlines. *Resolved for Köppen on 2026-09-13: class shares are now - area-weighted (§1).* - -3. **Partial precipitation years are accepted.** One valid monthly precipitation - value is sufficient to produce `annualPrecipIn`; absent months silently lower - the annual sum. Requiring all 12 months, or recording completeness, would be - safer. - -4. **The Köppen fallback can create false data.** A county with no valid raster - cells is labeled `Cfa` instead of missing. A null value plus an audit flag - would distinguish missing coverage from a genuine humid-subtropical class. - *Resolved on 2026-09-13: the Köppen builder leaves such counties blank. The - fallback remains only in `build_county_climate_data.py`, whose Köppen - value is replaced by the apply step.* - -5. **Extreme-day counts are not completeness-normalized.** A partially observed - year contributes a raw count and receives the same weight as a complete year. - Consider requiring a minimum number of valid days or annualizing partial - counts explicitly. - -6. **The absolute-extreme metric depends on unrelated percentile thresholds.** - `build_annual_counts` skips a county when its retired local p95/p05 thresholds - are missing, even though the active 95 °F / 0 °F calculation does not require - those percentiles. The absolute calculation should be separated from that - prerequisite. - -7. **Heat Index days are a daily-extrema proxy.** Daily Tmax and daily minimum - relative humidity are paired even though their observation times may differ. - The result should not be described as an observed hourly maximum Heat Index. - -8. **Heat-year completeness is permissive.** Any year with at least one valid - Tmax/RH pair is included in the equal-year average. A minimum valid-day rule - would reduce low-biased partial-year counts. - -9. **The GHI formula assumes hourly, 365-day input.** It is correct for the - current 60-minute, `leap_day=false` requests. If the request interval changes, - the energy sum needs an interval-hours multiplier; leap-day handling would - also need to change the divisor. - -10. **Clear-sky reduction averages ratios rather than energy totals.** This is a - valid but specific definition. It gives each retained time row equal weight, - rather than weighting rows by available clear-sky energy. The label and - documentation should retain this distinction. - -11. **Spatial weighting is inconsistent across metric families.** Base NOAA - normals use equal touched-cell weights, Köppen uses area-weighted class - shares, gridMET humidity uses - \(\cos(\phi)\) weights on cell centers, and NSRDB polygon metrics use - estimated overlap areas. Cross-metric comparisons should account for these - different county aggregation methods. - -## Verification status - -This reference was derived from the checked-in calculation and merge scripts, -not solely from UI descriptions. No calculation code was changed. The automated -test suite was not executed during this review because `pytest` is not installed -in either the system Python environment or the project virtual environment. - -Section 1 and findings 2, 4, and 11 were updated on 2026-09-14, after the -Köppen classification was reworked and applied. +Review findings and their status are kept with the filter reviews in +[reviews/](reviews/): findings specific to one filter in that filter's file, +and findings that affect several filters in +[reviews/00-cross-filter.md](reviews/00-cross-filter.md). Findings keep their +original numbers. diff --git a/docs/koppen-mixed-display-plan.md b/docs/koppen-mixed-display-plan.md index 8e34b7c..571e098 100644 --- a/docs/koppen-mixed-display-plan.md +++ b/docs/koppen-mixed-display-plan.md @@ -9,7 +9,8 @@ and the project documentation was updated (section 6). **Scope:** Show Köppen-Geiger "Mixed" counties as striped on the map, filter them by their top two classes, and carry the two stripe classes from the pipeline into the app. The classification rule itself is described in -[filter-calculations.md](filter-calculations.md) §1; overall progress is tracked in +[filter-calculations.md](filter-calculations.md) §1; the filter's review is +[reviews/01-koppen.md](reviews/01-koppen.md), and overall progress is tracked in [pipeline-plan.md](pipeline-plan.md). ## 1. Design diff --git a/docs/pipeline-plan.md b/docs/pipeline-plan.md index 1fb1aca..c92f3f8 100644 --- a/docs/pipeline-plan.md +++ b/docs/pipeline-plan.md @@ -3,7 +3,10 @@ This plan describes how the county data pipeline will move from scripts that edit one shared CSV in place to per-metric outputs assembled into the app CSV. It is a working reference for the filter-by-filter review. Calculation details -for each filter live in [filter-calculations.md](filter-calculations.md). +for each filter live in [filter-calculations.md](filter-calculations.md); +findings and tasks live in [reviews/](reviews/), one file per filter plus +[00-cross-filter.md](reviews/00-cross-filter.md); open and past decisions live +in [decisions.md](decisions.md). **Started:** 2026-09-12 @@ -64,10 +67,8 @@ The README's enrichment list starts at step 2 and omits steps 1 and 3. `scripts/county_data_sources.md`, and in each script's assumptions. 3. **Column ownership is unclear.** GHI is finalized by the extreme-temperature apply script; wettest/driest month are computed in two places. -4. **County aggregation is inconsistent.** NOAA uses touched raster cells, - Köppen uses area-weighted shares (since 2026-09-13), gridMET uses cell - centers with cos(latitude) weights, and NSRDB uses overlap areas (finding 11 - in `filter-calculations.md`). +4. **County aggregation is inconsistent.** See finding 11 in + [reviews/00-cross-filter.md](reviews/00-cross-filter.md). 5. **No single entry point or final check.** A new user must piece together about 20 scripts, several large downloads, and an NSRDB API key. @@ -125,50 +126,10 @@ existing CSV keeps working until Phase 3. ### Phase 2 — Shared helpers in `scripts/common/` -`scripts/common/` holds code used by more than one data source (NOAA, gridMET, -NSRDB, Köppen). Scripts import from it, for example -`from common.counties import load_counties`; nothing in it is run directly. - -Rules for `common/`: - -- **Cross-source only.** A helper goes in only if metrics from more than one - data source use it. Code shared by scripts of a single data source stays with - that source, for example a future NSRDB module for the NSRDB prompt, - redaction, and error-log helpers. -- **One topic per module.** Each module is named for its topic and has a - docstring. No catch-all `utils.py`. -- **Keep it small.** Before adding a helper, ask why it does not belong to any - one data source. - -Current modules: - -| Module | Contents | Why it is in `common/` | -| --- | --- | --- | -| `county_zonal_stats.py` | Raster windows, the 180th-meridian split, area-weighted class shares | Used by any raster-based metric | -| `counties.py` | County polygon loading, FIPS normalization, the state FIPS table | County identity is shared by nearly every pipeline | -| `koppen_legend.py` | The Köppen code map and legend loader | Temporary: also used by `build_county_climate_data.py`; moves next to the Köppen code in Phase 3 | - -**Rationale.** Helper functions make each step of a computation explicit, -avoid repeated code, and can be tested separately -([Brown CSCI 0111, "Helper Functions"](https://cs.brown.edu/courses/csci0111/fall2018/lectures/helper-functions.html)). -Shared helper folders, however, tend to lose cohesion and collect unrelated -code; the recommended alternative is to keep code with the part of the system -it belongs to, allowing a shared folder only if it stays small and documented -([Helpers and Utils Folders in Software Architecture](https://dev.to/knzt/helpers-and-utils-folders-in-software-architecture-3f8h)). -The rules above follow both: shared functions, organized by topic and limited -to code that crosses data sources. - -Other shared code moves when its filter is reviewed, so each move is tested -alongside that filter. A 2026-09-13 survey found 19 functions with identical -copies in several scripts and 19 with copies that have drifted apart. Most -identical copies are NSRDB helpers, which belong in an NSRDB module rather -than `common/`; `read_csv_rows` (4 identical copies in `apply_*` scripts) is -cross-source. Drifted copies need a decision on which version is correct -before merging. Notable drifts: `summarize_county_gridmet_humidity.py` has its -own county loader and FIPS normalizer, and the state FIPS table is also copied -in `build_county_representative_points.py`, -`summarize_county_gridmet_humidity.py`, and -`request_nsrdb_county_polygon_archives.py`. +Rules, current modules, and rationale are in +[scripts/common/README.md](../scripts/common/README.md). Duplicated helpers +found by the 2026-09-13 survey are finding 14 in +[reviews/00-cross-filter.md](reviews/00-cross-filter.md). ### Phase 3 — Assemble and restructure (after all 12 filters are clean) @@ -195,8 +156,10 @@ in `build_county_representative_points.py`, For each filter: -1. Verify the calculation against the source data and document findings. -2. Decide any rule or method changes with the project owner. +1. Verify the calculation against the source data and document findings in the + filter's review file under `docs/reviews/`. +2. Decide any rule or method changes with the project owner, and record them in + `decisions.md`. 3. Record the adopted definition in `filter-calculations.md`. 4. Implement the calculation, writing `data/metrics/.csv`. 5. Add a single-column apply step for the current CSV. @@ -208,53 +171,24 @@ For each filter: ## 6. Filter tracker -Known issues come from `filter-calculations.md` ("Calculation review findings") -and this review; none beyond Köppen have been investigated yet. +Each filter's findings, decisions, and tasks are in its review file. Findings +that affect more than one filter are in +[00-cross-filter.md](reviews/00-cross-filter.md). -| # | Filter | Status | Known issues to review | +| # | Filter | Status | Review | | --- | --- | --- | --- | -| 1 | Köppen-Geiger class | Done (2026-09-14): rule applied, Mixed display built, documentation updated | See tasks below | -| 2 | Annual avg temperature | Not started | Months weighted equally (finding 1); touched-cell aggregation (finding 2) | -| 3 | Diurnal temperature range | Not started | Lexington, VA (51678) blank, while heat-index days use Rockbridge County as a proxy | -| 4 | Extreme temperature days | Not started | Partial years not normalized (finding 5); depends on retired percentile thresholds (finding 6); Lexington, VA blank | -| 5 | 90 °F+ heat-index days | Not started | Daily-extrema proxy (finding 7); permissive year completeness (finding 8) | -| 6 | Annual precipitation | Not started | Partial-year sums accepted (finding 3); touched-cell aggregation (finding 2) | -| 7 | Seasonality index | Not started | Touched-cell aggregation (finding 2) | -| 8 | Wettest month | Not started | Computed in both the base build and the precipitation-month script | -| 9 | Driest month | Not started | Same as wettest month | -| 10 | Summer specific humidity | Not started | Cell-center cos(latitude) aggregation differs from other metrics (finding 11) | -| 11 | Solar GHI | Not started | Finalized by the extreme-temperature apply script; hourly, 365-day assumption (finding 9) | -| 12 | Clear-sky GHI reduction | Not started | Mean of ratios rather than energy totals (finding 10) | - -### Köppen-Geiger tasks - -Adopted rule: a county is predominantly its top class if and only if that class -covers at least 50% of the county's land and leads the runner-up by at least -5 percentage points; otherwise it is Mixed climate. Expected result for the -50 states and DC: 3,010 predominant, 133 Mixed. - -- [x] Investigate low-majority counties and adopt the rule. -- [x] Windowed raster reads and 180th-meridian split. -- [x] Area-weighted class shares (16 × 16 sub-cells per raster cell, scaled by - cos(latitude)) in `scripts/common/county_zonal_stats.py`. -- [x] Apply the 50% / 5-point rule (`scripts/build_county_koppen_metric.py`; - counties with no valid cells are left blank). -- [x] Run the builder to write `data/metrics/koppen.csv` and confirm the - expected 3,010 predominant / 133 Mixed (2026-09-13). -- [x] `koppenZone`-only apply step - (`scripts/apply_koppen_metric_to_climate_data.py`, with `--dry-run`). -- [x] Apply to `data/climate-data.csv` (2026-09-13; 142 counties changed to - Mixed, with `koppenPrimaryClass` and `koppenSecondaryClass` added). -- [x] Allow `Mixed` in `check_climate_data.py`. -- [x] Add a Mixed climate category to `app.js`, drawn as stripes of the - county's top two classes; see - [koppen-mixed-display-plan.md](koppen-mixed-display-plan.md). -- [x] Replace the plurality description in `filter-calculations.md` §1 and mark - review findings 2 and 4 resolved for Köppen (2026-09-14). -- [x] Update the Köppen descriptions and script lists in `README.md` and - `scripts/county_data_sources.md` (2026-09-14). -- [x] Tests for shares, the rule, boundary cases, and the apply step - (`tests/test_koppen_metric.py`). +| 1 | Köppen-Geiger class | Done (2026-09-14): rule applied, Mixed display built, documentation updated | [01-koppen.md](reviews/01-koppen.md) | +| 2 | Annual avg temperature | In progress: method decided (2026-09-15) | [02-annual-avg-temperature.md](reviews/02-annual-avg-temperature.md) | +| 3 | Diurnal temperature range | Not started | [03-diurnal-temperature-range.md](reviews/03-diurnal-temperature-range.md) | +| 4 | Extreme temperature days | Not started | [04-extreme-temperature-days.md](reviews/04-extreme-temperature-days.md) | +| 5 | 90 °F+ heat-index days | Not started | [05-heat-index-days.md](reviews/05-heat-index-days.md) | +| 6 | Annual precipitation | Not started | [06-annual-precipitation.md](reviews/06-annual-precipitation.md) | +| 7 | Seasonality index | Not started | [07-seasonality-index.md](reviews/07-seasonality-index.md) | +| 8 | Wettest month | Not started | [08-wettest-month.md](reviews/08-wettest-month.md) | +| 9 | Driest month | Not started | [09-driest-month.md](reviews/09-driest-month.md) | +| 10 | Summer specific humidity | Not started | [10-summer-specific-humidity.md](reviews/10-summer-specific-humidity.md) | +| 11 | Solar GHI | Not started | [11-solar-ghi.md](reviews/11-solar-ghi.md) | +| 12 | Clear-sky GHI reduction | Not started | [12-clear-sky-ghi-reduction.md](reviews/12-clear-sky-ghi-reduction.md) | ## 7. Guardrails until Phase 3 @@ -263,29 +197,3 @@ covers at least 50% of the county's land and leads the runner-up by at least `extremeDays`, and overwrite the Mixed classification in `koppenZone`. - Run `check_climate_data.py` after every apply step. - Change one filter at a time, and compare its before and after values. - -## 8. Open decisions - -| Decision | Options | Needed by | -| --- | --- | --- | -| Committing large intermediates | Commit metric files only, or also source summaries | Phase 3 | - -### Decided - -- **Köppen audit columns (2026-09-12):** `koppen.csv` stores the top class and - share and the runner-up class and share alongside `koppenZone`. -- **Köppen no-data fallback (2026-09-12):** a county with no valid raster cells - is left blank, not assigned `Cfa`. -- **Applying Köppen to the app CSV (2026-09-12):** wait until the app supports - the Mixed class. Done 2026-09-13. -- **Metric files (2026-09-13):** one CSV per metric under `data/metrics/`, - starting with `koppen.csv`. -- **Mixed climate display (2026-09-13):** diagonal stripes of each Mixed - county's top two classes; see - [koppen-mixed-display-plan.md](koppen-mixed-display-plan.md). -- **Puerto Rico (2026-09-13):** off the map and out of every filter. The app - already drops state FIPS 72; the data files keep the rows. -- **Shared helpers (2026-09-13):** `scripts/common/` holds only code used by - more than one data source, one topic per module; code shared within one data - source stays with that source. Duplicates move during their own filter's - review. diff --git a/docs/reviews/00-cross-filter.md b/docs/reviews/00-cross-filter.md new file mode 100644 index 0000000..459656d --- /dev/null +++ b/docs/reviews/00-cross-filter.md @@ -0,0 +1,121 @@ +# Cross-filter review + +Findings that affect more than one filter. Each finding lives in exactly one +review file; the filter reviews it affects link here and record only how it +applied to them. Findings keep their numbers across all review files, and new +findings take the next number wherever they are filed. A finding that turns out +to affect other filters moves here, leaving a link behind. + +## Findings + +**2. Base NOAA aggregation is not area-weighted.** Every touched raster cell +receives equal weight, including cells that intersect only a small portion +of a county. This can matter most for small or narrow counties and along +coastlines. *Resolved for Köppen on 2026-09-13: class shares are now +area-weighted ([filter-calculations.md](../filter-calculations.md) §1).* + +Affects filters 1 (resolved), 2, 6, and 7. + +**11. Spatial weighting is inconsistent across metric families.** Base NOAA +normals use equal touched-cell weights, Köppen uses area-weighted class +shares, gridMET humidity uses +\(\cos(\phi)\) weights on cell centers, and NSRDB polygon metrics use +estimated overlap areas. Cross-metric comparisons should account for these +different county aggregation methods. + +Affects filters 1, 2, 6, 7, 10, 11, and 12. The planned fix is item 3 of the +target design in [pipeline-plan.md](../pipeline-plan.md): one shared +county-aggregation module used by every raster-based metric. + +**12. Alaska and Hawaii are not covered by the NOAA and gridMET sources.** +NOAA nClimGrid and gridMET cover the contiguous U.S. only; the nClimGrid grid +spans latitude 24.56 to 49.35 and longitude −124.69 to −67.02. All 29 Alaska +and 5 Hawaii counties are therefore blank in filters 2–10. Solar GHI and +clear-sky reduction (filters 11 and 12) cover both states. +`check_climate_data.py` allows these blanks (`OUTSIDE_CONUS`). The notes on +the base build in `scripts/county_data_sources.md` say "fallback values are +applied" for counties outside NOAA coverage, but `build_county_climate_data.py` +leaves them blank (lines 499–520). + +Affects filters 2–10. Open decision: "Alaska and Hawaii coverage" in +[decisions.md](../decisions.md). + +**13. Lexington, VA (51678) is missing from the NOAA daily county files.** +Diurnal temperature range and extreme temperature days are blank for it, and +`check_climate_data.py` allows those blanks (`NOAA_DAILY_MISSING_FIPS`). +Heat-index days instead use the surrounding Rockbridge County (51163) as a +proxy, recorded in `humidHeatSourceFips` and `humidHeatFipsAdjustment`. The +three filters handle the same gap differently. + +Affects filters 3, 4, and 5. + +**14. Helper functions are duplicated across scripts, and some copies have +drifted.** A 2026-09-13 survey found 19 functions with identical copies in +several scripts and 19 with copies that have drifted apart. Most identical +copies are NSRDB helpers, which belong in an NSRDB module rather than +`common/`; `read_csv_rows` (4 identical copies in `apply_*` scripts) is +cross-source. Drifted copies need a decision on which version is correct +before merging. Notable drifts: `summarize_county_gridmet_humidity.py` has its +own county loader and FIPS normalizer, and the state FIPS table is also copied +in `build_county_representative_points.py`, +`summarize_county_gridmet_humidity.py`, and +`request_nsrdb_county_polygon_archives.py`. + +Affects every script that holds a copy. The rules for moving shared code are +in [scripts/common/README.md](../../scripts/common/README.md). + +**15. The app's source text for three NOAA metrics names the wrong product.** +In `app.js`, `avgTempF`, `annualPrecipIn`, and `seasonalityIndex` credit +"NOAA NCEI 1991-2020 U.S. Climate Normals" and link the station-based Normals +page. Their values are computed from the nClimGrid-Monthly series averaged +over 1991–2020, which is NCEI's gridded-normals method rather than the station +product. Source 2 in `scripts/county_data_sources.md` likewise says the build +reads monthly normals files, while the build command passes +`data/noaa/nclimgrid/nclimgrid_tavg.nc` and `nclimgrid_prcp.nc`. Found during +the filter 2 review (2026-09-15). + +Affects filters 2, 6, and 7. Filter 2's task list covers the `avgTempF` text. + +**16. Source links point to retired or missing pages.** Checked 2026-09-15. +In `app.js`, the "NOAA nClimGrid Monthly" link used by wettest and driest month +(`https://www.ncei.noaa.gov/products/land-based-station/nclimgrid`) returns +404, and the "NREL National Solar Radiation Database" link used by solar GHI +and clear-sky GHI reduction (`https://nsrdb.nrel.gov/`) no longer resolves. +NREL's sites have moved to `nlr.gov`: `https://nsrdb.nlr.gov/` loads, and the +NSRDB scripts already call `developer.nlr.gov`. In Source 4 of +`scripts/county_data_sources.md`, `https://developer.nrel.gov/docs/solar/nsrdb/` +also fails (its `developer.nlr.gov` counterpart loads), and +`https://www.nrel.gov/gis/solar-resource-maps` fails with no counterpart at the +same path on `nlr.gov`. Every other link in `app.js`, `README.md`, and +`scripts/county_data_sources.md` loads. + +Affects filters 8, 9, 11, and 12. + +**17. The data-sources metric list is out of date.** The "Metric definitions +in generated output" list in `scripts/county_data_sources.md` describes +`extremeDays` / `oldExtremeDays` as an audit column kept in the app CSV, but +neither column is in `data/climate-data.csv`. The list has no entry for +`avgDiurnalTempRangeF` or `absoluteExtremeDays`, 2 of the 12 app metrics. + +Affects filters 3 and 4. + +## Decisions + +In [decisions.md](../decisions.md): Shared helpers (2026-09-13), and the open +decision on Alaska and Hawaii coverage. + +## Tasks + +Fixes for these findings are carried out in the affected filters' reviews. + +## Original review + +`filter-calculations.md` was derived from the checked-in calculation and merge +scripts, not solely from UI descriptions. No calculation code was changed. The +automated test suite was not executed during this review because `pytest` is +not installed in either the system Python environment or the project virtual +environment. The suite uses `unittest` and runs without pytest: +`python -m unittest discover -s tests`. + +Section 1 and findings 2, 4, and 11 were updated on 2026-09-14, after the +Köppen classification was reworked and applied. diff --git a/docs/reviews/01-koppen.md b/docs/reviews/01-koppen.md new file mode 100644 index 0000000..17e82ab --- /dev/null +++ b/docs/reviews/01-koppen.md @@ -0,0 +1,78 @@ +# Filter 1: Köppen-Geiger class + +**Status:** Done (2026-09-14): rule applied, Mixed display built, documentation +updated. + +**Data keys:** `koppenZone`, `koppenPrimaryClass`, `koppenSecondaryClass`. +Calculation: [filter-calculations.md](../filter-calculations.md) §1. Display +design and history: +[koppen-mixed-display-plan.md](../koppen-mixed-display-plan.md). + +## Findings + +**4. The Köppen fallback can create false data.** A county with no valid raster +cells is labeled `Cfa` instead of missing. A null value plus an audit flag +would distinguish missing coverage from a genuine humid-subtropical class. +*Resolved on 2026-09-13: the Köppen builder leaves such counties blank. The +fallback remains only in `build_county_climate_data.py`, whose Köppen +value is replaced by the apply step.* + +**21. The documented Köppen labels differ from the app.** `filter-calculations.md` +lists the group as "Climate Classification" and the filter as "Köppen-Geiger +Climate Class"; `app.js` shows "Koppen-Geiger Classification" and +"Koppen-Geiger Climate Class". The app spells Koppen without the umlaut +throughout, while the docs use Köppen. *Resolved on 2026-09-15: every +prose and display use of the name now reads "Köppen", in `app.js`, +`README.md`, `scripts/county_data_sources.md`, script docstrings, help text, +and checker messages (with the matching test expectations), and +`filter-calculations.md` quotes the app's group label, "Köppen-Geiger +Classification". Code identifiers, file names, and data paths such as +`koppenZone`, `normalizeKoppenCode`, and `koppen_geiger_tif/` keep the ASCII +spelling.* + +**22. The base build's docstring calls its Köppen value a majority class.** +`build_county_climate_data.py` lists `koppenZone` as the "majority +Koppen-Geiger class", but it assigns the most common class among touched +cells, which need not be a majority. That value is replaced by the Köppen apply +step, and the script is retired in Phase 3. *Resolved on 2026-09-15: the +docstring now reads "largest-share Köppen-Geiger class among touched cells +(replaced by the Köppen apply step)".* + +Cross-filter finding 2 (touched-cell aggregation) was resolved for Köppen on +2026-09-13; see [00-cross-filter.md](00-cross-filter.md). + +## Decisions + +In [decisions.md](../decisions.md): Köppen audit columns, Köppen no-data +fallback, and Applying Köppen to the app CSV (2026-09-12); Metric files and +Mixed climate display (2026-09-13). + +## Tasks + +Adopted rule: a county is predominantly its top class if and only if that class +covers at least 50% of the county's land and leads the runner-up by at least +5 percentage points; otherwise it is Mixed climate. Expected result for the +50 states and DC: 3,010 predominant, 133 Mixed. + +- [x] Investigate low-majority counties and adopt the rule. +- [x] Windowed raster reads and 180th-meridian split. +- [x] Area-weighted class shares (16 × 16 sub-cells per raster cell, scaled by + cos(latitude)) in `scripts/common/county_zonal_stats.py`. +- [x] Apply the 50% / 5-point rule (`scripts/build_county_koppen_metric.py`; + counties with no valid cells are left blank). +- [x] Run the builder to write `data/metrics/koppen.csv` and confirm the + expected 3,010 predominant / 133 Mixed (2026-09-13). +- [x] `koppenZone`-only apply step + (`scripts/apply_koppen_metric_to_climate_data.py`, with `--dry-run`). +- [x] Apply to `data/climate-data.csv` (2026-09-13; 142 counties changed to + Mixed, with `koppenPrimaryClass` and `koppenSecondaryClass` added). +- [x] Allow `Mixed` in `check_climate_data.py`. +- [x] Add a Mixed climate category to `app.js`, drawn as stripes of the + county's top two classes; see + [koppen-mixed-display-plan.md](../koppen-mixed-display-plan.md). +- [x] Replace the plurality description in `filter-calculations.md` §1 and mark + review findings 2 and 4 resolved for Köppen (2026-09-14). +- [x] Update the Köppen descriptions and script lists in `README.md` and + `scripts/county_data_sources.md` (2026-09-14). +- [x] Tests for shares, the rule, boundary cases, and the apply step + (`tests/test_koppen_metric.py`). diff --git a/docs/reviews/02-annual-avg-temperature.md b/docs/reviews/02-annual-avg-temperature.md new file mode 100644 index 0000000..4834b3c --- /dev/null +++ b/docs/reviews/02-annual-avg-temperature.md @@ -0,0 +1,81 @@ +# Filter 2: Annual avg temperature + +**Status:** In progress: method decided (2026-09-15). + +**Data key:** `avgTempF`. Calculation: +[filter-calculations.md](../filter-calculations.md) §2. + +## Findings + +**1. Annual temperature weights months equally.** February has the same weight +as January or July. If the intended label means an average across all days, +monthly normals should instead be weighted by the number of days in each +month. *Resolved on 2026-09-15: equal weighting is the WMO and NOAA standard +for annual normals and is kept; see "Annual avg temperature month weighting" +in [decisions.md](../decisions.md).* + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 2 and 11 (county aggregation), +12 (Alaska and Hawaii not covered), and 15 (source text cites the station +Normals). + +## Decisions + +In [decisions.md](../decisions.md), all 2026-09-15: Annual avg temperature +month weighting, aggregation, completeness, and in Alaska and Hawaii. + +## Tasks + +Adopted definition: the equally weighted mean of the 12 monthly 1991–2020 +normals of nClimGrid-Monthly `tavg`, area-weighted to each county; blank +unless all 12 months are present. The per-cell normals follow NCEI's own +gridded-normals method, "a simple 30-year average of monthly grids" +(Rennie and Palecki, *U.S. Monthly Gridded Precipitation and Temperature +Climate Normals*). Alaska and Hawaii stay blank; see +[decisions.md](../decisions.md). + +- [x] Verify the calculation against the source data and the WMO and NOAA + normals definitions (2026-09-15). +- [x] Decide month weighting, aggregation, completeness, and Alaska/Hawaii + handling (2026-09-15; see [decisions.md](../decisions.md)). +- [x] Rewrite `filter-calculations.md` §2 with the adopted definition, citing + WMO-No. 1203 §4.3.3 and NOAA's 1991–2020 Normals methodology. State that + average temperature is (Tmax + Tmin)/2 and that the grid covers the + contiguous U.S. only (2026-09-15; the current touched-cell method is kept + as "Current method" until the new values are applied). +- [ ] Add a continuous-value area-weighted mean to + `scripts/common/county_zonal_stats.py`. It takes an array and transform, + because nClimGrid is read from netCDF rather than a rasterio file, and + reuses `cell_coverage_fractions`, the cos(latitude) scaling, and the + padded window. NaN cells are excluded. +- [ ] Build `scripts/build_county_avg_temp_metric.py`, writing + `data/metrics/avg_temp.csv` with `avgTempF` and a valid-month count. The + 1991–2020 climatology step is NOAA-specific, so it stays with the NOAA + code rather than `common/`. +- [ ] Single-column apply step + `scripts/apply_avg_temp_metric_to_climate_data.py` with `--dry-run`. + Move `read_csv_rows` (4 identical copies in `apply_*` scripts) into + `common/` as part of this step. +- [ ] `check_climate_data.py`: keep the 20–85 °F rule and the Alaska/Hawaii + blank allowance until that open decision in + [decisions.md](../decisions.md) is made; confirm no new blanks in the + contiguous U.S. +- [ ] Tests (`tests/test_avg_temp_metric.py`): partial-cell weights, + cos(latitude), NaN exclusion, the 12-month rule, equal month weights, + Celsius-to-Fahrenheit conversion and rounding, and the apply step. +- [ ] Apply to `data/climate-data.csv`, run the checker, and compare: the + spread of changes, the largest shifts (expected in small, narrow, and + coastal counties), and no new blanks among the 3,109 contiguous-U.S. + counties. +- [ ] Mark finding 2 resolved for this metric in + [00-cross-filter.md](00-cross-filter.md). +- [ ] Correct the `avgTempF` source text in `app.js`, which cites the + station-based U.S. Climate Normals; describe it as 1991–2020 normals + computed from nClimGrid-Monthly and link nClimGrid. The "(Normals)" + label stays. +- [ ] Correct `scripts/county_data_sources.md`: Source 2 says the build reads + monthly normals files, but it reads the nClimGrid monthly series and + averages 1991–2020. Update the `avgTempF` definition line to match. +- [ ] Add the area-weighted `avgTempF` to the §7 guardrail on rerunning + `build_county_climate_data.py` in + [pipeline-plan.md](../pipeline-plan.md). diff --git a/docs/reviews/03-diurnal-temperature-range.md b/docs/reviews/03-diurnal-temperature-range.md new file mode 100644 index 0000000..b68f4df --- /dev/null +++ b/docs/reviews/03-diurnal-temperature-range.md @@ -0,0 +1,23 @@ +# Filter 3: Diurnal temperature range + +**Status:** Not started. + +**Data key:** `avgDiurnalTempRangeF`. Calculation: +[filter-calculations.md](../filter-calculations.md) §3. + +## Findings + +No filter-specific findings yet. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 12 (Alaska and Hawaii not covered), +13 (Lexington, VA blank), and 17 (missing from the data-sources metric list). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/04-extreme-temperature-days.md b/docs/reviews/04-extreme-temperature-days.md new file mode 100644 index 0000000..4145fd8 --- /dev/null +++ b/docs/reviews/04-extreme-temperature-days.md @@ -0,0 +1,40 @@ +# Filter 4: Extreme temperature days + +**Status:** Not started. + +**Data key:** `absoluteExtremeDays`. Calculation: +[filter-calculations.md](../filter-calculations.md) §4. + +## Findings + +**5. Extreme-day counts are not completeness-normalized.** A partially observed +year contributes a raw count and receives the same weight as a complete year. +Consider requiring a minimum number of valid days or annualizing partial +counts explicitly. + +**6. The absolute-extreme metric depends on unrelated percentile thresholds.** +`build_annual_counts` skips a county when its retired local p95/p05 thresholds +are missing, even though the active 95 °F / 0 °F calculation does not require +those percentiles. The absolute calculation should be separated from that +prerequisite. + +**18. The data-sources doc describes the retired locally extreme metric.** +Source 5 in `scripts/county_data_sources.md` is headed `locallyExtremeDays` +and says `apply_locally_extreme_metric_to_climate_data.py` writes the locally +extreme average into `extremeDays`, keeps `oldExtremeDays`, and adds eight +detail and audit columns. None of those columns is in `data/climate-data.csv`, +and the script's docstring says locally percentile-based metrics are not +included in the app CSV. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 12 (Alaska and Hawaii not covered), +13 (Lexington, VA blank), and 17 (missing from the data-sources metric list). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/05-heat-index-days.md b/docs/reviews/05-heat-index-days.md new file mode 100644 index 0000000..9d859da --- /dev/null +++ b/docs/reviews/05-heat-index-days.md @@ -0,0 +1,30 @@ +# Filter 5: 90 °F+ heat-index days + +**Status:** Not started. + +**Data keys:** `humidHeatDays`, with the audit columns `humidHeatSourceFips` +and `humidHeatFipsAdjustment`. Calculation: +[filter-calculations.md](../filter-calculations.md) §5. + +## Findings + +**7. Heat Index days are a daily-extrema proxy.** Daily Tmax and daily minimum +relative humidity are paired even though their observation times may differ. +The result should not be described as an observed hourly maximum Heat Index. + +**8. Heat-year completeness is permissive.** Any year with at least one valid +Tmax/RH pair is included in the equal-year average. A minimum valid-day rule +would reduce low-biased partial-year counts. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 12 (Alaska and Hawaii not covered) +and 13 (Lexington, VA uses Rockbridge County as a proxy). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/06-annual-precipitation.md b/docs/reviews/06-annual-precipitation.md new file mode 100644 index 0000000..372686a --- /dev/null +++ b/docs/reviews/06-annual-precipitation.md @@ -0,0 +1,27 @@ +# Filter 6: Annual precipitation + +**Status:** Not started. + +**Data key:** `annualPrecipIn`. Calculation: +[filter-calculations.md](../filter-calculations.md) §6. + +## Findings + +**3. Partial precipitation years are accepted.** One valid monthly precipitation +value is sufficient to produce `annualPrecipIn`; absent months silently lower +the annual sum. Requiring all 12 months, or recording completeness, would be +safer. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 2 and 11 (county aggregation), +12 (Alaska and Hawaii not covered), and 15 (source text cites the station +Normals). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/07-seasonality-index.md b/docs/reviews/07-seasonality-index.md new file mode 100644 index 0000000..ec3f128 --- /dev/null +++ b/docs/reviews/07-seasonality-index.md @@ -0,0 +1,24 @@ +# Filter 7: Seasonality index + +**Status:** Not started. + +**Data key:** `seasonalityIndex`. Calculation: +[filter-calculations.md](../filter-calculations.md) §7. + +## Findings + +No filter-specific findings yet. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 2 and 11 (county aggregation), +12 (Alaska and Hawaii not covered), and 15 (source text cites the station +Normals). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/08-wettest-month.md b/docs/reviews/08-wettest-month.md new file mode 100644 index 0000000..a44b774 --- /dev/null +++ b/docs/reviews/08-wettest-month.md @@ -0,0 +1,24 @@ +# Filter 8: Wettest month + +**Status:** Not started. + +**Data key:** `wettestPrecipMonth`. Calculation: +[filter-calculations.md](../filter-calculations.md) §8. + +## Findings + +No numbered findings yet. Known issue to review: computed in both the base +build and the precipitation-month script. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 12 (Alaska and Hawaii not covered) +and 16 (the app's nClimGrid source link returns 404). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/09-driest-month.md b/docs/reviews/09-driest-month.md new file mode 100644 index 0000000..b130576 --- /dev/null +++ b/docs/reviews/09-driest-month.md @@ -0,0 +1,25 @@ +# Filter 9: Driest month + +**Status:** Not started. + +**Data key:** `driestPrecipMonth`. Calculation: +[filter-calculations.md](../filter-calculations.md) §9. + +## Findings + +No numbered findings yet. Known issue to review: same as wettest month, which +is computed in both the base build and the precipitation-month script; see +[08-wettest-month.md](08-wettest-month.md). + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 12 (Alaska and Hawaii not covered) +and 16 (the app's nClimGrid source link returns 404). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/10-summer-specific-humidity.md b/docs/reviews/10-summer-specific-humidity.md new file mode 100644 index 0000000..b8ef657 --- /dev/null +++ b/docs/reviews/10-summer-specific-humidity.md @@ -0,0 +1,24 @@ +# Filter 10: Summer specific humidity + +**Status:** Not started. + +**Data key:** `avgSummerSpecificHumidityGKg`. Calculation: +[filter-calculations.md](../filter-calculations.md) §10. + +## Findings + +No filter-specific findings yet. Known issue to review: cell-center +cos(latitude) aggregation differs from other metrics (finding 11). + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 11 (inconsistent aggregation) and +12 (Alaska and Hawaii not covered). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/11-solar-ghi.md b/docs/reviews/11-solar-ghi.md new file mode 100644 index 0000000..493a753 --- /dev/null +++ b/docs/reviews/11-solar-ghi.md @@ -0,0 +1,41 @@ +# Filter 11: Solar GHI + +**Status:** Not started. + +**Data key:** `meanDailyGlobalHorizontalRadiationKwhM2Day`. Calculation: +[filter-calculations.md](../filter-calculations.md) §11. + +## Findings + +**9. The GHI formula assumes hourly, 365-day input.** It is correct for the +current 60-minute, `leap_day=false` requests. If the request interval changes, +the energy sum needs an interval-hours multiplier; leap-day handling would +also need to change the divisor. + +**19. The data-sources doc recommends a GHI raster the pipeline does not use.** +Source 4 in `scripts/county_data_sources.md` says county means should come +from a gridded annual GHI raster passed with `--solar-ghi-raster`. The app +values come from NSRDB polygon archive summaries, with representative points +as fallback, applied by `apply_locally_extreme_metric_to_climate_data.py`. + +**20. The base build labels any solar CSV as representative-point.** In +`build_county_climate_data.py`, the `--solar-ghi-csv` help text calls it a +representative-point fallback, and the per-row `source` tag is always +`solar-ghi-representative-point`, but the documented build command passes +`data/nrel/county_polygon_ghi_summary.csv`. The apply step later replaces both +the value and the tag, so only the base build's output is mislabeled. + +Known issue to review: finalized by the extreme-temperature apply script. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 11 (inconsistent aggregation) and +16 (the app's NSRDB source link no longer resolves). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/docs/reviews/12-clear-sky-ghi-reduction.md b/docs/reviews/12-clear-sky-ghi-reduction.md new file mode 100644 index 0000000..782aaab --- /dev/null +++ b/docs/reviews/12-clear-sky-ghi-reduction.md @@ -0,0 +1,26 @@ +# Filter 12: Clear-sky GHI reduction + +**Status:** Not started. + +**Data key:** `clearSkyGhiReductionIndex`. Calculation: +[filter-calculations.md](../filter-calculations.md) §12. + +## Findings + +**10. Clear-sky reduction averages ratios rather than energy totals.** This is +a valid but specific definition. It gives each retained time row equal weight, +rather than weighting rows by available clear-sky energy. The label and +documentation should retain this distinction. + +Cross-filter findings that affect this filter, in +[00-cross-filter.md](00-cross-filter.md): 11 (inconsistent aggregation) and +16 (the app's NSRDB source link no longer resolves). + +## Decisions + +None yet. + +## Tasks + +Not started; follow the per-filter review checklist in +[pipeline-plan.md](../pipeline-plan.md) §5. diff --git a/scripts/apply_koppen_metric_to_climate_data.py b/scripts/apply_koppen_metric_to_climate_data.py index 5865fda..b7c626a 100644 --- a/scripts/apply_koppen_metric_to_climate_data.py +++ b/scripts/apply_koppen_metric_to_climate_data.py @@ -1,4 +1,4 @@ -"""Apply the county Koppen-Geiger metric to the app CSV. +"""Apply the county Köppen-Geiger metric to the app CSV. Replaces the koppenZone column of data/climate-data.csv with the values in data/metrics/koppen.csv and writes koppenPrimaryClass and koppenSecondaryClass: @@ -71,7 +71,7 @@ def load_koppen_values(koppen_metric: Path) -> Dict[str, Dict[str, str]]: def apply_koppen_metric( climate_data: Path, koppen_metric: Path, out: Path, dry_run: bool = False ) -> Tuple[List[Change], List[str]]: - """Update the Koppen columns; return the value changes and any columns that were added.""" + """Update the Köppen columns; return the value changes and any columns that were added.""" fields, rows = read_csv_rows(climate_data) if ZONE_FIELD not in fields: raise ValueError(f"{climate_data} has no {ZONE_FIELD} column.") @@ -101,9 +101,9 @@ def apply_koppen_metric( def parse_args() -> argparse.Namespace: """Define and parse command-line options for this apply step.""" - parser = argparse.ArgumentParser(description="Apply the county Koppen metric to the app CSV.") + parser = argparse.ArgumentParser(description="Apply the county Köppen metric to the app CSV.") parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA, help="App climate CSV to update.") - parser.add_argument("--koppen-metric", type=Path, default=DEFAULT_KOPPEN_METRIC, help="Koppen metric CSV.") + parser.add_argument("--koppen-metric", type=Path, default=DEFAULT_KOPPEN_METRIC, help="Köppen metric CSV.") parser.add_argument("--out", type=Path, default=None, help="Output path; defaults to updating --climate-data in place.") parser.add_argument("--dry-run", action="store_true", help="Report changes without writing.") return parser.parse_args() diff --git a/scripts/build_county_climate_data.py b/scripts/build_county_climate_data.py index 567a958..c552d48 100644 --- a/scripts/build_county_climate_data.py +++ b/scripts/build_county_climate_data.py @@ -6,7 +6,7 @@ Outputs a CSV file compatible with the browser app: climate-data.csv Metrics produced per county: -- koppenZone: majority Koppen-Geiger class +- koppenZone: largest-share Köppen-Geiger class among touched cells (replaced by the Köppen apply step) - avgTempF: annual mean temperature from NOAA 1991-2020 gridded normals - annualPrecipIn: annual total precipitation from NOAA 1991-2020 gridded normals - seasonalityIndex: precipitation seasonality, coefficient of variation of monthly totals (%) @@ -256,7 +256,7 @@ def _touched_raster_values(source, geometry: BaseGeometry, split_antimeridian: b def _zonal_majority_class(koppen_raster: Path, counties: gpd.GeoDataFrame, code_map: Dict[int, str]) -> List[str]: - """Assign each county its most common Koppen-Geiger class.""" + """Assign each county its most common Köppen-Geiger class.""" classes: List[str] = [] with rasterio.open(koppen_raster) as source: raster_counties = counties @@ -597,7 +597,7 @@ def parse_args() -> argparse.Namespace: """Define and parse command-line options for this generator.""" parser = argparse.ArgumentParser(description="Generate county climate records for the web app.") parser.add_argument("--counties-geojson", type=Path, required=True, help="County polygon GeoJSON path.") - parser.add_argument("--koppen-raster", type=Path, required=True, help="Koppen-Geiger raster TIFF path.") + parser.add_argument("--koppen-raster", type=Path, required=True, help="Köppen-Geiger raster TIFF path.") parser.add_argument("--koppen-legend", type=Path, default=None, help="Optional legend.txt mapping integer codes.") parser.add_argument( "--monthly-tavg-nc", diff --git a/scripts/build_county_koppen_metric.py b/scripts/build_county_koppen_metric.py index ad3d9d4..5d1b5e5 100644 --- a/scripts/build_county_koppen_metric.py +++ b/scripts/build_county_koppen_metric.py @@ -1,4 +1,4 @@ -"""Build the county Koppen-Geiger metric file from area-weighted class shares. +"""Build the county Köppen-Geiger metric file from area-weighted class shares. Writes data/metrics/koppen.csv. A county is predominantly its top class when that class covers at least 50% of the county's land and leads the runner-up by @@ -58,7 +58,7 @@ def rank_class_shares(weights: Dict[int, float], code_map: Dict[int, str]) -> Li return [] unknown = sorted(code for code in weights if code not in code_map) if unknown: - raise ValueError(f"Raster codes {unknown} are not in the Koppen legend.") + raise ValueError(f"Raster codes {unknown} are not in the Köppen legend.") ranked = sorted(weights.items(), key=lambda item: (-item[1], item[0])) return [(code_map[code], weight / total) for code, weight in ranked] @@ -117,7 +117,7 @@ def build_koppen_records( def write_records(records: List[dict], out_file: Path) -> None: - """Write the Koppen metric rows.""" + """Write the Köppen metric rows.""" out_file.parent.mkdir(parents=True, exist_ok=True) with out_file.open("w", encoding="utf-8", newline="") as csv_file: writer = csv.DictWriter(csv_file, fieldnames=FIELDS) @@ -127,9 +127,9 @@ def write_records(records: List[dict], out_file: Path) -> None: def parse_args() -> argparse.Namespace: """Define and parse command-line options for this builder.""" - parser = argparse.ArgumentParser(description="Build the county Koppen-Geiger metric file.") + parser = argparse.ArgumentParser(description="Build the county Köppen-Geiger metric file.") parser.add_argument("--counties-geojson", type=Path, default=DEFAULT_COUNTIES_GEOJSON, help="County polygon GeoJSON path.") - parser.add_argument("--koppen-raster", type=Path, default=DEFAULT_KOPPEN_RASTER, help="Koppen-Geiger raster TIFF path.") + parser.add_argument("--koppen-raster", type=Path, default=DEFAULT_KOPPEN_RASTER, help="Köppen-Geiger raster TIFF path.") parser.add_argument("--koppen-legend", type=Path, default=DEFAULT_KOPPEN_LEGEND, help="legend.txt mapping raster codes.") parser.add_argument("--subcells", type=int, default=DEFAULT_SUBCELLS, help="Sub-cells per raster cell edge.") parser.add_argument("--out", type=Path, default=DEFAULT_OUT, help="Output metric CSV path.") diff --git a/scripts/check_climate_data.py b/scripts/check_climate_data.py index 3df488a..a08ada9 100644 --- a/scripts/check_climate_data.py +++ b/scripts/check_climate_data.py @@ -110,7 +110,7 @@ METRIC_RULES: Dict[str, MetricRule] = { } IDENTITY_COLUMNS = ("countyFips", "countyName", "state") -# Stripe classes the app draws for Mixed Koppen counties; blank otherwise. +# Stripe classes the app draws for Mixed Köppen counties; blank otherwise. KOPPEN_STRIPE_COLUMNS = ("koppenPrimaryClass", "koppenSecondaryClass") AUDIT_COLUMNS = ("humidHeatSourceFips", "humidHeatFipsAdjustment", "source") EXPECTED_COLUMNS = IDENTITY_COLUMNS + tuple(METRIC_RULES) + KOPPEN_STRIPE_COLUMNS + AUDIT_COLUMNS @@ -341,13 +341,13 @@ def check_cross_fields(report: Report, rows: List[Dict[str, str]]) -> None: secondary = row.get("koppenSecondaryClass") or "" if zone == MIXED_KOPPEN_CLASS: if not (primary and secondary): - report.add(CHECK_CROSS, f"{fips}: Mixed Koppen county needs koppenPrimaryClass and koppenSecondaryClass") + report.add(CHECK_CROSS, f"{fips}: Mixed Köppen county needs koppenPrimaryClass and koppenSecondaryClass") elif primary not in KOPPEN_CODES or secondary not in KOPPEN_CODES: - report.add(CHECK_CROSS, f"{fips}: invalid Koppen stripe classes {primary!r}/{secondary!r}") + report.add(CHECK_CROSS, f"{fips}: invalid Köppen stripe classes {primary!r}/{secondary!r}") elif primary == secondary: - report.add(CHECK_CROSS, f"{fips}: Koppen stripe classes are both {primary}") + report.add(CHECK_CROSS, f"{fips}: Köppen stripe classes are both {primary}") elif primary or secondary: - report.add(CHECK_CROSS, f"{fips}: Koppen stripe classes are set but koppenZone is {zone or 'blank'}, not Mixed") + report.add(CHECK_CROSS, f"{fips}: Köppen stripe classes are set but koppenZone is {zone or 'blank'}, not Mixed") if not (row.get("source") or "").strip(): report.add(CHECK_CROSS, f"{fips}: source is blank") diff --git a/scripts/common/README.md b/scripts/common/README.md new file mode 100644 index 0000000..98d6f35 --- /dev/null +++ b/scripts/common/README.md @@ -0,0 +1,39 @@ +# `scripts/common/` + +`scripts/common/` holds code used by more than one data source (NOAA, gridMET, +NSRDB, Köppen). Scripts import from it, for example +`from common.counties import load_counties`; nothing in it is run directly. + +Rules for `common/`: + +- **Cross-source only.** A helper goes in only if metrics from more than one + data source use it. Code shared by scripts of a single data source stays with + that source, for example a future NSRDB module for the NSRDB prompt, + redaction, and error-log helpers. +- **One topic per module.** Each module is named for its topic and has a + docstring. No catch-all `utils.py`. +- **Keep it small.** Before adding a helper, ask why it does not belong to any + one data source. + +Other shared code moves when its filter is reviewed, so each move is tested +alongside that filter. Duplicated and drifted helpers found by the 2026-09-13 +survey are finding 14 in +[docs/reviews/00-cross-filter.md](../../docs/reviews/00-cross-filter.md). + +Current modules: + +| Module | Contents | Why it is in `common/` | +| --- | --- | --- | +| `county_zonal_stats.py` | Raster windows, the 180th-meridian split, area-weighted class shares | Used by any raster-based metric | +| `counties.py` | County polygon loading, FIPS normalization, the state FIPS table | County identity is shared by nearly every pipeline | +| `koppen_legend.py` | The Köppen code map and legend loader | Temporary: also used by `build_county_climate_data.py`; moves next to the Köppen code in Phase 3 | + +**Rationale.** Helper functions make each step of a computation explicit, +avoid repeated code, and can be tested separately +([Brown CSCI 0111, "Helper Functions"](https://cs.brown.edu/courses/csci0111/fall2018/lectures/helper-functions.html)). +Shared helper folders, however, tend to lose cohesion and collect unrelated +code; the recommended alternative is to keep code with the part of the system +it belongs to, allowing a shared folder only if it stays small and documented +([Helpers and Utils Folders in Software Architecture](https://dev.to/knzt/helpers-and-utils-folders-in-software-architecture-3f8h)). +The rules above follow both: shared functions, organized by topic and limited +to code that crosses data sources. diff --git a/scripts/common/koppen_legend.py b/scripts/common/koppen_legend.py index 7a07668..33d3d0b 100644 --- a/scripts/common/koppen_legend.py +++ b/scripts/common/koppen_legend.py @@ -1,4 +1,4 @@ -"""Koppen-Geiger raster codes and the Beck et al. legend loader.""" +"""Köppen-Geiger raster codes and the Beck et al. legend loader.""" from __future__ import annotations @@ -42,7 +42,7 @@ DEFAULT_KOPPEN_CODE_MAP = { def load_koppen_legend(legend_path: Path | None) -> Dict[int, str]: - """Load Koppen raster codes, using defaults when no legend exists.""" + """Load Köppen raster codes, using defaults when no legend exists.""" if legend_path is None: return DEFAULT_KOPPEN_CODE_MAP diff --git a/scripts/county_data_sources.md b/scripts/county_data_sources.md index aaaf08e..0a363ac 100644 --- a/scripts/county_data_sources.md +++ b/scripts/county_data_sources.md @@ -12,9 +12,9 @@ The browser blocks `fetch("data/climate-data.csv")` when `index.html` is opened Then open [http://localhost:8000/](http://localhost:8000/). This keeps the app CSV-only while allowing the map and filters to load normally. -## Source 1: Koppen-Geiger classes (`koppenZone`, `koppenPrimaryClass`, `koppenSecondaryClass`) +## Source 1: Köppen-Geiger classes (`koppenZone`, `koppenPrimaryClass`, `koppenSecondaryClass`) -- Dataset: Beck et al. updated 1-km Koppen-Geiger climate classes (historical + future windows) +- Dataset: Beck et al. updated 1-km Köppen-Geiger climate classes (historical + future windows) - Landing page: [https://www.gloh2o.org/koppen/](https://www.gloh2o.org/koppen/) - Primary paper for updated release: [https://www.nature.com/articles/s41597-023-02549-6](https://www.nature.com/articles/s41597-023-02549-6) - Coverage: 1901-2099 (use historical 1991-2020 layer for this project to align with NOAA baselines) @@ -266,7 +266,7 @@ Then update the app CSV. Polygon archive GHI is used first; representative-point ## Metric definitions in generated output -- `koppenZone`: the county's predominant Koppen-Geiger class, meaning the class covering at least 50% of the county's land area and leading the runner-up by at least 5 percentage points; otherwise `Mixed`. Shares are area-weighted, with ocean and no-data cells excluded. +- `koppenZone`: the county's predominant Köppen-Geiger class, meaning the class covering at least 50% of the county's land area and leading the runner-up by at least 5 percentage points; otherwise `Mixed`. Shares are area-weighted, with ocean and no-data cells excluded. - `koppenPrimaryClass` / `koppenSecondaryClass`: for Mixed counties only, the top and runner-up classes, drawn as stripes on the map; blank for predominant counties. - `avgTempF`: mean of 12 monthly county mean temperatures, converted C -> F. - `annualPrecipIn`: sum of 12 monthly county mean precipitation totals, converted mm -> inches. diff --git a/tests/test_check_climate_data.py b/tests/test_check_climate_data.py index b22ea3c..77c0976 100644 --- a/tests/test_check_climate_data.py +++ b/tests/test_check_climate_data.py @@ -122,7 +122,7 @@ class CheckClimateDataTests(unittest.TestCase): problems = self.run_report().checks[CHECK_CROSS] - self.assertEqual(problems, ["01001: Mixed Koppen county needs koppenPrimaryClass and koppenSecondaryClass"]) + self.assertEqual(problems, ["01001: Mixed Köppen county needs koppenPrimaryClass and koppenSecondaryClass"]) def test_stripe_classes_on_predominant_county_are_reported(self) -> None: rows = valid_rows() @@ -131,7 +131,7 @@ class CheckClimateDataTests(unittest.TestCase): problems = self.run_report().checks[CHECK_CROSS] - self.assertEqual(problems, ["01001: Koppen stripe classes are set but koppenZone is Cfa, not Mixed"]) + self.assertEqual(problems, ["01001: Köppen stripe classes are set but koppenZone is Cfa, not Mixed"]) def test_invalid_or_identical_stripe_classes_are_reported(self) -> None: rows = valid_rows() @@ -143,7 +143,7 @@ class CheckClimateDataTests(unittest.TestCase): self.assertEqual( problems, - ["01001: Koppen stripe classes are both Csb", "02013: invalid Koppen stripe classes 'Xyz'/'Dfc'"], + ["01001: Köppen stripe classes are both Csb", "02013: invalid Köppen stripe classes 'Xyz'/'Dfc'"], ) def test_bad_values_and_categories_are_reported(self) -> None: