Refine climate metrics and data pipeline

This commit is contained in:
2026-07-24 22:55:14 -04:00
parent 016f7386d8
commit 4e2d878e5b
32 changed files with 3557 additions and 3347 deletions
@@ -14,7 +14,6 @@ from pathlib import Path
from build_county_locally_extreme_data import OLD_APP_FIPS_TO_CURRENT_FIPS
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_DTR_CSV = REPO_ROOT / "data" / "noaa" / "county_diurnal_temperature_range.csv"
@@ -15,7 +15,6 @@ import argparse
import csv
from pathlib import Path
DEFAULT_CLIMATE_DATA = Path("data/climate-data.csv")
DEFAULT_GRIDMET_HUMIDITY = Path("data/gridmet/county_gridmet_humidity.csv")
METRIC_FIELDS = [
@@ -17,7 +17,6 @@ import argparse
import csv
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_COMPARISON = REPO_ROOT / "data" / "noaa" / "county_locally_extreme_days_comparison.csv"
@@ -84,13 +83,13 @@ def load_locally_extreme_lookup(comparison_csv: Path) -> dict[str, dict]:
def load_solar_ghi_lookup(solar_ghi_csv: Path) -> dict[str, str]:
"""Map county FIPS codes to average daily GHI values from a county CSV."""
"""Map county FIPS codes to mean daily GHI values from a county CSV."""
if not solar_ghi_csv.exists():
return {}
fieldnames, rows = read_csv_rows(solar_ghi_csv)
if "avgSolarGhiKwhM2Day" not in fieldnames:
raise ValueError(f"{solar_ghi_csv} must include avgSolarGhiKwhM2Day.")
if "meanDailyGlobalHorizontalRadiationKwhM2Day" not in fieldnames:
raise ValueError(f"{solar_ghi_csv} must include meanDailyGlobalHorizontalRadiationKwhM2Day.")
if "county_fips" in fieldnames:
fips_field = "county_fips"
@@ -102,14 +101,14 @@ def load_solar_ghi_lookup(solar_ghi_csv: Path) -> dict[str, str]:
lookup: dict[str, str] = {}
for row in rows:
county_fips = normalize_fips(row.get(fips_field, ""))
raw_value = (row.get("avgSolarGhiKwhM2Day") or "").strip()
raw_value = (row.get("meanDailyGlobalHorizontalRadiationKwhM2Day") or "").strip()
if not county_fips or not raw_value:
continue
try:
float(raw_value)
except ValueError as error:
raise ValueError(
f"Invalid avgSolarGhiKwhM2Day value for county {county_fips}: {raw_value}"
f"Invalid meanDailyGlobalHorizontalRadiationKwhM2Day value for county {county_fips}: {raw_value}"
) from error
lookup[county_fips] = raw_value
@@ -193,23 +192,23 @@ def apply_locally_extreme_metric(
polygon_solar_value = polygon_solar_lookup.get(county_fips)
representative_solar_value = representative_solar_lookup.get(county_fips)
current_solar_value = (row.get("avgSolarGhiKwhM2Day") or "").strip()
current_solar_value = (row.get("meanDailyGlobalHorizontalRadiationKwhM2Day") or "").strip()
if polygon_solar_value:
row["avgSolarGhiKwhM2Day"] = polygon_solar_value
row["meanDailyGlobalHorizontalRadiationKwhM2Day"] = polygon_solar_value
row["source"] = replace_solar_source_tag(
row.get("source", ""),
"solar-ghi-polygon-archive-area-weighted",
)
polygon_solar_count += 1
elif representative_solar_value or current_solar_value:
row["avgSolarGhiKwhM2Day"] = representative_solar_value or current_solar_value
row["meanDailyGlobalHorizontalRadiationKwhM2Day"] = representative_solar_value or current_solar_value
row["source"] = replace_solar_source_tag(
row.get("source", ""),
"solar-ghi-representative-point",
)
representative_solar_count += 1
else:
row["avgSolarGhiKwhM2Day"] = ""
row["meanDailyGlobalHorizontalRadiationKwhM2Day"] = ""
row["source"] = replace_solar_source_tag(row.get("source", ""), "missing-solar-ghi")
missing_solar_count += 1
@@ -221,9 +220,9 @@ def apply_locally_extreme_metric(
print(f"Wrote {len(climate_rows)} rows to {out}")
print(f"Updated absoluteExtremeDays for {absolute_updated_count} rows.")
print(f"Left absoluteExtremeDays blank for {absolute_missing_count} rows without NOAA data.")
print(f"Updated avgSolarGhiKwhM2Day from polygon archives for {polygon_solar_count} rows.")
print(f"Updated meanDailyGlobalHorizontalRadiationKwhM2Day from polygon archives for {polygon_solar_count} rows.")
print(f"Kept representative-point solar fallback for {representative_solar_count} rows.")
print(f"Left avgSolarGhiKwhM2Day blank for {missing_solar_count} rows without solar data.")
print(f"Left meanDailyGlobalHorizontalRadiationKwhM2Day blank for {missing_solar_count} rows without solar data.")
def parse_args() -> argparse.Namespace:
@@ -1,9 +1,9 @@
#!/usr/bin/env python3
"""
Merge NSRDB cloudiness metrics into climate-data.csv.
Merge NSRDB clear-sky GHI reduction metrics into climate-data.csv.
Adds:
- cloudinessIndexPct
- clearSkyGhiReductionIndex
Polygon area-weighted values are used first when available; representative-point
values remain the fallback.
@@ -15,33 +15,33 @@ import argparse
import csv
from pathlib import Path
DEFAULT_CLIMATE_DATA = Path("data/climate-data.csv")
DEFAULT_POLYGON_CLOUD_SUMMARY = Path("data/nrel/county_polygon_cloud_summary.csv")
DEFAULT_REPRESENTATIVE_POINT_CLOUD_SUMMARY = Path("data/nrel/county_representative_point_cloud_summary.csv")
METRIC_FIELD = "cloudinessIndexPct"
POLYGON_SOURCE_TAG = "nsrdb-polygon-area-weighted-cloudiness-tmy"
REPRESENTATIVE_POINT_SOURCE_TAG = "nsrdb-representative-point-cloudiness-tmy"
METRIC_FIELD = "clearSkyGhiReductionIndex"
POLYGON_METRIC_FIELD = "areaWeightedClearSkyGhiReductionIndex"
POLYGON_SOURCE_TAG = "nsrdb-polygon-area-weighted-clear-sky-ghi-reduction-tmy"
REPRESENTATIVE_POINT_SOURCE_TAG = "nsrdb-representative-point-clear-sky-ghi-reduction-tmy"
CLOUD_SOURCE_TAGS = {
POLYGON_SOURCE_TAG,
REPRESENTATIVE_POINT_SOURCE_TAG,
}
def load_cloud_values(path: Path) -> dict[str, str]:
"""Read cloudiness values keyed by county FIPS."""
def load_cloud_values(path: Path, metric_field: str = METRIC_FIELD) -> dict[str, str]:
"""Read clear-sky GHI reduction values keyed by county FIPS."""
values: dict[str, str] = {}
with path.open("r", encoding="utf-8-sig", newline="") as handle:
reader = csv.DictReader(handle)
fieldnames = reader.fieldnames or []
required_fields = {"county_fips", METRIC_FIELD}
required_fields = {"county_fips", metric_field}
missing_fields = sorted(required_fields - set(fieldnames))
if missing_fields:
raise ValueError(f"{path} is missing fields: {', '.join(missing_fields)}")
for row in reader:
county_fips = (row.get("county_fips") or "").strip().zfill(5)
value = (row.get(METRIC_FIELD) or "").strip()
value = (row.get(metric_field) or "").strip()
if county_fips and value:
values[county_fips] = value
return values
@@ -75,13 +75,13 @@ def merge_metric(
polygon_cloud_summary: Path,
representative_point_cloud_summary: Path,
) -> tuple[int, int, int, int]:
"""Merge cloudiness values into the app climate CSV."""
polygon_cloudiness_by_fips = (
load_cloud_values(polygon_cloud_summary)
"""Merge clear-sky GHI reduction values into the app climate CSV."""
polygon_reduction_by_fips = (
load_cloud_values(polygon_cloud_summary, POLYGON_METRIC_FIELD)
if polygon_cloud_summary.exists()
else {}
)
representative_cloudiness_by_fips = load_cloud_values(representative_point_cloud_summary)
representative_reduction_by_fips = load_cloud_values(representative_point_cloud_summary)
with climate_data.open("r", encoding="utf-8", newline="") as handle:
reader = csv.DictReader(handle)
@@ -91,15 +91,15 @@ def merge_metric(
if "countyFips" not in fieldnames:
raise ValueError(f"{climate_data} is missing countyFips")
ensure_field_after(fieldnames, METRIC_FIELD, "avgSolarGhiKwhM2Day")
ensure_field_after(fieldnames, METRIC_FIELD, "meanDailyGlobalHorizontalRadiationKwhM2Day")
polygon_count = 0
representative_count = 0
missing_count = 0
for row in rows:
county_fips = (row.get("countyFips") or "").strip().zfill(5)
polygon_value = polygon_cloudiness_by_fips.get(county_fips, "")
representative_value = representative_cloudiness_by_fips.get(county_fips, "")
polygon_value = polygon_reduction_by_fips.get(county_fips, "")
representative_value = representative_reduction_by_fips.get(county_fips, "")
value = polygon_value or representative_value
row[METRIC_FIELD] = value
if polygon_value:
@@ -120,7 +120,7 @@ def merge_metric(
def main() -> int:
parser = argparse.ArgumentParser(description="Merge NSRDB cloudiness metric into climate-data.csv.")
parser = argparse.ArgumentParser(description="Merge NSRDB clear-sky GHI reduction metric into climate-data.csv.")
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA)
parser.add_argument("--polygon-cloud-summary", type=Path, default=DEFAULT_POLYGON_CLOUD_SUMMARY)
parser.add_argument(
@@ -14,7 +14,6 @@ from pathlib import Path
import numpy as np
import xarray as xr
from build_county_climate_data import (
MONTH_NAMES,
_as_monthly_climatology,
@@ -24,7 +23,6 @@ from build_county_climate_data import (
_zonal_mean,
)
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_COUNTIES_GEOJSON = REPO_ROOT / "data" / "geojson-counties-fips.json"
+19 -21
View File
@@ -13,7 +13,7 @@ Metrics produced per county:
- wettestPrecipMonth: month with the highest 1991-2020 county mean precipitation
- driestPrecipMonth: month with the lowest 1991-2020 county mean precipitation
- extremeDays: count of normal-days with Tmax >= hot threshold or Tmin <= freeze threshold
- avgSolarGhiKwhM2Day: annual average daily global horizontal irradiance (GHI), when a solar raster or representative-point CSV is provided
- meanDailyGlobalHorizontalRadiationKwhM2Day: mean daily global horizontal radiation (GHI), when a solar raster or representative-point CSV is provided
This script is intended for offline generation of complete county records.
"""
@@ -22,15 +22,15 @@ from __future__ import annotations
import argparse
import csv
import json
from pathlib import Path
from typing import Dict, List, Tuple
import geopandas as gpd
import numpy as np
import rasterio
from rasterio.features import geometry_mask
import xarray as xr
from affine import Affine
from rasterio.features import geometry_mask
DEFAULT_COUNTIES_GEOJSON_URL = "https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json"
@@ -261,7 +261,7 @@ def _select_data_var(dataset: xr.Dataset, preferred: str) -> str:
def _load_solar_ghi_csv(solar_ghi_csv: Path, counties: gpd.GeoDataFrame) -> List[float]:
"""Load county-keyed average daily GHI values from a representative-point or area-average CSV."""
"""Load county-keyed mean daily GHI values from a representative-point or area-average CSV."""
with solar_ghi_csv.open(newline="", encoding="utf-8") as handle:
reader = csv.DictReader(handle)
if reader.fieldnames is None:
@@ -276,22 +276,22 @@ def _load_solar_ghi_csv(solar_ghi_csv: Path, counties: gpd.GeoDataFrame) -> List
f"Solar GHI CSV at {solar_ghi_csv} must include county_fips or countyFips."
)
if "avgSolarGhiKwhM2Day" not in reader.fieldnames:
if "meanDailyGlobalHorizontalRadiationKwhM2Day" not in reader.fieldnames:
raise ValueError(
f"Solar GHI CSV at {solar_ghi_csv} must include avgSolarGhiKwhM2Day."
f"Solar GHI CSV at {solar_ghi_csv} must include meanDailyGlobalHorizontalRadiationKwhM2Day."
)
solar_by_fips: Dict[str, float] = {}
for row in reader:
county_fips = _normalize_fips(row.get(fips_field, ""), 5)
raw_value = str(row.get("avgSolarGhiKwhM2Day", "")).strip()
raw_value = str(row.get("meanDailyGlobalHorizontalRadiationKwhM2Day", "")).strip()
if not county_fips or not raw_value:
continue
try:
solar_by_fips[county_fips] = float(raw_value)
except ValueError as exc:
raise ValueError(
f"Invalid avgSolarGhiKwhM2Day value for county {county_fips}: {raw_value}"
f"Invalid meanDailyGlobalHorizontalRadiationKwhM2Day value for county {county_fips}: {raw_value}"
) from exc
return [
@@ -357,7 +357,7 @@ def _infer_time_resolution_days(data_array: xr.DataArray) -> float:
return float(np.median(deltas))
def _extract_grid_2d(data_array: xr.DataArray) -> Tuple[np.ndarray, "Affine"]:
def _extract_grid_2d(data_array: xr.DataArray) -> Tuple[np.ndarray, Affine]:
"""Convert a lat/lon slice to a raster array and transform."""
# Expected shape for 2D arrays: lat, lon
# Build affine from center coordinates.
@@ -376,8 +376,6 @@ def _extract_grid_2d(data_array: xr.DataArray) -> Tuple[np.ndarray, "Affine"]:
x_res = abs(lon[1] - lon[0])
y_res = abs(lat[0] - lat[1])
from affine import Affine
top_left_x = lon.min() - (x_res / 2.0)
top_left_y = lat.max() + (y_res / 2.0)
transform = Affine.translation(top_left_x, top_left_y) * Affine.scale(x_res, -y_res)
@@ -477,7 +475,7 @@ def _compute_extreme_days(
day_tmax = _zonal_mean(tmax_arr, tmax_transform, counties)
day_tmin = _zonal_mean(tmin_arr, tmin_transform, counties)
for idx, (mx, mn) in enumerate(zip(day_tmax, day_tmin)):
for idx, (mx, mn) in enumerate(zip(day_tmax, day_tmin, strict=True)):
if np.isnan(mx) or np.isnan(mn):
continue
if mx >= hot_threshold_c or mn <= freeze_threshold_c:
@@ -509,7 +507,7 @@ def _compute_extreme_days_monthly_proxy(
month_tmax = _zonal_mean(tmax_arr, tmax_transform, counties)
month_tmin = _zonal_mean(tmin_arr, tmin_transform, counties)
for idx, (mx, mn) in enumerate(zip(month_tmax, month_tmin)):
for idx, (mx, mn) in enumerate(zip(month_tmax, month_tmin, strict=True)):
if np.isnan(mx) or np.isnan(mn):
continue
if mx >= hot_threshold_c or mn <= freeze_threshold_c:
@@ -634,9 +632,9 @@ def build_county_records(
solar_source_tag = "solar-ghi-representative-point"
else:
if solar_ghi_raster is not None:
print(f"Solar GHI raster not found at {solar_ghi_raster}; leaving avgSolarGhiKwhM2Day blank.")
print(f"Solar GHI raster not found at {solar_ghi_raster}; leaving meanDailyGlobalHorizontalRadiationKwhM2Day blank.")
if solar_ghi_csv is not None:
print(f"Solar GHI CSV not found at {solar_ghi_csv}; leaving avgSolarGhiKwhM2Day blank.")
print(f"Solar GHI CSV not found at {solar_ghi_csv}; leaving meanDailyGlobalHorizontalRadiationKwhM2Day blank.")
solar_ghi_kwh_m2_day = [float("nan")] * len(counties)
solar_source_tag = "no-solar-ghi-source"
@@ -699,14 +697,14 @@ def build_county_records(
if not use_missing_extreme
else None
)
avg_solar_ghi_value = (
mean_daily_global_horizontal_radiation_value = (
round(float(solar_ghi_kwh_m2_day[idx]), 2)
if np.isfinite(solar_ghi_kwh_m2_day[idx])
else None
)
source_suffix = " + missing-noaa-numeric" if used_any_missing_numeric else ""
solar_source_suffix = "" if avg_solar_ghi_value is not None else " + missing-solar-ghi"
solar_source_suffix = "" if mean_daily_global_horizontal_radiation_value is not None else " + missing-solar-ghi"
record = {
"countyName": county_name,
@@ -718,7 +716,7 @@ def build_county_records(
"wettestPrecipMonth": wettest_precip_month,
"driestPrecipMonth": driest_precip_month,
"extremeDays": extreme_days_value,
"avgSolarGhiKwhM2Day": avg_solar_ghi_value,
"meanDailyGlobalHorizontalRadiationKwhM2Day": mean_daily_global_horizontal_radiation_value,
"source": (
"kg-beck2023 + noaa-nclimgrid-1991-2020 "
f"({extreme_days_source_tag}) + {solar_source_tag}{source_suffix}{solar_source_suffix}"
@@ -751,7 +749,7 @@ def write_csv(records: Dict[str, dict], out_file: Path) -> None:
"wettestPrecipMonth",
"driestPrecipMonth",
"extremeDays",
"avgSolarGhiKwhM2Day",
"meanDailyGlobalHorizontalRadiationKwhM2Day",
"source",
]
@@ -818,14 +816,14 @@ def parse_args() -> argparse.Namespace:
"--solar-ghi-raster",
type=Path,
default=None,
help="Optional raster of annual average daily GHI in kWh/m2/day for avgSolarGhiKwhM2Day.",
help="Optional raster of mean daily GHI in kWh/m2/day for meanDailyGlobalHorizontalRadiationKwhM2Day.",
)
parser.add_argument(
"--solar-ghi-csv",
type=Path,
default=None,
help=(
"Optional county CSV with avgSolarGhiKwhM2Day. Used as a representative-point fallback "
"Optional county CSV with meanDailyGlobalHorizontalRadiationKwhM2Day. Used as a representative-point fallback "
"when --solar-ghi-raster is not supplied."
),
)
@@ -27,7 +27,6 @@ from build_county_locally_extreme_data import (
rows_by_fips,
)
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_OUT = REPO_ROOT / "data" / "noaa" / "county_diurnal_temperature_range.csv"
SOURCE_TAG = "noaa-nclimgrid-daily-county-area-averages-scaled"
@@ -90,7 +89,11 @@ def build_diurnal_temperature_range(
),
)
for tmax_value, tmin_value in zip(tmax_record.values, tmin_record.values):
for tmax_value, tmin_value in zip(
tmax_record.values,
tmin_record.values,
strict=True,
):
if tmax_value is None or tmin_value is None:
continue
daily_range_c = tmax_value - tmin_value
+5 -2
View File
@@ -36,7 +36,6 @@ from pathlib import Path
from tempfile import NamedTemporaryFile
from typing import Dict, Iterable, Iterator, List, Tuple
NOAA_AVERAGES_BASE_URL = "https://www.ncei.noaa.gov/data/nclimgrid-daily/access/averages"
NOAA_STATE_CROSSWALK_URL = (
"https://www.ncei.noaa.gov/data/nclimgrid-daily/doc/us-state-codes_ncei-to-fips.csv"
@@ -518,7 +517,11 @@ def build_annual_counts(
)
counts = annual_counts.setdefault((county_fips, year), AnnualCounts())
for tmax_value, tmin_value in zip(tmax_record.values, tmin_record.values):
for tmax_value, tmin_value in zip(
tmax_record.values,
tmin_record.values,
strict=True,
):
if tmax_value is None or tmin_value is None:
continue
@@ -13,7 +13,6 @@ from pathlib import Path
import geopandas as gpd
DEFAULT_COUNTIES_GEOJSON = Path("data/geojson-counties-fips.json")
DEFAULT_OUTPUT_CSV = Path("data/nrel/county_representative_points.csv")
+18 -16
View File
@@ -45,14 +45,14 @@ If you have monthly nClimGrid history files (for example `nclimgrid_tavg.nc`) ra
- Original Plotly county geometry source: [https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json](https://raw.githubusercontent.com/plotly/datasets/master/geojson-counties-fips.json)
- Official county geometry reference (Census TIGER/Line): [https://www.census.gov/geographies/mapping-files/time-series/geo/tiger-line-file.html](https://www.census.gov/geographies/mapping-files/time-series/geo/tiger-line-file.html)
## Source 4: Solar resource (`avgSolarGhiKwhM2Day`)
## Source 4: Solar resource (`meanDailyGlobalHorizontalRadiationKwhM2Day`)
- Recommended dataset: NREL National Solar Radiation Database (NSRDB)
- Data/API page: [https://developer.nrel.gov/docs/solar/nsrdb/](https://developer.nrel.gov/docs/solar/nsrdb/)
- Maps/geospatial data page: [https://www.nrel.gov/gis/solar-resource-maps](https://www.nrel.gov/gis/solar-resource-maps)
- Fallback point API option: NASA POWER `ALLSKY_SFC_SW_DWN` (surface shortwave downwelling radiation), [https://power.larc.nasa.gov/docs/tutorials/service-data-request/api/](https://power.larc.nasa.gov/docs/tutorials/service-data-request/api/)
The app metric is designed for annual average daily global horizontal irradiance (GHI), in `kWh/m2/day`.
The app metric is designed for mean daily global horizontal radiation (GHI), in `kWh/m2/day`.
For county means, use a gridded annual GHI raster and pass it to the generator with `--solar-ghi-raster`.
### First test: NSRDB county representative points
@@ -80,12 +80,12 @@ Outputs:
- `data/nrel/county_representative_points.csv`: county FIPS, name, state, latitude, and longitude.
- `data/nrel/representative_point_csv/`: cached raw NSRDB CSV responses by county FIPS.
- `data/nrel/county_representative_point_ghi_summary.csv`: summarized `avgSolarGhiKwhM2Day` values.
- `data/nrel/county_representative_point_ghi_summary.csv`: summarized `meanDailyGlobalHorizontalRadiationKwhM2Day` values.
The calculation is:
```text
avgSolarGhiKwhM2Day = sum(hourly GHI) / 1000 / 365
meanDailyGlobalHorizontalRadiationKwhM2Day = sum(hourly GHI) / 1000 / 365
```
If the raw cache is complete but the summary CSV only contains the last fetched batch, rebuild the summary from cached files without calling the API:
@@ -96,9 +96,9 @@ If the raw cache is complete but the summary CSV only contains the last fetched
This point-based workflow is easier to validate and resume than full polygon downloads, but it is an approximation of county sunlight rather than an area-weighted county mean.
### First cloud-cover pass: NSRDB representative points
### First clear-sky GHI reduction pass: NSRDB representative points
For a fast cloud-cover input layer, fetch representative-point NSRDB CSVs with observed GHI, Clearsky GHI, and Cloud Type:
For a fast clear-sky GHI reduction input layer, fetch representative-point NSRDB CSVs with observed GHI, Clearsky GHI, and Cloud Type:
```powershell
.venv\Scripts\python.exe scripts\fetch_nsrdb_representative_point_cloud_metrics.py --limit 10
@@ -113,18 +113,18 @@ Once the first batch looks right, fetch every county:
Outputs:
- `data/nrel/representative_point_cloud_csv/`: cached raw NSRDB CSV responses with `ghi,clearsky_ghi,cloud_type`.
- `data/nrel/county_representative_point_cloud_summary.csv`: summarized cloud-cover proxy fields.
- `data/nrel/county_representative_point_cloud_summary.csv`: summarized clear-sky GHI reduction fields.
- `data/nrel/county_representative_point_cloud_error_log.csv`: failed county requests with redacted API context.
The primary calculation uses daylight rows where Clearsky GHI is at least 50 W/m2:
```text
cloudinessIndexPct = 1 - mean(clamped(GHI / Clearsky GHI, 0, 1))
clearSkyGhiReductionIndex = 1 - mean(clamped(GHI / Clearsky GHI, 0, 1))
```
The same summary also stores daylight row counts, observed-to-clear-sky ratio, and broad Cloud Type frequency buckets. This is faster than the polygon archive workflow, but it remains a representative-point county approximation.
Apply the representative-point cloudiness metric to the browser app CSV:
Apply the representative-point clear-sky GHI reduction metric to the browser app CSV:
```powershell
.venv\Scripts\python.exe scripts\apply_nsrdb_cloud_metric_to_climate_data.py
@@ -134,7 +134,7 @@ The current cloud summary covers the 3,143 county representative points and leav
### County-average target: NSRDB polygon cloud archive requests
For a less noisy county-level cloudiness layer, submit county polygons to the NSRDB archive workflow with `ghi,clearsky_ghi,cloud_type`. This uses the same polygon tiling and pacing logic as the GHI archive workflow, but writes separate cloud manifests, response JSON files, archives, and summaries.
For a less noisy county-level clear-sky GHI reduction layer, submit county polygons to the NSRDB archive workflow with `ghi,clearsky_ghi,cloud_type`. This uses the same polygon tiling and pacing logic as the GHI archive workflow, but writes separate cloud manifests, response JSON files, archives, and summaries.
Start with a dry run:
@@ -166,7 +166,7 @@ Download completed archives:
.venv\Scripts\python.exe scripts\download_nsrdb_county_polygon_cloud_archives.py
```
Summarize downloaded archives into area-weighted county cloudiness:
Summarize downloaded archives into area-weighted county clear-sky GHI reduction:
```powershell
.venv\Scripts\python.exe scripts\summarize_nsrdb_county_polygon_cloud_archives.py --reuse-existing-output
@@ -178,7 +178,7 @@ Outputs:
- `data/nrel/county_polygon_cloud_error_log.csv`: cloud archive request errors.
- `data/nrel/polygon_cloud_request_responses/`: raw NSRDB cloud acknowledgement JSON files.
- `data/nrel/polygon_cloud_archives/`: downloaded cloud ZIP archives.
- `data/nrel/county_polygon_cloud_summary.csv`: area-weighted county cloudiness summaries.
- `data/nrel/county_polygon_cloud_summary.csv`: area-weighted county clear-sky GHI reduction summaries.
Then update the browser app CSV. Polygon area-weighted values are used first when `data/nrel/county_polygon_cloud_summary.csv` exists; representative-point values remain the fallback:
@@ -228,7 +228,7 @@ Outputs:
- `data/nrel/county_polygon_ghi_error_log.csv`: counties that need retrying or tiling.
- `data/nrel/polygon_request_responses/`: raw API acknowledgement JSON files.
- `data/nrel/polygon_archives/`: downloaded county or tiled county ZIP archives.
- `data/nrel/county_polygon_ghi_summary.csv`: summarized polygon archive GHI, including `avgSolarGhiKwhM2Day`.
- `data/nrel/county_polygon_ghi_summary.csv`: summarized polygon archive GHI, including `meanDailyGlobalHorizontalRadiationKwhM2Day`.
Download completed GHI archives:
@@ -262,8 +262,10 @@ Then update the app CSV. Polygon archive GHI is used first; representative-point
- `wettestPrecipMonth`: month with the highest 1991-2020 county mean precipitation total.
- `driestPrecipMonth`: month with the lowest 1991-2020 county mean precipitation total.
- previous `extremeDays` / `oldExtremeDays`: count of daily-normal or monthly-proxy days where county mean `tmax >= 95F` or `tmin <= 32F` (thresholds configurable in script). This is preserved only as an audit column after the NOAA nClimGrid-Daily metrics are applied.
- `avgSolarGhiKwhM2Day`: county mean annual average daily GHI, in `kWh/m2/day`. The current app CSV uses NSRDB polygon archive area-weighted values where available, with representative-point values kept as fallback.
- `cloudinessIndexPct`: representative-point NSRDB daylight cloudiness proxy derived from observed GHI divided by Clearsky GHI. Higher values mean observed irradiance is lower relative to modeled clear-sky irradiance. Despite the legacy field name, values are stored on a 0-1 scale.
- `humidHeatDays`: average annual count of days where estimated Heat Index is at least 90 F, the lower bound of the NWS Extreme Caution category. The NWS Heat Index algorithm is applied to NOAA nClimGrid-Daily county `tmax` and gridMET daily minimum relative humidity (`rmin`) for 1991-2020. Because the inputs are paired daily extrema rather than coincident hourly observations, this is an estimated daily-peak proxy.
- `avgSummerSpecificHumidityGKg`: county-cell-weighted mean gridMET specific humidity for June-August 1991-2020, converted from kg/kg to g/kg.
- `meanDailyGlobalHorizontalRadiationKwhM2Day`: county mean daily global horizontal radiation (GHI), in `kWh/m2/day`. The current app CSV uses NSRDB polygon archive area-weighted values where available, with representative-point values kept as fallback.
- `clearSkyGhiReductionIndex`: NSRDB daylight clear-sky GHI reduction index derived from observed GHI divided by modeled clear-sky GHI. Higher values mean observed irradiance is lower relative to clear-sky conditions. Values are stored on a 0-1 scale.
## Run the generator
@@ -297,7 +299,7 @@ Notes:
- For counties outside CONUS coverage in NOAA gridded files, fallback values are applied by the script when no valid grid values intersect.
- For physically-based daily `extremeDays`, provide true daily grids and set `--extreme-days-mode require-daily`.
- If `--counties-geojson` does not exist locally, the script will try to download the county GeoJSON automatically from the Plotly URL above and cache it at that path.
- `--solar-ghi-csv` is optional and can load county-keyed solar summaries into `avgSolarGhiKwhM2Day`.
- `--solar-ghi-csv` is optional and can load county-keyed solar summaries into `meanDailyGlobalHorizontalRadiationKwhM2Day`.
- `--solar-ghi-raster` is optional and takes precedence over `--solar-ghi-csv`. If you have a gridded annual GHI raster, add `--solar-ghi-raster path/to/annual_ghi_kwh_m2_day.tif` for a true county-area raster mean.
## Source 5: NOAA nClimGrid-Daily county area averages (`locallyExtremeDays`)
-1
View File
@@ -20,7 +20,6 @@ import urllib.error
import urllib.request
from pathlib import Path
DEFAULT_BASE_URL = "https://www.northwestknowledge.net/metdata/data"
DEFAULT_VARIABLES = ("sph", "rmax", "rmin")
@@ -20,7 +20,6 @@ from enum import Enum
from pathlib import Path
from typing import Any
DEFAULT_TIMEOUT = 300
DEFAULT_CHUNK_SIZE = 1024 * 1024
@@ -12,7 +12,6 @@ import sys
import download_nsrdb_county_polygon_archives as polygon_download
DEFAULT_ARGS = [
"--response-dir",
"data/nrel/polygon_cloud_request_responses",
@@ -13,7 +13,6 @@ import sys
import download_nsrdb_county_polygon_archives as polygon_download
DEFAULT_ARGS = [
"--response-dir",
"data/nrel/polygon_request_responses",
@@ -1,6 +1,6 @@
#!/usr/bin/env python3
"""
Fetch NSRDB representative-point inputs for a county cloud-cover metric.
Fetch NSRDB representative-point inputs for a county clear-sky GHI reduction metric.
This uses the direct single-point NSRDB CSV endpoint rather than polygon archive
requests. It is faster for first-pass cloud metrics because it downloads one CSV
@@ -12,10 +12,10 @@ ghi,clearsky_ghi,cloud_type
The summary metric is based on daylight rows with valid GHI and Clearsky GHI:
cloudinessIndexPct = 1 - mean(clamped(GHI / Clearsky GHI))
clearSkyGhiReductionIndex = 1 - mean(clamped(GHI / Clearsky GHI))
where the ratio is clamped to [0, 1] so occasional above-clear-sky modeled GHI
does not create negative cloudiness.
does not create negative reduction values.
"""
from __future__ import annotations
@@ -33,7 +33,6 @@ import urllib.parse
import urllib.request
from pathlib import Path
DEFAULT_POINTS_CSV = Path("data/nrel/county_representative_points.csv")
DEFAULT_OUTPUT_CSV = Path("data/nrel/county_representative_point_cloud_summary.csv")
DEFAULT_ERROR_CSV = Path("data/nrel/county_representative_point_cloud_error_log.csv")
@@ -73,7 +72,7 @@ CLOUD_SUMMARY_FIELDS = [
"state_abbr",
"lat",
"lon",
"cloudinessIndexPct",
"clearSkyGhiReductionIndex",
"avgObservedToClearskyRatio",
"daylightRows",
"allRows",
@@ -442,7 +441,7 @@ def summarize_cloud_metrics(
raise ValueError("No daylight rows with valid GHI and Clearsky GHI were found.")
avg_ratio = ratio_sum / daylight_rows
cloudiness_pct = 1 - avg_ratio
clear_sky_ghi_reduction = 1 - avg_ratio
clear_or_probably_clear = daylight_cloud_type_counts.get("0", 0) + daylight_cloud_type_counts.get("1", 0)
cloudy_or_obscured = sum(
daylight_cloud_type_counts.get(code, 0)
@@ -467,7 +466,7 @@ def summarize_cloud_metrics(
"state_abbr": point["state_abbr"],
"lat": point["lat"],
"lon": point["lon"],
"cloudinessIndexPct": fmt(cloudiness_pct, 4),
"clearSkyGhiReductionIndex": fmt(clear_sky_ghi_reduction, 4),
"avgObservedToClearskyRatio": fmt(avg_ratio, 4),
"daylightRows": str(daylight_rows),
"allRows": str(all_rows),
@@ -675,7 +674,7 @@ def log_county_result(
if summary is not None:
print(
" "
f"cloudiness={summary['cloudinessIndexPct']}, "
f"clear_sky_ghi_reduction={summary['clearSkyGhiReductionIndex']}, "
f"clear_ratio={summary['avgObservedToClearskyRatio']}, "
f"daylight_rows={summary['daylightRows']}"
)
@@ -783,7 +782,7 @@ def parse_args() -> argparse.Namespace:
"--min-clearsky-ghi",
type=float,
default=DEFAULT_MIN_CLEARSKY_GHI,
help="Minimum Clearsky GHI W/m2 for daylight cloudiness ratio rows.",
help="Minimum Clearsky GHI W/m2 for daylight GHI ratio rows.",
)
parser.add_argument(
"--no-polar-fallback",
@@ -20,7 +20,6 @@ import urllib.parse
import urllib.request
from pathlib import Path
DEFAULT_POINTS_CSV = Path("data/nrel/county_representative_points.csv")
DEFAULT_OUTPUT_CSV = Path("data/nrel/county_representative_point_ghi_summary.csv")
DEFAULT_ERROR_CSV = Path("data/nrel/county_representative_point_ghi_error_log.csv")
@@ -247,7 +246,7 @@ def summarize_ghi(point: dict[str, str], csv_text: str, source_file: Path, sourc
"state_abbr": point["state_abbr"],
"lat": point["lat"],
"lon": point["lon"],
"avgSolarGhiKwhM2Day": f"{avg_daily_ghi:.3f}",
"meanDailyGlobalHorizontalRadiationKwhM2Day": f"{avg_daily_ghi:.3f}",
"ghi_rows": str(len(ghi_values)),
"ghi_min": f"{min(ghi_values):.1f}",
"ghi_max": f"{max(ghi_values):.1f}",
@@ -269,7 +268,7 @@ def write_summary_csv(rows: list[dict[str, str]], output_csv: Path) -> None:
"state_abbr",
"lat",
"lon",
"avgSolarGhiKwhM2Day",
"meanDailyGlobalHorizontalRadiationKwhM2Day",
"ghi_rows",
"ghi_min",
"ghi_max",
@@ -4,7 +4,7 @@ Rebuild county representative-point GHI summaries from cached NSRDB raw CSV resp
This does not call the NSRDB API. It reads data/nrel/representative_point_csv/*.csv, computes:
avgSolarGhiKwhM2Day = sum(hourly GHI) / 1000 / 365
meanDailyGlobalHorizontalRadiationKwhM2Day = sum(hourly GHI) / 1000 / 365
and writes a county-keyed CSV compatible with build_county_climate_data.py.
"""
@@ -15,7 +15,6 @@ import argparse
import csv
from pathlib import Path
DEFAULT_POINTS_CSV = Path("data/nrel/county_representative_points.csv")
DEFAULT_REPRESENTATIVE_POINT_CSV_DIR = Path("data/nrel/representative_point_csv")
DEFAULT_OUTPUT_CSV = Path("data/nrel/county_representative_point_ghi_summary.csv")
@@ -82,7 +81,7 @@ def summarize_county(point: dict[str, str], raw_csv: Path) -> dict[str, str]:
"state_abbr": point["state_abbr"],
"lat": point["lat"],
"lon": point["lon"],
"avgSolarGhiKwhM2Day": f"{avg_daily_ghi:.3f}",
"meanDailyGlobalHorizontalRadiationKwhM2Day": f"{avg_daily_ghi:.3f}",
"ghi_rows": str(len(ghi_values)),
"ghi_min": f"{min(ghi_values):.1f}",
"ghi_max": f"{max(ghi_values):.1f}",
@@ -104,7 +103,7 @@ def write_summary_csv(rows: list[dict[str, str]], output_csv: Path) -> None:
"state_abbr",
"lat",
"lon",
"avgSolarGhiKwhM2Day",
"meanDailyGlobalHorizontalRadiationKwhM2Day",
"ghi_rows",
"ghi_min",
"ghi_max",
@@ -29,7 +29,6 @@ import geopandas as gpd
from shapely.geometry import box
from shapely.wkt import dumps as dump_wkt
DEFAULT_COUNTIES_GEOJSON = Path("data/geojson-counties-fips.json")
DEFAULT_ENDPOINT = "https://developer.nlr.gov/api/nsrdb/v2/solar/nsrdb-GOES-tmy-v4-0-0-download.json"
DEFAULT_POLAR_ENDPOINT = "https://developer.nlr.gov/api/nsrdb/v2/solar/nsrdb-polar-tmy-v4-0-0-download.json"
@@ -550,7 +549,7 @@ class ArchiveQueueMonitor:
)
)
now = time.monotonic()
for url, status in zip(unresolved_urls, statuses):
for url, status in zip(unresolved_urls, statuses, strict=True):
if status == "pending":
if url not in self.pending_urls:
ambiguous_403 += 1
@@ -584,7 +583,7 @@ class ArchiveQueueMonitor:
return {
"pending": sum(
status == "pending" and url in self.pending_urls
for url, status in zip(unresolved_urls, statuses)
for url, status in zip(unresolved_urls, statuses, strict=True)
),
"ambiguous_403": ambiguous_403,
"recently_downloaded": recently_downloaded,
@@ -1402,8 +1401,6 @@ def submit_county_request(
return [], build_error_row(county, profile, None, None, wkt, error)
for profile in request_profiles(args, county):
site_count: int | None = None
weight: int | None = None
final_profile = profile
try:
return [
@@ -1,6 +1,6 @@
#!/usr/bin/env python3
"""
Submit NSRDB polygon archive requests for county-average cloudiness inputs.
Submit NSRDB polygon archive requests for county-average clear-sky GHI reduction inputs.
This is a cloud-specific wrapper around request_nsrdb_county_polygon_archives.py.
It reuses the existing polygon tiling, site-count, Polar fallback, pacing, and
@@ -15,7 +15,6 @@ import sys
import request_nsrdb_county_polygon_archives as polygon_request
DEFAULT_ARGS = [
"--requests-csv",
"data/nrel/county_polygon_cloud_request_manifest.csv",
@@ -13,7 +13,6 @@ import sys
import request_nsrdb_county_polygon_archives as polygon_request
DEFAULT_ARGS = [
"--requests-csv",
"data/nrel/county_polygon_ghi_request_manifest.csv",
+28 -7
View File
@@ -25,7 +25,6 @@ import numpy as np
import pandas as pd
import xarray as xr
STATE_FIPS_TO_ABBR = {
"01": "AL",
"04": "AZ",
@@ -80,6 +79,7 @@ STATE_FIPS_TO_ABBR = {
CONUS_STATE_FIPS = set(STATE_FIPS_TO_ABBR)
REQUIRED_VARIABLES = ("sph", "rmax", "rmin")
SUMMER_MONTHS = {6, 7, 8}
DEFAULT_HEAT_INDEX_THRESHOLD_F = 90.0
NOAA_REGION_CODE_TO_FIPS_OVERRIDES = {
"18511": "11001",
}
@@ -436,8 +436,12 @@ def _county_means_chunk(
def _heat_index_f(t_f: np.ndarray, rh_pct: np.ndarray) -> np.ndarray:
"""Return the NWS Heat Index for Fahrenheit temperature and percent RH."""
rh = np.clip(rh_pct, 0, 100)
heat_index = (
simple_heat_index = 0.5 * (t_f + 61.0 + ((t_f - 68.0) * 1.2) + (rh * 0.094))
simple_heat_index = (simple_heat_index + t_f) / 2
regression_heat_index = (
-42.379
+ 2.04901523 * t_f
+ 10.14333127 * rh
@@ -451,9 +455,17 @@ def _heat_index_f(t_f: np.ndarray, rh_pct: np.ndarray) -> np.ndarray:
low_rh_adjustment = ((13 - rh) / 4) * np.sqrt(np.maximum((17 - np.abs(t_f - 95)) / 17, 0))
high_rh_adjustment = ((rh - 85) / 10) * ((87 - t_f) / 5)
heat_index = np.where((rh < 13) & (80 <= t_f) & (t_f <= 112), heat_index - low_rh_adjustment, heat_index)
heat_index = np.where((rh > 85) & (80 <= t_f) & (t_f <= 87), heat_index + high_rh_adjustment, heat_index)
return np.where(t_f >= 80, heat_index, t_f)
regression_heat_index = np.where(
(rh < 13) & (80 <= t_f) & (t_f <= 112),
regression_heat_index - low_rh_adjustment,
regression_heat_index,
)
regression_heat_index = np.where(
(rh > 85) & (80 <= t_f) & (t_f <= 87),
regression_heat_index + high_rh_adjustment,
regression_heat_index,
)
return np.where(simple_heat_index >= 80, regression_heat_index, simple_heat_index)
def _time_dimension(data_array: xr.DataArray, lat_dim: str, lon_dim: str) -> str:
@@ -646,7 +658,8 @@ def summarize(args: argparse.Namespace) -> None:
"gridmetSource": (
f"gridmet-{source_period} county-cell-weighted; "
"relative-humidity=(rmax+rmin)/2; "
"humidHeatDays=noaa-nclimgrid-tmax+gridmet-rmin heat-index-gte-90f"
"humidHeatDays=noaa-nclimgrid-tmax+gridmet-rmin "
f"heat-index-gte-{args.heat_index_threshold_f:g}f"
),
}
)
@@ -680,7 +693,15 @@ def main() -> int:
parser.add_argument("--years", type=_parse_years, default=None, help="Optional comma/range years, e.g. 1991-2020.")
parser.add_argument("--start-year", type=int, default=None)
parser.add_argument("--end-year", type=int, default=None)
parser.add_argument("--heat-index-threshold-f", type=float, default=90.0)
parser.add_argument(
"--heat-index-threshold-f",
type=float,
default=DEFAULT_HEAT_INDEX_THRESHOLD_F,
help=(
"Daily Heat Index threshold in F. The default 90 F is the lower "
"bound of the NWS Extreme Caution category."
),
)
parser.add_argument("--chunk-days", type=int, default=31)
summarize(parser.parse_args())
return 0
@@ -4,7 +4,7 @@ Summarize downloaded NSRDB county polygon GHI archives.
Each downloaded polygon archive contains one CSV per NSRDB site that intersects
the county polygon or tile. This script computes area-weighted county-level
average daily GHI values across those site CSVs, combining multiple tile
mean daily GHI values across those site CSVs, combining multiple tile
archives back into one county summary when present.
"""
@@ -18,7 +18,6 @@ import statistics
import zipfile
from pathlib import Path
DEFAULT_ARCHIVE_DIR = Path("data/nrel/polygon_archives")
DEFAULT_COUNTIES_GEOJSON = Path("data/geojson-counties-fips.json")
DEFAULT_REQUESTS_CSV = Path("data/nrel/county_polygon_ghi_request_manifest.csv")
@@ -37,7 +36,7 @@ SUMMARY_FIELDS = [
"request_site_count",
"ghi_rows_per_site_min",
"ghi_rows_per_site_max",
"avgSolarGhiKwhM2Day",
"meanDailyGlobalHorizontalRadiationKwhM2Day",
"areaWeightedAvgSolarGhiKwhM2Day",
"areaWeightedSites",
"weightedCellAreaKm2",
@@ -144,7 +143,10 @@ def extract_site_lon_lat(csv_text: str) -> tuple[float, float]:
if len(rows) < 2:
raise ValueError("Could not read NSRDB metadata rows.")
metadata = {key.strip().lower(): value.strip() for key, value in zip(rows[0], rows[1])}
metadata = {
key.strip().lower(): value.strip()
for key, value in zip(rows[0], rows[1], strict=True)
}
try:
lon = float(metadata["longitude"])
lat = float(metadata["latitude"])
@@ -185,7 +187,7 @@ def extract_ghi_values(csv_text: str) -> list[float]:
def site_average_daily_ghi(csv_text: str) -> tuple[float, int]:
"""Return average daily GHI in kWh/m2/day and number of GHI rows."""
"""Return mean daily GHI in kWh/m2/day and number of GHI rows."""
ghi_values = extract_ghi_values(csv_text)
return sum(ghi_values) / 1000 / 365, len(ghi_values)
@@ -353,7 +355,7 @@ def area_weighted_average(
weighted_sites = 0
county_area = sum(projected_polygon_area(polygon) for polygon in projected_county)
for site_average, (lon, lat) in zip(site_averages, site_lon_lats):
for site_average, (lon, lat) in zip(site_averages, site_lon_lats, strict=True):
x, y = project_lon_lat(lon, lat, reference_lat)
min_x = x - half_cell
min_y = y - half_cell
@@ -408,7 +410,7 @@ def summarize_archives(
"polygon_sites": len(site_averages),
"ghi_rows_per_site_min": min(row_counts),
"ghi_rows_per_site_max": max(row_counts),
"avgSolarGhiKwhM2Day": weighted["area_weighted_avg"],
"meanDailyGlobalHorizontalRadiationKwhM2Day": weighted["area_weighted_avg"],
"area_weighted_avg": weighted["area_weighted_avg"],
"area_weighted_sites": weighted["area_weighted_sites"],
"weighted_cell_area_km2": weighted["weighted_cell_area_km2"],
@@ -450,11 +452,11 @@ def build_summary_row(
if str(row.get("site_count", "")).strip().isdigit()
]
archive_zip = ";".join(str(path) for path in archive_paths)
final_avg = float(polygon_summary["avgSolarGhiKwhM2Day"])
final_avg = float(polygon_summary["meanDailyGlobalHorizontalRadiationKwhM2Day"])
area_weighted_avg = float(polygon_summary["area_weighted_avg"])
representative_point_avg = (
float(representative_point_row["avgSolarGhiKwhM2Day"])
if representative_point_row.get("avgSolarGhiKwhM2Day")
float(representative_point_row["meanDailyGlobalHorizontalRadiationKwhM2Day"])
if representative_point_row.get("meanDailyGlobalHorizontalRadiationKwhM2Day")
else None
)
weighted_minus_representative_point = (
@@ -475,7 +477,7 @@ def build_summary_row(
"request_site_count": str(sum(request_site_counts)) if request_site_counts else request_row.get("site_count", ""),
"ghi_rows_per_site_min": str(polygon_summary["ghi_rows_per_site_min"]),
"ghi_rows_per_site_max": str(polygon_summary["ghi_rows_per_site_max"]),
"avgSolarGhiKwhM2Day": format_float(final_avg),
"meanDailyGlobalHorizontalRadiationKwhM2Day": format_float(final_avg),
"areaWeightedAvgSolarGhiKwhM2Day": format_float(area_weighted_avg),
"areaWeightedSites": str(polygon_summary["area_weighted_sites"]),
"weightedCellAreaKm2": format_float(float(polygon_summary["weighted_cell_area_km2"]), 1),
@@ -4,15 +4,15 @@ Summarize downloaded NSRDB county polygon cloud archives.
Each downloaded polygon archive contains one CSV per NSRDB grid site that
intersects the county polygon or tile. This script computes area-weighted
county-level cloudiness metrics across those site CSVs, combining multiple tile
county-level clear-sky GHI reduction metrics across those site CSVs, combining multiple tile
archives back into one county summary when present.
Primary metric:
cloudinessIndexPct = 1 - mean(clamped(GHI / Clearsky GHI, 0, 1))
clearSkyGhiReductionIndex = 1 - mean(clamped(GHI / Clearsky GHI, 0, 1))
Despite the legacy "Pct" field name, the primary index is stored on a 0-1 scale.
Cloud Type bucket fields are stored as percentages.
The primary index is stored on a 0-1 scale. Cloud Type bucket fields are stored
as percentages.
"""
from __future__ import annotations
@@ -26,9 +26,9 @@ import zipfile
from pathlib import Path
from summarize_nsrdb_county_polygon_archives import (
area_weighted_average,
archive_groups_by_county,
archive_signature,
area_weighted_average,
county_fips_from_archive,
existing_archive_signature,
extract_site_lon_lat,
@@ -38,7 +38,6 @@ from summarize_nsrdb_county_polygon_archives import (
read_lookup,
)
DEFAULT_ARCHIVE_DIR = Path("data/nrel/polygon_cloud_archives")
DEFAULT_COUNTIES_GEOJSON = Path("data/geojson-counties-fips.json")
DEFAULT_REQUESTS_CSV = Path("data/nrel/county_polygon_cloud_request_manifest.csv")
@@ -59,15 +58,15 @@ SUMMARY_FIELDS = [
"rows_per_site_max",
"daylight_rows_per_site_min",
"daylight_rows_per_site_max",
"cloudinessIndexPct",
"areaWeightedCloudinessIndexPct",
"clearSkyGhiReductionIndex",
"areaWeightedClearSkyGhiReductionIndex",
"areaWeightedAvgObservedToClearskyRatio",
"areaWeightedSites",
"weightedCellAreaKm2",
"countyAreaKm2",
"site_cloudiness_min",
"site_cloudiness_max",
"site_cloudiness_stddev",
"site_clear_sky_ghi_reduction_min",
"site_clear_sky_ghi_reduction_max",
"site_clear_sky_ghi_reduction_stddev",
"clearOrProbablyClearPct",
"cloudyOrObscuredPct",
"fogPct",
@@ -75,7 +74,7 @@ SUMMARY_FIELDS = [
"iceCloudPct",
"cirrusPct",
"unknownCloudTypePct",
"representativePointCloudinessIndexPct",
"representativePointClearSkyGhiReductionIndex",
"areaWeightedMinusRepresentativePoint",
"areaWeightedPctDiffFromRepresentativePoint",
]
@@ -137,7 +136,7 @@ def pct(count: int, total: int) -> float:
def site_cloud_metrics(csv_text: str, min_clearsky_ghi: float) -> dict[str, float | int]:
"""Return site-level cloudiness metrics from one NSRDB CSV response."""
"""Return site-level clear-sky GHI reduction metrics from one NSRDB CSV response."""
rows = list(csv.reader(csv_text.splitlines()))
header_index = find_data_header(rows)
if header_index is None:
@@ -187,7 +186,7 @@ def site_cloud_metrics(csv_text: str, min_clearsky_ghi: float) -> dict[str, floa
raise ValueError("No daylight rows with valid GHI and Clearsky GHI were found.")
avg_ratio = ratio_sum / daylight_rows
cloudiness = 1 - avg_ratio
clear_sky_ghi_reduction = 1 - avg_ratio
clear_or_probably_clear = daylight_cloud_type_counts.get("0", 0) + daylight_cloud_type_counts.get("1", 0)
cloudy_or_obscured = sum(
daylight_cloud_type_counts.get(code, 0)
@@ -206,7 +205,7 @@ def site_cloud_metrics(csv_text: str, min_clearsky_ghi: float) -> dict[str, floa
unknown_cloud_type = daylight_cloud_type_counts.get("10", 0) + daylight_cloud_type_counts.get("-15", 0)
return {
"cloudiness": cloudiness,
"clear_sky_ghi_reduction": clear_sky_ghi_reduction,
"avg_ratio": avg_ratio,
"all_rows": all_rows,
"daylight_rows": daylight_rows,
@@ -262,11 +261,11 @@ def summarize_archives(
if not site_metrics:
raise ValueError(f"No CSV files found in {', '.join(str(path) for path in paths)}")
cloudiness_values = [float(metrics["cloudiness"]) for metrics in site_metrics]
reduction_values = [float(metrics["clear_sky_ghi_reduction"]) for metrics in site_metrics]
row_counts = [int(metrics["all_rows"]) for metrics in site_metrics]
daylight_row_counts = [int(metrics["daylight_rows"]) for metrics in site_metrics]
weighted_cloudiness = weighted_metric(site_metrics, "cloudiness", site_lon_lats, county_geometry, cell_size_m)
weighted_reduction = weighted_metric(site_metrics, "clear_sky_ghi_reduction", site_lon_lats, county_geometry, cell_size_m)
weighted_avg_ratio = weighted_metric(site_metrics, "avg_ratio", site_lon_lats, county_geometry, cell_size_m)
return {
@@ -275,14 +274,14 @@ def summarize_archives(
"rows_per_site_max": max(row_counts),
"daylight_rows_per_site_min": min(daylight_row_counts),
"daylight_rows_per_site_max": max(daylight_row_counts),
"area_weighted_cloudiness": weighted_cloudiness["area_weighted_avg"],
"area_weighted_clear_sky_ghi_reduction": weighted_reduction["area_weighted_avg"],
"area_weighted_avg_ratio": weighted_avg_ratio["area_weighted_avg"],
"area_weighted_sites": weighted_cloudiness["area_weighted_sites"],
"weighted_cell_area_km2": weighted_cloudiness["weighted_cell_area_km2"],
"county_area_km2": weighted_cloudiness["county_area_km2"],
"site_cloudiness_min": min(cloudiness_values),
"site_cloudiness_max": max(cloudiness_values),
"site_cloudiness_stddev": statistics.pstdev(cloudiness_values) if len(cloudiness_values) > 1 else 0.0,
"area_weighted_sites": weighted_reduction["area_weighted_sites"],
"weighted_cell_area_km2": weighted_reduction["weighted_cell_area_km2"],
"county_area_km2": weighted_reduction["county_area_km2"],
"site_clear_sky_ghi_reduction_min": min(reduction_values),
"site_clear_sky_ghi_reduction_max": max(reduction_values),
"site_clear_sky_ghi_reduction_stddev": statistics.pstdev(reduction_values) if len(reduction_values) > 1 else 0.0,
"clear_or_probably_clear_pct": weighted_metric(
site_metrics, "clear_or_probably_clear_pct", site_lon_lats, county_geometry, cell_size_m
)["area_weighted_avg"],
@@ -343,20 +342,20 @@ def build_summary_row(
if str(row.get("site_count", "")).strip().isdigit()
]
archive_zip = ";".join(str(path) for path in archive_paths)
area_weighted_cloudiness = float(polygon_summary["area_weighted_cloudiness"])
representative_point_cloudiness = (
float(representative_point_row["cloudinessIndexPct"])
if representative_point_row.get("cloudinessIndexPct")
area_weighted_clear_sky_ghi_reduction = float(polygon_summary["area_weighted_clear_sky_ghi_reduction"])
representative_point_reduction = (
float(representative_point_row["clearSkyGhiReductionIndex"])
if representative_point_row.get("clearSkyGhiReductionIndex")
else None
)
weighted_minus_representative_point = (
area_weighted_cloudiness - representative_point_cloudiness
if representative_point_cloudiness is not None
area_weighted_clear_sky_ghi_reduction - representative_point_reduction
if representative_point_reduction is not None
else None
)
weighted_pct_diff_representative_point = (
weighted_minus_representative_point / representative_point_cloudiness * 100
if representative_point_cloudiness not in (None, 0)
weighted_minus_representative_point / representative_point_reduction * 100
if representative_point_reduction not in (None, 0)
else None
)
@@ -371,15 +370,15 @@ def build_summary_row(
"rows_per_site_max": str(polygon_summary["rows_per_site_max"]),
"daylight_rows_per_site_min": str(polygon_summary["daylight_rows_per_site_min"]),
"daylight_rows_per_site_max": str(polygon_summary["daylight_rows_per_site_max"]),
"cloudinessIndexPct": format_float(area_weighted_cloudiness, 4),
"areaWeightedCloudinessIndexPct": format_float(area_weighted_cloudiness, 4),
"clearSkyGhiReductionIndex": format_float(area_weighted_clear_sky_ghi_reduction, 4),
"areaWeightedClearSkyGhiReductionIndex": format_float(area_weighted_clear_sky_ghi_reduction, 4),
"areaWeightedAvgObservedToClearskyRatio": format_float(float(polygon_summary["area_weighted_avg_ratio"]), 4),
"areaWeightedSites": str(polygon_summary["area_weighted_sites"]),
"weightedCellAreaKm2": format_float(float(polygon_summary["weighted_cell_area_km2"]), 1),
"countyAreaKm2": format_float(float(polygon_summary["county_area_km2"]), 1),
"site_cloudiness_min": format_float(float(polygon_summary["site_cloudiness_min"]), 4),
"site_cloudiness_max": format_float(float(polygon_summary["site_cloudiness_max"]), 4),
"site_cloudiness_stddev": format_float(float(polygon_summary["site_cloudiness_stddev"]), 4),
"site_clear_sky_ghi_reduction_min": format_float(float(polygon_summary["site_clear_sky_ghi_reduction_min"]), 4),
"site_clear_sky_ghi_reduction_max": format_float(float(polygon_summary["site_clear_sky_ghi_reduction_max"]), 4),
"site_clear_sky_ghi_reduction_stddev": format_float(float(polygon_summary["site_clear_sky_ghi_reduction_stddev"]), 4),
"clearOrProbablyClearPct": format_float(float(polygon_summary["clear_or_probably_clear_pct"]), 2),
"cloudyOrObscuredPct": format_float(float(polygon_summary["cloudy_or_obscured_pct"]), 2),
"fogPct": format_float(float(polygon_summary["fog_pct"]), 2),
@@ -387,7 +386,7 @@ def build_summary_row(
"iceCloudPct": format_float(float(polygon_summary["ice_cloud_pct"]), 2),
"cirrusPct": format_float(float(polygon_summary["cirrus_pct"]), 2),
"unknownCloudTypePct": format_float(float(polygon_summary["unknown_cloud_type_pct"]), 2),
"representativePointCloudinessIndexPct": format_float(representative_point_cloudiness, 4),
"representativePointClearSkyGhiReductionIndex": format_float(representative_point_reduction, 4),
"areaWeightedMinusRepresentativePoint": format_float(weighted_minus_representative_point, 4),
"areaWeightedPctDiffFromRepresentativePoint": format_float(weighted_pct_diff_representative_point),
}
@@ -465,8 +464,8 @@ def run(args: argparse.Namespace) -> None:
)
print(
f"[{index}/{total}] {county_fips}: "
f"sites={row['polygon_sites']}, weighted={row['areaWeightedCloudinessIndexPct']}, "
f"representative_point={row['representativePointCloudinessIndexPct'] or 'n/a'}"
f"sites={row['polygon_sites']}, weighted={row['areaWeightedClearSkyGhiReductionIndex']}, "
f"representative_point={row['representativePointClearSkyGhiReductionIndex'] or 'n/a'}"
f"{tile_note}{empty_note}"
)
@@ -500,7 +499,7 @@ def parse_args() -> argparse.Namespace:
"--min-clearsky-ghi",
type=float,
default=DEFAULT_MIN_CLEARSKY_GHI,
help="Minimum Clearsky GHI W/m2 for daylight cloudiness ratio rows.",
help="Minimum Clearsky GHI W/m2 for daylight GHI ratio rows.",
)
parser.add_argument(
"--reuse-existing-output",