Refine climate metrics and data pipeline

This commit is contained in:
2026-07-24 22:55:14 -04:00
parent 016f7386d8
commit 4e2d878e5b
32 changed files with 3557 additions and 3347 deletions
@@ -1,9 +1,9 @@
#!/usr/bin/env python3
"""
Merge NSRDB cloudiness metrics into climate-data.csv.
Merge NSRDB clear-sky GHI reduction metrics into climate-data.csv.
Adds:
- cloudinessIndexPct
- clearSkyGhiReductionIndex
Polygon area-weighted values are used first when available; representative-point
values remain the fallback.
@@ -15,33 +15,33 @@ import argparse
import csv
from pathlib import Path
DEFAULT_CLIMATE_DATA = Path("data/climate-data.csv")
DEFAULT_POLYGON_CLOUD_SUMMARY = Path("data/nrel/county_polygon_cloud_summary.csv")
DEFAULT_REPRESENTATIVE_POINT_CLOUD_SUMMARY = Path("data/nrel/county_representative_point_cloud_summary.csv")
METRIC_FIELD = "cloudinessIndexPct"
POLYGON_SOURCE_TAG = "nsrdb-polygon-area-weighted-cloudiness-tmy"
REPRESENTATIVE_POINT_SOURCE_TAG = "nsrdb-representative-point-cloudiness-tmy"
METRIC_FIELD = "clearSkyGhiReductionIndex"
POLYGON_METRIC_FIELD = "areaWeightedClearSkyGhiReductionIndex"
POLYGON_SOURCE_TAG = "nsrdb-polygon-area-weighted-clear-sky-ghi-reduction-tmy"
REPRESENTATIVE_POINT_SOURCE_TAG = "nsrdb-representative-point-clear-sky-ghi-reduction-tmy"
CLOUD_SOURCE_TAGS = {
POLYGON_SOURCE_TAG,
REPRESENTATIVE_POINT_SOURCE_TAG,
}
def load_cloud_values(path: Path) -> dict[str, str]:
"""Read cloudiness values keyed by county FIPS."""
def load_cloud_values(path: Path, metric_field: str = METRIC_FIELD) -> dict[str, str]:
"""Read clear-sky GHI reduction values keyed by county FIPS."""
values: dict[str, str] = {}
with path.open("r", encoding="utf-8-sig", newline="") as handle:
reader = csv.DictReader(handle)
fieldnames = reader.fieldnames or []
required_fields = {"county_fips", METRIC_FIELD}
required_fields = {"county_fips", metric_field}
missing_fields = sorted(required_fields - set(fieldnames))
if missing_fields:
raise ValueError(f"{path} is missing fields: {', '.join(missing_fields)}")
for row in reader:
county_fips = (row.get("county_fips") or "").strip().zfill(5)
value = (row.get(METRIC_FIELD) or "").strip()
value = (row.get(metric_field) or "").strip()
if county_fips and value:
values[county_fips] = value
return values
@@ -75,13 +75,13 @@ def merge_metric(
polygon_cloud_summary: Path,
representative_point_cloud_summary: Path,
) -> tuple[int, int, int, int]:
"""Merge cloudiness values into the app climate CSV."""
polygon_cloudiness_by_fips = (
load_cloud_values(polygon_cloud_summary)
"""Merge clear-sky GHI reduction values into the app climate CSV."""
polygon_reduction_by_fips = (
load_cloud_values(polygon_cloud_summary, POLYGON_METRIC_FIELD)
if polygon_cloud_summary.exists()
else {}
)
representative_cloudiness_by_fips = load_cloud_values(representative_point_cloud_summary)
representative_reduction_by_fips = load_cloud_values(representative_point_cloud_summary)
with climate_data.open("r", encoding="utf-8", newline="") as handle:
reader = csv.DictReader(handle)
@@ -91,15 +91,15 @@ def merge_metric(
if "countyFips" not in fieldnames:
raise ValueError(f"{climate_data} is missing countyFips")
ensure_field_after(fieldnames, METRIC_FIELD, "avgSolarGhiKwhM2Day")
ensure_field_after(fieldnames, METRIC_FIELD, "meanDailyGlobalHorizontalRadiationKwhM2Day")
polygon_count = 0
representative_count = 0
missing_count = 0
for row in rows:
county_fips = (row.get("countyFips") or "").strip().zfill(5)
polygon_value = polygon_cloudiness_by_fips.get(county_fips, "")
representative_value = representative_cloudiness_by_fips.get(county_fips, "")
polygon_value = polygon_reduction_by_fips.get(county_fips, "")
representative_value = representative_reduction_by_fips.get(county_fips, "")
value = polygon_value or representative_value
row[METRIC_FIELD] = value
if polygon_value:
@@ -120,7 +120,7 @@ def merge_metric(
def main() -> int:
parser = argparse.ArgumentParser(description="Merge NSRDB cloudiness metric into climate-data.csv.")
parser = argparse.ArgumentParser(description="Merge NSRDB clear-sky GHI reduction metric into climate-data.csv.")
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA)
parser.add_argument("--polygon-cloud-summary", type=Path, default=DEFAULT_POLYGON_CLOUD_SUMMARY)
parser.add_argument(