Refine climate metrics and data pipeline

This commit is contained in:
2026-07-24 22:55:14 -04:00
parent 016f7386d8
commit 4e2d878e5b
32 changed files with 3557 additions and 3347 deletions
@@ -4,15 +4,15 @@ Summarize downloaded NSRDB county polygon cloud archives.
Each downloaded polygon archive contains one CSV per NSRDB grid site that
intersects the county polygon or tile. This script computes area-weighted
county-level cloudiness metrics across those site CSVs, combining multiple tile
county-level clear-sky GHI reduction metrics across those site CSVs, combining multiple tile
archives back into one county summary when present.
Primary metric:
cloudinessIndexPct = 1 - mean(clamped(GHI / Clearsky GHI, 0, 1))
clearSkyGhiReductionIndex = 1 - mean(clamped(GHI / Clearsky GHI, 0, 1))
Despite the legacy "Pct" field name, the primary index is stored on a 0-1 scale.
Cloud Type bucket fields are stored as percentages.
The primary index is stored on a 0-1 scale. Cloud Type bucket fields are stored
as percentages.
"""
from __future__ import annotations
@@ -26,9 +26,9 @@ import zipfile
from pathlib import Path
from summarize_nsrdb_county_polygon_archives import (
area_weighted_average,
archive_groups_by_county,
archive_signature,
area_weighted_average,
county_fips_from_archive,
existing_archive_signature,
extract_site_lon_lat,
@@ -38,7 +38,6 @@ from summarize_nsrdb_county_polygon_archives import (
read_lookup,
)
DEFAULT_ARCHIVE_DIR = Path("data/nrel/polygon_cloud_archives")
DEFAULT_COUNTIES_GEOJSON = Path("data/geojson-counties-fips.json")
DEFAULT_REQUESTS_CSV = Path("data/nrel/county_polygon_cloud_request_manifest.csv")
@@ -59,15 +58,15 @@ SUMMARY_FIELDS = [
"rows_per_site_max",
"daylight_rows_per_site_min",
"daylight_rows_per_site_max",
"cloudinessIndexPct",
"areaWeightedCloudinessIndexPct",
"clearSkyGhiReductionIndex",
"areaWeightedClearSkyGhiReductionIndex",
"areaWeightedAvgObservedToClearskyRatio",
"areaWeightedSites",
"weightedCellAreaKm2",
"countyAreaKm2",
"site_cloudiness_min",
"site_cloudiness_max",
"site_cloudiness_stddev",
"site_clear_sky_ghi_reduction_min",
"site_clear_sky_ghi_reduction_max",
"site_clear_sky_ghi_reduction_stddev",
"clearOrProbablyClearPct",
"cloudyOrObscuredPct",
"fogPct",
@@ -75,7 +74,7 @@ SUMMARY_FIELDS = [
"iceCloudPct",
"cirrusPct",
"unknownCloudTypePct",
"representativePointCloudinessIndexPct",
"representativePointClearSkyGhiReductionIndex",
"areaWeightedMinusRepresentativePoint",
"areaWeightedPctDiffFromRepresentativePoint",
]
@@ -137,7 +136,7 @@ def pct(count: int, total: int) -> float:
def site_cloud_metrics(csv_text: str, min_clearsky_ghi: float) -> dict[str, float | int]:
"""Return site-level cloudiness metrics from one NSRDB CSV response."""
"""Return site-level clear-sky GHI reduction metrics from one NSRDB CSV response."""
rows = list(csv.reader(csv_text.splitlines()))
header_index = find_data_header(rows)
if header_index is None:
@@ -187,7 +186,7 @@ def site_cloud_metrics(csv_text: str, min_clearsky_ghi: float) -> dict[str, floa
raise ValueError("No daylight rows with valid GHI and Clearsky GHI were found.")
avg_ratio = ratio_sum / daylight_rows
cloudiness = 1 - avg_ratio
clear_sky_ghi_reduction = 1 - avg_ratio
clear_or_probably_clear = daylight_cloud_type_counts.get("0", 0) + daylight_cloud_type_counts.get("1", 0)
cloudy_or_obscured = sum(
daylight_cloud_type_counts.get(code, 0)
@@ -206,7 +205,7 @@ def site_cloud_metrics(csv_text: str, min_clearsky_ghi: float) -> dict[str, floa
unknown_cloud_type = daylight_cloud_type_counts.get("10", 0) + daylight_cloud_type_counts.get("-15", 0)
return {
"cloudiness": cloudiness,
"clear_sky_ghi_reduction": clear_sky_ghi_reduction,
"avg_ratio": avg_ratio,
"all_rows": all_rows,
"daylight_rows": daylight_rows,
@@ -262,11 +261,11 @@ def summarize_archives(
if not site_metrics:
raise ValueError(f"No CSV files found in {', '.join(str(path) for path in paths)}")
cloudiness_values = [float(metrics["cloudiness"]) for metrics in site_metrics]
reduction_values = [float(metrics["clear_sky_ghi_reduction"]) for metrics in site_metrics]
row_counts = [int(metrics["all_rows"]) for metrics in site_metrics]
daylight_row_counts = [int(metrics["daylight_rows"]) for metrics in site_metrics]
weighted_cloudiness = weighted_metric(site_metrics, "cloudiness", site_lon_lats, county_geometry, cell_size_m)
weighted_reduction = weighted_metric(site_metrics, "clear_sky_ghi_reduction", site_lon_lats, county_geometry, cell_size_m)
weighted_avg_ratio = weighted_metric(site_metrics, "avg_ratio", site_lon_lats, county_geometry, cell_size_m)
return {
@@ -275,14 +274,14 @@ def summarize_archives(
"rows_per_site_max": max(row_counts),
"daylight_rows_per_site_min": min(daylight_row_counts),
"daylight_rows_per_site_max": max(daylight_row_counts),
"area_weighted_cloudiness": weighted_cloudiness["area_weighted_avg"],
"area_weighted_clear_sky_ghi_reduction": weighted_reduction["area_weighted_avg"],
"area_weighted_avg_ratio": weighted_avg_ratio["area_weighted_avg"],
"area_weighted_sites": weighted_cloudiness["area_weighted_sites"],
"weighted_cell_area_km2": weighted_cloudiness["weighted_cell_area_km2"],
"county_area_km2": weighted_cloudiness["county_area_km2"],
"site_cloudiness_min": min(cloudiness_values),
"site_cloudiness_max": max(cloudiness_values),
"site_cloudiness_stddev": statistics.pstdev(cloudiness_values) if len(cloudiness_values) > 1 else 0.0,
"area_weighted_sites": weighted_reduction["area_weighted_sites"],
"weighted_cell_area_km2": weighted_reduction["weighted_cell_area_km2"],
"county_area_km2": weighted_reduction["county_area_km2"],
"site_clear_sky_ghi_reduction_min": min(reduction_values),
"site_clear_sky_ghi_reduction_max": max(reduction_values),
"site_clear_sky_ghi_reduction_stddev": statistics.pstdev(reduction_values) if len(reduction_values) > 1 else 0.0,
"clear_or_probably_clear_pct": weighted_metric(
site_metrics, "clear_or_probably_clear_pct", site_lon_lats, county_geometry, cell_size_m
)["area_weighted_avg"],
@@ -343,20 +342,20 @@ def build_summary_row(
if str(row.get("site_count", "")).strip().isdigit()
]
archive_zip = ";".join(str(path) for path in archive_paths)
area_weighted_cloudiness = float(polygon_summary["area_weighted_cloudiness"])
representative_point_cloudiness = (
float(representative_point_row["cloudinessIndexPct"])
if representative_point_row.get("cloudinessIndexPct")
area_weighted_clear_sky_ghi_reduction = float(polygon_summary["area_weighted_clear_sky_ghi_reduction"])
representative_point_reduction = (
float(representative_point_row["clearSkyGhiReductionIndex"])
if representative_point_row.get("clearSkyGhiReductionIndex")
else None
)
weighted_minus_representative_point = (
area_weighted_cloudiness - representative_point_cloudiness
if representative_point_cloudiness is not None
area_weighted_clear_sky_ghi_reduction - representative_point_reduction
if representative_point_reduction is not None
else None
)
weighted_pct_diff_representative_point = (
weighted_minus_representative_point / representative_point_cloudiness * 100
if representative_point_cloudiness not in (None, 0)
weighted_minus_representative_point / representative_point_reduction * 100
if representative_point_reduction not in (None, 0)
else None
)
@@ -371,15 +370,15 @@ def build_summary_row(
"rows_per_site_max": str(polygon_summary["rows_per_site_max"]),
"daylight_rows_per_site_min": str(polygon_summary["daylight_rows_per_site_min"]),
"daylight_rows_per_site_max": str(polygon_summary["daylight_rows_per_site_max"]),
"cloudinessIndexPct": format_float(area_weighted_cloudiness, 4),
"areaWeightedCloudinessIndexPct": format_float(area_weighted_cloudiness, 4),
"clearSkyGhiReductionIndex": format_float(area_weighted_clear_sky_ghi_reduction, 4),
"areaWeightedClearSkyGhiReductionIndex": format_float(area_weighted_clear_sky_ghi_reduction, 4),
"areaWeightedAvgObservedToClearskyRatio": format_float(float(polygon_summary["area_weighted_avg_ratio"]), 4),
"areaWeightedSites": str(polygon_summary["area_weighted_sites"]),
"weightedCellAreaKm2": format_float(float(polygon_summary["weighted_cell_area_km2"]), 1),
"countyAreaKm2": format_float(float(polygon_summary["county_area_km2"]), 1),
"site_cloudiness_min": format_float(float(polygon_summary["site_cloudiness_min"]), 4),
"site_cloudiness_max": format_float(float(polygon_summary["site_cloudiness_max"]), 4),
"site_cloudiness_stddev": format_float(float(polygon_summary["site_cloudiness_stddev"]), 4),
"site_clear_sky_ghi_reduction_min": format_float(float(polygon_summary["site_clear_sky_ghi_reduction_min"]), 4),
"site_clear_sky_ghi_reduction_max": format_float(float(polygon_summary["site_clear_sky_ghi_reduction_max"]), 4),
"site_clear_sky_ghi_reduction_stddev": format_float(float(polygon_summary["site_clear_sky_ghi_reduction_stddev"]), 4),
"clearOrProbablyClearPct": format_float(float(polygon_summary["clear_or_probably_clear_pct"]), 2),
"cloudyOrObscuredPct": format_float(float(polygon_summary["cloudy_or_obscured_pct"]), 2),
"fogPct": format_float(float(polygon_summary["fog_pct"]), 2),
@@ -387,7 +386,7 @@ def build_summary_row(
"iceCloudPct": format_float(float(polygon_summary["ice_cloud_pct"]), 2),
"cirrusPct": format_float(float(polygon_summary["cirrus_pct"]), 2),
"unknownCloudTypePct": format_float(float(polygon_summary["unknown_cloud_type_pct"]), 2),
"representativePointCloudinessIndexPct": format_float(representative_point_cloudiness, 4),
"representativePointClearSkyGhiReductionIndex": format_float(representative_point_reduction, 4),
"areaWeightedMinusRepresentativePoint": format_float(weighted_minus_representative_point, 4),
"areaWeightedPctDiffFromRepresentativePoint": format_float(weighted_pct_diff_representative_point),
}
@@ -465,8 +464,8 @@ def run(args: argparse.Namespace) -> None:
)
print(
f"[{index}/{total}] {county_fips}: "
f"sites={row['polygon_sites']}, weighted={row['areaWeightedCloudinessIndexPct']}, "
f"representative_point={row['representativePointCloudinessIndexPct'] or 'n/a'}"
f"sites={row['polygon_sites']}, weighted={row['areaWeightedClearSkyGhiReductionIndex']}, "
f"representative_point={row['representativePointClearSkyGhiReductionIndex'] or 'n/a'}"
f"{tile_note}{empty_note}"
)
@@ -500,7 +499,7 @@ def parse_args() -> argparse.Namespace:
"--min-clearsky-ghi",
type=float,
default=DEFAULT_MIN_CLEARSKY_GHI,
help="Minimum Clearsky GHI W/m2 for daylight cloudiness ratio rows.",
help="Minimum Clearsky GHI W/m2 for daylight GHI ratio rows.",
)
parser.add_argument(
"--reuse-existing-output",