Initial Climate Mood Analysis project

This commit is contained in:
Justin Fisher
2026-06-11 14:50:33 -04:00
commit 18e099ce84
34 changed files with 18807 additions and 0 deletions
@@ -0,0 +1,290 @@
#!/usr/bin/env python3
"""
Apply NOAA daily extreme-day and polygon solar metrics to the app CSV.
The browser app already reads the `extremeDays` column, so this script updates
that column to the average locally extreme days/year while preserving the older
proxy value in `oldExtremeDays`. It also writes an absolute daily heat/cold
bucket metric from the same NOAA nClimGrid-Daily county data.
When an NSRDB polygon archive summary is available, this also replaces the
older representative-point solar GHI value with the county polygon value. Rows
without a polygon summary keep the representative-point fallback when present.
"""
from __future__ import annotations
import argparse
import csv
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_COMPARISON = REPO_ROOT / "data" / "noaa" / "county_locally_extreme_days_comparison.csv"
DEFAULT_POLYGON_SOLAR_GHI = REPO_ROOT / "data" / "nrel" / "county_polygon_ghi_summary.csv"
DEFAULT_REPRESENTATIVE_POINT_SOLAR_GHI = REPO_ROOT / "data" / "nrel" / "county_representative_point_ghi_summary.csv"
BASE_FIELDS = [
"countyFips",
"countyName",
"state",
"koppenZone",
"avgTempF",
"annualPrecipIn",
"seasonalityIndex",
"wettestPrecipMonth",
"driestPrecipMonth",
"extremeDays",
"locallyExtremeDays",
"locallyExtremeHotDays",
"locallyExtremeColdDays",
"absoluteExtremeDays",
"oldExtremeDays",
"locallyExtremeAnalysisYears",
"locallyExtremeSourceFips",
"locallyExtremeFipsAdjustment",
"avgSolarGhiKwhM2Day",
"source",
]
LEGACY_FIELDS = {
"absoluteExtremeHeatDays",
"absoluteExtremeColdDays",
}
SOLAR_SOURCE_TAGS = {
"solar-ghi-centroid-preview",
"solar-ghi-representative-point",
"solar-ghi-polygon-area-weighted",
"solar-ghi-polygon-archive-area-weighted",
"no-solar-ghi-source",
"missing-solar-ghi",
}
def normalize_fips(value: object) -> str:
"""Return a 5-digit county FIPS string from mixed text or numeric input."""
digits = "".join(ch for ch in str(value).strip() if ch.isdigit())
return digits.zfill(5)[-5:] if digits else ""
def read_csv_rows(path: Path) -> tuple[list[str], list[dict]]:
"""Read a CSV while preserving the source field order."""
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
reader = csv.DictReader(csv_file)
if reader.fieldnames is None:
raise ValueError(f"{path} has no CSV header.")
return list(reader.fieldnames), list(reader)
def load_locally_extreme_lookup(comparison_csv: Path) -> dict[str, dict]:
"""Map current and supported old app FIPS codes to locally extreme rows."""
_, rows = read_csv_rows(comparison_csv)
lookup: dict[str, dict] = {}
for row in rows:
if row.get("hasLocallyExtremeData") != "yes":
continue
current_fips = normalize_fips(row.get("countyFips", ""))
if current_fips:
lookup[current_fips] = row
for old_fips in (row.get("oldSourceFips") or "").split(";"):
normalized_old_fips = normalize_fips(old_fips)
if normalized_old_fips:
lookup[normalized_old_fips] = row
return lookup
def load_solar_ghi_lookup(solar_ghi_csv: Path) -> dict[str, str]:
"""Map county FIPS codes to average daily GHI values from a county CSV."""
if not solar_ghi_csv.exists():
return {}
fieldnames, rows = read_csv_rows(solar_ghi_csv)
if "avgSolarGhiKwhM2Day" not in fieldnames:
raise ValueError(f"{solar_ghi_csv} must include avgSolarGhiKwhM2Day.")
if "county_fips" in fieldnames:
fips_field = "county_fips"
elif "countyFips" in fieldnames:
fips_field = "countyFips"
else:
raise ValueError(f"{solar_ghi_csv} must include county_fips or countyFips.")
lookup: dict[str, str] = {}
for row in rows:
county_fips = normalize_fips(row.get(fips_field, ""))
raw_value = (row.get("avgSolarGhiKwhM2Day") or "").strip()
if not county_fips or not raw_value:
continue
try:
float(raw_value)
except ValueError as error:
raise ValueError(
f"Invalid avgSolarGhiKwhM2Day value for county {county_fips}: {raw_value}"
) from error
lookup[county_fips] = raw_value
return lookup
def append_source_tag(source: str, has_local_value: bool) -> str:
"""Add the locally extreme provenance tags without duplicating old tags."""
old_source = source.strip() or "unknown-source"
old_source = old_source.replace(" + absoluteExtremeHeatDays=tmax-gte-95f", "")
old_source = old_source.replace(" + absoluteExtremeColdDays=tmin-lte-0f", "")
if not has_local_value:
return f"{old_source} + missing-noaa-locally-extreme-days"
has_local_tag = "extremeDays=locally-extreme-noaa-nclimgrid-daily-1991-2025" in old_source
has_absolute_tag = "absoluteExtremeDays=tmax-gte-95f-or-tmin-lte-0f" in old_source
if has_local_tag and has_absolute_tag:
return old_source
if has_local_tag:
return f"{old_source} + absoluteExtremeDays=tmax-gte-95f-or-tmin-lte-0f"
return (
f"{old_source} + extremeDays=locally-extreme-noaa-nclimgrid-daily-1991-2025 "
"+ absoluteExtremeDays=tmax-gte-95f-or-tmin-lte-0f "
"+ oldExtremeDays=previous-monthly-proxy"
)
def replace_solar_source_tag(source: str, solar_source_tag: str) -> str:
"""Replace any previous solar provenance tag with the active one."""
parts = [part.strip() for part in (source.strip() or "unknown-source").split("+")]
kept_parts = [
part
for part in parts
if part and part not in SOLAR_SOURCE_TAGS
]
if solar_source_tag not in kept_parts:
kept_parts.append(solar_source_tag)
return " + ".join(kept_parts)
def apply_locally_extreme_metric(
*,
climate_data: Path,
comparison_csv: Path,
polygon_solar_ghi_csv: Path,
representative_point_solar_ghi_csv: Path,
out: Path,
) -> None:
"""Replace app extreme-day fields and solar GHI with improved metrics."""
original_fields, climate_rows = read_csv_rows(climate_data)
local_lookup = load_locally_extreme_lookup(comparison_csv)
polygon_solar_lookup = load_solar_ghi_lookup(polygon_solar_ghi_csv)
representative_solar_lookup = load_solar_ghi_lookup(representative_point_solar_ghi_csv)
extra_fields = [
field
for field in original_fields
if field not in BASE_FIELDS and field not in LEGACY_FIELDS
]
fieldnames = BASE_FIELDS + extra_fields
updated_count = 0
missing_count = 0
polygon_solar_count = 0
representative_solar_count = 0
missing_solar_count = 0
for row in climate_rows:
county_fips = normalize_fips(row.get("countyFips", ""))
local_row = local_lookup.get(county_fips)
old_extreme_days = (row.get("oldExtremeDays") or row.get("extremeDays") or "").strip()
row["oldExtremeDays"] = old_extreme_days
if local_row:
row["extremeDays"] = local_row.get("avgLocallyExtremeDays", "")
row["locallyExtremeDays"] = local_row.get("avgLocallyExtremeDays", "")
row["locallyExtremeHotDays"] = local_row.get("avgHotExtremeDays", "")
row["locallyExtremeColdDays"] = local_row.get("avgColdExtremeDays", "")
row["absoluteExtremeDays"] = local_row.get("avgAbsoluteExtremeDays", "")
row["locallyExtremeAnalysisYears"] = local_row.get("analysisYears", "")
row["locallyExtremeSourceFips"] = local_row.get("countyFips", "")
row["locallyExtremeFipsAdjustment"] = local_row.get("fipsAdjustment", "")
row["source"] = append_source_tag(row.get("source", ""), has_local_value=True)
updated_count += 1
else:
row["extremeDays"] = ""
row["locallyExtremeDays"] = ""
row["locallyExtremeHotDays"] = ""
row["locallyExtremeColdDays"] = ""
row["absoluteExtremeDays"] = ""
row["locallyExtremeAnalysisYears"] = ""
row["locallyExtremeSourceFips"] = ""
row["locallyExtremeFipsAdjustment"] = ""
row["source"] = append_source_tag(row.get("source", ""), has_local_value=False)
missing_count += 1
polygon_solar_value = polygon_solar_lookup.get(county_fips)
representative_solar_value = representative_solar_lookup.get(county_fips)
current_solar_value = (row.get("avgSolarGhiKwhM2Day") or "").strip()
if polygon_solar_value:
row["avgSolarGhiKwhM2Day"] = polygon_solar_value
row["source"] = replace_solar_source_tag(
row.get("source", ""),
"solar-ghi-polygon-archive-area-weighted",
)
polygon_solar_count += 1
elif representative_solar_value or current_solar_value:
row["avgSolarGhiKwhM2Day"] = representative_solar_value or current_solar_value
row["source"] = replace_solar_source_tag(
row.get("source", ""),
"solar-ghi-representative-point",
)
representative_solar_count += 1
else:
row["avgSolarGhiKwhM2Day"] = ""
row["source"] = replace_solar_source_tag(row.get("source", ""), "missing-solar-ghi")
missing_solar_count += 1
with out.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=fieldnames, extrasaction="ignore")
writer.writeheader()
writer.writerows(climate_rows)
print(f"Wrote {len(climate_rows)} rows to {out}")
print(f"Updated active extremeDays from locally extreme data for {updated_count} rows.")
print(f"Left active extremeDays blank for {missing_count} rows without locally extreme data.")
print(f"Updated avgSolarGhiKwhM2Day from polygon archives for {polygon_solar_count} rows.")
print(f"Kept representative-point solar fallback for {representative_solar_count} rows.")
print(f"Left avgSolarGhiKwhM2Day blank for {missing_solar_count} rows without solar data.")
def parse_args() -> argparse.Namespace:
"""Parse command-line paths for the climate-data update."""
parser = argparse.ArgumentParser(
description="Apply locally extreme days/year and polygon solar GHI to climate-data.csv."
)
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA)
parser.add_argument("--comparison-csv", type=Path, default=DEFAULT_COMPARISON)
parser.add_argument("--polygon-solar-ghi-csv", type=Path, default=DEFAULT_POLYGON_SOLAR_GHI)
parser.add_argument(
"--representative-point-solar-ghi-csv",
type=Path,
default=DEFAULT_REPRESENTATIVE_POINT_SOLAR_GHI,
)
parser.add_argument("--out", type=Path, default=DEFAULT_CLIMATE_DATA)
return parser.parse_args()
def main() -> None:
"""Run the CSV update from parsed command-line arguments."""
args = parse_args()
apply_locally_extreme_metric(
climate_data=args.climate_data,
comparison_csv=args.comparison_csv,
polygon_solar_ghi_csv=args.polygon_solar_ghi_csv,
representative_point_solar_ghi_csv=args.representative_point_solar_ghi_csv,
out=args.out,
)
if __name__ == "__main__":
main()