Initial Climate Mood Analysis project
This commit is contained in:
@@ -0,0 +1,119 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Merge selected gridMET humidity and heat metrics into climate-data.csv.
|
||||
|
||||
Currently adds:
|
||||
- avgSummerSpecificHumidityGKg
|
||||
- humidHeatDays
|
||||
- humidHeatSourceFips
|
||||
- humidHeatFipsAdjustment
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
|
||||
DEFAULT_CLIMATE_DATA = Path("data/climate-data.csv")
|
||||
DEFAULT_GRIDMET_HUMIDITY = Path("data/gridmet/county_gridmet_humidity.csv")
|
||||
METRIC_FIELDS = [
|
||||
"avgSummerSpecificHumidityGKg",
|
||||
"humidHeatDays",
|
||||
"humidHeatSourceFips",
|
||||
"humidHeatFipsAdjustment",
|
||||
]
|
||||
SOURCE_TAGS_BY_FIELD = {
|
||||
"avgSummerSpecificHumidityGKg": "gridmet-specific-humidity-1991-2020",
|
||||
"humidHeatDays": "heat-index-noaa-tmax-gridmet-rmin-1991-2020",
|
||||
}
|
||||
|
||||
|
||||
def load_gridmet_values(path: Path) -> dict[str, dict[str, str]]:
|
||||
values: dict[str, dict[str, str]] = {}
|
||||
with path.open("r", encoding="utf-8", newline="") as handle:
|
||||
reader = csv.DictReader(handle)
|
||||
fieldnames = reader.fieldnames or []
|
||||
missing_fields = [field for field in METRIC_FIELDS if field not in fieldnames]
|
||||
if missing_fields:
|
||||
raise ValueError(f"{path} is missing fields: {', '.join(missing_fields)}")
|
||||
for row in reader:
|
||||
county_fips = (row.get("countyFips") or "").strip().zfill(5)
|
||||
if county_fips:
|
||||
values[county_fips] = {
|
||||
field: (row.get(field) or "").strip() for field in METRIC_FIELDS
|
||||
}
|
||||
return values
|
||||
|
||||
|
||||
def append_source_tag(source: str, tag: str) -> str:
|
||||
clean_source = source.strip()
|
||||
if tag in clean_source:
|
||||
return clean_source
|
||||
return f"{clean_source} + {tag}".strip(" +")
|
||||
|
||||
|
||||
def ensure_field_after(fieldnames: list[str], field: str, after_field: str | None = None) -> None:
|
||||
if field in fieldnames:
|
||||
return
|
||||
if after_field and after_field in fieldnames:
|
||||
fieldnames.insert(fieldnames.index(after_field) + 1, field)
|
||||
return
|
||||
insert_at = fieldnames.index("source") if "source" in fieldnames else len(fieldnames)
|
||||
fieldnames.insert(insert_at, field)
|
||||
|
||||
|
||||
def merge_metrics(climate_data: Path, humidity_csv: Path) -> tuple[int, dict[str, int]]:
|
||||
gridmet_by_fips = load_gridmet_values(humidity_csv)
|
||||
|
||||
with climate_data.open("r", encoding="utf-8", newline="") as handle:
|
||||
reader = csv.DictReader(handle)
|
||||
fieldnames = list(reader.fieldnames or [])
|
||||
rows = list(reader)
|
||||
|
||||
if "countyFips" not in fieldnames:
|
||||
raise ValueError(f"{climate_data} is missing countyFips")
|
||||
|
||||
ensure_field_after(fieldnames, "avgSummerSpecificHumidityGKg")
|
||||
ensure_field_after(fieldnames, "humidHeatDays", "avgSummerSpecificHumidityGKg")
|
||||
ensure_field_after(fieldnames, "humidHeatSourceFips", "humidHeatDays")
|
||||
ensure_field_after(fieldnames, "humidHeatFipsAdjustment", "humidHeatSourceFips")
|
||||
|
||||
missing_counts = {field: 0 for field in METRIC_FIELDS}
|
||||
for row in rows:
|
||||
county_fips = (row.get("countyFips") or "").strip().zfill(5)
|
||||
gridmet_values = gridmet_by_fips.get(county_fips, {})
|
||||
for field in METRIC_FIELDS:
|
||||
value = gridmet_values.get(field, "")
|
||||
row[field] = value
|
||||
if not value:
|
||||
missing_counts[field] += 1
|
||||
for field, tag in SOURCE_TAGS_BY_FIELD.items():
|
||||
if row.get(field):
|
||||
row["source"] = append_source_tag(row.get("source", ""), tag)
|
||||
|
||||
with climate_data.open("w", encoding="utf-8", newline="") as handle:
|
||||
writer = csv.DictWriter(handle, fieldnames=fieldnames)
|
||||
writer.writeheader()
|
||||
writer.writerows(rows)
|
||||
|
||||
return len(rows), missing_counts
|
||||
|
||||
|
||||
def main() -> int:
|
||||
parser = argparse.ArgumentParser(description="Merge gridMET humidity and Heat Index metrics into climate-data.csv.")
|
||||
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA)
|
||||
parser.add_argument("--humidity-csv", type=Path, default=DEFAULT_GRIDMET_HUMIDITY)
|
||||
args = parser.parse_args()
|
||||
|
||||
row_count, missing_counts = merge_metrics(args.climate_data, args.humidity_csv)
|
||||
print(f"Updated {args.climate_data} with {', '.join(METRIC_FIELDS)}")
|
||||
print(f"Rows: {row_count}")
|
||||
for field, missing_count in missing_counts.items():
|
||||
print(f"Missing {field}: {missing_count}")
|
||||
return 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
raise SystemExit(main())
|
||||
Reference in New Issue
Block a user