Initial Climate Mood Analysis project

This commit is contained in:
Justin Fisher
2026-06-11 14:50:33 -04:00
commit 18e099ce84
34 changed files with 18807 additions and 0 deletions
@@ -0,0 +1,159 @@
#!/usr/bin/env python3
"""
Apply county diurnal temperature range to the app CSV.
This joins only the final app-facing metric into climate-data.csv. Detailed ETL
fields stay in data/noaa/county_diurnal_temperature_range.csv.
"""
from __future__ import annotations
import argparse
import csv
from pathlib import Path
from build_county_locally_extreme_data import OLD_APP_FIPS_TO_CURRENT_FIPS
REPO_ROOT = Path(__file__).resolve().parents[1]
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
DEFAULT_DTR_CSV = REPO_ROOT / "data" / "noaa" / "county_diurnal_temperature_range.csv"
METRIC_FIELD = "avgDiurnalTempRangeF"
SOURCE_TAG_PREFIX = "diurnalTempRange="
SOURCE_TAG = "diurnalTempRange=noaa-nclimgrid-daily-1991-2020"
MISSING_SOURCE_TAG = "missing-noaa-diurnal-temperature-range"
def normalize_fips(value: object) -> str:
"""Return a 5-digit county FIPS string from mixed text or numeric input."""
digits = "".join(ch for ch in str(value).strip() if ch.isdigit())
return digits.zfill(5)[-5:] if digits else ""
def read_csv_rows(path: Path) -> tuple[list[str], list[dict]]:
"""Read a CSV while preserving the source field order."""
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
reader = csv.DictReader(csv_file)
if reader.fieldnames is None:
raise ValueError(f"{path} has no CSV header.")
return list(reader.fieldnames), list(reader)
def load_dtr_lookup(path: Path) -> dict[str, dict]:
"""Load DTR rows keyed by current county FIPS plus supported old app FIPS."""
fieldnames, rows = read_csv_rows(path)
required_fields = {"countyFips", METRIC_FIELD}
missing_fields = required_fields - set(fieldnames)
if missing_fields:
raise ValueError(f"{path} is missing required fields: {sorted(missing_fields)}")
current_lookup: dict[str, dict] = {}
for row in rows:
county_fips = normalize_fips(row.get("countyFips", ""))
metric_value = (row.get(METRIC_FIELD) or "").strip()
if not county_fips or not metric_value:
continue
float(metric_value)
current_lookup[county_fips] = row
lookup = dict(current_lookup)
for old_fips, (current_fips, reason) in OLD_APP_FIPS_TO_CURRENT_FIPS.items():
current_row = current_lookup.get(normalize_fips(current_fips))
if not current_row:
continue
proxy_row = dict(current_row)
proxy_row["fipsAdjustment"] = reason
lookup[normalize_fips(old_fips)] = proxy_row
return lookup
def fieldnames_with_metric(original_fields: list[str]) -> list[str]:
"""Insert the DTR metric near the other temperature field."""
fields = [field for field in original_fields if field != METRIC_FIELD]
if "avgTempF" in fields:
insert_at = fields.index("avgTempF") + 1
else:
insert_at = len(fields)
fields.insert(insert_at, METRIC_FIELD)
return fields
def replace_dtr_source_tag(source: str, has_value: bool) -> str:
"""Replace previous DTR provenance with the active one."""
parts = [
part.strip()
for part in (source.strip() or "unknown-source").split("+")
if part.strip()
]
kept_parts = [
part
for part in parts
if not part.startswith(SOURCE_TAG_PREFIX) and part != MISSING_SOURCE_TAG
]
kept_parts.append(SOURCE_TAG if has_value else MISSING_SOURCE_TAG)
return " + ".join(kept_parts)
def apply_diurnal_temperature_range(
*,
climate_data: Path,
dtr_csv: Path,
out: Path,
) -> None:
"""Join DTR values into climate-data.csv."""
original_fields, climate_rows = read_csv_rows(climate_data)
dtr_lookup = load_dtr_lookup(dtr_csv)
fieldnames = fieldnames_with_metric(original_fields)
updated_count = 0
missing_count = 0
adjusted_count = 0
for row in climate_rows:
county_fips = normalize_fips(row.get("countyFips", ""))
dtr_row = dtr_lookup.get(county_fips)
if dtr_row:
row[METRIC_FIELD] = dtr_row.get(METRIC_FIELD, "")
row["source"] = replace_dtr_source_tag(row.get("source", ""), has_value=True)
updated_count += 1
if dtr_row.get("fipsAdjustment"):
adjusted_count += 1
else:
row[METRIC_FIELD] = ""
row["source"] = replace_dtr_source_tag(row.get("source", ""), has_value=False)
missing_count += 1
with out.open("w", encoding="utf-8", newline="") as csv_file:
writer = csv.DictWriter(csv_file, fieldnames=fieldnames, extrasaction="ignore")
writer.writeheader()
writer.writerows(climate_rows)
print(f"Wrote {len(climate_rows)} rows to {out}")
print(f"Updated {METRIC_FIELD} for {updated_count} rows.")
print(f"Used supported county-vintage FIPS adjustments for {adjusted_count} rows.")
print(f"Left {METRIC_FIELD} blank for {missing_count} rows without DTR data.")
def parse_args() -> argparse.Namespace:
"""Parse command-line paths."""
parser = argparse.ArgumentParser(
description="Apply average diurnal temperature range to climate-data.csv."
)
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA)
parser.add_argument("--dtr-csv", type=Path, default=DEFAULT_DTR_CSV)
parser.add_argument("--out", type=Path, default=DEFAULT_CLIMATE_DATA)
return parser.parse_args()
def main() -> None:
"""Run the climate-data update."""
args = parse_args()
apply_diurnal_temperature_range(
climate_data=args.climate_data,
dtr_csv=args.dtr_csv,
out=args.out,
)
if __name__ == "__main__":
main()