Initial Climate Mood Analysis project
This commit is contained in:
@@ -0,0 +1,159 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Apply county diurnal temperature range to the app CSV.
|
||||
|
||||
This joins only the final app-facing metric into climate-data.csv. Detailed ETL
|
||||
fields stay in data/noaa/county_diurnal_temperature_range.csv.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
from pathlib import Path
|
||||
|
||||
from build_county_locally_extreme_data import OLD_APP_FIPS_TO_CURRENT_FIPS
|
||||
|
||||
|
||||
REPO_ROOT = Path(__file__).resolve().parents[1]
|
||||
DEFAULT_CLIMATE_DATA = REPO_ROOT / "data" / "climate-data.csv"
|
||||
DEFAULT_DTR_CSV = REPO_ROOT / "data" / "noaa" / "county_diurnal_temperature_range.csv"
|
||||
METRIC_FIELD = "avgDiurnalTempRangeF"
|
||||
SOURCE_TAG_PREFIX = "diurnalTempRange="
|
||||
SOURCE_TAG = "diurnalTempRange=noaa-nclimgrid-daily-1991-2020"
|
||||
MISSING_SOURCE_TAG = "missing-noaa-diurnal-temperature-range"
|
||||
|
||||
|
||||
def normalize_fips(value: object) -> str:
|
||||
"""Return a 5-digit county FIPS string from mixed text or numeric input."""
|
||||
digits = "".join(ch for ch in str(value).strip() if ch.isdigit())
|
||||
return digits.zfill(5)[-5:] if digits else ""
|
||||
|
||||
|
||||
def read_csv_rows(path: Path) -> tuple[list[str], list[dict]]:
|
||||
"""Read a CSV while preserving the source field order."""
|
||||
with path.open("r", encoding="utf-8-sig", newline="") as csv_file:
|
||||
reader = csv.DictReader(csv_file)
|
||||
if reader.fieldnames is None:
|
||||
raise ValueError(f"{path} has no CSV header.")
|
||||
return list(reader.fieldnames), list(reader)
|
||||
|
||||
|
||||
def load_dtr_lookup(path: Path) -> dict[str, dict]:
|
||||
"""Load DTR rows keyed by current county FIPS plus supported old app FIPS."""
|
||||
fieldnames, rows = read_csv_rows(path)
|
||||
required_fields = {"countyFips", METRIC_FIELD}
|
||||
missing_fields = required_fields - set(fieldnames)
|
||||
if missing_fields:
|
||||
raise ValueError(f"{path} is missing required fields: {sorted(missing_fields)}")
|
||||
|
||||
current_lookup: dict[str, dict] = {}
|
||||
for row in rows:
|
||||
county_fips = normalize_fips(row.get("countyFips", ""))
|
||||
metric_value = (row.get(METRIC_FIELD) or "").strip()
|
||||
if not county_fips or not metric_value:
|
||||
continue
|
||||
float(metric_value)
|
||||
current_lookup[county_fips] = row
|
||||
|
||||
lookup = dict(current_lookup)
|
||||
for old_fips, (current_fips, reason) in OLD_APP_FIPS_TO_CURRENT_FIPS.items():
|
||||
current_row = current_lookup.get(normalize_fips(current_fips))
|
||||
if not current_row:
|
||||
continue
|
||||
proxy_row = dict(current_row)
|
||||
proxy_row["fipsAdjustment"] = reason
|
||||
lookup[normalize_fips(old_fips)] = proxy_row
|
||||
|
||||
return lookup
|
||||
|
||||
|
||||
def fieldnames_with_metric(original_fields: list[str]) -> list[str]:
|
||||
"""Insert the DTR metric near the other temperature field."""
|
||||
fields = [field for field in original_fields if field != METRIC_FIELD]
|
||||
if "avgTempF" in fields:
|
||||
insert_at = fields.index("avgTempF") + 1
|
||||
else:
|
||||
insert_at = len(fields)
|
||||
fields.insert(insert_at, METRIC_FIELD)
|
||||
return fields
|
||||
|
||||
|
||||
def replace_dtr_source_tag(source: str, has_value: bool) -> str:
|
||||
"""Replace previous DTR provenance with the active one."""
|
||||
parts = [
|
||||
part.strip()
|
||||
for part in (source.strip() or "unknown-source").split("+")
|
||||
if part.strip()
|
||||
]
|
||||
kept_parts = [
|
||||
part
|
||||
for part in parts
|
||||
if not part.startswith(SOURCE_TAG_PREFIX) and part != MISSING_SOURCE_TAG
|
||||
]
|
||||
kept_parts.append(SOURCE_TAG if has_value else MISSING_SOURCE_TAG)
|
||||
return " + ".join(kept_parts)
|
||||
|
||||
|
||||
def apply_diurnal_temperature_range(
|
||||
*,
|
||||
climate_data: Path,
|
||||
dtr_csv: Path,
|
||||
out: Path,
|
||||
) -> None:
|
||||
"""Join DTR values into climate-data.csv."""
|
||||
original_fields, climate_rows = read_csv_rows(climate_data)
|
||||
dtr_lookup = load_dtr_lookup(dtr_csv)
|
||||
fieldnames = fieldnames_with_metric(original_fields)
|
||||
|
||||
updated_count = 0
|
||||
missing_count = 0
|
||||
adjusted_count = 0
|
||||
for row in climate_rows:
|
||||
county_fips = normalize_fips(row.get("countyFips", ""))
|
||||
dtr_row = dtr_lookup.get(county_fips)
|
||||
if dtr_row:
|
||||
row[METRIC_FIELD] = dtr_row.get(METRIC_FIELD, "")
|
||||
row["source"] = replace_dtr_source_tag(row.get("source", ""), has_value=True)
|
||||
updated_count += 1
|
||||
if dtr_row.get("fipsAdjustment"):
|
||||
adjusted_count += 1
|
||||
else:
|
||||
row[METRIC_FIELD] = ""
|
||||
row["source"] = replace_dtr_source_tag(row.get("source", ""), has_value=False)
|
||||
missing_count += 1
|
||||
|
||||
with out.open("w", encoding="utf-8", newline="") as csv_file:
|
||||
writer = csv.DictWriter(csv_file, fieldnames=fieldnames, extrasaction="ignore")
|
||||
writer.writeheader()
|
||||
writer.writerows(climate_rows)
|
||||
|
||||
print(f"Wrote {len(climate_rows)} rows to {out}")
|
||||
print(f"Updated {METRIC_FIELD} for {updated_count} rows.")
|
||||
print(f"Used supported county-vintage FIPS adjustments for {adjusted_count} rows.")
|
||||
print(f"Left {METRIC_FIELD} blank for {missing_count} rows without DTR data.")
|
||||
|
||||
|
||||
def parse_args() -> argparse.Namespace:
|
||||
"""Parse command-line paths."""
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Apply average diurnal temperature range to climate-data.csv."
|
||||
)
|
||||
parser.add_argument("--climate-data", type=Path, default=DEFAULT_CLIMATE_DATA)
|
||||
parser.add_argument("--dtr-csv", type=Path, default=DEFAULT_DTR_CSV)
|
||||
parser.add_argument("--out", type=Path, default=DEFAULT_CLIMATE_DATA)
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Run the climate-data update."""
|
||||
args = parse_args()
|
||||
apply_diurnal_temperature_range(
|
||||
climate_data=args.climate_data,
|
||||
dtr_csv=args.dtr_csv,
|
||||
out=args.out,
|
||||
)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user