"""Summer heat at Dallas/Fort Worth International Airport.

Reads NOAA's GHCN-Daily record for station USW00003927 and writes the CSVs
used by the charts in the post. Run it next to the downloaded file:

    curl -O https://www.ncei.noaa.gov/data/global-historical-climatology-network-daily/access/USW00003927.csv
    python3 analysis.py USW00003927.csv

Writes:
    dfw-summers.csv          one row per year since FIRST_YEAR, every measure
    dfw-summer-days.csv      one row per summer day since FIRST_YEAR
    dfw-summers-history.csv  one row per year since HISTORY_YEAR, the basics
"""
import csv
import sys
from datetime import date

FIRST_YEAR = 1992
# The station's daily highs start in 1953; 1948 to 1952 only have lows.
HISTORY_YEAR = 1953
SUMMER = (6, 7, 8)  # June, July, August


def to_f(tenths_c):
    # GHCN stores tenths of a degree Celsius, converted from the whole
    # Fahrenheit degrees the weather service records. Rounding recovers them.
    return round(int(tenths_c) / 10 * 9 / 5 + 32)


def to_inches(tenths_mm):
    return int(tenths_mm) / 254


def passed_qc(attributes):
    # Attributes are "measurement flag, quality flag, source flag". Any
    # quality flag means the value failed one of NOAA's checks.
    parts = attributes.split(",")
    return len(parts) < 2 or parts[1].strip() == ""


def value(row, field, convert):
    if not row[field].strip() or not passed_qc(row[field + "_ATTRIBUTES"]):
        return None
    return convert(row[field])


def read(path, since):
    days = {}
    for row in csv.DictReader(open(path, newline="")):
        d = date.fromisoformat(row["DATE"])
        if d.year < since:
            continue
        days[d] = (value(row, "TMAX", to_f), value(row, "TMIN", to_f), value(row, "PRCP", to_inches))
    return days


def longest_streak(dates):
    best = run = 0
    prev = None
    for d in sorted(dates):
        run = run + 1 if prev and (d - prev).days == 1 else 1
        best = max(best, run)
        prev = d
    return best


def summarize(days, year):
    ds = {d: v for d, v in days.items() if d.year == year}
    hot = [d for d, (hi, _, _) in ds.items() if hi is not None and hi >= 100]
    summer = [v for d, v in ds.items() if d.month in SUMMER]
    his = [hi for hi, _, _ in summer if hi is not None]
    los = [lo for _, lo, _ in summer if lo is not None]
    return {
        "year": year,
        "days_100": len(hot),
        "days_105": sum(1 for hi, _, _ in ds.values() if hi is not None and hi >= 105),
        "nights_80": sum(1 for _, lo, _ in ds.values() if lo is not None and lo >= 80),
        "summer_high": round(sum(his) / len(his), 1),
        "summer_low": round(sum(los) / len(los), 1),
        "summer_rain": round(sum(p for _, _, p in summer if p is not None), 2),
        "first_100": min(hot).isoformat() if hot else "",
        "last_100": max(hot).isoformat() if hot else "",
        "longest_100_streak": longest_streak(hot),
        "hottest": max(hi for hi, _, _ in ds.values() if hi is not None),
        "summer_days_missing": 92 - min(len(his), len(los)),
    }


def write(path, rows, fields=None):
    with open(path, "w", newline="") as f:
        w = csv.DictWriter(f, fieldnames=fields or list(rows[0]), extrasaction="ignore")
        w.writeheader()
        w.writerows(rows)


def main(path):
    days = read(path, HISTORY_YEAR)
    last = max(days)

    history = [summarize(days, y) for y in range(HISTORY_YEAR, last.year + 1)]
    write("dfw-summers-history.csv", history, ["year", "days_100", "nights_80", "summer_high", "summer_low", "summer_rain", "summer_days_missing"])
    write("dfw-summers.csv", [r for r in history if r["year"] >= FIRST_YEAR])

    # Every summer day since FIRST_YEAR, for the heatmap.
    with open("dfw-summer-days.csv", "w", newline="") as f:
        w = csv.writer(f)
        w.writerow(["date", "high", "low"])
        for d in sorted(days):
            if d.year >= FIRST_YEAR and d.month in SUMMER and days[d][0] is not None:
                w.writerow([d.isoformat(), days[d][0], "" if days[d][1] is None else days[d][1]])

    print(f"{HISTORY_YEAR}-{last.year}, last day {last}")


if __name__ == "__main__":
    main(sys.argv[1] if len(sys.argv) > 1 else "USW00003927.csv")
