Skip to content

Commit b4f9fea

Browse files
committed
updating the conda env usage
1 parent 3da298e commit b4f9fea

8 files changed

Lines changed: 282 additions & 2 deletions

File tree

‎demo/keeling-curve/.gitignore‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,3 @@
1+
*.csv
2+
*.png
3+
__pycache__/

‎demo/keeling-curve/analyze.py‎

Lines changed: 58 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,58 @@
1+
"""
2+
analyze.py — Compute summary statistics and trends from the Keeling Curve data.
3+
"""
4+
5+
import pandas as pd
6+
7+
INPUT_FILE = "co2_clean.csv"
8+
9+
10+
def analyze(input_path: str) -> None:
11+
"""Print key insights from the Keeling Curve data."""
12+
df = pd.read_csv(input_path, parse_dates=["date"])
13+
14+
# --- Annual averages ---
15+
annual = df.groupby("year")["co2"].mean()
16+
17+
print("=" * 50)
18+
print("KEELING CURVE ANALYSIS")
19+
print("=" * 50)
20+
21+
# Overall statistics
22+
print(f"\nDate range: {df['date'].min().year} – {df['date'].max().year}")
23+
print(f"Total measurements: {len(df)}")
24+
print(f"CO₂ start: {annual.iloc[0]:.2f} ppm ({int(annual.index[0])})")
25+
print(f"CO₂ latest: {annual.iloc[-1]:.2f} ppm ({int(annual.index[-1])})")
26+
print(f"Total increase: {annual.iloc[-1] - annual.iloc[0]:.2f} ppm")
27+
28+
# --- Growth rate by decade ---
29+
print("\n--- Average Annual Growth Rate by Decade ---")
30+
df_annual = annual.reset_index()
31+
df_annual.columns = ["year", "co2"]
32+
df_annual["decade"] = (df_annual["year"] // 10) * 10
33+
df_annual["co2_diff"] = df_annual["co2"].diff()
34+
35+
decade_growth = df_annual.groupby("decade")["co2_diff"].mean()
36+
for decade, rate in decade_growth.items():
37+
if pd.notna(rate):
38+
print(f" {int(decade)}s: +{rate:.2f} ppm/year")
39+
40+
# --- Seasonal amplitude ---
41+
print("\n--- Seasonal Cycle ---")
42+
monthly_avg = df.groupby("month")["co2"].mean()
43+
amplitude = monthly_avg.max() - monthly_avg.min()
44+
peak_month = monthly_avg.idxmax()
45+
trough_month = monthly_avg.idxmin()
46+
47+
month_names = [
48+
"", "Jan", "Feb", "Mar", "Apr", "May", "Jun",
49+
"Jul", "Aug", "Sep", "Oct", "Nov", "Dec"
50+
]
51+
52+
print(f" Seasonal amplitude: {amplitude:.2f} ppm")
53+
print(f" Peak month: {month_names[int(peak_month)]} ({monthly_avg.max():.2f} ppm)")
54+
print(f" Trough month: {month_names[int(trough_month)]} ({monthly_avg.min():.2f} ppm)")
55+
56+
57+
if __name__ == "__main__":
58+
analyze(INPUT_FILE)

‎demo/keeling-curve/clean_data.py‎

Lines changed: 86 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,86 @@
1+
"""
2+
clean_data.py — Load, clean, and prepare the Keeling Curve data.
3+
"""
4+
5+
import pandas as pd
6+
7+
INPUT_FILE = "co2_monthly.csv"
8+
OUTPUT_FILE = "co2_clean.csv"
9+
10+
11+
def load_and_clean(input_path: str) -> pd.DataFrame:
12+
"""
13+
Load the raw Scripps CSV and return a clean DataFrame.
14+
15+
The raw file has:
16+
- Comment lines starting with "
17+
- Missing values coded as -99.99
18+
- Columns: Yr, Mn, Date_Excel, Date, CO2, seasonally_adjusted, fit, ...
19+
"""
20+
# Read the CSV, skipping comment lines
21+
df = pd.read_csv(
22+
input_path,
23+
comment='"',
24+
header=0,
25+
skipinitialspace=True,
26+
na_values=["-99.99", -99.99],
27+
)
28+
29+
# Strip whitespace from column names
30+
df.columns = df.columns.str.strip()
31+
32+
# Keep only the columns we need
33+
# The exact column names may vary — let's inspect and adapt
34+
print(f"Raw columns: {list(df.columns)}")
35+
print(f"Raw shape: {df.shape}")
36+
37+
# Rename columns to something consistent
38+
# Typical columns: Yr, Mn, Date Excel, Date, CO2, seasonally adjusted, fit, ...
39+
col_map = {}
40+
for col in df.columns:
41+
col_lower = col.lower().strip()
42+
if col_lower in ("yr", "year"):
43+
col_map[col] = "year"
44+
elif col_lower in ("mn", "month"):
45+
col_map[col] = "month"
46+
elif col_lower == "co2":
47+
col_map[col] = "co2"
48+
elif "seasonally" in col_lower:
49+
col_map[col] = "co2_adjusted"
50+
elif col_lower == "fit":
51+
col_map[col] = "co2_fit"
52+
53+
df = df.rename(columns=col_map)
54+
55+
# Keep only the columns we mapped
56+
keep_cols = [c for c in ["year", "month", "co2", "co2_adjusted", "co2_fit"] if c in df.columns]
57+
df = df[keep_cols].copy()
58+
59+
# Convert types
60+
df["year"] = pd.to_numeric(df["year"], errors="coerce")
61+
df["month"] = pd.to_numeric(df["month"], errors="coerce")
62+
df["co2"] = pd.to_numeric(df["co2"], errors="coerce")
63+
64+
# Drop rows where year or co2 is missing
65+
df = df.dropna(subset=["year", "co2"])
66+
67+
# Create a proper date column
68+
df["date"] = pd.to_datetime(
69+
df["year"].astype(int).astype(str) + "-" + df["month"].astype(int).astype(str) + "-01"
70+
)
71+
72+
# Sort by date
73+
df = df.sort_values("date").reset_index(drop=True)
74+
75+
print(f"Clean shape: {df.shape}")
76+
print(f"Date range: {df['date'].min()} to {df['date'].max()}")
77+
print(f"CO₂ range: {df['co2'].min():.2f} to {df['co2'].max():.2f} ppm")
78+
79+
return df
80+
81+
82+
if __name__ == "__main__":
83+
df = load_and_clean(INPUT_FILE)
84+
df.to_csv(OUTPUT_FILE, index=False)
85+
print(f"\nSaved clean data to {OUTPUT_FILE}")
86+
print(df.head(10))

‎demo/keeling-curve/fetch_data.py‎

Lines changed: 25 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,25 @@
1+
"""
2+
fetch_data.py — Download the Keeling Curve dataset from Scripps CO2 Program.
3+
"""
4+
5+
import requests
6+
import os
7+
8+
DATA_URL = "https://scrippsco2.ucsd.edu/assets/data/atmospheric/stations/in_situ_co2/monthly/monthly_in_situ_co2_mlo.csv"
9+
OUTPUT_FILE = "co2_monthly.csv"
10+
11+
12+
def download_data(url: str, output_path: str) -> None:
13+
"""Download the raw CSV data from Scripps."""
14+
print(f"Downloading data from {url}...")
15+
response = requests.get(url, timeout=30)
16+
response.raise_for_status()
17+
18+
with open(output_path, "w") as f:
19+
f.write(response.text)
20+
21+
print(f"Saved to {output_path} ({len(response.text):,} bytes)")
22+
23+
24+
if __name__ == "__main__":
25+
download_data(DATA_URL, OUTPUT_FILE)
Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,3 @@
1+
pandas
2+
matplotlib
3+
requests

‎demo/keeling-curve/visualize.py‎

Lines changed: 105 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,105 @@
1+
"""
2+
visualize.py — Create plots of the Keeling Curve data.
3+
"""
4+
5+
import pandas as pd
6+
import matplotlib.pyplot as plt
7+
import matplotlib.dates as mdates
8+
9+
10+
INPUT_FILE = "co2_clean.csv"
11+
12+
13+
def plot_full_curve(df: pd.DataFrame) -> None:
14+
"""Plot the complete Keeling Curve."""
15+
fig, ax = plt.subplots(figsize=(12, 6))
16+
17+
ax.plot(df["date"], df["co2"], color="#2196F3", linewidth=0.8, alpha=0.8, label="Monthly average")
18+
19+
# Add a trend line using the adjusted/fit values if available
20+
if "co2_fit" in df.columns:
21+
ax.plot(df["date"], df["co2_fit"], color="#F44336", linewidth=1.5, label="Trend (fit)")
22+
23+
ax.set_xlabel("Year", fontsize=12)
24+
ax.set_ylabel("CO₂ Concentration (ppm)", fontsize=12)
25+
ax.set_title("The Keeling Curve — Atmospheric CO₂ at Mauna Loa", fontsize=14, fontweight="bold")
26+
ax.legend(loc="upper left")
27+
ax.grid(True, alpha=0.3)
28+
29+
# Format x-axis
30+
ax.xaxis.set_major_locator(mdates.YearLocator(10))
31+
ax.xaxis.set_major_formatter(mdates.DateFormatter("%Y"))
32+
33+
plt.tight_layout()
34+
plt.savefig("keeling_curve_full.png", dpi=150)
35+
print("Saved: keeling_curve_full.png")
36+
plt.close()
37+
38+
39+
def plot_seasonal_cycle(df: pd.DataFrame) -> None:
40+
"""Plot the average seasonal CO₂ cycle."""
41+
monthly_avg = df.groupby("month")["co2"].mean()
42+
43+
fig, ax = plt.subplots(figsize=(8, 5))
44+
months = range(1, 13)
45+
month_labels = ["Jan", "Feb", "Mar", "Apr", "May", "Jun",
46+
"Jul", "Aug", "Sep", "Oct", "Nov", "Dec"]
47+
48+
ax.bar(months, monthly_avg.values, color="#4CAF50", alpha=0.7, edgecolor="white")
49+
ax.set_xticks(months)
50+
ax.set_xticklabels(month_labels)
51+
ax.set_xlabel("Month", fontsize=12)
52+
ax.set_ylabel("Average CO₂ (ppm)", fontsize=12)
53+
ax.set_title("Average Seasonal CO₂ Cycle", fontsize=14, fontweight="bold")
54+
ax.grid(True, axis="y", alpha=0.3)
55+
56+
plt.tight_layout()
57+
plt.savefig("keeling_curve_seasonal.png", dpi=150)
58+
print("Saved: keeling_curve_seasonal.png")
59+
plt.close()
60+
61+
62+
def plot_decade_comparison(df: pd.DataFrame) -> None:
63+
"""Plot CO₂ trend for each decade."""
64+
fig, ax = plt.subplots(figsize=(12, 6))
65+
66+
colors = plt.cm.viridis_r # Color map: darker = more recent
67+
decades = sorted(df["year"].apply(lambda y: int(y) // 10 * 10).unique())
68+
69+
for i, decade in enumerate(decades):
70+
mask = (df["year"] >= decade) & (df["year"] < decade + 10)
71+
subset = df[mask]
72+
if len(subset) > 0:
73+
color = colors(i / len(decades))
74+
ax.plot(subset["month"], subset["co2"], alpha=0.3, color=color)
75+
76+
# Plot the mean for first and last decades
77+
for decade, style, label in [(decades[0], "--", f"{decades[0]}s avg"),
78+
(decades[-1], "-", f"{decades[-1]}s avg")]:
79+
mask = (df["year"] >= decade) & (df["year"] < decade + 10)
80+
subset = df[mask]
81+
if len(subset) > 0:
82+
monthly = subset.groupby("month")["co2"].mean()
83+
ax.plot(monthly.index, monthly.values, style, linewidth=2.5, label=label)
84+
85+
ax.set_xticks(range(1, 13))
86+
ax.set_xticklabels(["Jan", "Feb", "Mar", "Apr", "May", "Jun",
87+
"Jul", "Aug", "Sep", "Oct", "Nov", "Dec"])
88+
ax.set_xlabel("Month", fontsize=12)
89+
ax.set_ylabel("CO₂ (ppm)", fontsize=12)
90+
ax.set_title("CO₂ Seasonal Cycle by Decade", fontsize=14, fontweight="bold")
91+
ax.legend()
92+
ax.grid(True, alpha=0.3)
93+
94+
plt.tight_layout()
95+
plt.savefig("keeling_curve_decades.png", dpi=150)
96+
print("Saved: keeling_curve_decades.png")
97+
plt.close()
98+
99+
100+
if __name__ == "__main__":
101+
df = pd.read_csv(INPUT_FILE, parse_dates=["date"])
102+
plot_full_curve(df)
103+
plot_seasonal_cycle(df)
104+
plot_decade_comparison(df)
105+
print("\nAll plots saved!")

‎docs/index.md‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
# Data Engineering Workshop
22

3-
**A hands-on workshop for master's students — from zero to data pipeline with AI-assisted coding.**
3+
**CSP data workshop for capstone development | Spring 2026**
44

55
---
66

‎mkdocs.yml‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,5 +1,5 @@
11
site_name: Data Engineering Workshop
2-
site_description: A hands-on workshop for master's students — from zero to data pipeline with AI-assisted coding.
2+
site_description: CSP data workshop for capstone development | Spring 2026
33
site_url: https://connorjmack.github.io/csp-data-workshop/
44
repo_url: https://github.com/connorjmack/csp-data-workshop
55
repo_name: connorjmack/csp-data-workshop

0 commit comments

Comments
 (0)