mrtag08's picture
Upload folder using huggingface_hub
b9d43ad verified
Raw History Blame Contribute Delete
5.27 kB
"""
clean_dataset.py
----------------
Processes and standardizes raw IFTTT weather observations recorded in Thane, Maharashtra, India.
Features cleaned:
- Dropping unusable formula columns (#REF!) and empty columns
- Standardizing date and time to ISO formats
- Dual units (Celsius / Fahrenheit, km/h / mph) cleanly preserved and typed
- Feature engineering:
- Season classification (Winter, Summer/Pre-Monsoon, Monsoon, Post-Monsoon)
- Diurnal temperature range (High - Low)
- Astronomical day length (hours)
- Rain occurrence flag
- Geospatial metadata tags (Thane, Maharashtra, India; 19.22 N, 72.98 E)
Outputs:
- data/thane_weather_2022_2026.csv
- data/thane_weather_2022_2026.parquet
"""
import os
import csv
from datetime import datetime
import pandas as pd
from pathlib import Path
def parse_day_length(rise_str, set_str):
try:
t_rise = datetime.strptime(rise_str.split(' at ')[-1].strip(), '%I:%M%p')
t_set = datetime.strptime(set_str.split(' at ')[-1].strip(), '%I:%M%p')
rise_h = t_rise.hour + t_rise.minute / 60.0
set_h = t_set.hour + t_set.minute / 60.0
return round((set_h + 24.0) - rise_h, 2)
except Exception:
return None
def get_season(month):
# Seasons in Maharashtra / MMR climatology:
# Dec - Feb: Winter
# Mar - May: Summer / Pre-Monsoon
# Jun - Sep: Monsoon
# Oct - Nov: Post-Monsoon
if month in [12, 1, 2]:
return "Winter"
elif month in [3, 4, 5]:
return "Summer / Pre-Monsoon"
elif month in [6, 7, 8, 9]:
return "Monsoon"
else:
return "Post-Monsoon"
def clean_data(input_csv="weather_data_raw.csv", output_dir="data"):
out_path = Path(output_dir)
out_path.mkdir(parents=True, exist_ok=True)
with open(input_csv, 'r', encoding='utf-8') as f:
raw_rows = list(csv.reader(f))
print(f"Total raw rows: {len(raw_rows)}")
records = []
for r in raw_rows:
if len(r) < 19:
continue
# Col 0: 'March 17, 2022 at 09:59AM'
raw_dt_str = r[0].strip()
parts = raw_dt_str.split(' at ')
date_obj = datetime.strptime(parts[0], '%B %d, %Y')
date_str = date_obj.strftime('%Y-%m-%d')
time_rec = parts[1].strip() if len(parts) > 1 else '10:00AM'
temp_f = float(r[1])
temp_c = float(r[2])
condition = r[3].strip().title()
forecast_high_f = float(r[5])
forecast_high_c = float(r[6])
forecast_low_f = float(r[7])
forecast_low_c = float(r[8])
forecast_condition = r[9].strip().title()
humidity_pct = int(float(r[12]))
wind_speed_mph = float(r[13])
wind_speed_kmh = float(r[14])
wind_direction = r[15].strip().title()
# Sunrise & Sunset
raw_sunrise = r[17].strip()
raw_sunset = r[18].strip()
day_length = parse_day_length(raw_sunrise, raw_sunset)
diurnal_range_c = round(forecast_high_c - forecast_low_c, 1)
diurnal_range_f = round(forecast_high_f - forecast_low_f, 1)
season = get_season(date_obj.month)
rain_keywords = ['rain', 'drizzle', 'shower', 'thunderstorm']
is_rain = any(k in condition.lower() or k in forecast_condition.lower() for k in rain_keywords)
records.append({
"date": date_str,
"year": date_obj.year,
"month": date_obj.month,
"day": date_obj.day,
"day_of_week": date_obj.strftime('%A'),
"time_recorded": time_rec,
"temp_c": temp_c,
"temp_f": temp_f,
"condition": condition,
"forecast_high_c": forecast_high_c,
"forecast_high_f": forecast_high_f,
"forecast_low_c": forecast_low_c,
"forecast_low_f": forecast_low_f,
"forecast_condition": forecast_condition,
"humidity_pct": humidity_pct,
"wind_speed_kmh": wind_speed_kmh,
"wind_speed_mph": wind_speed_mph,
"wind_direction": wind_direction,
"day_length_hours": day_length,
"temp_diurnal_range_c": diurnal_range_c,
"temp_diurnal_range_f": diurnal_range_f,
"season": season,
"is_rain": is_rain,
"city": "Thane",
"state": "Maharashtra",
"country": "India",
"latitude": 19.22,
"longitude": 72.98
})
df = pd.DataFrame(records)
# Sort chronologically
df['date_dt'] = pd.to_datetime(df['date'])
df = df.sort_values('date_dt').drop(columns=['date_dt']).reset_index(drop=True)
csv_file = out_path / "thane_weather_2022_2026.csv"
parquet_file = out_path / "thane_weather_2022_2026.parquet"
df.to_csv(csv_file, index=False)
df.to_parquet(parquet_file, index=False)
print(f"Cleaned dataset successfully created!")
print(f"Shape: {df.shape}")
print(f"Date range: {df['date'].min()} to {df['date'].max()}")
print(f"CSV saved to: {csv_file}")
print(f"Parquet saved to: {parquet_file}")
print("\nSummary stats:")
print(df[['temp_c', 'forecast_high_c', 'forecast_low_c', 'humidity_pct', 'wind_speed_kmh', 'day_length_hours']].describe())
if __name__ == "__main__":
clean_data()