Download clean_dataset.py from mrtag08/thane-weather-dataset-2022-2026: direct link, hf CLI and curl.
- Browser
- Download file 5.27 kB
-
https://huggingface.co/datasets/mrtag08/thane-weather-dataset-2022-2026/resolve/main/clean_dataset.py
- Command line
-
hf download hf://datasets/mrtag08/thane-weather-dataset-2022-2026/clean_dataset.py
-
curl -L -o clean_dataset.py https://huggingface.co/datasets/mrtag08/thane-weather-dataset-2022-2026/resolve/main/clean_dataset.py
5.27 kB
| """ | |
| clean_dataset.py | |
| ---------------- | |
| Processes and standardizes raw IFTTT weather observations recorded in Thane, Maharashtra, India. | |
| Features cleaned: | |
| - Dropping unusable formula columns (#REF!) and empty columns | |
| - Standardizing date and time to ISO formats | |
| - Dual units (Celsius / Fahrenheit, km/h / mph) cleanly preserved and typed | |
| - Feature engineering: | |
| - Season classification (Winter, Summer/Pre-Monsoon, Monsoon, Post-Monsoon) | |
| - Diurnal temperature range (High - Low) | |
| - Astronomical day length (hours) | |
| - Rain occurrence flag | |
| - Geospatial metadata tags (Thane, Maharashtra, India; 19.22 N, 72.98 E) | |
| Outputs: | |
| - data/thane_weather_2022_2026.csv | |
| - data/thane_weather_2022_2026.parquet | |
| """ | |
| import os | |
| import csv | |
| from datetime import datetime | |
| import pandas as pd | |
| from pathlib import Path | |
| def parse_day_length(rise_str, set_str): | |
| try: | |
| t_rise = datetime.strptime(rise_str.split(' at ')[-1].strip(), '%I:%M%p') | |
| t_set = datetime.strptime(set_str.split(' at ')[-1].strip(), '%I:%M%p') | |
| rise_h = t_rise.hour + t_rise.minute / 60.0 | |
| set_h = t_set.hour + t_set.minute / 60.0 | |
| return round((set_h + 24.0) - rise_h, 2) | |
| except Exception: | |
| return None | |
| def get_season(month): | |
| # Seasons in Maharashtra / MMR climatology: | |
| # Dec - Feb: Winter | |
| # Mar - May: Summer / Pre-Monsoon | |
| # Jun - Sep: Monsoon | |
| # Oct - Nov: Post-Monsoon | |
| if month in [12, 1, 2]: | |
| return "Winter" | |
| elif month in [3, 4, 5]: | |
| return "Summer / Pre-Monsoon" | |
| elif month in [6, 7, 8, 9]: | |
| return "Monsoon" | |
| else: | |
| return "Post-Monsoon" | |
| def clean_data(input_csv="weather_data_raw.csv", output_dir="data"): | |
| out_path = Path(output_dir) | |
| out_path.mkdir(parents=True, exist_ok=True) | |
| with open(input_csv, 'r', encoding='utf-8') as f: | |
| raw_rows = list(csv.reader(f)) | |
| print(f"Total raw rows: {len(raw_rows)}") | |
| records = [] | |
| for r in raw_rows: | |
| if len(r) < 19: | |
| continue | |
| # Col 0: 'March 17, 2022 at 09:59AM' | |
| raw_dt_str = r[0].strip() | |
| parts = raw_dt_str.split(' at ') | |
| date_obj = datetime.strptime(parts[0], '%B %d, %Y') | |
| date_str = date_obj.strftime('%Y-%m-%d') | |
| time_rec = parts[1].strip() if len(parts) > 1 else '10:00AM' | |
| temp_f = float(r[1]) | |
| temp_c = float(r[2]) | |
| condition = r[3].strip().title() | |
| forecast_high_f = float(r[5]) | |
| forecast_high_c = float(r[6]) | |
| forecast_low_f = float(r[7]) | |
| forecast_low_c = float(r[8]) | |
| forecast_condition = r[9].strip().title() | |
| humidity_pct = int(float(r[12])) | |
| wind_speed_mph = float(r[13]) | |
| wind_speed_kmh = float(r[14]) | |
| wind_direction = r[15].strip().title() | |
| # Sunrise & Sunset | |
| raw_sunrise = r[17].strip() | |
| raw_sunset = r[18].strip() | |
| day_length = parse_day_length(raw_sunrise, raw_sunset) | |
| diurnal_range_c = round(forecast_high_c - forecast_low_c, 1) | |
| diurnal_range_f = round(forecast_high_f - forecast_low_f, 1) | |
| season = get_season(date_obj.month) | |
| rain_keywords = ['rain', 'drizzle', 'shower', 'thunderstorm'] | |
| is_rain = any(k in condition.lower() or k in forecast_condition.lower() for k in rain_keywords) | |
| records.append({ | |
| "date": date_str, | |
| "year": date_obj.year, | |
| "month": date_obj.month, | |
| "day": date_obj.day, | |
| "day_of_week": date_obj.strftime('%A'), | |
| "time_recorded": time_rec, | |
| "temp_c": temp_c, | |
| "temp_f": temp_f, | |
| "condition": condition, | |
| "forecast_high_c": forecast_high_c, | |
| "forecast_high_f": forecast_high_f, | |
| "forecast_low_c": forecast_low_c, | |
| "forecast_low_f": forecast_low_f, | |
| "forecast_condition": forecast_condition, | |
| "humidity_pct": humidity_pct, | |
| "wind_speed_kmh": wind_speed_kmh, | |
| "wind_speed_mph": wind_speed_mph, | |
| "wind_direction": wind_direction, | |
| "day_length_hours": day_length, | |
| "temp_diurnal_range_c": diurnal_range_c, | |
| "temp_diurnal_range_f": diurnal_range_f, | |
| "season": season, | |
| "is_rain": is_rain, | |
| "city": "Thane", | |
| "state": "Maharashtra", | |
| "country": "India", | |
| "latitude": 19.22, | |
| "longitude": 72.98 | |
| }) | |
| df = pd.DataFrame(records) | |
| # Sort chronologically | |
| df['date_dt'] = pd.to_datetime(df['date']) | |
| df = df.sort_values('date_dt').drop(columns=['date_dt']).reset_index(drop=True) | |
| csv_file = out_path / "thane_weather_2022_2026.csv" | |
| parquet_file = out_path / "thane_weather_2022_2026.parquet" | |
| df.to_csv(csv_file, index=False) | |
| df.to_parquet(parquet_file, index=False) | |
| print(f"Cleaned dataset successfully created!") | |
| print(f"Shape: {df.shape}") | |
| print(f"Date range: {df['date'].min()} to {df['date'].max()}") | |
| print(f"CSV saved to: {csv_file}") | |
| print(f"Parquet saved to: {parquet_file}") | |
| print("\nSummary stats:") | |
| print(df[['temp_c', 'forecast_high_c', 'forecast_low_c', 'humidity_pct', 'wind_speed_kmh', 'day_length_hours']].describe()) | |
| if __name__ == "__main__": | |
| clean_data() | |