-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathconditions_calculator.py
More file actions
executable file
·187 lines (135 loc) · 8.37 KB
/
Copy pathconditions_calculator.py
File metadata and controls
executable file
·187 lines (135 loc) · 8.37 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
# -*- coding: utf-8 -*-
"""CACRCGRB Final Copy.ipynb
Automatically generated by Colaboratory.
Original file is located at
https://colab.research.google.com/drive/1EID1XYTTPDqXKdQMGN0htGjBEdjZ01gN
"""
###################################################################################################################################################
# IMPORTS
###################################################################################################################################################
# For data collection & management
import os # for navigating computer files
import numpy as np # for managing arrays
import pandas as pd # for managing tables and csv files
import requests # for http requests
import csv # for handling csv files
import matplotlib.pyplot as plt # for data visualization
import seaborn as sns # for making it look nicer
import urllib # for fetching urls
import codecs # for compressing files over the internet
# For training and testing model
from sklearn.model_selection import train_test_split # for splitting data between training set and testing set
from sklearn.ensemble import GradientBoostingRegressor # for creating a gradient boosting regression model
from sklearn.pipeline import Pipeline # to make the training smoother
from sklearn import preprocessing # for standardizing dataset
from sklearn.model_selection import learning_curve # to visualize learning curve
from sklearn.metrics import mean_squared_error # for judging accuracy of model
# For debugging
import logging
from running_app import logger
###################################################################################################################################################
# CONSTANTS
###################################################################################################################################################
import datetime
current_time = str(datetime.datetime.now())
location = "queens%20new%20york"
date = current_time[:10:]
full_time = current_time[11:19:]
time = int(full_time[:2:])
historical_data_download_url = 'https://drive.google.com/uc?export=download&id=1Q6Wdyq2nW7LuEVKp25Lka7ZmKgn6wFYA' # sharing url is https://drive.google.com/file/d/1Q6Wdyq2nW7LuEVKp25Lka7ZmKgn6wFYA/view?usp=sharing
historical_data_file_name = "historical_data.csv"
api_key = "H86LKTEJLLQ6GG58ZJHZPSNWW"
current_data_download_url = f"https://weather.visualcrossing.com/VisualCrossingWebServices/rest/services/timeline/{location}/{date}/{date}?unitGroup=us&include=hours&key={api_key}&contentType=csv"
current_data_file_name = "current_data.csv"
up_to_date = False
###################################################################################################################################################
# DOWNLOADING DATA
###################################################################################################################################################
def download_data(download_url: str, file_name: str):
data = requests.get(download_url)
if data.status_code == 200:
with open(file_name, "wb") as file:
file.write(data.content)
print(f"CSV file downloaded and saved as {file_name}")
logger.info(f"CSV file was successfully downloaded as {file_name}")
else:
print("Failed to download the CSV file")
logger.warning(f"CSV file was not successfully downloaded")
# def download_weather_data():
# download_data(historical_data_download_url, historical_data_file_name)
# download_data(current_data_download_url, current_data_file_name)
def download_weather_data():
datetime = None
last_updated_date = None
try:
historical_weather_dataframe = pd.read_csv(historical_data_file_name)
current_weather_dataframe = pd.read_csv(current_data_file_name)
datetime = current_weather_dataframe["datetime"]
last_updated_date = datetime[0][0:10:1]
if last_updated_date == date:
up_to_date = True
else:
up_to_date = False
except:
up_to_date = False
logger.debug(f"Current date: {date}. Current time: {time}.")
if up_to_date:
logger.debug(f"Data is already up to date. No further action needed.")
else:
logger.debug(f"Data was last updated {last_updated_date}. Update needed. Currently downloading data...")
download_data(historical_data_download_url, historical_data_file_name)
download_data(current_data_download_url, current_data_file_name)
logger.debug(f"Update completed. Data is now up to date.")
###################################################################################################################################################
# CALCULATING RUNNING INDEX
###################################################################################################################################################
def calculate_running_index():
logger.info("Calculating running index")
download_weather_data()
############################################################## TRAINING ######################################################
logger.debug("Training started")
# 1. Making a dataframe out of our training data
training_dataframe = pd.read_csv(historical_data_file_name)
# 2. Setting the features (what our model will factor into its predictions) and true labels (the answer key)
X = ['humidity', 'cloudcover', 'dew', 'feelslike', 'precip', 'precipprob', 'winddir']
y = 'RWI'
# 3. Defining pipeline
pipeline = Pipeline([
('scaler', preprocessing.StandardScaler()),
('regressor', GradientBoostingRegressor(n_estimators=275, learning_rate=0.11, max_depth=2, random_state=69))
])
# 4. Creating training datasets
X_train = training_dataframe[X]
y_train = training_dataframe[y]
# 5. Fitting training data to pipeline
pipeline.fit(X_train, y_train)
# 6. Making predictions for training data
train_predictions = pipeline.predict(X_train)
train_mse = mean_squared_error(y_train, train_predictions)
logger.debug(f"Training completed. Mean Squared Error: {train_mse}")
############################################################## TESTING ######################################################
logger.debug("Testing started")
# 1. Making a dataframe out of our testing data
testing_dataframe = pd.read_csv(current_data_file_name)
# 1.5. Checks whether information is up to date
data_date_time = ""
for index in range(10):
data_date_time += testing_dataframe["datetime"][index]
if data_date_time == date:
up_to_date = True
# 2. Creating testing dataset
X_test = testing_dataframe[X]
# 3. Make predictions on the testing dataset
test_predictions = pipeline.predict(X_test)
final_predictions = [round(RWI) for RWI in test_predictions]
logger.debug(f"Testing completed. Final predictions: {final_predictions}")
logger.info(f"Calculations complete. Calculated running index: {final_predictions[time]}")
return final_predictions[int(time)]
###################################################################################################################################################
# CALCULATING WEATHER DESCRIPTION
###################################################################################################################################################
def calculate_weather():
download_weather_data()
current_weather = pd.read_csv(current_data_file_name)["icon"][time]
logger.info(f"Current weather: {current_weather}")
return current_weather