From b958a174ca0cca187eaa059f1a5956d9702aa544 Mon Sep 17 00:00:00 2001 From: Sourcery AI Date: Sun, 3 Apr 2022 10:01:30 +0000 Subject: [PATCH] 'Refactored by Sourcery' --- app.py | 11 +++++------ capture-data/web_scrape.py | 36 ++++++++++++++++++++++------------ src/fix_missing_values.py | 7 ++----- src/test_fix_missing_values.py | 29 +++++++++++++++------------ src/visualize_data.py | 17 ++++++++-------- 5 files changed, 56 insertions(+), 44 deletions(-) diff --git a/app.py b/app.py index adaff3a..1292dd5 100644 --- a/app.py +++ b/app.py @@ -52,7 +52,7 @@ def get_current_time(): current_min = int(dt.split(':')[1]) distances = [abs(current_min - _min) for _min in minutes] closest_min = minutes[distances.index(min(distances))] - current_time = dt.replace(':'+str(current_min), ':'+str(closest_min)) + current_time = dt.replace(f':{current_min}', f':{str(closest_min)}') current_time = datetime.datetime.strptime(current_time, "%Y/%m/%d %H:%M") return current_time @@ -118,9 +118,8 @@ def st_avg_data(boulderdf, selected_gym): # st_prediction(boulderdf, selected_gym) st_avg_data(boulderdf, selected_gym) - st.markdown(f""" - Does your gym show this occupancy data? Make a PR yourself or let us know and we'll add your gym 😎\n - Created by [anebz](https://github.com/anebz) and [AnglinaBhambra](https://github.com/AnglinaBhambra).\n - Follow us! [![@anebzt](https://img.shields.io/twitter/follow/anebzt?style=social)](https://www.twitter.com/anebzt) - [![@_AnglinaB](https://img.shields.io/twitter/follow/_AnglinaB?style=social)](https://www.twitter.com/_AnglinaB)""") + st.markdown( + """\x1f Does your gym show this occupancy data? Make a PR yourself or let us know and we'll add your gym 😎\\n\x1f Created by [anebz](https://github.com/anebz) and [AnglinaBhambra](https://github.com/AnglinaBhambra).\\n\x1f Follow us! [![@anebzt](https://img.shields.io/twitter/follow/anebzt?style=social)](https://www.twitter.com/anebzt)\x1f [![@_AnglinaB](https://img.shields.io/twitter/follow/_AnglinaB?style=social)](https://www.twitter.com/_AnglinaB)""" + ) + streamlit_analytics.stop_tracking(firestore_key_file="firestore-key.json", firestore_collection_name="counts") diff --git a/capture-data/web_scrape.py b/capture-data/web_scrape.py index d2d374d..963397b 100644 --- a/capture-data/web_scrape.py +++ b/capture-data/web_scrape.py @@ -34,13 +34,13 @@ def get_occupancy_boulderwelt(gym_name:str, url: str) -> tuple(): if 'queue' in data and int(data['queue']) > 0: occupancy += int(data['queue']) / 10 return occupancy - + # admin-ajax.php not working page = requests.get(url) if page.status_code != 200: return 0 try: - occupancy = int(float(re.search(r'style="margin-left:(.*?)%"', page.text).group(1))) + occupancy = int(float(re.search(r'style="margin-left:(.*?)%"', page.text)[1])) except: occupancy = 0 return occupancy @@ -68,7 +68,13 @@ def get_occupancy_boulderado(gym_name:str, url: str) -> tuple(): return 0 soup = BeautifulSoup(page.content, 'html.parser') try: - occupancy = int(re.search(r'left: (\d*)%', str(soup.find_all("div", class_="pointer-image")[0]['style'])).group(1)) + occupancy = int( + re.search( + r'left: (\d*)%', + str(soup.find_all("div", class_="pointer-image")[0]['style']), + )[1] + ) + except: occupancy = 0 return occupancy @@ -181,8 +187,8 @@ def get_occupancy_einstein(gym_name:str, url: str) -> tuple(): frame = soup.select("iframe")[0] frame_url = urllib.parse.urljoin(url, frame["src"]) response = session.get(frame_url) - frame_soup = BeautifulSoup(response.content, 'html.parser') - occupancy = re.search(r'left: (\d+)%', str(frame_soup)).group(1) + frame_soup = BeautifulSoup(response.content, 'html.parser') + occupancy = re.search(r'left: (\d+)%', str(frame_soup))[1] except: occupancy = 0 return occupancy @@ -216,12 +222,20 @@ def scrape_websites(current_time: str, gymdatadf: pd.DataFrame) -> pd.DataFrame: scrape_data = globals()[gym_data['function']] occupancy = scrape_data(gym_name, gym_data['url']) print(f"{gym_name}: occupancy={occupancy}, temp={weather_temp}, status={weather_status}") - if occupancy == 0 or occupancy == '0/0': + if occupancy in [0, '0/0']: continue webdata.append((current_time, gym_name, occupancy, weather_temp, weather_status)) - - webdf = pd.DataFrame(data=webdata, columns=['time', 'gym_name', 'occupancy', 'weather_temp', 'weather_status']) - return webdf + + return pd.DataFrame( + data=webdata, + columns=[ + 'time', + 'gym_name', + 'occupancy', + 'weather_temp', + 'weather_status', + ], + ) def get_current_time(): @@ -242,9 +256,7 @@ def get_current_time(): current_min = int(dt.split(':')[1]) distances = [abs(current_min - _min) for _min in minutes] closest_min = minutes[distances.index(min(distances))] - current_time = dt.replace(':'+str(current_min), ':'+str(closest_min)) - - return current_time + return dt.replace(f':{current_min}', f':{str(closest_min)}') def lambda_handler(event, context): diff --git a/src/fix_missing_values.py b/src/fix_missing_values.py index 8caa2ea..02b1a76 100644 --- a/src/fix_missing_values.py +++ b/src/fix_missing_values.py @@ -100,7 +100,7 @@ def fill_nan_values(df: pd.DataFrame) -> pd.DataFrame: for column in ['occupancy', 'waiting', 'weather_temp']: # this might work with the dev version of scipy func = interp1d(x_values, df_gym_day_noNan[column], kind='linear', bounds_error=False, fill_value='extrapolate') - + #fill in the interpolated value timestamp_unix = (timestamp-pd.Timestamp("1970-01-01")) // pd.Timedelta('1s') values = func([timestamp_unix]) @@ -120,10 +120,7 @@ def fill_nan_values(df: pd.DataFrame) -> pd.DataFrame: df_gym_day_oneNan = df_gym_day_noNan.append(df.loc[index], ignore_index=False) df_gym_day_oneNan = df_gym_day_oneNan.sort_values(by='current_time', ascending=True) #Case 1: it is at the beginning: - if df_gym_day_oneNan.iloc[0].isnull().sum() == 1: - shift = -1 - else: - shift = 1 + shift = -1 if df_gym_day_oneNan.iloc[0].isnull().sum() == 1 else 1 df_gym_day_oneNan_shift = df_gym_day_oneNan.shift(shift) df.loc[index, column] = df_gym_day_oneNan_shift.loc[index, column] return df diff --git a/src/test_fix_missing_values.py b/src/test_fix_missing_values.py index 99c553f..b85502f 100644 --- a/src/test_fix_missing_values.py +++ b/src/test_fix_missing_values.py @@ -16,8 +16,7 @@ class Testget_missing_aquisition_timestamps(unittest.TestCase): def gaussian(self, n, sigma, shift): - gauss = np.exp(-0.5*((n-shift)/sigma)**2) - return gauss + return np.exp(-0.5*((n-shift)/sigma)**2) def plot_gaussian(self): import matplotlib.pyplot as plt @@ -37,13 +36,13 @@ def get_testset(self): gym1 = 'muenchen-ost' gym2 = 'frankfurt' gyms_n = [gym1, gym1, gym1, gym2, gym2, gym2] - gyms = [[gyms_n[g] for n in day] for g,day in enumerate(days)] + gyms = [[gyms_n[g] for _ in day] for g,day in enumerate(days)] #occupancies=[[float(random.randint(0,100)) for n in day] for day in days] #making a gauss for occupancies occupancies = [self.gaussian(np.arange(0,len(day),1),len(day)/5,len(day)/2) for day in days] #waitings=[[float(random.randint(0,15)) for n in day] for day in days] waitings =[self.gaussian(np.arange(0,len(day),1),len(day)/5,len(day)/2) for day in days] - temps = [[float(random.randint(0,15)) for n in day] for day in days] + temps = [[float(random.randint(0,15)) for _ in day] for day in days] temps = [self.gaussian(np.arange(0,len(day),1),len(day)/5,len(day)/2)-0.5 for day in days] #statuss=[[random.choice(['Clouds', 'Rain', 'Drizzle', 'Clear', 'Thunderstorm', 'Mist']) for n in day] for day in days] statuss = [[random.choice(['Clouds', 'Rain', 'Drizzle', 'Clear', 'Thunderstorm', 'Mist'])]*len(day) for day in days] @@ -110,7 +109,7 @@ def test_testset(self): def test_add_missing_timestamps(self): df = self.get_testset() #remove 8 values at random - for n in range(8): + for _ in range(8): df = df.drop([df.index[random.randint(0, len(df)-1)]]) df2 = add_missing_timestamps(df, interval='20min', @@ -121,7 +120,7 @@ def test_add_missing_timestamps(self): def test_fill_nan(self): df = self.get_testset() #remove 8 values at random - for n in range(8): + for _ in range(8): df = df.drop([df.index[random.randint(0, len(df)-1)]]) df2 = add_missing_timestamps(df, interval='20min', @@ -133,7 +132,7 @@ def test_fill_nan(self): def test_drop_additional(self): df = self.get_testset() #change the time of 8 values at random - for n in range(8): + for _ in range(8): df.current_time[random.randint(0, len(df)-1)]+=pd.to_timedelta(16, unit='min') df2 = add_missing_timestamps(df, interval='20min', @@ -152,21 +151,25 @@ def test_correct_value_recovered(self): df = df_o[:] #remove 8 values at random dropped = [] - for n in range(8): + for _ in range(8): drop = df.index[random.randint(0, len(df)-1)] dropped.append(df.loc[drop]) df = df.drop([drop]) - + df2 = add_missing_timestamps(df, interval='20min', sample_start='07:20', sample_end='23:40') df2 = fill_nan_values(df2) #get the values it recovered: - recovered=[] - for n, drop in enumerate(dropped): - recovered.append(df2[(df2.gym_name == drop.gym_name)& - (df2.current_time == drop.current_time)]) + recovered = [ + df2[ + (df2.gym_name == drop.gym_name) + & (df2.current_time == drop.current_time) + ] + for drop in dropped + ] + ''' #in case of visual inspection plt.figure() diff --git a/src/visualize_data.py b/src/visualize_data.py index 7fe21bb..52b6314 100644 --- a/src/visualize_data.py +++ b/src/visualize_data.py @@ -118,12 +118,15 @@ def given_day(boulderdf: pd.DataFrame, date: str, gym: str) -> pd.DataFrame: def plot_data(df: pd.DataFrame): # to plot several graphs in the y axis, melt the df df = df.melt('time', var_name='name', value_name='occupancy') - chart = alt.Chart(df).mark_line(interpolate='basis').encode( - x=alt.X('time:N', axis=alt.Axis(grid=True)), - y=alt.Y('occupancy:Q', scale=alt.Scale(domain=[0, 100])), - color=alt.Color("name:N") + return ( + alt.Chart(df) + .mark_line(interpolate='basis') + .encode( + x=alt.X('time:N', axis=alt.Axis(grid=True)), + y=alt.Y('occupancy:Q', scale=alt.Scale(domain=[0, 100])), + color=alt.Color("name:N"), + ) ) - return chart def preprocess_current_data(boulderdf: pd.DataFrame, selected_gym: str, current_time: datetime.date): @@ -149,6 +152,4 @@ def preprocess_current_data(boulderdf: pd.DataFrame, selected_gym: str, current_ today_data[onehotlist.index('weather_temp')] = latest_gym_entry['weather_temp'] today_data[onehotlist.index(latest_gym_entry['weather_status'])] = 1 - # convert to numpy array - X_today = [np.asarray(today_data)] - return X_today + return [np.asarray(today_data)]