Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
11 changes: 5 additions & 6 deletions app.py
Original file line number Diff line number Diff line change
Expand Up @@ -52,7 +52,7 @@ def get_current_time():
current_min = int(dt.split(':')[1])
distances = [abs(current_min - _min) for _min in minutes]
closest_min = minutes[distances.index(min(distances))]
current_time = dt.replace(':'+str(current_min), ':'+str(closest_min))
current_time = dt.replace(f':{current_min}', f':{str(closest_min)}')

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function get_current_time refactored with the following changes:

current_time = datetime.datetime.strptime(current_time, "%Y/%m/%d %H:%M")

return current_time
Expand Down Expand Up @@ -118,9 +118,8 @@ def st_avg_data(boulderdf, selected_gym):
# st_prediction(boulderdf, selected_gym)
st_avg_data(boulderdf, selected_gym)

st.markdown(f"""
Does your gym show this occupancy data? Make a PR yourself or let us know and we'll add your gym 😎\n
Created by [anebz](https://github.com/anebz) and [AnglinaBhambra](https://github.com/AnglinaBhambra).\n
Follow us! [![@anebzt](https://img.shields.io/twitter/follow/anebzt?style=social)](https://www.twitter.com/anebzt)
[![@_AnglinaB](https://img.shields.io/twitter/follow/_AnglinaB?style=social)](https://www.twitter.com/_AnglinaB)""")
st.markdown(
"""\x1f Does your gym show this occupancy data? Make a PR yourself or let us know and we'll add your gym 😎\\n\x1f Created by [anebz](https://github.com/anebz) and [AnglinaBhambra](https://github.com/AnglinaBhambra).\\n\x1f Follow us! [![@anebzt](https://img.shields.io/twitter/follow/anebzt?style=social)](https://www.twitter.com/anebzt)\x1f [![@_AnglinaB](https://img.shields.io/twitter/follow/_AnglinaB?style=social)](https://www.twitter.com/_AnglinaB)"""
)

Comment on lines -121 to +124

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Lines 121-125 refactored with the following changes:

streamlit_analytics.stop_tracking(firestore_key_file="firestore-key.json", firestore_collection_name="counts")
36 changes: 24 additions & 12 deletions capture-data/web_scrape.py
Original file line number Diff line number Diff line change
Expand Up @@ -34,13 +34,13 @@ def get_occupancy_boulderwelt(gym_name:str, url: str) -> tuple():
if 'queue' in data and int(data['queue']) > 0:
occupancy += int(data['queue']) / 10
return occupancy

# admin-ajax.php not working
page = requests.get(url)
if page.status_code != 200:
return 0
try:
occupancy = int(float(re.search(r'style="margin-left:(.*?)%"', page.text).group(1)))
occupancy = int(float(re.search(r'style="margin-left:(.*?)%"', page.text)[1]))
Comment on lines -37 to +43

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function get_occupancy_boulderwelt refactored with the following changes:

except:
occupancy = 0
return occupancy
Expand Down Expand Up @@ -68,7 +68,13 @@ def get_occupancy_boulderado(gym_name:str, url: str) -> tuple():
return 0
soup = BeautifulSoup(page.content, 'html.parser')
try:
occupancy = int(re.search(r'left: (\d*)%', str(soup.find_all("div", class_="pointer-image")[0]['style'])).group(1))
occupancy = int(
re.search(
r'left: (\d*)%',
str(soup.find_all("div", class_="pointer-image")[0]['style']),
)[1]
)

Comment on lines -71 to +77

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function get_occupancy_boulderado refactored with the following changes:

except:
occupancy = 0
return occupancy
Expand Down Expand Up @@ -181,8 +187,8 @@ def get_occupancy_einstein(gym_name:str, url: str) -> tuple():
frame = soup.select("iframe")[0]
frame_url = urllib.parse.urljoin(url, frame["src"])
response = session.get(frame_url)
frame_soup = BeautifulSoup(response.content, 'html.parser')
occupancy = re.search(r'left: (\d+)%', str(frame_soup)).group(1)
frame_soup = BeautifulSoup(response.content, 'html.parser')
occupancy = re.search(r'left: (\d+)%', str(frame_soup))[1]
Comment on lines -184 to +191

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function get_occupancy_einstein refactored with the following changes:

except:
occupancy = 0
return occupancy
Expand Down Expand Up @@ -216,12 +222,20 @@ def scrape_websites(current_time: str, gymdatadf: pd.DataFrame) -> pd.DataFrame:
scrape_data = globals()[gym_data['function']]
occupancy = scrape_data(gym_name, gym_data['url'])
print(f"{gym_name}: occupancy={occupancy}, temp={weather_temp}, status={weather_status}")
if occupancy == 0 or occupancy == '0/0':
if occupancy in [0, '0/0']:
continue
webdata.append((current_time, gym_name, occupancy, weather_temp, weather_status))

webdf = pd.DataFrame(data=webdata, columns=['time', 'gym_name', 'occupancy', 'weather_temp', 'weather_status'])
return webdf

return pd.DataFrame(
data=webdata,
columns=[
'time',
'gym_name',
'occupancy',
'weather_temp',
'weather_status',
],
)
Comment on lines -219 to +238

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function scrape_websites refactored with the following changes:



def get_current_time():
Expand All @@ -242,9 +256,7 @@ def get_current_time():
current_min = int(dt.split(':')[1])
distances = [abs(current_min - _min) for _min in minutes]
closest_min = minutes[distances.index(min(distances))]
current_time = dt.replace(':'+str(current_min), ':'+str(closest_min))

return current_time
return dt.replace(f':{current_min}', f':{str(closest_min)}')
Comment on lines -245 to +259

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function get_current_time refactored with the following changes:



def lambda_handler(event, context):
Expand Down
7 changes: 2 additions & 5 deletions src/fix_missing_values.py
Original file line number Diff line number Diff line change
Expand Up @@ -100,7 +100,7 @@ def fill_nan_values(df: pd.DataFrame) -> pd.DataFrame:
for column in ['occupancy', 'waiting', 'weather_temp']:
# this might work with the dev version of scipy
func = interp1d(x_values, df_gym_day_noNan[column], kind='linear', bounds_error=False, fill_value='extrapolate')

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function fill_nan_values refactored with the following changes:

#fill in the interpolated value
timestamp_unix = (timestamp-pd.Timestamp("1970-01-01")) // pd.Timedelta('1s')
values = func([timestamp_unix])
Expand All @@ -120,10 +120,7 @@ def fill_nan_values(df: pd.DataFrame) -> pd.DataFrame:
df_gym_day_oneNan = df_gym_day_noNan.append(df.loc[index], ignore_index=False)
df_gym_day_oneNan = df_gym_day_oneNan.sort_values(by='current_time', ascending=True)
#Case 1: it is at the beginning:
if df_gym_day_oneNan.iloc[0].isnull().sum() == 1:
shift = -1
else:
shift = 1
shift = -1 if df_gym_day_oneNan.iloc[0].isnull().sum() == 1 else 1
df_gym_day_oneNan_shift = df_gym_day_oneNan.shift(shift)
df.loc[index, column] = df_gym_day_oneNan_shift.loc[index, column]
return df
Expand Down
29 changes: 16 additions & 13 deletions src/test_fix_missing_values.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,8 +16,7 @@
class Testget_missing_aquisition_timestamps(unittest.TestCase):

def gaussian(self, n, sigma, shift):
gauss = np.exp(-0.5*((n-shift)/sigma)**2)
return gauss
return np.exp(-0.5*((n-shift)/sigma)**2)

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function Testget_missing_aquisition_timestamps.gaussian refactored with the following changes:


def plot_gaussian(self):
import matplotlib.pyplot as plt
Expand All @@ -37,13 +36,13 @@ def get_testset(self):
gym1 = 'muenchen-ost'
gym2 = 'frankfurt'
gyms_n = [gym1, gym1, gym1, gym2, gym2, gym2]
gyms = [[gyms_n[g] for n in day] for g,day in enumerate(days)]
gyms = [[gyms_n[g] for _ in day] for g,day in enumerate(days)]
#occupancies=[[float(random.randint(0,100)) for n in day] for day in days]
#making a gauss for occupancies
occupancies = [self.gaussian(np.arange(0,len(day),1),len(day)/5,len(day)/2) for day in days]
#waitings=[[float(random.randint(0,15)) for n in day] for day in days]
waitings =[self.gaussian(np.arange(0,len(day),1),len(day)/5,len(day)/2) for day in days]
temps = [[float(random.randint(0,15)) for n in day] for day in days]
temps = [[float(random.randint(0,15)) for _ in day] for day in days]
Comment on lines -40 to +45

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function Testget_missing_aquisition_timestamps.get_testset refactored with the following changes:

temps = [self.gaussian(np.arange(0,len(day),1),len(day)/5,len(day)/2)-0.5 for day in days]
#statuss=[[random.choice(['Clouds', 'Rain', 'Drizzle', 'Clear', 'Thunderstorm', 'Mist']) for n in day] for day in days]
statuss = [[random.choice(['Clouds', 'Rain', 'Drizzle', 'Clear', 'Thunderstorm', 'Mist'])]*len(day) for day in days]
Expand Down Expand Up @@ -110,7 +109,7 @@ def test_testset(self):
def test_add_missing_timestamps(self):
df = self.get_testset()
#remove 8 values at random
for n in range(8):
for _ in range(8):
Comment on lines -113 to +112

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function Testget_missing_aquisition_timestamps.test_add_missing_timestamps refactored with the following changes:

df = df.drop([df.index[random.randint(0, len(df)-1)]])
df2 = add_missing_timestamps(df,
interval='20min',
Expand All @@ -121,7 +120,7 @@ def test_add_missing_timestamps(self):
def test_fill_nan(self):
df = self.get_testset()
#remove 8 values at random
for n in range(8):
for _ in range(8):
Comment on lines -124 to +123

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function Testget_missing_aquisition_timestamps.test_fill_nan refactored with the following changes:

df = df.drop([df.index[random.randint(0, len(df)-1)]])
df2 = add_missing_timestamps(df,
interval='20min',
Expand All @@ -133,7 +132,7 @@ def test_fill_nan(self):
def test_drop_additional(self):
df = self.get_testset()
#change the time of 8 values at random
for n in range(8):
for _ in range(8):
Comment on lines -136 to +135

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function Testget_missing_aquisition_timestamps.test_drop_additional refactored with the following changes:

df.current_time[random.randint(0, len(df)-1)]+=pd.to_timedelta(16, unit='min')
df2 = add_missing_timestamps(df,
interval='20min',
Expand All @@ -152,21 +151,25 @@ def test_correct_value_recovered(self):
df = df_o[:]
#remove 8 values at random
dropped = []
for n in range(8):
for _ in range(8):
drop = df.index[random.randint(0, len(df)-1)]
dropped.append(df.loc[drop])
df = df.drop([drop])

df2 = add_missing_timestamps(df,
interval='20min',
sample_start='07:20',
sample_end='23:40')
df2 = fill_nan_values(df2)
#get the values it recovered:
recovered=[]
for n, drop in enumerate(dropped):
recovered.append(df2[(df2.gym_name == drop.gym_name)&
(df2.current_time == drop.current_time)])
recovered = [
df2[
(df2.gym_name == drop.gym_name)
& (df2.current_time == drop.current_time)
]
for drop in dropped
]

Comment on lines -155 to +172

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function Testget_missing_aquisition_timestamps.test_correct_value_recovered refactored with the following changes:

'''
#in case of visual inspection
plt.figure()
Expand Down
17 changes: 9 additions & 8 deletions src/visualize_data.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,12 +118,15 @@ def given_day(boulderdf: pd.DataFrame, date: str, gym: str) -> pd.DataFrame:
def plot_data(df: pd.DataFrame):
# to plot several graphs in the y axis, melt the df
df = df.melt('time', var_name='name', value_name='occupancy')
chart = alt.Chart(df).mark_line(interpolate='basis').encode(
x=alt.X('time:N', axis=alt.Axis(grid=True)),
y=alt.Y('occupancy:Q', scale=alt.Scale(domain=[0, 100])),
color=alt.Color("name:N")
return (
alt.Chart(df)
.mark_line(interpolate='basis')
.encode(
x=alt.X('time:N', axis=alt.Axis(grid=True)),
y=alt.Y('occupancy:Q', scale=alt.Scale(domain=[0, 100])),
color=alt.Color("name:N"),
)
)
return chart
Comment on lines -121 to -126

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function plot_data refactored with the following changes:



def preprocess_current_data(boulderdf: pd.DataFrame, selected_gym: str, current_time: datetime.date):
Expand All @@ -149,6 +152,4 @@ def preprocess_current_data(boulderdf: pd.DataFrame, selected_gym: str, current_
today_data[onehotlist.index('weather_temp')] = latest_gym_entry['weather_temp']
today_data[onehotlist.index(latest_gym_entry['weather_status'])] = 1

# convert to numpy array
X_today = [np.asarray(today_data)]
return X_today
return [np.asarray(today_data)]
Comment on lines -152 to +155

Copy link
Copy Markdown
Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Function preprocess_current_data refactored with the following changes:

This removes the following comments ( why? ):

# convert to numpy array