restructure project with seperate app/ and notebooks/ sub directories
This commit is contained in:
1 parent
2ac6ac005a
commit
d5da086e66
16 files changed
+40
-14
No files matched your search
@@ -0,0 +1,115 @@
|
||||
import os
|
||||
import requests
|
||||
from dotenv import load_dotenv
|
||||
from app import constants
|
||||
from io import StringIO
|
||||
import pandas as pd
|
||||
|
||||
load_dotenv()
|
||||
|
||||
WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT")
|
||||
|
||||
|
||||
class TableNotFound(Exception):
|
||||
pass
|
||||
|
||||
|
||||
def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
|
||||
"""Get data from a wikipedia page matching the provided title.
|
||||
|
||||
Args:
|
||||
page_title: the title of the wikipedia page.
|
||||
|
||||
Returns:
|
||||
A pandas data frame of the found table.
|
||||
"""
|
||||
|
||||
page_url = f"{constants.WIKIPEDIA_REST_API_URL_PREFIX}/{page_title}"
|
||||
print(f"Fetching data from Wikipedia: {page_url}")
|
||||
with requests.Session() as session:
|
||||
session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT
|
||||
html = session.get(page_url, timeout=30).text
|
||||
|
||||
tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"})
|
||||
# TODO: add a way to specifically fetch a table
|
||||
if not tables:
|
||||
raise TableNotFound(f"No tables found in {page_url}")
|
||||
|
||||
else:
|
||||
df = tables[0]
|
||||
# removing last row with "nan" values
|
||||
if df.iloc[-1].isna().all():
|
||||
df = df.iloc[:-1]
|
||||
return df
|
||||
|
||||
|
||||
def get_city_population_data_frame_from_raw_data() -> pd.DataFrame:
|
||||
"""Get the population data from a static csv file downloaded from
|
||||
https://simplemaps.com/data/world-cities website.
|
||||
|
||||
Returns:
|
||||
A data frame of the cities population data.
|
||||
"""
|
||||
raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE)
|
||||
raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1)
|
||||
raw_df["population"] = raw_df["population"].astype("Int64")
|
||||
wrong_washington_condition = (
|
||||
(raw_df["city"] == "Washington")
|
||||
& (raw_df["country"] == "United States")
|
||||
& (raw_df["admin_name"] != "District of Columbia")
|
||||
)
|
||||
cities_population = raw_df[~wrong_washington_condition]
|
||||
cities_population.rename(columns={"population": "Population"}, inplace=True)
|
||||
return cities_population
|
||||
|
||||
|
||||
def update_city_population(museum_dataframe: pd.DataFrame) -> pd.DataFrame:
|
||||
"""Add a population column corresponding to city name to the provided museum data frame from
|
||||
a population dataframe.
|
||||
|
||||
Args:
|
||||
museum_dataframe (): The museum data frame.
|
||||
|
||||
Returns:
|
||||
The updated museum data frame with population column.
|
||||
"""
|
||||
city_pop = get_city_population_data_frame_from_raw_data(museum_dataframe)
|
||||
museum_dataframe = pd.merge(
|
||||
museum_dataframe,
|
||||
city_pop,
|
||||
left_on=["City_Clean", "Country"],
|
||||
right_on=["city", "country"],
|
||||
how="left"
|
||||
)
|
||||
museum_dataframe.drop(columns=["city", "country"], inplace=True)
|
||||
|
||||
return museum_dataframe
|
||||
|
||||
|
||||
def get_museum_data() -> pd.DataFrame:
|
||||
"""Fetch the museum data from the Wikipedia page, Convert and transform to appropriate format and
|
||||
data types. Fixes ambiguity where city name may have a detailed syntax like Washington, D.C. keeping
|
||||
only the name Washington.
|
||||
|
||||
Returns:
|
||||
A pandas data frame of museum data.
|
||||
"""
|
||||
page_title = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL.rsplit("/", 1)[-1]
|
||||
museum_df = get_first_data_table_from_wikipedia(page_title=page_title)
|
||||
|
||||
city_corrections = {
|
||||
'Washington, D.C.': 'Washington',
|
||||
'New York City': 'New York',
|
||||
'Vatican City, Rome': 'Vatican City',
|
||||
'London, South Kensington': 'London',
|
||||
}
|
||||
museum_df["City_Clean"] = museum_df['City'].replace(city_corrections)
|
||||
museum_df["Visitors_clean"] = museum_df["Visitors"].str.replace(',', '', regex=False)
|
||||
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].str.extract(r'^(\d+)')
|
||||
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int)
|
||||
museum_df["Visitors"] = museum_df["Visitors_clean"]
|
||||
museum_df = museum_df.drop(columns=["Visitors_clean"])
|
||||
museum_df = update_city_population(museum_df)
|
||||
return museum_df
|
||||
|
||||
|
||||
Reference in new issue
Block a user