doc string tweaks, cleanup code
This commit is contained in:
1 parent
82a1bdb0a5
commit
38d4daa63a
2 files changed
+51
-29
No files matched your search
@@ -1,7 +1,6 @@
|
|||||||
|
|
||||||
|
|
||||||
DEFAULT_MUSEUM_DATA_SOURCE_URL = "https://en.wikipedia.org/wiki/List_of_most_visited_museums"
|
DEFAULT_MUSEUM_DATA_SOURCE_URL = "https://en.wikipedia.org/wiki/List_of_most_visited_museums"
|
||||||
|
WIKIPEDIA_REST_API_URL_PREFIX = "https://en.wikipedia.org/api/rest_v1/page/html"
|
||||||
|
|
||||||
# csv file from https://simplemaps.com/data/world-cities
|
# csv file downloaded from https://simplemaps.com/data/world-cities
|
||||||
RAW_POPULATION_DATA_FILE= "data/worldcities.csv"
|
RAW_POPULATION_DATA_FILE = "data/worldcities.csv"
|
||||||
MUSEUM_DATA_FILE= "data/updated_museum_data.csv"
|
MUSEUM_DATA_FILE = "data/updated_museum_data.csv"
|
||||||
@@ -1,7 +1,5 @@
|
|||||||
import os
|
import os
|
||||||
import requests
|
import requests
|
||||||
import wikipediaapi
|
|
||||||
from bs4 import BeautifulSoup
|
|
||||||
from dotenv import load_dotenv
|
from dotenv import load_dotenv
|
||||||
from museum_analytics import constants
|
from museum_analytics import constants
|
||||||
from io import StringIO
|
from io import StringIO
|
||||||
@@ -10,17 +8,23 @@ import pandas as pd
|
|||||||
load_dotenv()
|
load_dotenv()
|
||||||
|
|
||||||
WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT")
|
WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT")
|
||||||
SOUP_PER_SECTION: dict[str, BeautifulSoup] = {}
|
|
||||||
|
|
||||||
|
class TableNotFound(Exception):
|
||||||
|
pass
|
||||||
|
|
||||||
|
|
||||||
def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
|
def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
|
||||||
|
"""Get data from a wikipedia page matching the provided title.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
page_title: the title of the wikipedia page.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
A pandas data frame of the found table.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
:param page_title:
|
page_url = f"{constants.WIKIPEDIA_REST_API_URL_PREFIX}/{page_title}"
|
||||||
:return:
|
|
||||||
"""
|
|
||||||
|
|
||||||
page_url = f"https://en.wikipedia.org/api/rest_v1/page/html/{page_title}"
|
|
||||||
print(f"Fetching data from Wikipedia: {page_url}")
|
print(f"Fetching data from Wikipedia: {page_url}")
|
||||||
with requests.Session() as session:
|
with requests.Session() as session:
|
||||||
session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT
|
session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT
|
||||||
@@ -28,18 +32,23 @@ def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
|
|||||||
|
|
||||||
tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"})
|
tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"})
|
||||||
# TODO: add a way to specifically fetch a table
|
# TODO: add a way to specifically fetch a table
|
||||||
df = tables[0]
|
if not tables:
|
||||||
|
raise TableNotFound(f"No tables found in {page_url}")
|
||||||
|
|
||||||
|
else:
|
||||||
|
df = tables[0]
|
||||||
|
# removing last row with "nan" values
|
||||||
if df.iloc[-1].isna().all():
|
if df.iloc[-1].isna().all():
|
||||||
df = df.iloc[:-1]
|
df = df.iloc[:-1]
|
||||||
return df
|
return df
|
||||||
|
|
||||||
|
|
||||||
def _get_city_population_data_frame_from_raw_data(museum_cities_df: pd.DataFrame) -> pd.DataFrame:
|
def get_city_population_data_frame_from_raw_data() -> pd.DataFrame:
|
||||||
"""
|
"""Get the population data from a static csv file downloaded from
|
||||||
|
https://simplemaps.com/data/world-cities website.
|
||||||
|
|
||||||
:param museum_cities_df:
|
Returns:
|
||||||
:return:
|
A data frame of the cities population data.
|
||||||
"""
|
"""
|
||||||
raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE)
|
raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE)
|
||||||
raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1)
|
raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1)
|
||||||
@@ -54,25 +63,39 @@ def _get_city_population_data_frame_from_raw_data(museum_cities_df: pd.DataFrame
|
|||||||
return cities_population
|
return cities_population
|
||||||
|
|
||||||
|
|
||||||
def _update_city_population(museum_df) -> pd.DataFrame:
|
def update_city_population(museum_dataframe: pd.DataFrame) -> pd.DataFrame:
|
||||||
city_pop = _get_city_population_data_frame_from_raw_data(museum_df)
|
"""Add a population column corresponding to city name to the provided museum data frame from
|
||||||
museum_df = pd.merge(
|
a population dataframe.
|
||||||
museum_df,
|
|
||||||
|
Args:
|
||||||
|
museum_dataframe (): The museum data frame.
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
The updated museum data frame with population column.
|
||||||
|
"""
|
||||||
|
city_pop = get_city_population_data_frame_from_raw_data(museum_dataframe)
|
||||||
|
museum_dataframe = pd.merge(
|
||||||
|
museum_dataframe,
|
||||||
city_pop,
|
city_pop,
|
||||||
left_on=["City_Clean", "Country"],
|
left_on=["City_Clean", "Country"],
|
||||||
right_on=["city", "country"],
|
right_on=["city", "country"],
|
||||||
how="left"
|
how="left"
|
||||||
)
|
)
|
||||||
museum_df.drop(columns=["city", "country"], inplace=True)
|
museum_dataframe.drop(columns=["city", "country"], inplace=True)
|
||||||
|
|
||||||
return museum_df
|
return museum_dataframe
|
||||||
|
|
||||||
|
|
||||||
def get_museum_data() -> pd.DataFrame:
|
def get_museum_data() -> pd.DataFrame:
|
||||||
url = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL
|
"""Fetch the museum data from the Wikipedia page, Convert and transform to appropriate format and
|
||||||
page = url.rsplit("/", 1)[-1]
|
data types. Fixes ambiguity where city name may have a detailed syntax like Washington, D.C. keeping
|
||||||
museum_df = get_first_data_table_from_wikipedia(page_title=page)
|
only the name Washington.
|
||||||
# print(museum_df.to_string())
|
|
||||||
|
Returns:
|
||||||
|
A pandas data frame of museum data.
|
||||||
|
"""
|
||||||
|
page_title = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL.rsplit("/", 1)[-1]
|
||||||
|
museum_df = get_first_data_table_from_wikipedia(page_title=page_title)
|
||||||
|
|
||||||
city_corrections = {
|
city_corrections = {
|
||||||
'Washington, D.C.': 'Washington',
|
'Washington, D.C.': 'Washington',
|
||||||
@@ -86,7 +109,7 @@ def get_museum_data() -> pd.DataFrame:
|
|||||||
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int)
|
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int)
|
||||||
museum_df["Visitors"] = museum_df["Visitors_clean"]
|
museum_df["Visitors"] = museum_df["Visitors_clean"]
|
||||||
museum_df = museum_df.drop(columns=["Visitors_clean"])
|
museum_df = museum_df.drop(columns=["Visitors_clean"])
|
||||||
museum_df = _update_city_population(museum_df)
|
museum_df = update_city_population(museum_df)
|
||||||
return museum_df
|
return museum_df
|
||||||
|
|
||||||
|
|
||||||
Reference in new issue
Block a user