doc string tweaks, cleanup code

This commit is contained in:
hpayer committed 2026-10-08 10:41:57 -04:00
1 parent 82a1bdb0a5
commit 38d4daa63a
2 files changed
+49 -27

No files matched your search

+2 -3
View File
@@ -1,7 +1,6 @@
DEFAULT_MUSEUM_DATA_SOURCE_URL = "https://en.wikipedia.org/wiki/List_of_most_visited_museums" DEFAULT_MUSEUM_DATA_SOURCE_URL = "https://en.wikipedia.org/wiki/List_of_most_visited_museums"
WIKIPEDIA_REST_API_URL_PREFIX = "https://en.wikipedia.org/api/rest_v1/page/html"
# csv file from https://simplemaps.com/data/world-cities # csv file downloaded from https://simplemaps.com/data/world-cities
RAW_POPULATION_DATA_FILE = "data/worldcities.csv" RAW_POPULATION_DATA_FILE = "data/worldcities.csv"
MUSEUM_DATA_FILE = "data/updated_museum_data.csv" MUSEUM_DATA_FILE = "data/updated_museum_data.csv"
+47 -24
View File
@@ -1,7 +1,5 @@
import os import os
import requests import requests
import wikipediaapi
from bs4 import BeautifulSoup
from dotenv import load_dotenv from dotenv import load_dotenv
from museum_analytics import constants from museum_analytics import constants
from io import StringIO from io import StringIO
@@ -10,17 +8,23 @@ import pandas as pd
load_dotenv() load_dotenv()
WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT") WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT")
SOUP_PER_SECTION: dict[str, BeautifulSoup] = {}
class TableNotFound(Exception):
pass
def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame: def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
"""Get data from a wikipedia page matching the provided title.
Args:
page_title: the title of the wikipedia page.
Returns:
A pandas data frame of the found table.
""" """
:param page_title: page_url = f"{constants.WIKIPEDIA_REST_API_URL_PREFIX}/{page_title}"
:return:
"""
page_url = f"https://en.wikipedia.org/api/rest_v1/page/html/{page_title}"
print(f"Fetching data from Wikipedia: {page_url}") print(f"Fetching data from Wikipedia: {page_url}")
with requests.Session() as session: with requests.Session() as session:
session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT
@@ -28,18 +32,23 @@ def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"}) tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"})
# TODO: add a way to specifically fetch a table # TODO: add a way to specifically fetch a table
df = tables[0] if not tables:
raise TableNotFound(f"No tables found in {page_url}")
else:
df = tables[0]
# removing last row with "nan" values
if df.iloc[-1].isna().all(): if df.iloc[-1].isna().all():
df = df.iloc[:-1] df = df.iloc[:-1]
return df return df
def _get_city_population_data_frame_from_raw_data(museum_cities_df: pd.DataFrame) -> pd.DataFrame: def get_city_population_data_frame_from_raw_data() -> pd.DataFrame:
""" """Get the population data from a static csv file downloaded from
https://simplemaps.com/data/world-cities website.
:param museum_cities_df: Returns:
:return: A data frame of the cities population data.
""" """
raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE) raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE)
raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1) raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1)
@@ -54,25 +63,39 @@ def _get_city_population_data_frame_from_raw_data(museum_cities_df: pd.DataFrame
return cities_population return cities_population
def _update_city_population(museum_df) -> pd.DataFrame: def update_city_population(museum_dataframe: pd.DataFrame) -> pd.DataFrame:
city_pop = _get_city_population_data_frame_from_raw_data(museum_df) """Add a population column corresponding to city name to the provided museum data frame from
museum_df = pd.merge( a population dataframe.
museum_df,
Args:
museum_dataframe (): The museum data frame.
Returns:
The updated museum data frame with population column.
"""
city_pop = get_city_population_data_frame_from_raw_data(museum_dataframe)
museum_dataframe = pd.merge(
museum_dataframe,
city_pop, city_pop,
left_on=["City_Clean", "Country"], left_on=["City_Clean", "Country"],
right_on=["city", "country"], right_on=["city", "country"],
how="left" how="left"
) )
museum_df.drop(columns=["city", "country"], inplace=True) museum_dataframe.drop(columns=["city", "country"], inplace=True)
return museum_df return museum_dataframe
def get_museum_data() -> pd.DataFrame: def get_museum_data() -> pd.DataFrame:
url = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL """Fetch the museum data from the Wikipedia page, Convert and transform to appropriate format and
page = url.rsplit("/", 1)[-1] data types. Fixes ambiguity where city name may have a detailed syntax like Washington, D.C. keeping
museum_df = get_first_data_table_from_wikipedia(page_title=page) only the name Washington.
# print(museum_df.to_string())
Returns:
A pandas data frame of museum data.
"""
page_title = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL.rsplit("/", 1)[-1]
museum_df = get_first_data_table_from_wikipedia(page_title=page_title)
city_corrections = { city_corrections = {
'Washington, D.C.': 'Washington', 'Washington, D.C.': 'Washington',
@@ -86,7 +109,7 @@ def get_museum_data() -> pd.DataFrame:
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int) museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int)
museum_df["Visitors"] = museum_df["Visitors_clean"] museum_df["Visitors"] = museum_df["Visitors_clean"]
museum_df = museum_df.drop(columns=["Visitors_clean"]) museum_df = museum_df.drop(columns=["Visitors_clean"])
museum_df = _update_city_population(museum_df) museum_df = update_city_population(museum_df)
return museum_df return museum_df