import os import requests from dotenv import load_dotenv from app import constants from io import StringIO import pandas as pd load_dotenv() WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT") class TableNotFound(Exception): pass def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame: """Get data from a wikipedia page matching the provided title. Args: page_title: the title of the wikipedia page. Returns: A pandas data frame of the found table. """ page_url = f"{constants.WIKIPEDIA_REST_API_URL_PREFIX}/{page_title}" print(f"Fetching data from Wikipedia: {page_url}") with requests.Session() as session: session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT html = session.get(page_url, timeout=30).text tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"}) # TODO: add a way to specifically fetch a table if not tables: raise TableNotFound(f"No tables found in {page_url}") else: df = tables[0] # removing last row with "nan" values if df.iloc[-1].isna().all(): df = df.iloc[:-1] return df def get_city_population_data_frame_from_raw_data() -> pd.DataFrame: """Get the population data from a static csv file downloaded from https://simplemaps.com/data/world-cities website. Returns: A data frame of the cities population data. """ raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE) raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1) raw_df["population"] = raw_df["population"].astype("Int64") wrong_washington_condition = ( (raw_df["city"] == "Washington") & (raw_df["country"] == "United States") & (raw_df["admin_name"] != "District of Columbia") ) cities_population = raw_df[~wrong_washington_condition] cities_population.rename(columns={"population": "Population"}, inplace=True) return cities_population def update_city_population(museum_dataframe: pd.DataFrame) -> pd.DataFrame: """Add a population column corresponding to city name to the provided museum data frame from a population dataframe. Args: museum_dataframe (): The museum data frame. Returns: The updated museum data frame with population column. """ city_pop = get_city_population_data_frame_from_raw_data(museum_dataframe) museum_dataframe = pd.merge( museum_dataframe, city_pop, left_on=["City_Clean", "Country"], right_on=["city", "country"], how="left" ) museum_dataframe.drop(columns=["city", "country"], inplace=True) return museum_dataframe def get_museum_data() -> pd.DataFrame: """Fetch the museum data from the Wikipedia page, Convert and transform to appropriate format and data types. Fixes ambiguity where city name may have a detailed syntax like Washington, D.C. keeping only the name Washington. Returns: A pandas data frame of museum data. """ page_title = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL.rsplit("/", 1)[-1] museum_df = get_first_data_table_from_wikipedia(page_title=page_title) city_corrections = { 'Washington, D.C.': 'Washington', 'New York City': 'New York', 'Vatican City, Rome': 'Vatican City', 'London, South Kensington': 'London', } museum_df["City_Clean"] = museum_df['City'].replace(city_corrections) museum_df["Visitors_clean"] = museum_df["Visitors"].str.replace(',', '', regex=False) museum_df["Visitors_clean"] = museum_df["Visitors_clean"].str.extract(r'^(\d+)') museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int) museum_df["Visitors"] = museum_df["Visitors_clean"] museum_df = museum_df.drop(columns=["Visitors_clean"]) museum_df = update_city_population(museum_df) return museum_df