import os import requests import wikipediaapi from bs4 import BeautifulSoup from dotenv import load_dotenv from api.app import constants from io import StringIO import pandas as pd load_dotenv() WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT") SOUP_PER_SECTION: dict[str, BeautifulSoup] = {} def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame: """ :param page_title: :return: """ page_url = f"https://en.wikipedia.org/api/rest_v1/page/html/{page_title}" print(f"Fetching data from Wikipedia: {page_url}") with requests.Session() as session: session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT html = session.get(page_url, timeout=30).text tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"}) # TODO: add a way to specifically fetch a table df = tables[0] if df.iloc[-1].isna().all(): df = df.iloc[:-1] return df def country_table(soup: BeautifulSoup, country: str): for h in soup.find_all(["h2", "h3"]): if h.get_text(strip=True) == country: # exact match, not substring tbl = h.find_next("table", class_="wikitable") return pd.read_html(StringIO(str(tbl)))[0] raise ValueError(f"No section for {country!r}") def get_country_cities_populations(country: str) -> pd.DataFrame | None: found_section = None for section in constants.CITY_POPULATION_WIKI_SUB_PAGE_NAMES: if country[0] in section: found_section = section break if not found_section: return soup = SOUP_PER_SECTION.get(found_section) if soup is None: page_title = f"{constants.CITY_POPULATION_WIKI_PAGE_NAME}: {found_section}" with requests.Session() as session: session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT response = session.get( "https://en.wikipedia.org/w/api.php", params={ "action": "parse", "page": page_title, "prop": "text", "format": "json", "formatversion": 2, }, timeout=30) response.raise_for_status() html = response.json()["parse"]["text"] soup = BeautifulSoup(html, "lxml") SOUP_PER_SECTION[found_section] = soup dataframe = country_table(soup, country) # Population column name includes the year, e.g. "Population (2021)" pop_col = next(c for c in dataframe.columns if str(c).startswith("Population")) dataframe = dataframe.rename(columns={pop_col: "Population"}) dataframe["Population"] = pd.to_numeric( dataframe["Population"].astype(str).str.replace(r"\[.*?\]|,", "", regex=True), errors="coerce", ) dataframe = dataframe.sort_values("Population", ascending=False) return dataframe def get_city_population_data_frame_from_wikipedia(museum_cities_df: pd.DataFrame) -> pd.DataFrame: print("Fetching all population data...") country_list = museum_cities_df["Country"].unique().tolist() data_frames = [] for country in country_list: dataframe = get_country_cities_populations(country) if dataframe is None: print(f"Cannot find population data for {country}") continue data_frames.append(dataframe) all_cities = pd.concat(data_frames) cities_to_filter = museum_cities_df["City"].unique().tolist() filtered_cities_population = all_cities[all_cities["City"].isin(cities_to_filter)] return filtered_cities_population def get_city_population_data_frame_from_raw_data(museum_cities_df: pd.DataFrame) -> pd.DataFrame: raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE) raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1) raw_df["population"] = raw_df["population"].astype("Int64") wrong_washington_condition = ( (raw_df["city"] == "Washington") & (raw_df["country"] == "United States") & (raw_df["admin_name"] != "District of Columbia") ) cities_population = raw_df[~wrong_washington_condition] cities_population.rename(columns={"population": "Population"},inplace=True) return cities_population def update_city_population(museum_df) -> pd.DataFrame: city_pop = get_city_population_data_frame_from_raw_data(museum_df) museum_df = pd.merge( museum_df, city_pop, left_on=["City_Clean", "Country"], right_on=["city", "country"], how="left" ) museum_df.drop(columns=["city", "country"], inplace=True) return museum_df def refresh_museum_data(): print("Refreshing museum data...") url = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL page = url.rsplit("/", 1)[-1] museum_df = get_first_data_table_from_wikipedia(page_title=page) # print(museum_df.to_string()) city_corrections = { 'Washington, D.C.': 'Washington', 'New York City': 'New York', 'Vatican City, Rome': 'Vatican City', 'London, South Kensington': 'London', } museum_df["City_Clean"] = museum_df['City'].replace(city_corrections) museum_df["Visitors_clean"] = museum_df["Visitors"].str.replace(',', '', regex=False) museum_df["Visitors_clean"] = museum_df["Visitors_clean"].str.extract(r'^(\d+)') museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int) museum_df["Visitors"] = museum_df["Visitors_clean"] museum_df = museum_df.drop(columns=["Visitors_clean"]) museum_df = update_city_population(museum_df) print(museum_df.to_string()) if __name__ == "__main__": refresh_museum_data()