restructure project with seperate app/ and notebooks/ sub directories
This commit is contained in:
1 parent
2ac6ac005a
commit
d5da086e66
16 files changed
+40
-14
No files matched your search
Whitespace-only changes.
@@ -0,0 +1,158 @@
|
||||
import os
|
||||
import requests
|
||||
import wikipediaapi
|
||||
from bs4 import BeautifulSoup
|
||||
from dotenv import load_dotenv
|
||||
from api.app import constants
|
||||
from io import StringIO
|
||||
import pandas as pd
|
||||
|
||||
load_dotenv()
|
||||
|
||||
WIKIPEDIA_USER_AGENT = os.getenv("WIKIPEDIA_USER_AGENT")
|
||||
SOUP_PER_SECTION: dict[str, BeautifulSoup] = {}
|
||||
|
||||
|
||||
def get_first_data_table_from_wikipedia(page_title: str) -> pd.DataFrame:
|
||||
"""
|
||||
|
||||
:param page_title:
|
||||
:return:
|
||||
"""
|
||||
|
||||
page_url = f"https://en.wikipedia.org/api/rest_v1/page/html/{page_title}"
|
||||
print(f"Fetching data from Wikipedia: {page_url}")
|
||||
with requests.Session() as session:
|
||||
session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT
|
||||
html = session.get(page_url, timeout=30).text
|
||||
|
||||
tables = pd.read_html(StringIO(html), attrs={"class": "wikitable"})
|
||||
# TODO: add a way to specifically fetch a table
|
||||
df = tables[0]
|
||||
|
||||
if df.iloc[-1].isna().all():
|
||||
df = df.iloc[:-1]
|
||||
return df
|
||||
|
||||
|
||||
def country_table(soup: BeautifulSoup, country: str):
|
||||
for h in soup.find_all(["h2", "h3"]):
|
||||
if h.get_text(strip=True) == country: # exact match, not substring
|
||||
tbl = h.find_next("table", class_="wikitable")
|
||||
return pd.read_html(StringIO(str(tbl)))[0]
|
||||
raise ValueError(f"No section for {country!r}")
|
||||
|
||||
|
||||
def get_country_cities_populations(country: str) -> pd.DataFrame | None:
|
||||
found_section = None
|
||||
for section in constants.CITY_POPULATION_WIKI_SUB_PAGE_NAMES:
|
||||
if country[0] in section:
|
||||
found_section = section
|
||||
break
|
||||
if not found_section:
|
||||
return
|
||||
|
||||
soup = SOUP_PER_SECTION.get(found_section)
|
||||
|
||||
if soup is None:
|
||||
page_title = f"{constants.CITY_POPULATION_WIKI_PAGE_NAME}: {found_section}"
|
||||
|
||||
with requests.Session() as session:
|
||||
session.headers["User-Agent"] = WIKIPEDIA_USER_AGENT
|
||||
response = session.get(
|
||||
"https://en.wikipedia.org/w/api.php", params={
|
||||
"action": "parse", "page": page_title, "prop": "text",
|
||||
"format": "json", "formatversion": 2,
|
||||
}, timeout=30)
|
||||
response.raise_for_status()
|
||||
html = response.json()["parse"]["text"]
|
||||
|
||||
soup = BeautifulSoup(html, "lxml")
|
||||
SOUP_PER_SECTION[found_section] = soup
|
||||
|
||||
dataframe = country_table(soup, country)
|
||||
|
||||
# Population column name includes the year, e.g. "Population (2021)"
|
||||
pop_col = next(c for c in dataframe.columns if str(c).startswith("Population"))
|
||||
dataframe = dataframe.rename(columns={pop_col: "Population"})
|
||||
dataframe["Population"] = pd.to_numeric(
|
||||
dataframe["Population"].astype(str).str.replace(r"\[.*?\]|,", "", regex=True),
|
||||
errors="coerce",
|
||||
)
|
||||
dataframe = dataframe.sort_values("Population", ascending=False)
|
||||
return dataframe
|
||||
|
||||
def get_city_population_data_frame_from_wikipedia(museum_cities_df: pd.DataFrame) -> pd.DataFrame:
|
||||
print("Fetching all population data...")
|
||||
country_list = museum_cities_df["Country"].unique().tolist()
|
||||
data_frames = []
|
||||
for country in country_list:
|
||||
dataframe = get_country_cities_populations(country)
|
||||
if dataframe is None:
|
||||
print(f"Cannot find population data for {country}")
|
||||
continue
|
||||
data_frames.append(dataframe)
|
||||
|
||||
all_cities = pd.concat(data_frames)
|
||||
cities_to_filter = museum_cities_df["City"].unique().tolist()
|
||||
filtered_cities_population = all_cities[all_cities["City"].isin(cities_to_filter)]
|
||||
return filtered_cities_population
|
||||
|
||||
|
||||
def get_city_population_data_frame_from_raw_data(museum_cities_df: pd.DataFrame) -> pd.DataFrame:
|
||||
|
||||
raw_df = pd.read_csv(constants.RAW_POPULATION_DATA_FILE)
|
||||
raw_df = raw_df.drop(["city_ascii", "lat", "lng", "iso2", "iso3", "capital", "id"], axis=1)
|
||||
raw_df["population"] = raw_df["population"].astype("Int64")
|
||||
wrong_washington_condition = (
|
||||
(raw_df["city"] == "Washington")
|
||||
& (raw_df["country"] == "United States")
|
||||
& (raw_df["admin_name"] != "District of Columbia")
|
||||
)
|
||||
cities_population = raw_df[~wrong_washington_condition]
|
||||
cities_population.rename(columns={"population": "Population"},inplace=True)
|
||||
return cities_population
|
||||
|
||||
def update_city_population(museum_df) -> pd.DataFrame:
|
||||
city_pop = get_city_population_data_frame_from_raw_data(museum_df)
|
||||
museum_df = pd.merge(
|
||||
museum_df,
|
||||
city_pop,
|
||||
left_on=["City_Clean", "Country"],
|
||||
right_on=["city", "country"],
|
||||
how="left"
|
||||
)
|
||||
museum_df.drop(columns=["city", "country"], inplace=True)
|
||||
|
||||
return museum_df
|
||||
|
||||
def refresh_museum_data():
|
||||
print("Refreshing museum data...")
|
||||
|
||||
url = constants.DEFAULT_MUSEUM_DATA_SOURCE_URL
|
||||
page = url.rsplit("/", 1)[-1]
|
||||
museum_df = get_first_data_table_from_wikipedia(page_title=page)
|
||||
# print(museum_df.to_string())
|
||||
|
||||
city_corrections = {
|
||||
'Washington, D.C.': 'Washington',
|
||||
'New York City': 'New York',
|
||||
'Vatican City, Rome': 'Vatican City',
|
||||
'London, South Kensington': 'London',
|
||||
}
|
||||
museum_df["City_Clean"] = museum_df['City'].replace(city_corrections)
|
||||
museum_df["Visitors_clean"] = museum_df["Visitors"].str.replace(',', '', regex=False)
|
||||
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].str.extract(r'^(\d+)')
|
||||
museum_df["Visitors_clean"] = museum_df["Visitors_clean"].astype(int)
|
||||
museum_df["Visitors"] = museum_df["Visitors_clean"]
|
||||
museum_df = museum_df.drop(columns=["Visitors_clean"])
|
||||
|
||||
museum_df = update_city_population(museum_df)
|
||||
|
||||
print(museum_df.to_string())
|
||||
|
||||
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
refresh_museum_data()
|
||||
Reference in new issue
Block a user