Files
museum_analytics/notebooks/work/museum_regression.ipynb
T
2026-10-08 15:15:39 -04:00

13 KiB

Museum visitors vs. city population

This notebook talks to the museum_api FastAPI server over HTTP: it starts the server (or uses one you already run), sends it the data, and plots the regression it returns.

Setup: put ../../api/app/museum_api.py next to this notebook, then pip install scikit-learn pandas fastapi uvicorn requests matplotlib.

In [1]:
import threading, time
import requests
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
from matplotlib.ticker import FuncFormatter

API = "http://127.0.0.1:8000"   # change if your server runs elsewhere
---------------------------------------------------------------------------
ModuleNotFoundError                       Traceback (most recent call last)
Cell In[1], line 5
      1 import threading, time
      2 import requests
      3 import numpy as np
      4 import pandas as pd
----> 5 import matplotlib.pyplot as plt
      6 from matplotlib.ticker import FuncFormatter
      7 
      8 API = "http://127.0.0.1:8000"   # change if your server runs elsewhere

ModuleNotFoundError: No module named 'matplotlib'

1. Start the server

If a server is already answering at API (e.g. you started it with uvicorn museum_api:app in a terminal), this cell just uses it. Otherwise it launches one in a background thread inside this kernel. Restart the kernel to stop it.

In [2]:
def server_up():
    try:
        return requests.get(f"{API}/health", timeout=1).ok
    except requests.ConnectionError:
        return False

if not server_up():
    import uvicorn
    from api.app.museum_api import create_app
    server = uvicorn.Server(uvicorn.Config(create_app(), host="127.0.0.1", port=8000, log_level="warning"))
    threading.Thread(target=server.run, daemon=True).start()
    for _ in range(50):
        if server_up(): break
        time.sleep(0.1)

requests.get(f"{API}/health").json()
---------------------------------------------------------------------------
NameError                                 Traceback (most recent call last)
Cell In[2], line 7
      3         return requests.get(f"{API}/health", timeout=1).ok
      4     except requests.ConnectionError:
      5         return False
      6 
----> 7 if not server_up():
      8     import uvicorn
      9     from museum_api import create_app
     10     server = uvicorn.Server(uvicorn.Config(create_app(), host="127.0.0.1", port=8000, log_level="warning"))

Cell In[2], line 4, in server_up()
      1 def server_up():
      2     try:
      3         return requests.get(f"{API}/health", timeout=1).ok
----> 4     except requests.ConnectionError:
      5         return False

NameError: name 'API' is not defined

2. Load your data

Replace this with however you build your DataFrame. It needs the columns Name, Visitors, City, Country, Population.

In [ ]:
df = pd.read_csv("museums.csv")
df.head()

3. Train the model through the API

scale="log" fits log(visitors) ~ log(population). Set aggregate_by_city=True to count each city once instead of once per museum.

In [ ]:
def train(df, scale="log", aggregate_by_city=False):
    records = df[["Name", "Visitors", "City", "Country", "Population"]].dropna().to_dict("records")
    r = requests.post(f"{API}/train", json={"records": records, "scale": scale,
                                            "aggregate_by_city": aggregate_by_city})
    r.raise_for_status()
    return r.json()

summary = train(df, scale="log")
pd.Series(summary).to_frame("value")

4. Plot the regression

In [ ]:
BLUE, ORANGE, INK, MUTED = "#2a78d6", "#eb6834", "#0b0b0b", "#52514e"
plt.rcParams.update({"axes.spines.top": False, "axes.spines.right": False,
                     "axes.edgecolor": MUTED, "axes.labelcolor": INK,
                     "xtick.color": MUTED, "ytick.color": MUTED,
                     "axes.grid": True, "axes.axisbelow": True, "grid.color": "#e6e5e0", "grid.linewidth": 0.8,
                     "figure.dpi": 110})

def human(x, _=None):
    for div, suf in ((1e9, "B"), (1e6, "M"), (1e3, "K")):
        if abs(x) >= div: return f"{x/div:g}{suf}"
    return f"{x:g}"

points = pd.DataFrame(requests.get(f"{API}/data").json())

# Fitted curve: ask the API to predict across the population range
grid = np.geomspace(points.Population.min(), points.Population.max(), 100)
curve = pd.DataFrame(requests.post(f"{API}/predict", json={"populations": grid.tolist()}).json())

fig, ax = plt.subplots(figsize=(9, 6))
ax.scatter(points.Population, points.Visitors, s=40, color=BLUE, alpha=0.75,
           edgecolor="white", linewidth=1, label="Museums", zorder=3)
ax.plot(curve.population, curve.predicted_visitors, color=ORANGE, lw=2,
        label=f"Fit: {summary['equation']}", zorder=4)

# Label the 5 biggest outliers (furthest from the line on the log scale)
points["logres"] = np.log10(points.Visitors / points.Predicted)
for _, r in points.reindex(points.logres.abs().sort_values(ascending=False).index).head(5).iterrows():
    ax.annotate(r.Name, (r.Population, r.Visitors), xytext=(6, 4),
                textcoords="offset points", fontsize=8, color=MUTED)

if summary["scale"] == "log":
    ax.set_xscale("log"); ax.set_yscale("log")
ax.xaxis.set_major_formatter(FuncFormatter(human))
ax.yaxis.set_major_formatter(FuncFormatter(human))
ax.set_xlabel("City population"); ax.set_ylabel("Annual visitors")
cv = summary["cv_r2_5fold"]
ax.set_title(f"Visitors vs. city population   R² = {summary['r2']:.2f}"
             + (f"   (cross-validated {cv:.2f})" if cv is not None else ""),
             loc="left", fontsize=12, color=INK)
ax.legend(frameon=False, loc="upper left")
plt.tight_layout(); plt.show()

print(summary["interpretation"])

5. Which museums beat (or miss) their city's size?

Ratio = actual ÷ predicted visitors. Above 1× means the museum draws more than its city's size alone would suggest.

In [ ]:
N = 8
over  = pd.DataFrame(requests.get(f"{API}/residuals", params={"top": N, "order": "over"}).json())
under = pd.DataFrame(requests.get(f"{API}/residuals", params={"top": N, "order": "under"}).json())
res = pd.concat([over, under.iloc[::-1]]).drop_duplicates("Name").iloc[::-1]

fig, ax = plt.subplots(figsize=(9, 0.35 * len(res) + 1))
colors = [BLUE if r >= 1 else ORANGE for r in res.Ratio]
ax.barh(res.Name + " (" + res.City + ")", np.log10(res.Ratio), color=colors, height=0.7)
ax.axvline(0, color=MUTED, lw=1)
ticks = np.array([0.1, 0.2, 0.5, 1, 2, 5, 10])
ticks = ticks[(np.log10(ticks) >= np.log10(res.Ratio).min() - 0.1) &
              (np.log10(ticks) <= np.log10(res.Ratio).max() + 0.1)]
ax.set_xticks(np.log10(ticks), [f"{t:g}×" for t in ticks])
ax.set_xlabel("Actual ÷ predicted visitors (log scale)")
ax.set_title("Over-performers (blue) and under-performers (orange)", loc="left", fontsize=12, color=INK)
ax.grid(axis="y", visible=False)
plt.tight_layout(); plt.show()

res[["Name", "City", "Country", "Population", "Visitors", "Predicted", "Ratio"]].round(2)

6. Compare model variants

Retrains the server with each setting and collects the fit statistics. The last setting tried stays loaded, so the final call re-trains the default.

In [ ]:
rows = []
for scale in ("linear", "log"):
    for agg in (False, True):
        s = train(df, scale=scale, aggregate_by_city=agg)
        rows.append({k: s[k] for k in ("scale", "aggregate_by_city", "n_samples",
                                       "r2", "cv_r2_5fold", "pearson_r", "spearman_rho")})
train(df, scale="log")   # restore the default model
pd.DataFrame(rows).round(3)

7. Predict for any city

In [ ]:
cities = {"Montréal": 1_780_000, "Lyon": 520_000, "Tokyo": 14_000_000}
pred = requests.post(f"{API}/predict", json={"populations": list(cities.values())}).json()
pd.DataFrame(pred, index=cities.keys()).round(0)