getting a bit further with the frontend and demoing the analysis to matti

This commit is contained in:
itsamejms
2025-09-19 14:57:03 +02:00
parent a44833d054
commit b2e412b3c5
8 changed files with 1354 additions and 233 deletions
+15 -231
View File
@@ -1,236 +1,20 @@
# %% [markdown]
# # Visualize Most Prepared Users
# This workflow loads user embeddings, generates a prompt embedding, computes similarity, and visualizes the most prepared users.
import logging
import uvicorn
import os
from api import app
# %%
import pandas as pd
import numpy as np
from sklearn.metrics.pairwise import cosine_similarity
from sklearn.decomposition import PCA
from sklearn.manifold import TSNE
import plotly.express as px
import matplotlib.pyplot as plt
import seaborn as sns
import umap
import umap.plot
from utils import get_ollama_embedding
# %% [markdown]
# ## Read in the file with user vectors
# Update the path/format as needed.
# %%
df = (
pd.read_csv("./data/eurobarometer_preparedness_model_data_v3.csv")
.reset_index()
.assign(
**{
"user_id": lambda x: x["index"].astype(str)
+ "_"
+ x["country_code_iso_3166"]
+ "_"
+ x["age_recoded_6_categories"].astype(str)
+ "_"
+ x["gender"].astype(str)
}
)
# select only relevant countries for visualization
.loc[
lambda x: x["country_code_iso_3166"].isin(
["FI", "DE-E", "DE-W", "FR", "ES", "PT"]
)
]
)
df.shape
# relabel None of the above/ Non binary/ do not recognize yourself in above categories/Prefer not to say to other
df["gender"] = df["gender"].replace({
"None of the above/ Non binary/ do not recognize yourself in above categories/Prefer not to say": "Other",
})
# %% [markdown]
# ## Convert string embeddings to lists if needed
# %%
def parse_embedding(x):
if isinstance(x, str):
return [float(i) for i in x.strip("[]").split(",")]
return x
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
df["embedding"] = df["embedding"].apply(parse_embedding)
def run_api() -> None:
"""Run the FastAPI application."""
if "james" in os.environ.get("USER", ""):
logger.info("Running in James's environment")
uvicorn.run("api:app", host="0.0.0.0", port=8080, reload=True)
else:
uvicorn.run(app, host="0.0.0.0", port=8080)
# %%
# Select best user:
# disaster_measures_in_hh_* == 1, and how_many_days_meet_* == (4 | 5)
# Following relabeling in data_preparation_raw.py:
# disaster_measures_in_hh_* != "Not mentioned"
# how_many_days_meet_* == "More than 7 days"
disaster_measures_in_hh_columns = [
col for col in df.columns if col.startswith("disaster_measures_in_hh_")
]
how_many_days_meet_columns = [
col for col in df.columns if col.startswith("how_many_days_meet_")
]
best_users = df[
(df[disaster_measures_in_hh_columns] != "Not mentioned").all(axis=1)
& (df[how_many_days_meet_columns] == "More than 7 days").all(axis=1)
]
print(f"Number of best users: {len(best_users)}")
best_users.head(5)[["user_id", "user_text"]].to_dict(orient="records")
# %% [markdown]
# ## Write your prompt and generate its embedding
# %%
# prompt = "The user is highly prepared for disasters, with emergency supplies and a clear plan."
prompt = best_users.head(1)["user_text"].values[
0
] # Example: use the first user's text as the prompt
prompt_embedding = get_ollama_embedding(prompt)
# %% [markdown]
# ## Compute similarity between each user and the prompt
# %%
user_embeddings = np.vstack(df["embedding"].values)
prompt_vec = np.array(prompt_embedding).reshape(1, -1)
similarities = cosine_similarity(user_embeddings, prompt_vec).flatten()
df["similarity"] = similarities
# %% [markdown]
# ## Visualize the users by similarity
# %%
_ = plt.figure(figsize=(5, 7))
_ = sns.violinplot(
x="similarity", y="country_code_iso_3166", hue="country_code_iso_3166", data=df
)
_ = sns.stripplot(
x="similarity",
y="country_code_iso_3166",
data=df,
hue="country_code_iso_3166",
alpha=0.8,
jitter=True,
linewidth=0.5,
edgecolor="white",
)
_ = plt.title("Distribution of User Similarities to Preparedness Prompt")
_ = plt.xlabel("Cosine Similarity")
# %%
# np.savetxt("./data/embeddings.tsv", np.vstack(df["embedding"].values), delimiter="\t")
# # %%
# df[["user_id"]].to_csv(
# "./data/metadata.tsv", sep="\t", index=False
# )
# %%
# Reduce embeddings to 3D with PCA
pca = PCA(n_components=3)
embeddings_3d = pca.fit_transform(np.vstack(df["embedding"].values))
# Add PCA components to dataframe
df["pca1"] = embeddings_3d[:, 0]
df["pca2"] = embeddings_3d[:, 1]
df["pca3"] = embeddings_3d[:, 2]
# %%
# Interactive 3D scatter plot
fig = px.scatter_3d(
df,
x="pca1",
y="pca2",
z="pca3",
color="age_recoded_6_categories",
hover_data=["user_id", "gender", "age_recoded_6_categories"],
title="User Embeddings (PCA 3D)"
)
fig.write_html("./output/user_embeddings_pca_3d.html")
# %%
# Reduce embeddings to 3D with t-SNE
tsne = TSNE(n_components=3, random_state=42, perplexity=30)
embeddings_3d = tsne.fit_transform(np.vstack(df["embedding"].values))
# Add t-SNE components to dataframe
df["tsne1"] = embeddings_3d[:, 0]
df["tsne2"] = embeddings_3d[:, 1]
df["tsne3"] = embeddings_3d[:, 2]
# Interactive 3D scatter plot
fig = px.scatter_3d(
df,
x="tsne1",
y="tsne2",
z="tsne3",
color="age_recoded_6_categories",
hover_data=["user_id", "gender", "country_code_iso_3166"],
title="User Embeddings (t-SNE 3D)"
)
fig.write_html("./output/user_embeddings_tsne_3d.html")
# %%
# example comparison based on t-sne projection
index_1 = 9561
index_2 = 5096
df.loc[[index_1, index_2]][["user_id", "user_text", "similarity"]].to_dict(orient="records")
# %%
row1 = df.loc[index_1].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
row2 = df.loc[index_2].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
# column_list = df.columns.difference(["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
column_list = disaster_measures_in_hh_columns + how_many_days_meet_columns
diffs = {}
for col in column_list:
val1 = row1[col]
val2 = row2[col]
if pd.isnull(val1) and pd.isnull(val2):
continue
if val1 != val2:
diffs[col] = (val1, val2)
# Print or display the differing columns and their values
for col, (v1, v2) in diffs.items():
print(f"{col}: {v1} | {v2}")
# %%
# Reduce embeddings to 3D with UMAP
umap_3d = umap.UMAP(n_components=3, random_state=42, metric="cosine").fit_transform(np.vstack(df["embedding"].values))
# Add UMAP components to dataframe
df["umap1"] = umap_3d[:, 0]
df["umap2"] = umap_3d[:, 1]
df["umap3"] = umap_3d[:, 2]
# Interactive 3D scatter plot
fig = px.scatter_3d(
df,
x="umap1",
y="umap2",
z="umap3",
color="country_code_iso_3166",
hover_data=["user_id", "gender", "country_code_iso_3166"],
title="User Embeddings (UMAP 3D)"
)
fig.write_html("./output/user_embeddings_umap_3d.html")
# %%
# example comparison based on UMAP projection
index_1 = 21396
index_2 = 10444
df.loc[[index_1, index_2]][["user_id", "user_text", "similarity"]].to_dict(orient="records")
# %%
row1 = df.loc[index_1].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
row2 = df.loc[index_2].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
# column_list = df.columns.difference(["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
column_list = disaster_measures_in_hh_columns + how_many_days_meet_columns
diffs = {}
for col in column_list:
val1 = row1[col]
val2 = row2[col]
if pd.isnull(val1) and pd.isnull(val2):
continue
if val1 != val2:
diffs[col] = (val1, val2)
# Print or display the differing columns and their values
for col, (v1, v2) in diffs.items():
print(f"{col}: {v1} | {v2}")
# %%
if __name__ == "__main__":
run_api()