getting a version where the visualization is possible via pca, t-sne, and umap
This commit is contained in:
@@ -4,6 +4,8 @@ import json
|
||||
import re
|
||||
import datetime
|
||||
|
||||
from utils import get_ollama_summary, get_ollama_embedding
|
||||
|
||||
# %%
|
||||
df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav')
|
||||
print(df_sav.shape)
|
||||
@@ -63,3 +65,73 @@ df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns]
|
||||
df_sav.head(1).to_dict(orient='records')
|
||||
|
||||
# %%
|
||||
cols_to_select = [
|
||||
'country_code_iso_3166',
|
||||
'risks_cntry_most_exposed_to_firstly',
|
||||
'risks_pers_most_exposed_to_firstly',
|
||||
'risks_pers_most_exposed_to_number_of_mentioned_risks',
|
||||
'pot_info_sources_to_learn_about_disaster_risks_firstly',
|
||||
*df_sav.columns[404:448], # Python is 0-indexed
|
||||
'occupation_of_respondent',
|
||||
'age_recoded_6_categories',
|
||||
'size_of_community',
|
||||
# 'social_class_self_assessment_5_cat', # not in the data due to mapping
|
||||
'direction_things_are_going_life_personally',
|
||||
'political_discussion_local_matters',
|
||||
'political_discussion_national_matters',
|
||||
# 'left_right_placement_recoded_5_cat', # not in the data due to mapping
|
||||
'internet_use_total',
|
||||
'gender',
|
||||
'age_education',
|
||||
'standard_of_living_last_5yrs_in_light_of_crises',
|
||||
'personal_living_conditions_in_one_years_time',
|
||||
'standard_of_living_next_5yrs'
|
||||
]
|
||||
|
||||
# Remove columns containing certain substrings
|
||||
exclude_patterns = ['2nd', 'spont', 'other']
|
||||
cols_to_exclude = [col for col in df_sav.columns if any(p in col for p in exclude_patterns)]
|
||||
cols_to_exclude += [
|
||||
'disaster_measures_in_hh_number_of_measures',
|
||||
'disaster_pers_experienced_past_10yrs_none',
|
||||
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
|
||||
]
|
||||
|
||||
# Add region_ and education_level_ columns
|
||||
cols_to_select += [col for col in df_sav.columns if col.startswith('region_')]
|
||||
cols_to_select += [col for col in df_sav.columns if col.startswith('education_level_')]
|
||||
|
||||
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
|
||||
|
||||
print(f"Final number of columns: {len(final_cols)}")
|
||||
final_cols
|
||||
|
||||
# %%
|
||||
df_model = df_sav[final_cols].copy()
|
||||
|
||||
# %%
|
||||
# %% [markdown]
|
||||
# #### Combine all columns into a single string per user
|
||||
# This step creates a text representation of each user, which can be sent to an embedding model.
|
||||
|
||||
def row_to_string(row):
|
||||
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
|
||||
|
||||
# Combine all columns into a single string per user (except country_code_iso_3166)
|
||||
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
|
||||
|
||||
df_model.head()
|
||||
|
||||
# %%
|
||||
# Convert character columns to category BEFORE adding embedding column
|
||||
for col in df_model.select_dtypes(include='object').columns:
|
||||
if col != 'user_text':
|
||||
df_model[col] = df_model[col].astype('category')
|
||||
|
||||
# %%
|
||||
# df_model["summary"] = df_model['user_text'].apply(get_ollama_summary)
|
||||
# df_model.head(1)['user_text'].apply(get_ollama_summary).to_list()
|
||||
df_model["embedding"] = df_model['user_text'].apply(get_ollama_embedding)
|
||||
|
||||
# %%
|
||||
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v3.csv', index=False)
|
||||
File diff suppressed because one or more lines are too long
+140
-4
@@ -6,8 +6,13 @@
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
from sklearn.metrics.pairwise import cosine_similarity
|
||||
from sklearn.decomposition import PCA
|
||||
from sklearn.manifold import TSNE
|
||||
import plotly.express as px
|
||||
import matplotlib.pyplot as plt
|
||||
import seaborn as sns
|
||||
import umap
|
||||
import umap.plot
|
||||
from utils import get_ollama_embedding
|
||||
|
||||
# %% [markdown]
|
||||
@@ -15,9 +20,13 @@ from utils import get_ollama_embedding
|
||||
# Update the path/format as needed.
|
||||
# %%
|
||||
df = (
|
||||
pd.read_csv("./data/eurobarometer_preparedness_model_data_v2.csv").assign(
|
||||
pd.read_csv("./data/eurobarometer_preparedness_model_data_v3.csv")
|
||||
.reset_index()
|
||||
.assign(
|
||||
**{
|
||||
"user_id": lambda x: x["country_code_iso_3166"]
|
||||
"user_id": lambda x: x["index"].astype(str)
|
||||
+ "_"
|
||||
+ x["country_code_iso_3166"]
|
||||
+ "_"
|
||||
+ x["age_recoded_6_categories"].astype(str)
|
||||
+ "_"
|
||||
@@ -33,6 +42,10 @@ df = (
|
||||
)
|
||||
df.shape
|
||||
|
||||
# relabel None of the above/ Non binary/ do not recognize yourself in above categories/Prefer not to say to other
|
||||
df["gender"] = df["gender"].replace({
|
||||
"None of the above/ Non binary/ do not recognize yourself in above categories/Prefer not to say": "Other",
|
||||
})
|
||||
|
||||
# %% [markdown]
|
||||
# ## Convert string embeddings to lists if needed
|
||||
@@ -48,6 +61,9 @@ df["embedding"] = df["embedding"].apply(parse_embedding)
|
||||
# %%
|
||||
# Select best user:
|
||||
# disaster_measures_in_hh_* == 1, and how_many_days_meet_* == (4 | 5)
|
||||
# Following relabeling in data_preparation_raw.py:
|
||||
# disaster_measures_in_hh_* != "Not mentioned"
|
||||
# how_many_days_meet_* == "More than 7 days"
|
||||
disaster_measures_in_hh_columns = [
|
||||
col for col in df.columns if col.startswith("disaster_measures_in_hh_")
|
||||
]
|
||||
@@ -55,11 +71,13 @@ how_many_days_meet_columns = [
|
||||
col for col in df.columns if col.startswith("how_many_days_meet_")
|
||||
]
|
||||
best_users = df[
|
||||
(df[disaster_measures_in_hh_columns] == 1).all(axis=1)
|
||||
& (df[how_many_days_meet_columns].isin([4, 5]).all(axis=1))
|
||||
(df[disaster_measures_in_hh_columns] != "Not mentioned").all(axis=1)
|
||||
& (df[how_many_days_meet_columns] == "More than 7 days").all(axis=1)
|
||||
]
|
||||
print(f"Number of best users: {len(best_users)}")
|
||||
|
||||
best_users.head(5)[["user_id", "user_text"]].to_dict(orient="records")
|
||||
|
||||
# %% [markdown]
|
||||
# ## Write your prompt and generate its embedding
|
||||
# %%
|
||||
@@ -98,3 +116,121 @@ _ = plt.title("Distribution of User Similarities to Preparedness Prompt")
|
||||
_ = plt.xlabel("Cosine Similarity")
|
||||
|
||||
# %%
|
||||
# np.savetxt("./data/embeddings.tsv", np.vstack(df["embedding"].values), delimiter="\t")
|
||||
|
||||
# # %%
|
||||
# df[["user_id"]].to_csv(
|
||||
# "./data/metadata.tsv", sep="\t", index=False
|
||||
# )
|
||||
|
||||
# %%
|
||||
# Reduce embeddings to 3D with PCA
|
||||
pca = PCA(n_components=3)
|
||||
embeddings_3d = pca.fit_transform(np.vstack(df["embedding"].values))
|
||||
|
||||
# Add PCA components to dataframe
|
||||
df["pca1"] = embeddings_3d[:, 0]
|
||||
df["pca2"] = embeddings_3d[:, 1]
|
||||
df["pca3"] = embeddings_3d[:, 2]
|
||||
|
||||
# %%
|
||||
# Interactive 3D scatter plot
|
||||
fig = px.scatter_3d(
|
||||
df,
|
||||
x="pca1",
|
||||
y="pca2",
|
||||
z="pca3",
|
||||
color="age_recoded_6_categories",
|
||||
hover_data=["user_id", "gender", "age_recoded_6_categories"],
|
||||
title="User Embeddings (PCA 3D)"
|
||||
)
|
||||
fig.write_html("./output/user_embeddings_pca_3d.html")
|
||||
# %%
|
||||
# Reduce embeddings to 3D with t-SNE
|
||||
tsne = TSNE(n_components=3, random_state=42, perplexity=30)
|
||||
embeddings_3d = tsne.fit_transform(np.vstack(df["embedding"].values))
|
||||
|
||||
# Add t-SNE components to dataframe
|
||||
df["tsne1"] = embeddings_3d[:, 0]
|
||||
df["tsne2"] = embeddings_3d[:, 1]
|
||||
df["tsne3"] = embeddings_3d[:, 2]
|
||||
|
||||
# Interactive 3D scatter plot
|
||||
fig = px.scatter_3d(
|
||||
df,
|
||||
x="tsne1",
|
||||
y="tsne2",
|
||||
z="tsne3",
|
||||
color="age_recoded_6_categories",
|
||||
hover_data=["user_id", "gender", "country_code_iso_3166"],
|
||||
title="User Embeddings (t-SNE 3D)"
|
||||
)
|
||||
fig.write_html("./output/user_embeddings_tsne_3d.html")
|
||||
|
||||
# %%
|
||||
# example comparison based on t-sne projection
|
||||
index_1 = 9561
|
||||
index_2 = 5096
|
||||
df.loc[[index_1, index_2]][["user_id", "user_text", "similarity"]].to_dict(orient="records")
|
||||
# %%
|
||||
row1 = df.loc[index_1].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
|
||||
row2 = df.loc[index_2].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
|
||||
# column_list = df.columns.difference(["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
|
||||
column_list = disaster_measures_in_hh_columns + how_many_days_meet_columns
|
||||
|
||||
diffs = {}
|
||||
for col in column_list:
|
||||
val1 = row1[col]
|
||||
val2 = row2[col]
|
||||
if pd.isnull(val1) and pd.isnull(val2):
|
||||
continue
|
||||
if val1 != val2:
|
||||
diffs[col] = (val1, val2)
|
||||
|
||||
# Print or display the differing columns and their values
|
||||
for col, (v1, v2) in diffs.items():
|
||||
print(f"{col}: {v1} | {v2}")
|
||||
# %%
|
||||
# Reduce embeddings to 3D with UMAP
|
||||
umap_3d = umap.UMAP(n_components=3, random_state=42, metric="cosine").fit_transform(np.vstack(df["embedding"].values))
|
||||
|
||||
# Add UMAP components to dataframe
|
||||
df["umap1"] = umap_3d[:, 0]
|
||||
df["umap2"] = umap_3d[:, 1]
|
||||
df["umap3"] = umap_3d[:, 2]
|
||||
|
||||
# Interactive 3D scatter plot
|
||||
fig = px.scatter_3d(
|
||||
df,
|
||||
x="umap1",
|
||||
y="umap2",
|
||||
z="umap3",
|
||||
color="country_code_iso_3166",
|
||||
hover_data=["user_id", "gender", "country_code_iso_3166"],
|
||||
title="User Embeddings (UMAP 3D)"
|
||||
)
|
||||
fig.write_html("./output/user_embeddings_umap_3d.html")
|
||||
|
||||
# %%
|
||||
# example comparison based on UMAP projection
|
||||
index_1 = 21396
|
||||
index_2 = 10444
|
||||
df.loc[[index_1, index_2]][["user_id", "user_text", "similarity"]].to_dict(orient="records")
|
||||
# %%
|
||||
row1 = df.loc[index_1].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
|
||||
row2 = df.loc[index_2].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
|
||||
# column_list = df.columns.difference(["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
|
||||
column_list = disaster_measures_in_hh_columns + how_many_days_meet_columns
|
||||
diffs = {}
|
||||
for col in column_list:
|
||||
val1 = row1[col]
|
||||
val2 = row2[col]
|
||||
if pd.isnull(val1) and pd.isnull(val2):
|
||||
continue
|
||||
if val1 != val2:
|
||||
diffs[col] = (val1, val2)
|
||||
|
||||
# Print or display the differing columns and their values
|
||||
for col, (v1, v2) in diffs.items():
|
||||
print(f"{col}: {v1} | {v2}")
|
||||
# %%
|
||||
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
@@ -4,4 +4,7 @@ requests
|
||||
scikit-learn
|
||||
matplotlib
|
||||
seaborn
|
||||
pyreadstat
|
||||
pyreadstat
|
||||
plotly
|
||||
nbformat>=4.2.0
|
||||
umap-learn[plot]
|
||||
@@ -1,6 +1,6 @@
|
||||
import requests
|
||||
|
||||
def get_ollama_summary(text, model="gpt-oss:20b"):
|
||||
def get_ollama_summary(text, model="granite3.1-moe:1b"):
|
||||
url = "http://localhost:11434/v1/chat/completions"
|
||||
payload = {
|
||||
"model": model,
|
||||
|
||||
Reference in New Issue
Block a user