getting a version where the visualization is possible via pca, t-sne, and umap

This commit is contained in:
itsamejms
2025-09-16 09:26:40 +02:00
parent 8a2e7fdb6e
commit fc0d58cd1b
8 changed files with 12593 additions and 6 deletions
+72
View File
@@ -4,6 +4,8 @@ import json
import re
import datetime
from utils import get_ollama_summary, get_ollama_embedding
# %%
df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav')
print(df_sav.shape)
@@ -63,3 +65,73 @@ df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns]
df_sav.head(1).to_dict(orient='records')
# %%
cols_to_select = [
'country_code_iso_3166',
'risks_cntry_most_exposed_to_firstly',
'risks_pers_most_exposed_to_firstly',
'risks_pers_most_exposed_to_number_of_mentioned_risks',
'pot_info_sources_to_learn_about_disaster_risks_firstly',
*df_sav.columns[404:448], # Python is 0-indexed
'occupation_of_respondent',
'age_recoded_6_categories',
'size_of_community',
# 'social_class_self_assessment_5_cat', # not in the data due to mapping
'direction_things_are_going_life_personally',
'political_discussion_local_matters',
'political_discussion_national_matters',
# 'left_right_placement_recoded_5_cat', # not in the data due to mapping
'internet_use_total',
'gender',
'age_education',
'standard_of_living_last_5yrs_in_light_of_crises',
'personal_living_conditions_in_one_years_time',
'standard_of_living_next_5yrs'
]
# Remove columns containing certain substrings
exclude_patterns = ['2nd', 'spont', 'other']
cols_to_exclude = [col for col in df_sav.columns if any(p in col for p in exclude_patterns)]
cols_to_exclude += [
'disaster_measures_in_hh_number_of_measures',
'disaster_pers_experienced_past_10yrs_none',
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
]
# Add region_ and education_level_ columns
cols_to_select += [col for col in df_sav.columns if col.startswith('region_')]
cols_to_select += [col for col in df_sav.columns if col.startswith('education_level_')]
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
print(f"Final number of columns: {len(final_cols)}")
final_cols
# %%
df_model = df_sav[final_cols].copy()
# %%
# %% [markdown]
# #### Combine all columns into a single string per user
# This step creates a text representation of each user, which can be sent to an embedding model.
def row_to_string(row):
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
# Combine all columns into a single string per user (except country_code_iso_3166)
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
df_model.head()
# %%
# Convert character columns to category BEFORE adding embedding column
for col in df_model.select_dtypes(include='object').columns:
if col != 'user_text':
df_model[col] = df_model[col].astype('category')
# %%
# df_model["summary"] = df_model['user_text'].apply(get_ollama_summary)
# df_model.head(1)['user_text'].apply(get_ollama_summary).to_list()
df_model["embedding"] = df_model['user_text'].apply(get_ollama_embedding)
# %%
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v3.csv', index=False)