diff --git a/.gitignore b/.gitignore index d1b1888..a9b090e 100644 --- a/.gitignore +++ b/.gitignore @@ -137,4 +137,6 @@ kp_determinants2 .Rproj.user *.Rproj -.Rhistory \ No newline at end of file +.Rhistory + +scratch* \ No newline at end of file diff --git a/preparedness/data_preparation.py b/preparedness/data_preparation.py new file mode 100644 index 0000000..1526e70 --- /dev/null +++ b/preparedness/data_preparation.py @@ -0,0 +1,142 @@ +# %% +import pandas as pd +import numpy as np +from utils import get_ollama_embedding + +# %%[markdown] +#### Data Preparation Steps + +# - Select relevant columns from the cleaned dataset, including a range and specific variables. +# - Exclude columns containing substrings like "2nd", "spont", "other", and some specific variables. +# - Add columns starting with "region_" and "education_level_". +# - Convert character columns to categorical type. +# - Generate a `subregion` variable by coalescing region columns, then drop the originals. +# - Generate an `education_level` variable by coalescing education columns, replacing "Not mentioned" with NA, then drop the originals. +# - Define core preparedness items and their human-readable labels. +# - Map country codes to country names. + +# %% +# Load your data +data_cleaned = pd.read_csv('./data/eurobarometer_data_cleaned_csv.csv') + +print(data_cleaned.shape) + +# Select columns by name and range +cols_to_select = [ + 'country_code_iso_3166', + 'risks_cntry_most_exposed_to_firstly', + 'risks_pers_most_exposed_to_firstly', + 'risks_pers_most_exposed_to_number_of_mentioned_risks', + 'pot_info_sources_to_learn_about_disaster_risks_firstly', + *data_cleaned.columns[404:448], # Python is 0-indexed + 'occupation_of_respondent', + 'age_recoded_6_categories', + 'size_of_community', + 'social_class_self_assessment_5_cat', + 'direction_things_are_going_life_personally', + 'political_discussion_local_matters', + 'political_discussion_national_matters', + 'left_right_placement_recoded_5_cat', + 'internet_use_total', + 'gender', + 'age_education', + 'standard_of_living_last_5yrs_in_light_of_crises', + 'personal_living_conditions_in_one_years_time', + 'standard_of_living_next_5yrs' +] + +# Remove columns containing certain substrings +exclude_patterns = ['2nd', 'spont', 'other'] +cols_to_exclude = [col for col in data_cleaned.columns if any(p in col for p in exclude_patterns)] +cols_to_exclude += [ + 'disaster_measures_in_hh_number_of_measures', + 'disaster_pers_experienced_past_10yrs_none', + 'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source' +] + +# Add region_ and education_level_ columns +cols_to_select += [col for col in data_cleaned.columns if col.startswith('region_')] +cols_to_select += [col for col in data_cleaned.columns if col.startswith('education_level_')] + +final_cols = [col for col in cols_to_select if col not in cols_to_exclude] +df_model = data_cleaned[final_cols].copy() + +# %% [markdown] +# #### Combine all columns into a single string per user +# This step creates a text representation of each user, which can be sent to an embedding model. + +def row_to_string(row): + return ' | '.join(f'{col}: {row[col]}' for col in row.index) + +# Combine all columns into a single string per user (except country_code_iso_3166) +df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1) + +df_model.head() + +# %% +# Convert character columns to category BEFORE adding embedding column +for col in df_model.select_dtypes(include='object').columns: + if col != 'user_text': + df_model[col] = df_model[col].astype('category') + +# Generate embeddings for each user using Ollama +df_model['embedding'] = df_model['user_text'].apply(get_ollama_embedding) + +# Add new columns to final_cols +final_cols.extend(["user_text", "embedding"]) + +# Check the lengths of all embeddings +embedding_lengths = df_model['embedding'].apply(lambda x: len(x) if isinstance(x, list) else None) +print('Embedding lengths:', embedding_lengths.tolist()) +df_model = df_model[final_cols].copy() + +# Generate subregion variable +region_cols = [col for col in df_model.columns if col.startswith('region_')] +df_model['subregion'] = df_model[region_cols].bfill(axis=1).iloc[:, 0] +df_model.drop(columns=region_cols, inplace=True) + +# Generate education_level variable +edu_cols = [col for col in df_model.columns if col.startswith('education_level_')] +for col in edu_cols: + df_model[col] = df_model[col].replace('Not mentioned', np.nan) +df_model['education_level'] = df_model[edu_cols].bfill(axis=1).iloc[:, 0] +df_model['education_level'] = df_model['education_level'].astype('category') +df_model.drop(columns=edu_cols, inplace=True) + +# Core preparedness items +CORE_ITEMS_MAPPED = [ + "disaster_measures_in_hh_emergency_supply_drinks_food", + "disaster_measures_in_hh_emergency_supply_water_cooking_hygiene", + "disaster_measures_in_hh_agreed_with_friends_family_to_contact", + "disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood", + "disaster_measures_in_hh_battery_powered_radio" +] + +CORE_ITEM_LABELS = { + "disaster_measures_in_hh_emergency_supply_drinks_food": "Emergency supply of drinks, food", + "disaster_measures_in_hh_emergency_supply_water_cooking_hygiene": "Emergency supply of cooking and hygiene water", + "disaster_measures_in_hh_agreed_with_friends_family_to_contact": "Agreed with family, friends on how to contact in an emergency", + "disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood": "Discussed precautions in neighbourhood", + "disaster_measures_in_hh_battery_powered_radio": "Battery-powered radio accessible" +} + +COUNTRY_NAME_MAP = { + "FI": "Finland", + "DE-E": "East Germany", + "DE-W": "West Germany", + "FR": "France", + "ES": "Spain", + "PT": "Portugal", + # "EE": "Estonia", + # "DK": "Denmark", + # "SE": "Sweden", + # "NL": "Netherlands" +} + +# %% +df_model.head() + +# %% +df_model.head().to_dict(orient='records') +# %% +df_model.to_csv('./data/eurobarometer_preparedness_model_data_v2.csv', index=False) \ No newline at end of file diff --git a/preparedness/data_preparation_raw.py b/preparedness/data_preparation_raw.py new file mode 100644 index 0000000..367a575 --- /dev/null +++ b/preparedness/data_preparation_raw.py @@ -0,0 +1,65 @@ +# %% +import pyreadstat +import json +import re +import datetime + +# %% +df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav') +print(df_sav.shape) + +# %% +df_sav.head(1).to_dict(orient='records') + +# %% +def safe_json(obj): + if isinstance(obj, (datetime.datetime, datetime.date)): + return obj.isoformat() + if isinstance(obj, set): + return list(obj) + if hasattr(obj, '__dict__'): + return str(obj) + return obj + +meta_dict = vars(meta) +meta_json = json.dumps(meta_dict, default=safe_json, indent=2) + +# %% +with open('./data/za8841_meta.json', 'w') as f: + f.write(meta_json) + +# %% +# Load your mapping dictionary (from JSON file or directly) +with open("./data/za8841_meta.json") as f: + meta = json.load(f) +value_labels = meta["variable_value_labels"] + +# %% +# Assume df_sav is your loaded SPSS dataframe +# For each column in the mapping, map values if the column exists in df_sav +for col, mapping in value_labels.items(): + if col in df_sav.columns: + # Convert keys to float if needed (SPSS values often are float) + mapping_float = {float(k): v for k, v in mapping.items()} + df_sav[col] = df_sav[col].map(mapping_float).fillna(df_sav[col]) + +# Now all mapped columns have human-readable values +print(df_sav.head()) + +# %% +labels_map = meta["column_names_to_labels"] +def make_pandas_friendly(col): + col = labels_map.get(col, col) + col = re.sub(r'[.\s]+', '_', col) + col = re.sub(r'[^0-9a-zA-Z_]', '', col) + col = col.lower() + col = re.sub(r'__+', '_', col) # Replace double (or more) underscores with single + col = col.strip('_') # Remove leading/trailing underscores + return col + +df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns] + +# %% +df_sav.head(1).to_dict(orient='records') + +# %% diff --git a/preparedness/main.py b/preparedness/main.py new file mode 100644 index 0000000..d72416c --- /dev/null +++ b/preparedness/main.py @@ -0,0 +1,100 @@ +# %% [markdown] +# # Visualize Most Prepared Users +# This workflow loads user embeddings, generates a prompt embedding, computes similarity, and visualizes the most prepared users. + +# %% +import pandas as pd +import numpy as np +from sklearn.metrics.pairwise import cosine_similarity +import matplotlib.pyplot as plt +import seaborn as sns +from utils import get_ollama_embedding + +# %% [markdown] +# ## Read in the file with user vectors +# Update the path/format as needed. +# %% +df = ( + pd.read_csv("./data/eurobarometer_preparedness_model_data_v2.csv").assign( + **{ + "user_id": lambda x: x["country_code_iso_3166"] + + "_" + + x["age_recoded_6_categories"].astype(str) + + "_" + + x["gender"].astype(str) + } + ) + # select only relevant countries for visualization + .loc[ + lambda x: x["country_code_iso_3166"].isin( + ["FI", "DE-E", "DE-W", "FR", "ES", "PT"] + ) + ] +) +df.shape + + +# %% [markdown] +# ## Convert string embeddings to lists if needed +# %% +def parse_embedding(x): + if isinstance(x, str): + return [float(i) for i in x.strip("[]").split(",")] + return x + + +df["embedding"] = df["embedding"].apply(parse_embedding) + +# %% +# Select best user: +# disaster_measures_in_hh_* == 1, and how_many_days_meet_* == (4 | 5) +disaster_measures_in_hh_columns = [ + col for col in df.columns if col.startswith("disaster_measures_in_hh_") +] +how_many_days_meet_columns = [ + col for col in df.columns if col.startswith("how_many_days_meet_") +] +best_users = df[ + (df[disaster_measures_in_hh_columns] == 1).all(axis=1) + & (df[how_many_days_meet_columns].isin([4, 5]).all(axis=1)) +] +print(f"Number of best users: {len(best_users)}") + +# %% [markdown] +# ## Write your prompt and generate its embedding +# %% +# prompt = "The user is highly prepared for disasters, with emergency supplies and a clear plan." +prompt = best_users.head(1)["user_text"].values[ + 0 +] # Example: use the first user's text as the prompt +prompt_embedding = get_ollama_embedding(prompt) + +# %% [markdown] +# ## Compute similarity between each user and the prompt +# %% +user_embeddings = np.vstack(df["embedding"].values) +prompt_vec = np.array(prompt_embedding).reshape(1, -1) +similarities = cosine_similarity(user_embeddings, prompt_vec).flatten() +df["similarity"] = similarities + +# %% [markdown] +# ## Visualize the users by similarity +# %% +_ = plt.figure(figsize=(5, 7)) +_ = sns.violinplot( + x="similarity", y="country_code_iso_3166", hue="country_code_iso_3166", data=df +) +_ = sns.stripplot( + x="similarity", + y="country_code_iso_3166", + data=df, + hue="country_code_iso_3166", + alpha=0.8, + jitter=True, + linewidth=0.5, + edgecolor="white", +) +_ = plt.title("Distribution of User Similarities to Preparedness Prompt") +_ = plt.xlabel("Cosine Similarity") + +# %% diff --git a/preparedness/requirements.txt b/preparedness/requirements.txt new file mode 100644 index 0000000..a7cffe2 --- /dev/null +++ b/preparedness/requirements.txt @@ -0,0 +1,7 @@ +pandas +ipykernel +requests +scikit-learn +matplotlib +seaborn +pyreadstat \ No newline at end of file diff --git a/preparedness/utils.py b/preparedness/utils.py new file mode 100644 index 0000000..18ec933 --- /dev/null +++ b/preparedness/utils.py @@ -0,0 +1,24 @@ +import requests + +def get_ollama_summary(text, model="gpt-oss:20b"): + url = "http://localhost:11434/v1/chat/completions" + payload = { + "model": model, + "messages": [ + {"role": "system", "content": "You are a helpful assistant that summarizes text."}, + {"role": "user", "content": text} + ] + } + response = requests.post(url, json=payload) + response.raise_for_status() + return response.json()["choices"][0]["message"]["content"] + +def get_ollama_embedding(text, model="nomic-embed-text"): + url = "http://localhost:11434/api/embeddings" + payload = { + "model": model, + "prompt": text + } + response = requests.post(url, json=payload) + response.raise_for_status() + return response.json()["embedding"] \ No newline at end of file