getting started on the preparedness data
This commit is contained in:
+3
-1
@@ -137,4 +137,6 @@ kp_determinants2
|
||||
.Rproj.user
|
||||
|
||||
*.Rproj
|
||||
.Rhistory
|
||||
.Rhistory
|
||||
|
||||
scratch*
|
||||
@@ -0,0 +1,142 @@
|
||||
# %%
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
from utils import get_ollama_embedding
|
||||
|
||||
# %%[markdown]
|
||||
#### Data Preparation Steps
|
||||
|
||||
# - Select relevant columns from the cleaned dataset, including a range and specific variables.
|
||||
# - Exclude columns containing substrings like "2nd", "spont", "other", and some specific variables.
|
||||
# - Add columns starting with "region_" and "education_level_".
|
||||
# - Convert character columns to categorical type.
|
||||
# - Generate a `subregion` variable by coalescing region columns, then drop the originals.
|
||||
# - Generate an `education_level` variable by coalescing education columns, replacing "Not mentioned" with NA, then drop the originals.
|
||||
# - Define core preparedness items and their human-readable labels.
|
||||
# - Map country codes to country names.
|
||||
|
||||
# %%
|
||||
# Load your data
|
||||
data_cleaned = pd.read_csv('./data/eurobarometer_data_cleaned_csv.csv')
|
||||
|
||||
print(data_cleaned.shape)
|
||||
|
||||
# Select columns by name and range
|
||||
cols_to_select = [
|
||||
'country_code_iso_3166',
|
||||
'risks_cntry_most_exposed_to_firstly',
|
||||
'risks_pers_most_exposed_to_firstly',
|
||||
'risks_pers_most_exposed_to_number_of_mentioned_risks',
|
||||
'pot_info_sources_to_learn_about_disaster_risks_firstly',
|
||||
*data_cleaned.columns[404:448], # Python is 0-indexed
|
||||
'occupation_of_respondent',
|
||||
'age_recoded_6_categories',
|
||||
'size_of_community',
|
||||
'social_class_self_assessment_5_cat',
|
||||
'direction_things_are_going_life_personally',
|
||||
'political_discussion_local_matters',
|
||||
'political_discussion_national_matters',
|
||||
'left_right_placement_recoded_5_cat',
|
||||
'internet_use_total',
|
||||
'gender',
|
||||
'age_education',
|
||||
'standard_of_living_last_5yrs_in_light_of_crises',
|
||||
'personal_living_conditions_in_one_years_time',
|
||||
'standard_of_living_next_5yrs'
|
||||
]
|
||||
|
||||
# Remove columns containing certain substrings
|
||||
exclude_patterns = ['2nd', 'spont', 'other']
|
||||
cols_to_exclude = [col for col in data_cleaned.columns if any(p in col for p in exclude_patterns)]
|
||||
cols_to_exclude += [
|
||||
'disaster_measures_in_hh_number_of_measures',
|
||||
'disaster_pers_experienced_past_10yrs_none',
|
||||
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
|
||||
]
|
||||
|
||||
# Add region_ and education_level_ columns
|
||||
cols_to_select += [col for col in data_cleaned.columns if col.startswith('region_')]
|
||||
cols_to_select += [col for col in data_cleaned.columns if col.startswith('education_level_')]
|
||||
|
||||
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
|
||||
df_model = data_cleaned[final_cols].copy()
|
||||
|
||||
# %% [markdown]
|
||||
# #### Combine all columns into a single string per user
|
||||
# This step creates a text representation of each user, which can be sent to an embedding model.
|
||||
|
||||
def row_to_string(row):
|
||||
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
|
||||
|
||||
# Combine all columns into a single string per user (except country_code_iso_3166)
|
||||
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
|
||||
|
||||
df_model.head()
|
||||
|
||||
# %%
|
||||
# Convert character columns to category BEFORE adding embedding column
|
||||
for col in df_model.select_dtypes(include='object').columns:
|
||||
if col != 'user_text':
|
||||
df_model[col] = df_model[col].astype('category')
|
||||
|
||||
# Generate embeddings for each user using Ollama
|
||||
df_model['embedding'] = df_model['user_text'].apply(get_ollama_embedding)
|
||||
|
||||
# Add new columns to final_cols
|
||||
final_cols.extend(["user_text", "embedding"])
|
||||
|
||||
# Check the lengths of all embeddings
|
||||
embedding_lengths = df_model['embedding'].apply(lambda x: len(x) if isinstance(x, list) else None)
|
||||
print('Embedding lengths:', embedding_lengths.tolist())
|
||||
df_model = df_model[final_cols].copy()
|
||||
|
||||
# Generate subregion variable
|
||||
region_cols = [col for col in df_model.columns if col.startswith('region_')]
|
||||
df_model['subregion'] = df_model[region_cols].bfill(axis=1).iloc[:, 0]
|
||||
df_model.drop(columns=region_cols, inplace=True)
|
||||
|
||||
# Generate education_level variable
|
||||
edu_cols = [col for col in df_model.columns if col.startswith('education_level_')]
|
||||
for col in edu_cols:
|
||||
df_model[col] = df_model[col].replace('Not mentioned', np.nan)
|
||||
df_model['education_level'] = df_model[edu_cols].bfill(axis=1).iloc[:, 0]
|
||||
df_model['education_level'] = df_model['education_level'].astype('category')
|
||||
df_model.drop(columns=edu_cols, inplace=True)
|
||||
|
||||
# Core preparedness items
|
||||
CORE_ITEMS_MAPPED = [
|
||||
"disaster_measures_in_hh_emergency_supply_drinks_food",
|
||||
"disaster_measures_in_hh_emergency_supply_water_cooking_hygiene",
|
||||
"disaster_measures_in_hh_agreed_with_friends_family_to_contact",
|
||||
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood",
|
||||
"disaster_measures_in_hh_battery_powered_radio"
|
||||
]
|
||||
|
||||
CORE_ITEM_LABELS = {
|
||||
"disaster_measures_in_hh_emergency_supply_drinks_food": "Emergency supply of drinks, food",
|
||||
"disaster_measures_in_hh_emergency_supply_water_cooking_hygiene": "Emergency supply of cooking and hygiene water",
|
||||
"disaster_measures_in_hh_agreed_with_friends_family_to_contact": "Agreed with family, friends on how to contact in an emergency",
|
||||
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood": "Discussed precautions in neighbourhood",
|
||||
"disaster_measures_in_hh_battery_powered_radio": "Battery-powered radio accessible"
|
||||
}
|
||||
|
||||
COUNTRY_NAME_MAP = {
|
||||
"FI": "Finland",
|
||||
"DE-E": "East Germany",
|
||||
"DE-W": "West Germany",
|
||||
"FR": "France",
|
||||
"ES": "Spain",
|
||||
"PT": "Portugal",
|
||||
# "EE": "Estonia",
|
||||
# "DK": "Denmark",
|
||||
# "SE": "Sweden",
|
||||
# "NL": "Netherlands"
|
||||
}
|
||||
|
||||
# %%
|
||||
df_model.head()
|
||||
|
||||
# %%
|
||||
df_model.head().to_dict(orient='records')
|
||||
# %%
|
||||
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v2.csv', index=False)
|
||||
@@ -0,0 +1,65 @@
|
||||
# %%
|
||||
import pyreadstat
|
||||
import json
|
||||
import re
|
||||
import datetime
|
||||
|
||||
# %%
|
||||
df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav')
|
||||
print(df_sav.shape)
|
||||
|
||||
# %%
|
||||
df_sav.head(1).to_dict(orient='records')
|
||||
|
||||
# %%
|
||||
def safe_json(obj):
|
||||
if isinstance(obj, (datetime.datetime, datetime.date)):
|
||||
return obj.isoformat()
|
||||
if isinstance(obj, set):
|
||||
return list(obj)
|
||||
if hasattr(obj, '__dict__'):
|
||||
return str(obj)
|
||||
return obj
|
||||
|
||||
meta_dict = vars(meta)
|
||||
meta_json = json.dumps(meta_dict, default=safe_json, indent=2)
|
||||
|
||||
# %%
|
||||
with open('./data/za8841_meta.json', 'w') as f:
|
||||
f.write(meta_json)
|
||||
|
||||
# %%
|
||||
# Load your mapping dictionary (from JSON file or directly)
|
||||
with open("./data/za8841_meta.json") as f:
|
||||
meta = json.load(f)
|
||||
value_labels = meta["variable_value_labels"]
|
||||
|
||||
# %%
|
||||
# Assume df_sav is your loaded SPSS dataframe
|
||||
# For each column in the mapping, map values if the column exists in df_sav
|
||||
for col, mapping in value_labels.items():
|
||||
if col in df_sav.columns:
|
||||
# Convert keys to float if needed (SPSS values often are float)
|
||||
mapping_float = {float(k): v for k, v in mapping.items()}
|
||||
df_sav[col] = df_sav[col].map(mapping_float).fillna(df_sav[col])
|
||||
|
||||
# Now all mapped columns have human-readable values
|
||||
print(df_sav.head())
|
||||
|
||||
# %%
|
||||
labels_map = meta["column_names_to_labels"]
|
||||
def make_pandas_friendly(col):
|
||||
col = labels_map.get(col, col)
|
||||
col = re.sub(r'[.\s]+', '_', col)
|
||||
col = re.sub(r'[^0-9a-zA-Z_]', '', col)
|
||||
col = col.lower()
|
||||
col = re.sub(r'__+', '_', col) # Replace double (or more) underscores with single
|
||||
col = col.strip('_') # Remove leading/trailing underscores
|
||||
return col
|
||||
|
||||
df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns]
|
||||
|
||||
# %%
|
||||
df_sav.head(1).to_dict(orient='records')
|
||||
|
||||
# %%
|
||||
@@ -0,0 +1,100 @@
|
||||
# %% [markdown]
|
||||
# # Visualize Most Prepared Users
|
||||
# This workflow loads user embeddings, generates a prompt embedding, computes similarity, and visualizes the most prepared users.
|
||||
|
||||
# %%
|
||||
import pandas as pd
|
||||
import numpy as np
|
||||
from sklearn.metrics.pairwise import cosine_similarity
|
||||
import matplotlib.pyplot as plt
|
||||
import seaborn as sns
|
||||
from utils import get_ollama_embedding
|
||||
|
||||
# %% [markdown]
|
||||
# ## Read in the file with user vectors
|
||||
# Update the path/format as needed.
|
||||
# %%
|
||||
df = (
|
||||
pd.read_csv("./data/eurobarometer_preparedness_model_data_v2.csv").assign(
|
||||
**{
|
||||
"user_id": lambda x: x["country_code_iso_3166"]
|
||||
+ "_"
|
||||
+ x["age_recoded_6_categories"].astype(str)
|
||||
+ "_"
|
||||
+ x["gender"].astype(str)
|
||||
}
|
||||
)
|
||||
# select only relevant countries for visualization
|
||||
.loc[
|
||||
lambda x: x["country_code_iso_3166"].isin(
|
||||
["FI", "DE-E", "DE-W", "FR", "ES", "PT"]
|
||||
)
|
||||
]
|
||||
)
|
||||
df.shape
|
||||
|
||||
|
||||
# %% [markdown]
|
||||
# ## Convert string embeddings to lists if needed
|
||||
# %%
|
||||
def parse_embedding(x):
|
||||
if isinstance(x, str):
|
||||
return [float(i) for i in x.strip("[]").split(",")]
|
||||
return x
|
||||
|
||||
|
||||
df["embedding"] = df["embedding"].apply(parse_embedding)
|
||||
|
||||
# %%
|
||||
# Select best user:
|
||||
# disaster_measures_in_hh_* == 1, and how_many_days_meet_* == (4 | 5)
|
||||
disaster_measures_in_hh_columns = [
|
||||
col for col in df.columns if col.startswith("disaster_measures_in_hh_")
|
||||
]
|
||||
how_many_days_meet_columns = [
|
||||
col for col in df.columns if col.startswith("how_many_days_meet_")
|
||||
]
|
||||
best_users = df[
|
||||
(df[disaster_measures_in_hh_columns] == 1).all(axis=1)
|
||||
& (df[how_many_days_meet_columns].isin([4, 5]).all(axis=1))
|
||||
]
|
||||
print(f"Number of best users: {len(best_users)}")
|
||||
|
||||
# %% [markdown]
|
||||
# ## Write your prompt and generate its embedding
|
||||
# %%
|
||||
# prompt = "The user is highly prepared for disasters, with emergency supplies and a clear plan."
|
||||
prompt = best_users.head(1)["user_text"].values[
|
||||
0
|
||||
] # Example: use the first user's text as the prompt
|
||||
prompt_embedding = get_ollama_embedding(prompt)
|
||||
|
||||
# %% [markdown]
|
||||
# ## Compute similarity between each user and the prompt
|
||||
# %%
|
||||
user_embeddings = np.vstack(df["embedding"].values)
|
||||
prompt_vec = np.array(prompt_embedding).reshape(1, -1)
|
||||
similarities = cosine_similarity(user_embeddings, prompt_vec).flatten()
|
||||
df["similarity"] = similarities
|
||||
|
||||
# %% [markdown]
|
||||
# ## Visualize the users by similarity
|
||||
# %%
|
||||
_ = plt.figure(figsize=(5, 7))
|
||||
_ = sns.violinplot(
|
||||
x="similarity", y="country_code_iso_3166", hue="country_code_iso_3166", data=df
|
||||
)
|
||||
_ = sns.stripplot(
|
||||
x="similarity",
|
||||
y="country_code_iso_3166",
|
||||
data=df,
|
||||
hue="country_code_iso_3166",
|
||||
alpha=0.8,
|
||||
jitter=True,
|
||||
linewidth=0.5,
|
||||
edgecolor="white",
|
||||
)
|
||||
_ = plt.title("Distribution of User Similarities to Preparedness Prompt")
|
||||
_ = plt.xlabel("Cosine Similarity")
|
||||
|
||||
# %%
|
||||
@@ -0,0 +1,7 @@
|
||||
pandas
|
||||
ipykernel
|
||||
requests
|
||||
scikit-learn
|
||||
matplotlib
|
||||
seaborn
|
||||
pyreadstat
|
||||
@@ -0,0 +1,24 @@
|
||||
import requests
|
||||
|
||||
def get_ollama_summary(text, model="gpt-oss:20b"):
|
||||
url = "http://localhost:11434/v1/chat/completions"
|
||||
payload = {
|
||||
"model": model,
|
||||
"messages": [
|
||||
{"role": "system", "content": "You are a helpful assistant that summarizes text."},
|
||||
{"role": "user", "content": text}
|
||||
]
|
||||
}
|
||||
response = requests.post(url, json=payload)
|
||||
response.raise_for_status()
|
||||
return response.json()["choices"][0]["message"]["content"]
|
||||
|
||||
def get_ollama_embedding(text, model="nomic-embed-text"):
|
||||
url = "http://localhost:11434/api/embeddings"
|
||||
payload = {
|
||||
"model": model,
|
||||
"prompt": text
|
||||
}
|
||||
response = requests.post(url, json=payload)
|
||||
response.raise_for_status()
|
||||
return response.json()["embedding"]
|
||||
Reference in New Issue
Block a user