142 lines
5.5 KiB
Python
142 lines
5.5 KiB
Python
# %%
|
|
import pandas as pd
|
|
import numpy as np
|
|
from utils import get_ollama_embedding
|
|
|
|
# %%[markdown]
|
|
#### Data Preparation Steps
|
|
|
|
# - Select relevant columns from the cleaned dataset, including a range and specific variables.
|
|
# - Exclude columns containing substrings like "2nd", "spont", "other", and some specific variables.
|
|
# - Add columns starting with "region_" and "education_level_".
|
|
# - Convert character columns to categorical type.
|
|
# - Generate a `subregion` variable by coalescing region columns, then drop the originals.
|
|
# - Generate an `education_level` variable by coalescing education columns, replacing "Not mentioned" with NA, then drop the originals.
|
|
# - Define core preparedness items and their human-readable labels.
|
|
# - Map country codes to country names.
|
|
|
|
# %%
|
|
# Load your data
|
|
data_cleaned = pd.read_csv('./data/eurobarometer_data_cleaned_csv.csv')
|
|
|
|
print(data_cleaned.shape)
|
|
|
|
# Select columns by name and range
|
|
cols_to_select = [
|
|
'country_code_iso_3166',
|
|
'risks_cntry_most_exposed_to_firstly',
|
|
'risks_pers_most_exposed_to_firstly',
|
|
'risks_pers_most_exposed_to_number_of_mentioned_risks',
|
|
'pot_info_sources_to_learn_about_disaster_risks_firstly',
|
|
*data_cleaned.columns[404:448], # Python is 0-indexed
|
|
'occupation_of_respondent',
|
|
'age_recoded_6_categories',
|
|
'size_of_community',
|
|
'social_class_self_assessment_5_cat',
|
|
'direction_things_are_going_life_personally',
|
|
'political_discussion_local_matters',
|
|
'political_discussion_national_matters',
|
|
'left_right_placement_recoded_5_cat',
|
|
'internet_use_total',
|
|
'gender',
|
|
'age_education',
|
|
'standard_of_living_last_5yrs_in_light_of_crises',
|
|
'personal_living_conditions_in_one_years_time',
|
|
'standard_of_living_next_5yrs'
|
|
]
|
|
|
|
# Remove columns containing certain substrings
|
|
exclude_patterns = ['2nd', 'spont', 'other']
|
|
cols_to_exclude = [col for col in data_cleaned.columns if any(p in col for p in exclude_patterns)]
|
|
cols_to_exclude += [
|
|
'disaster_measures_in_hh_number_of_measures',
|
|
'disaster_pers_experienced_past_10yrs_none',
|
|
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
|
|
]
|
|
|
|
# Add region_ and education_level_ columns
|
|
cols_to_select += [col for col in data_cleaned.columns if col.startswith('region_')]
|
|
cols_to_select += [col for col in data_cleaned.columns if col.startswith('education_level_')]
|
|
|
|
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
|
|
df_model = data_cleaned[final_cols].copy()
|
|
|
|
# %% [markdown]
|
|
# #### Combine all columns into a single string per user
|
|
# This step creates a text representation of each user, which can be sent to an embedding model.
|
|
|
|
def row_to_string(row):
|
|
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
|
|
|
|
# Combine all columns into a single string per user (except country_code_iso_3166)
|
|
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
|
|
|
|
df_model.head()
|
|
|
|
# %%
|
|
# Convert character columns to category BEFORE adding embedding column
|
|
for col in df_model.select_dtypes(include='object').columns:
|
|
if col != 'user_text':
|
|
df_model[col] = df_model[col].astype('category')
|
|
|
|
# Generate embeddings for each user using Ollama
|
|
df_model['embedding'] = df_model['user_text'].apply(get_ollama_embedding)
|
|
|
|
# Add new columns to final_cols
|
|
final_cols.extend(["user_text", "embedding"])
|
|
|
|
# Check the lengths of all embeddings
|
|
embedding_lengths = df_model['embedding'].apply(lambda x: len(x) if isinstance(x, list) else None)
|
|
print('Embedding lengths:', embedding_lengths.tolist())
|
|
df_model = df_model[final_cols].copy()
|
|
|
|
# Generate subregion variable
|
|
region_cols = [col for col in df_model.columns if col.startswith('region_')]
|
|
df_model['subregion'] = df_model[region_cols].bfill(axis=1).iloc[:, 0]
|
|
df_model.drop(columns=region_cols, inplace=True)
|
|
|
|
# Generate education_level variable
|
|
edu_cols = [col for col in df_model.columns if col.startswith('education_level_')]
|
|
for col in edu_cols:
|
|
df_model[col] = df_model[col].replace('Not mentioned', np.nan)
|
|
df_model['education_level'] = df_model[edu_cols].bfill(axis=1).iloc[:, 0]
|
|
df_model['education_level'] = df_model['education_level'].astype('category')
|
|
df_model.drop(columns=edu_cols, inplace=True)
|
|
|
|
# Core preparedness items
|
|
CORE_ITEMS_MAPPED = [
|
|
"disaster_measures_in_hh_emergency_supply_drinks_food",
|
|
"disaster_measures_in_hh_emergency_supply_water_cooking_hygiene",
|
|
"disaster_measures_in_hh_agreed_with_friends_family_to_contact",
|
|
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood",
|
|
"disaster_measures_in_hh_battery_powered_radio"
|
|
]
|
|
|
|
CORE_ITEM_LABELS = {
|
|
"disaster_measures_in_hh_emergency_supply_drinks_food": "Emergency supply of drinks, food",
|
|
"disaster_measures_in_hh_emergency_supply_water_cooking_hygiene": "Emergency supply of cooking and hygiene water",
|
|
"disaster_measures_in_hh_agreed_with_friends_family_to_contact": "Agreed with family, friends on how to contact in an emergency",
|
|
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood": "Discussed precautions in neighbourhood",
|
|
"disaster_measures_in_hh_battery_powered_radio": "Battery-powered radio accessible"
|
|
}
|
|
|
|
COUNTRY_NAME_MAP = {
|
|
"FI": "Finland",
|
|
"DE-E": "East Germany",
|
|
"DE-W": "West Germany",
|
|
"FR": "France",
|
|
"ES": "Spain",
|
|
"PT": "Portugal",
|
|
# "EE": "Estonia",
|
|
# "DK": "Denmark",
|
|
# "SE": "Sweden",
|
|
# "NL": "Netherlands"
|
|
}
|
|
|
|
# %%
|
|
df_model.head()
|
|
|
|
# %%
|
|
df_model.head().to_dict(orient='records')
|
|
# %%
|
|
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v2.csv', index=False) |