# %% import pandas as pd import numpy as np from utils import get_ollama_embedding # %%[markdown] #### Data Preparation Steps # - Select relevant columns from the cleaned dataset, including a range and specific variables. # - Exclude columns containing substrings like "2nd", "spont", "other", and some specific variables. # - Add columns starting with "region_" and "education_level_". # - Convert character columns to categorical type. # - Generate a `subregion` variable by coalescing region columns, then drop the originals. # - Generate an `education_level` variable by coalescing education columns, replacing "Not mentioned" with NA, then drop the originals. # - Define core preparedness items and their human-readable labels. # - Map country codes to country names. # %% # Load your data data_cleaned = pd.read_csv('./data/eurobarometer_data_cleaned_csv.csv') print(data_cleaned.shape) # Select columns by name and range cols_to_select = [ 'country_code_iso_3166', 'risks_cntry_most_exposed_to_firstly', 'risks_pers_most_exposed_to_firstly', 'risks_pers_most_exposed_to_number_of_mentioned_risks', 'pot_info_sources_to_learn_about_disaster_risks_firstly', *data_cleaned.columns[404:448], # Python is 0-indexed 'occupation_of_respondent', 'age_recoded_6_categories', 'size_of_community', 'social_class_self_assessment_5_cat', 'direction_things_are_going_life_personally', 'political_discussion_local_matters', 'political_discussion_national_matters', 'left_right_placement_recoded_5_cat', 'internet_use_total', 'gender', 'age_education', 'standard_of_living_last_5yrs_in_light_of_crises', 'personal_living_conditions_in_one_years_time', 'standard_of_living_next_5yrs' ] # Remove columns containing certain substrings exclude_patterns = ['2nd', 'spont', 'other'] cols_to_exclude = [col for col in data_cleaned.columns if any(p in col for p in exclude_patterns)] cols_to_exclude += [ 'disaster_measures_in_hh_number_of_measures', 'disaster_pers_experienced_past_10yrs_none', 'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source' ] # Add region_ and education_level_ columns cols_to_select += [col for col in data_cleaned.columns if col.startswith('region_')] cols_to_select += [col for col in data_cleaned.columns if col.startswith('education_level_')] final_cols = [col for col in cols_to_select if col not in cols_to_exclude] df_model = data_cleaned[final_cols].copy() # %% [markdown] # #### Combine all columns into a single string per user # This step creates a text representation of each user, which can be sent to an embedding model. def row_to_string(row): return ' | '.join(f'{col}: {row[col]}' for col in row.index) # Combine all columns into a single string per user (except country_code_iso_3166) df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1) df_model.head() # %% # Convert character columns to category BEFORE adding embedding column for col in df_model.select_dtypes(include='object').columns: if col != 'user_text': df_model[col] = df_model[col].astype('category') # Generate embeddings for each user using Ollama df_model['embedding'] = df_model['user_text'].apply(get_ollama_embedding) # Add new columns to final_cols final_cols.extend(["user_text", "embedding"]) # Check the lengths of all embeddings embedding_lengths = df_model['embedding'].apply(lambda x: len(x) if isinstance(x, list) else None) print('Embedding lengths:', embedding_lengths.tolist()) df_model = df_model[final_cols].copy() # Generate subregion variable region_cols = [col for col in df_model.columns if col.startswith('region_')] df_model['subregion'] = df_model[region_cols].bfill(axis=1).iloc[:, 0] df_model.drop(columns=region_cols, inplace=True) # Generate education_level variable edu_cols = [col for col in df_model.columns if col.startswith('education_level_')] for col in edu_cols: df_model[col] = df_model[col].replace('Not mentioned', np.nan) df_model['education_level'] = df_model[edu_cols].bfill(axis=1).iloc[:, 0] df_model['education_level'] = df_model['education_level'].astype('category') df_model.drop(columns=edu_cols, inplace=True) # Core preparedness items CORE_ITEMS_MAPPED = [ "disaster_measures_in_hh_emergency_supply_drinks_food", "disaster_measures_in_hh_emergency_supply_water_cooking_hygiene", "disaster_measures_in_hh_agreed_with_friends_family_to_contact", "disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood", "disaster_measures_in_hh_battery_powered_radio" ] CORE_ITEM_LABELS = { "disaster_measures_in_hh_emergency_supply_drinks_food": "Emergency supply of drinks, food", "disaster_measures_in_hh_emergency_supply_water_cooking_hygiene": "Emergency supply of cooking and hygiene water", "disaster_measures_in_hh_agreed_with_friends_family_to_contact": "Agreed with family, friends on how to contact in an emergency", "disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood": "Discussed precautions in neighbourhood", "disaster_measures_in_hh_battery_powered_radio": "Battery-powered radio accessible" } COUNTRY_NAME_MAP = { "FI": "Finland", "DE-E": "East Germany", "DE-W": "West Germany", "FR": "France", "ES": "Spain", "PT": "Portugal", # "EE": "Estonia", # "DK": "Denmark", # "SE": "Sweden", # "NL": "Netherlands" } # %% df_model.head() # %% df_model.head().to_dict(orient='records') # %% df_model.to_csv('./data/eurobarometer_preparedness_model_data_v2.csv', index=False)