# %% import pyreadstat import json import re import datetime from utils import get_ollama_summary, get_ollama_embedding # %% df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav') print(df_sav.shape) # %% df_sav.head(1).to_dict(orient='records') # %% def safe_json(obj): if isinstance(obj, (datetime.datetime, datetime.date)): return obj.isoformat() if isinstance(obj, set): return list(obj) if hasattr(obj, '__dict__'): return str(obj) return obj meta_dict = vars(meta) meta_json = json.dumps(meta_dict, default=safe_json, indent=2) # %% with open('./data/za8841_meta.json', 'w') as f: f.write(meta_json) # %% # Load your mapping dictionary (from JSON file or directly) with open("./data/za8841_meta.json") as f: meta = json.load(f) value_labels = meta["variable_value_labels"] # %% # Assume df_sav is your loaded SPSS dataframe # For each column in the mapping, map values if the column exists in df_sav for col, mapping in value_labels.items(): if col in df_sav.columns: # Convert keys to float if needed (SPSS values often are float) mapping_float = {float(k): v for k, v in mapping.items()} df_sav[col] = df_sav[col].map(mapping_float).fillna(df_sav[col]) # Now all mapped columns have human-readable values print(df_sav.head()) # %% labels_map = meta["column_names_to_labels"] def make_pandas_friendly(col): col = labels_map.get(col, col) col = re.sub(r'[.\s]+', '_', col) col = re.sub(r'[^0-9a-zA-Z_]', '', col) col = col.lower() col = re.sub(r'__+', '_', col) # Replace double (or more) underscores with single col = col.strip('_') # Remove leading/trailing underscores return col df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns] # %% df_sav.head(1).to_dict(orient='records') # %% cols_to_select = [ 'country_code_iso_3166', 'risks_cntry_most_exposed_to_firstly', 'risks_pers_most_exposed_to_firstly', 'risks_pers_most_exposed_to_number_of_mentioned_risks', 'pot_info_sources_to_learn_about_disaster_risks_firstly', *df_sav.columns[404:448], # Python is 0-indexed 'occupation_of_respondent', 'age_recoded_6_categories', 'size_of_community', # 'social_class_self_assessment_5_cat', # not in the data due to mapping 'direction_things_are_going_life_personally', 'political_discussion_local_matters', 'political_discussion_national_matters', # 'left_right_placement_recoded_5_cat', # not in the data due to mapping 'internet_use_total', 'gender', 'age_education', 'standard_of_living_last_5yrs_in_light_of_crises', 'personal_living_conditions_in_one_years_time', 'standard_of_living_next_5yrs' ] # Remove columns containing certain substrings exclude_patterns = ['2nd', 'spont', 'other'] cols_to_exclude = [col for col in df_sav.columns if any(p in col for p in exclude_patterns)] cols_to_exclude += [ 'disaster_measures_in_hh_number_of_measures', 'disaster_pers_experienced_past_10yrs_none', 'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source' ] # Add region_ and education_level_ columns cols_to_select += [col for col in df_sav.columns if col.startswith('region_')] cols_to_select += [col for col in df_sav.columns if col.startswith('education_level_')] final_cols = [col for col in cols_to_select if col not in cols_to_exclude] print(f"Final number of columns: {len(final_cols)}") final_cols # %% df_model = df_sav[final_cols].copy() # %% # %% [markdown] # #### Combine all columns into a single string per user # This step creates a text representation of each user, which can be sent to an embedding model. def row_to_string(row): return ' | '.join(f'{col}: {row[col]}' for col in row.index) # Combine all columns into a single string per user (except country_code_iso_3166) df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1) df_model.head() # %% # Convert character columns to category BEFORE adding embedding column for col in df_model.select_dtypes(include='object').columns: if col != 'user_text': df_model[col] = df_model[col].astype('category') # %% # df_model["summary"] = df_model['user_text'].apply(get_ollama_summary) # df_model.head(1)['user_text'].apply(get_ollama_summary).to_list() df_model["embedding"] = df_model['user_text'].apply(get_ollama_embedding) # %% df_model.to_csv('./data/eurobarometer_preparedness_model_data_v3.csv', index=False)