137 lines
4.4 KiB
Python
137 lines
4.4 KiB
Python
# %%
|
|
import pyreadstat
|
|
import json
|
|
import re
|
|
import datetime
|
|
|
|
from utils import get_ollama_summary, get_ollama_embedding
|
|
|
|
# %%
|
|
df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav')
|
|
print(df_sav.shape)
|
|
|
|
# %%
|
|
df_sav.head(1).to_dict(orient='records')
|
|
|
|
# %%
|
|
def safe_json(obj):
|
|
if isinstance(obj, (datetime.datetime, datetime.date)):
|
|
return obj.isoformat()
|
|
if isinstance(obj, set):
|
|
return list(obj)
|
|
if hasattr(obj, '__dict__'):
|
|
return str(obj)
|
|
return obj
|
|
|
|
meta_dict = vars(meta)
|
|
meta_json = json.dumps(meta_dict, default=safe_json, indent=2)
|
|
|
|
# %%
|
|
with open('./data/za8841_meta.json', 'w') as f:
|
|
f.write(meta_json)
|
|
|
|
# %%
|
|
# Load your mapping dictionary (from JSON file or directly)
|
|
with open("./data/za8841_meta.json") as f:
|
|
meta = json.load(f)
|
|
value_labels = meta["variable_value_labels"]
|
|
|
|
# %%
|
|
# Assume df_sav is your loaded SPSS dataframe
|
|
# For each column in the mapping, map values if the column exists in df_sav
|
|
for col, mapping in value_labels.items():
|
|
if col in df_sav.columns:
|
|
# Convert keys to float if needed (SPSS values often are float)
|
|
mapping_float = {float(k): v for k, v in mapping.items()}
|
|
df_sav[col] = df_sav[col].map(mapping_float).fillna(df_sav[col])
|
|
|
|
# Now all mapped columns have human-readable values
|
|
print(df_sav.head())
|
|
|
|
# %%
|
|
labels_map = meta["column_names_to_labels"]
|
|
def make_pandas_friendly(col):
|
|
col = labels_map.get(col, col)
|
|
col = re.sub(r'[.\s]+', '_', col)
|
|
col = re.sub(r'[^0-9a-zA-Z_]', '', col)
|
|
col = col.lower()
|
|
col = re.sub(r'__+', '_', col) # Replace double (or more) underscores with single
|
|
col = col.strip('_') # Remove leading/trailing underscores
|
|
return col
|
|
|
|
df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns]
|
|
|
|
# %%
|
|
df_sav.head(1).to_dict(orient='records')
|
|
|
|
# %%
|
|
cols_to_select = [
|
|
'country_code_iso_3166',
|
|
'risks_cntry_most_exposed_to_firstly',
|
|
'risks_pers_most_exposed_to_firstly',
|
|
'risks_pers_most_exposed_to_number_of_mentioned_risks',
|
|
'pot_info_sources_to_learn_about_disaster_risks_firstly',
|
|
*df_sav.columns[404:448], # Python is 0-indexed
|
|
'occupation_of_respondent',
|
|
'age_recoded_6_categories',
|
|
'size_of_community',
|
|
# 'social_class_self_assessment_5_cat', # not in the data due to mapping
|
|
'direction_things_are_going_life_personally',
|
|
'political_discussion_local_matters',
|
|
'political_discussion_national_matters',
|
|
# 'left_right_placement_recoded_5_cat', # not in the data due to mapping
|
|
'internet_use_total',
|
|
'gender',
|
|
'age_education',
|
|
'standard_of_living_last_5yrs_in_light_of_crises',
|
|
'personal_living_conditions_in_one_years_time',
|
|
'standard_of_living_next_5yrs'
|
|
]
|
|
|
|
# Remove columns containing certain substrings
|
|
exclude_patterns = ['2nd', 'spont', 'other']
|
|
cols_to_exclude = [col for col in df_sav.columns if any(p in col for p in exclude_patterns)]
|
|
cols_to_exclude += [
|
|
'disaster_measures_in_hh_number_of_measures',
|
|
'disaster_pers_experienced_past_10yrs_none',
|
|
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
|
|
]
|
|
|
|
# Add region_ and education_level_ columns
|
|
cols_to_select += [col for col in df_sav.columns if col.startswith('region_')]
|
|
cols_to_select += [col for col in df_sav.columns if col.startswith('education_level_')]
|
|
|
|
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
|
|
|
|
print(f"Final number of columns: {len(final_cols)}")
|
|
final_cols
|
|
|
|
# %%
|
|
df_model = df_sav[final_cols].copy()
|
|
|
|
# %%
|
|
# %% [markdown]
|
|
# #### Combine all columns into a single string per user
|
|
# This step creates a text representation of each user, which can be sent to an embedding model.
|
|
|
|
def row_to_string(row):
|
|
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
|
|
|
|
# Combine all columns into a single string per user (except country_code_iso_3166)
|
|
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
|
|
|
|
df_model.head()
|
|
|
|
# %%
|
|
# Convert character columns to category BEFORE adding embedding column
|
|
for col in df_model.select_dtypes(include='object').columns:
|
|
if col != 'user_text':
|
|
df_model[col] = df_model[col].astype('category')
|
|
|
|
# %%
|
|
# df_model["summary"] = df_model['user_text'].apply(get_ollama_summary)
|
|
# df_model.head(1)['user_text'].apply(get_ollama_summary).to_list()
|
|
df_model["embedding"] = df_model['user_text'].apply(get_ollama_embedding)
|
|
|
|
# %%
|
|
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v3.csv', index=False) |