Files
matti_jms_collabs/preparedness/api.py
T

306 lines
17 KiB
Python

import re
from fastapi import FastAPI, Response
import json
from fastapi.middleware.cors import CORSMiddleware
import os
from sklearn.metrics.pairwise import cosine_similarity
from schemas import PredictionRequest
from utils import get_ollama_embedding, parse_combined_text_to_dict
app = FastAPI(
title="preparedness-api",
version="0.0.1",
description="General API for the preparedness campaign.",
contact={
"email": "contact@jamestwose.com",
},
# lifespan=lifespan,
)
# Whitelist: only these variables will be exposed to the client form renderer.
# Update this list when you want to add/remove fields shown in the UI.
WHITELIST = {
"risks_cntry_most_exposed_to_firstly",
"risks_pers_most_exposed_to_firstly",
"risks_pers_most_exposed_to_number_of_mentioned_risks",
"pot_info_sources_to_learn_about_disaster_risks_firstly",
"statements_disaster_risks_readseenheard_info_in_last_12m",
"statements_disaster_risks_feel_well_informed",
"statements_disaster_risks_trust_information_by_pub_auth_on_risks_where_you_live",
"statements_disaster_risks_easy_to_find_information_by_pub_auth_on_risks_where_you_live",
"statements_disaster_risks_know_where_to_find_info_when_travelling_to_oth_eu_cntry",
"disaster_measures_in_hh_emergency_supply_drinksfood",
"disaster_measures_in_hh_emergency_supply_water_cookinghygiene",
"disaster_measures_in_hh_flashlightcandles",
"disaster_measures_in_hh_batterypowered_radio",
"disaster_measures_in_hh_emergency_pharmacy",
"disaster_measures_in_hh_copies_imp_documentsstored_safely",
"disaster_measures_in_hh_emergency_grabbag",
"disaster_measures_in_hh_signed_up_for_alerts",
"disaster_measures_in_hh_participated_in_trainingexercise",
"disaster_measures_in_hh_informed_about_official_response_plan",
"disaster_measures_in_hh_agreed_with_friendsfamily_to_contact",
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood",
"disaster_measures_in_hh_invested_in_prot_measures_in_home",
"how_many_days_meet_water_needs_if_water_services_disrupted",
"how_many_days_power_essent_appliances_if_elec_interrupted",
"how_many_days_cook_mealsheat_if_gas_disrupted",
"how_many_days_provide_food_if_transportation_disrupted",
"how_many_days_continued_treatment_if_medication_supply_disrupted",
"personal_disaster_preparedness_better_able_to_cope_by_prep",
"personal_disaster_preparedness_feel_well_prepared",
"personal_disaster_preparedness_no_timefin_resources_to_prep",
"personal_disaster_preparedness_easy_to_find_info_on_how_to_prep",
"personal_disaster_preparedness_need_more_info_to_prep",
"personal_disaster_preparedness_know_how_emerg_services_will_alert",
"personal_disaster_preparedness_know_what_to_do_in_event_of_disaster",
"personal_disaster_preparedness_employerschool_encourages_trainingprep",
"personal_disaster_preparedness_emerg_services_encourage_trainingprep",
"relying_on_in_first_days_of_disaster_familyfriends",
"relying_on_in_first_days_of_disaster_people_in_neighbourhood",
"relying_on_in_first_days_of_disaster_assocsnonprofit_orgs",
"relying_on_in_first_days_of_disaster_emerg_services",
"relying_on_in_first_days_of_disaster_local_authgvmt_services",
"relying_on_in_first_days_of_disaster_workemployerschooledu_institution",
"relying_on_in_first_days_of_disaster_private_sector_entities",
"trust_in_emerg_services_to_handle_disastersemerg_situations_properly",
"engaging_in_voluntary_work_for_emerg_responder_orgs",
"occupation_of_respondent",
"age_recoded_6_categories",
"size_of_community",
"direction_things_are_going_life_personally",
"political_discussion_local_matters",
"political_discussion_national_matters",
"internet_use_total",
"gender",
"age_education",
"standard_of_living_last_5yrs_in_light_of_crises",
"personal_living_conditions_in_one_years_time",
"standard_of_living_next_5yrs",
# region and education levels are included but will often be null in examples
"region_spain",
"region_austria",
"region_belgium",
"education_level_bachelor_or_equivalent",
}
def format_column_name(col: str) -> str:
"""Format a column name to match the style used in the WHITELIST."""
col = re.sub(r"[.\s]+", "_", col)
col = re.sub(r"[^0-9a-zA-Z_]", "", col)
col = col.lower()
col = re.sub(r"__+", "_", col) # Replace double (or more) underscores with single
col = col.strip("_") # Remove leading/trailing underscores
return col
origins = ["*"]
app.add_middleware(
CORSMiddleware,
allow_origins=origins,
allow_credentials=True,
allow_methods=["*"],
allow_headers=["*"],
)
@app.get("/")
async def root():
template_path = os.path.join(
os.path.dirname(os.path.dirname(__file__)),
"preparedness/templates",
"user_survey.html",
)
try:
with open(template_path, "r", encoding="utf-8") as f:
html_content = f.read()
return Response(content=html_content, media_type="text/html")
except Exception as e:
return Response(
content=f"Error loading template: {e}",
media_type="text/plain",
status_code=500,
)
@app.post("/predict")
async def predict(request: PredictionRequest):
example_best_user = "risks_cntry_most_exposed_to_firstly: Terrorist attacks | risks_pers_most_exposed_to_firstly: Extreme weather events (violent storms, droughts, heatwaves, cold waves, etc.) | risks_pers_most_exposed_to_number_of_mentioned_risks: 2 mentions | pot_info_sources_to_learn_about_disaster_risks_firstly: National media | statements_disaster_risks_readseenheard_info_in_last_12m: Tend to agree | statements_disaster_risks_feel_well_informed: Tend to agree | statements_disaster_risks_trust_information_by_pub_auth_on_risks_where_you_live: Totally agree | statements_disaster_risks_easy_to_find_information_by_pub_auth_on_risks_where_you_live: Tend to agree | statements_disaster_risks_know_where_to_find_info_when_travelling_to_oth_eu_cntry: Totally agree | disaster_measures_in_hh_emergency_supply_drinksfood: Keep an emergency supply stock/pack of drinks, food | disaster_measures_in_hh_emergency_supply_water_cookinghygiene: Keep an emergency supply of water for cooking and hygiene | disaster_measures_in_hh_flashlightcandles: Have flashlight or candles accessible | disaster_measures_in_hh_batterypowered_radio: Have a battery-powered radio accessible | disaster_measures_in_hh_emergency_pharmacy: Keep a home pharmacy for emergencies | disaster_measures_in_hh_copies_imp_documentsstored_safely: Have made sure you have copies of your most important documents or have stored them safely | disaster_measures_in_hh_emergency_grabbag: Have prepared a grab-bag, in case you need to evacuate rapidly in an emergency | disaster_measures_in_hh_signed_up_for_alerts: Have signed up for alerts and warnings from emergency services or authorities | disaster_measures_in_hh_participated_in_trainingexercise: Have participated in a training or exercise, to learn how to react in an emergency | disaster_measures_in_hh_informed_about_official_response_plan: Got informed on the response plan your city, region or country has for a disaster or emergency (e.g. (...) | disaster_measures_in_hh_agreed_with_friendsfamily_to_contact: Agreed with family, friends on how to contact each other in case of an emergency | disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood: Discussed common protective measures in your neighbourhood | disaster_measures_in_hh_invested_in_prot_measures_in_home: Have invested in protective measures in your home (e.g. flood-proofed the electricity installation, cleared (...) | how_many_days_meet_water_needs_if_water_services_disrupted: More than 7 days | how_many_days_power_essent_appliances_if_elec_interrupted: More than 7 days | how_many_days_cook_mealsheat_if_gas_disrupted: More than 7 days | how_many_days_provide_food_if_transportation_disrupted: More than 7 days | how_many_days_continued_treatment_if_medication_supply_disrupted: More than 7 days | personal_disaster_preparedness_better_able_to_cope_by_prep: Totally agree | personal_disaster_preparedness_feel_well_prepared: Tend to agree | personal_disaster_preparedness_no_timefin_resources_to_prep: Tend to disagree | personal_disaster_preparedness_easy_to_find_info_on_how_to_prep: Tend to agree | personal_disaster_preparedness_need_more_info_to_prep: Totally agree | personal_disaster_preparedness_know_how_emerg_services_will_alert: Tend to agree | personal_disaster_preparedness_know_what_to_do_in_event_of_disaster: Tend to agree | personal_disaster_preparedness_employerschool_encourages_trainingprep: Totally disagree | personal_disaster_preparedness_emerg_services_encourage_trainingprep: Tend to agree | relying_on_in_first_days_of_disaster_familyfriends: A great deal | relying_on_in_first_days_of_disaster_people_in_neighbourhood: A great deal | relying_on_in_first_days_of_disaster_assocsnonprofit_orgs: A great deal | relying_on_in_first_days_of_disaster_emerg_services: A great deal | relying_on_in_first_days_of_disaster_local_authgvmt_services: A great deal | relying_on_in_first_days_of_disaster_workemployerschooledu_institution: Not a lot | relying_on_in_first_days_of_disaster_private_sector_entities: Not a lot | trust_in_emerg_services_to_handle_disastersemerg_situations_properly: Tend to trust | engaging_in_voluntary_work_for_emerg_responder_orgs: No, you have never engaged in voluntary work and do not plan to do so | occupation_of_respondent: Skilled manual worker | age_recoded_6_categories: 35-44 | size_of_community: Towns/suburbs | direction_things_are_going_life_personally: Things are going in the right direction | political_discussion_local_matters: Never | political_discussion_national_matters: Never | internet_use_total: Everyday/almost everyday (at least once 1 in d62_1 to d62_4) | gender: Woman | age_education: 23.0 | standard_of_living_last_5yrs_in_light_of_crises: Your standard of living has not changed | personal_living_conditions_in_one_years_time: Worse | standard_of_living_next_5yrs: Your standard of living will not change | region_austria: nan | region_belgium: nan | region_bulgaria: nan | region_croatia: nan | region_cyprus: nan | region_czechia: nan | region_denmark: nan | region_germany: nan | region_estonia: nan | region_finland: nan | region_france: nan | region_greece: nan | region_hungary: nan | region_ireland: nan | region_italy: nan | region_latvia: nan | region_lithuania: nan | region_luxembourg: nan | region_malta: nan | region_netherlands: nan | region_poland: nan | region_portugal: nan | region_romania: nan | region_slovenia: nan | region_slovakia: nan | region_spain: ES61 - Andalucia | region_sweden: nan | education_level_preprimary_education_incl_no_education: Not mentioned | education_level_primary_education: Not mentioned | education_level_lower_secondary_education: Not mentioned | education_level_upper_secondary_education: Not mentioned | education_level_postsecondary_non_tertiary_incl_prevocationalvocational: Not mentioned | education_level_shortcycle_tertiary: Not mentioned | education_level_bachelor_or_equivalent: Bachelor or equivalent | education_level_master_or_equivalent: Not mentioned | education_level_doctoral_or_equivalent: Not mentioned"
# parse combined example and request text into dicts for field-wise comparison
example_dict = parse_combined_text_to_dict(example_best_user)
# Load metadata and build an inverted mapping from internal_label -> canonical variable id
meta_path = os.path.join(
os.path.dirname(os.path.dirname(__file__)),
"preparedness",
"data",
"za8841_meta.json",
)
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f)
var_labels = meta.get("column_names_to_labels", {})
col_names = meta.get("column_names", [])
# Build mapping from internal_label -> canonical id using the var_labels and column list
internal_to_id = {}
for col in col_names:
human_label = var_labels.get(col)
if isinstance(human_label, str):
internal = format_column_name(human_label)
internal_to_id[internal] = col
except Exception:
internal_to_id = {}
request_parsed_internal = parse_combined_text_to_dict(request.text)
# # Map internal_label keys back to canonical variable ids when possible
# request_dict = {}
# for ik, val in request_parsed_internal.items():
# # prefer exact match by internal label
# varid = internal_to_id.get(ik)
# if not varid:
# # try case-insensitive normalized match
# for internal_label, cid in internal_to_id.items():
# if internal_label.strip().lower() == ik.strip().lower():
# varid = cid
# break
# # fall back to the original internal key if we couldn't map
# request_dict[varid or ik] = val
# compute simple field diffs: keys present in either dict; value equal, different or missing
keys = sorted(set(list(example_dict.keys()) + list(request_parsed_internal.keys())))
field_diffs = {}
for k in keys:
if "region" in k:
continue
a = example_dict.get(k)
b = request_parsed_internal.get(k)
if a == b:
# field_diffs[k] = {"status": "same", "example": a, "request": b}
continue
else:
field_diffs[k] = {"status": "different", "example": a, "request": b}
example_best_user_embedding = get_ollama_embedding(example_best_user)
prompt_embedding = get_ollama_embedding(request.text)
similarity = float(
cosine_similarity([example_best_user_embedding], [prompt_embedding])[0][0]
)
return {
"similarity": similarity,
"example_parsed": example_dict,
# "request_parsed": request_dict,
"request_parsed_internal_keys": request_parsed_internal,
"field_diffs": field_diffs,
}
@app.get("/value_labels")
async def value_labels():
"""Return the variable-level value labels extracted from the za8841 metadata JSON.
This endpoint returns the `variable_value_labels` section as a JSON object so the client
can use canonical, human-friendly labels for variables where available.
"""
meta_path = os.path.join(
os.path.dirname(os.path.dirname(__file__)),
"preparedness",
"data",
"za8841_meta.json",
)
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f)
variable_value_labels = meta.get("variable_value_labels", {})
# filter to whitelist
filtered = {k: v for k, v in variable_value_labels.items() if k in WHITELIST}
return filtered
except Exception as e:
return Response(
content=json.dumps({"error": str(e)}),
media_type="application/json",
status_code=500,
)
@app.get("/variable_map")
async def variable_map():
"""Return a mapping from human-readable variable label -> variable id.
The source JSON contains a mapping of variable id -> human label (e.g. "qc5_3": "STATEMENTS ...").
This endpoint inverts that mapping so the client can look up the variable id by visible label.
"""
meta_path = os.path.join(
os.path.dirname(os.path.dirname(__file__)),
"preparedness",
"data",
"za8841_meta.json",
)
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f)
var_labels = meta.get("column_names_to_labels", {})
# filter to whitelist
filtered = {k: v for k, v in var_labels.items() if k in WHITELIST}
# invert: map human label -> variable id
inv = {v: k for k, v in filtered.items() if isinstance(v, str)}
return inv
except Exception as e:
return Response(
content=json.dumps({"error": str(e)}),
media_type="application/json",
status_code=500,
)
@app.get("/variables")
async def variables():
"""Return an ordered list of variable metadata suitable for client-side form rendering.
The response format is a list of objects:
[{"id": "qc5_3", "label": "STATEMENTS ...", "values": {"1.0": "Totally agree", ...}}, ...]
This lets the client render each variable using canonical labels and codes.
"""
meta_path = os.path.join(
os.path.dirname(os.path.dirname(__file__)),
"preparedness",
"data",
"za8841_meta.json",
)
try:
with open(meta_path, "r", encoding="utf-8") as f:
meta = json.load(f)
col_names = meta.get("column_names", [])
# human-friendly labels live under `column_names_to_labels` in the metadata
var_labels = meta.get("column_names_to_labels", {})
value_labels = meta.get("variable_value_labels", {})
vars_out = []
for col in col_names:
if format_column_name(var_labels.get(col)) not in WHITELIST:
print(format_column_name(var_labels.get(col)))
continue
item = {
"id": col,
"internal_label": format_column_name(var_labels.get(col)),
"label": var_labels.get(col) or col,
"values": value_labels.get(col) or {},
}
vars_out.append(item)
return vars_out
except Exception as e:
return Response(
content=json.dumps({"error": str(e)}),
media_type="application/json",
status_code=500,
)