Compare commits

...

8 Commits

23 changed files with 25755 additions and 35 deletions
+5
View File
@@ -0,0 +1,5 @@
{
"projects": {
"default": "itsamejms"
}
}
-33
View File
@@ -1,33 +0,0 @@
name: Sphinx build
on:
push:
branches:
- 'main'
jobs:
build:
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest]
steps:
- uses: actions/checkout@v2
- name: Build HTML
uses: ammaraskar/sphinx-action@0.4
with:
# docs-folder: "docs/"
# build-command: "sphinx-build -b html docs/source/ docs/build/html"
pre-build-command: "apt-get update -y && apt-get install -y pandoc"
- name: Upload artifacts
uses: actions/upload-artifact@v1
with:
name: html-docs
path: docs/build/html/
- name: Deploy
uses: peaceiris/actions-gh-pages@v3
if: github.ref == 'refs/heads/main'
with:
github_token: ${{ secrets.GITHUB_TOKEN }}
publish_dir: docs/build/html
+2
View File
@@ -139,3 +139,5 @@ kp_determinants2
*.Rproj
.Rhistory
scratch*
# Firebase
.firebase/
Executable
+5
View File
@@ -0,0 +1,5 @@
#!/usr/bin/env bash
# Build the Sphinx docs and deploy to Firebase Hosting (site: matti-jms).
set -euo pipefail
sphinx-build -b html docs/source/ docs/build/html
firebase deploy --only hosting:matti-jms
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+3
View File
@@ -0,0 +1,3 @@
{
"path": "../../preparedness/analysis.ipynb"
}
+20 -1
View File
@@ -14,7 +14,7 @@ Welcome to Matti Jms Collabs's documentation!
Citizen Shield
----------------------------
.. nbgallery::
:caption: Example Notebooks
:caption: Citizen Shield Notebooks
:name: notebook-gallery
:glob:
@@ -43,6 +43,25 @@ References
|
Preparedness
----------------------------
.. nbgallery::
:caption: Preparedness Notebooks
:name: preparedness-notebook-gallery
:glob:
analysis.ipynb
.. raw:: html
<div id="content">
<li><a href="_static/html/user_embeddings_pca_3d.html">PCA Preparedness</a></li>
<li><a href="_static/html/user_embeddings_tsne_3d.html">t-SNE Preparedness</a></li>
<li><a href="_static/html/user_embeddings_umap_3d.html">UMAP Preparedness</a></li>
</div>
|
Indices and tables
==================
+9
View File
@@ -0,0 +1,9 @@
{
"hosting": {
"site": "matti-jms",
"public": "docs/build/html",
"ignore": ["firebase.json", "**/.*", "**/node_modules/**"],
"cleanUrls": true,
"trailingSlash": false
}
}
File diff suppressed because one or more lines are too long
+239
View File
@@ -0,0 +1,239 @@
# %% [markdown]
# # Visualize Most Prepared Users
# This workflow loads user embeddings, generates a prompt embedding, computes similarity, and visualizes the most prepared users.
# %%
import pandas as pd
import numpy as np
from sklearn.metrics.pairwise import cosine_similarity
from sklearn.decomposition import PCA
from sklearn.manifold import TSNE
import plotly.express as px
import matplotlib.pyplot as plt
import seaborn as sns
import umap
import umap.plot
from utils import get_ollama_embedding
# %% [markdown]
# ## Read in the file with user vectors
# Update the path/format as needed.
# %%
df = (
pd.read_csv("./data/eurobarometer_preparedness_model_data_v3.csv")
.reset_index()
.assign(
**{
"user_id": lambda x: x["index"].astype(str)
+ "_"
+ x["country_code_iso_3166"]
+ "_"
+ x["age_recoded_6_categories"].astype(str)
+ "_"
+ x["gender"].astype(str)
}
)
# select only relevant countries for visualization
.loc[
lambda x: x["country_code_iso_3166"].isin(
["FI", "DE-E", "DE-W", "FR", "ES", "PT"]
)
]
)
df.shape
# relabel None of the above/ Non binary/ do not recognize yourself in above categories/Prefer not to say to other
df["gender"] = df["gender"].replace({
"None of the above/ Non binary/ do not recognize yourself in above categories/Prefer not to say": "Other",
})
# %% [markdown]
# ## Convert string embeddings to lists if needed
# %%
def parse_embedding(x):
if isinstance(x, str):
return [float(i) for i in x.strip("[]").split(",")]
return x
df["embedding"] = df["embedding"].apply(parse_embedding)
# %%
# Select best user:
# disaster_measures_in_hh_* == 1, and how_many_days_meet_* == (4 | 5)
# Following relabeling in data_preparation_raw.py:
# disaster_measures_in_hh_* != "Not mentioned"
# how_many_days_meet_* == "More than 7 days"
disaster_measures_in_hh_columns = [
col for col in df.columns if col.startswith("disaster_measures_in_hh_")
]
how_many_days_meet_columns = [
col for col in df.columns if col.startswith("how_many_days_meet_")
]
best_users = df[
(df[disaster_measures_in_hh_columns] != "Not mentioned").all(axis=1)
& (df[how_many_days_meet_columns] == "More than 7 days").all(axis=1)
]
print(f"Number of best users: {len(best_users)}")
best_users.head(5)[["user_id", "user_text"]].to_dict(orient="records")
# %%
best_users.columns.tolist()
# %% [markdown]
# ## Write your prompt and generate its embedding
# %%
# prompt = "The user is highly prepared for disasters, with emergency supplies and a clear plan."
prompt = best_users.head(1)["user_text"].values[
0
] # Example: use the first user's text as the prompt
prompt_embedding = get_ollama_embedding(prompt)
# %% [markdown]
# ## Compute similarity between each user and the prompt
# %%
user_embeddings = np.vstack(df["embedding"].values)
prompt_vec = np.array(prompt_embedding).reshape(1, -1)
similarities = cosine_similarity(user_embeddings, prompt_vec).flatten()
df["similarity"] = similarities
# %% [markdown]
# ## Visualize the users by similarity
# %%
_ = plt.figure(figsize=(5, 7))
_ = sns.violinplot(
x="similarity", y="country_code_iso_3166", hue="country_code_iso_3166", data=df
)
_ = sns.stripplot(
x="similarity",
y="country_code_iso_3166",
data=df,
hue="country_code_iso_3166",
alpha=0.8,
jitter=True,
linewidth=0.5,
edgecolor="white",
)
_ = plt.title("Distribution of User Similarities to Preparedness Prompt")
_ = plt.xlabel("Cosine Similarity")
# %%
# np.savetxt("./data/embeddings.tsv", np.vstack(df["embedding"].values), delimiter="\t")
# # %%
# df[["user_id"]].to_csv(
# "./data/metadata.tsv", sep="\t", index=False
# )
# %%
# Reduce embeddings to 3D with PCA
pca = PCA(n_components=3)
embeddings_3d = pca.fit_transform(np.vstack(df["embedding"].values))
# Add PCA components to dataframe
df["pca1"] = embeddings_3d[:, 0]
df["pca2"] = embeddings_3d[:, 1]
df["pca3"] = embeddings_3d[:, 2]
# %%
# Interactive 3D scatter plot
fig = px.scatter_3d(
df,
x="pca1",
y="pca2",
z="pca3",
color="age_recoded_6_categories",
hover_data=["user_id", "gender", "age_recoded_6_categories"],
title="User Embeddings (PCA 3D)"
)
fig.write_html("./output/user_embeddings_pca_3d.html")
# %%
# Reduce embeddings to 3D with t-SNE
tsne = TSNE(n_components=3, random_state=42, perplexity=30)
embeddings_3d = tsne.fit_transform(np.vstack(df["embedding"].values))
# Add t-SNE components to dataframe
df["tsne1"] = embeddings_3d[:, 0]
df["tsne2"] = embeddings_3d[:, 1]
df["tsne3"] = embeddings_3d[:, 2]
# Interactive 3D scatter plot
fig = px.scatter_3d(
df,
x="tsne1",
y="tsne2",
z="tsne3",
color="age_recoded_6_categories",
hover_data=["user_id", "gender", "country_code_iso_3166"],
title="User Embeddings (t-SNE 3D)"
)
fig.write_html("./output/user_embeddings_tsne_3d.html")
# %%
# example comparison based on t-sne projection
index_1 = 9561
index_2 = 5096
df.loc[[index_1, index_2]][["user_id", "user_text", "similarity"]].to_dict(orient="records")
# %%
row1 = df.loc[index_1].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
row2 = df.loc[index_2].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
# column_list = df.columns.difference(["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3"])
column_list = disaster_measures_in_hh_columns + how_many_days_meet_columns
diffs = {}
for col in column_list:
val1 = row1[col]
val2 = row2[col]
if pd.isnull(val1) and pd.isnull(val2):
continue
if val1 != val2:
diffs[col] = (val1, val2)
# Print or display the differing columns and their values
for col, (v1, v2) in diffs.items():
print(f"{col}: {v1} | {v2}")
# %%
# Reduce embeddings to 3D with UMAP
umap_3d = umap.UMAP(n_components=3, random_state=42, metric="cosine").fit_transform(np.vstack(df["embedding"].values))
# Add UMAP components to dataframe
df["umap1"] = umap_3d[:, 0]
df["umap2"] = umap_3d[:, 1]
df["umap3"] = umap_3d[:, 2]
# Interactive 3D scatter plot
fig = px.scatter_3d(
df,
x="umap1",
y="umap2",
z="umap3",
color="country_code_iso_3166",
hover_data=["user_id", "gender", "country_code_iso_3166"],
title="User Embeddings (UMAP 3D)"
)
fig.write_html("./output/user_embeddings_umap_3d.html")
# %%
# example comparison based on UMAP projection
index_1 = 21396
index_2 = 10444
df.loc[[index_1, index_2]][["user_id", "user_text", "similarity"]].to_dict(orient="records")
# %%
row1 = df.loc[index_1].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
row2 = df.loc[index_2].drop(labels=["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
# column_list = df.columns.difference(["user_id", "embedding", "user_text", "similarity", "pca1", "pca2", "pca3", "tsne1", "tsne2", "tsne3", "umap1", "umap2", "umap3"])
column_list = disaster_measures_in_hh_columns + how_many_days_meet_columns
diffs = {}
for col in column_list:
val1 = row1[col]
val2 = row2[col]
if pd.isnull(val1) and pd.isnull(val2):
continue
if val1 != val2:
diffs[col] = (val1, val2)
# Print or display the differing columns and their values
for col, (v1, v2) in diffs.items():
print(f"{col}: {v1} | {v2}")
# %%
+305
View File
File diff suppressed because one or more lines are too long
+142
View File
@@ -0,0 +1,142 @@
# %%
import pandas as pd
import numpy as np
from utils import get_ollama_embedding
# %%[markdown]
#### Data Preparation Steps
# - Select relevant columns from the cleaned dataset, including a range and specific variables.
# - Exclude columns containing substrings like "2nd", "spont", "other", and some specific variables.
# - Add columns starting with "region_" and "education_level_".
# - Convert character columns to categorical type.
# - Generate a `subregion` variable by coalescing region columns, then drop the originals.
# - Generate an `education_level` variable by coalescing education columns, replacing "Not mentioned" with NA, then drop the originals.
# - Define core preparedness items and their human-readable labels.
# - Map country codes to country names.
# %%
# Load your data
data_cleaned = pd.read_csv('./data/eurobarometer_data_cleaned_csv.csv')
print(data_cleaned.shape)
# Select columns by name and range
cols_to_select = [
'country_code_iso_3166',
'risks_cntry_most_exposed_to_firstly',
'risks_pers_most_exposed_to_firstly',
'risks_pers_most_exposed_to_number_of_mentioned_risks',
'pot_info_sources_to_learn_about_disaster_risks_firstly',
*data_cleaned.columns[404:448], # Python is 0-indexed
'occupation_of_respondent',
'age_recoded_6_categories',
'size_of_community',
'social_class_self_assessment_5_cat',
'direction_things_are_going_life_personally',
'political_discussion_local_matters',
'political_discussion_national_matters',
'left_right_placement_recoded_5_cat',
'internet_use_total',
'gender',
'age_education',
'standard_of_living_last_5yrs_in_light_of_crises',
'personal_living_conditions_in_one_years_time',
'standard_of_living_next_5yrs'
]
# Remove columns containing certain substrings
exclude_patterns = ['2nd', 'spont', 'other']
cols_to_exclude = [col for col in data_cleaned.columns if any(p in col for p in exclude_patterns)]
cols_to_exclude += [
'disaster_measures_in_hh_number_of_measures',
'disaster_pers_experienced_past_10yrs_none',
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
]
# Add region_ and education_level_ columns
cols_to_select += [col for col in data_cleaned.columns if col.startswith('region_')]
cols_to_select += [col for col in data_cleaned.columns if col.startswith('education_level_')]
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
df_model = data_cleaned[final_cols].copy()
# %% [markdown]
# #### Combine all columns into a single string per user
# This step creates a text representation of each user, which can be sent to an embedding model.
def row_to_string(row):
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
# Combine all columns into a single string per user (except country_code_iso_3166)
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
df_model.head()
# %%
# Convert character columns to category BEFORE adding embedding column
for col in df_model.select_dtypes(include='object').columns:
if col != 'user_text':
df_model[col] = df_model[col].astype('category')
# Generate embeddings for each user using Ollama
df_model['embedding'] = df_model['user_text'].apply(get_ollama_embedding)
# Add new columns to final_cols
final_cols.extend(["user_text", "embedding"])
# Check the lengths of all embeddings
embedding_lengths = df_model['embedding'].apply(lambda x: len(x) if isinstance(x, list) else None)
print('Embedding lengths:', embedding_lengths.tolist())
df_model = df_model[final_cols].copy()
# Generate subregion variable
region_cols = [col for col in df_model.columns if col.startswith('region_')]
df_model['subregion'] = df_model[region_cols].bfill(axis=1).iloc[:, 0]
df_model.drop(columns=region_cols, inplace=True)
# Generate education_level variable
edu_cols = [col for col in df_model.columns if col.startswith('education_level_')]
for col in edu_cols:
df_model[col] = df_model[col].replace('Not mentioned', np.nan)
df_model['education_level'] = df_model[edu_cols].bfill(axis=1).iloc[:, 0]
df_model['education_level'] = df_model['education_level'].astype('category')
df_model.drop(columns=edu_cols, inplace=True)
# Core preparedness items
CORE_ITEMS_MAPPED = [
"disaster_measures_in_hh_emergency_supply_drinks_food",
"disaster_measures_in_hh_emergency_supply_water_cooking_hygiene",
"disaster_measures_in_hh_agreed_with_friends_family_to_contact",
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood",
"disaster_measures_in_hh_battery_powered_radio"
]
CORE_ITEM_LABELS = {
"disaster_measures_in_hh_emergency_supply_drinks_food": "Emergency supply of drinks, food",
"disaster_measures_in_hh_emergency_supply_water_cooking_hygiene": "Emergency supply of cooking and hygiene water",
"disaster_measures_in_hh_agreed_with_friends_family_to_contact": "Agreed with family, friends on how to contact in an emergency",
"disaster_measures_in_hh_discussed_common_prot_measures_in_neighbourhood": "Discussed precautions in neighbourhood",
"disaster_measures_in_hh_battery_powered_radio": "Battery-powered radio accessible"
}
COUNTRY_NAME_MAP = {
"FI": "Finland",
"DE-E": "East Germany",
"DE-W": "West Germany",
"FR": "France",
"ES": "Spain",
"PT": "Portugal",
# "EE": "Estonia",
# "DK": "Denmark",
# "SE": "Sweden",
# "NL": "Netherlands"
}
# %%
df_model.head()
# %%
df_model.head().to_dict(orient='records')
# %%
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v2.csv', index=False)
+137
View File
@@ -0,0 +1,137 @@
# %%
import pyreadstat
import json
import re
import datetime
from utils import get_ollama_summary, get_ollama_embedding
# %%
df_sav, meta = pyreadstat.read_sav('./data/ZA8841_v1-0-0.sav')
print(df_sav.shape)
# %%
df_sav.head(1).to_dict(orient='records')
# %%
def safe_json(obj):
if isinstance(obj, (datetime.datetime, datetime.date)):
return obj.isoformat()
if isinstance(obj, set):
return list(obj)
if hasattr(obj, '__dict__'):
return str(obj)
return obj
meta_dict = vars(meta)
meta_json = json.dumps(meta_dict, default=safe_json, indent=2)
# %%
with open('./data/za8841_meta.json', 'w') as f:
f.write(meta_json)
# %%
# Load your mapping dictionary (from JSON file or directly)
with open("./data/za8841_meta.json") as f:
meta = json.load(f)
value_labels = meta["variable_value_labels"]
# %%
# Assume df_sav is your loaded SPSS dataframe
# For each column in the mapping, map values if the column exists in df_sav
for col, mapping in value_labels.items():
if col in df_sav.columns:
# Convert keys to float if needed (SPSS values often are float)
mapping_float = {float(k): v for k, v in mapping.items()}
df_sav[col] = df_sav[col].map(mapping_float).fillna(df_sav[col])
# Now all mapped columns have human-readable values
print(df_sav.head())
# %%
labels_map = meta["column_names_to_labels"]
def make_pandas_friendly(col):
col = labels_map.get(col, col)
col = re.sub(r'[.\s]+', '_', col)
col = re.sub(r'[^0-9a-zA-Z_]', '', col)
col = col.lower()
col = re.sub(r'__+', '_', col) # Replace double (or more) underscores with single
col = col.strip('_') # Remove leading/trailing underscores
return col
df_sav.columns = [make_pandas_friendly(col) for col in df_sav.columns]
# %%
df_sav.head(1).to_dict(orient='records')
# %%
cols_to_select = [
'country_code_iso_3166',
'risks_cntry_most_exposed_to_firstly',
'risks_pers_most_exposed_to_firstly',
'risks_pers_most_exposed_to_number_of_mentioned_risks',
'pot_info_sources_to_learn_about_disaster_risks_firstly',
*df_sav.columns[404:448], # Python is 0-indexed
'occupation_of_respondent',
'age_recoded_6_categories',
'size_of_community',
# 'social_class_self_assessment_5_cat', # not in the data due to mapping
'direction_things_are_going_life_personally',
'political_discussion_local_matters',
'political_discussion_national_matters',
# 'left_right_placement_recoded_5_cat', # not in the data due to mapping
'internet_use_total',
'gender',
'age_education',
'standard_of_living_last_5yrs_in_light_of_crises',
'personal_living_conditions_in_one_years_time',
'standard_of_living_next_5yrs'
]
# Remove columns containing certain substrings
exclude_patterns = ['2nd', 'spont', 'other']
cols_to_exclude = [col for col in df_sav.columns if any(p in col for p in exclude_patterns)]
cols_to_exclude += [
'disaster_measures_in_hh_number_of_measures',
'disaster_pers_experienced_past_10yrs_none',
'pot_info_sources_to_learn_about_disaster_risks_interested_in_at_least_one_source'
]
# Add region_ and education_level_ columns
cols_to_select += [col for col in df_sav.columns if col.startswith('region_')]
cols_to_select += [col for col in df_sav.columns if col.startswith('education_level_')]
final_cols = [col for col in cols_to_select if col not in cols_to_exclude]
print(f"Final number of columns: {len(final_cols)}")
final_cols
# %%
df_model = df_sav[final_cols].copy()
# %%
# %% [markdown]
# #### Combine all columns into a single string per user
# This step creates a text representation of each user, which can be sent to an embedding model.
def row_to_string(row):
return ' | '.join(f'{col}: {row[col]}' for col in row.index)
# Combine all columns into a single string per user (except country_code_iso_3166)
df_model['user_text'] = df_model.drop(columns=['country_code_iso_3166']).apply(row_to_string, axis=1)
df_model.head()
# %%
# Convert character columns to category BEFORE adding embedding column
for col in df_model.select_dtypes(include='object').columns:
if col != 'user_text':
df_model[col] = df_model[col].astype('category')
# %%
# df_model["summary"] = df_model['user_text'].apply(get_ollama_summary)
# df_model.head(1)['user_text'].apply(get_ollama_summary).to_list()
df_model["embedding"] = df_model['user_text'].apply(get_ollama_embedding)
# %%
df_model.to_csv('./data/eurobarometer_preparedness_model_data_v3.csv', index=False)
+20
View File
@@ -0,0 +1,20 @@
import logging
import uvicorn
import os
from api import app
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger(__name__)
def run_api() -> None:
"""Run the FastAPI application."""
if "james" in os.environ.get("USER", ""):
logger.info("Running in James's environment")
uvicorn.run("api:app", host="0.0.0.0", port=8080, reload=True)
else:
uvicorn.run(app, host="0.0.0.0", port=8080)
if __name__ == "__main__":
run_api()
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
+12
View File
@@ -0,0 +1,12 @@
pandas
ipykernel
requests
scikit-learn
matplotlib
seaborn
pyreadstat
plotly
nbformat>=4.2.0
umap-learn[plot]
fastapi
uvicorn
+4
View File
@@ -0,0 +1,4 @@
from pydantic import BaseModel
class PredictionRequest(BaseModel):
text: str
+757
View File
@@ -0,0 +1,757 @@
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8">
<meta name="viewport" content="width=device-width,initial-scale=1">
<title>Preparedness Survey — Quick Form</title>
<style>
body {
font-family: system-ui, Segoe UI, Roboto, Helvetica, Arial, sans-serif;
margin: 28px;
background: #f7fafc
}
.container {
max-width: 980px;
margin: 0 auto;
background: #fff;
padding: 28px;
border-radius: 10px;
box-shadow: 0 6px 20px rgba(2, 6, 23, .08)
}
h1 {
margin-top: 0;
margin-bottom: 18px;
font-size: 28px;
}
.row {
display: flex;
gap: 16px;
margin-bottom: 14px
}
label {
display: block;
font-weight: 600;
margin-bottom: 4px
}
input[type=text],
select,
textarea,
input[type=number] {
width: 100%;
padding: 12px;
border: 1px solid #d1d5db;
border-radius: 8px;
font-size: 14px;
}
.grid {
display: block;
margin-top: 18px;
margin-bottom: 18px;
}
/* form rows: labels above inputs for cleaner reading and to avoid overlap */
.form-row {
display: block;
padding: 12px 0;
border-bottom: 1px solid #eef2f6;
margin: 0;
}
/* label shown above the input area */
.form-row>label,
.label-top {
display: block;
width: 100%;
margin: 0 0 8px 0;
font-weight: 600;
font-size: 14px;
color: #111827;
}
.form-row .input-wrap {
width: 100%;
margin: 0;
}
/* keep Likert rows visually compact with the question label above */
.form-row.likert-row {
display: block
}
.form-row.likert-row .input-wrap.likert {
display: flex;
gap: 12px;
flex-wrap: wrap
}
@media (max-width:700px) {
.form-row {
flex-direction: column;
align-items: stretch
}
.form-row label {
width: 100%
}
.form-row .input-wrap {
width: 100%
}
}
.likert {
display: flex;
gap: 12px;
align-items: center;
flex-wrap: wrap
}
/* each option is an inline radio + text so the options form a horizontal row */
.likert label {
display: flex;
flex-direction: row;
align-items: center;
gap: 6px;
font-weight: 400;
white-space: nowrap;
padding: 4px 6px;
border-radius: 6px
}
.likert input[type=radio] {
margin: 0 6px 0 0
}
/* Normalize likert groups so all questions display consistently */
.input-wrap.likert {
display: flex;
}
.input-wrap.likert .likert,
.input-wrap.likert {
width: 100%;
align-items: center
}
.input-wrap.likert .likert {
justify-content: flex-start;
gap: 14px;
}
.input-wrap.likert label {
padding: 6px 10px;
border-radius: 8px;
background: transparent
}
.input-wrap.likert label:hover {
background: rgba(2, 6, 23, 0.03)
}
.input-wrap.likert input[type=radio] {
transform: scale(1.05);
}
/* toggle switch for binary yes/no selects */
.toggle-switch {
display: inline-flex;
align-items: center;
cursor: pointer
}
.toggle-switch input {
display: none
}
.toggle-switch .slider {
width: 48px;
height: 26px;
background: #e5e7eb;
border-radius: 999px;
position: relative;
transition: background .18s ease
}
.toggle-switch .slider::after {
content: '';
position: absolute;
left: 4px;
top: 4px;
width: 18px;
height: 18px;
background: #fff;
border-radius: 50%;
transition: transform .18s ease
}
.toggle-switch input:checked+.slider {
background: #10b981
}
.toggle-switch input:checked+.slider::after {
transform: translateX(22px)
}
.actions {
display: flex;
gap: 12px;
margin-top: 20px
}
button {
padding: 10px 14px;
border-radius: 8px;
border: 0;
background: #0069ff;
color: #fff;
font-weight: 600;
}
pre {
background: #0b1220;
color: #e6eef8;
padding: 16px;
border-radius: 8px;
overflow: auto
}
h3 {
margin-top: 28px;
margin-bottom: 12px;
font-size: 18px;
color: #0f172a
}
p {
margin-bottom: 18px;
color: #374151
}
small {
color: #6b7280
}
</style>
</head>
<body>
<div class="container">
<h1>Preparedness survey — quick form</h1>
<p>Fill the fields you have, then use <strong>Predict</strong> to send the form as JSON to the server's
<code>/predict</code> endpoint and see the model response below.
</p>
<form id="survey">
<!-- dynamic form will be rendered here from metadata -->
<div id="dynamic-grid" class="grid"></div>
<!-- All static questions removed. The form is fully metadata-driven and rendered into #dynamic-grid -->
<div class="actions">
<button type="button" id="predict">Predict</button>
<button type="button" id="selftest">Self-test</button>
</div>
<h3>JSON preview</h3>
<pre id="output">{ }</pre>
</form>
<script>
// Note: selects were converted to static markup; label/likert styling is handled in the CSS above.
const form = document.getElementById('survey');
const output = document.getElementById('output');
function collectForm() {
const data = {};
// include all inputs/selects
const elements = Array.from(form.elements).filter(e => e.name);
// handle radios by grouping
const handled = new Set();
elements.forEach(el => {
const name = el.name;
if (handled.has(name)) return;
if (el.type === 'radio') {
const radios = form.querySelectorAll(`input[name="${name}"]`);
const checked = Array.from(radios).find(r => r.checked);
data[name] = checked ? checked.value : null;
handled.add(name);
return;
}
if (el.type === 'checkbox') {
// if this checkbox was created as a toggle with data-yes/data-no, return the string
const yes = el.dataset && el.dataset.yes;
const no = el.dataset && el.dataset.no;
if (yes !== undefined && no !== undefined) {
data[name] = el.checked ? yes : no;
} else {
data[name] = el.checked ? 1 : 0;
}
handled.add(name);
return;
}
if (el.type === 'number' || el.type === 'range') {
const v = el.value;
data[name] = v === '' ? null : Number(v);
handled.add(name);
return;
}
// select, text, textarea, etc.
data[name] = el.value === '' ? null : el.value;
handled.add(name);
});
return data;
}
// Map compact option values to human-friendly labels used in the canonical example
const VALUE_LABELS = {
// general likert / frequency
'very_much': 'A great deal',
'somewhat': 'Somewhat',
'not_much': 'Not much',
'not_at_all': 'Not at all',
'strongly_agree': 'Strongly agree',
'agree': 'Agree',
'neutral': 'Neutral',
'disagree': 'Disagree',
'strongly_disagree': 'Strongly disagree',
'very_trust': 'Very trust',
'somewhat_trust': 'Somewhat trust',
'somewhat_distrust': 'Somewhat distrust',
'very_distrust': 'Very distrust',
'daily': 'Daily',
'weekly': 'Weekly',
'monthly': 'Monthly',
'rarely': 'Rarely',
'never': 'Never',
'female': 'Female',
'male': 'Male',
'other': 'Other',
'prefer_not': 'Prefer not to say',
'getting_better': 'Getting better',
'staying_same': 'Staying the same',
'getting_worse': 'Getting worse',
'often': 'Often',
'sometimes': 'Sometimes',
'improved': 'Improved',
'same': 'Same',
'worse': 'Worse',
'better': 'Better',
'rural': 'Rural',
'small_town': 'Small town',
'suburb': 'Suburb',
'city': 'City',
// toggles / yes-no
'yes': 'Yes',
'no': 'No'
};
// SERVER_VALUE_LABELS will hold the variable-level mappings returned from the server
// structure: { varName: { codeStr: label, ... }, ... }
let SERVER_VALUE_LABELS = {};
// FIELD_LABELS maps our stable field name (input.name) -> human-friendly label
// FIELD_INTERNALS maps stable field name -> server-provided internal_label (snake_case id)
// populated when we render the variables list so we can use internal_label when sending
// the combined text (and still show human-friendly labels in the UI).
let FIELD_LABELS = {};
let FIELD_INTERNALS = {};
// FIELD_VARNAMES maps stable field name -> canonical variable id (as returned in variables[].id)
// we need this so we can look up SERVER_VALUE_LABELS by the canonical id when resolving labels
let FIELD_VARNAMES = {};
// VAR_VALUES stores the values map returned in /variables for each canonical var id
// structure: { varId: { code: label, ... }, ... }
let VAR_VALUES = {};
// Load server-provided variable metadata and render the entire form from it.
(async function loadVariablesAndRender() {
try {
// Load variable metadata (ordered list)
const varsResp = await fetch('/variables');
if (!varsResp.ok) throw new Error('Could not load /variables');
const variables = await varsResp.json();
// load server-side value labels for quick lookup too
const vlResp = await fetch('/value_labels');
if (vlResp.ok) {
SERVER_VALUE_LABELS = await vlResp.json();
}
const grid = document.getElementById('dynamic-grid');
grid.innerHTML = '';
console.log('[survey] Loaded', variables.length, 'variables from /variables');
// heuristic: certain variable name prefixes indicate a range/numeric field
function looksLikeRange(varId) {
return varId.startsWith('how_many_days') || varId.startsWith('age_') || varId.startsWith('age') || varId.startsWith('age_recoded');
}
variables.forEach(v => {
const row = document.createElement('div');
row.className = 'form-row';
const lbl = document.createElement('label');
lbl.textContent = v.label || v.id;
row.appendChild(lbl);
const wrap = document.createElement('div');
wrap.className = 'input-wrap';
// stable field name for submission: use snake_case of id
const fieldName = String(v.id).toLowerCase().replace(/[^a-z0-9]+/g, '_');
wrap.setAttribute('data-field', fieldName);
wrap.setAttribute('data-var', v.id);
// remember the human label and internal_label for display and submission
FIELD_LABELS[fieldName] = v.label || v.id;
FIELD_INTERNALS[fieldName] = v.internal_label || fieldName;
FIELD_VARNAMES[fieldName] = v.id;
VAR_VALUES[v.id] = v.values || {};
console.log('[survey] FIELD_LABELS set:', fieldName, '=>', FIELD_LABELS[fieldName]);
console.log('[survey] FIELD_INTERNALS set:', fieldName, '=>', FIELD_INTERNALS[fieldName]);
// if we have value labels for this var, render radios or toggles
const vals = v.values || {};
const codes = Object.keys(vals || {});
if (codes.length === 0) {
// no canonical values: choose text or range based on heuristics
if (looksLikeRange(v.id)) {
const inp = document.createElement('input');
inp.type = 'range';
inp.name = fieldName;
inp.id = fieldName;
inp.min = 0;
inp.max = 30;
inp.step = 1;
wrap.appendChild(inp);
const span = document.createElement('span');
span.className = 'range-value';
span.setAttribute('data-for', fieldName);
span.textContent = '0';
wrap.appendChild(span);
} else {
const inp = document.createElement('input');
inp.type = 'text';
inp.name = fieldName;
inp.id = fieldName;
wrap.appendChild(inp);
}
} else {
// render canonical codes -> labels; detect binary numeric yes/no
const numericCodes = codes.map(c => Number(String(c))).filter(n => !Number.isNaN(n));
const isBinary = (codes.length === 2) && (numericCodes.includes(0) || numericCodes.includes(1) || numericCodes.includes(2));
if (isBinary) {
const yesCode = codes.find(c => Number(c) === 1) || codes.find(c => Number(c) === 2) || codes[0];
const noCode = codes.find(c => c !== yesCode) || null;
const wrapper = document.createElement('div');
const label = document.createElement('label');
label.className = 'toggle-switch';
const input = document.createElement('input');
input.type = 'checkbox';
input.name = fieldName;
input.setAttribute('data-yes', yesCode.replace(/\.0$/, ''));
input.setAttribute('data-no', noCode ? noCode.replace(/\.0$/, '') : '0');
const slider = document.createElement('span');
slider.className = 'slider';
label.appendChild(input);
label.appendChild(slider);
wrapper.appendChild(label);
const small = document.createElement('small');
small.style.marginLeft = '10px';
small.textContent = vals[yesCode] || 'Yes';
wrapper.appendChild(small);
wrap.appendChild(wrapper);
} else {
const likert = document.createElement('div');
likert.className = 'likert';
codes.slice().sort((a,b) => {
const na = Number(a), nb = Number(b);
if (!Number.isNaN(na) && !Number.isNaN(nb)) return na - nb;
return String(a).localeCompare(String(b));
}).forEach(code => {
const label = document.createElement('label');
const input = document.createElement('input');
input.type = 'radio';
input.name = fieldName;
input.value = String(code).replace(/\.0$/, '');
label.appendChild(input);
label.appendChild(document.createTextNode(' ' + (vals[code] || code)));
likert.appendChild(label);
});
wrap.appendChild(likert);
}
}
row.appendChild(wrap);
grid.appendChild(row);
});
// init sliders to show values
initSliders();
} catch (e) {
console.warn('Failed to load variables:', e && e.message ? e.message : e);
}
})();
// Render inputs for elements with data-var attribute using SERVER_VALUE_LABELS.
// For each container with data-var="<varname>", we create controls based on the mapping:
// - If the mapping looks binary (codes like 0/1) we render a toggle checkbox (data-yes/data-no)
// - Otherwise we render radio inputs for each code->label pair (values are code without .0)
function renderVariableControls() {
console.log('[survey] renderVariableControls start, SERVER_VALUE_LABELS keys:', Object.keys(SERVER_VALUE_LABELS).length);
const containers = Array.from(document.querySelectorAll('[data-var]'));
containers.forEach(container => {
const varName = container.getAttribute('data-var');
if (!varName) return;
const map = SERVER_VALUE_LABELS[varName];
if (!map || typeof map !== 'object') return;
// clear existing content
container.innerHTML = '';
const codes = Object.keys(map);
// detect binary mapping (common pattern: 0.0/1.0 or 1.0/2.0 for yes/no)
const numericCodes = codes.map(c => Number(String(c)) ).filter(n => !Number.isNaN(n));
const isBinary = (codes.length === 2) && (numericCodes.includes(0) || numericCodes.includes(1) || numericCodes.includes(2));
const fieldName = container.getAttribute('data-field') || varName;
if (isBinary) {
// determine the 'yes' code (prefer 1, then 2)
const yesCode = codes.find(c => Number(c) === 1) || codes.find(c => Number(c) === 2) || codes[0];
const noCode = codes.find(c => c !== yesCode) || null;
const yesLabel = map[yesCode] || 'Yes';
const noLabel = noCode ? map[noCode] : 'No';
// checkbox: when checked -> yes (1), unchecked -> no (0)
const wrapper = document.createElement('div');
const label = document.createElement('label');
label.className = 'toggle-switch';
const input = document.createElement('input');
input.type = 'checkbox';
input.name = fieldName;
// encode yes/no as data attributes so collectForm interprets correctly
input.setAttribute('data-yes', yesCode.replace(/\.0$/, ''));
input.setAttribute('data-no', noCode ? noCode.replace(/\.0$/, '') : '0');
const slider = document.createElement('span');
slider.className = 'slider';
label.appendChild(input);
label.appendChild(slider);
wrapper.appendChild(label);
const small = document.createElement('small');
small.style.marginLeft = '10px';
small.textContent = yesLabel;
wrapper.appendChild(small);
container.appendChild(wrapper);
} else {
// render radios; sort codes numerically when possible for stable ordering
const sorted = codes.slice().sort((a, b) => {
const na = Number(a), nb = Number(b);
if (!Number.isNaN(na) && !Number.isNaN(nb)) return na - nb;
return String(a).localeCompare(String(b));
});
sorted.forEach(code => {
const labelText = map[code];
const valueToken = String(code).replace(/\.0$/, '');
const label = document.createElement('label');
const input = document.createElement('input');
input.type = 'radio';
input.name = fieldName;
input.value = valueToken;
label.appendChild(input);
label.appendChild(document.createTextNode(' ' + labelText));
container.appendChild(label);
});
}
});
}
// Helper: return the human-friendly label for a given field and token.
// Lookup order:
// 1) If SERVER_VALUE_LABELS has an entry for fieldName and a matching code -> return label.
// We attempt exact match on token, token+'.0', and token with trailing '.0' removed.
// 2) If VALUE_LABELS has an entry for token -> return that.
// 3) Else return null to indicate no mapping.
function labelFor(fieldName, token) {
if (!token || token === 'nan') return null;
// server-side per-variable lookup
// determine canonical var id for this field so we can resolve per-variable value labels
const varKey = FIELD_VARNAMES[fieldName] || fieldName;
// Prefer VAR_VALUES (values shipped with /variables) as it contains the recoded labels
if (VAR_VALUES && VAR_VALUES[varKey]) {
const map = VAR_VALUES[varKey];
// 1) If the token directly matches a code key, return the server label
if (map.hasOwnProperty(token)) return map[token];
// try token + '.0' and stripped
if (map.hasOwnProperty(token + '.0')) return map[token + '.0'];
const strippedToken = token.replace(/\.0$/, '');
if (map.hasOwnProperty(strippedToken)) return map[strippedToken];
// 2) If the token maps locally to a human label, see if the server map contains that label
// (this lets us map semantic tokens like 'very_much' -> 'A great deal' if the server
// contains that exact human label for the field)
if (VALUE_LABELS.hasOwnProperty(token)) {
const human = VALUE_LABELS[token];
// try exact value match (case-sensitive); also try trimmed/case-insensitive
for (const k of Object.keys(map)) {
const serverLabel = map[k];
if (!serverLabel) continue;
if (serverLabel === human) return serverLabel;
if (serverLabel.trim().toLowerCase() === human.trim().toLowerCase()) return serverLabel;
}
}
// 3) additional numeric equivalence handled above
// 4) numeric equivalence match
try {
const numToken = Number(token);
if (!Number.isNaN(numToken)) {
for (const k of Object.keys(map)) {
if (!isNaN(Number(k)) && Number(k) === numToken) return map[k];
}
}
} catch (e) { /* ignore */ }
}
// fallback to server-provided per-variable mapping if present
if (SERVER_VALUE_LABELS && SERVER_VALUE_LABELS[varKey]) {
const map = SERVER_VALUE_LABELS[varKey];
if (map.hasOwnProperty(token)) return map[token];
if (map.hasOwnProperty(token + '.0')) return map[token + '.0'];
const stripped2 = token.replace(/\.0$/, '');
if (map.hasOwnProperty(stripped2)) return map[stripped2];
try {
const numToken = Number(token);
if (!Number.isNaN(numToken)) {
for (const k of Object.keys(map)) {
if (!isNaN(Number(k)) && Number(k) === numToken) return map[k];
}
}
} catch (e) { /* ignore */ }
}
// fallback to global VALUE_LABELS exact match
if (VALUE_LABELS.hasOwnProperty(token)) {
console.log('[survey] labelFor fallback global VALUE_LABELS for', token, '->', VALUE_LABELS[token]);
return VALUE_LABELS[token];
}
console.log('[survey] labelFor no mapping for', fieldName, token);
return null;
}
document.getElementById('predict').addEventListener('click', async () => {
const d = collectForm();
// Build a combined string of key:value for all form fields in a stable order
// Missing values become 'nan' to mirror downstream expectations
const parts = [];
Object.keys(d).forEach(k => {
let v = d[k];
if (v === null || v === undefined || v === '') {
v = 'nan';
} else if (typeof v === 'object') {
v = JSON.stringify(v);
} else {
v = String(v);
}
// Attempt to map token to friendly label using per-field labels first,
// then global fallback.
if (typeof v === 'string') {
const fieldLabelVal = labelFor(k, v);
if (fieldLabelVal) {
v = fieldLabelVal;
}
}
// Use server-provided internal_label (if available) as the key in the combined text
// Fall back to the human-friendly FIELD_LABELS or the field name
const internalKey = FIELD_INTERNALS[k] || FIELD_LABELS[k] || k;
parts.push(`${internalKey}: ${v}`);
});
const combined = parts.join(' | ');
output.textContent = 'Sending...';
console.log('[survey] Sending combined payload to /predict');
try {
// Build an internal-keyed mapping to send alongside the combined text
const request_parsed_internal = {};
Object.keys(d).forEach(k => {
const internalKey = FIELD_INTERNALS[k] || FIELD_LABELS[k] || k;
request_parsed_internal[internalKey] = d[k] === null || d[k] === undefined || d[k] === '' ? 'nan' : d[k];
});
const res = await fetch('/predict', {
method: 'POST',
headers: { 'Content-Type': 'application/json' },
body: JSON.stringify({ text: combined })
});
if (!res.ok) {
const text = await res.text();
output.textContent = `Error ${res.status}: ${text}`;
return;
}
const json = await res.json();
output.textContent = JSON.stringify(json, null, 2);
} catch (err) {
output.textContent = 'Network error: ' + (err && err.message ? err.message : String(err));
}
});
// Self-test: programmatically choose first option in each likert group, enable toggles and fill a couple fields,
// then trigger the Predict flow so testers can quickly exercise the endpoint.
function fillSelfTest() {
// pick first radio of each group
console.log('[survey] fillSelfTest starting');
const radios = Array.from(document.querySelectorAll('input[type=radio]'));
const grouped = {};
radios.forEach(r => { if (!grouped[r.name]) grouped[r.name] = []; grouped[r.name].push(r); });
Object.values(grouped).forEach(g => { if (g.length) g[0].checked = true; });
// enable all toggle checkboxes (rendered as .toggle-switch input)
document.querySelectorAll('.toggle-switch input[type=checkbox]').forEach(ch => ch.checked = true);
// fill first text inputs
const firstText = document.querySelector('input[type=text]'); if (firstText) firstText.value = 'Sample';
// set ranges to mid values
document.querySelectorAll('input[type=range]').forEach(r => {
const min = Number(r.min || 0);
const max = Number(r.max || 10);
r.value = Math.floor((min + max) / 2);
r.dispatchEvent(new Event('input'));
});
console.log('[survey] fillSelfTest done populating fields');
// trigger predict
const btn = document.getElementById('predict');
if (btn) btn.click();
}
document.getElementById('selftest').addEventListener('click', fillSelfTest);
// Keep range-value spans in sync with range inputs
function initSliders() {
const ranges = Array.from(document.querySelectorAll('input[type=range]'));
ranges.forEach(r => {
const span = document.querySelector(`.range-value[data-for="${r.id}"]`);
const update = () => {
if (span) span.textContent = r.value;
// set ARIA value
r.setAttribute('aria-valuenow', r.value);
};
// initialize
update();
r.addEventListener('input', update);
});
}
document.addEventListener('DOMContentLoaded', initSliders);
</script>
<p><small>If you need CSV export or different field typing, tell me and I'll add it.</small></p>
</div>
</body>
</html>
+54
View File
@@ -0,0 +1,54 @@
import requests
def get_ollama_summary(text, model="granite3.1-moe:1b"):
url = "http://localhost:11434/v1/chat/completions"
payload = {
"model": model,
"messages": [
{"role": "system", "content": "You are a helpful assistant that summarizes text."},
{"role": "user", "content": text}
]
}
response = requests.post(url, json=payload)
response.raise_for_status()
return response.json()["choices"][0]["message"]["content"]
def get_ollama_embedding(text, model="nomic-embed-text"):
url = "http://localhost:11434/api/embeddings"
payload = {
"model": model,
"prompt": text
}
response = requests.post(url, json=payload)
response.raise_for_status()
return response.json()["embedding"]
def parse_combined_text_to_dict(s: str) -> dict:
"""Parse a string of the form 'key1: value1 | key2: value2' into a dict.
Rules:
- Split on ' | ' to get key:value segments.
- For each segment, split on the first ':' to separate key and value.
- Strip whitespace. If a value is 'nan' (case-insensitive) or empty, use None.
- Return a dict mapping keys to values or None.
"""
if not s:
return {}
result = {}
parts = [p.strip() for p in s.split("|")]
for part in parts:
if not part:
continue
# split on the first colon
if ':' in part:
k, v = part.split(':', 1)
k = k.strip()
v = v.strip()
if v.lower() == 'nan' or v == '':
result[k] = None
else:
result[k] = v
else:
# fallback: store whole segment under a numeric key
result.setdefault('_extra', []).append(part)
return result