Starting an EDA on the mult_wave part of the citizen_shield project and looking into a bunch of networks linking the determinants with the least amount of missing data

This commit is contained in:
James Twose
2022-06-28 21:03:40 +02:00
4 changed files with 4439 additions and 0 deletions
+184
View File
@@ -0,0 +1,184 @@
# Setup data
## https://gisumd.github.io/COVID-19-API-Documentation/
library(tidyverse)
library(dplyr)
country_in_question <- "Netherlands"
## World Survey
# matti date range 2021-06-09 to 2022-01-18
# Ask for an indicator that's not found:
path <- paste0("https://covidmap.umd.edu/api/resources?indicator=all&type=smoothed&country=", country_in_question, "&daterange=20210609-20220118")
request <- httr::GET(url = path)
response <- httr::content(request, as = "text", encoding = "UTF-8")
# The error message contains all available indicators
returned_error <- jsonlite::fromJSON(response, flatten = TRUE) %>% data.frame()
# Pull available indicators
all_indicators <- returned_error %>%
dplyr::pull(error) %>%
stringr::str_replace_all(string = .,
pattern = "\'",
replacement = "") %>%
stringr::str_replace_all(string = .,
pattern = "need parameter:|indicator:",
replacement = "") %>%
stringr::str_replace_all(string = .,
pattern = "[]\\[]",
replacement = "") %>%
# Split based on "or", preceded or followed by any number of spaces
strsplit("[ \t]+or[ \t]+|[ \t]+or") %>%
purrr::map(.x = .,
.f = ~gsub(pattern = " ", replacement = "", x = .x)) %>%
unlist()
worldsurvey_df <- list()
for (i in 1:length(all_indicators)){
print(all_indicators[i])
# add url
path <- paste0("https://covidmap.umd.edu/api/resources?indicator=",
all_indicators[i],
"&type=daily&country=",
country_in_question,
"&daterange=20210609-20220118")
# request data from api
request <- httr::GET(url = path)
# make sure the content is encoded with 'UTF-8'
response <- httr::content(request, as = "text", encoding = "UTF-8")
# now we have a dataframe for use!
worldsurvey_df[[i]] <- jsonlite::fromJSON(response, flatten = FALSE)[[1]] %>%
# If there's no data for the indicator, return empty data frame
{if(is.list(.)) data.frame(.) else
data.frame(empty = character(),
data.survey_date = as.Date(character())) }
}
worldsurvey_df_nonempty <- purrr::discard(worldsurvey_df, ~nrow(.) == 0)
worldsurvey_df_selected <- purrr::map(.x = worldsurvey_df_nonempty,
.f = ~.x %>%
dplyr::select(1, "survey_date"))
worldsurvey_df_selected_full <- purrr::map(.x = worldsurvey_df_nonempty,
.f = ~.x %>%
dplyr::select(1, std_error = 2,
"survey_date") %>%
dplyr::mutate(name = names(.[[1]]))) %>%
purrr::reduce(.x = .,
.f = dplyr::full_join,
by = "survey_date")
worldsurvey_analysis_allvars <- purrr::reduce(.x = worldsurvey_df_selected,
.f = dplyr::full_join,
by = "survey_date") %>%
dplyr::arrange(survey_date) %>%
dplyr::mutate(date = as.Date(survey_date, format = "%Y%m%d")) %>%
dplyr::select(-survey_date)
### DATA FOR PCA
data_for_pca <- worldsurvey_analysis_allvars %>%
dplyr::filter(date > "2021-06-08") %>%
dplyr::arrange(date) %>%
dplyr::select(date,
where(~sum(is.na(.x)) <= 10),
-contains("vaccin")) %>%
tidyr::drop_na()
latest_date <- data_for_pca$date %>% tail(1)
readr::write_csv(x = data_for_pca,
file = paste0("shield-complexity/data/",
country_in_question,
"_worldsurvey_nonmissing_since_2021-06-09_to_",
latest_date, ".csv"))
#### Coefficient of variation
worldsurvey_df_selected_cov <- purrr::map(
.x = worldsurvey_df_nonempty,
.f = ~.x %>%
dplyr::select("survey_date",
1,
std_error = 2,
sample_size) %>%
dplyr::mutate(std_dev = std_error * sqrt(sample_size),
coef_of_variation = std_dev/.[[2]])
)
worldsurvey_df_selected_cov_filtered <-
purrr::map(.x = worldsurvey_df_selected_cov,
.f = ~.x %>%
dplyr::mutate(date = as.Date(survey_date,
format = "%Y%m%d")) %>%
dplyr::filter(date > "2021-06-08"))
coevar_data_from_2021_06_08_long <-
purrr::map(.x = worldsurvey_df_selected_cov_filtered,
.f = ~.x %>%
dplyr::mutate(name = names(.x[2]),
value = coef_of_variation) %>%
dplyr::select(date, name, value,
sample_size, std_error)) %>%
purrr::reduce(.x = .,
.f = full_join)
coevar_data_from_2021_06_08 <- coevar_data_from_2021_06_08_long %>%
dplyr::select(-sample_size, -std_error) %>%
tidyr::pivot_wider()
worldsurvey_sample_sizes_from_2021_06_08 <-
coevar_data_from_2021_06_08_long %>%
dplyr::select(date,
name,
sample_size)
worldsurvey_std_errors_from_2021_06_08 <-
coevar_data_from_2021_06_08_long %>%
dplyr::select(date,
name,
std_error)
coevar_data_from_2021_06_08_dropna <- coevar_data_from_2021_06_08 %>%
dplyr::select(date,
where(~sum(is.na(.x)) <= 10),
-contains("vaccin")) %>%
tidyr::drop_na()
latest_date_coevar <- coevar_data_from_2021_06_08_dropna$date %>% max()
coevar_data_from_2021_06_08_dropna %>%
readr::write_csv(file = paste0(
"shield-complexity/data/",
country_in_question,
"_worldsurvey_nonmissing_c_of_v_since_2021-06-08_to_",
latest_date_coevar, ".csv"))
worldsurvey_sample_sizes_from_2021_06_08 %>%
readr::write_csv(file = paste0(
"shield-complexity/data/",
country_in_question,
"_sample_sizes_since_2021-06-08",
".csv"))
worldsurvey_std_errors_from_2021_06_08 %>%
readr::write_csv(file = paste0(
"shield-complexity/data/",
country_in_question,
"_std_errors_since_2021-06-08",
".csv"))
File diff suppressed because it is too large Load Diff
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long