Starting an EDA on the mult_wave part of the citizen_shield project and looking into a bunch of networks linking the determinants with the least amount of missing data
This commit is contained in:
@@ -0,0 +1,184 @@
|
|||||||
|
|
||||||
|
|
||||||
|
# Setup data
|
||||||
|
|
||||||
|
## https://gisumd.github.io/COVID-19-API-Documentation/
|
||||||
|
|
||||||
|
library(tidyverse)
|
||||||
|
library(dplyr)
|
||||||
|
|
||||||
|
country_in_question <- "Netherlands"
|
||||||
|
|
||||||
|
## World Survey
|
||||||
|
|
||||||
|
# matti date range 2021-06-09 to 2022-01-18
|
||||||
|
|
||||||
|
# Ask for an indicator that's not found:
|
||||||
|
path <- paste0("https://covidmap.umd.edu/api/resources?indicator=all&type=smoothed&country=", country_in_question, "&daterange=20210609-20220118")
|
||||||
|
request <- httr::GET(url = path)
|
||||||
|
response <- httr::content(request, as = "text", encoding = "UTF-8")
|
||||||
|
|
||||||
|
# The error message contains all available indicators
|
||||||
|
returned_error <- jsonlite::fromJSON(response, flatten = TRUE) %>% data.frame()
|
||||||
|
|
||||||
|
# Pull available indicators
|
||||||
|
all_indicators <- returned_error %>%
|
||||||
|
dplyr::pull(error) %>%
|
||||||
|
stringr::str_replace_all(string = .,
|
||||||
|
pattern = "\'",
|
||||||
|
replacement = "") %>%
|
||||||
|
stringr::str_replace_all(string = .,
|
||||||
|
pattern = "need parameter:|indicator:",
|
||||||
|
replacement = "") %>%
|
||||||
|
stringr::str_replace_all(string = .,
|
||||||
|
pattern = "[]\\[]",
|
||||||
|
replacement = "") %>%
|
||||||
|
# Split based on "or", preceded or followed by any number of spaces
|
||||||
|
strsplit("[ \t]+or[ \t]+|[ \t]+or") %>%
|
||||||
|
purrr::map(.x = .,
|
||||||
|
.f = ~gsub(pattern = " ", replacement = "", x = .x)) %>%
|
||||||
|
unlist()
|
||||||
|
|
||||||
|
|
||||||
|
worldsurvey_df <- list()
|
||||||
|
|
||||||
|
for (i in 1:length(all_indicators)){
|
||||||
|
print(all_indicators[i])
|
||||||
|
# add url
|
||||||
|
path <- paste0("https://covidmap.umd.edu/api/resources?indicator=",
|
||||||
|
all_indicators[i],
|
||||||
|
"&type=daily&country=",
|
||||||
|
country_in_question,
|
||||||
|
"&daterange=20210609-20220118")
|
||||||
|
|
||||||
|
# request data from api
|
||||||
|
request <- httr::GET(url = path)
|
||||||
|
|
||||||
|
# make sure the content is encoded with 'UTF-8'
|
||||||
|
response <- httr::content(request, as = "text", encoding = "UTF-8")
|
||||||
|
|
||||||
|
# now we have a dataframe for use!
|
||||||
|
worldsurvey_df[[i]] <- jsonlite::fromJSON(response, flatten = FALSE)[[1]] %>%
|
||||||
|
# If there's no data for the indicator, return empty data frame
|
||||||
|
{if(is.list(.)) data.frame(.) else
|
||||||
|
data.frame(empty = character(),
|
||||||
|
data.survey_date = as.Date(character())) }
|
||||||
|
|
||||||
|
}
|
||||||
|
|
||||||
|
worldsurvey_df_nonempty <- purrr::discard(worldsurvey_df, ~nrow(.) == 0)
|
||||||
|
|
||||||
|
worldsurvey_df_selected <- purrr::map(.x = worldsurvey_df_nonempty,
|
||||||
|
.f = ~.x %>%
|
||||||
|
dplyr::select(1, "survey_date"))
|
||||||
|
|
||||||
|
worldsurvey_df_selected_full <- purrr::map(.x = worldsurvey_df_nonempty,
|
||||||
|
.f = ~.x %>%
|
||||||
|
dplyr::select(1, std_error = 2,
|
||||||
|
"survey_date") %>%
|
||||||
|
dplyr::mutate(name = names(.[[1]]))) %>%
|
||||||
|
|
||||||
|
purrr::reduce(.x = .,
|
||||||
|
.f = dplyr::full_join,
|
||||||
|
by = "survey_date")
|
||||||
|
|
||||||
|
worldsurvey_analysis_allvars <- purrr::reduce(.x = worldsurvey_df_selected,
|
||||||
|
.f = dplyr::full_join,
|
||||||
|
by = "survey_date") %>%
|
||||||
|
dplyr::arrange(survey_date) %>%
|
||||||
|
dplyr::mutate(date = as.Date(survey_date, format = "%Y%m%d")) %>%
|
||||||
|
dplyr::select(-survey_date)
|
||||||
|
|
||||||
|
|
||||||
|
### DATA FOR PCA
|
||||||
|
data_for_pca <- worldsurvey_analysis_allvars %>%
|
||||||
|
dplyr::filter(date > "2021-06-08") %>%
|
||||||
|
dplyr::arrange(date) %>%
|
||||||
|
dplyr::select(date,
|
||||||
|
where(~sum(is.na(.x)) <= 10),
|
||||||
|
-contains("vaccin")) %>%
|
||||||
|
tidyr::drop_na()
|
||||||
|
|
||||||
|
latest_date <- data_for_pca$date %>% tail(1)
|
||||||
|
|
||||||
|
readr::write_csv(x = data_for_pca,
|
||||||
|
file = paste0("shield-complexity/data/",
|
||||||
|
country_in_question,
|
||||||
|
"_worldsurvey_nonmissing_since_2021-06-09_to_",
|
||||||
|
latest_date, ".csv"))
|
||||||
|
|
||||||
|
#### Coefficient of variation
|
||||||
|
|
||||||
|
worldsurvey_df_selected_cov <- purrr::map(
|
||||||
|
.x = worldsurvey_df_nonempty,
|
||||||
|
.f = ~.x %>%
|
||||||
|
dplyr::select("survey_date",
|
||||||
|
1,
|
||||||
|
std_error = 2,
|
||||||
|
sample_size) %>%
|
||||||
|
dplyr::mutate(std_dev = std_error * sqrt(sample_size),
|
||||||
|
coef_of_variation = std_dev/.[[2]])
|
||||||
|
)
|
||||||
|
|
||||||
|
worldsurvey_df_selected_cov_filtered <-
|
||||||
|
purrr::map(.x = worldsurvey_df_selected_cov,
|
||||||
|
.f = ~.x %>%
|
||||||
|
dplyr::mutate(date = as.Date(survey_date,
|
||||||
|
format = "%Y%m%d")) %>%
|
||||||
|
dplyr::filter(date > "2021-06-08"))
|
||||||
|
|
||||||
|
|
||||||
|
coevar_data_from_2021_06_08_long <-
|
||||||
|
purrr::map(.x = worldsurvey_df_selected_cov_filtered,
|
||||||
|
.f = ~.x %>%
|
||||||
|
dplyr::mutate(name = names(.x[2]),
|
||||||
|
value = coef_of_variation) %>%
|
||||||
|
dplyr::select(date, name, value,
|
||||||
|
sample_size, std_error)) %>%
|
||||||
|
purrr::reduce(.x = .,
|
||||||
|
.f = full_join)
|
||||||
|
|
||||||
|
coevar_data_from_2021_06_08 <- coevar_data_from_2021_06_08_long %>%
|
||||||
|
dplyr::select(-sample_size, -std_error) %>%
|
||||||
|
tidyr::pivot_wider()
|
||||||
|
|
||||||
|
worldsurvey_sample_sizes_from_2021_06_08 <-
|
||||||
|
coevar_data_from_2021_06_08_long %>%
|
||||||
|
dplyr::select(date,
|
||||||
|
name,
|
||||||
|
sample_size)
|
||||||
|
|
||||||
|
worldsurvey_std_errors_from_2021_06_08 <-
|
||||||
|
coevar_data_from_2021_06_08_long %>%
|
||||||
|
dplyr::select(date,
|
||||||
|
name,
|
||||||
|
std_error)
|
||||||
|
|
||||||
|
coevar_data_from_2021_06_08_dropna <- coevar_data_from_2021_06_08 %>%
|
||||||
|
dplyr::select(date,
|
||||||
|
where(~sum(is.na(.x)) <= 10),
|
||||||
|
-contains("vaccin")) %>%
|
||||||
|
tidyr::drop_na()
|
||||||
|
|
||||||
|
latest_date_coevar <- coevar_data_from_2021_06_08_dropna$date %>% max()
|
||||||
|
|
||||||
|
coevar_data_from_2021_06_08_dropna %>%
|
||||||
|
readr::write_csv(file = paste0(
|
||||||
|
"shield-complexity/data/",
|
||||||
|
country_in_question,
|
||||||
|
"_worldsurvey_nonmissing_c_of_v_since_2021-06-08_to_",
|
||||||
|
latest_date_coevar, ".csv"))
|
||||||
|
|
||||||
|
worldsurvey_sample_sizes_from_2021_06_08 %>%
|
||||||
|
readr::write_csv(file = paste0(
|
||||||
|
"shield-complexity/data/",
|
||||||
|
country_in_question,
|
||||||
|
"_sample_sizes_since_2021-06-08",
|
||||||
|
".csv"))
|
||||||
|
|
||||||
|
worldsurvey_std_errors_from_2021_06_08 %>%
|
||||||
|
readr::write_csv(file = paste0(
|
||||||
|
"shield-complexity/data/",
|
||||||
|
country_in_question,
|
||||||
|
"_std_errors_since_2021-06-08",
|
||||||
|
".csv"))
|
||||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because one or more lines are too long
File diff suppressed because one or more lines are too long
Reference in New Issue
Block a user