## ----include = FALSE---------------------------------------------------------- knitr::opts_chunk$set( collapse = TRUE, comment = "#>", message = FALSE, warning = FALSE ) ## ----setup-------------------------------------------------------------------- library(mariposa) library(dplyr) data(survey_data) ## ----------------------------------------------------------------------------- survey_data %>% describe(age) ## ----------------------------------------------------------------------------- survey_data %>% describe(age, income, life_satisfaction) ## ----------------------------------------------------------------------------- survey_data %>% describe(age, income, life_satisfaction, weights = sampling_weight) ## ----------------------------------------------------------------------------- survey_data %>% group_by(region) %>% describe(income, life_satisfaction, weights = sampling_weight) ## ----------------------------------------------------------------------------- # Just the essentials survey_data %>% describe(age, income, show = c("mean", "sd", "range")) ## ----------------------------------------------------------------------------- # Everything available survey_data %>% describe(age, show = "all") ## ----------------------------------------------------------------------------- survey_data %>% frequency(education) ## ----------------------------------------------------------------------------- survey_data %>% frequency(education, weights = sampling_weight) ## ----------------------------------------------------------------------------- survey_data %>% frequency(education, employment, region, weights = sampling_weight) ## ----------------------------------------------------------------------------- survey_data %>% group_by(gender) %>% frequency(education, weights = sampling_weight) ## ----------------------------------------------------------------------------- survey_data %>% crosstab(education, employment) ## ----------------------------------------------------------------------------- survey_data %>% crosstab(education, employment, weights = sampling_weight) ## ----------------------------------------------------------------------------- survey_data %>% group_by(region) %>% crosstab(education, employment, weights = sampling_weight) ## ----------------------------------------------------------------------------- # Example set: institutions a respondent trusts highly (rating 4-5) trust_set <- survey_data %>% mutate( gov = as.integer(trust_government >= 4), media = as.integer(trust_media >= 4), science = as.integer(trust_science >= 4) ) trust_set %>% multiple_response(gov, media, science, weights = sampling_weight) %>% summary() ## ----------------------------------------------------------------------------- trust_set %>% multiple_response(gov, media, science, by = gender, weights = sampling_weight) %>% summary() ## ----eval = FALSE------------------------------------------------------------- # codebook(survey_data) ## ----------------------------------------------------------------------------- # 1. Explore the dataset find_var(survey_data, "trust|satisfaction") # 2. Summarize numeric variables survey_data %>% describe(age, income, life_satisfaction, weights = sampling_weight) # 3. Check categorical distributions survey_data %>% frequency(education, employment, weights = sampling_weight) # 4. Cross-tabulate key relationships survey_data %>% crosstab(education, employment, weights = sampling_weight) # 5. Compare across regions survey_data %>% group_by(region) %>% describe(income, life_satisfaction, weights = sampling_weight)