## ----include = FALSE----------------------------------------------------------
knitr::opts_chunk$set(
  collapse = TRUE,
  comment = "#>",
  message = FALSE,
  warning = FALSE
)

## ----setup--------------------------------------------------------------------
library(mariposa)
library(dplyr)
data(survey_data)

## -----------------------------------------------------------------------------
survey_data %>%
  describe(age)

## -----------------------------------------------------------------------------
survey_data %>%
  describe(age, income, life_satisfaction)

## -----------------------------------------------------------------------------
survey_data %>%
  describe(age, income, life_satisfaction, weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  group_by(region) %>%
  describe(income, life_satisfaction, weights = sampling_weight)

## -----------------------------------------------------------------------------
# Just the essentials
survey_data %>%
  describe(age, income, show = c("mean", "sd", "range"))

## -----------------------------------------------------------------------------
# Everything available
survey_data %>%
  describe(age, show = "all")

## -----------------------------------------------------------------------------
survey_data %>%
  frequency(education)

## -----------------------------------------------------------------------------
survey_data %>%
  frequency(education, weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  frequency(education, employment, region, weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  group_by(gender) %>%
  frequency(education, weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  crosstab(education, employment)

## -----------------------------------------------------------------------------
survey_data %>%
  crosstab(education, employment, weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  group_by(region) %>%
  crosstab(education, employment, weights = sampling_weight)

## -----------------------------------------------------------------------------
# Example set: institutions a respondent trusts highly (rating 4-5)
trust_set <- survey_data %>%
  mutate(
    gov     = as.integer(trust_government >= 4),
    media   = as.integer(trust_media >= 4),
    science = as.integer(trust_science >= 4)
  )

trust_set %>%
  multiple_response(gov, media, science, weights = sampling_weight) %>%
  summary()

## -----------------------------------------------------------------------------
trust_set %>%
  multiple_response(gov, media, science, by = gender,
                    weights = sampling_weight) %>%
  summary()

## ----eval = FALSE-------------------------------------------------------------
# codebook(survey_data)

## -----------------------------------------------------------------------------
# 1. Explore the dataset
find_var(survey_data, "trust|satisfaction")

# 2. Summarize numeric variables
survey_data %>%
  describe(age, income, life_satisfaction,
           weights = sampling_weight)

# 3. Check categorical distributions
survey_data %>%
  frequency(education, employment,
            weights = sampling_weight)

# 4. Cross-tabulate key relationships
survey_data %>%
  crosstab(education, employment,
           weights = sampling_weight)

# 5. Compare across regions
survey_data %>%
  group_by(region) %>%
  describe(income, life_satisfaction,
           weights = sampling_weight)

