## ----include = FALSE----------------------------------------------------------
knitr::opts_chunk$set(
  collapse = TRUE,
  comment = "#>",
  message = FALSE,
  warning = FALSE
)

## ----setup--------------------------------------------------------------------
library(mariposa)
library(dplyr)
data(survey_data)

## -----------------------------------------------------------------------------
# Get labels for specific variables
var_label(survey_data, gender, education, life_satisfaction)

## -----------------------------------------------------------------------------
# Get labels for all variables
var_label(survey_data)

## -----------------------------------------------------------------------------
# Get value labels for a single variable
val_labels(survey_data, gender)

## -----------------------------------------------------------------------------
# Get value labels for multiple variables
val_labels(survey_data, education, employment)

## -----------------------------------------------------------------------------
# Search in both names and labels (default)
find_var(survey_data, "trust")

## -----------------------------------------------------------------------------
# Search only in variable labels
find_var(survey_data, "satisfaction", search = "label")

## -----------------------------------------------------------------------------
# Search only in variable names
find_var(survey_data, "age|income", search = "name")

## -----------------------------------------------------------------------------
# Set labels for specific variables
labeled_data <- var_label(survey_data,
  age = "Age of respondent in years",
  income = "Monthly net income in euros"
)

# Verify
var_label(labeled_data, age, income)

## -----------------------------------------------------------------------------
# Set value labels
labeled_data <- val_labels(survey_data,
  gender = c("Male" = 1, "Female" = 2)
)

# Verify
val_labels(labeled_data, gender)

## -----------------------------------------------------------------------------
# Add labels without replacing existing ones
labeled_data <- val_labels(survey_data,
  gender = c("Diverse" = 3),
  .add = TRUE
)

val_labels(labeled_data, gender)

## -----------------------------------------------------------------------------
# Convert gender from labelled to factor
factor_data <- to_label(survey_data, gender, education)

# Check the result
class(factor_data$gender)
levels(factor_data$gender)
head(factor_data$gender)

## -----------------------------------------------------------------------------
# Create ordered factors (useful for ordinal variables)
ordered_data <- to_label(survey_data, education, ordered = TRUE)
levels(ordered_data$education)

## -----------------------------------------------------------------------------
char_data <- to_character(survey_data, gender, region)
head(char_data$gender)

## -----------------------------------------------------------------------------
# First convert to factor, then back to numeric
factor_data <- to_label(survey_data, education)

# Parse numeric values from factor levels
numeric_data <- to_numeric(factor_data, education)
head(numeric_data$education)

## -----------------------------------------------------------------------------
# Convert a factor back to haven_labelled
plain_data <- data.frame(
  gender = factor(c("Male", "Female", "Male")),
  score = c(3, 4, 2)
)

labelled_data <- to_labelled(plain_data, gender)
class(labelled_data$gender)

## ----eval = FALSE-------------------------------------------------------------
# # After importing SPSS data where -9 = refused, -8 = don't know
# data <- read_spss("survey.sav", tag_na = FALSE)
# 
# # Declare -9 and -8 as missing across all numeric columns
# data <- set_na(data, -9, -8, tag = TRUE)
# 
# # Declare different missing codes for specific variables
# data <- set_na(data, q1 = c(-9, -8), q2 = c(99))

## ----eval = FALSE-------------------------------------------------------------
# # Strip all labels from entire dataset
# plain_data <- unlabel(survey_data)
# 
# # Strip labels from specific variables only
# plain_data <- unlabel(survey_data, gender, education)

## -----------------------------------------------------------------------------
filtered <- survey_data %>%
  filter(age >= 30) %>%
  select(gender, education, income)

# Labels may be lost after certain operations

## -----------------------------------------------------------------------------
# Restore labels from the source dataset
filtered <- copy_labels(filtered, survey_data)

# Verify labels are back
var_label(filtered, gender, education)

## -----------------------------------------------------------------------------
# Subset to one region only
subset_data <- survey_data %>% filter(region == 1)

# Remove value labels for regions that are no longer in the data
clean_data <- drop_labels(subset_data, region)
val_labels(clean_data, region)

## ----eval = FALSE-------------------------------------------------------------
# # 1. Import SPSS file
# data <- read_spss("survey_2024.sav")
# 
# # 2. Explore what's in the data
# codebook(data)
# find_var(data, "satisf")
# find_var(data, "trust")
# 
# # 3. Check labels
# var_label(data, q1, q2, q3)
# val_labels(data, q1)
# 
# # 4. Rename variables for clarity
# data <- data %>%
#   rename(life_satisfaction = q1, income = q2, education = q3)
# 
# # Update variable labels
# data <- var_label(data,
#   life_satisfaction = "Overall life satisfaction (1-5)",
#   income = "Monthly net income in euros",
#   education = "Highest education level"
# )
# 
# # 5. Convert categorical variables for analysis
# data <- to_label(data, education, gender)
# 
# # 6. Analyze
# data %>%
#   describe(life_satisfaction, income, weights = sampling_weight)
# 
# data %>%
#   t_test(life_satisfaction, group = gender, weights = sampling_weight)

