## ----setup, include = FALSE---------------------------------------------------
knitr::opts_chunk$set(
  collapse = TRUE,
  comment = "#>",
  message = FALSE,
  warning = FALSE
)

## -----------------------------------------------------------------------------
library(mariposa)
library(dplyr)
data(survey_data)

## -----------------------------------------------------------------------------
linear_regression(survey_data, life_satisfaction ~ age)

## -----------------------------------------------------------------------------
result <- linear_regression(survey_data, life_satisfaction ~ age)
summary(result)

## -----------------------------------------------------------------------------
linear_regression(survey_data,
                  life_satisfaction ~ age + income + trust_government)

## -----------------------------------------------------------------------------
linear_regression(survey_data,
                  dependent = life_satisfaction,
                  predictors = c(age, income, trust_government))

## -----------------------------------------------------------------------------
linear_regression(survey_data,
                  life_satisfaction ~ age + income,
                  weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  group_by(region) %>%
  linear_regression(life_satisfaction ~ age + income)

## -----------------------------------------------------------------------------
# Standardize predictors for comparable coefficients
survey_data_z <- survey_data %>%
  std(age, income, suffix = "_z")

linear_regression(survey_data_z,
                  life_satisfaction ~ age_z + income_z + trust_government,
                  weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data <- survey_data %>%
  mutate(high_satisfaction = ifelse(life_satisfaction >= 4, 1, 0))

## -----------------------------------------------------------------------------
logistic_regression(survey_data, high_satisfaction ~ age + income)

## -----------------------------------------------------------------------------
log_result <- logistic_regression(survey_data, high_satisfaction ~ age + income)
summary(log_result)

## -----------------------------------------------------------------------------
logistic_regression(survey_data,
                    high_satisfaction ~ age + income + trust_government + education)

## -----------------------------------------------------------------------------
logistic_regression(survey_data,
                    dependent = high_satisfaction,
                    predictors = c(age, income, trust_government))

## -----------------------------------------------------------------------------
logistic_regression(survey_data,
                    high_satisfaction ~ age + income,
                    weights = sampling_weight)

## -----------------------------------------------------------------------------
survey_data %>%
  group_by(region) %>%
  logistic_regression(high_satisfaction ~ age + income)

## -----------------------------------------------------------------------------
model <- survey_data %>%
  logistic_regression(high_satisfaction ~ age + income + education)

marginal_effects(model)
summary(marginal_effects(model))

## -----------------------------------------------------------------------------
# 1. Explore relationships first
survey_data %>%
  pearson_cor(life_satisfaction, age, income, trust_government)

# 2. Run linear regression
lm_result <- linear_regression(survey_data,
                               life_satisfaction ~ age + income + trust_government,
                               weights = sampling_weight)
lm_result
summary(lm_result)

# 3. Create binary outcome
survey_data <- survey_data %>%
  mutate(high_satisfaction = ifelse(life_satisfaction >= 4, 1, 0))

# 4. Run logistic regression
log_result <- logistic_regression(survey_data,
                                  high_satisfaction ~ age + income + trust_government,
                                  weights = sampling_weight)
log_result
summary(log_result)

