# Case 013: a reverse-scoring error can hide a real effect. Base R only.
# Deterministic teaching data (no random numbers), so R and Python give identical results.
i <- 1:120
unit <- function(a, m) (((i * a) %% m) - (m - 1) / 2) / ((m - 1) / 2)   # repeatable values in [-1, 1]

support <- ((i * 37) %% 11 - 5) / 2.5
trait   <- 0.8 * support + 0.6 * unit(53, 13)
item    <- function(a, m) pmin(5, pmax(1, floor(3 + trait + 0.8 * unit(a, m) + 0.5)))

# Four-item well-being scale (1-5). Items 2 and 4 are reverse-worded:
# a HIGH raw score on them means LOW well-being.
d <- data.frame(support,
                wb1 = item(17, 7),
                wb2 = 6 - item(19, 7),
                wb3 = item(23, 7),
                wb4 = 6 - item(29, 7))
items <- c("wb1", "wb2", "wb3", "wb4")

# Check: inter-item correlations. Negative values show items that point the other way.
round(cor(d[items]), 2)

# The mistake: sum the raw items without reverse-scoring items 2 and 4.
d$wb_wrong <- rowSums(d[items])
wrong <- summary(lm(wb_wrong ~ support, data = d))$coefficients["support", ]

# The fix: reverse-score items 2 and 4 (minimum + maximum - response), then sum.
d$wb2r <- 6 - d$wb2
d$wb4r <- 6 - d$wb4
d$wb_right <- d$wb1 + d$wb2r + d$wb3 + d$wb4r
round(cor(d[c("wb1", "wb2r", "wb3", "wb4r")]), 2)
right <- summary(lm(wb_right ~ support, data = d))$coefficients["support", ]

print(round(rbind(unreversed = wrong, reversed = right), 3))
stopifnot(abs(wrong[["Estimate"]] - 0.105) < 5e-4, abs(right[["Estimate"]] - 3.042) < 5e-4, nrow(d) == 120L)
sessionInfo()
