# ========================================================
# GETTING STARTED WITH R AND RSTUDIO
# Run it line by line: three lines give errors on purpose
# ========================================================
# Any line that starts with a hash symbol (#) is a comment.
# R ignores comments, so they are notes for human readers.
# ========================================================


# ========================================================
# SECTION 1: R AS A CALCULATOR
# ========================================================
2 + 3
(19 + 22 + 20) / 3
sqrt(16)        # the square root of 16


# ========================================================
# SECTION 2: OBJECTS
# ========================================================
# The arrow <- stores a value under a name. The stored
# value is called an object, and it is listed in the
# Environment pane.
my_age <- 20
my_age
my_age + 1

# --- 2.1 c() combines several values into one object
ages <- c(19, 22, 20, 25, 21)
ages

# --- 2.2 Names are case-sensitive: Ages is not ages
Ages

# --- 2.3 Text values go inside quotation marks
programs <- c("Health Sciences", "Kinesiology", "Health Sciences",
              "Biology", "Health Sciences")
programs
class(ages)
class(programs)

# --- 2.4 A data frame is a table: one column per variable
class_data <- data.frame(age = ages, program = programs)
class_data
class_data$age      # the $ sign picks out one column
nrow(class_data)    # the number of rows


# ========================================================
# SECTION 3: FUNCTIONS
# ========================================================
# A function does a job. Its inputs, called arguments, go
# inside the round brackets and are separated by commas.
mean(ages)
round(21.4567, digits = 1)
round(mean(ages), digits = 1)
table(class_data$program)


# ========================================================
# SECTION 4: PACKAGES
# ========================================================
describe(ages)      # an error: describe() comes from a package

# install.packages("psych")   # run once per computer
library(psych)      # run once in every new R session
describe(ages)


# ========================================================
# SECTION 5: REAL DATA
# ========================================================
# The Canadian Social Connection Survey (CSCS), stored on GitHub
github_url <- "https://raw.githubusercontent.com/jorgeandr3s/heal/main/cscs/public_data/CSCS2025_full_cleaned_deidentified_data_and_metadata.RData"
load(url(github_url))

# Keep the 2021 wave, which gives one row per person
data <- data[data$SURVEY_collection_year == 2021, ]
dim(data)           # rows, then columns
head(names(data), 12)

# The PHQ-2 depressive symptom score runs from 0 to 6
summary(data$WELLNESS_phq_score)
barplot(table(data$WELLNESS_phq_score))   # people with each score
mean(data$WELLNESS_phq_score)


# ========================================================
# SECTION 6: HELP FILES
# ========================================================
?mean
mean(data$WELLNESS_phq_score, na.rm = TRUE)


# ========================================================
# SECTION 7: AN AI ASSISTANT AS A CODING BUDDY
# ========================================================
# Goal: compare the average PHQ-2 score across gender
# groups in the 2021 wave.

# --- 7.1 Collect the facts the assistant needs
grep("phq", names(data), value = TRUE)
grep("gender", names(data), value = TRUE)
class(data$WELLNESS_phq_score)
table(data$DEMO_gender, useNA = "ifany")

# --- 7.2 Code suggested by the assistant, read line by line
phq_by_gender <- tapply(data$WELLNESS_phq_score, data$DEMO_gender,
                        mean, na.rm = TRUE)
round(phq_by_gender, 2)

# --- 7.3 An error to take back to the assistant
phq <- data$WELLNESS_phq_score
mean(phq$WELLNESS_phq_score, na.rm = TRUE)

# --- 7.4 The corrected line
mean(phq, na.rm = TRUE)
