### histograms and heatmaps for scoping review

library(tidyverse)
library(plotly)
library(ggplot2)
#install.packages("extrafont")
#library(extrafont)
#font_import(prompt = FALSE)
#loadfonts(device = "win")
library(dplyr)
#Set Matt's working directory:
#setwd("D:/OneDrive - The University of Sydney (Staff)/CRE/Standing on Giants/Scoping review")
#OR set Deanna's working directory:
setwd("~/Data Harmonisation Project/Scoping Review Data and R Code")
#setwd("~/Postdoc - Matilda Centre/Scoping Review - Mental Health Data Available in Representative Surveys")

####### DATA SETUP FOR PLOTS #######

## import csv file

#data <- read_csv("review-data-r.csv") #Original data file
#data <- read_csv("Scoping Review Data 31.07.2024.csv") #Updated data file 1
data <- read_csv("Scoping Review Data_SPPE Revision.csv") #Updated data file 2, with new dataset available since first submission

data$`Study type` <- as.factor(data$`Study type`)

data$year_continuous <- ifelse(substr(data$Year, 1, 1) == "?", 
                               "2017", 
                               substr(data$Year, 1, 4))


#Summing variants of the SF-12 scale into the same column: 
data$`SF-12` <- ifelse(
  data$`SF-12` == "Yes" | data$`SF-12 v2` == "Yes" | data$`MCS-12` == "Yes",
  "Yes",
  "No"
)
#Summing variants of the DASS-21 into the same column:
data$`DASS-21` <- ifelse(
  data$`DASS-21` == "Yes" | data$`DASS-21 Anxiety Subscale` == "Yes",
  "Yes",
  "No"
)
#Summing variants of the PHQ-9 into the same column:
data$`PHQ-9` <- ifelse(
  data$`PHQ-9` == "Yes" | data$`PHQ-9 (Modified for Teens)` == "Yes",
  "Yes",
  "No"
)
#Remove columns `SF-12 v2`, `MCS-12`, `DASS-21 Anxiety Subscale`, 
#`PHQ-9 (Modified for Teens)`:
data <- data[, !names(data) %in% c("SF-12 v2", 
                                   "MCS-12",
                                   "DASS-21 Anxiety Subscale",
                                   "PHQ-9 (Modified for Teens)")]

#Changing coding of instrumentation and other surveyed domains to binary numeric:
data2 <- data %>%
  mutate(across(c(K10, K6, K5, `SF-8`, `SF-12`, `SF-36`, SDQ, CHQ, CIDI, 
                  `DISC-IV`, `DASS-21`, YSR, 
                  `CES-D`,`MINI`,GADS, `PHQ-9`,`GAD-7`,PedsQL, `SCAS`,`GHQ-12`,
                  SMFQ, DQ5, HBSC, CBQ, RBPC, RCMAS, `GDS-SF`,`CIS-R`,
                  `Other Instrument`, Demographics,`Physical Health`,
                  `Other Mental Health, Wellbeing and Cognitive Function`,
                  `Alcohol and Drug Use`,`Social Wellbeing`,
                  `Financial and Socioeconomic Information`,`Social Attitudes`,
                  `Disability`,`Caring Responsibilities`,`Personality`,
                  `Physical and Biomedical Measurements`,Technology,
                  `Major Life Events`), ~ case_when(
                    . == "Yes" ~ 1,
                    . == "No" ~ 0,
                    TRUE ~ NaN
                  )))

#Amending ID numbers so all cohorts of LSAC, LSIC, LSAY and ALSWH are recorded
#as being the same study, instead of separate studies:
#(This will ensure study labels on the heatmap created later display correctly).
data2 <- data2 %>%
  mutate(`ID number` = case_when(`ID number` == 40 ~ 39, #Recode LSIC ID #s
                                 `ID number` == 42 ~ 41, #Recode LSAY ID #s
                                 `ID number` == 43 ~ 41, #Recode LSAY ID #s
                                 `ID number` == 44 ~ 41, #Recode LSAY ID #s
                                 `ID number` == 51 ~ 41, #Recode LSAY ID #s
                                 `ID number` == 30 ~ 29, #Recode ALSWH ID #s
                                 `ID number` == 31 ~ 29, #Recode ALSWH ID #s
                                 `ID number` == 32 ~ 29, #Recode ALSWH ID #s
                                 `ID number` == 46 ~ 45, #Recode LSIC ID #s
                                 TRUE ~ `ID number`))

## collapse some of the data by study ID number. 
data_summed <- data2 %>%
  group_by(`ID number`) %>%
  summarise(
    tot_k10 = sum(K10, na.rm=TRUE),
    tot_k6 = sum(K6, na.rm=TRUE),
    tot_k5 = sum(K5, na.rm=TRUE),
    tot_sf8 = sum(`SF-8`, na.rm=TRUE),
    tot_sf12 = sum(`SF-12`, na.rm=TRUE),
    tot_sf36 = sum(`SF-36`, na.rm=TRUE),
    tot_sdq = sum(SDQ, na.rm=TRUE),
    tot_chq = sum(CHQ, na.rm=TRUE),
    tot_cidi = sum(CIDI, na.rm=TRUE),
    tot_disc = sum(`DISC-IV`, na.rm=TRUE),
    tot_dass21 = sum(`DASS-21`, na.rm=TRUE),
    tot_ysr = sum(YSR, na.rm=TRUE),
    tot_cesd = sum(`CES-D`, na.rm=TRUE),
    tot_mini = sum(`MINI`, na.rm=TRUE),
    tot_gads = sum(GADS, na.rm=TRUE),
    tot_phq = sum(`PHQ-9`, na.rm=TRUE),
    tot_gad7 = sum(`GAD-7`, na.rm=TRUE),
    tot_pedsql = sum(PedsQL, na.rm=TRUE),
    tot_spence = sum(`SCAS`, na.rm=TRUE),
    tot_ghq12 = sum(`GHQ-12`, na.rm=TRUE),
    tot_smfq = sum(SMFQ, na.rm=TRUE),
    tot_dq5 = sum(DQ5, na.rm=TRUE),
    tot_hbsc = sum(HBSC, na.rm=TRUE),
    tot_cbq = sum(CBQ, na.rm=TRUE),
    tot_rbpc = sum(RBPC, na.rm=TRUE),
    tot_rcmas = sum(RCMAS, na.rm=TRUE),
    tot_gdssf = sum(`GDS-SF`, na.rm=TRUE),
    tot_cisr = sum(`CIS-R`, na.rm=TRUE),
    tot_oth = sum(`Other Instrument`, na.rm=TRUE),
    total_iterations = n(), #Total number of waves/repeats of each study
    tot_demo = sum(Demographics, na.rm=TRUE),
    tot_phys = sum(`Physical Health`, na.rm=TRUE),
    tot_omh = sum(`Other Mental Health, Wellbeing and Cognitive Function`, na.rm=TRUE),
    tot_alc = sum(`Alcohol and Drug Use`,na.rm=TRUE),
    tot_swb = sum(`Social Wellbeing`,na.rm=TRUE),
    tot_fin = sum(`Financial and Socioeconomic Information`,na.rm=TRUE),
    tot_sa = sum(`Social Attitudes`,na.rm=TRUE),
    tot_dis = sum(`Disability`,na.rm=TRUE),
    tot_care = sum(`Caring Responsibilities`,na.rm=TRUE),
    tot_pers = sum(`Personality`,na.rm=TRUE),
    tot_pbm = sum(`Physical and Biomedical Measurements`,na.rm=TRUE),
    tot_tech = sum(Technology,na.rm=TRUE),
    tot_mle = sum(`Major Life Events`, na.rm=TRUE)
  )

data_summed <- data_summed %>%
  mutate(Study = case_when(
    `ID number` == 1 ~ "YMM",
    `ID number` == 2 ~ 'ACMS',
    `ID number` == 3 ~ 'ACWP',
    `ID number` == 4 ~ 'AusDiab',
    `ID number` == 5 ~ 'AHS',
    `ID number` == 6 ~ 'NSMHW',
    `ID number` == 7 ~ 'SDAC',
    `ID number` == 8 ~ 'AuSSA',
    `ID number` == 9 ~ 'GSS',
    `ID number` == 11 ~ 'HWS',
    `ID number` == 12 ~ 'MCS',
    `ID number` == 13 ~ 'NATSIHS',
    `ID number` == 14 ~ 'NATSISS',
    `ID number` == 15 ~ 'NDSHS',
    `ID number` == 16 ~ 'NHS',
    `ID number` == 17 ~ 'NSW Adult PHS',
    `ID number` == 18 ~ 'NSW Child PHS',
    `ID number` == 19 ~ 'SSHBS',
    `ID number` == 20 ~ 'SAPHS',
    `ID number` == 21 ~ 'SAMSS',
    `ID number` == 22 ~ 'SOAR',
    `ID number` == 23 ~ 'TTPN',
    `ID number` == 25 ~ 'VPHS',
    `ID number` == 26 ~ 'WA HWSS',
    `ID number` == 27 ~ 'WAACHS',
    `ID number` == 28 ~ '45 and Up Study',
    `ID number` == 29 ~ 'ALSWH',
    `ID number` == 33 ~ 'ALSA',
    `ID number` == 34 ~ 'Ten to Men',
    `ID number` == 35 ~ 'ATP',
    `ID number` == 36 ~ 'AYS',
    `ID number` == 37 ~ 'HILDA',
    `ID number` == 38 ~ 'IYDS',
    `ID number` == 39 ~ 'LSAC',
    `ID number` == 41 ~ 'LSAY',
    `ID number` == 45 ~ 'LSIC',
    `ID number` == 47 ~ 'Mayi Kuwayu Study',
    `ID number` == 48 ~ 'PATH Through Life',
    `ID number` == 49 ~ 'COVID MHBRCS',
    `ID number` == 50 ~ 'VAHCS',
    `ID number` == 51 ~ 'LSAY',
    `ID number` == 52 ~ 'HOYVS'
  ),
  FullStudyName = case_when(
    `ID number` == 1 ~ "Australian Child and Adolescent Surveys of Mental Health and Wellbeing (Young Minds Matter)",
    `ID number` == 2 ~ "Australian Child Maltreatment Study",
    `ID number` == 3 ~ "Australian Child Wellbeing Project",
    `ID number` == 4 ~ "Australian Diabetes, Obesity and Lifestyle Study (AusDiab)",
    `ID number` == 5 ~ "Australian Health Survey",
    `ID number` == 6 ~ "National Study of Mental Health and Wellbeing",
    `ID number` == 7 ~ "Australian Survey of Disability, Ageing and Carers (SDAC)",
    `ID number` == 8 ~ "Australian Survey of Social Attitudes",
    `ID number` == 9 ~ "General Social Survey",
    `ID number` == 11 ~ "Health and Well-being Survey (WANTS Study)",
    `ID number` == 12 ~ "Middle Childhood Survey (NSW Child Development Cohort)",
    `ID number` == 13 ~ "National Aboriginal and Torres Strait Islander Health Survey",
    `ID number` == 14 ~ "National Aboriginal and Torres Strait Islander Social Survey",
    `ID number` == 15 ~ "National Drug Strategy Household survey",
    `ID number` == 16 ~ "National Health Survey",
    `ID number` == 17 ~ "NSW Adult Population Health Survey",
    `ID number` == 18 ~ "NSW Child Population Health Survey",
    `ID number` == 19 ~ "NSW School Students Health Behaviours Survey",
    `ID number` == 20 ~ "South Australian Population Health Survey",
    `ID number` == 21 ~ "South Australian Monitoring and Surveillance System (SAMSS)",
    `ID number` == 22 ~ "Speak Out Against Racism (SOAR) Survey",
    `ID number` == 23 ~ "Taking the Pulse of the Nation Surveys",
    `ID number` == 25 ~ "Victorian Population Health Survey (VPHS)",
    `ID number` == 26 ~ "Western Australia Health and Wellbeing Surveillance System (Adult Surveys)",
    `ID number` == 27 ~ "Western Australian Aboriginal Child Health Survey (WAACHS)",
    `ID number` == 28 ~ "45 and Up Study",
    `ID number` == 29 ~ "Australian Longitudinal Study of Women's Health",
    `ID number` == 33 ~ "Australian Longitudinal Study of Ageing",
    `ID number` == 34 ~ "Australian Longitudinal Study on Male Health (Ten to Men)",
    `ID number` == 35 ~ "Australian Temperament Project",
    `ID number` == 36 ~ "Australian Youth Survey",
    `ID number` == 37 ~ "The Household, Income and Labour Dynamics in Australia (HILDA) Survey",
    `ID number` == 38 ~ "International Youth Development Survey",
    `ID number` == 39 ~ "Longitudinal Study of Australian Children",
    `ID number` == 41 ~ "Longitudinal Study of Australian Youth",
    `ID number` == 45 ~ "Longitudinal Study of Indigenous Children",
    `ID number` == 47 ~ "Mayi Kuwayu National Study of Aboriginal and Torres Strait Islander Wellbeing",
    `ID number` == 48 ~ "The Personality and Total Health (PATH) Through Life Study",
    `ID number` == 49 ~ "The Australian National COVID-19 Mental Health, Behaviour and Risk Communication Survey",
    `ID number` == 50 ~ "Victorian Adolescent Health Cohort Study (VAHCS)",
    `ID number` == 51 ~ "Longitudinal Study of Australian Youth",
    `ID number` == 52 ~ "Health of Young Victorians Study"
  )
  )

data_percentages <- data_summed %>%
  mutate(
    `Percent of Waves/Repeats of Study Using the K10` = (tot_k10/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the K6` = (tot_k6/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the K5` = (tot_k5/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the SF-8`= (tot_sf8/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the SF-12` = (tot_sf12/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the SF-36` = (tot_sf36/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the SDQ` = (tot_sdq/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the CHQ` = (tot_chq/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the CIDI` = (tot_cidi/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the DISC-IV` = (tot_disc/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the DASS-21` = (tot_dass21/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the YSR` = (tot_ysr/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the CES-D` = (tot_cesd/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the MINI` = (tot_mini/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the GADS` = (tot_gads/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the PHQ-9` = (tot_phq/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the GAD-7` = (tot_gad7/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the PedsQL` = (tot_pedsql/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the SCAS` = (tot_spence/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the GHQ-12` = (tot_ghq12/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the SMFQ` = (tot_smfq/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the DQ5` = (tot_dq5/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the HBSC` = (tot_hbsc/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the CBQ` = (tot_cbq/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the RBPC` = (tot_rbpc/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the RCMAS` = (tot_rcmas/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the GDS-SF` = (tot_gdssf/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using the CIS-R` = (tot_cisr/ total_iterations)*100,
    `Percent of Waves/Repeats of Study Using Other Instrument` = (tot_oth/ total_iterations)*100
  )

data_percentages2 <- data_summed %>%
  mutate(`Percent of Waves/Repeats of Study that Surveyed Demographics` = (tot_demo/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Physical Health` = (tot_phys/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Other Mental Health Variables` = (tot_omh/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Alcohol and/or Drug Use` = (tot_alc/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Social Wellbeing` = (tot_swb/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Financial and Socioeconomic Information` = (tot_fin/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Social Attitudes` = (tot_sa/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Disability` = (tot_dis/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Caring Responsibilities` = (tot_care/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Personality` = (tot_pers/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Biometrics` = (tot_pbm/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Technology Use` = (tot_tech/ total_iterations)*100,
         `Percent of Waves/Repeats of Study that Surveyed Major or Adverse Life Experiences` = (tot_mle/ total_iterations)*100
  )

#Convert data_percentages data frame to long form and add variable with full 
#names of psych distress instruments for later use in interactive figures
data_long <- data_percentages %>%
  pivot_longer(cols = starts_with("Percent"), names_to = "Variable", values_to = "Percentage")%>%
  select(`ID number`,Variable,Percentage,Study,FullStudyName) %>%
  mutate(ID = factor(`ID number`),
         FullInstrumentName = case_when(
           `Variable` == "Percent of Waves/Repeats of Study Using the K10" ~ "10-Item Kessler Psychological Distress Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the K6" ~ "6-Item Kessler Psychological Distress Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the K5" ~ "5-Item Kessler Psychological Distress Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the SF-8" ~ "8-Item Short Form Health Survey",
           `Variable` == "Percent of Waves/Repeats of Study Using the SF-12" ~ "12-Item Short Form Health Survey",
           `Variable` == "Percent of Waves/Repeats of Study Using the SF-36" ~ "36-Item Short Form Health Survey",
           `Variable` == "Percent of Waves/Repeats of Study Using the SDQ" ~ "Strengths and Difficulties Questionnaire",
           `Variable` == "Percent of Waves/Repeats of Study Using the CHQ" ~ "Child Health Questionnaire",
           `Variable` == "Percent of Waves/Repeats of Study Using the CIDI" ~ "Composite International Diagnostic Interview",
           `Variable` == "Percent of Waves/Repeats of Study Using the DISC-IV" ~ "National Institute of Mental Health Diagnostic Interview Schedule for Children Version IV",
           `Variable` == "Percent of Waves/Repeats of Study Using the DASS-21" ~ "21-Item Depression Anxiety Stress Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the YSR" ~ "Youth Self-Report",
           `Variable` == "Percent of Waves/Repeats of Study Using the CES-D" ~ "Centre for Epidemiological Studies Depression Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the MINI" ~ "Mini International Neuropsychiatric Interview",
           `Variable` == "Percent of Waves/Repeats of Study Using the GADS" ~ "Goldberg Anxiety and Depression Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the PHQ-9" ~ "9-Item Patient Health Questionnaire",
           `Variable` == "Percent of Waves/Repeats of Study Using the GAD-7" ~ "7-Item Generalized Anxiety Disorder Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the PedsQL" ~ "Paediatric Quality of Life Inventory",
           `Variable` == "Percent of Waves/Repeats of Study Using the SCAS" ~ "Spence Children’s Anxiety Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the GHQ-12" ~ "12-Item General Health Questionnaire",
           `Variable` == "Percent of Waves/Repeats of Study Using the SMFQ" ~ "Short Mood and Feelings Questionnaire",
           `Variable` == "Percent of Waves/Repeats of Study Using the DQ5" ~ "Distress Questionnaire-5",
           `Variable` == "Percent of Waves/Repeats of Study Using the HBSC" ~ "Health Behaviours of School-Aged Children Study Devised Measure",
           `Variable` == "Percent of Waves/Repeats of Study Using the CBQ" ~ "Child Behaviour Questionnaire",
           `Variable` == "Percent of Waves/Repeats of Study Using the RBPC" ~ "Revised Behaviour Problem Checklist",
           `Variable` == "Percent of Waves/Repeats of Study Using the RCMAS" ~ "Revised Children’s Manifest Anxiety Scale",
           `Variable` == "Percent of Waves/Repeats of Study Using the GDS-SF" ~ "Geriatric Depression Scale – Short Form",
           `Variable` == "Percent of Waves/Repeats of Study Using the CIS-R" ~ "Clinical Interview Schedule – Revised",
           `Variable` == "Percent of Waves/Repeats of Study Using Other Instrument" ~ "Other"
         ))

#Convert data_percentages2 data frame to long form
data_long2 <- data_percentages2 %>%
  pivot_longer(cols = starts_with("Percent of Waves/Repeats of Study that"), names_to = "Variable", values_to = "Percentage")%>%
  select(`ID number`,Variable,Percentage,Study,FullStudyName) %>%
  mutate(ID = factor(`ID number`))


###### PLOTS ###########

data2$year_continuous <- as.numeric(data2$year_continuous)

#Downloading and loading times new roman font:
install.packages("extrafont")
library(extrafont)
font_import()
loadfonts(device="win")       #Register fonts for Windows bitmap output
fonts()

# Plot histogram to capture year coverage by number of datasets across study type
p <- ggplot(data2, aes(x = year_continuous, fill=`Study type`)) +
  geom_histogram(binwidth = 1, na.rm = TRUE, position="stack", color = "black") +
  labs(title = " ",
       x = "Year",
       y = "Count") +
  theme_classic() +
  theme(
    plot.title = element_text(face = "bold", hjust = 0.5),
    axis.title.x = element_text(face = "bold"),
    axis.title.y = element_text(face = "bold"),
    legend.title = element_text(face = "bold"),
    legend.position = "bottom",
    legend.background = element_blank(),
    legend.key = element_blank(),
    text=element_text(family = "Times New Roman", size = 15)
  ) +
  scale_fill_manual(
    name = "Study Type",
    labels = c("cross-sectional" = "Cross-sectional", "longitudinal" = "Longitudinal"),
    values = c("cross-sectional" = "#FF7777", "longitudinal" = "#33CCCC")
  ) +
  scale_x_continuous(breaks = waiver(), n.breaks = 8, limits = c(1989, 2023))

p
ggsave("high_quality_survey_year.jpeg", plot = p, width = 8, height = 7, dpi = 300)

#Counting number of surveys available in each year overall, and for 
#cross-sectional and longitudinal surveys:
year_n <- data2 %>% count(year_continuous)
year_n_cross <- data2 %>%
  filter(`Study type` == "cross-sectional") %>%
  count(year_continuous)
year_n_long <- data2 %>%
  filter(`Study type` == "longitudinal") %>%
  count(year_continuous)

## plot heatmap to show percentage of study iterations with each measure

variable_labels <-c(
  `Percent of Waves/Repeats of Study Using the K10` = "K-10",
  `Percent of Waves/Repeats of Study Using the K6` = "K-6",
  `Percent of Waves/Repeats of Study Using the K5` = "K-5",
  `Percent of Waves/Repeats of Study Using the SF-8` = "SF-8",
  `Percent of Waves/Repeats of Study Using the SF-12` = "SF-12",
  `Percent of Waves/Repeats of Study Using the SF-36` = "SF-36",
  `Percent of Waves/Repeats of Study Using the SDQ` = "SDQ",
  `Percent of Waves/Repeats of Study Using the CHQ` = "CHQ",
  `Percent of Waves/Repeats of Study Using the CIDI` = "CIDI",
  `Percent of Waves/Repeats of Study Using the DISC-IV` = "DISC-IV",
  `Percent of Waves/Repeats of Study Using the DASS-21` = "DASS-21",
  `Percent of Waves/Repeats of Study Using the YSR` = "YSR",
  `Percent of Waves/Repeats of Study Using the CES-D` = "CES-D",
  `Percent of Waves/Repeats of Study Using the MINI` = "MINI",
  `Percent of Waves/Repeats of Study Using the GADS` = "GADS",
  `Percent of Waves/Repeats of Study Using the PHQ-9` = "PHQ-9",
  `Percent of Waves/Repeats of Study Using the GAD-7` = "GAD-7",
  `Percent of Waves/Repeats of Study Using the PedsQL` = "PedsQL",
  `Percent of Waves/Repeats of Study Using the SCAS` = "SCAS",
  `Percent of Waves/Repeats of Study Using the GHQ-12` = "GHQ-12",
  `Percent of Waves/Repeats of Study Using the SMFQ` = "SMFQ",
  `Percent of Waves/Repeats of Study Using Other Instrument` = "Other",
  `Percent of Waves/Repeats of Study Using the DQ5` = "DQ5",
  `Percent of Waves/Repeats of Study Using the HBSC` = "HBSC",
  `Percent of Waves/Repeats of Study Using the CBQ` = "CBQ",
  `Percent of Waves/Repeats of Study Using the RBPC` = "RBPC",
  `Percent of Waves/Repeats of Study Using the RCMAS` = "RCMAS",
  `Percent of Waves/Repeats of Study Using the GDS-SF` = "GDS-SF",
  `Percent of Waves/Repeats of Study Using the CIS-R` = "CIS-R"
)

variable_labels2 <-c(
  `Percent of Waves/Repeats of Study that Surveyed Demographics` = "Demographics",
  `Percent of Waves/Repeats of Study that Surveyed Physical Health` = "Physical Health",
  `Percent of Waves/Repeats of Study that Surveyed Other Mental Health Variables` = "Other Mental Health",
  `Percent of Waves/Repeats of Study that Surveyed Alcohol and/or Drug Use` = "Alcohol and Drug Use",
  `Percent of Waves/Repeats of Study that Surveyed Social Wellbeing` = "Social Wellbeing",
  `Percent of Waves/Repeats of Study that Surveyed Financial and Socioeconomic Information` = "Socioeconomics",
  `Percent of Waves/Repeats of Study that Surveyed Social Attitudes` = "Social Attitudes",
  `Percent of Waves/Repeats of Study that Surveyed Disability` = "Disability",
  `Percent of Waves/Repeats of Study that Surveyed Caring Responsibilities` = "Caring Responsibilities",
  `Percent of Waves/Repeats of Study that Surveyed Personality` = "Personality",
  `Percent of Waves/Repeats of Study that Surveyed Biometrics` = "Biometrics",
  `Percent of Waves/Repeats of Study that Surveyed Technology Use` = "Technology",
  `Percent of Waves/Repeats of Study that Surveyed Major or Adverse Life Experiences` = "Major or Adverse Life Events"
)

#Convert 'variable' to a factor and reorder levels so 'Other' appears last in
#heatmap:
data_long <- data_long %>%
  mutate(Variable = factor(
    Variable, 
    levels = c(setdiff(unique(Variable), "Other"), "Other")))


#Creating heatmap of psychological distress instrument by study:
p <- ggplot(data_long, aes(x = Variable, y = Study, fill = Percentage, text = 
                             paste0("Study: ", FullStudyName, 
                                    "\nVariable: ", Variable, 
                                    "\nInstrument: ", FullInstrumentName, 
                                    "\nPercentage: ", Percentage, "%"))) +
  geom_tile(color = "white") +
  scale_fill_gradient2(low = "deepskyblue", high = "coral", mid="white", midpoint=50, na.value = "grey50") +
  scale_x_discrete(labels=variable_labels) +
  labs(title = "Heatmap of Psychological Distress Instrument Use by Study",
       x = "Psychological Distress Instrument",
       y = "Study") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(axis.text.x = element_text(angle = 90, hjust = 1),
        plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"))
p
ggsave("high_quality_heatmap_instruments.jpeg", plot = p, width = 12, height = 10, dpi = 300)

interactive_plot <- ggplotly(p, tooltip = "text")
interactive_plot

#Creating heatmap of additional surveyed domains instrument by study:
p <- ggplot(data_long2, aes(x = Variable, y = Study, fill = Percentage, text = 
                              paste0("Study: ", FullStudyName, 
                                     "\nVariable: ", Variable, 
                                     "\nPercentage: ", Percentage, "%"))) +
  geom_tile(color = "white") +
  scale_fill_gradient2(low = "deepskyblue", high = "coral", mid="white", midpoint=50, na.value = "grey50") +
  scale_x_discrete(labels=variable_labels2) +
  labs(title = "Heatmap of Additional Domains Surveyed by Study",
       x = "Other Surveyed Domains",
       y = "Study") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(axis.text.x = element_text(angle = 90, hjust = 1),
        plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"))

p
ggsave("high_quality_heatmap_domains.jpeg", plot = p, width = 12, height = 10, dpi = 300)

interactive_plot <- ggplotly(p, tooltip = "text")
interactive_plot

#### Scatterplot for minimum and maximum age. 
ggplot(data, aes(x = `Min Age Usable Values`, y = `Max Age Usable Values`, color = `Study type`)) +
  stat_sum(aes(size = after_stat(n)), geom = "point") +
  scale_size_continuous(name = "Count", range = c(3, 10)) +
  scale_color_manual(values = c("red", "blue"), name = "Study design") +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Scatter Plot of Minimum Age vs Maximum Age") +
  theme_minimal()

ggplot(data, aes(x = `Min Age Usable Values`, y = `Max Age Usable Values`, color = factor(`ID number`))) +
  stat_sum(aes(size = after_stat(n)), geom = "point") +
  scale_size_continuous(name = "Count") +
  #scale_color_manual(values = c("red", "blue"), name = "Study design") +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Scatter Plot of Minimum Age vs Maximum Age") +
  theme_minimal()

ggplot(data, aes(x = `Min Age Usable Values`, y = `Max Age Usable Values`, color = factor(Year))) +
  stat_sum(aes(size = after_stat(n)), geom = "point") +
  scale_size_continuous(name = "Count", range = c(3, 10)) +
  #scale_color_manual(values = c("red", "blue"), name = "Study design") +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Scatter Plot of Minimum Age vs Maximum Age") +
  theme_minimal()

#Changing variable labels for use in interactive plot:
data$`Minimum Age` <- data$`Min Age Usable Values`
data$`Maximum Age` <- data$`Max Age Usable Values`

## Creating data for dot plot and interactive dot plot of min age, max age by year 
agg_data <- data %>%
  group_by(`Minimum Age`, `Maximum Age`, Year) %>% 
  summarise(`Number of Datasets` = n(), .groups = 'drop')

agg_data_cross <- data %>%
  filter(`Study type` == "cross-sectional") %>%
  group_by(`Minimum Age`, `Maximum Age`, Year) %>% 
  summarise(`Number of Datasets` = n(), .groups = 'drop')

agg_data_long <- data %>%
  filter(`Study type` == "longitudinal") %>%
  group_by(`Minimum Age`, `Maximum Age`, Year) %>% 
  summarise(`Number of Datasets` = n(), .groups = 'drop')

#Create plot of sample age range for all surveys:
p<-ggplot(agg_data, aes(x = `Minimum Age`, y = `Maximum Age`, color = Year)) +
  geom_jitter(aes(size = `Number of Datasets`), width = 0.2, height = 0.2) +
  scale_size_continuous(name = "Number of Datasets", range = c(3, 10)) +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Bubble Plot of Sample Age Range by Study Year") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"),
        axis.line = element_line(linewidth = 0.3))
p
ggsave("high_quality_age_bubble_plot.jpeg", plot = p, width = 12, height = 10, dpi = 300)

#Create plot of sample age range for cross-sectional surveys:
p<-ggplot(agg_data_cross, aes(x = `Minimum Age`, y = `Maximum Age`, color = Year)) +
  geom_jitter(aes(size = `Number of Datasets`), width = 0.2, height = 0.2) +
  scale_size_continuous(name = "Number of Datasets", 
                        range = c(3,5),
                        breaks = c(1,2)) +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Bubble Plot of Cross-Sectional Sample Age Ranges by Study Year") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"),
        axis.line = element_line(linewidth = 0.3))
p
ggsave("high_quality_age_bubble_plot_cross.jpeg", plot = p, width = 12, height = 10, dpi = 300)

#Create plot of sample age range for longitudinal surveys:
p<-ggplot(agg_data_long, aes(x = `Minimum Age`, y = `Maximum Age`, color = Year)) +
  geom_jitter(aes(size = `Number of Datasets`), width = 0.2, height = 0.2) +
  scale_size_continuous(name = "Number of Datasets", 
                        range = c(3,10),
                        breaks = c(1,2,4,6,8)) +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Bubble Plot of Longitudinal Sample Age Ranges by Study Year") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"),
        axis.line = element_line(linewidth = 0.3))
p
ggsave("high_quality_age_bubble_plot_long.jpeg", plot = p, width = 12, height = 10, dpi = 300)


#Interactive version of the plots for supplementary materials:

#Creating data with 'text' column that will show study names in interactive plot:
#First, combine the agg_data with the full dataset:
combined_data <- agg_data %>%
  inner_join(data, by = c("Minimum Age", "Maximum Age", "Year"))
#Group by the key columns from agg_data and concatenate Name values
agg_data_with_text <- combined_data %>%
  group_by(`Minimum Age`, `Maximum Age`, Year) %>%
  summarise(Text = paste(Name, collapse = ", ")) %>%
  right_join(agg_data, by = c("Minimum Age", "Maximum Age", "Year"))

#Creating a function that will ensure study names are visible when text is too
#long for one line in the plot:
insert_line_breaks <- function(text, line_length = 40) {
  sapply(text, function(x) {
    if (nchar(x) > line_length) {
      paste(strwrap(x, width = line_length), collapse = "<br>")
    } else {
      x
    }
  })
}

#Apply the function to the Text column in your data
agg_data_with_text$Text <- insert_line_breaks(agg_data_with_text$Text)

#Creating the interactive plot for all surveys:
p<-ggplot(agg_data_with_text, aes(x = `Minimum Age`, y = `Maximum Age`, color = Year)) +
  geom_jitter(aes(size = `Number of Datasets`, text = Text), width = 0.2, height = 0.2) +
  scale_size_continuous(name = "", range = c(3, 10)) +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Bubble Plot of Sample Age Range by Study Year") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"),
        axis.line = element_line(linewidth = 0.3))

interactive_plot <- ggplotly(p, tooltip = "all")
interactive_plot

#Creating the data for cross-sectional interactive plot:
#First, combine agg_data_cross with the full dataset:
combined_data <- agg_data_cross %>%
  inner_join(data, by = c("Minimum Age", "Maximum Age", "Year"))
#Group by the key columns from agg_data_cross and concatenate Name values
agg_data_with_text <- combined_data %>%
  group_by(`Minimum Age`, `Maximum Age`, Year) %>%
  summarise(Text = paste(Name, collapse = ", ")) %>%
  right_join(agg_data_cross, by = c("Minimum Age", "Maximum Age", "Year"))

#Creating the interactive plot for cross-sectional surveys:
p<-ggplot(agg_data_with_text, aes(x = `Minimum Age`, y = `Maximum Age`, color = Year)) +
  geom_jitter(aes(size = `Number of Datasets`, text = Text), width = 0.2, height = 0.2) +
  scale_size_continuous(name = "", range = c(3, 5)) +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Bubble Plot of Cross-Sectional Sample Age Ranges by Study Year") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"),
        axis.line = element_line(linewidth = 0.3))

interactive_plot <- ggplotly(p, tooltip = "all")
interactive_plot

#Creating the data for longitudinal interactive plot:
#First, combine agg_data_long with the full dataset:
combined_data <- agg_data_long %>%
  inner_join(data, by = c("Minimum Age", "Maximum Age", "Year"))
#Group by the key columns from agg_data_long and concatenate Name values
agg_data_with_text <- combined_data %>%
  group_by(`Minimum Age`, `Maximum Age`, Year) %>%
  summarise(Text = paste(Name, collapse = ", ")) %>%
  right_join(agg_data_long, by = c("Minimum Age", "Maximum Age", "Year"))

#Apply the function to the Text column in your data
agg_data_with_text$Text <- insert_line_breaks(agg_data_with_text$Text)

#Creating the interactive plot for Longitudinal surveys:
p<-ggplot(agg_data_with_text, aes(x = `Minimum Age`, y = `Maximum Age`, color = Year)) +
  geom_jitter(aes(size = `Number of Datasets`, text = Text), width = 0.2, height = 0.2) +
  scale_size_continuous(name = "", range = c(3, 10)) +
  labs(x = "Minimum Age", y = "Maximum Age", title = "Bubble Plot of Longitudinal Sample Age Ranges by Study Year") +
  theme_minimal(base_size = 12, base_family = "Times New Roman") +
  theme(plot.title = element_text(face = "bold", hjust = 0.5),
        axis.title.x = element_text(face = "bold"),
        axis.title.y = element_text(face = "bold"),
        legend.title = element_text(face = "bold"),
        axis.line = element_line(linewidth = 0.3))

interactive_plot <- ggplotly(p, tooltip = "all")
interactive_plot


