Research Software Engineer, Department of Computer Science, University of Sheffield
a computer program is given as an explanation of how it works in a natural language, such as English, interspersed (embedded) with snippets of macros and traditional source code, from which compilable source code can be generated.1
My programs are not only explained better than ever before; they also are better programs, because the new methodology encourages me to do a better job.1
❱ tree -d -L 2
[4.0K Mar 19 12:36] .
├── [4.0K Nov 22 14:57] ./data
│ ├── [4.0K Jan 9 13:32] ./data/csv
│ └── [4.0K Feb 29 17:27] ./data/r
├── [4.0K Mar 14 17:36] ./docs
│ ├── [ 7 Jan 5 14:45] ./docs/data -> ../data
│ ├── [4.0K Feb 29 15:38] ./docs/modelling_cache
│ ├── [4.0K Mar 14 17:21] ./docs/modelling_files
│ ├── [ 4 Jan 5 14:44] ./docs/r -> ../r
│ └── [4.0K Mar 14 17:21] ./docs/_site
├── [4.0K Oct 19 11:40] ./inst
├── [4.0K Nov 16 10:54] ./log
├── [4.0K Oct 19 11:24] ./node_modules
├── [4.0K Oct 19 12:39] ./pages
├── [4.0K Oct 12 10:43] ./papers
├── [4.0K Nov 9 11:44] ./quarto
│ └── [4.0K Oct 19 13:12] ./quarto/_site
├── [4.0K Mar 14 16:59] ./r
├── [4.0K Mar 14 17:28] ./renv
│ ├── [4.0K Nov 16 10:57] ./renv/library
│ └── [4.0K Feb 29 14:28] ./renv/staging
├── [4.0K Jan 31 11:00] ./_site
│ ├── [4.0K Dec 19 16:36] ./_site/data_files
│ ├── [4.0K Feb 15 15:15] ./_site/docs
│ ├── [4.0K Dec 19 16:36] ./_site/modelling_files
│ └── [4.0K Jan 25 17:22] ./_site/site_libs
└── [4.0K Feb 1 17:47] ./tmp
29 directories
Key top-level directories
data/ - holds data.docs/ - my Quarto files.r/ - my R-scripts.data/❱ tree data
[4.0K Nov 22 14:57] data
├── [4.0K Jan 9 13:32] data/csv
│ ├── [ 20K Jan 9 13:31] data/csv/sheffield_thyroid_nodule.csv
│ └── [393K Nov 2 11:08] data/csv/Thy3000_DATA_LABELS_Raw.csv
└── [4.0K Feb 29 17:27] data/r
├── [ 30K Mar 14 11:12] data/r/clean.rds
├── [136K Feb 29 17:56] data/r/elastic_net.RData
├── [136K Feb 29 17:56] data/r/lasso.RData
├── [129K Feb 29 17:56] data/r/rf.RData
├── [137K Feb 29 17:58] data/r/svm.RData
└── [ 14K Feb 29 17:56] data/r/xgboost.RData
3 directories, 9 files
data/csv/*.csv - CSV filesdata/r/*.rds - R filesdata/r/*.RData - R filesdocs/❱ l docs | grep -v "~"
drwxr-xr-x neil neil 4.0 KB Thu Mar 14 17:36:04 2024 .
drwxr-xr-x neil neil 4.0 KB Thu Mar 28 16:35:45 2024 ..
.rw-r--r-- neil neil 10 B Thu Feb 15 15:16:48 2024 .gitignore
.rw-r--r-- neil neil 12 KB Thu Jan 25 15:35:49 2024 .modelling.qmd
drwxr-xr-x neil neil 4.0 KB Thu Mar 14 17:15:43 2024 .quarto
.rw-r--r-- neil neil 483 B Thu Mar 14 16:59:28 2024 _quarto.yml
drwxr-xr-x neil neil 4.0 KB Thu Mar 14 17:21:20 2024 _site
.rw-r--r-- neil neil 408 B Thu Mar 14 16:57:10 2024 about.qmd
.rw-r--r-- neil neil 1.2 KB Thu Mar 14 16:57:10 2024 citations.qmd
lrwxrwxrwx neil neil 7 B Fri Jan 5 14:45:10 2024 data ⇒ ../data
.rw-r--r-- neil neil 33 KB Thu Mar 14 16:57:10 2024 data.qmd
.rw-r--r-- neil neil 363 B Thu Mar 14 16:57:10 2024 index.qmd
.rw-r--r-- neil neil 569 B Thu Mar 14 16:57:10 2024 links.qmd
.rw-r--r-- neil neil 2.0 KB Thu Mar 14 16:57:10 2024 literature.qmd
.rw-r--r-- neil neil 32 KB Thu Mar 14 17:36:04 2024 modelling.qmd
drwxr-xr-x neil neil 4.0 KB Thu Feb 29 15:38:53 2024 modelling_cache
.rw-r--r-- neil neil 3.5 KB Thu Mar 14 17:31:46 2024 modelling_elastic_net.qmd
drwxr-xr-x neil neil 4.0 KB Thu Mar 14 17:21:18 2024 modelling_files
.rw-r--r-- neil neil 5.4 KB Thu Mar 14 17:31:46 2024 modelling_lasso.qmd
.rw-r--r-- neil neil 3.6 KB Thu Mar 14 17:31:46 2024 modelling_random_forest.qmd
.rw-r--r-- neil neil 7.0 KB Thu Mar 14 17:31:46 2024 modelling_tidymodel.qmd
lrwxrwxrwx neil neil 4 B Fri Jan 5 14:44:09 2024 r ⇒ ../r
.rw-r--r-- neil neil 18 KB Thu Mar 14 16:57:10 2024 references.bib
.rw-r--r-- neil neil 2.7 KB Thu Mar 14 16:57:10 2024 reproducibility.qmd
.rw-r--r-- neil neil 17 B Thu Mar 14 16:59:28 2024 styles.css
r/❱ l r | grep -v "~"
drwxr-xr-x neil neil 4.0 KB Thu Mar 14 16:59:28 2024 .
drwxr-xr-x neil neil 4.0 KB Thu Mar 28 16:35:45 2024 ..
.rw-r--r-- neil neil 28 KB Thu Mar 14 16:56:18 2024 clean.R
.rw-r--r-- neil neil 684 B Thu Mar 14 16:59:28 2024 master.R
.rw-r--r-- neil neil 15 KB Thu Feb 29 14:40:13 2024 modelling.R
.rw-r--r-- neil neil 2.6 KB Thu Mar 14 16:59:28 2024 simulate_data.R
.rw-r--r-- neil neil 233 B Thu Jan 25 17:02:10 2024 summary.R
.rw-r--r-- neil neil 2.9 KB Thu Mar 14 16:57:10 2024 tidymodel.R
master.R## Filename : master.R
## Description : Master file that controls running of all subsequent scripts.
library(dplyr)
library(forcats)
library(Hmisc)
library(lubridate)
library(tidymodels)
library(tidyverse)
library(vip)
## Set directories based on current location
base_dir <- getwd()
data_dir <- paste(base_dir, "data", sep = "/")
csv_dir <- paste(data_dir, "csv", sep = "/")
r_dir <- paste(data_dir, "r", sep = "/")
r_scripts <- paste(base_dir, "r", sep = "/")
## Clean the data
source(paste(r_scripts, "clean.R", sep = "/"))
## Simulate data
source(paste(r_scripts, "simulate_data.R", sep = "/"))
## Run Statistical models
source(paste(r_scripts, "tidymodel.R", sep = "/"))r/clean.R## Filename : clean.R
## Description : Load and clean raw data, saving as a .RData file for subsequent analyses.
## Read the data and convert to tibble
df_raw <- read_csv(paste(csv_dir, "Thy3000_DATA_LABELS_Raw.csv", sep = "/"))
df_raw <- as_tibble(df_raw)
## Rename columns using dplyr::rename()
df_raw <- dplyr::rename(df_raw,
record_id = "Record ID",
data_access_group = "Data Access Group",
study_id = "Study ID",
date_referral = "1.1 Date of referral",
clinic_recruiting = "1.2 Which clinic was the patient recruited from?",
clinic_recruiting_other = "If Other",
date_clinic = "1.3. The date the patient was seen in clinic",
referral_source = "1.4 Referral source",
referral_source_other = "If Other, please specify",
two_week_wait_referral = "1.4.1 If GP, was it 2-week wait referral?",
presentation = "1.5. Presentation",
presentation_complete = "Complete?...12",
age = "2.1. Age of the patient when seen in clinic",
bmi = "2.2. Body Mass Index of patient",
smoking_status = "2.3. Smoking status",
...
)r/clean.R (cont.)
## Convert dates to elapsed dates using lubridate
df <- df |>
mutate(
date_referral = lubridate::dmy(date_referral),
date_clinic = lubridate::dmy(date_clinic),
date_initial_management_decision = lubridate::dmy(date_initial_management_decision),
date_treatment = lubridate::dmy(date_treatment),
routine_review_date_last_seen = lubridate::dmy(routine_review_date_last_seen)
)
## Convert variables that are meant to be numeric but aren't
df <- df |>
dplyr::mutate(
bmi = as.numeric(bmi),
)r/clean.R (cont.)
check_cols <- c(
"previous_neck_irradiation",
"incidental_imaging",
"retrosternal",
"palpable_lymphadenopathy",
"nodule_rapid_growth",
"thyroid_function_3months",
"ultrasound",
"elastography",
"ct_neck",
"mri_neck",
"iodine_scan",
"nodule_fna",
"core_biopsy",
"routine_review_ultrasound",
"routine_review_fna",
"routine_review_patient_signposting_information"
)
df <- df %>%
dplyr::mutate(across(
all_of(check_cols),
~ dplyr::recode(.x,
"Yes" = 1,
"No" = 0,
"Not Known" = NA_real_
)
))r/clean.R (cont.)docs/*.qmd
Two repositories on GitHub
thyroid-cancer-predictionmulticenter-thyroidmulticenter-thyroid (cont.)It really makes you think about what numbers, statistics you are calculating and the tables and graphs you are including at each step and ensures they are what you want or mean them to be. It’s much better than pointing and clicking in SPSS.
renv to the rescue{renv} package helps with this.renv/ directory.renv.lock file.renv.lock{
"R": {
"Version": "4.3.2",
"Repositories": [
{
"Name": "RStudio",
"URL": "https://cran.rstudio.com"
}
]
},
...
"dplyr": {
"Package": "dplyr",
"Version": "1.1.4",
"Source": "Repository",
"Repository": "CRAN",
"Requirements": [
"R",
"R6",
"cli",
"generics",
"glue",
"lifecycle",
"magrittr",
"methods",
"pillar",
"rlang",
"tibble",
"tidyselect",
"utils",
"vctrs"
],
"Hash": "fedd9d00c2944ff00a0e2696ccf048ec"
},
...
"ggplot2": {
"Package": "ggplot2",
"Version": "3.5.0",
"Source": "Repository",
"Repository": "CRAN",
"Requirements": [
"MASS",
"R",
"cli",
"glue",
"grDevices",
"grid",
"gtable",
"isoband",
"lifecycle",
"mgcv",
"rlang",
"scales",
"stats",
"tibble",
"vctrs",
"withr"
],
"Hash": "52ef83f93f74833007f193b2d4c159a2"
},
...
}The official Quarto documentation is really good.
Reproducible Research in Practice