Repository navigation
Expand file tree
/
Copy pathScriptTiO2.Rmd
More file actions
397 lines (315 loc) · 13.1 KB
/
Copy pathScriptTiO2.Rmd
File metadata and controls
397 lines (315 loc) · 13.1 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
# Script and session info
Data pre-processing:
Purpose of script: pre-process gene-expression data that will be used in: https://github.com/laurent2207/TiO2-scripts.
Gene-expression data: GEO:GSE42069
Part 1:
Purpose of script: Takes one gene list for a process, finds relevant pathways and runs enrichment analysis (using transcriptomics data) to study how process is affected in dataset
Part 2:
Purpose of script: pre-process GO-term genelists that will be used in: https://github.com/laurent2207/TiO2-scripts.
GO-term genelists: Apoptopic process (GO:0006915), Inflammatory response (GO:0006954)
GO-term genelists: Cellular response to DNA damage stimulus (GO:0006974), response to oxidative stress (GO:0006979)
Part 3:
Purpose of script: pre-process GO-term genelists that will be used in: https://github.com/laurent2207/TiO2-scripts.
Author: Laurent Winckers, Martina Kutmon
Date Created: 2020-03-24
# Session info:
R version 3.6.3 (2020-03-24)
Platform: x86_64-w64-mingw32/x64 (64-bit)
Running under: Windows 10 x64 (build 17763)
Pacakges Data pre-processing: rstudioapi_0.11, biomaRt_2.42.0, dplyr_0.8.5, EnhancedVolcano_1.4.0
Packages Part 1: rstudioapi_0.11, clusterProfiler_3.14.3, plyr_1.8.6, biomaRt_2.42.0, dplyr_0.8.5, data.table_1.12.8, pheatmap_1.0.12,
colorRamps_2.3, RColorBrewer_1.1-2, enrichplot_1.8.1, DOSE_3.14.0, org.Hs.eg.db_3.11.4, ggpubr_0.4., ggplot2_3.3.2
Packages Part 2: rstudioapi_0.11, biomaRt_2.42.0, clusterProfiler_3.14.3, tidyr_1.1.1, dplyr_0.8.5, ggplot2_3.3.2
Packages Part 3: rstudioapi_0.11, data.table_1.12.8, pheatmap_1.0.12, colorRamps_2.3, RColorBrewer_1.1-2, igraph_1.2.5, ggplot2_3.3.2
###############################
##### Data pre-processing #####
###############################
# Install required packages and set up BioMart
```{r}
library(rstudioapi)
library(biomaRt)
library(dplyr)
library(EnhancedVolcano)
```
# Set up environment
```{r}
#clear workspace and set string as factors to false
rm(list=ls())
options(stringsAsFactors = F)
#set working directroy
#setwd(dirname(rstudioapi::getActiveDocumentContext()$path)) does not work in markdown as they are not linked to RStudio
```
# Load gene expression data (analysed with arrayanalysis.org)
```{r}
# Caco2
caco2_low <- read.table("./data/data_pre_processing/TiO2_24hrs_10_Caco2.txt", sep = "\t", header = T)
caco2_high <- read.table("./data/data_pre_processing/TiO2_24hrs_100_Caco2.txt", sep = "\t", header = T)
# SAE
SAE_low <- read.table("./data/data_pre_processing/TiO2_24hrs_10_SAE.txt", sep = "\t", header = T)
SAE_high <- read.table("./data/data_pre_processing/TiO2_24hrs_100_SAE.txt", sep = "\t", header = T)
# THP1
THP1_low <- read.table("./data/data_pre_processing/TiO2_24hrs_10_THP1.txt", sep = "\t", header = T)
THP1_high <- read.table("./data/data_pre_processing/TiO2_24hrs_100_THP1.txt", sep = "\t", header = T)
# Select necessary columns, remove columns that are not needed further down the process
# Caco2
caco2_low <- caco2_low[c(1,2,3,6)]
caco2_high <- caco2_high[c(1,2,3,6)]
# SAE
SAE_low <- SAE_low[c(1,2,3,6)]
SAE_high <- SAE_high[c(1,2,3,6)]
# THP1
THP1_low <- THP1_low[c(1,2,3,6)]
THP1_high <- THP1_high[c(1,2,3,6)]
# Change column names adressing respective cell line and either 10 ug/ml, low (L) or 100 ug/ml, high (H) concentration
# Caco2
colnames(caco2_low)[c(2,3,4)] <- c("caco2_L_logFC", "caco2_L_FC", "caco2_L_pval")
colnames(caco2_high)[c(2,3,4)] <- c("caco2_H_logFC", "caco2_H_FC", "caco2_H_pval")
# SAE
colnames(SAE_low)[c(2,3,4)] <- c("SAE_L_logFC", "SAE_L_FC", "SAE_L_pval")
colnames(SAE_high)[c(2,3,4)] <- c("SAE_H_logFC", "SAE_H_FC", "SAE_H_pval")
# THP1
colnames(THP1_low)[c(2,3,4)] <- c("THP1_L_logFC", "THP1_L_FC", "THP1_L_pval")
colnames(THP1_high)[c(2,3,4)] <- c("THP1_H_logFC", "THP1_H_FC", "THP1_H_pval")
```
# Identifier mapping
```{r}
### select ensembl IDs from one of the datasets
ids <- as.data.frame(caco2_low$ENSG_ID)
colnames(ids) <- "ENSG_ID"
### map Ensemble IDs to Entrez Gene and HGNC symbols
ensembl <- useEnsembl("ensembl", dataset = "hsapiens_gene_ensembl", mirror = "useast")
genes <- getBM(attributes = c('ensembl_gene_id', 'hgnc_symbol', 'entrezgene_id'), filters = 'ensembl_gene_id', values = ids$ENSG_ID, mart = ensembl)
### remove genes without Entrez Gene identifier from gene list 11791 -> 11324
genes <- genes[!(is.na(genes$entrezgene_id)),]
genes <- genes[!duplicated(genes$entrezgene_id),]
```
# Merge datasets with annotated gene identifiers
```{r}
colnames(genes)[1] <- "ENSG_ID"
data <- merge(genes, caco2_low, by = "ENSG_ID")
data <- merge(data, caco2_high, by = "ENSG_ID")
data <- merge(data, SAE_low, by = "ENSG_ID")
data <- merge(data, SAE_high, by = "ENSG_ID")
data <- merge(data, THP1_low, by = "ENSG_ID")
data <- merge(data, THP1_high, by = "ENSG_ID")
```
# Data visualization (volcano plots)
```{r}
cl <- EnhancedVolcano(data,title = "CACO2/HT29-MTX 10 μg/ml 24h", lab = data$hgnc_symbol, x = 'caco2_L_logFC',y = 'caco2_L_pval', xlim = c(-2, 2), FCcutoff = 0.26, pCutoff = 0.05, col = c("grey30", "orange", "royalblue", "darkorange4"), selectLab = '', ylim = c(0, 13.5))
ch <- EnhancedVolcano(data,title = "CACO2/HT29-MTX 100 μg/ml 24h", lab = data$hgnc_symbol, x = 'caco2_H_logFC',y = 'caco2_H_pval', xlim = c(-2, 2), FCcutoff = 0.26, pCutoff = 0.05, col = c("grey30", "orange", "royalblue", "darkorange4"), selectLab = '', ylim = c(0, 13.5))
sl <- EnhancedVolcano(data,title = "SAE 10 μg/ml 24h", lab = data$hgnc_symbol, x = 'SAE_L_logFC',y = 'SAE_L_pval', xlim = c(-2, 2), FCcutoff = 0.26, pCutoff = 0.05, col = c("grey30", "orange", "royalblue", "darkorange4"), selectLab = '', ylim = c(0, 13.5))
sh <- EnhancedVolcano(data,title = "SAE 100 μg/ml 24h", lab = data$hgnc_symbol, x = 'SAE_H_logFC',y = 'SAE_H_pval', xlim = c(-2, 2), FCcutoff = 0.26, pCutoff = 0.05, col = c("grey30", "orange", "royalblue", "darkorange4"), selectLab = '', ylim = c(0, 13.5))
tl <- EnhancedVolcano(data,title = "THP1 10 μg/ml 24h", lab = data$hgnc_symbol, x = 'THP1_L_logFC',y = 'THP1_L_pval', xlim = c(-2, 2), FCcutoff = 0.26, pCutoff = 0.05, col = c("grey30", "orange", "royalblue", "darkorange4"), selectLab = '', ylim = c(0, 13.5))
th <- EnhancedVolcano(data,title = "THP1 100 μg/ml 24h", lab = data$hgnc_symbol, x = 'THP1_H_logFC',y = 'THP1_H_pval', xlim = c(-2, 2), FCcutoff = 0.26, pCutoff = 0.05, col = c("grey30", "orange", "royalblue", "darkorange4"), selectLab = '', ylim = c(0, 13.5))
svg(file="./output/volcanoplots.svg", width = 17, height = 18)
par(mar=c(1, 1, 1, 1))
cowplot::plot_grid(cl, ch, sl, sh, tl, th, ncol=2, nrow=3)
dev.off()
```
# Save results and ranked value for GSEA
```{r}
### save combined annotated gene-expression file
write.table(data, "./output/TiO2-dataset.txt", sep = "\t", quote = F, row.names = F)
```
##################
##### PART 1 #####
##################
# Install required packages and set up BioMart
```{r}
library(rstudioapi)
library(clusterProfiler)
library(plyr)
library(biomaRt)
library(dplyr)
library(data.table)
library(pheatmap)
library(colorRamps)
library(RColorBrewer)
library(enrichplot)
library(DOSE)
library(org.Hs.eg.db)
library(ggpubr)
library(ggplot2)
```
# Set up environment
```{r}
#clear workspace and set string as factors to false
rm(list=ls())
options(stringsAsFactors = F)
#set working directory
#setwd(dirname(rstudioapi::getActiveDocumentContext()$path)) does not work in markdown as they are not linked to RStudio
```
# Enrichment analysis
```{r}
# load gene-expression file
data <- read.table("output/TiO2-dataset.txt", header = T, sep ="\t")
### load geneset
wp2gene <- clusterProfiler::read.gmt("data/gmt_wp_Homo_sapiens.gmt")
wp2gene <- wp2gene %>% tidyr::separate(term, c("name","version","wpid","org"), "%")
wpid2gene <- wp2gene %>% dplyr::select(wpid,gene) #TERM2GENE
wpid2name <- wp2gene %>% dplyr::select(wpid,name) #TERM2NAME
comparisons <- c("caco2_L","caco2_H","SAE_L","SAE_H","THP1_L", "THP1_H")
source("functions/enrichment.R")
# set output destination
path = "output/ORApw_cb_"
enrichment(wpid2gene,wpid2name, data, comparisons, 0.26, path)
```
##################
##### PART 2 #####
##################
# Set up environment
```{r}
#clear workspace and set string as factors to false
rm(list=ls())
options(stringsAsFactors = F)
#set working directroy
#setwd(dirname(rstudioapi::getActiveDocumentContext()$path)) does not work in markdown as they are not linked to RStudio
```
# Install required packages and set up BioMart
```{r}
library(rstudioapi)
library(biomaRt)
library(clusterProfiler)
library(tidyr)
library(dplyr)
library(ggplot2)
biomart <- biomaRt::useEnsembl("ensembl", dataset = "hsapiens_gene_ensembl", host="uswest.ensembl.org")
```
# Define GO terms of interest
```{r}
go.terms <- c("GO:0006915","GO:0006954","GO:0006974","GO:0006979")
```
# Load genesets
```{r}
### load genesets
wp2gene <- clusterProfiler::read.gmt("data/gmt_wp_Homo_sapiens.gmt")
wp2gene <- wp2gene %>% tidyr::separate(term, c("name","version","wpid","org"), "%")
wpid2gene <- wp2gene %>% dplyr::select(wpid,gene) #TERM2GENE
wpid2name <- wp2gene %>% dplyr::select(wpid,name) #TERM2NAME
```
# Create GO annotations for all GO terms of interest and run enrichment analysis
```{r}
source("functions/go_annotations.R")
source("functions/ora_enrichment.R")
# set output directory for ora_enrichtment function
path = "output/"
for(i in go.terms) {
res <- go_annotations(go.term = i, biomart = biomart, output = paste("output/",gsub(":", "", i),".txt", sep=""))
ora_enrichment(genes = res$entrezgene_id, wpid2gene = wpid2gene, wpid2name = wpid2name, path = path, prefix = paste0(gsub(":", "", i), "_ORA"))
# update workflow data file
file.copy(from = paste("output/", gsub(":", "", i),".txt",sep=""), to = paste("data/",gsub(":", "", i),".txt",sep=""), overwrite = TRUE)
}
```
##################
##### PART 3 #####
##################
# Set up environment
```{r}
#clear workspace and set string as factors to false
rm(list=ls())
options(stringsAsFactors = F)
#set working directroy
#setwd(dirname(rstudioapi::getActiveDocumentContext()$path)) does not work in markdown as they are not linked to RStudio
```
# Install required packages
```{r}
library(rstudioapi)
library(data.table)
library(pheatmap)
library(colorRamps)
library(RColorBrewer)
library(igraph)
library(ggplot2)
```
# Load ORA results for GO-terms - only select rows that are significant (adjusted p-value < 0.05)
```{r}
path = paste0(getwd(), "/output")
ls_GO <- list.files(path = path, pattern = "_ORA.txt")
for (i in 1:length(ls_GO)){
assign(paste0("GO", i), read.table(paste0(path,"/",ls_GO[i]), header = T, sep = "\t", quote = ""))
}
GO1 <- GO1[GO1$p.adjust < 0.05,]
GO2 <- GO2[GO2$p.adjust < 0.05,]
GO3 <- GO3[GO3$p.adjust < 0.05,]
GO4 <- GO4[GO4$p.adjust < 0.05,]
for (i in 1:nrow(GO1)){
GO1$Perc[i] <- (as.numeric(sub("\\/.*", "", GO1$GeneRatio[i])) / as.numeric(sub("\\/.*", "", GO1$BgRatio[i])))*100
}
for (i in 1:nrow(GO2)){
GO2$Perc[i] <- (as.numeric(sub("\\/.*", "", GO2$GeneRatio[i])) / as.numeric(sub("\\/.*", "", GO2$BgRatio[i])))*100
}
for (i in 1:nrow(GO3)){
GO3$Perc[i] <- (as.numeric(sub("\\/.*", "", GO3$GeneRatio[i])) / as.numeric(sub("\\/.*", "", GO3$BgRatio[i])))*100
}
for (i in 1:nrow(GO4)){
GO4$Perc[i] <- (as.numeric(sub("\\/.*", "", GO4$GeneRatio[i])) / as.numeric(sub("\\/.*", "", GO4$BgRatio[i])))*100
}
GO150 <- GO1[GO1$Perc > 50,]
GO160 <- GO1[GO1$Perc > 60,]
GO170 <- GO1[GO1$Perc > 70,]
GO180 <- GO1[GO1$Perc > 80,]
GO250 <- GO2[GO2$Perc > 50,]
GO260 <- GO2[GO2$Perc > 60,]
GO270 <- GO2[GO2$Perc > 70,]
GO280 <- GO2[GO2$Perc > 80,]
GO350 <- GO3[GO3$Perc > 50,]
GO360 <- GO3[GO3$Perc > 60,]
GO370 <- GO3[GO3$Perc > 70,]
GO380 <- GO3[GO3$Perc > 80,]
GO450 <- GO4[GO4$Perc > 50,]
GO460 <- GO4[GO4$Perc > 60,]
GO470 <- GO4[GO4$Perc > 70,]
GO480 <- GO4[GO4$Perc > 80,]
write.table(GO150, "output/GO1.txt", quote = F, sep = "\t", row.names = F)
write.table(GO250, "output/GO2.txt", quote = F, sep = "\t", row.names = F)
write.table(GO350, "output/GO3.txt", quote = F, sep = "\t", row.names = F)
write.table(GO450, "output/GO4.txt", quote = F, sep = "\t", row.names = F)
```
# Combine the significant results of all four GO-terms
```{r}
sigORA <- rbind(GO150, GO250, GO350, GO450)
sigORA <- as.data.frame(unique(sigORA[,c(1)]))
colnames(sigORA) <- c("ID")
for (i in 1:nrow(sigORA)) {
if (sigORA$ID[i] %in% GO1$ID) {
sigORA$pid1[i] <- "1"
} else {
sigORA$pid1[i] <- "0"
}
if (sigORA$ID[i] %in% GO2$ID) {
sigORA$pid2[i] <- "1"
} else {
sigORA$pid2[i] <- "0"
}
if (sigORA$ID[i] %in% GO3$ID) {
sigORA$pid3[i] <- "1"
} else {
sigORA$pid3[i] <- "0"
}
if (sigORA$ID[i] %in% GO4$ID) {
sigORA$pid4[i] <- "1"
} else {
sigORA$pid4[i] <- "0"
}}
```
# Read pathway ORA results and only select rows where pvalue < 0.05
```{r}
path = paste0(getwd(), "/output")
ls_pw <- list.files(path = path, pattern = "ORApw_")
for (i in 1:length(ls_pw)){
assign(gsub(".txt", "", paste0(ls_pw[i])), read.table(paste0(path,"/",ls_pw[i]), header = T, sep = "\t", quote = ""))
}
ls_pw <- mget(ls(pattern = 'ORApw_'))
ls_pw <- lapply(ls_pw, function(x){x[x$pvalue<0.05,]})
#list2env(ls_pw, envir = .GlobalEnv)
```
# Filter ORA results of pathways - create plots
```{r}
source("functions/filteredPlot.R")
# set output directory
path = "output/"
dir.create(paste0(path, "nodesEdges"))
for (i in 1:length(ls_pw)){
filteredPlot(data = ls_pw[[i]], path, fileName = gsub("ORApw_", "", names(ls_pw[i])))
}
```