【问题标题】:How can I quantify the difference between these plots in R?我如何量化 R 中这些图之间的差异?
【发布时间】:2022-12-11 13:52:15
【问题描述】:

我有四个条形图。其中三个具有相似的模式,但一个行为不同。我如何在 R 中显示这种差异?如您所见,我使用了彩色编码箭头,但我喜欢量化三个图之间的相似性以及它们与第四个图的差异。 谢谢你的帮助。 这是我的数据:

dput(data)
structure(list(Gene.name = c("Gene1", "Gene1", "Gene1", "Gene1", 
"Gene1", "Gene1", "Gene1", "Gene1", "Gene1", "Gene1", "Gene1", 
"Gene1", "Gene1", "Gene1", "Gene1", "Gene1", "Gene1", "Gene1", 
"Gene1", "Gene1", "Gene1", "Gene1", "Gene1", "Gene2", "Gene2", 
"Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", 
"Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", 
"Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", 
"Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", "Gene2", 
"Gene2", "Gene2", "Gene2", "Gene2", "Gene3", "Gene3", "Gene3", 
"Gene3", "Gene3", "Gene3", "Gene3", "Gene3", "Gene3", "Gene3", 
"Gene3", "Gene3", "Gene3", "Gene3", "Gene3", "Gene3", "Gene3", 
"Gene3", "Gene3", "Gene3", "Gene3", "Gene3", "Gene3", "Gene3", 
"Gene4", "Gene4", "Gene4", "Gene4", "Gene4", "Gene4", "Gene4", 
"Gene4", "Gene4", "Gene4", "Gene4", "Gene4", "Gene4", "Gene4", 
"Gene4", "Gene4", "Gene4", "Gene4", "Gene4", "Gene4", "Gene4", 
"Gene4", "Gene4"), Cancer.Study = c("Stomach Adenocarcinoma ", 
"Stomach Adenocarcinoma ", "Uterine Corpus Endometrial Carcinoma ", 
"Uterine Corpus Endometrial Carcinoma ", "Colorectal Adenocarcinoma ", 
"Colorectal Adenocarcinoma ", "Colorectal Adenocarcinoma ", "Breast Invasive Carcinoma ", 
"Breast Invasive Carcinoma ", "Esophageal Carcinoma ", "Esophageal Carcinoma ", 
"Lung Adenocarcinoma ", "Lung Adenocarcinoma ", "Liver Hepatocellular Carcinoma ", 
"Liver Hepatocellular Carcinoma ", "Liver Hepatocellular Carcinoma ", 
"Kidney Renal Clear Cell Carcinoma ", "Bladder Urothelial Carcinoma ", 
"Bladder Urothelial Carcinoma ", "Prostate Adenocarcinoma ", 
"Prostate Adenocarcinoma ", "Lung Squamous Cell Carcinoma ", 
"Glioblastoma Multiforme ", "Esophageal Carcinoma", "Esophageal Carcinoma", 
"Esophageal Carcinoma", "Liver Hepatocellular Carcinoma", "Liver Hepatocellular Carcinoma", 
"Liver Hepatocellular Carcinoma", "Liver Hepatocellular Carcinoma", 
"Breast Invasive Carcinoma", "Breast Invasive Carcinoma", "Breast Invasive Carcinoma", 
"Breast Invasive Carcinoma", "Stomach Adenocarcinoma", "Stomach Adenocarcinoma", 
"Stomach Adenocarcinoma", "Lung Adenocarcinoma", "Lung Adenocarcinoma", 
"Lung Adenocarcinoma", "Lung Squamous Cell Carcinoma", "Lung Squamous Cell Carcinoma", 
"Lung Squamous Cell Carcinoma", "Uterine Corpus Endometrial Carcinoma", 
"Uterine Corpus Endometrial Carcinoma", "Uterine Corpus Endometrial Carcinoma", 
"Prostate Adenocarcinoma", "Prostate Adenocarcinoma", "Prostate Adenocarcinoma", 
"Bladder Urothelial Carcinoma", "Bladder Urothelial Carcinoma", 
"Bladder Urothelial Carcinoma", "Colorectal Adenocarcinoma", 
"Colorectal Adenocarcinoma", "Kidney Renal Clear Cell Carcinoma", 
"Kidney Renal Clear Cell Carcinoma", "Glioblastoma Multiforme", 
"Uterine Corpus Endometrial Carcinoma ", "Uterine Corpus Endometrial Carcinoma ", 
"Esophageal Carcinoma ", "Esophageal Carcinoma ", "Lung Adenocarcinoma ", 
"Lung Adenocarcinoma ", "Liver Hepatocellular Carcinoma ", "Liver Hepatocellular Carcinoma ", 
"Liver Hepatocellular Carcinoma ", "Breast Invasive Carcinoma ", 
"Breast Invasive Carcinoma ", "Bladder Urothelial Carcinoma ", 
"Bladder Urothelial Carcinoma ", "Colorectal Adenocarcinoma ", 
"Colorectal Adenocarcinoma ", "Colorectal Adenocarcinoma ", "Stomach Adenocarcinoma ", 
"Prostate Adenocarcinoma ", "Prostate Adenocarcinoma ", "Lung Squamous Cell Carcinoma ", 
"Lung Squamous Cell Carcinoma ", "Glioblastoma Multiforme ", 
"Glioblastoma Multiforme ", "Kidney Renal Clear Cell Carcinoma ", 
"Esophageal Carcinoma ", "Esophageal Carcinoma ", "Stomach Adenocarcinoma ", 
"Stomach Adenocarcinoma ", "Uterine Corpus Endometrial Carcinoma ", 
"Uterine Corpus Endometrial Carcinoma ", "Liver Hepatocellular Carcinoma ", 
"Liver Hepatocellular Carcinoma ", "Liver Hepatocellular Carcinoma ", 
"Bladder Urothelial Carcinoma ", "Bladder Urothelial Carcinoma ", 
"Colorectal Adenocarcinoma ", "Colorectal Adenocarcinoma ", "Colorectal Adenocarcinoma ", 
"Breast Invasive Carcinoma ", "Breast Invasive Carcinoma ", "Lung Adenocarcinoma ", 
"Lung Adenocarcinoma ", "Kidney Renal Clear Cell Carcinoma ", 
"Prostate Adenocarcinoma ", "Prostate Adenocarcinoma ", "Glioblastoma Multiforme ", 
"Lung Squamous Cell Carcinoma "), Alteration.Frequency = c(1.046025105, 
3.347280335, 2.018348624, 0.733944954, 0.161550889, 0.161550889, 
1.453957997, 1.000909918, 0.727934486, 1.081081081, 0.540540541, 
0.968992248, 0.581395349, 0.265251989, 0.265251989, 0.795755968, 
1.31826742, 0.97323601, 0.243309002, 0.400801603, 0.200400802, 
0.399201597, 0.336700337, 0.540540541, 16.75675676, 2.162162162, 
0.265251989, 0.265251989, 15.64986737, 0.530503979, 0.090991811, 
0.454959054, 14.83166515, 0.727934486, 0.627615063, 6.694560669, 
4.184100418, 0.19379845, 6.395348837, 1.356589147, 0.199600798, 
7.185628743, 0.199600798, 0.550458716, 4.403669725, 2.018348624, 
0.801603206, 5.410821643, 0.200400802, 0.243309002, 5.109489051, 
0.729927007, 4.684975767, 0.484652666, 0.753295669, 1.129943503, 
1.01010101, 3.119266055, 2.018348624, 1.621621622, 1.081081081, 
0.968992248, 0.775193798, 0.265251989, 0.265251989, 1.061007958, 
1.18289354, 0.363967243, 0.97323601, 0.486618005, 0.161550889, 
0.161550889, 0.969305331, 1.255230126, 0.400801603, 0.400801603, 
0.199600798, 0.598802395, 0.505050505, 0.168350168, 0.564971751, 
1.081081081, 2.702702703, 1.046025105, 2.30125523, 1.651376147, 
1.100917431, 0.265251989, 0.265251989, 1.591511936, 0.97323601, 
0.486618005, 0.161550889, 0.161550889, 0.969305331, 0.818926297, 
0.363967243, 0.968992248, 0.19379845, 1.129943503, 0.400801603, 
0.200400802, 0.336700337, 0.199600798), Alteration.Type = c("amp", 
"mutated", "amp", "mutated", "homdel", "amp", "mutated", "amp", 
"mutated", "amp", "mutated", "amp", "mutated", "homdel", "amp", 
"mutated", "mutated", "amp", "mutated", "homdel", "mutated", 
"mutated", "amp", "homdel", "amp", "mutated", "multiple", "homdel", 
"amp", "mutated", "multiple", "homdel", "amp", "mutated", "homdel", 
"amp", "mutated", "multiple", "amp", "mutated", "homdel", "amp", 
"mutated", "homdel", "amp", "mutated", "homdel", "amp", "mutated", 
"homdel", "amp", "mutated", "amp", "mutated", "amp", "mutated", 
"amp", "amp", "mutated", "amp", "mutated", "amp", "mutated", 
"homdel", "amp", "mutated", "amp", "mutated", "amp", "mutated", 
"homdel", "amp", "mutated", "amp", "homdel", "mutated", "amp", 
"mutated", "amp", "mutated", "mutated", "amp", "mutated", "amp", 
"mutated", "amp", "mutated", "homdel", "amp", "mutated", "amp", 
"mutated", "homdel", "amp", "mutated", "amp", "mutated", "amp", 
"mutated", "mutated", "homdel", "mutated", "amp", "mutated")), class = "data.frame", row.names = c(NA, 
-104L))

【问题讨论】:

  • how can I show this... 是什么意思?显示如何?以图形方式?数值上?还有别的吗?

标签: r


【解决方案1】:

如果您仍想使用图表,但比较容易的图表,您可以尝试分离变更类型(编辑:您的癌症研究栏中有一些尾随空格,这些空格创建了看似不同的类别)

library(ggplot2)

df$Cancer.Study=trimws(df$Cancer.Study)

ggplot(df,aes(y=Alteration.Frequency,x=Cancer.Study,color=Gene.name)) +
  geom_point() + geom_jitter() +
  facet_wrap(~Alteration.Type,ncol=1) + theme_minimal() +
  theme(axis.text.x=element_text(angle=60,hjust=1))

【讨论】:

  • 感谢你的回复。 X轴部分癌症类型重复,如何解决?
  • 这个问题可以追溯到 OP。你想如何聚合重复项?
【解决方案2】:

像这样配置上面的分面图也可以帮助突出显示对“Gene2”的影响,改变类型为“amp”:

cancer_test <- cancer %>%
group_by(Cancer.Study) 
ggplot(cancer_test, mapping = aes(x = Cancer.Study, y = Alteration.Frequency, color = Alteration.Type)) + geom_jitter() + facet_wrap(~Gene.name, ncol=1) + theme(axis.text.x=element_text(angle=60, hjust=1))

【讨论】:

    【解决方案3】:

    可以使用 Chisquare 或 Bray-Curtis 等相异指数进行数值比较。这里我们首先需要将整洁的数据格式(这当然是个好主意)转换成交叉表。然后我们可以应用标准的dist函数或vegdistfrom包素食主义者.

    以下方法使用pivot_wider来自整洁的对于交叉表,然后是vegdist对于距离矩阵。由于 tidyverse 的数据类型不是 100% 兼容 vegan,我们可能会将其转换为标准的 data.frame,然后分配名称。

    函数vegdist 计算案例之间的差异(此处:基因)。作为替代方案,也可以使用热图。这里我使用包作弊图(漂亮的热图)。

    library(dplyr)
    library(tidyr)
    library(vegan)
    library(pheatmap)
    
    ## re-arrange as crosstable
    crosstable <- 
      df %>% 
        mutate(Cancer.Study = sub(" $", "", Cancer.Study)) %>%
        pivot_wider(id_cols = c(Alteration.Type, Gene.name),
                    names_from = Cancer.Study, 
                    values_from = Alteration.Frequency)
    
    ## show the result
    crosstable
    
    ## convert to standard data frame, because tibbles don't support row names
    crosstable <- as.data.frame(crosstable)
    
    ## assign row names and remove the ID columns
    rownames(crosstable) <- with(crosstable, paste(Alteration.Type, Gene.name))
    crosstable <- crosstable[,-c(1, 2)]
    
    ## calculate pairwise dissimilarities
    vegdist(crosstable, na.rm=TRUE, method="bray")  # this is the default in vegan
    
    ## another possibility is to combine a heatmap with a cluster analysis
    ## here we use also Bray-Curtis dissimilarity for the hierarchical clusters
    pheatmap(as.matrix(crosstable), 
            distfun = function(x) vegdist(x, method="bray", na.rm=TRUE))
    

    编辑

    • 从 Cancer.Study 变量中删除了尾随空格(感谢 @user2974951 发现了这一点)
    • 添加了热图

    【讨论】:

      猜你喜欢
      • 2010-09-16
      • 1970-01-01
      • 2011-05-16
      • 2015-10-12
      • 2017-06-27
      • 1970-01-01
      • 1970-01-01
      • 1970-01-01
      • 2019-07-25
      相关资源
      最近更新 更多