R语言随机抽取数据,并作两组数据间t检验,并保存抽取的数据,并绘制boxplot

前提:接着上述R脚本输出的seed结果来选择应该使用哪个seed比较合理,上个R脚本名字:

"5utr_计算ABD中Ge1和Lt1的个数和均值以及按照TE个数小的进行随机100次抽样.R"

1.输入数据:"5utr-5d做ABD中有RG4和没有RG4的TE之间的T检验.csv"

2.代码:"5utr_5d_ABD中有RG4和无RG4的TE之间的T检验函数+保存符合要求的seed+保存符合要求的数据框+绘制boxplot.R"

r 复制代码
setwd("E:\\R\\Rscripts\\5UTR_extended_TE")
# 载入必要的库
library(tidyverse)
library(dplyr)
library(openxlsx)

# 读取数据
data <- read.csv("5utr-5d做ABD中有RG4和没有RG4的TE之间的T检验.csv", na.strings = "#N/A")

# 将所有的NA值转换为0
data <- data %>% mutate_all(~ifelse(is.na(.), 0, .))

############################################################  
# 调整后的process_scores函数1,适用于le1的个数小于ge1的个数且ave-le1大于ave-ge1的情况
############################################################
  process_scores <- function(df, score_name, TE_name) {
    successful_seeds <- list() # 初始化一个列表来保存成功的seed值
    combined_samples_list <- list() # 新增:初始化一个列表来保存符合条件的组合数据框
    
    for (seed_val in 1) {
      set.seed(seed_val)
      ge1 <- df %>% filter(!!sym(score_name) >= 1) %>% select(!!sym(TE_name)) %>% mutate(Source = "ge1")
      le1 <- df %>% filter(!!sym(score_name) < 1) %>% select(!!sym(TE_name)) %>% mutate(Source = "sample_le1")
      
      sample_le1 <- sample_n(le1, nrow(ge1)) # 取单一样本进行比较
      
      t_test <- t.test(ge1[[1]], sample_le1[[1]])
      mean1 <- mean(ge1[[1]])
      mean2 <- mean(sample_le1[[1]])
      
      if (mean2 < mean1 && t_test$p.value <= 0.09) {
        successful_seeds[[paste0(seed_val, "_", score_name)]] <- list(
          seed = seed_val,
          mean1 = mean1,
          mean2 = mean2,
          pvalue = t_test$p.value
        )
        # 新增:将符合条件的ge1和sample_le1合并到一个数据框中,并保存到列表中
        combined_samples <- bind_rows(ge1, sample_le1)
        combined_samples_list[[paste0(seed_val, "_", score_name)]] <- combined_samples
      }
    }
  
  # 将成功的seeds信息转换为数据框
  if (length(successful_seeds) > 0) {
    successful_seeds_df <- bind_rows(successful_seeds, .id = "seed_score") %>% mutate(Comparison = seed_score)
  } else {
    successful_seeds_df <- tibble(Comparison = character(), mean1 = numeric(), mean2 = numeric(), pvalue = numeric())
  }
  
  # 新增:将combined_samples_list中的数据框合并或以其他形式输出
  combined_samples_output <- if (length(combined_samples_list) > 0) {
    # 例如,这里我们简单地将所有符合条件的数据框合并
    bind_rows(combined_samples_list)
  } else {
    # 如果没有符合条件的,则返回空数据框
    tibble()
  }
  
  return(list(successful_seeds = successful_seeds_df, combined_samples = combined_samples_output))
}

# 对AScore5d进行处理示例
results_AScore5d <- process_scores(data, "AScore5d", "ATe5d")
results_BScore5d <- process_scores(data, "BScore5d", "BTe5d")
results_DScore5d <- process_scores(data, "DScore5d", "DTe5d")
# 打印出符合条件的successful_seeds结果进行检查
bind_results_AScore5d_successful_seeds<-rbind(results_AScore5d$successful_seeds,results_BScore5d$successful_seeds,results_DScore5d$successful_seeds)
write.xlsx(bind_results_AScore5d_successful_seeds, file = "5utr_bind_results_ABDScore5d_successful_seeds_seed1.xlsx")

# 将符合条件的组合数据框写入文件
write.table(results_AScore5d$combined_samples, "combined_samples_seed1_5utr5dAScored.csv", quote = FALSE, row.names = FALSE, sep = ",")
write.table(results_BScore5d$combined_samples, "combined_samples_seed1_5utr5dBScored.csv", quote = FALSE, row.names = FALSE, sep = ",")
write.table(results_DScore5d$combined_samples, "combined_samples_seed1_5utr5dDScored.csv", quote = FALSE, row.names = FALSE, sep = ",")

####################################################################
##
##
#接着上面的结果绘制boxplot
##
##
####################################################################
library(tidyverse)
library(ggplot2)
library(patchwork)


results_AScore5d$combined_samples$Source<-factor(results_AScore5d$combined_samples$Source,
                                                 levels=c("ge1","sample_le1"),labels=c("A with rG4","A without rG4"),ordered=TRUE)
p1<-ggplot(results_AScore5d$combined_samples, aes(x=Source,y=ATe5d,fill=Source))+#根据Type进行填充,fill=Type
  stat_boxplot(geom = "errorbar",width=0.1)+  #添加误差线
  geom_boxplot(outlier.size = -1,width=0.25)+
  theme_classic()+#背景设置为白色
  scale_fill_manual(values = c( "#8DD3C7", "#FC8D62"))+
  labs(y="TE")+
  scale_y_continuous(limits = c(0,5),breaks=seq(0,5,1))+
  theme(
    strip.background = element_rect(colour="black", fill="#FFFFFF"),
    plot.title=element_text (hjust = 0.5,vjust =1,lineheight=1,color="black"),
    panel.background=element_rect(fill="white",colour="black",linewidth =0.5),
    axis.title.y=element_text(size=25,face="plain",color="black"),
    axis.title.x=element_blank(),
    axis.text = element_text(size=20,face="plain",color="black"),
    #axis.tex用来调整描述x轴的文本,比如图中的conserved等
    panel.border = element_blank(),
    panel.grid.major = element_blank(),
    panel.grid.minor = element_blank(),
    axis.ticks.x=element_line(colour="black"),
    axis.ticks.length.x=grid::unit(0.2, "cm")
  )+guides(fill="none")


results_BScore5d$combined_samples$Source<-factor(results_BScore5d$combined_samples$Source,
                                                 levels=c("ge1","sample_le1"),labels=c("B with rG4","B without rG4"),ordered=TRUE)
p2<-ggplot(results_BScore5d$combined_samples, aes(x=Source,y=BTe5d,fill=Source))+#根据Type进行填充,fill=Type
  stat_boxplot(geom = "errorbar",width=0.1)+  #添加误差线
  geom_boxplot(outlier.size = -1,width=0.25)+
  theme_classic()+#背景设置为白色
  scale_fill_manual(values = c( "#8DD3C7", "#FC8D62"))+
  labs(y="TE")+
  scale_y_continuous(limits = c(0,5),breaks=seq(0,5,1))+
  theme(
    strip.background = element_rect(colour="black", fill="#FFFFFF"),
    plot.title=element_text (hjust = 0.5,vjust =1,lineheight=1,color="black"),
    panel.background=element_rect(fill="white",colour="black",linewidth =0.5),
    axis.title.y=element_text(size=25,face="plain",color="black"),
    axis.title.x=element_blank(),
    axis.text = element_text(size=20,face="plain",color="black"),
    #axis.tex用来调整描述x轴的文本,比如图中的conserved等
    panel.border = element_blank(),
    panel.grid.major = element_blank(),
    panel.grid.minor = element_blank(),
    axis.ticks.x=element_line(colour="black"),
    axis.ticks.length.x=grid::unit(0.2, "cm")
  )+guides(fill="none")

results_DScore5d$combined_samples$Source<-factor(results_DScore5d$combined_samples$Source,
                                                 levels=c("ge1","sample_le1"),labels=c("D with rG4","D without rG4"),ordered=TRUE)
p3<-ggplot(results_DScore5d$combined_samples, aes(x=Source,y=DTe5d,fill=Source))+#根据Type进行填充,fill=Type
  stat_boxplot(geom = "errorbar",width=0.1)+  #添加误差线
  geom_boxplot(outlier.size = -1,width=0.25)+
  theme_classic()+#背景设置为白色
  scale_fill_manual(values = c( "#8DD3C7", "#FC8D62"))+
  labs(y="TE")+
  scale_y_continuous(limits = c(0,5),breaks=seq(0,5,1))+
  theme(
    strip.background = element_rect(colour="black", fill="#FFFFFF"),
    plot.title=element_text (hjust = 0.5,vjust =1,lineheight=1,color="black"),
    panel.background=element_rect(fill="white",colour="black",linewidth =0.5),
    axis.title.y=element_text(size=25,face="plain",color="black"),
    axis.title.x=element_blank(),
    axis.text = element_text(size=20,face="plain",color="black"),
    #axis.tex用来调整描述x轴的文本,比如图中的conserved等
    panel.border = element_blank(),
    panel.grid.major = element_blank(),
    panel.grid.minor = element_blank(),
    axis.ticks.x=element_line(colour="black"),
    axis.ticks.length.x=grid::unit(0.2, "cm")
  )+guides(fill="none")
p4<-p1+p2+p3+plot_layout(widths = c(1,1,1))
ggsave("boxplot-5utr-5d做ABD中有RG4和没有RG4的TE之间的T检验.pdf",plot=p4,width=24,height=10)

3.输出数据:"5utr_bind_results_ABDScore5d_successful_seeds_seed1.xlsx"

4.输出boxplot:"boxplot-5utr-5d做ABD中有RG4和没有RG4的TE之间的T检验.pdf"

相关推荐
czhc11400756634 天前
LINUX913 shell:set ip [lindex $argv 0],\r,send_user,spawn ssh root@ip “cat “
tcp/ip·r语言·ssh
zhangfeng11334 天前
win7 R 4.4.0和RStudio1.25的版本兼容性以及系统区域设置有关 导致Plots绘图面板被禁用,但是单独页面显示
开发语言·人工智能·r语言·生物信息
zhangfeng11334 天前
在 R 语言里,`$` 只有一个作用 按名字提取“列表型”对象里的单个元素 对象 $ 名字
开发语言·windows·r语言
高-老师4 天前
R语言生物群落(生态)数据统计分析与绘图实践技术应用
开发语言·r语言·生物群落
WangYan20225 天前
R语言:数据读取与重构、试验设计(RCB/BIB/正交/析因)、ggplot2高级绘图与统计检验(t检验/方差分析/PCA/聚类)
r语言·ggplot2·dplyr
zhangfeng11336 天前
错误于make.names(vnames, unique = TRUE): invalid multibyte string 9 使用 R 语言进行数据处理时
开发语言·r语言·生物信息
zhangfeng11336 天前
R geo 然后读取数据的时候 make.names(vnames, unique = TRUE): invalid multibyte string 9
开发语言·chrome·r语言·生物信息
梦想的初衷~7 天前
R语言生物群落数据分析全流程:从数据清洗到混合模型与结构方程
机器学习·r语言·生态·环境
没有梦想的咸鱼185-1037-16639 天前
基于R语言机器学习方法在生态经济学领域中的实践技术应用
开发语言·机器学习·数据分析·r语言
zhangfeng11339 天前
R 语法高亮为什么没有,是需要安装专用的编辑软件,R语言自带的R-gui 功能还是比较简单
开发语言·r语言