我用命令for sample in 12.ARS;do
genome=$datadir/$sample.fa
gff=$datadir/$sample.gff
#保留蛋白编码基因
agat_sp_filter_feature_by_attribute_value.pl --gff $gff --attribute biotype --value protein_coding -t '!' -o $sample.protein_coding.gff3
#保留最长转录本
agat_sp_keep_longest_isoform.pl --gff $sample.protein_coding.gff3 -o $sample.longest_isoform.gff3
#Ensembl 下载的基因组 会在基因和mRNA ID前面加gene: 和 transcript:这里给去掉
sed -i 's/gene://' $sample.longest_isoform.gff3
sed -i 's/transcript://' $sample.longest_isoform.gff3
#提取cds序列
/share/work/biosoft/TransDecoder/latest/util/gff3_file_to_proteins.pl --gff3 $sample.longest_isoform.gff3 --fasta $genome --seqType CDS >$sample.cds.fa
#提取pep序列
/share/work/biosoft/TransDecoder/latest/util/gff3_file_to_proteins.pl --gff3 $sample.longest_isoform.gff3 --fasta $genome --seqType prot >$sample.pep.fa
#提取cds 转换成OrthoFinder 需要的序列格式 基因ID作为序列ID
perl $scriptsdir/get_gene_longest_fa_for_OrthoFinder.pl -l 150 --gff $sample.longest_isoform.gff3 --fa $sample.cds.fa -p $sample -o cds
#提取pep 转换成OrthoFinder 需要的序列格式 基因ID作为序列ID
perl $scriptsdir/get_gene_longest_fa_for_OrthoFinder.pl -l 50 --gff $sample.longest_isoform.gff3 --fa $sample.pep.fa -p $sample -o pep
done
进行分析的时候,这个样本一直不出现结果,下面是我的gff 和报错
下面是我gff 文件