You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Julia求解Rosalind开放阅读框问题的对象类型操作错误

问题:Rosalind开放阅读框(ORF)实现错误排查

问题背景

正在解决Rosalind的“Open Reading Frames”问题,思路是将每个开放阅读框(ORF)存储在orf(Char类型向量)中,最终存入Proteins(String类型向量),但运行时出现类型转换错误。

错误信息

ERROR: LoadError: MethodError: Cannot `convert` an object of type String to an object of type Char

Closest candidates are:
  convert(::Type{T}, ::Number) where T<:AbstractChar
   @ Base char.jl:184
  convert(::Type{T}, ::T) where T<:AbstractChar
   @ Base char.jl:187
  convert(::Type{T}, ::AbstractChar) where T<:AbstractChar
   @ Base char.jl:186
  ...

Stacktrace:
 [1] push!(a::Vector{Char}, item::String)
   @ Base ./array.jl:1060
 [2] find_orfs(sequence::String)
   @ Main ~/Documents/Codigos/RSLD_Open_Reading_Frames.jl:66
 [3] top-level scope
   @ ~/Documents/Codigos/RSLD_Open_Reading_Frames.jl:111

原始代码

#=
The following code is proposed to complete the Rosalind activity "Open Reading Frames".
Given any DNA sequence, the algorithm should be capable to identify protein sequences that starts in an start codon and finishes at stop codon.
It should be achieved by navigating through the original and complementary sequences and accessing the 3 reading frames in each.
If the start codon is identified, the code should insert a 'M' into the sequence and consequently add the amino acids that corresponds to each codon.
Finally, the Protein vector of Strings should contain all possibilities of proteins.
=#

#Function to get the complementary strand given the DNA sequence
function reverse_complement(x::String)
    comp = Vector{Char}()
    for i in x
        if i == 'A'
            push!(comp, 'T')
        elseif i == 'T'
            push!(comp, 'A')
        elseif i == 'C'
            push!(comp, 'G')
        elseif i == 'G'
            push!(comp, 'C')
        end
    end
    return join(reverse(comp))
end

function find_orfs(sequence::String)

    #Creation of Codon dictionary
    codon_table = Dict(
        "TTT" => "F", "TTC" => "F",
        "TTA" => "L", "TTG" => "L", "CTT" => "L", "CTC" => "L", "CTA" => "L", "CTG" => "L",
        "ATT" => "I", "ATC" => "I", "ATA" => "I",
        "ATG" => "M",
        "GTT" => "V", "GTC" => "V", "GTA" => "V", "GTG" => "V",
        "TCT" => "S", "TCC" => "S", "TCA" => "S", "TCG" => "S", "AGT" => "S", "AGC" => "S",
        "CCT" => "P", "CCC" => "P", "CCA" => "P", "CCG" => "P",
        "ACT" => "T", "ACC" => "T", "ACA" => "T", "ACG" => "T",
        "GCT" => "A", "GCC" => "A", "GCA" => "A", "GCG" => "A",
        "TAT" => "Y", "TAC" => "Y",
        "TAA" => "STOP", "TAG" => "STOP", "TGA" => "STOP",
        "CAT" => "H", "CAC" => "H",
        "CAA" => "Q", "CAG" => "Q",
        "AAT" => "N", "AAC" => "N",
        "AAA" => "K", "AAG" => "K",
        "GAT" => "D", "GAC" => "D",
        "GAA" => "E", "GAG" => "E",
        "TGT" => "C", "TGC" => "C",
        "TGG" => "W",
        "CGT" => "R", "CGC" => "R", "CGA" => "R", "CGG" => "R", "AGA" => "R", "AGG" => "R",
        "GGT" => "G", "GGC" => "G", "GGA" => "G", "GGG" => "G"
    )  
   
    Proteins = Vector{String}()


    # Consider all three forward reading frames
        frames = [1, 2, 3]
    for j in frames
            for i in j:3:length(sequence) - 2
                orf = Vector{Char}()    #Vetor
                codon = sequence[i:i+2]

                if haskey(codon_table, codon)
                    amino_acid = codon_table[codon]
                    if amino_acid == "M" 
                        push!(orf, amino_acid)
                        continue  
                    elseif amino_acid == "STOP"
                        push!(Proteins, orf)
                        break  
                    else
                        push!(orf, amino_acid)
                    end
                else
                    error("Invalid codon: $codon")
                end
        end
    end


    comp_seq = reverse_complement(sequence)

    # Consider all three reverse reading frames
    for j in frames
            for i in j:3:length(comp_seq) - 2
                orf = Vector{Char}()
                codon = comp_seq[i:i+2]

                if haskey(codon_table, codon)
                    amino_acid = codon_table[codon]
                    if amino_acid == "M"
                        push!(orf, amino_acid)
                        continue  
                    elseif amino_acid == "STOP"
                        push!(Proteins, orf)
                        break  
                    else
                        push!(orf, amino_acid)
                    end
                else
                    error("Invalid codon: $codon")
                end
        end
    end

    return Proteins
end


sequence = "ATGGCCATGGCGCCCAGAACTGAGATCAATAGTACCCGTATAACGGGTGA"
result = find_orfs(sequence)
println(result)

问题分析与解决方案

核心错误原因

  1. 类型不匹配:密码子表中氨基酸值为String类型(如"M"),但orf是Vector{Char},push!(orf, amino_acid)会触发String转Char的类型错误。
  2. ORF收集逻辑错误:每次循环新建orf,无法持续收集从起始到终止密码子的完整序列;终止时直接将Char向量存入String类型的Proteins,再次触发类型不匹配。

修复步骤

1. 修正密码子表类型

将密码子表中氨基酸值改为Char类型,终止密码子保留String用于判断:

codon_table = Dict(
    "TTT" => 'F', "TTC" => 'F',
    "TTA" => 'L', "TTG" => 'L', "CTT" => 'L', "CTC" => 'L', "CTA" => 'L', "CTG" => 'L',
    "ATT" => 'I', "ATC" => 'I', "ATA" => 'I',
    "ATG" => 'M',
    "GTT" => 'V', "GTC" => 'V', "GTA" => 'V', "GTG" => 'V',
    "TCT" => 'S', "TCC" => 'S', "TCA" => 'S', "TCG" => 'S', "AGT" => 'S', "AGC" => 'S',
    "CCT" => 'P', "CCC" => 'P', "CCA" => 'P', "CCG" => 'P',
    "ACT" => 'T', "ACC" => 'T', "ACA" => 'T', "ACG" => 'T',
    "GCT" => 'A', "GCC" => 'A', "GCA" => 'A', "GCG" => 'A',
    "TAT" => 'Y', "TAC" => 'Y',
    "TAA" => "STOP", "TAG" => "STOP", "TGA" => "STOP",
    "CAT" => 'H', "CAC" => 'H',
    "CAA" => 'Q', "CAG" => 'Q',
    "AAT" => 'N', "AAC" => 'N',
    "AAA" => 'K', "AAG" => 'K',
    "GAT" => 'D', "GAC" => 'D',
    "GAA" => 'E', "GAG" => 'E',
    "TGT" => 'C', "TGC" => 'C',
    "TGG" => 'W',
    "CGT" => 'R', "CGC" => 'R', "CGA" => 'R', "CGG" => 'R', "AGA" => 'R', "AGG" => 'R',
    "GGT" => 'G', "GGC" => 'G', "GGA" => 'G', "GGG" => 'G'
)

2. 修复ORF收集逻辑

每个阅读帧维护一个活跃ORF列表,处理起始/终止密码子的逻辑:

  • 遇到起始密码子'M'时,新建ORF加入活跃列表;
  • 遇到终止密码子时,将所有活跃ORF转为String存入Proteins,并清空列表;
  • 普通氨基酸添加到所有活跃ORF中。

修改后的正向阅读帧处理代码:

# 考虑所有三个正向阅读帧
frames = [1, 2, 3]
for j in frames
    active_orfs = Vector{Vector{Char}}()  # 存储当前帧中正在收集的ORF
    for i in j:3:length(sequence)-2
        codon = sequence[i:i+2]
        amino_acid = codon_table[codon]
        
        if amino_acid == 'M'
            push!(active_orfs, [amino_acid])
        elseif amino_acid == "STOP"
            for orf in active_orfs
                push!(Proteins, join(orf))
            end
            empty!(active_orfs)
        else
            for orf in active_orfs
                push!(orf, amino_acid)
            end
        end
    end
end

反向阅读帧逻辑与正向一致,替换sequence为comp_seq即可。

3. 类型匹配修正

用join(orf)将Char向量转为String,符合Proteins的Vector{String}类型要求。

完整修复代码

#=
The following code is proposed to complete the Rosalind activity "Open Reading Frames".
Given any DNA sequence, the algorithm should be capable to identify protein sequences that starts in an start codon and finishes at stop codon.
It should be achieved by navigating through the original and complementary sequences and accessing the 3 reading frames in each.
If the start codon is identified, the code should insert a 'M' into the sequence and consequently add the amino acids that corresponds to each codon.
Finally, the Protein vector of Strings should contain all possibilities of proteins.
=#

#Function to get the complementary strand given the DNA sequence
function reverse_complement(x::String)
    comp = Vector{Char}()
    for i in x
        if i == 'A'
            push!(comp, 'T')
        elseif i == 'T'
            push!(comp, 'A')
        elseif i == 'C'
            push!(comp, 'G')
        elseif i == 'G'
            push!(comp, 'C')
        end
    end
    return join(reverse(comp))
end

function find_orfs(sequence::String)

    #Creation of Codon dictionary (values are Char except STOP)
    codon_table = Dict(
        "TTT" => 'F', "TTC" => 'F',
        "TTA" => 'L', "TTG" => 'L', "CTT" => 'L', "CTC" => 'L', "CTA" => 'L', "CTG" => 'L',
        "ATT" => 'I', "ATC" => 'I', "ATA" => 'I',
        "ATG" => 'M',
        "GTT" => 'V', "GTC" => 'V', "GTA" => 'V', "GTG" => 'V',
        "TCT" => 'S', "TCC" => 'S', "TCA" => 'S', "TCG" => 'S', "AGT" => 'S', "AGC" => 'S',
        "CCT" => 'P', "CCC" => 'P', "CCA" => 'P', "CCG" => 'P',
        "ACT" => 'T', "ACC" => 'T', "ACA" => 'T', "ACG" => 'T',
        "GCT" => 'A', "GCC" => 'A', "GCA" => 'A', "GCG" => 'A',
        "TAT" => 'Y', "TAC" => 'Y',
        "TAA" => "STOP", "TAG" => "STOP", "TGA" => "STOP",
        "CAT" => 'H', "CAC" => 'H',
        "CAA" => 'Q', "CAG" => 'Q',
        "AAT" => 'N', "AAC" => 'N',
        "AAA" => 'K', "AAG" => 'K',
        "GAT" => 'D', "GAC" => 'D',
        "GAA" => 'E', "GAG" => 'E',
        "TGT" => 'C', "TGC" => 'C',
        "TGG" => 'W',
        "CGT" => 'R', "CGC" => 'R', "CGA" => 'R', "CGG" => 'R', "AGA" => 'R', "AGG" => 'R',
        "GGT" => 'G', "GGC" => 'G', "GGA" => 'G', "GGG" => 'G'
    )  
   
    Proteins = Vector{String}()

    # Consider all three forward reading frames
    frames = [1, 2, 3]
    for j in frames
        active_orfs = Vector{Vector{Char}}()
        for i in j:3:length(sequence)-2
            codon = sequence[i:i+2]
            amino_acid = codon_table[codon]
            
            if amino_acid == 'M'
                push!(active_orfs, [amino_acid])
            elseif amino_acid == "STOP"
                for orf in active_orfs
                    push!(Proteins, join(orf))
                end
                empty!(active_orfs)
            else
                for orf in active_orfs
                    push!(orf, amino_acid)
                end
            end
        end
    end

    comp_seq = reverse_complement(sequence)

    # Consider all three reverse reading frames
    for j in frames
        active_orfs = Vector{Vector{Char}}()
        for i in j:3:length(comp_seq)-2
            codon = comp_seq[i:i+2]
            amino_acid = codon_table[codon]
            
            if amino_acid == 'M'
                push!(active_orfs, [amino_acid])
            elseif amino_acid == "STOP"
                for orf in active_orfs
                    push!(Proteins, join(orf))
                end
                empty!(active_orfs)
            else
                for orf in active_orfs
                    push!(orf, amino_acid)
                end
            end
        end
    end

    # 去重(符合Rosalind题目要求)
    unique!(Proteins)
    return Proteins
end

sequence = "ATGGCCATGGCGCCCAGAACTGAGATCAATAGTACCCGTATAACGGGTGA"
result = find_orfs(sequence)
for protein in result
    println(protein)
end

额外优化

添加unique!(Proteins)去重,符合Rosalind输出不重复蛋白质序列的要求;最后遍历打印结果,匹配题目输出格式。


内容的提问来源于stack exchange,提问作者Hugo Oliveira

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.07 13:45:54