You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

纯R实现MD5哈希函数输出与tools::md5sum不匹配问题排查

纯R语言实现MD5哈希函数的输出不匹配问题

我尝试用纯R语言编写不调用任何C例程的MD5哈希函数,代码能正常执行,但输出始终和tools::md5sum(其结果与各类在线示例一致)不匹配。我怀疑是字节序(或字序)的问题,代码里已经尝试加入flip操作但未成功,也检查过是否是字节顺序颠倒的简单情况,没有收获。

运行此函数需要Rmpfr和bigBits库,以及本文提供的flip32、flip16和bigRotate辅助函数:

# 需要加载的库
library(Rmpfr)
library(bigBits)

mymd5 <- function(msg){

  if(!is(msg,'raw')) msg <- charToRaw(as.character(msg))
  # 用于截断超过32位的数值
  two32 <- as.bigz(2^(as.bigz(32))) 

  sidx <- c( 7, 12, 17, 22,  7, 12, 17, 22,  7, 12, 17, 22,  7, 12, 17, 22 , 5,  9, 14, 20,  5,  9, 14, 20,  5,  9, 14, 20,  5,  9, 14, 20 , 4, 11, 16, 23,  4, 11, 16, 23,  4, 11, 16, 23,  4, 11, 16, 23 , 6, 10, 15, 21,  6, 10, 15, 21,  6, 10, 15, 21,  6, 10, 15, 21 )

  # K值已验证正确
  K <- as.bigz(2^32 * abs(sin(1:64))) 

  # IETF文档显示初始值无需调整字节序,A的表示为01234567(低字节在前)
  adef <- '67452301'
  bdef <- 'efcdab89'
  cdef <- '98badcfe'
  ddef <- '10325476'

  # 尝试过的flip操作(当前启用)
  adef <- flip32(adef)
  bdef <- flip32(bdef)
  cdef <- flip32(cdef)
  ddef <- flip32(ddef)

  a0 <- (base2base(adef,16,10))[[1]]
  b0 <-(base2base(bdef,16,10))[[1]]
  c0 <-(base2base(cdef,16,10))[[1]]
  d0 <-(base2base(ddef,16,10))[[1]]

  # 追加0x80,再填充0x00使消息字节数 ≡56 mod64
  origlenbit <- length(msg) * 8
  msg <- c(msg, as.raw('0x80') )
  bytemod <- length(msg)%%64
  raw00 <- as.raw('0x00')
  if (bytemod <  56 && bytemod > 0 ) {
      msg <- c(msg,rep(raw00,times=(56-bytemod)))
  } else if (bytemod > 56) {
      msg <- c(msg, rep(raw00, times = bytemod-56))
  }
   
  # 追加原始消息的比特长度(64位,低32位在前)
  len16 <-    base2base(origlenbit,10,16)[[1]]
  lenbit2 <- base2base(origlenbit,10,2)[[1]]
  if(nchar(lenbit2) < 64 ){
      lenbit2 <- paste0(c(rep('0',times = 64 -nchar(lenbit2)) ,lenbit2), collapse='')
  } else if (nchar(lenbit2) > 64) lenbit2 <- rev(rev(lenbit2)[1:64])
  newbytes2 <- substring(lenbit2, seq(1,nchar(lenbit2)-7,by=8), last = seq(8, nchar(lenbit2),by = 8) )    
  newdec <- base2base(newbytes2,2,10)
  newdbl <- rep(0,times= 8)
  for (jd in 1: 8) newdbl[jd] <- as.numeric(newdec[[jd]])
  newraw <- as.raw(newdbl)
  # 尝试过的不同长度字节顺序
  # msg <- c(msg,newraw[c(7,8,5,6,3,4,1,2)])
  msg <- c(msg, newraw[5:8], newraw[1:4])
  # msg <- c(msg, newraw[1:4], newraw[5:8])
  chunkit <- as.bigz(as.numeric(msg ) ) 

  # 将512位消息块拆分为16个32位字
  for(jchunk in 1: (length(msg)/64)) {  
      thischunk <- matrix(chunkit[ ((jchunk-1)*64)+(1:64)],ncol=16)
      thisM <- as.bigz(rep(0,16) )
      # 尝试过的两种字节顺序计算方式
      for(jm in 1:16) thisM[jm] <- sum(thischunk[,jm] * c(256^3,256^2,256,1) )
      # for(jm in 1:16) thisM[jm] <- sum(thischunk[,jm] * c(1,256,256^2,256^3) )
      # 初始化当前块的哈希值
      Ah<- a0
      Bh<- b0
      Ch<- c0
      Dh<- d0
      # 四轮循环计算
      for (jone in 1:16) {
          Fh <- bigOr (bigAnd(Bh,Ch, inTwo=FALSE)[[1]] ,bigAnd(bigNot(Bh, inTwo=FALSE, outTwo=FALSE)[[1]],Dh, inTwo=FALSE)[[1]] )[[1]]
          g <- jone
          Fh<- (Fh + Ah + K[jone] + thisM[g]) %%two32  
          Ah<- Dh
          Dh<- Ch
          Ch<- Bh
          Bh<- (Bh + bigRotate(Fh, sidx[jone], inTwo=FALSE)[[1]] )%%two32   
      }

      for (j2 in 17:32) {
          Fh <- bigOr( bigAnd(Dh,Bh,inTwo=FALSE)[[1]] ,bigAnd(bigNot(Dh[[1]],inTwo=FALSE, outTwo=FALSE), Ch, inTwo=FALSE)[[1]] ,inTwo=FALSE)[[1]] 
          g <-(1 + 5*(j2-1))%%16 +1
          Fh<- (Fh + Ah + K[j2] + thisM[g])%%two32  
          Ah<- Dh
          Dh<- Ch
          Ch<- Bh
          Bh<- (Bh + bigRotate(Fh, sidx[j2], inTwo=FALSE)[[1]])%%two32   
      }
      for(j3 in 33:48){
          Fh <- bigXor(Bh,bigXor(Ch,Dh, inTwo=FALSE)[[1]], inTwo=FALSE)[[1]]
          g <-(5 + 3*(j3-1))%%16 +1
          Fh<- (Fh + Ah + K[j3] + thisM[g])%%two32  
          Ah<- Dh
          Dh<- Ch
          Ch<- Bh
          Bh<- (Bh + bigRotate(Fh, sidx[j3],inTwo=FALSE)[[1]])%%two32 
      }
      for(j4 in 49:64){
          Fh <- bigXor(Ch,bigOr(Bh,bigNot(Dh,inTwo=FALSE, outTwo=FALSE)[[1]],inTwo=FALSE)[[1]],inTwo=FALSE)[[1]]
          g <- (7 * (j4 -1))%%16 +1
          Fh<- (Fh + Ah + K[j4] + thisM[g])%%two32  
          Ah<- Dh
          Dh<- Ch
          Ch<- Bh
          Bh<- (Bh + bigRotate(Fh, sidx[j4], inTwo=FALSE)[[1]])%%two32 
      }
      # 累加当前块的哈希结果
      a0<- (a0 + Ah)%%two32
      b0<- (b0 + Bh)%%two32
      c0<- (c0 + Ch)%%two32
      d0<- (d0 + Dh)%%two32
  }  

  # 拼接最终结果,补全前导零
  a0x <-(base2base(a0,10,16))
  a0x <- paste0(c(rep(0,times=max(8-(nchar(a0x)),0)), a0x), collapse='')
  b0x <-(base2base(b0,10,16))
  b0x <- paste0(c(rep(0,times=max(8-(nchar(b0x)),0)), b0x), collapse='')
  c0x <-(base2base(c0,10,16))
  c0x <- paste0(c(rep(0,times=max(8-(nchar(c0x)),0)), c0x), collapse='')
  d0x <-(base2base(d0,10,16))
  d0x <- paste0(c(rep(0,times=max(8-(nchar(d0x)),0)), d0x), collapse='')

  thesum <- paste0(c(a0x,b0x,c0x,d0x) , sep='',collapse='')
  return(thesum)
} 

# 辅助函数:翻转32位十六进制字符串的字节顺序
flip32 <- function(x){
    xsep <- unlist(strsplit(x,'') )
    xout <- paste0(c(xsep[c(7,8,5,6,3,4,1,2)]),collapse='')
    return(xout)
}
flip16 <- function(x) {
        xsep <- unlist(strsplit(x,''))
    xout <- paste0(c(xsep[c(3,4,1,2)]),collapse='')
    return(xout)
}

# 辅助函数:32位循环左移
bigRotate <- function(x, shift,  inBase = 10,binSize = 32, outBase = 10, inTwosComp = TRUE) {
 shift = floor(shift[1])

 out <- rep('0',times = length(x))  
    for(jj in 1:length(x)) {
        bintmp <- base2base(x[[jj]],inBase,2, binSize = binSize, inTwosComp = inTwosComp, outTwosComp = FALSE)[[1]]
        isPos = TRUE
        if(length(grep('-',bintmp)) ) {
            isPos = FALSE
            bintmp <- gsub('^-','',bintmp)
        }
        bintmp <- strsplit(bintmp,'')[[1]]
        xlen = length(bintmp)
        # 循环移位
        otmp  <- unlist(paste0(c(bintmp[ ((1:xlen) + shift -1) %% (xlen) + 1]),collapse=''))
        out[jj] <- base2base(otmp, 2, outBase, binSize=binSize, outTwosComp = FALSE, classOut = "character")[[1]]
        if(!isPos) out[jj] <- paste0('-',out[jj], collapse='')
    }
    # 匹配输入类型输出
    if(is(x,'mpfr') || is(x,'mpfr1')) {
        out <- Rmpfr::.bigz2mpfr(gmp::as.bigz(out))
    } else if(is(x,'numeric')) {
        out <- as.numeric(out)
    } else if(is(x,'bigz')) out <- gmp::as.bigz(out)
    
 return(out) 
}

核心问题修正点

  1. 初始哈希值的字节序错误
    MD5规定初始值是小端字节序的数值表示,不需要做flip32操作。直接使用原始字符串转换即可:

    # 移除所有flip32调用
    # adef <- flip32(adef)
    # bdef <- flip32(bdef)
    # ...
    
  2. 消息分块的字节序解析错误
    拆分512位块为32位字时,MD5要求按小端字节序解析,应修改计算方式:

    for(jm in 1:16) thisM[jm] <- sum(thischunk[,jm] * c(1,256,256^2,256^3) )
    
  3. 消息长度追加的字节序错误
    追加64位长度时需按小端字节序存储,替换原长度处理代码:

    len_raw <- raw(8)
    len_val <- as.bigz(origlenbit)
    for(i in 1:8) {
        len_raw[i] <- as.raw(len_val %% 256)
        len_val <- len_val %/% 256
    }
    msg <- c(msg, len_raw)
    
  4. 最终结果的字节序转换
    最终输出需要将每个32位哈希值按小端字节序拼接,修改结果拼接代码:

    thesum <- paste0(flip32(a0x), flip32(b0x), flip32(c0x), flip32(d0x), collapse='')
    

内容的提问来源于stack exchange,提问作者Carl Witthoft

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.17 07:02:35