纯R实现MD5哈希函数输出与tools::md5sum不匹配问题排查
纯R语言实现MD5哈希函数的输出不匹配问题
我尝试用纯R语言编写不调用任何C例程的MD5哈希函数,代码能正常执行,但输出始终和tools::md5sum(其结果与各类在线示例一致)不匹配。我怀疑是字节序(或字序)的问题,代码里已经尝试加入flip操作但未成功,也检查过是否是字节顺序颠倒的简单情况,没有收获。
运行此函数需要Rmpfr和bigBits库,以及本文提供的flip32、flip16和bigRotate辅助函数:
# 需要加载的库 library(Rmpfr) library(bigBits) mymd5 <- function(msg){ if(!is(msg,'raw')) msg <- charToRaw(as.character(msg)) # 用于截断超过32位的数值 two32 <- as.bigz(2^(as.bigz(32))) sidx <- c( 7, 12, 17, 22, 7, 12, 17, 22, 7, 12, 17, 22, 7, 12, 17, 22 , 5, 9, 14, 20, 5, 9, 14, 20, 5, 9, 14, 20, 5, 9, 14, 20 , 4, 11, 16, 23, 4, 11, 16, 23, 4, 11, 16, 23, 4, 11, 16, 23 , 6, 10, 15, 21, 6, 10, 15, 21, 6, 10, 15, 21, 6, 10, 15, 21 ) # K值已验证正确 K <- as.bigz(2^32 * abs(sin(1:64))) # IETF文档显示初始值无需调整字节序,A的表示为01234567(低字节在前) adef <- '67452301' bdef <- 'efcdab89' cdef <- '98badcfe' ddef <- '10325476' # 尝试过的flip操作(当前启用) adef <- flip32(adef) bdef <- flip32(bdef) cdef <- flip32(cdef) ddef <- flip32(ddef) a0 <- (base2base(adef,16,10))[[1]] b0 <-(base2base(bdef,16,10))[[1]] c0 <-(base2base(cdef,16,10))[[1]] d0 <-(base2base(ddef,16,10))[[1]] # 追加0x80,再填充0x00使消息字节数 ≡56 mod64 origlenbit <- length(msg) * 8 msg <- c(msg, as.raw('0x80') ) bytemod <- length(msg)%%64 raw00 <- as.raw('0x00') if (bytemod < 56 && bytemod > 0 ) { msg <- c(msg,rep(raw00,times=(56-bytemod))) } else if (bytemod > 56) { msg <- c(msg, rep(raw00, times = bytemod-56)) } # 追加原始消息的比特长度(64位,低32位在前) len16 <- base2base(origlenbit,10,16)[[1]] lenbit2 <- base2base(origlenbit,10,2)[[1]] if(nchar(lenbit2) < 64 ){ lenbit2 <- paste0(c(rep('0',times = 64 -nchar(lenbit2)) ,lenbit2), collapse='') } else if (nchar(lenbit2) > 64) lenbit2 <- rev(rev(lenbit2)[1:64]) newbytes2 <- substring(lenbit2, seq(1,nchar(lenbit2)-7,by=8), last = seq(8, nchar(lenbit2),by = 8) ) newdec <- base2base(newbytes2,2,10) newdbl <- rep(0,times= 8) for (jd in 1: 8) newdbl[jd] <- as.numeric(newdec[[jd]]) newraw <- as.raw(newdbl) # 尝试过的不同长度字节顺序 # msg <- c(msg,newraw[c(7,8,5,6,3,4,1,2)]) msg <- c(msg, newraw[5:8], newraw[1:4]) # msg <- c(msg, newraw[1:4], newraw[5:8]) chunkit <- as.bigz(as.numeric(msg ) ) # 将512位消息块拆分为16个32位字 for(jchunk in 1: (length(msg)/64)) { thischunk <- matrix(chunkit[ ((jchunk-1)*64)+(1:64)],ncol=16) thisM <- as.bigz(rep(0,16) ) # 尝试过的两种字节顺序计算方式 for(jm in 1:16) thisM[jm] <- sum(thischunk[,jm] * c(256^3,256^2,256,1) ) # for(jm in 1:16) thisM[jm] <- sum(thischunk[,jm] * c(1,256,256^2,256^3) ) # 初始化当前块的哈希值 Ah<- a0 Bh<- b0 Ch<- c0 Dh<- d0 # 四轮循环计算 for (jone in 1:16) { Fh <- bigOr (bigAnd(Bh,Ch, inTwo=FALSE)[[1]] ,bigAnd(bigNot(Bh, inTwo=FALSE, outTwo=FALSE)[[1]],Dh, inTwo=FALSE)[[1]] )[[1]] g <- jone Fh<- (Fh + Ah + K[jone] + thisM[g]) %%two32 Ah<- Dh Dh<- Ch Ch<- Bh Bh<- (Bh + bigRotate(Fh, sidx[jone], inTwo=FALSE)[[1]] )%%two32 } for (j2 in 17:32) { Fh <- bigOr( bigAnd(Dh,Bh,inTwo=FALSE)[[1]] ,bigAnd(bigNot(Dh[[1]],inTwo=FALSE, outTwo=FALSE), Ch, inTwo=FALSE)[[1]] ,inTwo=FALSE)[[1]] g <-(1 + 5*(j2-1))%%16 +1 Fh<- (Fh + Ah + K[j2] + thisM[g])%%two32 Ah<- Dh Dh<- Ch Ch<- Bh Bh<- (Bh + bigRotate(Fh, sidx[j2], inTwo=FALSE)[[1]])%%two32 } for(j3 in 33:48){ Fh <- bigXor(Bh,bigXor(Ch,Dh, inTwo=FALSE)[[1]], inTwo=FALSE)[[1]] g <-(5 + 3*(j3-1))%%16 +1 Fh<- (Fh + Ah + K[j3] + thisM[g])%%two32 Ah<- Dh Dh<- Ch Ch<- Bh Bh<- (Bh + bigRotate(Fh, sidx[j3],inTwo=FALSE)[[1]])%%two32 } for(j4 in 49:64){ Fh <- bigXor(Ch,bigOr(Bh,bigNot(Dh,inTwo=FALSE, outTwo=FALSE)[[1]],inTwo=FALSE)[[1]],inTwo=FALSE)[[1]] g <- (7 * (j4 -1))%%16 +1 Fh<- (Fh + Ah + K[j4] + thisM[g])%%two32 Ah<- Dh Dh<- Ch Ch<- Bh Bh<- (Bh + bigRotate(Fh, sidx[j4], inTwo=FALSE)[[1]])%%two32 } # 累加当前块的哈希结果 a0<- (a0 + Ah)%%two32 b0<- (b0 + Bh)%%two32 c0<- (c0 + Ch)%%two32 d0<- (d0 + Dh)%%two32 } # 拼接最终结果,补全前导零 a0x <-(base2base(a0,10,16)) a0x <- paste0(c(rep(0,times=max(8-(nchar(a0x)),0)), a0x), collapse='') b0x <-(base2base(b0,10,16)) b0x <- paste0(c(rep(0,times=max(8-(nchar(b0x)),0)), b0x), collapse='') c0x <-(base2base(c0,10,16)) c0x <- paste0(c(rep(0,times=max(8-(nchar(c0x)),0)), c0x), collapse='') d0x <-(base2base(d0,10,16)) d0x <- paste0(c(rep(0,times=max(8-(nchar(d0x)),0)), d0x), collapse='') thesum <- paste0(c(a0x,b0x,c0x,d0x) , sep='',collapse='') return(thesum) } # 辅助函数:翻转32位十六进制字符串的字节顺序 flip32 <- function(x){ xsep <- unlist(strsplit(x,'') ) xout <- paste0(c(xsep[c(7,8,5,6,3,4,1,2)]),collapse='') return(xout) } flip16 <- function(x) { xsep <- unlist(strsplit(x,'')) xout <- paste0(c(xsep[c(3,4,1,2)]),collapse='') return(xout) } # 辅助函数:32位循环左移 bigRotate <- function(x, shift, inBase = 10,binSize = 32, outBase = 10, inTwosComp = TRUE) { shift = floor(shift[1]) out <- rep('0',times = length(x)) for(jj in 1:length(x)) { bintmp <- base2base(x[[jj]],inBase,2, binSize = binSize, inTwosComp = inTwosComp, outTwosComp = FALSE)[[1]] isPos = TRUE if(length(grep('-',bintmp)) ) { isPos = FALSE bintmp <- gsub('^-','',bintmp) } bintmp <- strsplit(bintmp,'')[[1]] xlen = length(bintmp) # 循环移位 otmp <- unlist(paste0(c(bintmp[ ((1:xlen) + shift -1) %% (xlen) + 1]),collapse='')) out[jj] <- base2base(otmp, 2, outBase, binSize=binSize, outTwosComp = FALSE, classOut = "character")[[1]] if(!isPos) out[jj] <- paste0('-',out[jj], collapse='') } # 匹配输入类型输出 if(is(x,'mpfr') || is(x,'mpfr1')) { out <- Rmpfr::.bigz2mpfr(gmp::as.bigz(out)) } else if(is(x,'numeric')) { out <- as.numeric(out) } else if(is(x,'bigz')) out <- gmp::as.bigz(out) return(out) }
核心问题修正点
初始哈希值的字节序错误
MD5规定初始值是小端字节序的数值表示,不需要做flip32操作。直接使用原始字符串转换即可:# 移除所有flip32调用 # adef <- flip32(adef) # bdef <- flip32(bdef) # ...消息分块的字节序解析错误
拆分512位块为32位字时,MD5要求按小端字节序解析,应修改计算方式:for(jm in 1:16) thisM[jm] <- sum(thischunk[,jm] * c(1,256,256^2,256^3) )消息长度追加的字节序错误
追加64位长度时需按小端字节序存储,替换原长度处理代码:len_raw <- raw(8) len_val <- as.bigz(origlenbit) for(i in 1:8) { len_raw[i] <- as.raw(len_val %% 256) len_val <- len_val %/% 256 } msg <- c(msg, len_raw)最终结果的字节序转换
最终输出需要将每个32位哈希值按小端字节序拼接,修改结果拼接代码:thesum <- paste0(flip32(a0x), flip32(b0x), flip32(c0x), flip32(d0x), collapse='')
内容的提问来源于stack exchange,提问作者Carl Witthoft
相关产品推荐
相关产品推荐

