You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

计算字符串长度时切换段寄存器的原因及异常问题咨询

汇编中ES/DS寄存器与字符串长度计算的疑问

我有两种计算字符串长度的汇编实现,搞不懂为什么要切换额外段(ES寄存器)和数据段(DS寄存器)——准确说是把DS内容复制到ES里。移除push ds和pop es语句后结果没变化,但把les di, [bp+4]换成mov di, [bp+4]后,屏幕能显示消息但带大量乱码。作为汇编新手,想知道这是怎么回事?

原始实现代码

org 0x0100
jmp start

message: db 'Hello world!', 0      ; First string to be concatenated


; Subroutine to calculate the length of a string
; Takes the segment and offset of a string as parameters
strlen:
    push bp
    mov bp, sp
    push es
    push cx
    push di
    les di, [bp+4]        ; Point es:di to string
    mov cx, 0xffff        ; Load maximum number in cx
    xor al, al            ; Load a zero in al
    repne scasb           ; Find zero in the string
    mov ax, 0xffff        ; Load maximum number in ax
    sub ax, cx            ; Find change in cx
    dec ax                ; Exclude null from length
    pop di
    pop cx
    pop es
    pop bp
    ret 4

; Subroutine to print a string
; Takes the x position, y position, attribute, and address of a null-terminated string as parameters
printstr:
    push bp
    mov bp, sp
    pusha
    push di
    push ds               ; Push segment of string
    mov ax, [bp+4]
    push ax               ; Push offset of string
    call strlen           ; Calculate string length

    cmp ax, 0             ; Is the string empty?
    jz exit               ; No printing if string is empty

    mov cx, ax            ; Save length in cx
    mov ax, 0xb800        ; Video memory base address
    mov es, ax            ; Point es to video base
    mov al, 80            ; Load al with columns per row
    mul byte [bp+8]       ; Multiply with y position
    add ax, [bp+10]       ; Add x position
    shl ax, 1             ; Turn into byte offset
    mov di, ax            ; Point di to required location
    mov si, [bp+4]        ; Point si to string
    mov ah, [bp+6]        ; Load attribute in ah
    cld                   ; Clear direction flag for auto-increment mode

 nextchar:
    lodsb                 ; Load next char in al
    stosw                 ; Print char/attribute pair
    loop nextchar         ; Repeat for the whole string

 exit:
    pop ds
    pop di
    popa
    pop bp
    ret 8

start:
    mov ax, 30
    push ax               ; Push x position
    mov ax, 20
    push ax               ; Push y position
    mov ax, 0x7           ; Blue on white attribute
    push ax               ; Push attribute
    mov ax, message
    push ax               ; Push address of message
    call printstr         ; Call the printstr subroutine

    mov ax, 0x4c00        ; Terminate program
    int 0x21

替代实现代码片段

push bp
    mov bp, sp
    pusha

    push ds
    pop es         ; load ds in es
    mov di, [bp+4] ; point di to string
    mov cx, 0xffff ; load maximum number in cx
    xor al, al 
    repne scasb 
    mov ax, 0xffff 
    sub ax, cx 
    dec ax 
    jz done 

    mov cx, ax
    ...

问题解答

1. 为什么要把DS复制到ES?

核心原因是**scasb指令的寻址规则**:scasb专门用来比较AL寄存器的值与ES:DI指向的内存字节,执行后还会根据方向标志DF自动增减DI。这个指令只能使用ES段来寻址目标数据,无法直接用DS段。

  • 原始的strlen子程序是通用设计:它通过栈接收字符串的段地址+偏移地址,用les di, [bp+4]把栈中的段值加载到ES、偏移加载到DI,这样不管字符串在哪个段(比如DS、ES甚至其他自定义段),都能正确寻址。
  • 替代版本里复制DS到ES,是当前场景的简化处理:因为你的程序是DOS .com格式,字符串默认在DS段里,让ES和DS指向同一段后,ES:DI就等价于DS:DI,同样能正确找到字符串。

2. 移除push ds和pop es后结果不变?

这是因为当前程序的ES和DS初始值完全相同。DOS加载.com程序时,会把DS、ES、SS等段寄存器都设置为同一个值(指向程序的PSP和代码数据区)。所以即使你移除push ds/pop es,ES还是和DS指向同一段,scasb用ES:DI寻址的依然是正确的字符串,结果自然不变。但如果字符串放在非DS的其他段,移除这两句就会导致寻址错误。

3. 替换les di, [bp+4]为mov di, [bp+4]后出现乱码?

这是因为两个指令的功能完全不同:

  • les di, [bp+4]:从栈的[bp+4]位置读取4字节数据(低2字节是字符串偏移,高2字节是字符串段地址),把段地址加载到ES,偏移加载到DI,确保ES:DI指向正确的字符串。
  • mov di, [bp+4]:只读取2字节的偏移地址到DI,ES寄存器的值还是之前printstr里设置的0xb800(视频内存段)!

此时scasb是在0xb800:DI(也就是视频内存区域)里查找0字节,计算出来的字符串长度完全错误。printstr用这个错误的长度去循环打印字符,自然会输出大量不属于你的字符串的乱码内容。


内容的提问来源于stack exchange,提问作者Fatima sami

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.16 08:16:09