You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何利用awk的FIELDWIDTHS读取固定宽度记录数据?

问题

已知单条记录长度为29字符,每条记录包含3个字段,宽度分别为15、10、4。能否通过awk的FIELDWIDTHS特性读取以下固定宽度格式的数据?数据大小约2MB,示例数据如下:

4harddisk-------412999238---13cdrom-------------488388---22floppy-----------4994942---31mouse-----------20303202---4

第一条记录为4harddisk-------412999238---1,字段处理要求:

  • 第一个字段(15位):从4harddisk-------中提取有效内容4harddisk
  • 第二个字段(10位):从-------412999238中提取有效内容412999238
  • 第三个字段(4位):从---1中提取有效内容1

我尝试了两种方法,但都不够简便:

方法一(速度极慢)

通过RS=".{29}"用getline把数据读入RT,再通过双向管道调用cat并设置RS="^$"强制awk分割字段,代码如下:

awk '
function load_fixedWidth_record (width, setFields,     command)
   {
   if (width==0) { $0=""; RT=""; return 1; }
   
   RS=".{" width "}"; command="cat";
   if (getline)
      {
      if (length(RT)>0)                      #-- RT contains the record, $0 is empty
         {
         if (setFields)
            {
            printf "%s",RT |& command; close (command, "to");
            RS="^$"; command |& getline;     #-- Sets $0, NF, and RT

            close(command, "from");
            }
         else { $0=RT; RT=""; }
         return 1;
         }
      else return 0;                         #-- record is last one and record length was smaller than width, $0 contains the record, RT=""
      }
   else return -1;
   }
   
BEGIN {
FIELDWIDTHS="15 10 4";
load_fixedWidth_record(29, 1);
for (i=1;i<=NF;i++) print $i;
} ' <<< "4harddisk-------412999238---13cdrom-------------488388---22floppy-----------4994942---31mouse-----------20303202---4"

方法二(自行分割字段)

编写函数根据FIELDWIDTHS手动分割字段,但希望有更简便的、让awk自动处理的方法,代码如下:

awk '
#-- loads a fixed width record of the actual datastream into $0, returns the load success
#--   if setFields is true, set fields according to PROCINFO["FS"]
#--   success: -1 ... already on end of datastream, 0 ... record smaller than width, end of datastream reached, 1 ... success
function load_fixedWidth_record (width, setFields,    new_RS, fields, i, j, l, str_pos, skip, take, dp, star_found, star_i, star_strPos, star_skip, star_take, substring_ARR)
   {
   if (width==0) { $0=""; RT=""; return 1; }
   
   new_RS=".{" width "}";             #-- setting RS to often can immensly slow down the execution, i saved the old RS and restored it back
   if (RS!=new_RS) RS=new_RS;         #-- every time after function execution, this slowed down the function by 100times, i removed it
   
   if (getline)
      {
      if (length(RT)>0)               #-- RT contains the record, $0 is empty
         {
         if (setFields)
            {
            switch (PROCINFO["FS"])
               {
               case "FS": split(RT, fields); for (i in fields) $i=fields[i]; NF=length(fields); break;
               
               case "FIELDWIDTHS":
                  split(FIELDWIDTHS,fields," "); str_pos=1; star_i=0; j=1;
                  for (i in fields)
                     {
                     star_found=0; skip=0; dp=index(fields[i],":");                                               #-- search for doublepoint dp
                     if (dp) 
                        {
                        skip=substr(fields[i],1,dp-1); take=substr(fields[i],dp+1);                               #-- field descriptor ... skip:take ... supports also - *:3 or 5:* and any * in the middle of fields, but awk supports only * at last field
                        if (take=="*") star_found=1; else take=strtonum(take);                                    #-- search for take *, else str to number
                        if (skip=="*") { if (take=="*") skip=0; else star_found=1; } else skip=strtonum(skip);    #-- search for skip *, else str to number, "*:*" becomes "0:*"
                        }
                     else if (take=="*") star_found=1; else take=strtonum(take);
       
                     if (star_found)
                        {
                        if (star_i==0)                                                                            #-- star found - save star values
                           {                                                                                      #-- and reserve two element places in substring_ARR (j+=2)  
                           star_i=i; star_strPos=str_pos; 
                           star_skip=skip; star_take=take;
                           j+=2; continue;
                           }
                        continue;                                                                                 #-- if more stars are found in other field descriptors, skip the complete field
                        }
                     else
                        {
                        str_pos+=skip; substring_ARR[j++]=str_pos;                                                #-- build substring_ARR  elements 1,3,5, ... save index where to cut
                        substring_ARR[j++]=take; str_pos+=take;                                                   #--                      elements 2,4,6, ... save how many chars to cut
                        }                                                                                         #-- field 1: element 1, element 2; field 2: element 3, element 4; ...                                                                                   
                     }
                      
                  if (star_i>0)                                                                                      #-- star found - fill reserved empty places in substring_ARR for star field
                     {
                     if (star_take=="*") { star_take=length(RT)-star_skip; if (length(star_take<0)) star_take=0; }   #-- calculate star_take; if (<0) star_take=0;
                     else { star_skip=length(RT)-star_take; if (star_skip<0) star_skip=0; }                          #-- calculate star_skip; if (<0) star_skip=0;
                      
                     i=star_skip+star_take;                                                                          #-- save star field length in i
                     if (i>0)                                                                                        #-- if field length >0 fill reserved places
                        {
                        j=star_i*2-1; substring_ARR[j++]=star_strPos+star_skip; substring_ARR[j++]=star_take;        
                        l=length(substring_ARR); for (;j<=l;j+=2) substring_ARR[j]+=i;                               #-- update the cut position of the fields, after the inserted star field
                        }
                     else { j=star_i*2-1; delete substring_ARR[j]; delete substring_ARR[j+1]; }                      #-- field length ==0 -> delete reserved places
                     }
                  #-- write fields -- star_found is used to save the cut position
                  i=1; star_found=0; for (j in substring_ARR) { if (star_found==0) star_found=j; else { $i=substr(RT, substring_ARR[star_found], substring_ARR[j]); star_found=0; i++ }}
                  NF=i-1; break;
                  
               case "FPAT": patsplit(RT, fields); for (i in fields) $i=fields[i]; NF=length(fields); break;
               }
            }
         $0=RT; RT=""; return 1;
         }
      else return 0;                  #-- record is last one and record length was smaller than width, $0 contains the record, RT=""
      }
   else return -1;
   }
   
BEGIN {
FIELDWIDTHS="15 10 4";
load_fixedWidth_record(29, 1);
for (i=1;i<=NF;i++) print $i;
} ' <<< "4harddisk-------412999238---13cdrom-------------488388---22floppy-----------4994942---31mouse-----------20303202---4"
解决方案

完全可以利用awk的FIELDWIDTHS特性实现,无需复杂的自定义函数或管道操作,核心思路是直接基于FIELDWIDTHS自动分割字段,同时高效截取固定长度的记录并清理字段内容:

awk '
BEGIN {
    FIELDWIDTHS = "15 10 4"
    RECORD_LEN = 29
}
{
    # 循环截取所有完整的29字符记录
    for (pos = 1; pos + RECORD_LEN - 1 <= length($0); pos += RECORD_LEN) {
        # 提取当前记录并赋值给$0,触发FIELDWIDTHS自动分割
        $0 = substr($0, pos, RECORD_LEN)
        # 清理字段中的多余分隔符
        $1 = gensub(/-+$/, "", 1, $1)  # 去掉第一个字段末尾的-
        $2 = gensub(/^-+/, "", 1, $2)  # 去掉第二个字段开头的-
        $3 = gensub(/^-+/, "", 1, $3)  # 去掉第三个字段开头的-
        # 输出处理后的字段,可按需调整格式
        print $1, $2, $3
    }
}
' <<< "4harddisk-------412999238---13cdrom-------------488388---22floppy-----------4994942---31mouse-----------20303202---4"

代码说明

  • FIELDWIDTHS = "15 10 4":直接告诉awk按指定宽度分割字段,无需手动计算字段位置
  • 循环截取记录:通过substr从输入字符串中逐个提取完整的29字符记录
  • 字段清理:使用gensub精准去除字段中多余的-,匹配规则完全贴合你的提取需求
  • 效率优势:全程在awk内部处理,无外部进程调用,对于2MB的数据可快速完成处理

测试输出:

4harddisk 412999238 1
cdrom 488388 22
floppy 4994942 31
mouse 20303202 4

内容的提问来源于stack exchange,提问作者Schmaehgrunza

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.02 00:40:58