You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何在Bash中基于非整数子串(文件名时间戳)排序文件数组?

按文件名时间戳排序清理JFR文件的脚本优化方案

核心思路

要实现忽略目录,仅按文件名的时间戳排序并优先删除最早文件,关键是提取每个JFR文件的时间戳文件名作为排序依据,而非依赖find的目录排序结果。由于你的文件名是年_月_日_时_分_秒_xxx.jfr格式,时间戳字符串的字典序与时间先后完全一致,直接按该字符串排序即可。

具体修改步骤

1. 重构文件列表获取与排序逻辑

替换原fillarray函数,通过find输出文件名和路径,提取时间戳部分排序后再获取文件路径,同时解决原脚本中文件名含空格时数组存储错误的问题:

fillarray() { 
    files=()
    # 遍历find输出的排序后文件路径,存入数组
    while IFS= read -r file_path; do
        files+=("$file_path")
    done < <(
        # 1. 查找所有.jfr文件,输出"文件名 完整路径"(制表符分隔)
        find "$pathToFolder" -name "*.jfr" -type f -printf "%f\t%p\n" |
        # 2. 去掉文件名的.jfr后缀,得到纯时间戳字符串
        sed 's/\.jfr$//' |
        # 3. 按时间戳字符串排序(字典序=时间顺序,最早的在前)
        sort -k1,1 |
        # 4. 提取排序后的完整文件路径
        cut -f2
    )
    totalFiles=${#files[@]}
}

2. 修复文件大小获取函数

原getfilesize函数没有返回值,导致调用时无法正确获取大小,修改为直接输出结果:

getfilesize() {  
    du -k "$1" | cut -f1
}

3. 修复空文件检测逻辑

原scanfornullfiles中! getfilesize的写法错误,改为正确调用并判断,同时改用新数组存储非空文件,避免原数组删除元素时的错误:

scanfornullfiles() { 
    local filtered_files=()
    for file in "${files[@]}"; do
        filesize=$(getfilesize "$file")
        if [[ $filesize -eq 0 ]]; then
            echo "msg=\"Removal result\", result=false, file=$(realpath "$file"), reason=\"empty file\""
        else
            filtered_files+=("$file")
        fi
    done
    files=("${filtered_files[@]}")
}

4. 优化文件清理逻辑

原checkfiles中每次修改数组的方式效率低且易出错,改为直接计算需要删除的文件数量(总文件数 - 保留数),遍历前N个文件进行删除:

checkfiles() {
    local files_to_remove=$(( ${#files[@]} - maxFiles ))
    if [[ $files_to_remove -le 0 ]]; then
        for file in "${files[@]}"; do
            echo "msg=\"Removal result\", result=false, file=$(realpath "$file"), reason=\"within max file boundary\""
        done
        return
    fi

    # 删除前files_to_remove个文件(最早的时间戳)
    for ((i=0; i<files_to_remove; i++)); do
        file="${files[$i]}"
        filesize=$(getfilesize "$file")
        if [[ $dryRun == "false" ]]; then
            echo "msg=\"Removal result\", result=true, file=$(realpath "$file"), filesize=$filesize, reason=\"outside max file boundary\""
            rm "$file"
        else
            echo "msg=\"Removal result\", result=true, file=$(realpath "$file"), filesize=$filesize, reason=\"outside max file boundary\""
        fi
        ((count++))
    done

    # 输出保留的文件结果
    for ((i=files_to_remove; i<${#files[@]}; i++)); do
        file="${files[$i]}"
        echo "msg=\"Removal result\", result=false, file=$(realpath "$file"), reason=\"within max file boundary\""
    done
}

优化后的完整脚本

#!/bin/bash

# handle -h (help)
if [[ "$1" == "-h" || "$1" == "" ]]; then 
  echo -e '-p [目标文件夹路径] \n-f [需保留的文件数量] \n-d [设为false关闭模拟运行]'
  exit 0
fi

# handle parameters
while getopts p:f:d: flag 
do
    case "${flag}" in
        p) pathToFolder=${OPTARG};;
        f) maxFiles=${OPTARG};;
        d) dryRun=${OPTARG};;
        *) echo -e '-p [目标文件夹路径] \n-f [需保留的文件数量] \n-d [设为false关闭模拟运行]'
    esac
done

if [[ -z $dryRun ]]; then
    dryRun=true
fi

# 填充数组:按文件名时间戳排序,最早的在前
fillarray() { 
    files=()
    while IFS= read -r file_path; do
        files+=("$file_path")
    done < <(
        find "$pathToFolder" -name "*.jfr" -type f -printf "%f\t%p\n" |
        sed 's/\.jfr$//' |
        sort -k1,1 |
        cut -f2
    )
    totalFiles=${#files[@]}
}

# 返回文件大小(KB)
getfilesize() {  
    du -k "$1" | cut -f1
}

count=0

# 检查并清理超出保留数量的文件
checkfiles() {
    local files_to_remove=$(( ${#files[@]} - maxFiles ))
    if [[ $files_to_remove -le 0 ]]; then
        for file in "${files[@]}"; do
            echo "msg=\"Removal result\", result=false, file=$(realpath "$file"), reason=\"within max file boundary\""
        done
        return
    fi

    # 删除最早的files_to_remove个文件
    for ((i=0; i<files_to_remove; i++)); do
        file="${files[$i]}"
        filesize=$(getfilesize "$file")
        if [[ $dryRun == "false" ]]; then
            echo "msg=\"Removal result\", result=true, file=$(realpath "$file"), filesize=$filesize, reason=\"outside max file boundary\""
            rm "$file"
        else
            echo "msg=\"Removal result\", result=true, file=$(realpath "$file"), filesize=$filesize, reason=\"outside max file boundary\""
        fi
        ((count++))
    done

    # 输出保留文件的结果
    for ((i=files_to_remove; i<${#files[@]}; i++)); do
        file="${files[$i]}"
        echo "msg=\"Removal result\", result=false, file=$(realpath "$file"), reason=\"within max file boundary\""
    done
}

# 扫描并移除空文件
scanfornullfiles() { 
    local filtered_files=()
    for file in "${files[@]}"; do
        filesize=$(getfilesize "$file")
        if [[ $filesize -eq 0 ]]; then
            echo "msg=\"Removal result\", result=false, file=$(realpath "$file"), reason=\"empty file\""
        else
            filtered_files+=("$file")
        fi
    done
    files=("${filtered_files[@]}")
}

echo "msg=\"jfrcleanup.sh started\", maxFiles=$maxFiles, dryRun=$dryRun, directory=$pathToFolder"
{
    cd "$pathToFolder" > /dev/null 2>&1
} || {
    echo "msg=\"no permission in directory\""
    echo "msg=\"jfrcleanup.sh stopped\""
    exit 0
}

fillarray

scanfornullfiles
checkfiles

echo "msg=\"jfrcleanup.sh finished\", totalFileCount=$totalFiles, filesRemoved=$count"

内容的提问来源于stack exchange,提问作者Tim Reber

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.08.19 08:10:32