You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何使用Polars内置方法实现短语缩写?

需求:Polars短语缩写优化

核心需求

  • 对Polars Series或表达式中的短语执行缩写,步骤如下:
    1. 提取短语中的首字母大写单词;
    2. 根据大写单词总长度计算每个单词的比例长度;
    3. 调整长度,使最终缩写总字符数达到目标值(例如4个字符)。
  • 待办:若缩写出现重复,需自动处理(如添加数字)或标记警告。
  • 当前问题:现有实现通过map_elements调用Python函数处理,寻求更高效的Polars内置方法替代。

当前实现代码

import polars as pl

def _abbreviate_phrase(phrase: str, length: int) -> str:
    """将单个短语缩写为指定长度的字符串。
    核心逻辑:聚焦首字母大写的单词,按比例分配每个单词的缩写长度,最终拼接成目标长度的缩写。
    
    示例:
        phrase = 'Commercial & Professional'
        length = 4
        res = _abbreviate_phrase(phrase, length)
        print(res)
        # 输出:CoPr
    """
    # 提取首字母大写的单词
    capitalized_words = [word for word in phrase.split(' ') if word[0].isupper()]
    word_lengths = [len(word) for word in capitalized_words]
    total_word_length = sum(word_lengths)

    if total_word_length == 0:
        return ''  # 无大写单词时返回空字符串

    # 计算每个单词的比例缩写长度
    proportional_lengths = [round(wl / total_word_length * length) for wl in word_lengths]
    total_proportional_length = sum(proportional_lengths)

    # 调整长度,确保总长度匹配目标值
    if total_proportional_length < length:
        for i in range(length - total_proportional_length):
            proportional_lengths[i] += 1
    elif total_proportional_length > length:
        for i in range(total_proportional_length - length):
            proportional_lengths[i] -= 1

    # 拼接缩写结果
    abbreviated_phrase = ''.join([word[:plength] for word, plength in zip(capitalized_words, proportional_lengths)])
    return abbreviated_phrase

def abbreviate_phrases(phrases: pl.Series, length: int) -> pl.Series:
    """批量缩写Polars Series中的短语。
    
    示例:
        phrases = pl.Series([
            'Sunshine',
            'Sunset',
            'Climate Change and Environmental Impact',
            'Health and Wellness',
            'Quantum Computing and Physics',
            'Global Warming and Renewable Resources'
        ])
        length = 4
        res = abbreviate_phrases(phrases, length)
        print(res)
        # 输出:
        # Series: '' [str]
        # [
        #   "Suns"
        #   "Suns"
        #   "CEnI"
        #   "HeWe"
        #   "QCoP"
        #   "GWRR"
        # ]
    """
    abbreviates = phrases.map_elements(lambda x: _abbreviate_phrase(x, length), return_dtype=pl.String)
    # if not abbreviates.is_unique().all():
    #     print('警告:存在重复缩写。')
    return abbreviates

性能对比测试设置

import sys
import timeit

def generate_phrases(n: int) -> pl.DataFrame:
    # 将原始短语重复n次生成测试数据
    phrases = pl.DataFrame({
        "p": ['Climate Change and Environmental Impact',
              'Health and Wellness',
              'Quantum Computing and Physics',
              'Global Warming and Renewable Resources',
              'no capital letters']
    })
    return pl.concat([phrases.with_columns(pl.col('p')+'_'+str(i)) for i in range(n)])

def compare_performance(n: int, length: int = 4):
    phrases = generate_phrases(n)

    abbreviate_phrases_time = timeit.timeit(lambda: abbreviate_phrases(phrases['p'], length), number=10)
    abbreviate_phrases_harbeck_time = timeit.timeit(lambda: abbreviate_phrases_harbeck(phrases, phrase_column="p", length=length), number=10)
    abbreviate_phrases_jqurious_time = timeit.timeit(lambda: abbreviate_phrases_jqurious(phrases['p'], length=length), number=10)
    abbreviate_phrases_rle_time = timeit.timeit(lambda: abbreviate_phrases_rle(phrases['p'], length=length), number=10)

    ratio_harbeck = abbreviate_phrases_time / abbreviate_phrases_harbeck_time
    ratio_jqurious = abbreviate_phrases_time / abbreviate_phrases_jqurious_time
    ratio_rle = abbreviate_phrases_time / abbreviate_phrases_rle_time

    return ratio_harbeck, ratio_jqurious, ratio_rle

性能对比测试结果

n = 200_000
ratio_harbeck, ratio_jqurious, ratio_rle = compare_performance(n=n, length=4)

print(f"性能对比(总行数:{n*5}):")
print(f"    原实现/harbeck实现  {ratio_harbeck:.2f}x")
print(f"    原实现/jqurious实现 {ratio_jqurious:.2f}x")
print(f"    原实现/rle实现     {ratio_rle:.2f}x")
print()
print(f'Python版本:{sys.version.split(' ')[0]}')
print(f'Polars版本:{pl.__version__}')

# 输出结果:
# 性能对比(总行数:1000000):
#     原实现/harbeck实现  0.70x
#     原实现/jqurious实现 1.30x
#     原实现/rle实现     1.79x

# Python版本:3.12.0
# Polars版本:1.12.0

内容的提问来源于stack exchange,提问作者user11062613

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.16 07:09:55