You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

如何将英国假期等级规则(1-16级)转换为Python代码实现?

英国假期等级规则的Python实现方案

我正试图完成一项理论简单但实际编码难度大的任务——把一份文档第14页的英国假期等级规则(1-16级,忽略更高等级)转换成Python代码,给2012-2022年的日期序列标注对应等级。我不想手动处理,希望写出通用逻辑,方便后续适配其他国家的规则。

我已经做了初步尝试:先获取英国假期日历,创建rank列占位,再基于假期信息写逻辑,但实现起来很棘手。想请教更优方案或具体实现方法。

我的初步尝试代码:

import holidays
import numpy as np
import pandas as pd

dates = pd.date_range(start='2012-01-01', end='2022-12-31', freq='D')
df = pd.DataFrame({'Value': np.random.rand(len(dates))}, index=dates)

def get_british_holidays(df):
    gb_holidays = holidays.UnitedKingdom()
    holiday_dates = pd.Series(index=df.index)
    for single_date in holiday_dates.index:
        if single_date in gb_holidays:
            holiday = gb_holidays[single_date]
            holiday_parts = [part.strip() for part in holiday.split(',')]
            holiday_parts = [part for part in holiday_parts if '[Northern Ireland]' not in part]
            holiday_dates.loc[single_date] = ', '.join(holiday_parts)
    holiday_dates = holiday_dates.replace('', np.nan).fillna(value=np.nan)
    holiday_dates = holiday_dates.to_frame(name='holiday')
    return pd.merge(df, holiday_dates, left_index=True, right_index=True, how='left')


df = get_british_holidays(df)
df['rank'] = 0

# first attempt
for date, rank in df.loc[df.index.month.isin([12, 1])].iterrows():
    if df.loc[date, 'holiday'] == 'Christmas Day':
        if date.dayofweek <= 3:
            holiday_period_start = pd.Timestamp(date.year, date.month, date.day - 3 - date.dayofweek)
        else:
            holiday_period_start = pd.Timestamp(date.year, date.month, date.day - date.dayofweek)

    if df.loc[date, 'holiday'] == 'New Year Holiday [Scotland]':
        holiday_period_end = pd.Timestamp(date.year, date.month, date.day + date.dayofweek)

# second attempt
for date, rank in df.loc[df.index.month.isin([12, 1])].iterrows():

    if date in pd.date_range(pd.Timestamp(date.year, 12, 24), pd.Timestamp(date.year + 1, 1, 2)):
        df.loc[date, 'rank'] = 5
    elif date == pd.Timestamp(date.year, 12, 25):
        df.loc[date, 'rank'] = 1
    elif date in pd.date_range(pd.Timestamp(date.year, 12, 26), pd.Timestamp(date.year, 12, 27)) or \
            date in pd.date_range(pd.Timestamp(date.year, 1, 1), pd.Timestamp(date.year, 1, 2)):
        df.loc[date, 'rank'] = 2
    elif date.dayofweek < 5 and date in pd.date_range(pd.Timestamp(date.year, 12, 24),
                                                      pd.Timestamp(date.year, 1, 1)):
        if df.loc[date, 'rank'] == 0:
            df.loc[date, 'rank'] = 3

优化方案与实现步骤

1. 重构假期获取逻辑,提升效率

原代码用循环遍历日期,效率较低,改用批量映射+过滤的方式:

def get_british_holidays(df):
    gb_holidays = holidays.UnitedKingdom()
    # 批量映射日期到假期名称
    holiday_series = pd.Series(
        df.index.map(lambda d: gb_holidays.get(d) if d in gb_holidays else np.nan),
        index=df.index,
        name='holiday'
    )
    # 过滤北爱尔兰专属假期
    holiday_series = holiday_series.apply(
        lambda x: ', '.join([p.strip() for p in str(x).split(',') if '[Northern Ireland]' not in p]) 
        if pd.notna(x) else np.nan
    )
    holiday_series.replace('', np.nan, inplace=True)
    return df.join(holiday_series)

2. 用向量化操作替代循环,提升性能

避免iterrows()遍历,利用Pandas的向量化方法批量处理等级规则(以下以圣诞-新年时段为例,需根据文档补充完整1-16级规则):

def assign_holiday_ranks(df):
    # 初始化默认等级(非特殊时段的基础等级,需按文档调整)
    df['rank'] = 10

    # 等级1:圣诞节当天
    df.loc[df['holiday'] == 'Christmas Day', 'rank'] = 1

    # 等级2:节礼日及次日(工作日)、新年元旦及次日(工作日)
    xmas_boxing_days = []
    new_year_days = []
    for year in range(2012, 2023):
        # 节礼日(12.26)及次日(若为工作日)
        boxing_day = pd.Timestamp(year, 12, 26)
        xmas_boxing_days.append(boxing_day)
        if boxing_day.weekday() < 4:  # 若节礼日是周四及之前,次日为工作日
            xmas_boxing_days.append(boxing_day + pd.Timedelta(days=1))
        # 新年元旦(1.1)及次日(若为工作日)
        new_year = pd.Timestamp(year, 1, 1)
        new_year_days.append(new_year)
        if new_year.weekday() < 4:
            new_year_days.append(new_year + pd.Timedelta(days=1))
    
    df.loc[df.index.isin(xmas_boxing_days + new_year_days), 'rank'] = 2

    # 等级3:圣诞前的工作日(12.24-12.31之间的工作日,排除已设等级的日期)
    pre_xmas_workdays = (
        (df.index.month == 12) & 
        (df.index.day >= 24) & 
        (df.index.day <= 31) & 
        (df.index.weekday() < 5)
    )
    df.loc[pre_xmas_workdays & (df['rank'] == 10), 'rank'] = 3

    # 等级5:圣诞-新年窗口(12.24-次年1.2),优先级低于已设置的等级
    holiday_window = pd.Series(False, index=df.index)
    for year in range(2012, 2023):
        window_start = pd.Timestamp(year, 12, 24)
        window_end = pd.Timestamp(year + 1, 1, 2)
        holiday_window |= (df.index >= window_start) & (df.index <= window_end)
    df.loc[holiday_window & (df['rank'] == 10), 'rank'] = 5

    # 按文档补充其他等级规则(如复活节、银行假期等对应的等级)
    return df

3. 构建通用规则框架,适配不同国家

把规则抽象成配置字典,切换国家时只需修改配置,无需改动核心逻辑:

# 英国假期等级配置(示例,需补充完整1-16级)
UK_HOLIDAY_RANKS = [
    {
        'rank': 1,
        'condition': lambda df: df['holiday'] == 'Christmas Day'
    },
    {
        'rank': 2,
        'condition': lambda df: df.index.isin([
            pd.Timestamp(y,12,26) for y in range(2012,2023)
        ] + [
            pd.Timestamp(y,12,27) for y in range(2012,2023) if pd.Timestamp(y,12,27).weekday() <5
        ] + [
            pd.Timestamp(y,1,1) for y in range(2012,2023)
        ] + [
            pd.Timestamp(y,1,2) for y in range(2012,2023) if pd.Timestamp(y,1,2).weekday() <5
        ])
    },
    # 其他等级配置...
]

def apply_rank_rules(df, rank_config):
    df['rank'] = 10  # 默认等级
    # 按等级数字从小到大应用(数字越小优先级越高)
    for rule in sorted(rank_config, key=lambda x: x['rank']):
        mask = rule['condition'](df)
        df.loc[mask, 'rank'] = rule['rank']
    return df

完整调用流程

import holidays
import numpy as np
import pandas as pd

# 生成日期数据集
dates = pd.date_range(start='2012-01-01', end='2022-12-31', freq='D')
df = pd.DataFrame({'Value': np.random.rand(len(dates))}, index=dates)

# 获取英国假期(过滤北爱尔兰专属假期)
df = get_british_holidays(df)

# 应用等级规则
df = apply_rank_rules(df, UK_HOLIDAY_RANKS)

# 查看结果示例
print(df[['holiday', 'rank']].sample(10))

内容的提问来源于stack exchange,提问作者13sen1

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.07.25 03:07:05