Ubuntu下如何自动化多终端执行Python批量删除在线应用条目任务?
问题描述
我有一个存储需从在线应用删除的所有条目的文本文件,每个待删除条目需单独发送删除请求。为加快删除速度,我将原文件拆分为多个文本文件,在约130个终端中运行删除脚本(处理约7000条条目耗时可控制在30分钟内)。以下是删除脚本代码:
from fileinput import filename from WitApiClient import WitApiClient import os dirname = os.path.dirname(__file__) parent_dirname = os.path.dirname(dirname) token = input("Enter the token") file_name = os.path.join(parent_dirname, 'data/deletion_pair.txt') with open(file_name, encoding="utf-8") as file: templates = [line.strip() for line in file.readlines()] for template in templates: entity, keyword = template.split(", ") print(entity, keyword) resp = WitApiClient(token).delete_keyword(entity, keyword) print(resp)
请问是否有办法自动化该流程或采用更高效的实现方式?
解决方案
一、自动化流程(无需手动拆分文件、开启终端)
1. 用并发池批量处理(推荐)
不需要手动拆分文件和打开多个终端,直接在单脚本中用多进程/线程池实现并发请求,自动读取原文件的所有条目。示例代码如下:
from WitApiClient import WitApiClient import os from concurrent.futures import ThreadPoolExecutor, as_completed def delete_entry(token, entry, fail_file): try: entity, keyword = entry.strip().split(", ") resp = WitApiClient(token).delete_keyword(entity, keyword) return f"Success: {entity}, {keyword} - {resp}" except Exception as e: # 记录失败条目,方便后续重试 with open(fail_file, "a", encoding="utf-8") as ff: ff.write(entry) return f"Failed: {entry.strip()} - {str(e)}" if __name__ == "__main__": token = input("Enter the token: ") dirname = os.path.dirname(__file__) parent_dirname = os.path.dirname(dirname) file_name = os.path.join(parent_dirname, 'data/deletion_pair.txt') fail_file = os.path.join(parent_dirname, 'data/failed_deletions.txt') # 清空历史失败记录 with open(fail_file, "w", encoding="utf-8"): pass # 读取所有有效条目 with open(file_name, encoding="utf-8") as file: entries = [line for line in file if line.strip()] # 用线程池处理(API请求属于IO密集型,线程池更省资源) # 并发数根据API限流规则调整,比如设为30-50 with ThreadPoolExecutor(max_workers=40) as executor: futures = [executor.submit(delete_entry, token, entry, fail_file) for entry in entries] for future in as_completed(futures): print(future.result())
- 如果WitApiClient存在CPU密集型逻辑,可替换为
ProcessPoolExecutor,否则线程池性能更优。
2. 自动拆分文件+批量启动脚本(兼容原有多终端模式)
写一个辅助脚本自动拆分大文件,并批量启动终端运行删除脚本:
import os import subprocess def split_file(input_file, num_chunks): with open(input_file, encoding="utf-8") as f: lines = [line for line in f if line.strip()] chunk_size = len(lines) // num_chunks + 1 chunk_files = [] for i in range(num_chunks): chunk_lines = lines[i*chunk_size : (i+1)*chunk_size] if not chunk_lines: break chunk_path = os.path.join(os.path.dirname(input_file), f"deletion_pair_{i}.txt") with open(chunk_path, "w", encoding="utf-8") as cf: cf.writelines(chunk_lines) chunk_files.append(chunk_path) return chunk_files if __name__ == "__main__": token = input("Enter the token: ") dirname = os.path.dirname(__file__) parent_dirname = os.path.dirname(dirname) input_file = os.path.join(parent_dirname, 'data/deletion_pair.txt') num_chunks = 130 # 拆分文件 chunk_files = split_file(input_file, num_chunks) # 批量启动脚本(Windows示例,Linux/macOS替换为bash命令) delete_script = os.path.join(dirname, "你的删除脚本名.py") for chunk in chunk_files: subprocess.Popen(f"cmd /k python {delete_script} {chunk} {token}", shell=True)
同时修改原删除脚本,支持命令行传参(无需手动输入token和文件名):
from WitApiClient import WitApiClient import os import sys if __name__ == "__main__": if len(sys.argv) != 3: print("Usage: python delete_script.py <chunk_file> <token>") sys.exit(1) chunk_file = sys.argv[1] token = sys.argv[2] with open(chunk_file, encoding="utf-8") as file: templates = [line.strip() for line in file.readlines()] for template in templates: entity, keyword = template.split(", ") print(entity, keyword) resp = WitApiClient(token).delete_keyword(entity, keyword) print(resp)
二、更高效的优化方向
1. 加入请求重试机制
针对网络波动、API限流导致的失败,自动重试避免重复手动处理:
from tenacity import retry, stop_after_attempt, wait_exponential, retry_if_exception_type import requests # 针对请求异常重试,最多3次,间隔指数增长 @retry(stop=stop_after_attempt(3), wait=wait_exponential(multiplier=1, min=2, max=10), retry=retry_if_exception_type((requests.exceptions.RequestException,))) def delete_entry(token, entry, fail_file): entity, keyword = entry.strip().split(", ") resp = WitApiClient(token).delete_keyword(entity, keyword) return f"Success: {entity}, {keyword} - {resp}"
需先安装依赖:pip install tenacity
2. 严格控制并发数,适配API限流
先查阅Wit API的限流规则(比如每秒允许的请求数),调整并发池的max_workers值,避免触发429限流错误,反而拖慢整体速度。
3. 异步请求(IO密集型场景最优)
用异步HTTP客户端替代同步请求,资源占用更低、处理速度更快:
import aiohttp import asyncio import os async def delete_entry(session, token, entry, fail_file): try: entity, keyword = entry.strip().split(", ") # 替换为Wit API实际的删除接口地址 url = f"https://api.wit.ai/entities/{entity}/keywords/{keyword}" headers = {"Authorization": f"Bearer {token}"} async with session.delete(url, headers=headers) as resp: result = await resp.json() return f"Success: {entity}, {keyword} - {result}" except Exception as e: with open(fail_file, "a", encoding="utf-8") as ff: ff.write(entry) return f"Failed: {entry.strip()} - {str(e)}" async def main(): token = input("Enter the token: ") dirname = os.path.dirname(__file__) parent_dirname = os.path.dirname(dirname) file_name = os.path.join(parent_dirname, 'data/deletion_pair.txt') fail_file = os.path.join(parent_dirname, 'data/failed_deletions.txt') with open(fail_file, "w", encoding="utf-8"): pass with open(file_name, encoding="utf-8") as file: entries = [line for line in file if line.strip()] # 用信号量控制并发数 semaphore = asyncio.Semaphore(50) async def bounded_delete(session, token, entry, fail_file): async with semaphore: return await delete_entry(session, token, entry, fail_file) async with aiohttp.ClientSession() as session: tasks = [bounded_delete(session, token, entry, fail_file) for entry in entries] results = await asyncio.gather(*tasks) for result in results: print(result) if __name__ == "__main__": asyncio.run(main())
内容的提问来源于stack exchange,提问作者Mritunjay Prasad
相关产品推荐
相关产品推荐

