如何修改TrackmaniaExchange搜索链接以生成地图下载链接
批量处理TrackmaniaExchange地图链接生成下载地址
我正在编写Python脚本,从TrackmaniaExchange网站的搜索结果批量下载地图。已知地图下载链接格式为https://trackmania.exchange/maps/download/(map id),但搜索结果的href格式为/maps/(map id)/(map name)。计划用Selenium获取href,通过正则修改链接并移除末尾的地图名,但不知具体实现方法。现有脚本如下:
import requests import os.path import os import selenium.webdriver as webdriver from selenium.webdriver.firefox.options import Options import time import re def Search(): link="https://trackmania.exchange/mapsearch2?limit=100" #Trackmania Exchange link, will scrape all 100 results checkedlink = re.sub("\s", "+", link) #Replaces spaces with + for track names (this shouldnt happen with authors/tags) options = Options() #This is for selenium options.binary_location = "C:/Program Files/Mozilla Firefox/firefox.exe" driver = webdriver.Firefox(options=options) search_box = driver.find_element_by_name("trackname") sitelinks = driver.find_element_by_xpath("/html/[div/@id='container'/@data-select2-id='container']/[div/@class='container-inner']/[div/@class='ly-box-open']/[div/@class='box-col-6']/[div/@class='windowv2-panel']/[div/@id='searchResults-container']/div/div/table/tbody/[tr/@class='WindowTableCell2v2 with-hover has-image']/[td/@class='cell-ellipsis']") results = [] name=input("Track Name (if nothing, hit enter)") #Prompts the user to input stuff author=input("Track Author (if nothing, hit enter)") tags=input("Tags (separate with %2C if there's multiple, if nothing, hit enter)") path=input("Map download directory (do not leave blank, use forward slashes)") print("WARNING: Download wget for this script to work.") type(name) #These are to make a link to find html with type(author) type(tags) type(path) if path == "": print("Please put a path next time you start this") time.sleep(3) os.exit() else: #And so begins the if/else hellhole to find out what needs to be added to the link if tags == "": if name == "": if author == "": print("Chief, you cant just enter nothing. Put something in here next time") time.sleep(3) os.exit() else: link = link+"&author="+author else: link = link+"&trackname="+name if author != "": link = link+"&author="+author else: link = link+"&tags="+tags if name != "": link = link+"&trackname="+name if author != "": link = link+"&author="+author else: if author != "": link = link+"&author="+author print("Checking link...") checkedlink() #this is to make sure there's no spaces in the link. tags are separated by %2C, but track names are separated by + print("Attempting to download...") driver.get(link) links = sitelinks for link in links href = link.get_attribute("href") browser.close() with open("list.txt", "w", encoding="utf-8") as f: f.write(href) for line in f: h = re.findall("\d") #My failed attempt at removing the end of the link re.sub("/maps/", "https://trackmania.exchange/maps/download", f) re.sub("") #unfinished part cause i was stubbed os.system("wget --directory-prefix="path" -i list.txt") Search()
解决方案
1. 核心:生成下载链接的方法
不需要复杂的字符串替换,直接从href中提取地图ID,再拼接成标准下载链接。用正则表达式可以快速捕获ID:
def convert_to_download_link(href): # 匹配/maps/后面的数字ID match_result = re.search(r'/maps/(\d+)/', href) if match_result: map_id = match_result.group(1) return f"https://trackmania.exchange/maps/download/{map_id}" # 匹配失败时返回None return None
比如输入/maps/91677/cloudy-day,会返回https://trackmania.exchange/maps/download/91677。
2. 修正脚本中的其他关键错误
原脚本存在语法、逻辑错误,以下是修正后的完整可运行脚本:
import os import time import re import selenium.webdriver as webdriver from selenium.webdriver.firefox.options import Options from selenium.webdriver.common.by import By from selenium.webdriver.support.ui import WebDriverWait from selenium.webdriver.support import expected_conditions as EC def convert_to_download_link(href): match_result = re.search(r'/maps/(\d+)/', href) if match_result: map_id = match_result.group(1) return f"https://trackmania.exchange/maps/download/{map_id}" return None def Search(): base_link = "https://trackmania.exchange/mapsearch2?limit=100" options = Options() options.binary_location = "C:/Program Files/Mozilla Firefox/firefox.exe" driver = webdriver.Firefox(options=options) # 获取用户输入 name = input("Track Name (if nothing, hit enter): ") author = input("Track Author (if nothing, hit enter): ") tags = input("Tags (separate with %2C if multiple, if nothing, hit enter): ") path = input("Map download directory (do not leave blank, use forward slashes): ") print("WARNING: Download wget for this script to work.") # 校验路径 if not path: print("Please provide a valid path next time.") time.sleep(3) driver.quit() os.exit() # 构建搜索链接 params = [] if name: params.append(f"trackname={re.sub(r'\s', '+', name)}") if author: params.append(f"author={author}") if tags: params.append(f"tags={tags}") if not params: print("You must enter at least one search criteria.") time.sleep(3) driver.quit() os.exit() final_link = f"{base_link}&{'&'.join(params)}" print(f"Checking link: {final_link}") # 加载搜索页面并等待结果加载 driver.get(final_link) try: WebDriverWait(driver, 10).until( EC.presence_of_element_located((By.XPATH, "//table[@id='searchResults']//td[@class='cell-ellipsis']/a")) ) except: print("Failed to load search results.") driver.quit() return # 获取所有地图链接并转换为下载链接 map_links = driver.find_elements(By.XPATH, "//table[@id='searchResults']//td[@class='cell-ellipsis']/a") download_links = [] for link in map_links: href = link.get_attribute("href") download_link = convert_to_download_link(href) if download_link: download_links.append(download_link) # 写入下载链接到文件 with open("list.txt", "w", encoding="utf-8") as f: f.write("\n".join(download_links)) # 调用wget批量下载 os.system(f'wget --directory-prefix="{path}" -i list.txt') # 关闭浏览器 driver.quit() print("Download process completed.") if __name__ == "__main__": Search()
关键修正点说明
- 用
WebDriverWait等待页面加载,避免元素未找到的错误 - 修正XPATH选择器,正确定位搜索结果中的地图链接
- 重构搜索链接构建逻辑,替代嵌套if/else的复杂写法
- 修复文件写入逻辑,一次性写入所有下载链接
- 修正循环语法、函数调用等基础错误
内容的提问来源于stack exchange,提问作者j0cffex2
相关产品推荐
相关产品推荐

