Python爬取YouTube数据时video_df返回空Dataframe的排查与解决
问题分析与解决方法
核心原因:get_video_details函数无返回值
你的video_df为空的直接原因是get_video_details函数收集完视频数据后没有返回结果,调用该函数时得到的是None,自然无法向video_df添加任何数据。
具体修正步骤
1. 给get_video_details添加返回语句
在函数末尾添加返回逻辑,将收集到的视频数据转换成DataFrame返回:
def get_video_details(youtube, video_ids): # 原有代码保持不变... all_video_info = [] # 原有循环收集数据的代码... # 新增返回语句 return pd.DataFrame(all_video_info)
2. 修正API字段拼写错误
YouTube API使用美式拼写,你代码里的favouriteCount要改成favoriteCount,否则该字段会始终返回None:
stats_to_keep = {'snippet': ['channelTitle', 'title', 'description', 'tags', 'publishedAt'], 'statistics': ['viewCount', 'likeCount', 'favoriteCount', 'commentCount'], # 修正拼写 'contentDetails': ['duration', 'definition', 'caption'] }
3. 替换过时的append方法(推荐)
Pandas的append方法已被弃用,建议用pd.concat合并DataFrame:
# 原代码 video_df = video_df.append(video_data, ignore_index=True) comments_df = comments_df.append(comments_data, ignore_index=True) # 替换为 video_df = pd.concat([video_df, video_data], ignore_index=True) comments_df = pd.concat([comments_df, comments_data], ignore_index=True)
4. 添加Excel导出代码
在获取完数据后,添加以下代码将结果导出到Excel:
# 导出视频数据 video_df.to_excel('youtube_video_stats.xlsx', index=False) # 导出评论数据(可选) comments_df.to_excel('youtube_comments.xlsx', index=False) print("数据已成功导出至Excel文件")
完整修正后的代码
import pandas as pd import numpy as np from googleapiclient.discovery import build api_key = '<api key>' channel_ids = ['<channel_id>',] youtube = build('youtube', 'v3', developerKey=api_key) def get_channel_stats(youtube, channel_ids): all_data = [] request = youtube.channels().list( part='snippet,contentDetails,statistics', id=','.join(channel_ids)) response = request.execute() for i in range(len(response['items'])): data = dict(channelName=response['items'][i]['snippet']['title'], subscribers=response['items'][i]['statistics']['subscriberCount'], views=response['items'][i]['statistics']['viewCount'], totalVideos=response['items'][i]['statistics']['videoCount'], playlistId=response['items'][i]['contentDetails']['relatedPlaylists']['uploads']) all_data.append(data) return pd.DataFrame(all_data) def get_video_ids(youtube, playlist_id): request = youtube.playlistItems().list( part='contentDetails', playlistId=playlist_id, maxResults=50) response = request.execute() video_ids = [] for i in range(len(response['items'])): video_ids.append(response['items'][i]['contentDetails']['videoId']) next_page_token = response.get('nextPageToken') more_pages = True while more_pages: if next_page_token is None: more_pages = False else: request = youtube.playlistItems().list( part='contentDetails', playlistId=playlist_id, maxResults=50, pageToken=next_page_token) response = request.execute() for i in range(len(response['items'])): video_ids.append(response['items'][i]['contentDetails']['videoId']) next_page_token = response.get('nextPageToken') return video_ids def get_video_details(youtube, video_ids): all_video_info = [] for i in range(0, len(video_ids), 50): request = youtube.videos().list( part="snippet,contentDetails,statistics", id=','.join(video_ids[i:i + 50]) ) response = request.execute() for video in response['items']: stats_to_keep = {'snippet': ['channelTitle', 'title', 'description', 'tags', 'publishedAt'], 'statistics': ['viewCount', 'likeCount', 'favoriteCount', 'commentCount'], 'contentDetails': ['duration', 'definition', 'caption'] } video_info = {} video_info['video_id'] = video['id'] for k in stats_to_keep.keys(): for v in stats_to_keep[k]: try: video_info[v] = video[k][v] except: video_info[v] = None all_video_info.append(video_info) # 新增返回语句 return pd.DataFrame(all_video_info) def get_comments_in_videos(youtube, video_ids): all_comments = [] for video_id in video_ids: try: request = youtube.commentThreads().list( part="snippet,replies", videoId=video_id ) response = request.execute() comments_in_video = [comment['snippet']['topLevelComment']['snippet']['textOriginal'] for comment in response['items'][0:10]] comments_in_video_info = {'video_id': video_id, 'comments': comments_in_video} all_comments.append(comments_in_video_info) except: print('Could not get comments for video ' + video_id) return pd.DataFrame(all_comments) channel_data = get_channel_stats(youtube, channel_ids) numeric_cols = ['subscribers', 'views', 'totalVideos'] channel_data[numeric_cols] = channel_data[numeric_cols].apply(pd.to_numeric, errors='coerce') video_df = pd.DataFrame() comments_df = pd.DataFrame() for c in channel_data['channelName'].unique(): print("Getting video information from channel: " + c) playlist_id = channel_data.loc[channel_data['channelName'] == c, 'playlistId'].iloc[0] video_ids = get_video_ids(youtube, playlist_id) video_data = get_video_details(youtube, video_ids) comments_data = get_comments_in_videos(youtube, video_ids) # 替换为pd.concat video_df = pd.concat([video_df, video_data], ignore_index=True) comments_df = pd.concat([comments_df, comments_data], ignore_index=True) print(video_df) # 导出Excel video_df.to_excel('youtube_video_stats.xlsx', index=False) comments_df.to_excel('youtube_comments.xlsx', index=False) print("数据已成功导出至Excel文件")
内容的提问来源于stack exchange,提问作者youtube_tryhard
相关产品推荐
相关产品推荐

