访问DataFrame列众数时触发KeyError问题求助
美国共享单车数据分析代码错误修复
问题描述
运行以下美国共享单车数据分析Python代码时,出现KeyError: 0错误,无法完成月份、星期、小时的统计分析:
import time import pandas as pd import numpy as np CITY_DATA = {'chicago': 'chicago.csv', 'new york city': 'new_york_city.csv', 'washington': 'washington.csv'} def get_filters(): """ Asks user to specify a city, month, and day to analyze. Returns: (str) city - name of the city to analyze (str) month - name of the month to filter by, or "all" to apply no month filter (str) day - name of the day of week to filter by, or "all" to apply no day filter """ print('Hello! Let\'s explore some US bikeshare data!') # get user input for city (chicago, new york city, washington). HINT: Use a while loop to handle invalid inputs while True: city = input('Which city you would like to explore : "chicago" , "new york city" , or "washington" :' ) if city not in ('chicago', 'new york city', 'washington'): print(" You entered wrong choice , please try again") continue else: break # get user input for month (all, january, february, ... , june) while True: month = input('Enter "all" for all data or chose a month : "january" , "february" , "march", "april" , "may" or "june " :') if month not in ("all", "january", "february", "march", "april", "may", "june"): print(" You entered wrong choice , please try again") continue else: break # get user input for day of week (all, monday, tuesday, ... sunday) while True: day = input('Enter "all" for all days or chose a day : "saturday", "sunday", "monday", "tuesday", "wednesday", "thursday", "friday": ') if day not in ("all","saturday", "sunday", "monday", "tuesday", "wednesday", "thursday", "friday"): print(" You entered wrong choice , please try again") continue else: break print('-'*60) return city, month, day def load_data(city, month, day): """ Loads data for the specified city and filters by month and day if applicable. Args: (str) city - name of the city to analyze (str) month - name of the month to filter by, or "all" to apply no month filter (str) day - name of the day of week to filter by, or "all" to apply no day filter Returns: df - Pandas DataFrame containing city data filtered by month and day """ df = pd.read_csv(CITY_DATA[city]) # convert the Start Time column to datetime df['Start Time'] = pd.to_datetime(df['Start Time']) # extract month , day of week , and hour from Start Time to new columns df['month'] = df['Start Time'].dt.month df['day_of_week'] = df['Start Time'].dt.day_name df['hour'] = df['Start Time'].dt.hour # filter by month if applicable if month != 'all': # use the index of the month_list to get the corresponding int months = ['january', 'february', 'march', 'april', 'may', 'june'] month = months.index(month) + 1 # filter by month to create the new dataframe df = df[df['month'] == month] # filter by day of week if applicable if day != 'all': # filter by day of week to create the new dataframe df = df[df['day_of_week'] == day.title()] return df def time_stats(df): """Displays statistics on the most frequent times of travel.""" print('\nCalculating The Most Frequent Times of Travel...\n') start_time = time.time() # display the most common month popular_month = df['month'].mode()[0] print('\n The most popular month is : \n', popular_month) # display the most common day of week popular_day = df['day_of_week'].mode()[0] print('\n The most popular day of the week is : \n', str(popular_day)) # display the most common start hour popular_hour = df['hour'].mode()[0] print('\n The most popular hour of the day is :\n ', popular_hour) print("\nThis took %s seconds.\n" % (time.time() - start_time)) print('-'*60) def station_stats(df): """Displays statistics on the most popular stations and trip.""" print('\nCalculating The Most Popular Stations and Trip...\n') start_time = time.time() # display most commonly used start station start_station = df['Start Station'].value_counts().idxmax() print('\n The most commonly used start station is : \n', start_station) # display most commonly used end station end_station = df['End Station'].value_counts().idxmax() print('\nThe most commonly used end station is: \n', end_station) # display most frequent combination of start station and end station trip combination = df.groupby(['Start Station','End Station']).value_counts().idxmax() print('\nThe most frequent combination of start station and end station are: \n', combination) print("\nThis took %s seconds." % (time.time() - start_time)) print('-'*40) def trip_duration_stats(df): """Displays statistics on the total and average trip duration.""" start_time = time.time() travel_time = sum(df['Trip Duration']) print('Total travel time:', travel_time / 86400, " Days") # display total travel time total_time = sum(df['Trip Duration']) print('\nThe total travel time is {} seconds: \n', total_time) # display mean travel time mean_time = df['Trip Duration'].mean() print('\n The average travel time is \n', mean_time) print("\nThis took %s seconds." % (time.time() - start_time)) print('-'*40) def user_stats(df): """Displays statistics on bikeshare users.""" print('\nCalculating User Stats...\n') start_time = time.time() # TO DO: Display counts of user types user_types = df['User Type'].value_counts() #print(user_types) print('User Types:\n', user_types) # TO DO: Display counts of gender print("\nThis took %s seconds." % (time.time() - start_time)) print('-'*40) def main(): while True: city, month, day = get_filters() df = load_data(city, month, day) time_stats(df) station_stats(df) trip_duration_stats(df) user_stats(df) restart = input('\nWould you like to restart? Enter yes or no.\n') if restart.lower() != 'yes': break if __name__ == "__main__": main()
错误信息
> Traceback (most recent call last): File "C:\Users\DELL\PycharmProjects\Professional\venv\Lib\site-packages\pandas\core\indexes\range.py", line 391, in get_loc return self._range.index(new_key) ^^^^^^^^^^^^^^^^^^^^^^^^^^ ValueError: 0 is not in range The above exception was the direct cause of the following exception: Traceback (most recent call last): File "C:\Users\DELL\PycharmProjects\Professional\Bikeshare.py", line 203, in <module> main() File "C:\Users\DELL\PycharmProjects\Professional\Bikeshare.py", line 192, in main time_stats(df) File "C:\Users\DELL\PycharmProjects\Professional\Bikeshare.py", line 100, in time_stats popular_month = df['month'].mode()[0] ~~~~~~~~~~~~~~~~~~^^^ File "C:\Users\DELL\PycharmProjects\Professional\venv\Lib\site-packages\pandas\core\series.py", line 981, in __getitem__ Calculating The Most Frequent Times of Travel... return self._get_value(key) ^^^^^^^^^^^^^^^^^^^^ File "C:\Users\DELL\PycharmProjects\Professional\venv\Lib\site-packages\pandas\core\series.py", line 1089, in _get_value loc = self.index.get_loc(label) ^^^^^^^^^^^^^^^^^^^^^^^^^ File "C:\Users\DELL\PycharmProjects\Professional\venv\Lib\site-packages\pandas\core\indexes\range.py", line 393, in get_loc raise KeyError(key) from err KeyError: 0
错误原因分析
- 空DataFrame访问错误:当用户选择的月份/日期组合在数据中没有匹配记录时,过滤后的
df为空,此时df['month'].mode()返回空Series,尝试取索引[0]会触发KeyError。 - day_name方法调用错误:
df['Start Time'].dt.day_name是方法对象,未加括号()调用,导致day_of_week列存储的是方法而非实际星期名称,后续过滤和统计完全失效。 - 重复计算与格式化错误:
trip_duration_stats中重复计算总旅行时间,且字符串格式化语法错误(print('\nThe total travel time is {} seconds: \n', total_time)应改为print('\nThe total travel time is {} seconds: \n'.format(total_time)))。 - 未处理缺失列情况:
user_stats中未完成性别统计代码,且未考虑Washington数据没有Gender和Birth Year列的情况。
修复后的完整代码
import time import pandas as pd import numpy as np CITY_DATA = {'chicago': 'chicago.csv', 'new york city': 'new_york_city.csv', 'washington': 'washington.csv'} def get_filters(): """ Asks user to specify a city, month, and day to analyze. Returns: (str) city - name of the city to analyze (str) month - name of the month to filter by, or "all" to apply no month filter (str) day - name of the day of week to filter by, or "all" to apply no day filter """ print('Hello! Let\'s explore some US bikeshare data!') # 获取城市输入 while True: city = input('Which city you would like to explore : "chicago" , "new york city" , or "washington" :' ).lower() if city not in CITY_DATA: print(" You entered wrong choice , please try again") continue else: break # 获取月份输入 while True: month = input('Enter "all" for all data or chose a month : "january" , "february" , "march", "april" , "may" or "june" :').lower() if month not in ("all", "january", "february", "march", "april", "may", "june"): print(" You entered wrong choice , please try again") continue else: break # 获取星期输入 while True: day = input('Enter "all" for all days or chose a day : "saturday", "sunday", "monday", "tuesday", "wednesday", "thursday", "friday": ').lower() if day not in ("all","saturday", "sunday", "monday", "tuesday", "wednesday", "thursday", "friday"): print(" You entered wrong choice , please try again") continue else: break print('-'*60) return city, month, day def load_data(city, month, day): """ Loads data for the specified city and filters by month and day if applicable. Args: (str) city - name of the city to analyze (str) month - name of the month to filter by, or "all" to apply no month filter (str) day - name of the day of week to filter by, or "all" to apply no day filter Returns: df - Pandas DataFrame containing city data filtered by month and day """ df = pd.read_csv(CITY_DATA[city]) # 转换Start Time为datetime类型 df['Start Time'] = pd.to_datetime(df['Start Time']) # 提取月份、星期、小时到新列 df['month'] = df['Start Time'].dt.month # 修复day_name调用:添加括号执行方法 df['day_of_week'] = df['Start Time'].dt.day_name() df['hour'] = df['Start Time'].dt.hour # 按月份过滤 if month != 'all': months = ['january', 'february', 'march', 'april', 'may', 'june'] month = months.index(month) + 1 df = df[df['month'] == month] # 按星期过滤 if day != 'all': df = df[df['day_of_week'] == day.title()] return df def time_stats(df): """Displays statistics on the most frequent times of travel.""" print('\nCalculating The Most Frequent Times of Travel...\n') start_time = time.time() # 检查DataFrame是否为空 if df.empty: print("No data available for the selected filters.") print("\nThis took %s seconds.\n" % (time.time() - start_time)) print('-'*60) return # 最常见月份 popular_month = df['month'].mode()[0] month_names = ['January', 'February', 'March', 'April', 'May', 'June'] print(f'\n The most popular month is : \n {month_names[popular_month-1]}') # 最常见星期 popular_day = df['day_of_week'].mode()[0] print(f'\n The most popular day of the week is : \n {popular_day}') # 最常见小时 popular_hour = df['hour'].mode()[0] print(f'\n The most popular hour of the day is :\n {popular_hour}:00') print("\nThis took %s seconds.\n" % (time.time() - start_time)) print('-'*60) def station_stats(df): """Displays statistics on the most popular stations and trip.""" print('\nCalculating The Most Popular Stations and Trip...\n') start_time = time.time() if df.empty: print("No data available for the selected filters.") print("\nThis took %s seconds." % (time.time() - start_time)) print('-'*40) return # 最常用起始站 start_station = df['Start Station'].value_counts().idxmax() print(f'\n The most commonly used start station is : \n {start_station}') # 最常用终点站 end_station = df['End Station'].value_counts().idxmax() print(f'\nThe most commonly used end station is: \n {end_station}') # 最常见的起始-终点组合 # 用size()更高效统计组合次数 combination_counts = df.groupby(['Start Station', 'End Station']).size() most_common_combination = combination_counts.idxmax() print(f'\nThe most frequent combination of start station and end station are: \n {most_common_combination[0]} -> {most_common_combination[1]}') print("\nThis took %s seconds." % (time.time() - start_time)) print('-'*40) def trip_duration_stats(df): """Displays statistics on the total and average trip duration.""" start_time = time.time() if df.empty: print("No data available for the selected filters.") print("\nThis took %s seconds." % (time.time() - start_time)) print('-'*40) return # 总旅行时间(天和秒) total_time = df['Trip Duration'].sum() print(f'Total travel time: {total_time / 86400:.2f} Days') print(f'\nThe total travel time is {total_time
相关产品推荐
相关产品推荐

