如何从ipywidgets输出中获取Pandas DataFrame?
嘿Tom,我懂你现在的困扰——你能在Jupyter的Output组件里看到过滤后的表格,但就是没办法把这些数据提取出来当成普通的Pandas DataFrame来用,对吧?其实问题很简单,咱们只要把过滤后的结果存到一个外部能访问的地方就行,给你两种靠谱的解决方案:
方案一:用全局变量快速实现(适合简单场景)
原代码里的过滤结果是common_filtering函数里的局部变量,函数执行完就会被销毁,所以外部访问不到。咱们只需要在函数外定义一个全局变量,每次过滤后把结果赋值给它就行:
import pandas as pd import numpy as np import ipywidgets as widgets from ipywidgets import Layout, AppLayout from IPython.display import display import functools data = {'year': ['2000', '2000','2000','2000','2001','2001','2001','2001', '2002', '2002', '2002', '2002', '2003','2003','2003','2003','2004', '2004','2004','2004', '2005', '2005', '2005', '2005', '2006', '2006', '2006', '2006', '2006', '2007', '2007', '2007', '2007', '2008', '2008', '2008', '2008', '2009', '2009', '2009', '2009'], 'purpose':['Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday' ], 'market':['Belgium', 'Luxembourg', 'France', 'Spain', 'Norway', 'Sweden', 'Germany', 'Austria', 'Denmark', 'Portugal', 'Greece', 'Croatia', 'Belgium', 'Luxembourg', 'France', 'Spain', 'Norway', 'Sweden', 'Germany', 'Austria', 'Denmark', 'Portugal', 'Greece', 'Croatia', 'Belgium', 'Luxembourg', 'France', 'Spain', 'Norway', 'Sweden', 'Germany', 'Austria', 'Denmark', 'Portugal', 'Greece', 'Croatia', 'Belgium', 'Luxembourg', 'France', 'Spain', 'Norway' ]} df_london = pd.DataFrame (data, columns = ['year','purpose', 'market']) # Get our unique values ALL = 'ALL' def unique_sorted_values_plus_ALL(array): unique = array.unique().tolist() unique.sort() unique.insert(0, ALL) return unique output = widgets.Output() # 定义全局变量存储过滤后的结果 filtered_df = df_london.copy() # Dropdown listbox dropdown_year = widgets.Dropdown(description='Year', options = unique_sorted_values_plus_ALL(df_london.year)) # Function to filter our dropdown listboxe def common_filtering(year): global filtered_df # 声明要使用全局变量 df = df_london.copy() filters = [] # Evaluate our dropdown listbox and return booleans for our selections if year is not ALL: filters.append(df['year'] == year) output.clear_output() with output: if filters: df_filter = functools.reduce(lambda x,y: x&y, filters) filtered_df = df.loc[df_filter] # 将过滤结果赋值给全局变量 display(filtered_df) else: filtered_df = df # 未过滤时返回原数据 display(df) def dropdown_year_eventhandler(change): common_filtering(change.new) dropdown_year.observe(dropdown_year_eventhandler, names='value') ui = widgets.HBox([dropdown_year]) display(ui, output)
现在你只要在Jupyter单元格里输入filtered_df,就能直接拿到最新的过滤结果,还能对它做任何Pandas操作,比如filtered_df.groupby('purpose').size()。
方案二:用类封装(适合复杂多过滤器场景)
如果之后你要加更多过滤器(比如purpose、market的下拉框),用全局变量会显得混乱。这时候用类来封装状态会更优雅:
import pandas as pd import numpy as np import ipywidgets as widgets from IPython.display import display import functools data = {'year': ['2000', '2000','2000','2000','2001','2001','2001','2001', '2002', '2002', '2002', '2002', '2003','2003','2003','2003','2004', '2004','2004','2004', '2005', '2005', '2005', '2005', '2006', '2006', '2006', '2006', '2006', '2007', '2007', '2007', '2007', '2008', '2008', '2008', '2008', '2009', '2009', '2009', '2009'], 'purpose':['Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday', 'Business', 'VFR', 'Study', 'Holiday' ], 'market':['Belgium', 'Luxembourg', 'France', 'Spain', 'Norway', 'Sweden', 'Germany', 'Austria', 'Denmark', 'Portugal', 'Greece', 'Croatia', 'Belgium', 'Luxembourg', 'France', 'Spain', 'Norway', 'Sweden', 'Germany', 'Austria', 'Denmark', 'Portugal', 'Greece', 'Croatia', 'Belgium', 'Luxembourg', 'France', 'Spain', 'Norway', 'Sweden', 'Germany', 'Austria', 'Denmark', 'Portugal', 'Greece', 'Croatia', 'Belgium', 'Luxembourg', 'France', 'Spain', 'Norway' ]} df_london = pd.DataFrame (data, columns = ['year','purpose', 'market']) ALL = 'ALL' def unique_sorted_values_plus_ALL(array): unique = array.unique().tolist() unique.sort() unique.insert(0, ALL) return unique class DataFilter: def __init__(self, df): self.raw_df = df self.filtered_df = df.copy() # 创建控件 self.dropdown_year = widgets.Dropdown(description='Year', options=unique_sorted_values_plus_ALL(df.year)) self.output = widgets.Output() # 绑定事件 self.dropdown_year.observe(self._on_year_change, names='value') # 初始化显示 self._common_filtering(self.dropdown_year.value) def _common_filtering(self, year): df = self.raw_df.copy() filters = [] if year is not ALL: filters.append(df['year'] == year) self.output.clear_output() with self.output: if filters: df_filter = functools.reduce(lambda x,y: x&y, filters) self.filtered_df = df.loc[df_filter] display(self.filtered_df) else: self.filtered_df = df display(df) def _on_year_change(self, change): self._common_filtering(change.new) def show_ui(self): ui = widgets.HBox([self.dropdown_year]) display(ui, self.output) # 创建过滤器实例 filter_app = DataFilter(df_london) filter_app.show_ui()
这个方案里,过滤后的结果存在filter_app.filtered_df里,你随时可以调用它。后续要加新的过滤器,只要在类里新增控件和过滤逻辑就行,代码结构会非常清晰。
内容的提问来源于stack exchange,提问作者twelsh37
相关产品推荐
相关产品推荐

