Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
19 changes: 10 additions & 9 deletions buckaroo/buckaroo_widget.py
Original file line number Diff line number Diff line change
Expand Up @@ -58,16 +58,16 @@ def _bk_flash(event, **extra):

class PdSampling(Sampling):
@classmethod
def pre_stats_sample(kls, df):
def pre_stats_sample(cls, df):
# this is a bad place for fixing the dataframe, but for now
# it's expedient. There probably should be a nother processing
# step
df = check_and_fix_df(df)
if len(df.columns) > kls.max_columns:
if len(df.columns) > cls.max_columns:
print("Removing excess columns, found %d columns" % len(df.columns))
df = df[df.columns[:kls.max_columns]]
if kls.pre_limit and len(df) > kls.pre_limit:
sampled = df.sample(kls.pre_limit)
df = df[df.columns[:cls.max_columns]]
if cls.pre_limit and len(df) > cls.pre_limit:
sampled = df.sample(cls.pre_limit)
if isinstance(sampled, pd.DataFrame):
return sampled.sort_index()
return sampled
Expand Down Expand Up @@ -115,7 +115,7 @@ def get_story_config(self, include_summary_stats=False, test_name=None) -> str:
'secondary_df_viewer_config': EMPTY_DFVIEWER_CONFIG}}
args_dict['args']['summary_stats_data'] = []
if include_summary_stats:
1/0 # not supported yet
raise NotImplementedError("include_summary_stats isn't supported yet")
# summary_stats data is big, and most of the time you won't want to serialize it
#args_dict['summary_stats_data'] = {} #desrialize here

Expand Down Expand Up @@ -145,15 +145,16 @@ def __init__(self, orig_df, debug=False,
self.record_transcript = record_transcript
self.exception = None
kls = self.__class__
widget = self
class InnerDataFlow(kls.dataflow_klass):
sampling_klass = kls.sampling_klass
autocleaning_klass = kls.autocleaning_klass
DFStatsClass = kls.DFStatsClass
autoclean_conf= kls.autoclean_conf
analysis_klasses = kls.analysis_klasses

def _df_to_obj(idfself, df:pd.DataFrame):
return self._df_to_obj(df)
def _df_to_obj(self, df:pd.DataFrame):
return widget._df_to_obj(df)

self.dataflow = InnerDataFlow(
orig_df,
Expand Down Expand Up @@ -258,7 +259,7 @@ def add_processing(self, df_processing_func, /):
class DecoratedProcessing(ColAnalysis):
provides_defaults = {}
@classmethod
def post_process_df(kls, df):
def post_process_df(cls, df):
new_df = df_processing_func(df)
return [new_df, {}]
post_processing_method = proc_func_name
Expand Down
4 changes: 2 additions & 2 deletions buckaroo/customizations/styling.py
Original file line number Diff line number Diff line change
Expand Up @@ -69,7 +69,7 @@ class DefaultMainStyling(StylingAnalysis):


@classmethod
def style_column(kls, col:str, column_metadata: Any) -> Any:
def style_column(cls, col:str, column_metadata: Any) -> Any:
# `_type` is the discriminator for displayer choice. If column_metadata
# is empty (polars/index edge case) or arrives without `_type` — e.g.
# a no-cleaning run where only an op-contributed entry (highlight_*)
Expand Down Expand Up @@ -126,7 +126,7 @@ def style_column(kls, col:str, column_metadata: Any) -> Any:
header_name = column_metadata.get('orig_col_name', col)
has_histogram = any(
pr.get('displayer_args', {}).get('displayer') == 'histogram'
for pr in kls.pinned_rows)
for pr in cls.pinned_rows)
min_w = estimate_min_width_px(disp, header_name, column_metadata, has_histogram)
base_config['ag_grid_specs'] = {'minWidth': min_w}
# ag_grid_specs from column_metadata (e.g. init_sd carrying
Expand Down
10 changes: 5 additions & 5 deletions buckaroo/server/data_loading.py
Original file line number Diff line number Diff line change
Expand Up @@ -24,12 +24,12 @@ class ServerSampling(Sampling):
pre_limit = 1_000_000

@classmethod
def pre_stats_sample(kls, df):
def pre_stats_sample(cls, df):
df = check_and_fix_df(df)
if len(df.columns) > kls.max_columns:
df = df[df.columns[:kls.max_columns]]
if kls.pre_limit and len(df) > kls.pre_limit:
sampled = df.sample(kls.pre_limit)
if len(df.columns) > cls.max_columns:
df = df[df.columns[:cls.max_columns]]
if cls.pre_limit and len(df) > cls.pre_limit:
sampled = df.sample(cls.pre_limit)
if isinstance(sampled, pd.DataFrame):
return sampled.sort_index()
return sampled
Expand Down
2 changes: 1 addition & 1 deletion buckaroo/xorq_buckaroo.py
Original file line number Diff line number Diff line change
Expand Up @@ -264,7 +264,7 @@ class DecoratedXorqProcessing(ColAnalysis):
provides_defaults = {}

@classmethod
def post_process_df(kls, expr):
def post_process_df(cls, expr):
return [expr_processing_func(expr), {}]

post_processing_method = proc_func_name
Expand Down
Loading