43 class simple_aggregations_collector;
44 class aggregation_finalizer;
148 [[nodiscard]]
virtual size_t do_hash()
const {
return std::hash<int>{}(
kind); }
155 [[nodiscard]]
virtual std::unique_ptr<aggregation>
clone()
const = 0;
166 data_type col_type, cudf::detail::simple_aggregations_collector& collector)
const = 0;
174 virtual void finalize(cudf::detail::aggregation_finalizer& finalizer)
const = 0;
191 using aggregation::aggregation;
256 template <
typename Base = aggregation>
261 template <
typename Base = aggregation>
266 template <
typename Base = aggregation>
271 template <
typename Base = aggregation>
280 template <
typename Base = aggregation>
285 template <
typename Base = aggregation>
290 template <
typename Base = aggregation>
295 template <
typename Base = aggregation>
300 template <
typename Base = aggregation>
305 template <
typename Base = aggregation>
320 template <
typename Base = aggregation>
332 template <
typename Base = aggregation>
344 template <
typename Base = aggregation>
349 template <
typename Base = aggregation>
359 template <
typename Base = aggregation>
369 template <
typename Base = aggregation>
378 template <
typename Base = aggregation>
388 template <
typename Base = aggregation>
405 template <
typename Base = aggregation>
411 template <
typename Base = aggregation>
486 template <
typename Base = aggregation>
504 template <
typename Base = aggregation>
524 template <
typename Base = aggregation>
536 template <
typename Base = aggregation>
545 template <
typename Base = aggregation>
557 template <
typename Base = aggregation>
559 std::string
const& user_defined_aggregator,
573 template <
typename Base = aggregation>
598 template <
typename Base = aggregation>
617 template <
typename Base = aggregation>
628 template <
typename Base = aggregation>
641 template <
typename Base = aggregation>
654 template <
typename Base = aggregation>
692 template <
typename Base>
730 template <
typename Base>
Abstract base class for specifying the desired aggregation in an aggregation_request.
virtual std::vector< std::unique_ptr< aggregation > > get_simple_aggregations(data_type col_type, cudf::detail::simple_aggregations_collector &collector) const =0
Get the simple aggregations that this aggregation requires to compute.
virtual void finalize(cudf::detail::aggregation_finalizer &finalizer) const =0
Compute the aggregation after pre-requisite simple aggregations have been computed.
Kind
Possible aggregation operations.
@ PRODUCT
product reduction
@ M2
sum of squares of differences from the mean
@ TDIGEST
create a tdigest from a set of input values
@ MEAN
arithmetic mean reduction
@ MERGE_M2
merge partial values of M2 aggregation,
@ PTX
PTX UDF based reduction.
@ MERGE_SETS
merge multiple lists values into one list then drop duplicate entries
@ NUNIQUE
count number of unique elements
@ MERGE_HISTOGRAM
merge partial values of HISTOGRAM aggregation,
@ ARGMIN
Index of min element.
@ CORRELATION
correlation between two sets of elements
@ QUANTILE
compute specified quantile(s)
@ COVARIANCE
covariance between two sets of elements
@ COLLECT_SET
collect values into a list without duplicate entries
@ LAG
window function, accesses row at specified offset preceding current row
@ CUDA
CUDA UDF based reduction.
@ LEAD
window function, accesses row at specified offset following current row
@ SUM_OF_SQUARES
sum of squares reduction
@ NTH_ELEMENT
get the nth element
@ MERGE_LISTS
merge multiple lists values into one list
@ MERGE_TDIGEST
create a tdigest by merging multiple tdigests together
@ COLLECT_LIST
collect values into a list
@ COUNT_VALID
count number of valid elements
@ ROW_NUMBER
get row-number of current index (relative to rolling window)
@ ARGMAX
Index of max element.
@ HISTOGRAM
compute frequency of each element
@ RANK
get rank of current index
@ COUNT_ALL
count number of elements
virtual bool is_equal(aggregation const &other) const
Compares two aggregation objects for equality.
aggregation(aggregation::Kind a)
Construct a new aggregation object.
virtual size_t do_hash() const
Computes the hash value of the aggregation.
virtual std::unique_ptr< aggregation > clone() const =0
Clones the aggregation object.
Kind kind
The aggregation to perform.
Indicator for the logical data type of an element in a column.
Derived class intended for groupby specific aggregation usage.
Derived class intended for groupby specific scan usage.
Derived class intended for reduction usage.
Derived class intended for rolling_window specific aggregation usage.
Derived class intended for scan usage.
Derived class intended for segmented reduction usage.
std::unique_ptr< Base > make_median_aggregation()
correlation_type
Type of correlation method.
std::unique_ptr< Base > make_lag_aggregation(size_type offset)
Factory to create a LAG aggregation.
std::unique_ptr< Base > make_tdigest_aggregation(int max_centroids=1000)
Factory to create a TDIGEST aggregation.
rank_percentage
Whether returned rank should be percentage or not and mention the type of percentage normalization.
std::unique_ptr< Base > make_covariance_aggregation(size_type min_periods=1, size_type ddof=1)
Factory to create a COVARIANCE aggregation.
std::unique_ptr< Base > make_std_aggregation(size_type ddof=1)
Factory to create a STD aggregation.
std::unique_ptr< Base > make_correlation_aggregation(correlation_type type, size_type min_periods=1)
Factory to create a CORRELATION aggregation.
std::unique_ptr< Base > make_merge_sets_aggregation(null_equality nulls_equal=null_equality::EQUAL, nan_equality nans_equal=nan_equality::ALL_EQUAL)
Factory to create a MERGE_SETS aggregation.
std::unique_ptr< Base > make_variance_aggregation(size_type ddof=1)
Factory to create a VARIANCE aggregation.
std::unique_ptr< Base > make_lead_aggregation(size_type offset)
Factory to create a LEAD aggregation.
std::unique_ptr< Base > make_any_aggregation()
std::unique_ptr< Base > make_nunique_aggregation(null_policy null_handling=null_policy::EXCLUDE)
Factory to create a NUNIQUE aggregation.
std::unique_ptr< Base > make_max_aggregation()
std::unique_ptr< Base > make_histogram_aggregation()
std::unique_ptr< Base > make_rank_aggregation(rank_method method, order column_order=order::ASCENDING, null_policy null_handling=null_policy::EXCLUDE, null_order null_precedence=null_order::AFTER, rank_percentage percentage=rank_percentage::NONE)
Factory to create a RANK aggregation.
std::unique_ptr< Base > make_udf_aggregation(udf_type type, std::string const &user_defined_aggregator, data_type output_type)
Factory to create an aggregation base on UDF for PTX or CUDA.
std::unique_ptr< Base > make_row_number_aggregation()
std::unique_ptr< Base > make_merge_histogram_aggregation()
Factory to create a MERGE_HISTOGRAM aggregation.
std::unique_ptr< Base > make_count_aggregation(null_policy null_handling=null_policy::EXCLUDE)
Factory to create a COUNT aggregation.
std::unique_ptr< Base > make_collect_list_aggregation(null_policy null_handling=null_policy::INCLUDE)
Factory to create a COLLECT_LIST aggregation.
std::unique_ptr< Base > make_argmax_aggregation()
Factory to create an ARGMAX aggregation.
std::unique_ptr< Base > make_sum_aggregation()
std::unique_ptr< Base > make_all_aggregation()
std::unique_ptr< Base > make_m2_aggregation()
Factory to create a M2 aggregation.
std::unique_ptr< Base > make_merge_m2_aggregation()
Factory to create a MERGE_M2 aggregation.
std::unique_ptr< Base > make_sum_of_squares_aggregation()
std::unique_ptr< Base > make_min_aggregation()
std::unique_ptr< Base > make_product_aggregation()
std::unique_ptr< Base > make_nth_element_aggregation(size_type n, null_policy null_handling=null_policy::INCLUDE)
Factory to create a NTH_ELEMENT aggregation.
udf_type
Type of code in the user defined function string.
std::unique_ptr< Base > make_merge_lists_aggregation()
Factory to create a MERGE_LISTS aggregation.
std::unique_ptr< Base > make_collect_set_aggregation(null_policy null_handling=null_policy::INCLUDE, null_equality nulls_equal=null_equality::EQUAL, nan_equality nans_equal=nan_equality::ALL_EQUAL)
Factory to create a COLLECT_SET aggregation.
std::unique_ptr< Base > make_argmin_aggregation()
Factory to create an ARGMIN aggregation.
std::unique_ptr< Base > make_quantile_aggregation(std::vector< double > const &quantiles, interpolation interp=interpolation::LINEAR)
Factory to create a QUANTILE aggregation.
std::unique_ptr< Base > make_mean_aggregation()
std::unique_ptr< Base > make_merge_tdigest_aggregation(int max_centroids=1000)
Factory to create a MERGE_TDIGEST aggregation.
@ ONE_NORMALIZED
(rank - 1) / (count - 1)
@ ZERO_NORMALIZED
rank / count
std::unique_ptr< table > quantiles(table_view const &input, std::vector< double > const &q, interpolation interp=interpolation::NEAREST, cudf::sorted is_input_sorted=sorted::NO, std::vector< order > const &column_order={}, std::vector< null_order > const &null_precedence={}, rmm::mr::device_memory_resource *mr=rmm::mr::get_current_device_resource())
Returns the rows of the input corresponding to the requested quantiles.
rank_method
Tie-breaker method to use for ranking the column.
@ DENSE
rank always increases by 1 between groups
@ AVERAGE
mean of first in the group
@ MAX
max of first in the group
@ FIRST
stable sort order ranking (no ties)
@ MIN
min of first in the group
null_order
Indicates how null values compare against all other values.
null_equality
Enum to consider two nulls as equal or unequal.
int32_t size_type
Row index type for columns and tables.
null_policy
Enum to specify whether to include nulls or exclude nulls.
order
Indicates the order in which elements should be sorted.
interpolation
Interpolation method to use when the desired quantile lies between two data points i and j.
nan_equality
Enum to consider different elements (of floating point types) holding NaN value as equal or unequal.
@ AFTER
NULL values ordered after all other values.
@ EQUAL
nulls compare equal
@ INCLUDE
include null elements
@ EXCLUDE
exclude null elements
@ ASCENDING
Elements ordered from small to large.
@ LINEAR
Linear interpolation between i and j.
@ ALL_EQUAL
All NaNs compare equal, regardless of sign.
Type declarations for libcudf.