diff --git a/be/src/agent/be_exec_version_manager.cpp b/be/src/agent/be_exec_version_manager.cpp index ded7a31240afe5..a154e9e85ffd08 100644 --- a/be/src/agent/be_exec_version_manager.cpp +++ b/be/src/agent/be_exec_version_manager.cpp @@ -135,8 +135,10 @@ void BeExecVersionManager::check_function_compatibility(int current_be_exec_vers // a. support TIMESTAMP_NS in Thrift descriptors and PBlock exchange. // 15: start from master // a. distinguish Hive OpenCSVSerde row semantics from generic CSV decoding during upgrades. +// 16: start from master +// a. support pluggable hash algorithms for table distribution and bucket-local exchanges. -const int BeExecVersionManager::max_be_exec_version = SUPPORT_HIVE_OPEN_CSV_VERSION; +const int BeExecVersionManager::max_be_exec_version = SUPPORT_DISTRIBUTION_HASH_TYPE_VERSION; const int BeExecVersionManager::min_be_exec_version = 0; std::map> BeExecVersionManager::_function_change_map {}; std::set BeExecVersionManager::_function_restrict_map; diff --git a/be/src/agent/be_exec_version_manager.h b/be/src/agent/be_exec_version_manager.h index 4a7ca9de6e4588..2b733e9a8d5c3f 100644 --- a/be/src/agent/be_exec_version_manager.h +++ b/be/src/agent/be_exec_version_manager.h @@ -35,6 +35,7 @@ constexpr inline int SUPPORT_ICEBERG_VARIANT_VERSION = 12; constexpr inline int SUPPORT_EXTERNAL_TABLE_SINK_HASH_VERSION = 13; constexpr inline int SUPPORT_TIMESTAMP_NS_VERSION = 14; constexpr inline int SUPPORT_HIVE_OPEN_CSV_VERSION = 15; +constexpr inline int SUPPORT_DISTRIBUTION_HASH_TYPE_VERSION = 16; class BeExecVersionManager { public: diff --git a/be/src/exec/exchange/local_exchange_sink_operator.cpp b/be/src/exec/exchange/local_exchange_sink_operator.cpp index 40a69d86ff1905..a9e39f9ccfc5dd 100644 --- a/be/src/exec/exchange/local_exchange_sink_operator.cpp +++ b/be/src/exec/exchange/local_exchange_sink_operator.cpp @@ -50,7 +50,17 @@ Status LocalExchangeSinkOperatorX::_create_partitioner(RuntimeState* state, int RETURN_IF_ERROR(_partitioner->init(_texprs)); } else if (_type == TLocalPartitionType::BUCKET_HASH_SHUFFLE) { DCHECK_GT(bucket_count, 0); - _partitioner = std::make_unique>(bucket_count); + switch (_distribution_hash_type) { + case TDistributionHashType::CRC32: + _partitioner = std::make_unique>(bucket_count); + break; + case TDistributionHashType::IDENTITY: + _partitioner = std::make_unique(bucket_count); + break; + default: + return Status::InternalError("unsupported distribution_hash_type {}", + static_cast(_distribution_hash_type)); + } RETURN_IF_ERROR(_partitioner->init(_texprs)); } return Status::OK(); diff --git a/be/src/exec/exchange/local_exchange_sink_operator.h b/be/src/exec/exchange/local_exchange_sink_operator.h index 357da9fd83849c..db9c33aed54a14 100644 --- a/be/src/exec/exchange/local_exchange_sink_operator.h +++ b/be/src/exec/exchange/local_exchange_sink_operator.h @@ -73,8 +73,10 @@ class LocalExchangeSinkOperatorX final : public DataSinkOperatorX; LocalExchangeSinkOperatorX(int sink_id, int dest_id, int num_partitions, const std::vector& texprs, - const std::map& bucket_seq_to_instance_idx) + const std::map& bucket_seq_to_instance_idx, + TDistributionHashType::type distribution_hash_type) : Base(sink_id, dest_id, dest_id), + _distribution_hash_type(distribution_hash_type), _num_partitions(num_partitions), _texprs(texprs), _partitioned_exprs_num(texprs.size()), @@ -85,6 +87,9 @@ class LocalExchangeSinkOperatorX final : public DataSinkOperatorX& shuffle_id_to_instance_idx) : Base(operator_id, tnode, dest_id), _type(tnode.local_exchange_node.partition_type), + _distribution_hash_type(tnode.local_exchange_node.__isset.distribution_hash_type + ? tnode.local_exchange_node.distribution_hash_type + : TDistributionHashType::CRC32), _num_partitions(num_partitions), _texprs(tnode.local_exchange_node.distribute_expr_lists), _partitioned_exprs_num(tnode.local_exchange_node.distribute_expr_lists.size()), @@ -124,6 +129,10 @@ class LocalExchangeSinkOperatorX final : public DataSinkOperatorXset_low_memory_mode(); } +#ifdef BE_TEST + PartitionerBase* partitioner_for_test() const { return _partitioner.get(); } +#endif + private: friend class LocalExchangeSinkLocalState; friend class ShuffleExchanger; @@ -135,6 +144,7 @@ class LocalExchangeSinkOperatorX final : public DataSinkOperatorX& _texprs; const size_t _partitioned_exprs_num; diff --git a/be/src/exec/operator/exchange_sink_operator.cpp b/be/src/exec/operator/exchange_sink_operator.cpp index 449c71e3339281..4abc3a83a068b6 100644 --- a/be/src/exec/operator/exchange_sink_operator.cpp +++ b/be/src/exec/operator/exchange_sink_operator.cpp @@ -138,11 +138,24 @@ Status ExchangeSinkLocalState::init(RuntimeState* state, LocalSinkStateInfo& inf "Partitioner", fmt::format("Crc32CHashPartitioner({})", _partition_count)); } else if (_part_type == TPartitionType::BUCKET_SHFFULE_HASH_PARTITIONED) { _partition_count = channels.size(); - _partitioner = std::make_unique>(channels.size()); + switch (p._distribution_hash_type) { + case TDistributionHashType::CRC32: + _partitioner = + std::make_unique>(channels.size()); + custom_profile()->add_info_string( + "Partitioner", fmt::format("Crc32HashPartitioner({})", _partition_count)); + break; + case TDistributionHashType::IDENTITY: + _partitioner = std::make_unique(channels.size()); + custom_profile()->add_info_string( + "Partitioner", fmt::format("IdentityHashPartitioner({})", _partition_count)); + break; + default: + return Status::InternalError("unsupported distribution_hash_type {}", + static_cast(p._distribution_hash_type)); + } RETURN_IF_ERROR(_partitioner->init(p._texprs)); RETURN_IF_ERROR(_partitioner->prepare(state, p._row_desc)); - custom_profile()->add_info_string( - "Partitioner", fmt::format("Crc32HashPartitioner({})", _partition_count)); } else if (_part_type == TPartitionType::OLAP_TABLE_SINK_HASH_PARTITIONED) { // in ExchangeOlapWriter we rely on type of _partitioner here _partition_count = channels.size(); @@ -301,6 +314,9 @@ ExchangeSinkOperatorX::ExchangeSinkOperatorX( _texprs(sink.output_partition.partition_exprs), _row_desc(row_desc), _part_type(sink.output_partition.type), + _distribution_hash_type(sink.output_partition.__isset.distribution_hash_type + ? sink.output_partition.distribution_hash_type + : TDistributionHashType::CRC32), _dests(destinations), _dest_node_id(sink.dest_node_id), _transfer_large_data_by_brpc(config::transfer_large_data_by_brpc), diff --git a/be/src/exec/operator/exchange_sink_operator.h b/be/src/exec/operator/exchange_sink_operator.h index 10351154d1d8cd..2c89129fbd19a8 100644 --- a/be/src/exec/operator/exchange_sink_operator.h +++ b/be/src/exec/operator/exchange_sink_operator.h @@ -250,6 +250,7 @@ class ExchangeSinkOperatorX MOCK_REMOVE(final) : public DataSinkOperatorX_partition_expr_ctxs); } +void IdentityHashPartitioner::_do_hash(const ColumnPtr& column, HashValType* __restrict result, + int idx) const { + const PrimitiveType type = _partition_expr_ctxs[idx]->root()->data_type()->get_primitive_type(); + for (size_t row = 0; row < column->size(); ++row) { + auto val = column->get_data_at(row); + result[row] = + RawValue::identity_hash(val.data, val.size, type, result[row], _partition_count); + } +} + +Status IdentityHashPartitioner::clone(RuntimeState* state, + std::unique_ptr& partitioner) { + auto* new_partitioner = new IdentityHashPartitioner(_partition_count); + partitioner.reset(new_partitioner); + return _clone_expr_ctxs(state, new_partitioner->_partition_expr_ctxs); +} + template class Crc32HashPartitioner; template class Crc32HashPartitioner; template class Crc32HashPartitioner; diff --git a/be/src/exec/partitioner/partitioner.h b/be/src/exec/partitioner/partitioner.h index 98607c3623634f..cf67162eed408b 100644 --- a/be/src/exec/partitioner/partitioner.h +++ b/be/src/exec/partitioner/partitioner.h @@ -191,6 +191,26 @@ class Crc32CHashPartitioner : public Crc32HashPartitioner { } }; +// Bucket-shuffle repartitioner for tables bucketed with the identity hash. Each distribution +// column's canonical bytes are interpreted as an unsigned integer with the first byte as the least +// significant, then appended to the preceding columns; the combined value is kept modulo the +// bucket count. Must stay bit-identical with FE HashDistributionPruner and BE tablet routing. +class IdentityHashPartitioner : public Crc32HashPartitioner { +public: + IdentityHashPartitioner(int partition_count) + : Crc32HashPartitioner(partition_count) {} + + Status clone(RuntimeState* state, std::unique_ptr& partitioner) override; + +private: + void _do_hash(const ColumnPtr& column, HashValType* __restrict result, int idx) const override; + + void _initialize_hash_vals(size_t rows) const override { + _hash_vals.resize(rows); + std::ranges::fill(_hash_vals, 0); + } +}; + /// Instantiated once in partitioner.cpp; suppresses per-TU implicit instantiation. extern template class Crc32HashPartitioner; extern template class Crc32HashPartitioner; diff --git a/be/src/exec/pipeline/dependency.h b/be/src/exec/pipeline/dependency.h index 5802a8ddefc255..c719881b2000d3 100644 --- a/be/src/exec/pipeline/dependency.h +++ b/be/src/exec/pipeline/dependency.h @@ -794,8 +794,13 @@ struct DataDistribution { DataDistribution(const DataDistribution& other) = default; bool need_local_exchange() const { return distribution_type != TLocalPartitionType::NOOP; } DataDistribution& operator=(const DataDistribution& other) = default; + // Hash type is fragment-scoped by design: PipelineFragmentContext::_add_local_exchange_impl + // stamps the fragment's distribution_hash_type onto every BUCKET_HASH_SHUFFLE local exchange + // (FE derives it from the fragment root and rejects mixed-layout fragments before sending), + // so per-operator construction never needs to carry one. TLocalPartitionType::type distribution_type; std::vector partition_exprs; + TDistributionHashType::type distribution_hash_type = TDistributionHashType::CRC32; }; class ExchangerBase; diff --git a/be/src/exec/pipeline/pipeline_fragment_context.cpp b/be/src/exec/pipeline/pipeline_fragment_context.cpp index 5fa5c4c2932294..60764db6bce707 100644 --- a/be/src/exec/pipeline/pipeline_fragment_context.cpp +++ b/be/src/exec/pipeline/pipeline_fragment_context.cpp @@ -1017,9 +1017,14 @@ Status PipelineFragmentContext::_add_local_exchange_impl( const bool use_global_hash_shuffle = bucket_seq_to_instance_idx.empty() && !shuffle_idx_to_instance_idx.contains(-1) && followed_by_shuffled_operator && !_use_serial_source; + if (data_distribution.distribution_type == TLocalPartitionType::BUCKET_HASH_SHUFFLE && + _params.fragment.__isset.distribution_hash_type) { + data_distribution.distribution_hash_type = _params.fragment.distribution_hash_type; + } sink = std::make_shared( sink_id, local_exchange_id, use_global_hash_shuffle ? _total_instances : _num_instances, - data_distribution.partition_exprs, bucket_seq_to_instance_idx); + data_distribution.partition_exprs, bucket_seq_to_instance_idx, + data_distribution.distribution_hash_type); if (bucket_seq_to_instance_idx.empty() && data_distribution.distribution_type == TLocalPartitionType::BUCKET_HASH_SHUFFLE) { data_distribution.distribution_type = diff --git a/be/src/exec/runtime_filter/runtime_filter_bucket_pruner.cpp b/be/src/exec/runtime_filter/runtime_filter_bucket_pruner.cpp index f22c094baae9e7..e4777c73cd4072 100644 --- a/be/src/exec/runtime_filter/runtime_filter_bucket_pruner.cpp +++ b/be/src/exec/runtime_filter/runtime_filter_bucket_pruner.cpp @@ -23,6 +23,7 @@ #include #include +#include "common/cast_set.h" #include "exprs/hybrid_set.h" #include "exprs/runtime_filter_expr.h" #include "exprs/vexpr.h" @@ -40,14 +41,21 @@ Status RuntimeFilterBucketPruner::prune_by_runtime_filters( return Status::OK(); } - phmap::flat_hash_set eligible_filter_ids; + phmap::flat_hash_map eligible_filter_hash_types; for (const auto& desc : rf_descs) { if (desc.__isset.bucket_pruning_target_ids && desc.bucket_pruning_target_ids.contains(scan_node_id)) { - eligible_filter_ids.insert(desc.filter_id); + TDistributionHashType::type hash_type = TDistributionHashType::CRC32; + if (desc.__isset.bucket_pruning_target_hash_types) { + auto it = desc.bucket_pruning_target_hash_types.find(scan_node_id); + if (it != desc.bucket_pruning_target_hash_types.end()) { + hash_type = it->second; + } + } + eligible_filter_hash_types.emplace(desc.filter_id, hash_type); } } - if (eligible_filter_ids.empty()) { + if (eligible_filter_hash_types.empty()) { return Status::OK(); } @@ -57,7 +65,8 @@ Status RuntimeFilterBucketPruner::prune_by_runtime_filters( continue; } auto* rf_expr = assert_cast(root.get()); - if (!eligible_filter_ids.contains(rf_expr->filter_id())) { + auto hash_type_it = eligible_filter_hash_types.find(rf_expr->filter_id()); + if (hash_type_it == eligible_filter_hash_types.end()) { continue; } @@ -77,8 +86,6 @@ Status RuntimeFilterBucketPruner::prune_by_runtime_filters( VExprSPtr target_expr = impl->children()[0]; DORIS_CHECK_EQ(target_expr->node_type(), TExprNodeType::SLOT_REF); - std::shared_ptr> hashes = - rf_expr->get_bucket_prune_hashes(target_expr->data_type()); phmap::flat_hash_map> new_selected_buckets_by_num; for (const auto& range_ptr : ranges) { DORIS_CHECK(range_ptr != nullptr); @@ -92,6 +99,10 @@ Status RuntimeFilterBucketPruner::prune_by_runtime_filters( auto [selected_it, inserted] = new_selected_buckets_by_num.try_emplace(range.bucket_num); if (inserted) { + std::shared_ptr> hashes = + rf_expr->get_bucket_prune_hashes(target_expr->data_type(), + hash_type_it->second, + cast_set(range.bucket_num)); auto& selected_buckets = selected_it->second; selected_buckets.reserve( std::min(hashes->size(), static_cast(range.bucket_num))); diff --git a/be/src/exec/runtime_filter/runtime_filter_wrapper.cpp b/be/src/exec/runtime_filter/runtime_filter_wrapper.cpp index ccd63238ff142e..b63927e298f487 100644 --- a/be/src/exec/runtime_filter/runtime_filter_wrapper.cpp +++ b/be/src/exec/runtime_filter/runtime_filter_wrapper.cpp @@ -17,12 +17,16 @@ #include "exec/runtime_filter/runtime_filter_wrapper.h" +#include + #include "core/data_type/define_primitive_type.h" #include "core/string_ref.h" +#include "exec/common/hash_table/phmap_fwd_decl.h" #include "exec/runtime_filter/runtime_filter_definitions.h" #include "exprs/create_predicate_function.h" #include "exprs/function/cast/cast_to_date_or_datetime_impl.hpp" #include "util/hash_util.hpp" +#include "util/raw_value.h" namespace doris { RuntimeFilterWrapper::RuntimeFilterWrapper(const RuntimeFilterParams* params) @@ -632,14 +636,60 @@ bool RuntimeFilterWrapper::contain_null() const { return false; } +std::shared_ptr> RuntimeFilterWrapper::_get_or_compute_identity_buckets( + PrimitiveType primitive_type, uint32_t bucket_num) const { + std::scoped_lock lock(_identity_bucket_prune_hashes_mutex); + if (auto it = _identity_bucket_prune_hashes.find(bucket_num); + it != _identity_bucket_prune_hashes.end()) { + return it->second; + } + _bucket_prune_hashes_started.store(true); + // Cache bucket membership, not one entry per IN value: different partitions can use + // different bucket counts, and retaining every value would multiply the set size by + // the number of counts. Once all buckets are selected, membership cannot change. + flat_hash_set selected_buckets; + selected_buckets.reserve(std::min( + static_cast(bucket_num), + static_cast(_hybrid_set->size()) + (_hybrid_set->contain_null() ? 1 : 0))); + if (_hybrid_set->contain_null()) { + selected_buckets.insert(RawValue::identity_hash(nullptr, 0, primitive_type, 0, bucket_num)); + } + auto* iter = _hybrid_set->begin(); + while (selected_buckets.size() < bucket_num && iter->has_next()) { + const void* value = iter->get_value(); + DORIS_CHECK(value != nullptr); + if (is_string_type(primitive_type) || primitive_type == TYPE_VARBINARY) { + const auto* string_value = reinterpret_cast(value); + selected_buckets.insert(RawValue::identity_hash(string_value->data, string_value->size, + primitive_type, 0, bucket_num)); + } else { + selected_buckets.insert( + RawValue::identity_hash(value, 0, primitive_type, 0, bucket_num)); + } + iter->next(); + } + auto buckets = std::make_shared>(selected_buckets.begin(), + selected_buckets.end()); + _identity_bucket_prune_hashes.emplace(bucket_num, buckets); + return buckets; +} + std::shared_ptr> -RuntimeFilterWrapper::get_or_compute_bucket_prune_hashes(const DataTypePtr& target_type) const { +RuntimeFilterWrapper::get_or_compute_bucket_prune_hashes(const DataTypePtr& target_type, + TDistributionHashType::type hash_type, + uint32_t bucket_num) const { DORIS_CHECK(_state.load() == State::READY); DORIS_CHECK(_hybrid_set != nullptr); DORIS_CHECK(target_type != nullptr); + DORIS_CHECK_GT(bucket_num, 0); PrimitiveType primitive_type = target_type->get_primitive_type(); DORIS_CHECK_EQ(primitive_type, _column_return_type); + if (hash_type == TDistributionHashType::IDENTITY) { + return _get_or_compute_identity_buckets(primitive_type, bucket_num); + } + + DORIS_CHECK_EQ(hash_type, TDistributionHashType::CRC32); std::call_once(_bucket_prune_hashes_once, [&] { _bucket_prune_hashes_started.store(true); // Materialize the exact-set values into a column so bucket pruning uses the diff --git a/be/src/exec/runtime_filter/runtime_filter_wrapper.h b/be/src/exec/runtime_filter/runtime_filter_wrapper.h index 7fb770e8855c74..729c64f864f1e6 100644 --- a/be/src/exec/runtime_filter/runtime_filter_wrapper.h +++ b/be/src/exec/runtime_filter/runtime_filter_wrapper.h @@ -21,6 +21,7 @@ #include #include +#include #include #include "common/status.h" @@ -87,10 +88,13 @@ class RuntimeFilterWrapper { bool contain_null() const; - // The shared vector includes the NULL hash whenever the exact set contains NULL, regardless - // of target nullability. A non-nullable target may therefore retain one conservative bucket. + // CRC32 returns raw hashes shared across bucket counts; + // IDENTITY returns distinct bucket IDs for the requested count, with no ordering guarantee. + // Both include the NULL bucket whenever the exact set contains NULL, regardless + // of target nullability, so a non-nullable target may conservatively retain that bucket. std::shared_ptr> get_or_compute_bucket_prune_hashes( - const DataTypePtr& target_type) const; + const DataTypePtr& target_type, TDistributionHashType::type hash_type, + uint32_t bucket_num) const; bool disable_always_true_logic() const { return _disable_always_true_logic; } @@ -147,6 +151,8 @@ class RuntimeFilterWrapper { Status _assign(const PBloomFilter& bloom_filter, butil::IOBufAsZeroCopyInputStream* data, bool contain_null); Status _assign(const PMinMaxFilter& minmax_filter, bool contain_null); + std::shared_ptr> _get_or_compute_identity_buckets( + PrimitiveType primitive_type, uint32_t bucket_num) const; Status _change_to_bloom_filter(); // When a runtime filter received from remote and it is a bloom filter, _column_return_type will be invalid. const PrimitiveType _column_return_type; // column type @@ -171,5 +177,8 @@ class RuntimeFilterWrapper { mutable std::once_flag _bucket_prune_hashes_once; mutable std::atomic_bool _bucket_prune_hashes_started = false; mutable std::shared_ptr> _bucket_prune_hashes; + mutable std::mutex _identity_bucket_prune_hashes_mutex; + mutable std::unordered_map>> + _identity_bucket_prune_hashes; }; } // namespace doris diff --git a/be/src/exprs/function/function_string_misc.cpp b/be/src/exprs/function/function_string_misc.cpp index 663fa0fe018591..7821c7829a9085 100644 --- a/be/src/exprs/function/function_string_misc.cpp +++ b/be/src/exprs/function/function_string_misc.cpp @@ -33,6 +33,7 @@ #include #include #include +#include #include #include #include @@ -1507,6 +1508,108 @@ class FunctionCrc32Internal : public IFunction { } }; +// ATTN: for debug only +// compute identity bucket hash as the same way in `VOlapTablePartitionParam::find_tablets()` +// for tables whose distribution_hash_type is identity. `mod` is the bucket count; the returned +// value is the bucket index, so callers can compare it against crc32_internal's raw hash taken +// modulo the same bucket count. +class FunctionIdentityHashInternal : public IFunction { +public: + static constexpr auto name = "identity_hash_internal"; + static FunctionPtr create() { return std::make_shared(); } + String get_name() const override { return name; } + size_t get_number_of_arguments() const override { return 0; } + bool is_variadic() const override { return true; } + bool use_default_implementation_for_nulls() const override { return false; } + DataTypePtr get_return_type_impl(const DataTypes& arguments) const override { + return std::make_shared(); + } + + Status execute_impl(FunctionContext* context, Block& block, const ColumnNumbers& arguments, + uint32_t result, size_t input_rows_count) const override { + DCHECK_GE(arguments.size(), 1); + // The trailing literal is the bucket count to take the modulus against; every leading + // argument is a distribution column. FE rejects non-literal / non-positive counts at + // analysis time; these checks keep BE safe on its own (e.g. against forged thrift). + if (!context->is_col_constant(arguments.back())) { + return Status::InvalidArgument( + "the bucket count argument of {} must be a constant integer, got a variable " + "column", + name); + } + const auto* mod_col_wrapper = context->get_constant_col(arguments.back()); + if (mod_col_wrapper == nullptr || !mod_col_wrapper->column_ptr) { + return Status::InvalidArgument( + "the bucket count argument of {} must be a constant integer, but the constant " + "column is missing", + name); + } + auto mod_ref = mod_col_wrapper->column_ptr->get_data_at(0); + int64_t signed_mod = 0; + switch (block.get_by_position(arguments.back()).type->get_primitive_type()) { + case TYPE_TINYINT: + signed_mod = *reinterpret_cast(mod_ref.data); + break; + case TYPE_SMALLINT: + signed_mod = *reinterpret_cast(mod_ref.data); + break; + case TYPE_INT: + signed_mod = *reinterpret_cast(mod_ref.data); + break; + case TYPE_BIGINT: + signed_mod = *reinterpret_cast(mod_ref.data); + break; + default: + // The FE signature casts any integer literal family to BIGINT before it reaches BE. + return Status::InvalidArgument( + "the bucket count argument of {} must be an integer literal, got {}", name, + block.get_by_position(arguments.back()).type->get_name()); + } + if (signed_mod <= 0 || signed_mod > std::numeric_limits::max()) { + return Status::InvalidArgument( + "the bucket count argument of {} must be a positive integer, got {}", name, + signed_mod); + } + auto mod = static_cast(signed_mod); + + auto argument_size = arguments.size() - 1; + std::vector argument_columns(argument_size); + std::vector argument_primitive_types(argument_size); + + for (size_t i = 0; i < argument_size; ++i) { + argument_columns[i] = + block.get_by_position(arguments[i]).column->convert_to_full_column_if_const(); + argument_primitive_types[i] = + block.get_by_position(arguments[i]).type->get_primitive_type(); + } + + auto res_col = ColumnInt64::create(); + auto& res_data = res_col->get_data(); + res_data.resize_fill(input_rows_count, 0); + + for (size_t i = 0; i < input_rows_count; ++i) { + uint32_t hash_val = 0; + for (size_t j = 0; j < argument_size; ++j) { + const auto& column = argument_columns[j]; + auto primitive_type = argument_primitive_types[j]; + auto val = column->get_data_at(i); + if (val.data != nullptr) { + hash_val = RawValue::identity_hash(val.data, val.size, primitive_type, hash_val, + mod); + } else { + // A null distribution value contributes four zero canonical bytes, the same + // convention RawValue::identity_hash applies to a null value. + hash_val = RawValue::identity_hash(nullptr, 0, primitive_type, hash_val, mod); + } + } + res_data[i] = hash_val; + } + + block.replace_by_position(result, std::move(res_col)); + return Status::OK(); + } +}; + class FunctionUnicodeNormalize : public IFunction { public: static constexpr auto name = "unicode_normalize"; @@ -1661,6 +1764,7 @@ void register_function_string_misc(SimpleFunctionFactory& factory) { factory.register_function(); factory.register_function(); factory.register_function(); + factory.register_function(); factory.register_function(); factory.register_function(); factory.register_function(); diff --git a/be/src/exprs/runtime_filter_expr.cpp b/be/src/exprs/runtime_filter_expr.cpp index e491b4329f3d76..fb83ac99c5d998 100644 --- a/be/src/exprs/runtime_filter_expr.cpp +++ b/be/src/exprs/runtime_filter_expr.cpp @@ -86,9 +86,11 @@ Status RuntimeFilterExpr::clone_node(VExprSPtr* cloned_expr) const { } std::shared_ptr> RuntimeFilterExpr::get_bucket_prune_hashes( - const DataTypePtr& target_type) const { + const DataTypePtr& target_type, TDistributionHashType::type hash_type, + uint32_t bucket_num) const { DORIS_CHECK(_runtime_filter_wrapper != nullptr); - return _runtime_filter_wrapper->get_or_compute_bucket_prune_hashes(target_type); + return _runtime_filter_wrapper->get_or_compute_bucket_prune_hashes(target_type, hash_type, + bucket_num); } Status RuntimeFilterExpr::prepare(RuntimeState* state, const RowDescriptor& desc, diff --git a/be/src/exprs/runtime_filter_expr.h b/be/src/exprs/runtime_filter_expr.h index 32004500f9ca1f..907d7d3eb56992 100644 --- a/be/src/exprs/runtime_filter_expr.h +++ b/be/src/exprs/runtime_filter_expr.h @@ -127,7 +127,8 @@ class RuntimeFilterExpr final : public VExpr { int filter_id() const { return _filter_id; } std::shared_ptr> get_bucket_prune_hashes( - const DataTypePtr& target_type) const; + const DataTypePtr& target_type, TDistributionHashType::type hash_type, + uint32_t bucket_num) const; std::shared_ptr predicate_filtered_rows_counter() const { return _rf_filter_rows; diff --git a/be/src/storage/tablet_info.cpp b/be/src/storage/tablet_info.cpp index fb5c0be4f9f3db..196d44bfc1e8a6 100644 --- a/be/src/storage/tablet_info.cpp +++ b/be/src/storage/tablet_info.cpp @@ -29,6 +29,7 @@ #include #include #include +#include #include #include #include diff --git a/be/src/storage/tablet_info.h b/be/src/storage/tablet_info.h index 1ea346844d89d6..e71c08f3c9500e 100644 --- a/be/src/storage/tablet_info.h +++ b/be/src/storage/tablet_info.h @@ -248,24 +248,40 @@ class VOlapTablePartitionParam { std::map* partition_tablets_buffer = nullptr) const { std::function compute_function; if (!_distributed_slot_locs.empty()) { - //TODO: refactor by saving the hash values. then we can calculate in columnwise. - compute_function = [this](Block* block, uint32_t row, - const VOlapTablePartition& partition) -> uint32_t { - uint32_t hash_val = 0; - for (unsigned short _distributed_slot_loc : _distributed_slot_locs) { - auto* slot_desc = _slots[_distributed_slot_loc]; - auto& column = block->get_by_position(_distributed_slot_loc).column; - auto val = column->get_data_at(row); - if (val.data != nullptr) { - hash_val = RawValue::zlib_crc32(val.data, val.size, - slot_desc->type()->get_primitive_type(), - hash_val); - } else { - hash_val = HashUtil::zlib_crc_hash_null(hash_val); + if (_t_param.distribution_hash_type == TDistributionHashType::IDENTITY) { + compute_function = [this](Block* block, uint32_t row, + const VOlapTablePartition& partition) -> uint32_t { + uint32_t bucket = 0; + for (unsigned short distributed_slot_loc : _distributed_slot_locs) { + auto* slot_desc = _slots[distributed_slot_loc]; + const auto& column = block->get_by_position(distributed_slot_loc).column; + auto val = column->get_data_at(row); + bucket = RawValue::identity_hash( + val.data, val.size, slot_desc->type()->get_primitive_type(), bucket, + cast_set(partition.num_buckets)); } - } - return cast_set(hash_val % partition.num_buckets); - }; + return bucket; + }; + } else { + //TODO: refactor by saving the hash values. then we can calculate in columnwise. + compute_function = [this](Block* block, uint32_t row, + const VOlapTablePartition& partition) -> uint32_t { + uint32_t hash_val = 0; + for (unsigned short _distributed_slot_loc : _distributed_slot_locs) { + auto* slot_desc = _slots[_distributed_slot_loc]; + auto& column = block->get_by_position(_distributed_slot_loc).column; + auto val = column->get_data_at(row); + if (val.data != nullptr) { + hash_val = RawValue::zlib_crc32(val.data, val.size, + slot_desc->type()->get_primitive_type(), + hash_val); + } else { + hash_val = HashUtil::zlib_crc_hash_null(hash_val); + } + } + return cast_set(hash_val % partition.num_buckets); + }; + } } else { // random distribution compute_function = [](Block* block, uint32_t row, const VOlapTablePartition& partition) -> uint32_t { diff --git a/be/src/util/raw_value.h b/be/src/util/raw_value.h index ca9914e4064d9d..0a4efdaa468947 100644 --- a/be/src/util/raw_value.h +++ b/be/src/util/raw_value.h @@ -22,6 +22,7 @@ #include +#include "common/check.h" #include "common/consts.h" #include "common/logging.h" #include "core/data_type/define_primitive_type.h" @@ -38,8 +39,101 @@ class RawValue { // Same as the up function, only use in vec exec engine. static uint32_t zlib_crc32(const void* value, size_t len, const PrimitiveType& type, uint32_t seed); + + // Treat the canonical distribution bytes of a value as an unsigned integer with the first byte + // as the least-significant byte, then append it to the preceding distribution columns. The + // returned value is kept modulo mod throughout, so values of any byte width and any number of + // columns do not require a wide integer. + static uint32_t identity_hash(const void* value, size_t len, const PrimitiveType& type, + uint32_t seed, uint32_t mod); }; +// Keep the canonical byte-width dispatch together so it can be audited against zlib_crc32 below. +// NOLINTNEXTLINE(readability-function-size) +inline uint32_t RawValue::identity_hash(const void* v, size_t len, const PrimitiveType& type, + uint32_t seed, uint32_t mod) { + DCHECK_GT(mod, 0); + auto append_little_endian = [&seed, mod](const void* value, size_t size) { + const auto* bytes = reinterpret_cast(value); + uint64_t remainder = seed; + size_t bytes_since_mod = 0; + for (size_t i = size; i > 0; --i) { + remainder = remainder * 256 + bytes[i - 1]; + if (++bytes_since_mod == sizeof(uint32_t)) { + remainder %= mod; + bytes_since_mod = 0; + } + } + seed = static_cast(remainder % mod); + }; + + if (v == nullptr) { + static constexpr uint32_t NULL_VALUE = 0; + append_little_endian(&NULL_VALUE, sizeof(NULL_VALUE)); + return seed; + } + + switch (type) { + case TYPE_VARCHAR: + case TYPE_VARBINARY: + case TYPE_HLL: + case TYPE_STRING: + case TYPE_CHAR: + append_little_endian(v, len); + break; + case TYPE_BOOLEAN: + case TYPE_TINYINT: + append_little_endian(v, 1); + break; + case TYPE_SMALLINT: + append_little_endian(v, 2); + break; + case TYPE_INT: + case TYPE_FLOAT: + case TYPE_DATEV2: + case TYPE_DECIMAL32: + case TYPE_IPV4: + append_little_endian(v, 4); + break; + case TYPE_BIGINT: + case TYPE_DOUBLE: + case TYPE_TIMEV2: + case TYPE_DATETIMEV2: + case TYPE_TIMESTAMP_NS: + case TYPE_TIMESTAMPTZ: + case TYPE_DECIMAL64: + append_little_endian(v, 8); + break; + case TYPE_LARGEINT: + case TYPE_DECIMAL128I: + case TYPE_IPV6: + append_little_endian(v, 16); + break; + case TYPE_DECIMAL256: + append_little_endian(v, 32); + break; + case TYPE_DATE: + case TYPE_DATETIME: { + const auto* date_val = reinterpret_cast(v); + char buf[64]; + int date_len = date_val->to_buffer(buf); + append_little_endian(buf, date_len); + break; + } + case TYPE_DECIMALV2: { + const auto* dec_val = reinterpret_cast(v); + int64_t int_val = dec_val->int_value(); + int32_t frac_val = dec_val->frac_value(); + append_little_endian(&frac_val, sizeof(frac_val)); + append_little_endian(&int_val, sizeof(int_val)); + break; + } + default: + DORIS_CHECK(false) << "invalid type: " << type; + } + return seed; +} + // NOTE: this is just for split data, decimal use old doris hash function // Because crc32 hardware is not equal with zlib crc32 inline uint32_t RawValue::zlib_crc32(const void* v, size_t len, const PrimitiveType& type, @@ -75,7 +169,7 @@ inline uint32_t RawValue::zlib_crc32(const void* v, size_t len, const PrimitiveT return HashUtil::zlib_crc_hash(v, 8, seed); case TYPE_DATE: case TYPE_DATETIME: { - auto* date_val = (const VecDateTimeValue*)v; + const auto* date_val = reinterpret_cast(v); char buf[64]; int date_len = date_val->to_buffer(buf); return HashUtil::zlib_crc_hash(buf, date_len, seed); @@ -95,7 +189,7 @@ inline uint32_t RawValue::zlib_crc32(const void* v, size_t len, const PrimitiveT } case TYPE_DECIMALV2: { - const DecimalV2Value* dec_val = (const DecimalV2Value*)v; + const auto* dec_val = reinterpret_cast(v); int64_t int_val = dec_val->int_value(); int32_t frac_val = dec_val->frac_value(); seed = HashUtil::zlib_crc_hash(&int_val, sizeof(int_val), seed); diff --git a/be/test/exec/partitioner/identity_partitioner_test.cpp b/be/test/exec/partitioner/identity_partitioner_test.cpp new file mode 100644 index 00000000000000..cadcbe09f7dc18 --- /dev/null +++ b/be/test/exec/partitioner/identity_partitioner_test.cpp @@ -0,0 +1,327 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +#include + +#include +#include +#include +#include +#include + +#include "common/object_pool.h" +#include "core/block/block.h" +#include "core/data_type/data_type_number.h" +#include "core/data_type/data_type_string.h" +#include "core/value/decimalv2_value.h" +#include "core/value/ipv4_value.h" +#include "core/value/ipv6_value.h" +#include "core/value/vdatetime_value.h" +#include "exec/partitioner/partitioner.h" +#include "runtime/descriptor_helper.h" +#include "runtime/descriptors.h" +#include "testutil/column_helper.h" +#include "testutil/mock/mock_runtime_state.h" +#include "util/raw_value.h" + +namespace doris { + +// Unit tests for the BE-side identity reshuffle partitioner used by bucket-shuffle join when the +// target table is bucketed with distribution_hash_type = identity. It must interpret every value's +// canonical little-endian bytes as unsigned and compose multiple columns identically to FE pruning +// and BE tablet routing. +class IdentityPartitionerTest : public ::testing::Test { +protected: + void SetUp() override { + TDescriptorTableBuilder dtb; + TTupleDescriptorBuilder tuple_builder; + tuple_builder.add_slot(TSlotDescriptorBuilder() + .type(TYPE_INT) + .nullable(true) + .column_name("c1") + .column_pos(1) + .build()); + tuple_builder.add_slot(TSlotDescriptorBuilder() + .type(TYPE_STRING) + .nullable(false) + .column_name("c2") + .column_pos(2) + .build()); + tuple_builder.build(&dtb); + TDescriptorTable thrift_tbl = dtb.desc_tbl(); + + DescriptorTbl* desc_tbl = nullptr; + auto st = DescriptorTbl::create(&_pool, thrift_tbl, &desc_tbl); + ASSERT_TRUE(st.ok()) << st.to_string(); + _state.set_desc_tbl(desc_tbl); + + _tuple_id = thrift_tbl.tupleDescriptors[0].id; + _row_desc = std::make_unique(*desc_tbl, std::vector {_tuple_id}); + _slot_ids.push_back(thrift_tbl.slotDescriptors[0].id); + _slot_ids.push_back(thrift_tbl.slotDescriptors[1].id); + } + + TExpr make_slot_ref(size_t slot_index, PrimitiveType type, bool nullable) { + TExprNode node; + node.__set_node_type(TExprNodeType::SLOT_REF); + node.__set_num_children(0); + TSlotRef slot_ref; + slot_ref.__set_slot_id(_slot_ids[slot_index]); + slot_ref.__set_tuple_id(_tuple_id); + node.__set_slot_ref(slot_ref); + TTypeDesc type_desc = create_type_desc(type); + type_desc.__set_is_nullable(nullable); + node.__set_type(type_desc); + node.__set_is_nullable(nullable); + TExpr expr; + expr.nodes.emplace_back(std::move(node)); + return expr; + } + + TExpr make_int_slot_ref() { return make_slot_ref(0, TYPE_INT, true); } + + TExpr make_string_slot_ref() { return make_slot_ref(1, TYPE_STRING, false); } + + template + std::vector run(int partition_count, Block block, + std::vector exprs) { + Partitioner partitioner(partition_count); + EXPECT_TRUE(partitioner.init(exprs).ok()); + EXPECT_TRUE(partitioner.prepare(&_state, *_row_desc).ok()); + EXPECT_TRUE(partitioner.open(&_state).ok()); + EXPECT_TRUE(partitioner.do_partitioning(&_state, &block).ok()); + return partitioner.get_channel_ids(); + } + + template + std::vector run(int partition_count, Block block) { + return run(partition_count, std::move(block), {make_int_slot_ref()}); + } + + ObjectPool _pool; + MockRuntimeState _state; + std::unique_ptr _row_desc; + TTupleId _tuple_id = 0; + std::vector _slot_ids; +}; + +// Positive integers retain value-modulo behavior; for a power-of-two bucket count, two's-complement +// unsigned bytes also place negative values in the same buckets as negative-safe signed modulo. +TEST_F(IdentityPartitionerTest, ChannelIsValueModBucketCount) { + constexpr int n = 8; + std::vector values = {3, 8, 100, 999, -1, -8}; + auto channels = + run(n, ColumnHelper::create_block(values)); + ASSERT_EQ(values.size(), channels.size()); + for (size_t i = 0; i < values.size(); i++) { + EXPECT_EQ(static_cast(((values[i] % n) + n) % n), channels[i]) + << "row " << i << " value " << values[i]; + } +} + +// Canonical two's-complement bytes are unsigned, so negative values need no special branch. +TEST_F(IdentityPartitionerTest, NegativeValueUsesUnsignedBytes) { + constexpr int n = 10; + auto channels = + run(n, ColumnHelper::create_block({-1, -8})); + ASSERT_EQ(2U, channels.size()); + EXPECT_EQ(5U, channels[0]); // UINT32_MAX % 10 + EXPECT_EQ(8U, channels[1]); // (UINT32_MAX - 7) % 10 +} + +TEST_F(IdentityPartitionerTest, SupportsMultipleTypedColumns) { + constexpr int n = 257; + auto block = ColumnHelper::create_block({1, 2}); + auto strings = ColumnHelper::create_block({"A", "BC"}); + block.insert(strings.get_by_position(0)); + auto channels = run(n, std::move(block), + {make_int_slot_ref(), make_string_slot_ref()}); + ASSERT_EQ(2U, channels.size()); + EXPECT_EQ(64U, channels[0]); // (1 * 256 + 'A') % 257 + // unsigned_le("BC") = 0x4342; append it after uint32_le(2). + EXPECT_EQ((2U * 256U * 256U + 0x4342U) % n, channels[1]); +} + +// A null distribution value is represented by four zero bytes. +TEST_F(IdentityPartitionerTest, NullGoesToChannelZero) { + constexpr int n = 8; + // row 0 null -> 0; row 1 = 300 -> 300 % 8 = 4 + auto channels = run( + n, ColumnHelper::create_nullable_block({0, 300}, {1, 0})); + ASSERT_EQ(2U, channels.size()); + EXPECT_EQ(0U, channels[0]); + EXPECT_EQ(4U, channels[1]); +} + +// Guard against the two branches being swapped: crc32 reshuffle must differ from identity for at +// least one row (crc32 does not collapse to value % n). +TEST_F(IdentityPartitionerTest, Crc32DiffersFromIdentity) { + constexpr int n = 8; + std::vector values = {3, 8, 100, 999, 5, 6, 7, 12}; + auto identity = + run(n, ColumnHelper::create_block(values)); + auto crc32 = run>( + n, ColumnHelper::create_block(values)); + ASSERT_EQ(values.size(), identity.size()); + ASSERT_EQ(values.size(), crc32.size()); + bool differs = false; + for (size_t i = 0; i < values.size(); i++) { + if (identity[i] != crc32[i]) { + differs = true; + break; + } + } + EXPECT_TRUE(differs); +} + +namespace { + +// Like the FE pruning tests' BigInteger oracle, construct the entire unsigned value before +// taking the modulus. This deliberately does not reproduce RawValue's chunked modular loop. +boost::multiprecision::cpp_int append_bytes(boost::multiprecision::cpp_int prefix, + const void* value, size_t size) { + const auto* bytes = static_cast(value); + boost::multiprecision::cpp_int suffix = 0; + for (size_t i = 0; i < size; ++i) { + suffix += boost::multiprecision::cpp_int(bytes[i]) << (8 * i); + } + return (prefix << (8 * size)) + suffix; +} + +std::array make_high_bytes(size_t width, bool negative) { + std::array bytes {}; + // Nonzero on both sides of every 4/8/16-byte boundary, including the high byte. + for (size_t i = 0; i < width; ++i) { + bytes[i] = static_cast(17 + 7 * i); + } + bytes[width - 1] = negative ? 0xe3 : 0x63; + return bytes; +} + +uint32_t bucket(const boost::multiprecision::cpp_int& value, uint32_t modulus) { + return (value % modulus).convert_to(); +} + +} // namespace + +TEST(IdentityHashTest, FixedWidthHighBytes) { + const std::vector> types = { + {TYPE_BOOLEAN, 1}, {TYPE_TINYINT, 1}, {TYPE_SMALLINT, 2}, + {TYPE_INT, 4}, {TYPE_FLOAT, 4}, {TYPE_DATEV2, 4}, + {TYPE_DECIMAL32, 4}, {TYPE_IPV4, 4}, {TYPE_BIGINT, 8}, + {TYPE_DOUBLE, 8}, {TYPE_TIMEV2, 8}, {TYPE_DATETIMEV2, 8}, + {TYPE_TIMESTAMP_NS, 8}, {TYPE_TIMESTAMPTZ, 8}, {TYPE_DECIMAL64, 8}, + {TYPE_LARGEINT, 16}, {TYPE_DECIMAL128I, 16}, {TYPE_IPV6, 16}, + {TYPE_DECIMAL256, 32}}; + for (auto [type, width] : types) { + for (bool negative : {false, true}) { + auto bytes = make_high_bytes(width, negative); + for (uint32_t seed : {0U, 37U, 0xfedcba98U}) { + auto expected = append_bytes(seed, bytes.data(), width); + for (uint32_t modulus : {251U, 1009U, 1024U}) { + SCOPED_TRACE(::testing::Message() + << "type=" << type << " width=" << width << " negative=" + << negative << " seed=" << seed << " modulus=" << modulus); + EXPECT_EQ(bucket(expected, modulus), + RawValue::identity_hash(bytes.data(), bytes.size(), type, seed, + modulus)); + } + // A power-of-two modulus alone cannot expose loss of high bytes. Ensure these + // vectors distinguish the common 2/4/8/16-byte truncations with an odd modulus. + for (size_t truncated : {2U, 4U, 8U, 16U}) { + if (truncated < width) { + auto wrong = append_bytes(seed, bytes.data(), truncated); + EXPECT_TRUE(bucket(expected, 251) != bucket(wrong, 251) || + bucket(expected, 1009) != bucket(wrong, 1009)); + } + } + } + } + } +} + +TEST(IdentityHashTest, WideColumnsWithNullTail) { + std::array wide {}; + for (size_t i = 0; i < wide.size(); ++i) { + wide[i] = static_cast(0xf1 - 3 * i); + } + const int64_t second = -0x123456789abcdefLL; + const uint32_t null_bytes = 0; + const std::string tail = "identity"; + for (uint32_t seed : {37U, 0xfedcba98U}) { + auto expected = append_bytes(seed, wide.data(), wide.size()); + expected = append_bytes(expected, &second, sizeof(second)); + expected = append_bytes(expected, &null_bytes, sizeof(null_bytes)); + for (uint32_t modulus : {251U, 1009U, 1024U}) { + uint32_t hash = RawValue::identity_hash(wide.data(), wide.size(), TYPE_DECIMAL256, seed, + modulus); + hash = RawValue::identity_hash(&second, sizeof(second), TYPE_BIGINT, hash, modulus); + hash = RawValue::identity_hash(nullptr, 0, TYPE_LARGEINT, hash, modulus); + EXPECT_EQ(bucket(expected, modulus), hash); + EXPECT_EQ( + bucket(append_bytes(expected, tail.data(), tail.size()), modulus), + RawValue::identity_hash(tail.data(), tail.size(), TYPE_STRING, hash, modulus)); + } + } +} + +TEST(IdentityHashTest, LegacyTypes) { + auto date = VecDateTimeValue::create_from_olap_date(20260102); + char date_buffer[64]; + const int date_length = date.to_buffer(date_buffer); + for (uint32_t seed : {0U, 37U, 0xfedcba98U}) { + for (uint32_t modulus : {251U, 1009U, 1024U}) { + EXPECT_EQ(bucket(append_bytes(seed, date_buffer, date_length), modulus), + RawValue::identity_hash(&date, sizeof(date), TYPE_DATE, seed, modulus)); + for (int64_t signed_integer : {123456789012LL, -123456789012LL}) { + const DecimalV2Value decimal(signed_integer, 456000000); + const int32_t fraction = decimal.frac_value(); + const int64_t integer = decimal.int_value(); + auto expected = append_bytes(seed, &fraction, sizeof(fraction)); + expected = append_bytes(expected, &integer, sizeof(integer)); + EXPECT_EQ(bucket(expected, modulus), + RawValue::identity_hash(&decimal, sizeof(decimal), TYPE_DECIMALV2, seed, + modulus)); + } + } + } +} + +TEST(IdentityHashTest, TimestampNsCanonicalBytes) { + constexpr uint32_t n = 257; + const TimeStampNsValue one_nanosecond(1); + EXPECT_EQ(1U, RawValue::identity_hash(&one_nanosecond, sizeof(one_nanosecond), + TYPE_TIMESTAMP_NS, 0, n)); + + const TimeStampNsValue before_epoch(-1); + EXPECT_EQ( + std::numeric_limits::max() % n, + RawValue::identity_hash(&before_epoch, sizeof(before_epoch), TYPE_TIMESTAMP_NS, 0, n)); +} + +TEST(IdentityHashTest, IpCanonicalBytes) { + constexpr uint32_t n = 257; + IPv4 ipv4 = 0; + ASSERT_TRUE(IPv4Value::from_string(ipv4, "1.2.3.4")); + EXPECT_EQ(2U, RawValue::identity_hash(&ipv4, sizeof(ipv4), TYPE_IPV4, 0, n)); + + IPv6 ipv6 = 0; + ASSERT_TRUE(IPv6Value::from_string(ipv6, "::1")); + EXPECT_EQ(1U, RawValue::identity_hash(&ipv6, sizeof(ipv6), TYPE_IPV6, 0, n)); +} + +} // namespace doris diff --git a/be/test/exec/pipeline/local_exchanger_test.cpp b/be/test/exec/pipeline/local_exchanger_test.cpp index 09c3dfa26c0642..56dc663224350f 100644 --- a/be/test/exec/pipeline/local_exchanger_test.cpp +++ b/be/test/exec/pipeline/local_exchanger_test.cpp @@ -18,7 +18,11 @@ #include #include +#include +#include #include +#include +#include #include "common/status.h" #include "core/assert_cast.h" @@ -71,6 +75,40 @@ class LocalExchangerTest : public testing::TestWithParam { const int DUMMY_PORT = config::brpc_port; }; +TEST_F(LocalExchangerTest, BucketShufflePartitionerHashType) { + const std::vector exprs; + const std::map bucket_seq_to_instance_idx {{0, 0}}; + + LocalExchangeSinkOperatorX crc32_op(0, 0, 1, exprs, bucket_seq_to_instance_idx, + TDistributionHashType::CRC32); + EXPECT_TRUE(crc32_op.init(_runtime_state.get(), TLocalPartitionType::BUCKET_HASH_SHUFFLE, 1, + bucket_seq_to_instance_idx) + .ok()); + EXPECT_NE( + dynamic_cast*>(crc32_op.partitioner_for_test()), + nullptr); + EXPECT_EQ(dynamic_cast(crc32_op.partitioner_for_test()), nullptr); + + LocalExchangeSinkOperatorX identity_op(1, 0, 1, exprs, bucket_seq_to_instance_idx, + TDistributionHashType::IDENTITY); + EXPECT_TRUE(identity_op + .init(_runtime_state.get(), TLocalPartitionType::BUCKET_HASH_SHUFFLE, 1, + bucket_seq_to_instance_idx) + .ok()); + EXPECT_NE(dynamic_cast(identity_op.partitioner_for_test()), nullptr); + + // Mirror Thrift's i32-to-enum read to test rejection of an unknown wire value. + // This deliberately injects an invalid C++ enum value, not a supported hash type. + LocalExchangeSinkOperatorX invalid_op( + 2, 0, 1, exprs, bucket_seq_to_instance_idx, + // NOLINTNEXTLINE(clang-analyzer-optin.core.EnumCastOutOfRange) + static_cast(std::numeric_limits::max())); + auto status = invalid_op.init(_runtime_state.get(), TLocalPartitionType::BUCKET_HASH_SHUFFLE, 1, + bucket_seq_to_instance_idx); + EXPECT_TRUE(status.is()); + EXPECT_NE(status.to_string().find("unsupported distribution_hash_type"), std::string::npos); +} + TEST_F(LocalExchangerTest, ShuffleExchanger) { int num_sink = 4; int num_sources = 4; diff --git a/be/test/exec/runtime_filter/runtime_filter_bucket_pruner_test.cpp b/be/test/exec/runtime_filter/runtime_filter_bucket_pruner_test.cpp index 59737c0c3d6de1..47e2b7516f0f1a 100644 --- a/be/test/exec/runtime_filter/runtime_filter_bucket_pruner_test.cpp +++ b/be/test/exec/runtime_filter/runtime_filter_bucket_pruner_test.cpp @@ -21,7 +21,10 @@ #include #include +#include +#include #include +#include #include #include #include @@ -32,6 +35,7 @@ #include "exec/runtime_filter/runtime_filter_definitions.h" #include "exec/runtime_filter/runtime_filter_wrapper.h" #include "exprs/create_predicate_function.h" +#include "exprs/hybrid_set.h" #include "exprs/runtime_filter_expr.h" #include "exprs/vdirect_in_predicate.h" #include "exprs/vexpr_context.h" @@ -48,12 +52,13 @@ class RuntimeFilterBucketPrunerTest : public testing::Test { std::shared_ptr make_in_wrapper(int filter_id, const std::vector& values, - bool null_aware = false) { + bool null_aware = false, + int max_in_num = 1024) { RuntimeFilterParams params {.filter_id = filter_id, .filter_type = RuntimeFilterType::IN_FILTER, .column_return_type = TYPE_INT, .null_aware = null_aware, - .max_in_num = 1024}; + .max_in_num = max_in_num}; auto wrapper = std::make_shared(¶ms); for (const int32_t value : values) { wrapper->hybrid_set()->insert(&value); @@ -123,10 +128,12 @@ class RuntimeFilterBucketPrunerTest : public testing::Test { return std::make_shared(wrapper); } - TRuntimeFilterDesc bucket_prune_desc(int filter_id) { + TRuntimeFilterDesc bucket_prune_desc( + int filter_id, TDistributionHashType::type hash_type = TDistributionHashType::CRC32) { TRuntimeFilterDesc desc; desc.__set_filter_id(filter_id); desc.__set_bucket_pruning_target_ids({SCAN_NODE_ID}); + desc.__set_bucket_pruning_target_hash_types({{SCAN_NODE_ID, hash_type}}); return desc; } @@ -165,16 +172,24 @@ TEST_F(RuntimeFilterBucketPrunerTest, ExactSetHashesSharedAcrossConsumers) { auto second = make_in_conjunct(filter_id, {}, runtime_filter_wrapper); auto target_type = first->root()->get_impl()->children()[0]->data_type(); - auto first_hashes = assert_cast(first->root().get()) - ->get_bucket_prune_hashes(target_type); - auto second_hashes = assert_cast(second->root().get()) - ->get_bucket_prune_hashes(target_type); - auto nullable_hashes = assert_cast(first->root().get()) - ->get_bucket_prune_hashes(std::make_shared( - std::make_shared())); + auto first_hashes = + assert_cast(first->root().get()) + ->get_bucket_prune_hashes(target_type, TDistributionHashType::CRC32, 8); + auto second_hashes = + assert_cast(second->root().get()) + ->get_bucket_prune_hashes(target_type, TDistributionHashType::CRC32, 8); + auto nullable_hashes = + assert_cast(first->root().get()) + ->get_bucket_prune_hashes( + std::make_shared(std::make_shared()), + TDistributionHashType::CRC32, 8); EXPECT_EQ(first_hashes.get(), second_hashes.get()); EXPECT_EQ(first_hashes.get(), nullable_hashes.get()); + EXPECT_EQ(first_hashes.get(), runtime_filter_wrapper + ->get_or_compute_bucket_prune_hashes( + target_type, TDistributionHashType::CRC32, 97) + .get()); ASSERT_EQ(first_hashes->size(), 4); EXPECT_EQ(first_hashes->back(), HashUtil::zlib_crc_hash_null(0)); } @@ -184,12 +199,232 @@ TEST_F(RuntimeFilterBucketPrunerTest, RejectsMergeAfterBucketHashesStart) { auto wrapper = make_in_wrapper(filter_id, {1}); auto other = make_in_wrapper(filter_id, {2}); - static_cast( - wrapper->get_or_compute_bucket_prune_hashes(std::make_shared())); + static_cast(wrapper->get_or_compute_bucket_prune_hashes(std::make_shared(), + TDistributionHashType::CRC32, 8)); EXPECT_DEATH({ static_cast(wrapper->merge(other.get())); }, "Check failed"); } +TEST_F(RuntimeFilterBucketPrunerTest, IdentityExactInKeepsIdentityBucket) { + constexpr int filter_id = 17; + // value 2 separates the algorithms on 4 buckets: crc32(2) % 4 == 3 while 2 % 4 == 2, so + // this case fails if the pruning silently fell back to CRC32 (the previous value 1 gave + // crc32(1) % 4 == 1 == 1 % 4 and could not tell the two apart). + constexpr int32_t value = 2; + VExprContextSPtrs conjuncts {make_in_conjunct(filter_id, {value})}; + std::vector rf_descs { + bucket_prune_desc(filter_id, TDistributionHashType::IDENTITY)}; + + RuntimeFilterBucketPruner pruner; + int64_t newly_pruned = 0; + ASSERT_TRUE(pruner.prune_by_runtime_filters(four_bucket_ranges(), conjuncts, rf_descs, + SCAN_NODE_ID, 1024, &newly_pruned) + .ok()); + EXPECT_EQ(newly_pruned, 3); + for (int32_t bucket_seq = 0; bucket_seq < 4; ++bucket_seq) { + EXPECT_EQ(pruner.is_bucket_pruned(bucket_seq, 4), bucket_seq != value % 4); + } +} + +TEST_F(RuntimeFilterBucketPrunerTest, IdentityExactInSeparatesFromCrc32AcrossCounts) { + constexpr int filter_id = 18; + constexpr int32_t value = 10; + VExprContextSPtrs conjuncts {make_in_conjunct(filter_id, {value})}; + + // For every bucket count, the IDENTITY descriptor must keep exactly value % n while the + // CRC32 fallback would keep zlib_crc32(value) % n; the two must disagree for at least + // one count so a regression to CRC32 cannot pass unnoticed. + int disagreements = 0; + for (int32_t bucket_num : {4, 5, 8, 97, 257}) { + BucketPruneRanges ranges; + for (int32_t bucket_seq = 0; bucket_seq < bucket_num; ++bucket_seq) { + add_range(&ranges, ranges.size(), bucket_seq, bucket_num); + } + std::vector identity_descs { + bucket_prune_desc(filter_id, TDistributionHashType::IDENTITY)}; + RuntimeFilterBucketPruner identity_pruner; + int64_t newly_pruned = 0; + ASSERT_TRUE(identity_pruner + .prune_by_runtime_filters(ranges, conjuncts, identity_descs, + SCAN_NODE_ID, 1024, &newly_pruned) + .ok()); + EXPECT_EQ(newly_pruned, bucket_num - 1); + for (int32_t bucket_seq = 0; bucket_seq < bucket_num; ++bucket_seq) { + EXPECT_EQ(identity_pruner.is_bucket_pruned(bucket_seq, bucket_num), + bucket_seq != value % bucket_num); + } + if (bucket_for_value(value, bucket_num) != value % bucket_num) { + ++disagreements; + } + } + EXPECT_GE(disagreements, 1); +} + +// Count values actually visited, so the full-coverage shortcut is checked without timing tests. +class CountingIntSet : public HybridSet { +public: + CountingIntSet() : HybridSet(false) {} + + class CountingIterator : public IteratorBase { + public: + CountingIterator(IteratorBase* inner, size_t& visited) : _inner(inner), _visited(visited) {} + const void* get_value() override { + ++_visited; + return _inner->get_value(); + } + bool has_next() const override { return _inner->has_next(); } + void next() override { _inner->next(); } + + private: + IteratorBase* _inner; + size_t& _visited; + }; + + IteratorBase* begin() override { + _iterator = + std::make_unique(HybridSet::begin(), values_visited); + return _iterator.get(); + } + + size_t values_visited = 0; + +private: + std::unique_ptr _iterator; +}; + +TEST_F(RuntimeFilterBucketPrunerTest, IdentityCacheStopsAfterFullCoverage) { + auto wrapper = make_in_wrapper(17, {}); + auto values = std::make_shared(); + for (int32_t value = 0; value < 1024; ++value) { + values->insert(&value); + } + wrapper->_hybrid_set = values; + auto buckets = wrapper->get_or_compute_bucket_prune_hashes(std::make_shared(), + TDistributionHashType::IDENTITY, 1); + ASSERT_EQ(buckets->size(), 1); + EXPECT_EQ(buckets->front(), 0U); + EXPECT_EQ(values->values_visited, 1); + EXPECT_EQ(values->size(), 1024); +} + +TEST_F(RuntimeFilterBucketPrunerTest, IdentityCacheDeduplicatesAndSharesBuckets) { + auto wrapper = make_in_wrapper(17, {0, 4, 8, 12, -4}, true); + auto first = make_in_conjunct(17, {}, wrapper); + auto second = make_in_conjunct(17, {}, wrapper); + auto target_type = std::make_shared(); + auto buckets = + assert_cast(first->root().get()) + ->get_bucket_prune_hashes(target_type, TDistributionHashType::IDENTITY, 4); + // Every non-null value and NULL select the same bucket; retain it only once. + ASSERT_EQ(buckets->size(), 1); + EXPECT_EQ(buckets->front(), 0U); + EXPECT_EQ(buckets.get(), + assert_cast(second->root().get()) + ->get_bucket_prune_hashes(target_type, TDistributionHashType::IDENTITY, 4) + .get()); + EXPECT_EQ(buckets.get(), wrapper->get_or_compute_bucket_prune_hashes( + std::make_shared(target_type), + TDistributionHashType::IDENTITY, 4) + .get()); +} + +TEST_F(RuntimeFilterBucketPrunerTest, IdentityCacheHandlesEmptyAndNullOnlySets) { + auto target_type = std::make_shared(); + for (bool null_aware : {false, true}) { + auto wrapper = make_in_wrapper(17, {}, null_aware); + for (uint32_t bucket_num : {1U, 7U, 768U}) { + auto buckets = wrapper->get_or_compute_bucket_prune_hashes( + target_type, TDistributionHashType::IDENTITY, bucket_num); + if (null_aware) { + ASSERT_EQ(buckets->size(), 1); + EXPECT_EQ(buckets->front(), 0U); + } else { + EXPECT_TRUE(buckets->empty()); + } + } + } +} + +TEST_F(RuntimeFilterBucketPrunerTest, IdentityCacheHandlesLargeBucketCount) { + auto wrapper = make_in_wrapper(17, {-1, 0, 1}, true); + auto buckets = wrapper->get_or_compute_bucket_prune_hashes( + std::make_shared(), TDistributionHashType::IDENTITY, + std::numeric_limits::max()); + // Scratch space must also be bounded by the set size, not by this huge bucket count. + const std::set expected {0, 1}; + EXPECT_EQ(std::set(buckets->begin(), buckets->end()), expected); + EXPECT_LE(buckets->capacity(), expected.size()); +} + +TEST_F(RuntimeFilterBucketPrunerTest, IdentityCacheRetainsOnlyBucketsAcrossManyCounts) { + constexpr int value_count = 40960; + constexpr uint32_t max_bucket_num = 768; + std::vector values(value_count); + std::iota(values.begin(), values.end(), 0); + auto wrapper = make_in_wrapper(17, values, true, value_count); + auto target_type = std::make_shared(); + size_t retained_capacity = 0; + for (uint32_t bucket_num = 1; bucket_num <= max_bucket_num; ++bucket_num) { + SCOPED_TRACE(bucket_num); + auto buckets = wrapper->get_or_compute_bucket_prune_hashes( + target_type, TDistributionHashType::IDENTITY, bucket_num); + // Contiguous values cover every bucket. The old cache retained value_count + 1 + // entries per count (~120 MiB); duplicates must not survive in size OR capacity. + ASSERT_EQ(buckets->size(), bucket_num); + EXPECT_LE(buckets->capacity(), bucket_num); + EXPECT_EQ(std::set(buckets->begin(), buckets->end()).size(), bucket_num); + for (uint32_t bucket : *buckets) { + EXPECT_LT(bucket, bucket_num); + } + retained_capacity += buckets->capacity(); + EXPECT_EQ(buckets.get(), + wrapper->get_or_compute_bucket_prune_hashes( + target_type, TDistributionHashType::IDENTITY, bucket_num) + .get()); + } + EXPECT_LE(retained_capacity, max_bucket_num * (max_bucket_num + 1) / 2); + EXPECT_EQ(wrapper->hybrid_set()->size(), value_count); + EXPECT_TRUE(wrapper->hybrid_set()->contain_null()); +} + +TEST_F(RuntimeFilterBucketPrunerTest, IdentitySparseBucketsRemainCorrectAcrossCounts) { + constexpr int filter_id = 17; + const std::vector values {-1, 0, 4, 8, 12}; + auto wrapper = make_in_wrapper(filter_id, values, true); + VExprContextSPtrs conjuncts {make_in_conjunct(filter_id, {}, wrapper)}; + std::vector rf_descs { + bucket_prune_desc(filter_id, TDistributionHashType::IDENTITY)}; + BucketPruneRanges ranges; + std::map> expected_by_num; + int64_t expected_pruned = 0; + for (int32_t bucket_num : {4, 7, 97}) { + auto& expected = expected_by_num[bucket_num]; + expected.insert(0); // NULL's canonical bytes select bucket zero. + for (int32_t value : values) { + expected.insert(static_cast(value) % static_cast(bucket_num)); + } + expected_pruned += bucket_num - static_cast(expected.size()); + for (int32_t bucket = 0; bucket < bucket_num; ++bucket) { + add_range(&ranges, ranges.size(), bucket, bucket_num); + } + } + RuntimeFilterBucketPruner pruner; + int64_t newly_pruned = 0; + ASSERT_TRUE(pruner.prune_by_runtime_filters(ranges, conjuncts, rf_descs, SCAN_NODE_ID, 1024, + &newly_pruned) + .ok()); + EXPECT_EQ(newly_pruned, expected_pruned); + for (const auto& [bucket_num, expected] : expected_by_num) { + auto buckets = wrapper->get_or_compute_bucket_prune_hashes( + std::make_shared(), TDistributionHashType::IDENTITY, bucket_num); + EXPECT_EQ(std::set(buckets->begin(), buckets->end()), expected); + EXPECT_EQ(buckets->size(), expected.size()); + for (int32_t bucket = 0; bucket < bucket_num; ++bucket) { + EXPECT_EQ(pruner.is_bucket_pruned(bucket, bucket_num), !expected.contains(bucket)); + } + } +} + TEST_F(RuntimeFilterBucketPrunerTest, ExactInKeepsOnlyMatchingBucket) { constexpr int filter_id = 7; constexpr int32_t value = 10; diff --git a/be/test/exec/runtime_filter/runtime_filter_wrapper_test.cpp b/be/test/exec/runtime_filter/runtime_filter_wrapper_test.cpp index 2bac50f3cad6c6..4c3ac7bc7bc644 100644 --- a/be/test/exec/runtime_filter/runtime_filter_wrapper_test.cpp +++ b/be/test/exec/runtime_filter/runtime_filter_wrapper_test.cpp @@ -275,9 +275,20 @@ TEST_F(RuntimeFilterWrapperTest, DateInFilterRoundTripPreservesBucketHash) { EXPECT_EQ(assigned_date->type(), TIME_DATE); EXPECT_EQ(*assigned_date, date); - auto hashes = consumer->get_or_compute_bucket_prune_hashes(std::make_shared()); - ASSERT_EQ(hashes->size(), 1); - EXPECT_EQ(hashes->front(), RawValue::zlib_crc32(&date, sizeof(date), TYPE_DATE, 0)); + auto target_type = std::make_shared(); + // Verify both algorithms against the original DATE across non-power-of-two bucket counts. + for (uint32_t bucket_num : {3U, 7U, 97U}) { + auto hashes = consumer->get_or_compute_bucket_prune_hashes( + target_type, TDistributionHashType::CRC32, bucket_num); + ASSERT_EQ(hashes->size(), 1); + EXPECT_EQ(hashes->front(), RawValue::zlib_crc32(&date, sizeof(date), TYPE_DATE, 0)); + + auto buckets = consumer->get_or_compute_bucket_prune_hashes( + target_type, TDistributionHashType::IDENTITY, bucket_num); + ASSERT_EQ(buckets->size(), 1); + EXPECT_EQ(buckets->front(), + RawValue::identity_hash(&date, sizeof(date), TYPE_DATE, 0, bucket_num)); + } } TEST_F(RuntimeFilterWrapperTest, TestMinMaxAssign) { diff --git a/be/test/exec/sink/sink_test_utils.h b/be/test/exec/sink/sink_test_utils.h index 9e653549f906b8..ca8f92f953a37c 100644 --- a/be/test/exec/sink/sink_test_utils.h +++ b/be/test/exec/sink/sink_test_utils.h @@ -222,6 +222,40 @@ inline TOlapTableLocationParam build_location_param() { return location; } +// A single range partition [-1000, 1000) with `num_buckets` tablets (ids 300, 301, ...), +// bucketed by the integer column "c1" using the given distribution hash type. The range spans +// negatives so identity's unsigned two's-complement byte handling can be exercised end-to-end. +inline TOlapTablePartitionParam build_single_col_partition_param( + int64_t schema_index_id, int32_t num_buckets, TDistributionHashType::type hash_type) { + TOlapTablePartitionParam param; + param.db_id = 1; + param.table_id = 2; + param.version = 0; + + param.__set_partition_type(TPartitionType::RANGE_PARTITIONED); + param.__set_partition_columns({"c1"}); + param.__set_distributed_columns({"c1"}); + param.__set_distribution_hash_type(hash_type); + + TOlapTablePartition p1; + p1.id = 1; + p1.num_buckets = num_buckets; + p1.__set_is_mutable(true); + { + TOlapTableIndexTablets index_tablets; + index_tablets.index_id = schema_index_id; + for (int32_t i = 0; i < num_buckets; i++) { + index_tablets.tablets.push_back(300 + i); + } + p1.indexes = {index_tablets}; + } + p1.__set_start_keys({make_int_literal(-1000)}); + p1.__set_end_keys({make_int_literal(1000)}); + + param.partitions = {p1}; + return param; +} + } // namespace sink_test_utils } // namespace doris diff --git a/be/test/exec/sink/tablet_sink_hash_partitioner_test.cpp b/be/test/exec/sink/tablet_sink_hash_partitioner_test.cpp index e1e4e7b7716f71..4ae868b9dddad9 100644 --- a/be/test/exec/sink/tablet_sink_hash_partitioner_test.cpp +++ b/be/test/exec/sink/tablet_sink_hash_partitioner_test.cpp @@ -340,6 +340,148 @@ TEST(TabletSinkHashPartitionerTest, OlapTabletFinderRoundRobinEveryBatch) { } } +// identity distribution_hash_type: canonical bytes are interpreted as unsigned, bit-identical +// with FE pruning. +TEST(TabletSinkHashPartitionerTest, IdentityBucketingModsValueByNumBuckets) { + OperatorContext ctx; + constexpr int32_t num_buckets = 8; + + TOlapTableSchemaParam tschema; + TTupleId tablet_sink_tuple_id = 0; + int64_t schema_index_id = 0; + sink_test_utils::build_desc_tbl_and_schema(ctx, tschema, tablet_sink_tuple_id, schema_index_id, + false); + + auto schema = std::make_shared(); + auto st = schema->init(tschema); + ASSERT_TRUE(st.ok()) << st.to_string(); + + auto tpartition = sink_test_utils::build_single_col_partition_param( + schema_index_id, num_buckets, TDistributionHashType::IDENTITY); + auto vpartition = std::make_unique(schema, tpartition); + st = vpartition->init(); + ASSERT_TRUE(st.ok()) << st.to_string(); + + OlapTabletFinder finder(vpartition.get(), + OlapTabletFinder::FindTabletMode::FIND_TABLET_EVERY_ROW); + + // 3 -> 3, 8 -> 0, 100 -> 4, 999 -> 7, -1 -> 7, -8 -> 0. + auto block = ColumnHelper::create_block({3, 8, 100, 999, -1, -8}); + std::vector partitions(block.rows(), nullptr); + std::vector tablet_index(block.rows(), 0); + std::vector skip(block.rows(), false); + + st = finder.find_tablets(&ctx.state, &block, cast_set(block.rows()), partitions, + tablet_index, skip, nullptr); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_EQ(tablet_index[0], 3U); + EXPECT_EQ(tablet_index[1], 0U); + EXPECT_EQ(tablet_index[2], 4U); + EXPECT_EQ(tablet_index[3], 7U); + EXPECT_EQ(tablet_index[4], 7U); // UINT32_MAX % 8 + EXPECT_EQ(tablet_index[5], 0U); // (UINT32_MAX - 7) % 8 +} + +// identity with a null distribution value falls into bucket 0 (FE/BE write the same rule). +TEST(TabletSinkHashPartitionerTest, IdentityNullGoesToBucketZero) { + OperatorContext ctx; + constexpr int32_t num_buckets = 8; + + TOlapTableSchemaParam tschema; + TTupleId tablet_sink_tuple_id = 0; + int64_t schema_index_id = 0; + // nullable distribution column + sink_test_utils::build_desc_tbl_and_schema(ctx, tschema, tablet_sink_tuple_id, schema_index_id, + true); + + auto schema = std::make_shared(); + auto st = schema->init(tschema); + ASSERT_TRUE(st.ok()) << st.to_string(); + + auto tpartition = sink_test_utils::build_single_col_partition_param( + schema_index_id, num_buckets, TDistributionHashType::IDENTITY); + auto vpartition = std::make_unique(schema, tpartition); + st = vpartition->init(); + ASSERT_TRUE(st.ok()) << st.to_string(); + + OlapTabletFinder finder(vpartition.get(), + OlapTabletFinder::FindTabletMode::FIND_TABLET_EVERY_ROW); + + // row 0 null -> bucket 0; row 1 = 300 -> 300 % 8 = 4 + auto block = ColumnHelper::create_nullable_block({0, 300}, {1, 0}); + std::vector partitions(block.rows(), nullptr); + std::vector tablet_index(block.rows(), 0); + std::vector skip(block.rows(), false); + + st = finder.find_tablets(&ctx.state, &block, cast_set(block.rows()), partitions, + tablet_index, skip, nullptr); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_EQ(tablet_index[0], 0U); // null -> 0 + EXPECT_EQ(tablet_index[1], 4U); // 300 % 8 = 4 +} + +// crc32 (default) must NOT collapse to value % n; guards the two branches from being swapped. +TEST(TabletSinkHashPartitionerTest, Crc32DiffersFromIdentity) { + OperatorContext ctx; + constexpr int32_t num_buckets = 8; + + TOlapTableSchemaParam tschema; + TTupleId tablet_sink_tuple_id = 0; + int64_t schema_index_id = 0; + sink_test_utils::build_desc_tbl_and_schema(ctx, tschema, tablet_sink_tuple_id, schema_index_id, + false); + + auto schema = std::make_shared(); + auto st = schema->init(tschema); + ASSERT_TRUE(st.ok()) << st.to_string(); + + std::vector values = {3, 8, 100, 999, 5, 6, 7, 12}; + + auto identity_param = sink_test_utils::build_single_col_partition_param( + schema_index_id, num_buckets, TDistributionHashType::IDENTITY); + auto identity_part = std::make_unique(schema, identity_param); + ASSERT_TRUE(identity_part->init().ok()); + OlapTabletFinder identity_finder(identity_part.get(), + OlapTabletFinder::FindTabletMode::FIND_TABLET_EVERY_ROW); + auto block1 = ColumnHelper::create_block(values); + std::vector parts1(block1.rows(), nullptr); + std::vector identity_index(block1.rows(), 0); + std::vector skip1(block1.rows(), false); + ASSERT_TRUE(identity_finder + .find_tablets(&ctx.state, &block1, cast_set(block1.rows()), parts1, + identity_index, skip1, nullptr) + .ok()); + + auto crc32_param = sink_test_utils::build_single_col_partition_param( + schema_index_id, num_buckets, TDistributionHashType::CRC32); + auto crc32_part = std::make_unique(schema, crc32_param); + ASSERT_TRUE(crc32_part->init().ok()); + OlapTabletFinder crc32_finder(crc32_part.get(), + OlapTabletFinder::FindTabletMode::FIND_TABLET_EVERY_ROW); + auto block2 = ColumnHelper::create_block(values); + std::vector parts2(block2.rows(), nullptr); + std::vector crc32_index(block2.rows(), 0); + std::vector skip2(block2.rows(), false); + ASSERT_TRUE(crc32_finder + .find_tablets(&ctx.state, &block2, cast_set(block2.rows()), parts2, + crc32_index, skip2, nullptr) + .ok()); + + // identity locates value % n; crc32 must differ for at least one row. + for (size_t i = 0; i < values.size(); i++) { + EXPECT_EQ(identity_index[i], + static_cast(((values[i] % num_buckets) + num_buckets) % num_buckets)); + } + bool differs = false; + for (size_t i = 0; i < values.size(); i++) { + if (crc32_index[i] != identity_index[i]) { + differs = true; + break; + } + } + EXPECT_TRUE(differs); +} + TEST(TabletSinkHashPartitionerTest, TimeStampNsRangePartitionKey) { OperatorContext ctx; diff --git a/be/test/exprs/function/identity_hash_internal_function_test.cpp b/be/test/exprs/function/identity_hash_internal_function_test.cpp new file mode 100644 index 00000000000000..e0e9418c57cb89 --- /dev/null +++ b/be/test/exprs/function/identity_hash_internal_function_test.cpp @@ -0,0 +1,135 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +#include + +#include +#include + +#include "agent/be_exec_version_manager.h" +#include "common/status.h" +#include "core/block/block.h" +#include "core/block/column_with_type_and_name.h" +#include "core/column/column_const.h" +#include "core/column/column_vector.h" +#include "core/data_type/data_type_number.h" +#include "exprs/function/simple_function_factory.h" +#include "exprs/function_context.h" +#include "runtime/runtime_state.h" +#include "testutil/mock/mock_runtime_state.h" +#include "util/raw_value.h" + +namespace doris { + +// Unit tests for FunctionIdentityHashInternal's execute_impl defensive validation. FE rejects +// malformed calls at analysis time; the BE checks guard against any path that delivers a +// non-constant, missing, or non-positive bucket count constant (e.g. forged thrift), and must +// return an error instead of crashing (the previous DCHECK-based guards aborted the process). +class IdentityHashInternalFunctionTest : public ::testing::Test { +protected: + void SetUp() override { + _fn = SimpleFunctionFactory::instance().get_function( + "identity_hash_internal", _block.get_columns_with_type_and_name(), _return_type, {}, + BeExecVersionManager::get_newest_version()); + ASSERT_NE(_fn, nullptr); + } + + // A non-null, non-const Int32 argument column. + static ColumnWithTypeAndName make_int32_column(const std::string& name, + const std::vector& values) { + auto column = ColumnInt32::create(); + for (auto v : values) { + column->insert_value(v); + } + ColumnPtr result = std::move(column); + return {std::move(result), std::make_shared(), name}; + } + + // The trailing bucket-count argument as a BIGINT constant (what FE delivers for a literal). + static ColumnWithTypeAndName make_bigint_const(int64_t value) { + auto column = ColumnInt64::create(); + column->insert_value(value); + ColumnPtr result = std::move(column); + result = ColumnConst::create(result, 1); + return {std::move(result), std::make_shared(), "mod"}; + } + + Status execute(const ColumnNumbers& arguments, uint32_t result) { + auto context = FunctionContext::create_context(&_state, _return_type, _argument_types); + context->set_constant_cols(_constant_cols); + _block.insert({nullptr, _return_type, "result"}); + _status = _fn->execute(context.get(), _block, arguments, result, 3); + return _status; + } + + void set_constant_col(size_t index, const ColumnPtr& column) { + if (_constant_cols.size() <= index) { + _constant_cols.resize(index + 1); + } + _constant_cols[index] = std::make_shared(column); + } + + DataTypes _argument_types {std::make_shared(), + std::make_shared()}; + DataTypePtr _return_type = std::make_shared(); + FunctionBasePtr _fn; + MockRuntimeState _state; + Block _block {make_int32_column("c1", {1, 2, 3}), make_bigint_const(8)}; + std::vector> _constant_cols; + Status _status; +}; + +// The happy path: a valid constant bucket count still computes the identity bucket. +TEST_F(IdentityHashInternalFunctionTest, ValidConstantBucketCount) { + set_constant_col(1, _block.get_by_position(1).column); + ASSERT_TRUE(execute({0, 1}, 2).ok()) << _status.to_string(); + // identity bucket of values 1,2,3 with mod 8: 1%8, 2%8, 3%8 + const auto* res = assert_cast(_block.get_by_position(2).column.get()); + ASSERT_EQ(res->size(), 3); + EXPECT_EQ(res->get_element(0), 1); + EXPECT_EQ(res->get_element(1), 2); + EXPECT_EQ(res->get_element(2), 3); +} + +// Non-positive bucket counts must return an error, not crash (previously DCHECK abort / UB). +TEST_F(IdentityHashInternalFunctionTest, RejectsZeroBucketCount) { + _block = ::doris::Block {make_int32_column("c1", {1, 2, 3}), make_bigint_const(0)}; + set_constant_col(1, _block.get_by_position(1).column); + auto st = execute({0, 1}, 2); + ASSERT_FALSE(st.ok()); + EXPECT_TRUE(st.is()); + EXPECT_NE(st.to_string().find("positive integer"), std::string::npos) << st.to_string(); +} + +TEST_F(IdentityHashInternalFunctionTest, RejectsNegativeBucketCount) { + _block = ::doris::Block {make_int32_column("c1", {1, 2, 3}), make_bigint_const(-8)}; + set_constant_col(1, _block.get_by_position(1).column); + auto st = execute({0, 1}, 2); + ASSERT_FALSE(st.ok()); + EXPECT_TRUE(st.is()); + EXPECT_NE(st.to_string().find("positive integer"), std::string::npos) << st.to_string(); +} + +// A missing (nullptr) constant column for the bucket count must return an error, not +// dereference nullptr (previously the DCHECK guarded release builds nowhere). +TEST_F(IdentityHashInternalFunctionTest, RejectsMissingConstantColumn) { + auto st = execute({0, 1}, 2); + ASSERT_FALSE(st.ok()); + EXPECT_TRUE(st.is()); +} + +} // namespace doris diff --git a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv4Literal.java b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv4Literal.java index 1c57e69cf9b0dd..3d34b6523b15f3 100644 --- a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv4Literal.java +++ b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv4Literal.java @@ -17,11 +17,15 @@ package org.apache.doris.analysis; +import org.apache.doris.catalog.PrimitiveType; import org.apache.doris.catalog.Type; import org.apache.doris.common.AnalysisException; import com.google.gson.annotations.SerializedName; +import java.nio.ByteBuffer; +import java.nio.ByteOrder; + public class IPv4Literal extends LiteralExpr { public static final long IPV4_MIN = 0L; // 0.0.0.0 @@ -159,6 +163,14 @@ public String getStringValue() { return parseLongToIPv4(this.value); } + @Override + public ByteBuffer getHashValue(PrimitiveType type) { + ByteBuffer buffer = ByteBuffer.allocate(Integer.BYTES).order(ByteOrder.LITTLE_ENDIAN); + buffer.putInt((int) value); + buffer.flip(); + return buffer; + } + public long getValue() { return value; } diff --git a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv6Literal.java b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv6Literal.java index fb9a06b7ac8847..265be262338aa7 100644 --- a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv6Literal.java +++ b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/IPv6Literal.java @@ -17,12 +17,14 @@ package org.apache.doris.analysis; +import org.apache.doris.catalog.PrimitiveType; import org.apache.doris.catalog.Type; import org.apache.doris.common.AnalysisException; import com.google.gson.annotations.SerializedName; import com.googlecode.ipv6.IPv6Address; +import java.nio.ByteBuffer; import java.util.regex.Pattern; public class IPv6Literal extends LiteralExpr { @@ -141,6 +143,17 @@ public String getStringValue() { return this.value; } + @Override + public ByteBuffer getHashValue(PrimitiveType type) { + byte[] networkOrder = parseAddress(value).toByteArray(); + ByteBuffer buffer = ByteBuffer.allocate(networkOrder.length); + for (int i = networkOrder.length - 1; i >= 0; i--) { + buffer.put(networkOrder[i]); + } + buffer.flip(); + return buffer; + } + public String getValue() { return value; } diff --git a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/TimeV2Literal.java b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/TimeV2Literal.java index 96a4014bd59a00..b01e73f814a170 100644 --- a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/TimeV2Literal.java +++ b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/TimeV2Literal.java @@ -17,9 +17,13 @@ package org.apache.doris.analysis; +import org.apache.doris.catalog.PrimitiveType; import org.apache.doris.catalog.ScalarType; import org.apache.doris.catalog.Type; +import java.nio.ByteBuffer; +import java.nio.ByteOrder; + public class TimeV2Literal extends LiteralExpr { public static final TimeV2Literal MIN_VALUE = new TimeV2Literal(838, 59, 59, 999999, 6, true); public static final TimeV2Literal MAX_VALUE = new TimeV2Literal(838, 59, 59, 999999, 6, false); @@ -126,6 +130,14 @@ public String getStringValue() { return sb.toString(); } + @Override + public ByteBuffer getHashValue(PrimitiveType type) { + ByteBuffer buffer = ByteBuffer.allocate(Double.BYTES).order(ByteOrder.LITTLE_ENDIAN); + buffer.putDouble(getValue()); + buffer.flip(); + return buffer; + } + protected static boolean checkRange(int hour, int minute, int second, int microsecond) { return hour > 838 || minute > 59 || second > 59 || microsecond > 999999 || minute < 0 || second < 0 || microsecond < 0; diff --git a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/VarBinaryLiteral.java b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/VarBinaryLiteral.java index b081e7be17f0ba..9cda95a4697aa7 100644 --- a/fe/fe-catalog/src/main/java/org/apache/doris/analysis/VarBinaryLiteral.java +++ b/fe/fe-catalog/src/main/java/org/apache/doris/analysis/VarBinaryLiteral.java @@ -17,12 +17,14 @@ package org.apache.doris.analysis; +import org.apache.doris.catalog.PrimitiveType; import org.apache.doris.catalog.Type; import org.apache.doris.common.AnalysisException; import com.google.common.io.BaseEncoding; import com.google.gson.annotations.SerializedName; +import java.nio.ByteBuffer; import java.nio.charset.StandardCharsets; public class VarBinaryLiteral extends LiteralExpr { @@ -115,6 +117,11 @@ public int compareLiteral(LiteralExpr other) { + this + " (" + this.type + ") vs " + other + " (" + ((LiteralExpr) other).type + ")"); } + @Override + public ByteBuffer getHashValue(PrimitiveType type) { + return ByteBuffer.wrap(value); + } + @Override public String getStringValue() { return new String(value, StandardCharsets.ISO_8859_1); diff --git a/fe/fe-common/src/main/java/org/apache/doris/common/Config.java b/fe/fe-common/src/main/java/org/apache/doris/common/Config.java index c13a29dc507846..66d9654c49ac8e 100644 --- a/fe/fe-common/src/main/java/org/apache/doris/common/Config.java +++ b/fe/fe-common/src/main/java/org/apache/doris/common/Config.java @@ -2012,9 +2012,10 @@ public class Config extends ConfigBase { public static final int TIMESTAMP_NS_MIN_BE_EXEC_VERSION = 14; // Older backends ignore the optional OpenCSV flag and would silently use different row semantics. public static final int HIVE_OPEN_CSV_MIN_BE_EXEC_VERSION = 15; + public static final int DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION = 16; @ConfField(mutable = false) - public static int max_be_exec_version = HIVE_OPEN_CSV_MIN_BE_EXEC_VERSION; + public static int max_be_exec_version = DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION; /** * Min data version of backends serialize block. diff --git a/fe/fe-common/src/main/java/org/apache/doris/common/ErrorCode.java b/fe/fe-common/src/main/java/org/apache/doris/common/ErrorCode.java index bb7c1c994beb97..a0a95b32921d3e 100644 --- a/fe/fe-common/src/main/java/org/apache/doris/common/ErrorCode.java +++ b/fe/fe-common/src/main/java/org/apache/doris/common/ErrorCode.java @@ -1138,6 +1138,8 @@ public enum ErrorCode { "Colocate tables distribution columns size must be same: %s should be %s"), ERR_COLOCATE_TABLE_MUST_HAS_SAME_DISTRIBUTION_COLUMN_TYPE(5063, new byte[]{'4', '2', '0', '0', '0'}, "Colocate tables distribution columns must have the same data type: %s should be %s"), + ERR_COLOCATE_TABLE_MUST_HAS_SAME_DISTRIBUTION_HASH_TYPE(5063, new byte[]{'4', '2', '0', '0', '0'}, + "Colocate tables must have same distribution hash type: %s should be %s"), ERR_COLOCATE_NOT_COLOCATE_TABLE(5064, new byte[]{'4', '2', '0', '0', '0'}, "Table %s is not a colocated table"), ERR_INVALID_OPERATION(5065, new byte[]{'4', '2', '0', '0', '0'}, "Operation %s is invalid"), diff --git a/fe/fe-common/src/main/java/org/apache/doris/common/FeMetaVersion.java b/fe/fe-common/src/main/java/org/apache/doris/common/FeMetaVersion.java index 746ca81f6f1dc3..07b8e0a77eadfa 100644 --- a/fe/fe-common/src/main/java/org/apache/doris/common/FeMetaVersion.java +++ b/fe/fe-common/src/main/java/org/apache/doris/common/FeMetaVersion.java @@ -102,9 +102,11 @@ public final class FeMetaVersion { public static final int VERSION_139 = 139; public static final int VERSION_140 = 140; + // add group-level distribution_hash_type in ColocateGroupSchema + public static final int VERSION_141 = 141; // note: when increment meta version, should assign the latest version to VERSION_CURRENT - public static final int VERSION_CURRENT = VERSION_140; + public static final int VERSION_CURRENT = VERSION_141; // all logs meta version should >= the minimum version, so that we could remove many if clause, for example diff --git a/fe/fe-core/src/main/java/org/apache/doris/analysis/HashDistributionDesc.java b/fe/fe-core/src/main/java/org/apache/doris/analysis/HashDistributionDesc.java index 4509a71440c7d7..feb8a002a56dde 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/analysis/HashDistributionDesc.java +++ b/fe/fe-core/src/main/java/org/apache/doris/analysis/HashDistributionDesc.java @@ -20,6 +20,7 @@ import org.apache.doris.catalog.Column; import org.apache.doris.catalog.DistributionInfo; import org.apache.doris.catalog.HashDistributionInfo; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.catalog.KeysType; import org.apache.doris.common.AnalysisException; import org.apache.doris.common.DdlException; @@ -33,15 +34,25 @@ public class HashDistributionDesc extends DistributionDesc { private List distributionColumnNames; + private HashType hashType; public HashDistributionDesc(int numBucket, List distributionColumnNames) { super(numBucket); this.distributionColumnNames = distributionColumnNames; + this.hashType = HashType.CRC32; } public HashDistributionDesc(int numBucket, boolean autoBucket, List distributionColumnNames) { super(numBucket, autoBucket); this.distributionColumnNames = distributionColumnNames; + this.hashType = HashType.CRC32; + } + + public HashDistributionDesc(int numBucket, boolean autoBucket, List distributionColumnNames, + HashType hashType) { + super(numBucket, autoBucket); + this.distributionColumnNames = distributionColumnNames; + this.hashType = hashType; } @Override @@ -126,13 +137,16 @@ public DistributionInfo toDistributionInfo(List columns) throws DdlExcep } } - HashDistributionInfo hashDistributionInfo = - new HashDistributionInfo(numBucket, autoBucket, distributionColumns); + HashDistributionInfo hashDistributionInfo + = new HashDistributionInfo(numBucket, autoBucket, distributionColumns, hashType); return hashDistributionInfo; } @Override public DistributionDescriptor toDistributionDescriptor() { - return new DistributionDescriptor(true, this.autoBucket, this.numBucket, this.distributionColumnNames); + DistributionDescriptor descriptor + = new DistributionDescriptor(true, this.autoBucket, this.numBucket, this.distributionColumnNames); + descriptor.updateHashType(hashType); + return descriptor; } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/BuiltinScalarFunctions.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/BuiltinScalarFunctions.java index a26f94f5ab946c..1310fa0f7e5150 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/BuiltinScalarFunctions.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/BuiltinScalarFunctions.java @@ -253,6 +253,7 @@ import org.apache.doris.nereids.trees.expressions.functions.scalar.HoursAdd; import org.apache.doris.nereids.trees.expressions.functions.scalar.HoursDiff; import org.apache.doris.nereids.trees.expressions.functions.scalar.HoursSub; +import org.apache.doris.nereids.trees.expressions.functions.scalar.IdentityHashInternal; import org.apache.doris.nereids.trees.expressions.functions.scalar.If; import org.apache.doris.nereids.trees.expressions.functions.scalar.Ignore; import org.apache.doris.nereids.trees.expressions.functions.scalar.Initcap; @@ -923,6 +924,7 @@ public class BuiltinScalarFunctions implements FunctionHelper { scalar(Levenshtein.class, "levenshtein", "levenshtein_distance", "edit_distance"), scalar(Crc32.class, "crc32"), scalar(Crc32Internal.class, "crc32_internal"), + scalar(IdentityHashInternal.class, "identity_hash_internal"), scalar(Like.class, "like"), scalar(Ln.class, "ln", "dlog1"), scalar(Locate.class, "position", "locate"), diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateGroupSchema.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateGroupSchema.java index 0860f89eb169b4..4d9f0f7737a603 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateGroupSchema.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateGroupSchema.java @@ -21,6 +21,8 @@ import org.apache.doris.common.DdlException; import org.apache.doris.common.ErrorCode; import org.apache.doris.common.ErrorReport; +import org.apache.doris.common.FeMetaVersion; +import org.apache.doris.common.io.Text; import org.apache.doris.common.io.Writable; import com.google.common.collect.Lists; @@ -45,17 +47,25 @@ public class ColocateGroupSchema implements Writable { private int bucketsNum; @SerializedName(value = "replicaAlloc") private ReplicaAllocation replicaAlloc; + @SerializedName(value = "hashType") + private HashDistributionInfo.HashType hashType; private ColocateGroupSchema() { } - public ColocateGroupSchema(GroupId groupId, List distributionCols, - int bucketsNum, ReplicaAllocation replicaAlloc) { + public ColocateGroupSchema(GroupId groupId, List distributionCols, int bucketsNum, + ReplicaAllocation replicaAlloc) { + this(groupId, distributionCols, bucketsNum, replicaAlloc, HashDistributionInfo.HashType.CRC32); + } + + public ColocateGroupSchema(GroupId groupId, List distributionCols, int bucketsNum, + ReplicaAllocation replicaAlloc, HashDistributionInfo.HashType hashType) { this.groupId = groupId; this.distributionColTypes = distributionCols.stream().map(c -> c.getType()).collect(Collectors.toList()); this.bucketsNum = bucketsNum; this.replicaAlloc = replicaAlloc; + this.hashType = hashType; } public GroupId getGroupId() { @@ -78,6 +88,12 @@ public List getDistributionColTypes() { return distributionColTypes; } + public HashDistributionInfo.HashType getHashType() { + return hashType == null + ? HashDistributionInfo.HashType.CRC32 + : hashType; + } + public void checkColocateSchema(OlapTable tbl) throws DdlException { checkDistribution(tbl.getDefaultDistributionInfo()); // We add a table with many partitions to the colocate group, @@ -91,6 +107,11 @@ public void checkColocateSchema(OlapTable tbl) throws DdlException { public void checkDistribution(DistributionInfo distributionInfo) throws DdlException { if (distributionInfo instanceof HashDistributionInfo) { HashDistributionInfo info = (HashDistributionInfo) distributionInfo; + // hash type + if (info.getHashType() != getHashType()) { + ErrorReport.reportDdlException(ErrorCode.ERR_COLOCATE_TABLE_MUST_HAS_SAME_DISTRIBUTION_HASH_TYPE, + info.getHashType(), getHashType()); + } // buckets num if (info.getBucketNum() != bucketsNum) { ErrorReport.reportDdlException(ErrorCode.ERR_COLOCATE_TABLE_MUST_HAS_SAME_BUCKET_NUM, @@ -159,6 +180,7 @@ public void write(DataOutput out) throws IOException { } out.writeInt(bucketsNum); this.replicaAlloc.write(out); + Text.writeString(out, getHashType().name()); } public void readFields(DataInput in) throws IOException { @@ -169,5 +191,10 @@ public void readFields(DataInput in) throws IOException { } bucketsNum = in.readInt(); this.replicaAlloc = ReplicaAllocation.read(in); + if (Env.getCurrentEnvJournalVersion() >= FeMetaVersion.VERSION_141) { + this.hashType = HashDistributionInfo.HashType.valueOf(Text.readString(in)); + } else { + this.hashType = HashDistributionInfo.HashType.CRC32; + } } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateTableIndex.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateTableIndex.java index 29ef3be84d9d67..931efec49f67fe 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateTableIndex.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/ColocateTableIndex.java @@ -223,7 +223,7 @@ public GroupId addTableToGroup(long dbId, OlapTable tbl, String fullGroupName, G HashDistributionInfo distributionInfo = (HashDistributionInfo) tbl.getDefaultDistributionInfo(); ColocateGroupSchema groupSchema = new ColocateGroupSchema(groupId, distributionInfo.getDistributionColumns(), distributionInfo.getBucketNum(), - tbl.getDefaultReplicaAllocation()); + tbl.getDefaultReplicaAllocation(), distributionInfo.getHashType()); groupName2Id.put(fullGroupName, groupId); group2Schema.put(groupId, groupSchema); group2ErrMsgs.put(groupId, ""); diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/Env.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/Env.java index 63c3bf15ec3aea..0094a4b7297b07 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/Env.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/Env.java @@ -4022,6 +4022,14 @@ private static void addOlapTablePropertyInfo(OlapTable olapTable, StringBuilder sb.append(colocateTable).append("\""); } + // distribution hash type (only emit when non-default to keep output stable) + DistributionInfo defaultDistInfo = olapTable.getDefaultDistributionInfo(); + if (defaultDistInfo instanceof HashDistributionInfo + && ((HashDistributionInfo) defaultDistInfo).getHashType() != HashDistributionInfo.HashType.CRC32) { + sb.append(",\n\"").append(PropertyAnalyzer.PROPERTIES_DISTRIBUTION_HASH_TYPE).append("\" = \""); + sb.append(((HashDistributionInfo) defaultDistInfo).getHashType().name().toLowerCase()).append("\""); + } + // dynamic partition if (olapTable.dynamicPartitionExists()) { sb.append(olapTable.getTableProperty().getDynamicPartitionProperty().getProperties(replicaAlloc)); diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/HashDistributionInfo.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/HashDistributionInfo.java index a1f4688cb66693..37f19c1e5646d8 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/HashDistributionInfo.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/HashDistributionInfo.java @@ -33,28 +33,58 @@ * Hash Distribution Info. */ public class HashDistributionInfo extends DistributionInfo { + + /** + * Hash function type used by HASH distribution to map rows to buckets. + * + * CRC32 (legacy behavior) is the default for backward compatibility. + */ + public enum HashType { + CRC32, IDENTITY; + } + @SerializedName(value = "distributionColumns") private List distributionColumns; + @SerializedName(value = "hashType") + private HashType hashType; + public HashDistributionInfo() { super(); this.distributionColumns = new ArrayList(); + this.hashType = HashType.CRC32; } public HashDistributionInfo(int bucketNum, List distributionColumns) { super(DistributionInfoType.HASH, bucketNum); this.distributionColumns = distributionColumns; + this.hashType = HashType.CRC32; } public HashDistributionInfo(int bucketNum, boolean autoBucket, List distributionColumns) { super(DistributionInfoType.HASH, bucketNum, autoBucket); this.distributionColumns = distributionColumns; + this.hashType = HashType.CRC32; + } + + public HashDistributionInfo(int bucketNum, boolean autoBucket, List distributionColumns, + HashType hashType) { + super(DistributionInfoType.HASH, bucketNum, autoBucket); + this.distributionColumns = distributionColumns; + this.hashType = hashType; } public List getDistributionColumns() { return distributionColumns; } + // null-safe defense against old versions persisted before hashType existed. + public HashType getHashType() { + return hashType == null + ? HashType.CRC32 + : hashType; + } + public static void checkDistributionColumnType(String columnName, Type type) throws DdlException { if (type.isArrayType()) { throw new DdlException("Array Type should not be used in distribution column[" + columnName + "]."); @@ -101,12 +131,12 @@ public boolean equals(Object o) { return false; } HashDistributionInfo that = (HashDistributionInfo) o; - return bucketNum == that.bucketNum && sameDistributionColumns(that); + return bucketNum == that.bucketNum && sameDistributionColumns(that) && getHashType() == that.getHashType(); } @Override public int hashCode() { - return Objects.hash(super.hashCode(), distributionColumns, bucketNum); + return Objects.hash(super.hashCode(), distributionColumns, bucketNum, getHashType()); } @Override @@ -115,7 +145,8 @@ public DistributionDesc toDistributionDesc() { for (Column col : distributionColumns) { distriColNames.add(col.getName()); } - DistributionDesc distributionDesc = new HashDistributionDesc(bucketNum, autoBucket, distriColNames); + DistributionDesc distributionDesc + = new HashDistributionDesc(bucketNum, autoBucket, distriColNames, getHashType()); return distributionDesc; } @@ -169,4 +200,8 @@ public RandomDistributionInfo toRandomDistributionInfo() { public void setDistributionColumns(List column) { this.distributionColumns = column; } + + public void setHashType(HashType hashType) { + this.hashType = hashType; + } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/OlapTable.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/OlapTable.java index 4645a7796c9d99..e2f8243e6c04f4 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/OlapTable.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/OlapTable.java @@ -2026,6 +2026,16 @@ public String getSignature(int signatureVersion, List partNames) { sb.append(Util.getSchemaSignatureString(partitionColumns)); } + // Restore compares only intersecting partition names, which can be empty when appending + // partitions. The table-wide hash must still match because writes use the default layout. + // Keep the legacy signature unchanged for CRC32 and random distribution. + if (defaultDistributionInfo instanceof HashDistributionInfo) { + HashDistributionInfo.HashType hashType = ((HashDistributionInfo) defaultDistributionInfo).getHashType(); + if (hashType != HashDistributionInfo.HashType.CRC32) { + sb.append(hashType); + } + } + // partition and distribution Collections.sort(partNames, String.CASE_INSENSITIVE_ORDER); for (String partName : partNames) { @@ -2038,6 +2048,7 @@ public String getSignature(int signatureVersion, List partNames) { HashDistributionInfo hashDistributionInfo = (HashDistributionInfo) distributionInfo; sb.append(Util.getSchemaSignatureString(hashDistributionInfo.getDistributionColumns())); sb.append(hashDistributionInfo.getBucketNum()); + sb.append(hashDistributionInfo.getHashType()); } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/Partition.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/Partition.java index 20e7b73cdc3aab..036a38182bf32c 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/Partition.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/Partition.java @@ -301,6 +301,10 @@ public String getMetaChecksum() { updateMetaChecksum(digest, (byte) 17, distType == null ? -1L : distType.ordinal()); updateMetaChecksum(digest, (byte) 18, distributionInfo.getBucketNum()); updateMetaChecksum(digest, (byte) 19, distributionInfo.getAutoBucket() ? 1L : 0L); + if (distributionInfo instanceof HashDistributionInfo) { + updateMetaChecksum(digest, (byte) 20, + ((HashDistributionInfo) distributionInfo).getHashType().ordinal()); + } } else { updateMetaChecksum(digest, (byte) 17, -1L); } diff --git a/fe/fe-core/src/main/java/org/apache/doris/catalog/PartitionKey.java b/fe/fe-core/src/main/java/org/apache/doris/catalog/PartitionKey.java index 13f5a64f54822b..56eae5b65ba2b3 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/catalog/PartitionKey.java +++ b/fe/fe-core/src/main/java/org/apache/doris/catalog/PartitionKey.java @@ -257,6 +257,23 @@ public long getHashValue() { return hashValue.getValue(); } + /** + * Treat each distribution value's canonical bytes as an unsigned integer with the first byte + * as the least-significant byte, then append it to the preceding values. Keeping only the + * remainder avoids constructing an arbitrarily wide integer for multi-column keys. + */ + public int getIdentityHashValue(int hashMod) { + Preconditions.checkArgument(hashMod > 0, "hash modulus must be positive"); + long result = 0; + for (int keyIndex = 0; keyIndex < keys.size(); keyIndex++) { + ByteBuffer buffer = keys.get(keyIndex).getHashValue(types.get(keyIndex)); + for (int byteIndex = buffer.limit() - 1; byteIndex >= 0; byteIndex--) { + result = (result * 256 + Byte.toUnsignedInt(buffer.get(byteIndex))) % hashMod; + } + } + return (int) result; + } + public boolean isMinValue() { for (LiteralExpr literalExpr : keys) { if (!literalExpr.isMinValue()) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/common/util/PropertyAnalyzer.java b/fe/fe-core/src/main/java/org/apache/doris/common/util/PropertyAnalyzer.java index 9e26accffd13e6..0e123277c703e0 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/common/util/PropertyAnalyzer.java +++ b/fe/fe-core/src/main/java/org/apache/doris/common/util/PropertyAnalyzer.java @@ -27,6 +27,7 @@ import org.apache.doris.catalog.DatabaseIf; import org.apache.doris.catalog.Env; import org.apache.doris.catalog.EnvFactory; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.catalog.KeysType; import org.apache.doris.catalog.Partition; import org.apache.doris.catalog.PrimitiveType; @@ -117,6 +118,8 @@ public class PropertyAnalyzer { public static final String PROPERTIES_ENABLE_LIGHT_SCHEMA_CHANGE = "light_schema_change"; public static final String PROPERTIES_DISTRIBUTION_TYPE = "distribution_type"; + // hash function type when distribution_type is "HASH" + public static final String PROPERTIES_DISTRIBUTION_HASH_TYPE = "distribution_hash_type"; public static final String PROPERTIES_SEND_CLEAR_ALTER_TASK = "send_clear_alter_tasks"; /* * for upgrade alpha rowset to beta rowset, valid value: v1, v2 @@ -812,6 +815,25 @@ public static String analyzeColocate(Map properties) { return colocateGroup; } + // analyze the hash function type of table; defaults to CRC32 + public static HashDistributionInfo.HashType analyzeDistributionHashType(Map properties) + throws AnalysisException { + HashDistributionInfo.HashType hashType = HashDistributionInfo.HashType.CRC32; + if (properties != null && properties.containsKey(PROPERTIES_DISTRIBUTION_HASH_TYPE)) { + String value = properties.get(PROPERTIES_DISTRIBUTION_HASH_TYPE); + properties.remove(PROPERTIES_DISTRIBUTION_HASH_TYPE); + if (value.equalsIgnoreCase("crc32")) { + hashType = HashDistributionInfo.HashType.CRC32; + } else if (value.equalsIgnoreCase("identity")) { + hashType = HashDistributionInfo.HashType.IDENTITY; + } else { + throw new AnalysisException("Invalid " + PROPERTIES_DISTRIBUTION_HASH_TYPE + ": " + value + + ". Supported values are 'crc32' and 'identity'."); + } + } + return hashType; + } + public static long analyzeTimeout(Map properties, long defaultTimeout) throws AnalysisException { long timeout = defaultTimeout; if (properties != null && properties.containsKey(PROPERTIES_TIMEOUT)) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/datasource/InternalCatalog.java b/fe/fe-core/src/main/java/org/apache/doris/datasource/InternalCatalog.java index 1f3a52c24d7b75..a8daa5b804ae56 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/datasource/InternalCatalog.java +++ b/fe/fe-core/src/main/java/org/apache/doris/datasource/InternalCatalog.java @@ -1693,6 +1693,9 @@ public void addPartition(Database db, String tableName, AddPartitionOp addPartit + "new is: " + hashDistributionInfo.getDistributionColumns() + " default is: " + ((HashDistributionInfo) defaultDistributionInfo).getDistributionColumns()); } + // New partition inherits the table's hash type, otherwise BE would bucket rows with one + // hash function while FE prunes with another, making the data unreadable. + hashDistributionInfo.setHashType(((HashDistributionInfo) defaultDistributionInfo).getHashType()); } else if (distributionInfo.getType() == DistributionInfoType.RANDOM) { RandomDistributionInfo randomDistributionInfo = (RandomDistributionInfo) distributionInfo; if (randomDistributionInfo.getBucketNum() <= 0) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/PhysicalPlanTranslator.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/PhysicalPlanTranslator.java index 9bfe285a13e645..7ffcf4eef59b3d 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/PhysicalPlanTranslator.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/PhysicalPlanTranslator.java @@ -383,6 +383,7 @@ public PlanFragment visitPhysicalDistribute(PhysicalDistribute d // target data partition DataPartition targetDataPartition = toDataPartition(targetDistribution, validOutputIds, context); exchangeNode.setPartitionType(targetDataPartition.getType()); + exchangeNode.setDistributionHashType(targetDataPartition.getHashType()); exchangeNode.setDistributeExprLists(getDistributeExpr(distribute)); exchangeNode.setChildrenDistributeExprLists(upstreamDistributeExprs); // its source partition is targetDataPartition. and outputPartition is UNPARTITIONED now, will be set when @@ -3737,7 +3738,9 @@ private DataPartition toDataPartition(DistributionSpec distributionSpec/* target switch (distributionSpecHash.getShuffleType()) { case STORAGE_BUCKETED: partitionType = TPartitionType.BUCKET_SHFFULE_HASH_PARTITIONED; - break; + // Bucket-shuffle re-partitions the shuffled side to the target table's storage + // layout, so the storage hashType must ride along for BE to pick the right partitioner. + return new DataPartition(partitionType, partitionExprs, distributionSpecHash.getHashType()); case EXECUTION_BUCKETED: partitionType = TPartitionType.HASH_PARTITIONED; break; diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslator.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslator.java index 8912502d1f43ed..41c0937aeba61d 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslator.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslator.java @@ -372,7 +372,8 @@ private void setPruningMetadata(org.apache.doris.planner.RuntimeFilter runtimeFi runtimeFilter.setTargetPartitionMonotonicity( scanNode.getId(), nereidsFilter.getPartitionMonotonicity()); if (nereidsFilter.canPruneBuckets()) { - runtimeFilter.markTargetCanPruneBuckets(scanNode.getId()); + runtimeFilter.markTargetCanPruneBuckets( + scanNode.getId(), nereidsFilter.getBucketPruningHashType()); } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterContext.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterContext.java index 64a321ff03848f..7118fd3e06693b 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterContext.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterContext.java @@ -95,7 +95,8 @@ public void generateRuntimeFilterPruneMetadata(RuntimeFilter filter) { RuntimeFilterPruneClassifier.Classification classification = RuntimeFilterPruneClassifier.classify(filter, sessionVariable); filter.setPruningMetadata( - classification.canPruneBuckets(), classification.getPartitionMonotonicity()); + classification.canPruneBuckets(), classification.getBucketHashType(), + classification.getPartitionMonotonicity()); } public void setTargetExprIdToFilter(ExprId id, RuntimeFilter filter) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifier.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifier.java index 6aed407d96bdd0..7fd602ef89439f 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifier.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifier.java @@ -96,6 +96,7 @@ private static BucketClassification classifyBucketPruning(RuntimeFilter filter) } Column distributionColumn = null; + HashDistributionInfo.HashType distributionHashType = null; for (Long partitionId : scan.getSelectedPartitionIds()) { Partition partition = table.getPartition(partitionId); if (partition == null) { @@ -118,9 +119,15 @@ private static BucketClassification classifyBucketPruning(RuntimeFilter filter) return BucketClassification.unsupported( "selected partitions use different distribution columns"); } + if (distributionHashType != null + && distributionHashType != hashDistributionInfo.getHashType()) { + return BucketClassification.unsupported( + "selected partitions use different distribution hash types"); + } distributionColumn = currentDistributionColumn; + distributionHashType = hashDistributionInfo.getHashType(); } - return BucketClassification.supported(); + return BucketClassification.supported(distributionHashType); } private static PartitionClassification classifyPartitionPruning(RuntimeFilter filter) { @@ -408,6 +415,10 @@ Map getPartitionMonotonicity() { return partitionClassification.partitionMonotonicity; } + HashDistributionInfo.HashType getBucketHashType() { + return bucketClassification.hashType; + } + String getBucketUnsupportedReason() { return bucketClassification.unsupportedReason; } @@ -420,18 +431,21 @@ String getPartitionUnsupportedReason() { private static final class BucketClassification { private final boolean canPruneBuckets; private final String unsupportedReason; + private final HashDistributionInfo.HashType hashType; - private BucketClassification(boolean canPruneBuckets, String unsupportedReason) { + private BucketClassification(boolean canPruneBuckets, String unsupportedReason, + HashDistributionInfo.HashType hashType) { this.canPruneBuckets = canPruneBuckets; this.unsupportedReason = unsupportedReason; + this.hashType = hashType; } - private static BucketClassification supported() { - return new BucketClassification(true, ""); + private static BucketClassification supported(HashDistributionInfo.HashType hashType) { + return new BucketClassification(true, "", hashType); } private static BucketClassification unsupported(String reason) { - return new BucketClassification(false, reason); + return new BucketClassification(false, reason, null); } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/ShuffleKeyPruner.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/ShuffleKeyPruner.java index 39548da4b13694..06b4a440b35189 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/ShuffleKeyPruner.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/ShuffleKeyPruner.java @@ -489,7 +489,7 @@ private static PhysicalHashAggregate tryPruneGlobalAgg(PhysicalH private static DistributionSpecHash sliceHashSpec(DistributionSpecHash origin, List newOrderedKeys) { return new DistributionSpecHash(newOrderedKeys, origin.getShuffleType(), - origin.getTableId(), origin.getSelectedIndexId(), origin.getPartitionIds()); + origin.getTableId(), origin.getSelectedIndexId(), origin.getPartitionIds(), origin.getHashType()); } private static PhysicalDistribute rebuildDistribute(PhysicalDistribute origin, diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriver.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriver.java index 2df7723a7ab052..6cdb69782b5806 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriver.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriver.java @@ -17,6 +17,7 @@ package org.apache.doris.nereids.properties; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.nereids.PlanContext; import org.apache.doris.nereids.exceptions.AnalysisException; import org.apache.doris.nereids.memo.GroupExpression; @@ -453,6 +454,7 @@ public PhysicalProperties visitPhysicalPartitionTopN(PhysicalPartitionTopN childrenDistribution = childrenOutputProperties.stream() .map(PhysicalProperties::getDistributionSpec) .collect(Collectors.toList()); @@ -515,8 +517,12 @@ public PhysicalProperties visitPhysicalSetOperation(PhysicalSetOperation setOper } setOperationDistributeColumnIds.add(setOperation.getOutput().get(index).getExprId()); } - // check whether the set operation output all distribution columns of the child - if (setOperationDistributeColumnIds.size() == orderedShuffledColumns.size()) { + // check whether the set operation output all distribution columns of the child. + // An empty result must fall through instead of advertising a zero-key hash spec: + // containsSatisfy() is vacuously true on the empty equivalence map, so such a spec + // satisfies any hash REQUIRE demand and suppresses the parent's exchange. + if (setOperationDistributeColumnIds.size() == orderedShuffledColumns.size() + && !setOperationDistributeColumnIds.isEmpty()) { // Keep the basic child's specific storage layout as the set operation output. When // the basic child is on the right (shuffleToRight) the output rows are physically // placed by the right child's storage bucket function, so advertising that layout is @@ -531,7 +537,8 @@ public PhysicalProperties visitPhysicalSetOperation(PhysicalSetOperation setOper childDistribution.getShuffleType(), childDistribution.getTableId(), childDistribution.getSelectedIndexId(), - childDistribution.getPartitionIds() + childDistribution.getPartitionIds(), + childDistribution.getHashType() ) ); } @@ -561,10 +568,12 @@ public PhysicalProperties visitPhysicalSetOperation(PhysicalSetOperation setOper } } if (offsetsOfFirstChild == null) { - firstType = ((DistributionSpecHash) childDistribution).getShuffleType(); + firstType = distributionSpecHash.getShuffleType(); + firstHashType = distributionSpecHash.getHashType(); offsetsOfFirstChild = offsetsOfCurrentChild; } else if (!Arrays.equals(offsetsOfFirstChild, offsetsOfCurrentChild) - || firstType != ((DistributionSpecHash) childDistribution).getShuffleType()) { + || firstType != distributionSpecHash.getShuffleType() + || firstHashType != distributionSpecHash.getHashType()) { // NOTICE: if come here, the first child output must be DistributionSpecHash return PhysicalProperties.createAnyFromHash((DistributionSpecHash) childrenDistribution.get(0)); } @@ -574,7 +583,11 @@ public PhysicalProperties visitPhysicalSetOperation(PhysicalSetOperation setOper for (int offset : offsetsOfFirstChild) { request.add(setOperation.getOutput().get(offset).getExprId()); } - return PhysicalProperties.createHash(request, firstType); + // Keep createHash's empty-key normalization: offsetsOfFirstChild is empty only when the + // first child has no shuffled columns, and a zero-key DistributionSpecHash would satisfy + // any hash REQUIRE demand (containsSatisfy() is vacuously true on an empty equivalence + // map), suppressing an exchange the parent actually needs. + return PhysicalProperties.createHash(request, firstType, firstHashType); } @Override @@ -754,7 +767,8 @@ private DistributionSpecHash mockAnotherSideSpecFromConjuncts( } anotherSideOrderedExprIds.add(rightExprIds.get(index)); } - return new DistributionSpecHash(anotherSideOrderedExprIds, oneSideSpec.getShuffleType()); + return new DistributionSpecHash(anotherSideOrderedExprIds, oneSideSpec.getShuffleType(), + -1L, -1L, Collections.emptySet(), oneSideSpec.getHashType()); } private static boolean isSameHashValue(DataType originType, DataType castType) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildrenPropertiesRegulator.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildrenPropertiesRegulator.java index 9741d1de68a828..6d282588e25572 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildrenPropertiesRegulator.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/ChildrenPropertiesRegulator.java @@ -64,6 +64,7 @@ import org.apache.logging.log4j.Logger; import java.util.ArrayList; +import java.util.Collections; import java.util.List; import java.util.Optional; import java.util.Set; @@ -497,7 +498,8 @@ public List> visitPhysicalHashJoin( } else if (leftHashSpec.getShuffleType() == ShuffleType.NATURAL && rightHashSpec.getShuffleType() == ShuffleType.STORAGE_BUCKETED) { shouldCheckLeftBucketDownGrade = true; - if (!bothSideShuffleKeysAreSameOrder(leftHashSpec, rightHashSpec, + if (leftHashSpec.getHashType() != rightHashSpec.getHashType() + || !bothSideShuffleKeysAreSameOrder(leftHashSpec, rightHashSpec, (DistributionSpecHash) requiredProperties.get(0).getDistributionSpec(), (DistributionSpecHash) requiredProperties.get(1).getDistributionSpec())) { updatedForRight = Optional.of(calAnotherSideRequired( @@ -592,7 +594,8 @@ public List> visitPhysicalHashJoin( } else if ((leftHashSpec.getShuffleType() == ShuffleType.STORAGE_BUCKETED && rightHashSpec.getShuffleType() == ShuffleType.STORAGE_BUCKETED)) { - if (!bothSideShuffleKeysAreSameOrder(rightHashSpec, leftHashSpec, + if (leftHashSpec.getHashType() != rightHashSpec.getHashType() + || !bothSideShuffleKeysAreSameOrder(rightHashSpec, leftHashSpec, (DistributionSpecHash) requiredProperties.get(1).getDistributionSpec(), (DistributionSpecHash) requiredProperties.get(0).getDistributionSpec())) { if (children.get(0).getPlan() instanceof PhysicalDistribute) { @@ -786,7 +789,8 @@ && canMapBucketKeysToRequire((DistributionSpecHash) childDistribution, List shuffleSideIds = calAnotherSideRequiredShuffleIds( notNeedShuffleOutput, notShuffleSideRequire, currentRequire); PhysicalProperties target = new PhysicalProperties( - new DistributionSpecHash(shuffleSideIds, ShuffleType.STORAGE_BUCKETED)); + new DistributionSpecHash(shuffleSideIds, ShuffleType.STORAGE_BUCKETED, -1L, -1L, + Collections.emptySet(), notNeedShuffleOutput.getHashType())); updateChildEnforceAndCost(i, target); } } else { @@ -957,7 +961,7 @@ private PhysicalProperties calAnotherSideRequired(ShuffleType shuffleType, notNeedShuffleSideRequired, needShuffleSideRequired); return new PhysicalProperties(new DistributionSpecHash(shuffleSideIds, shuffleType, needShuffleSideOutput.getTableId(), needShuffleSideOutput.getSelectedIndexId(), - needShuffleSideOutput.getPartitionIds())); + needShuffleSideOutput.getPartitionIds(), notNeedShuffleSideOutput.getHashType())); } private void updateChildEnforceAndCost(int index, PhysicalProperties targetProperties) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/DistributionSpecHash.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/DistributionSpecHash.java index ab96960684a154..d738c6fc417571 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/DistributionSpecHash.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/DistributionSpecHash.java @@ -17,10 +17,12 @@ package org.apache.doris.nereids.properties; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.nereids.annotation.Developing; import org.apache.doris.nereids.trees.expressions.ExprId; import org.apache.doris.nereids.util.Utils; +import com.google.common.base.Preconditions; import com.google.common.collect.ImmutableList; import com.google.common.collect.ImmutableMap; import com.google.common.collect.ImmutableSet; @@ -54,6 +56,10 @@ public class DistributionSpecHash extends DistributionSpec { private final Set partitionIds; private final long selectedIndexId; + // storage bucketing hash function of the NATURAL side; only equal hashType tables may share + // a distribution (colocate / bucket-shuffle). Non-bucketing specs default to CRC32. + private final HashDistributionInfo.HashType hashType; + /** * Use for no need set table related attributes. */ @@ -69,11 +75,17 @@ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shu this(orderedShuffledColumns, shuffleType, tableId, -1L, partitionIds); } + public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shuffleType, long tableId, + long selectedIndexId, Set partitionIds) { + this(orderedShuffledColumns, shuffleType, tableId, selectedIndexId, partitionIds, + HashDistributionInfo.HashType.CRC32); + } + /** * Normal constructor. */ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shuffleType, - long tableId, long selectedIndexId, Set partitionIds) { + long tableId, long selectedIndexId, Set partitionIds, HashDistributionInfo.HashType hashType) { this.orderedShuffledColumns = ImmutableList.copyOf( Objects.requireNonNull(orderedShuffledColumns, "orderedShuffledColumns should not null")); this.shuffleType = Objects.requireNonNull(shuffleType, "shuffleType should not null"); @@ -81,6 +93,8 @@ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shu Objects.requireNonNull(partitionIds, "partitionIds should not null")); this.tableId = tableId; this.selectedIndexId = selectedIndexId; + this.hashType = normalizeHashType(shuffleType, + Objects.requireNonNull(hashType, "hashType should not null")); ImmutableList.Builder> equivalenceExprIdsBuilder = ImmutableList.builderWithExpectedSize(orderedShuffledColumns.size()); ImmutableMap.Builder exprIdToEquivalenceSetBuilder @@ -101,7 +115,14 @@ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shu long tableId, Set partitionIds, List> equivalenceExprIds, Map exprIdToEquivalenceSet) { this(orderedShuffledColumns, shuffleType, tableId, -1L, partitionIds, - equivalenceExprIds, exprIdToEquivalenceSet); + equivalenceExprIds, exprIdToEquivalenceSet, HashDistributionInfo.HashType.CRC32); + } + + public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shuffleType, long tableId, + long selectedIndexId, Set partitionIds, List> equivalenceExprIds, + Map exprIdToEquivalenceSet) { + this(orderedShuffledColumns, shuffleType, tableId, selectedIndexId, partitionIds, equivalenceExprIds, + exprIdToEquivalenceSet, HashDistributionInfo.HashType.CRC32); } /** @@ -109,12 +130,14 @@ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shu */ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shuffleType, long tableId, long selectedIndexId, Set partitionIds, List> equivalenceExprIds, - Map exprIdToEquivalenceSet) { + Map exprIdToEquivalenceSet, HashDistributionInfo.HashType hashType) { this.orderedShuffledColumns = ImmutableList.copyOf(Objects.requireNonNull(orderedShuffledColumns, "orderedShuffledColumns should not null")); this.shuffleType = Objects.requireNonNull(shuffleType, "shuffleType should not null"); this.tableId = tableId; this.selectedIndexId = selectedIndexId; + this.hashType = normalizeHashType(shuffleType, + Objects.requireNonNull(hashType, "hashType should not null")); this.partitionIds = ImmutableSet.copyOf( Objects.requireNonNull(partitionIds, "partitionIds should not null")); this.equivalenceExprIds = ImmutableList.copyOf( @@ -123,7 +146,20 @@ public DistributionSpecHash(List orderedShuffledColumns, ShuffleType shu Objects.requireNonNull(exprIdToEquivalenceSet, "exprIdToEquivalenceSet should not null")); } - static DistributionSpecHash merge(DistributionSpecHash left, DistributionSpecHash right, ShuffleType shuffleType) { + private static HashDistributionInfo.HashType normalizeHashType( + ShuffleType shuffleType, HashDistributionInfo.HashType hashType) { + // EXECUTION_BUCKETED is produced by the ordinary execution exchange, not by table storage + // bucketing. It must not retain an IDENTITY label inherited from a source table. + return shuffleType == ShuffleType.EXECUTION_BUCKETED + ? HashDistributionInfo.HashType.CRC32 + : hashType; + } + + static DistributionSpecHash merge(DistributionSpecHash left, DistributionSpecHash right, + ShuffleType shuffleType) { + Preconditions.checkState(left.hashType == right.hashType, + "can not merge distribution specs with different hash types: %s vs %s", + left.hashType, right.hashType); List orderedShuffledColumns = left.getOrderedShuffledColumns(); ImmutableList.Builder> equivalenceExprIds = ImmutableList.builderWithExpectedSize(orderedShuffledColumns.size()); @@ -140,7 +176,7 @@ static DistributionSpecHash merge(DistributionSpecHash left, DistributionSpecHas exprIdToEquivalenceSet.putAll(right.getExprIdToEquivalenceSet()); return new DistributionSpecHash(orderedShuffledColumns, shuffleType, left.getTableId(), left.getSelectedIndexId(), left.getPartitionIds(), equivalenceExprIds.build(), - exprIdToEquivalenceSet.buildKeepingLast()); + exprIdToEquivalenceSet.buildKeepingLast(), left.getHashType()); } static DistributionSpecHash merge(DistributionSpecHash left, DistributionSpecHash right) { @@ -163,6 +199,10 @@ public long getSelectedIndexId() { return selectedIndexId; } + public HashDistributionInfo.HashType getHashType() { + return hashType; + } + public Set getPartitionIds() { return partitionIds; } @@ -202,6 +242,7 @@ public boolean satisfy(DistributionSpec required) { return containsSatisfy(requiredHash.getOrderedShuffledColumns()); } return requiredHash.getShuffleType() == this.getShuffleType() + && this.hashType == requiredHash.hashType && equalsSatisfy(requiredHash.getOrderedShuffledColumns()); } @@ -229,12 +270,12 @@ private boolean equalsSatisfy(List required) { public DistributionSpecHash withShuffleType(ShuffleType shuffleType) { return new DistributionSpecHash(orderedShuffledColumns, shuffleType, tableId, selectedIndexId, partitionIds, - equivalenceExprIds, exprIdToEquivalenceSet); + equivalenceExprIds, exprIdToEquivalenceSet, hashType); } public DistributionSpecHash withShuffleTypeAndForbidColocateJoin(ShuffleType shuffleType) { return new DistributionSpecHash(orderedShuffledColumns, shuffleType, -1, -1, partitionIds, - equivalenceExprIds, exprIdToEquivalenceSet); + equivalenceExprIds, exprIdToEquivalenceSet, hashType); } /** @@ -266,7 +307,7 @@ public DistributionSpecHash withShuffleExprs(List prunedOrderedColumns) } return new DistributionSpecHash(ImmutableList.copyOf(prunedOrderedColumns), shuffleType, tableId, selectedIndexId, partitionIds, equivBuilder.build(), - mapBuilder.buildKeepingLast()); + mapBuilder.buildKeepingLast(), hashType); } /** @@ -304,7 +345,7 @@ public DistributionSpec project(Map projections, } } return new DistributionSpecHash(orderedShuffledColumns, shuffleType, tableId, selectedIndexId, partitionIds, - equivalenceExprIds, exprIdToEquivalenceSet); + equivalenceExprIds, exprIdToEquivalenceSet, hashType); } @Override @@ -313,12 +354,13 @@ public boolean equals(Object o) { return false; } DistributionSpecHash that = (DistributionSpecHash) o; - return shuffleType == that.shuffleType && orderedShuffledColumns.equals(that.orderedShuffledColumns); + return shuffleType == that.shuffleType && hashType == that.hashType + && orderedShuffledColumns.equals(that.orderedShuffledColumns); } @Override public int hashCode() { - return Objects.hash(shuffleType, orderedShuffledColumns); + return Objects.hash(shuffleType, hashType, orderedShuffledColumns); } @Override @@ -326,6 +368,7 @@ public String toString() { return Utils.toSqlString("DistributionSpecHash", "orderedShuffledColumns", orderedShuffledColumns, "shuffleType", shuffleType, + "hashType", hashType, "tableId", tableId, "selectedIndexId", selectedIndexId, "partitionIds", partitionIds, diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/PhysicalProperties.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/PhysicalProperties.java index c28d6ac3cb4d47..4d0c9bb42a83a2 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/PhysicalProperties.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/properties/PhysicalProperties.java @@ -17,12 +17,14 @@ package org.apache.doris.nereids.properties; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.nereids.properties.DistributionSpecHash.ShuffleType; import org.apache.doris.nereids.trees.expressions.ExprId; import org.apache.doris.nereids.trees.expressions.Expression; import org.apache.doris.nereids.trees.expressions.SlotReference; import java.util.Collection; +import java.util.Collections; import java.util.List; import java.util.Objects; import java.util.stream.Collectors; @@ -100,6 +102,20 @@ public static PhysicalProperties createHash(List orderedShuffledColumns, : new PhysicalProperties(new DistributionSpecHash(orderedShuffledColumns, shuffleType)); } + /** + * Like {@link #createHash(List, ShuffleType)}, but keeps the storage hash type used by + * STORAGE_BUCKETED/NATURAL layouts. An empty column list is still normalized to GATHER: + * a zero-key hash spec would satisfy every hash REQUIRE demand (its empty equivalence map + * makes containsSatisfy() always true), wrongly suppressing the exchange a parent needs. + */ + public static PhysicalProperties createHash(List orderedShuffledColumns, ShuffleType shuffleType, + HashDistributionInfo.HashType hashType) { + return orderedShuffledColumns.isEmpty() + ? PhysicalProperties.GATHER + : new PhysicalProperties(new DistributionSpecHash(orderedShuffledColumns, shuffleType, + -1L, -1L, Collections.emptySet(), hashType)); + } + public static PhysicalProperties createHash(DistributionSpecHash distributionSpecHash) { return new PhysicalProperties(distributionSpecHash); } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/implementation/LogicalOlapScanToPhysicalOlapScan.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/implementation/LogicalOlapScanToPhysicalOlapScan.java index 8448f14831cfa7..cb2a310af8acea 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/implementation/LogicalOlapScanToPhysicalOlapScan.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/implementation/LogicalOlapScanToPhysicalOlapScan.java @@ -86,7 +86,12 @@ public static DistributionSpec convertDistribution(LogicalOlapScan olapScan) { boolean isBelongStableCG = Utils.isBelongStableCG(olapTable); boolean isSelectUnpartition = Utils.isSelectUnpartition(olapTable, olapScan.getSelectedPartitionIds()); // TODO: find a better way to handle both tablet num == 1 and colocate table together in future - if (distributionInfo instanceof HashDistributionInfo && (isBelongStableCG || isSelectUnpartition)) { + // Any HASH-bucketed table advertises a NATURAL distribution carrying its bucketing hashType. + // Colocate / bucket-shuffle compatibility is then gated by comparing both sides' hashType, + // so hash types can participate as long as both sides agree. + boolean isHashBucketed = distributionInfo instanceof HashDistributionInfo; + if (isHashBucketed && (isBelongStableCG || isSelectUnpartition)) { + HashDistributionInfo.HashType hashType = ((HashDistributionInfo) distributionInfo).getHashType(); if (olapScan.getSelectedIndexId() != olapScan.getTable().getBaseIndexId()) { HashDistributionInfo hashDistributionInfo = (HashDistributionInfo) distributionInfo; List output = olapScan.getOutput(); @@ -115,7 +120,8 @@ public static DistributionSpec convertDistribution(LogicalOlapScan olapScan) { } } return new DistributionSpecHash(hashColumns, ShuffleType.NATURAL, olapScan.getTable().getId(), - olapScan.getSelectedIndexId(), Sets.newLinkedHashSet(olapScan.getSelectedPartitionIds())); + olapScan.getSelectedIndexId(), Sets.newLinkedHashSet(olapScan.getSelectedPartitionIds()), + hashType); } else { HashDistributionInfo hashDistributionInfo = (HashDistributionInfo) distributionInfo; List output = olapScan.getOutput(); @@ -133,7 +139,8 @@ public static DistributionSpec convertDistribution(LogicalOlapScan olapScan) { } } return new DistributionSpecHash(hashColumns, ShuffleType.NATURAL, olapScan.getTable().getId(), - olapScan.getSelectedIndexId(), Sets.newLinkedHashSet(olapScan.getSelectedPartitionIds())); + olapScan.getSelectedIndexId(), Sets.newLinkedHashSet(olapScan.getSelectedPartitionIds()), + hashType); } } else { // RandomDistributionInfo diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/PruneOlapScanTablet.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/PruneOlapScanTablet.java index 20f8c1d6e48d20..3663cc102941f6 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/PruneOlapScanTablet.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/PruneOlapScanTablet.java @@ -110,6 +110,7 @@ private Collection getSelectedTabletIds(List schema, Map SIGNATURES = ImmutableList.of( + FunctionSignature.ret(BigIntType.INSTANCE).varArgs(AnyDataType.INSTANCE_WITHOUT_INDEX)); + + /** + * constructor with 2 or more arguments: distribution columns plus the bucket count. + */ + public IdentityHashInternal(Expression arg, Expression... varArgs) { + super("identity_hash_internal", ExpressionUtils.mergeArguments(arg, varArgs)); + } + + /** constructor for withChildren and reuse signature */ + private IdentityHashInternal(ScalarFunctionParams functionParams) { + super(functionParams); + checkArguments(functionParams.arguments); + } + + /** + * The trailing bucket count must be a positive integer literal. Analyzing it here rejects + * malformed calls (non-constant or non-positive count) at plan time, before the expression + * reaches BE, whose identity_hash_internal expects the modulus as a constant column. + */ + private void checkArguments(List children) { + Expression last = children.get(children.size() - 1); + if (!(last instanceof IntegerLikeLiteral)) { + throw new AnalysisException(String.format( + "the bucket count argument of %s must be an integer literal, but is %s", + getName(), last.toSql())); + } + long bucketCount = ((IntegerLikeLiteral) last).getLongValue(); + if (bucketCount <= 0 || bucketCount > Integer.MAX_VALUE) { + throw new AnalysisException(String.format( + "the bucket count argument of %s must be a positive integer, but is %s", + getName(), last.toSql())); + } + } + + /** + * withChildren. + */ + @Override + public IdentityHashInternal withChildren(List children) { + Preconditions.checkArgument(children.size() >= 2, + "identity_hash_internal needs at least one distribution column and the bucket count"); + return new IdentityHashInternal(getFunctionParams(children)); + } + + @Override + public List getSignatures() { + return SIGNATURES; + } + + @Override + public FunctionSignature computePrecision(FunctionSignature signature) { + return signature; + } + + @Override + public R accept(ExpressionVisitor visitor, C context) { + return visitor.visitIdentityHashInternal(this, context); + } + + /** + * Override computeSignature to skip legacy date type conversion, mirroring Crc32Internal: + * the distribution columns must keep their original DateTime/Date encodings. + */ + @Override + public FunctionSignature computeSignature(FunctionSignature signature) { + FunctionSignature sig = signature; + sig = ComputeSignatureHelper.implementAnyDataTypeWithOutIndexNoLegacyDateUpgrade(sig, getArguments()); + sig = ComputeSignatureHelper.implementAnyDataTypeWithIndexNoLegacyDateUpgrade(sig, getArguments()); + sig = ComputeSignatureHelper.computePrecision(this, sig, getArguments()); + sig = ComputeSignatureHelper.implementFollowToArgumentReturnType(sig, getArguments()); + sig = ComputeSignatureHelper.normalizeDecimalV2(sig, getArguments()); + sig = ComputeSignatureHelper.ensureNestedNullableOfArray(sig, getArguments()); + return sig; + } +} diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/expressions/visitor/ScalarFunctionVisitor.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/expressions/visitor/ScalarFunctionVisitor.java index 3ba1ab10352bac..1a8764eee85c10 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/expressions/visitor/ScalarFunctionVisitor.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/expressions/visitor/ScalarFunctionVisitor.java @@ -271,6 +271,7 @@ import org.apache.doris.nereids.trees.expressions.functions.scalar.HoursAdd; import org.apache.doris.nereids.trees.expressions.functions.scalar.HoursDiff; import org.apache.doris.nereids.trees.expressions.functions.scalar.HoursSub; +import org.apache.doris.nereids.trees.expressions.functions.scalar.IdentityHashInternal; import org.apache.doris.nereids.trees.expressions.functions.scalar.If; import org.apache.doris.nereids.trees.expressions.functions.scalar.Ignore; import org.apache.doris.nereids.trees.expressions.functions.scalar.Initcap; @@ -1907,6 +1908,10 @@ default R visitCrc32Internal(Crc32Internal crc32Internal, C context) { return visitScalarFunction(crc32Internal, context); } + default R visitIdentityHashInternal(IdentityHashInternal identityHashInternal, C context) { + return visitScalarFunction(identityHashInternal, context); + } + default R visitLike(Like like, C context) { return visitStringRegexPredicate(like, context); } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/CreateTableInfo.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/CreateTableInfo.java index 8307f965bf8e15..f9ea5e3a723de6 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/CreateTableInfo.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/CreateTableInfo.java @@ -664,6 +664,13 @@ public void validate(ConnectContext ctx) { // validate distribution descriptor distribution.updateCols(columns.get(0).getName()); distribution.validate(columnMap, keysType); + if (distribution.isHash()) { + try { + distribution.updateHashType(PropertyAnalyzer.analyzeDistributionHashType(properties)); + } catch (Exception e) { + throw new AnalysisException(e.getMessage(), e.getCause()); + } + } // validate key set. if (!distribution.isHash()) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/DistributionDescriptor.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/DistributionDescriptor.java index 35dbf8a8b34c5d..74a3df9195c5ce 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/DistributionDescriptor.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/commands/info/DistributionDescriptor.java @@ -21,6 +21,7 @@ import org.apache.doris.analysis.HashDistributionDesc; import org.apache.doris.analysis.RandomDistributionDesc; import org.apache.doris.catalog.AggregateType; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.catalog.KeysType; import org.apache.doris.common.Config; import org.apache.doris.nereids.exceptions.AnalysisException; @@ -42,6 +43,9 @@ public class DistributionDescriptor { private final boolean isAutoBucket; private int bucketNum; private List cols; + // Default to CRC32 so hash paths that never call setHashType (e.g. CreateMTMVInfo/CreateTableInfo) still + // translate to a non-null hash type. + private HashType hashType = HashType.CRC32; public DistributionDescriptor(boolean isHash, boolean isAutoBucket, int bucketNum, List cols) { this.isHash = isHash; @@ -69,6 +73,10 @@ public void updateBucketNum(int bucketNum) { this.bucketNum = bucketNum; } + public void updateHashType(HashType hashType) { + this.hashType = hashType; + } + /** * analyze distribution descriptor */ @@ -122,7 +130,7 @@ public void validate(Map columnMap, KeysType keysType) public DistributionDesc translateToCatalogStyle() { if (isHash) { - return new HashDistributionDesc(bucketNum, isAutoBucket, cols); + return new HashDistributionDesc(bucketNum, isAutoBucket, cols, hashType); } return new RandomDistributionDesc(bucketNum, isAutoBucket); } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/physical/RuntimeFilter.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/physical/RuntimeFilter.java index 2236550937f5d0..1b39baf1ce6b71 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/physical/RuntimeFilter.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/trees/plans/physical/RuntimeFilter.java @@ -17,6 +17,7 @@ package org.apache.doris.nereids.trees.plans.physical; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.nereids.trees.expressions.Expression; import org.apache.doris.nereids.trees.expressions.Slot; import org.apache.doris.planner.RuntimeFilterId; @@ -59,6 +60,7 @@ public class RuntimeFilter { // Generated once with the runtime filter at its final target scan. Translation only // maps this target-scoped metadata to the legacy scan node id. private boolean canPruneBuckets; + private HashDistributionInfo.HashType bucketPruningHashType = HashDistributionInfo.HashType.CRC32; private Map partitionMonotonicity = ImmutableMap.of(); /** @@ -205,8 +207,12 @@ public boolean isBloomFilterSizeCalculatedByNdv() { } public void setPruningMetadata(boolean canPruneBuckets, + HashDistributionInfo.HashType bucketPruningHashType, Map partitionMonotonicity) { this.canPruneBuckets = canPruneBuckets; + if (canPruneBuckets) { + this.bucketPruningHashType = Preconditions.checkNotNull(bucketPruningHashType); + } this.partitionMonotonicity = ImmutableMap.copyOf(partitionMonotonicity); } @@ -214,6 +220,10 @@ public boolean canPruneBuckets() { return canPruneBuckets; } + public HashDistributionInfo.HashType getBucketPruningHashType() { + return bucketPruningHashType; + } + public boolean canPrunePartitions() { return !partitionMonotonicity.isEmpty(); } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/util/JoinUtils.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/util/JoinUtils.java index 92f2d54d93c4af..81c1171730017a 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/util/JoinUtils.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/util/JoinUtils.java @@ -269,6 +269,9 @@ public static boolean couldColocateJoin(DistributionSpecHash leftHashSpec, Distr || rightHashSpec.getShuffleType() != ShuffleType.NATURAL) { return false; } + if (leftHashSpec.getHashType() != rightHashSpec.getHashType()) { + return false; + } final long leftTableId = leftHashSpec.getTableId(); final long rightTableId = rightHashSpec.getTableId(); diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/DataPartition.java b/fe/fe-core/src/main/java/org/apache/doris/planner/DataPartition.java index 0ef85f8ee67170..7ee59491779c16 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/DataPartition.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/DataPartition.java @@ -24,7 +24,10 @@ import org.apache.doris.analysis.ExprToSqlVisitor; import org.apache.doris.analysis.ExprToThriftVisitor; import org.apache.doris.analysis.ToSqlParams; +import org.apache.doris.catalog.HashDistributionInfo; +import org.apache.doris.common.Config; import org.apache.doris.thrift.TDataPartition; +import org.apache.doris.thrift.TDistributionHashType; import org.apache.doris.thrift.TExplainLevel; import org.apache.doris.thrift.TIcebergPartitionField; import org.apache.doris.thrift.TMergePartitionInfo; @@ -55,6 +58,8 @@ public class DataPartition { // for hash partition: exprs used to compute hash value private ImmutableList partitionExprs; private MergePartitionInfo mergePartitionInfo; + // storage bucketing hash for BUCKET_SHFFULE_HASH_PARTITIONED; defaults to CRC32 (legacy behavior) + private HashDistributionInfo.HashType hashType = HashDistributionInfo.HashType.CRC32; public DataPartition(TPartitionType type, List exprs) { Preconditions.checkNotNull(exprs); @@ -67,6 +72,11 @@ public DataPartition(TPartitionType type, List exprs) { this.partitionExprs = ImmutableList.copyOf(exprs); } + public DataPartition(TPartitionType type, List exprs, HashDistributionInfo.HashType hashType) { + this(type, exprs); + this.hashType = hashType == null ? HashDistributionInfo.HashType.CRC32 : hashType; + } + public DataPartition(TPartitionType type) { Preconditions.checkState(type == TPartitionType.UNPARTITIONED || type == TPartitionType.RANDOM @@ -102,6 +112,22 @@ public List getPartitionExprs() { return partitionExprs; } + public HashDistributionInfo.HashType getHashType() { + return hashType; + } + + public static TDistributionHashType toTHashType(HashDistributionInfo.HashType hashType) { + if (hashType == HashDistributionInfo.HashType.IDENTITY) { + Preconditions.checkState( + Config.be_exec_version >= Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION, + "IDENTITY distribution requires all participating backends to support execution version %s " + + "or newer; current be_exec_version is %s", + Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION, Config.be_exec_version); + return TDistributionHashType.IDENTITY; + } + return TDistributionHashType.CRC32; + } + public TDataPartition toThrift() { TDataPartition result = new TDataPartition(type); if (partitionExprs != null) { @@ -110,6 +136,9 @@ public TDataPartition toThrift() { if (mergePartitionInfo != null) { result.setMergePartitionInfo(mergePartitionInfo.toThrift()); } + if (type == TPartitionType.BUCKET_SHFFULE_HASH_PARTITIONED) { + result.setDistributionHashType(toTHashType(hashType)); + } return result; } diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/ExchangeNode.java b/fe/fe-core/src/main/java/org/apache/doris/planner/ExchangeNode.java index 8898987d9a75ef..fd546594e8c138 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/ExchangeNode.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/ExchangeNode.java @@ -23,6 +23,7 @@ import org.apache.doris.analysis.SortInfo; import org.apache.doris.analysis.TupleDescriptor; import org.apache.doris.analysis.TupleId; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.common.Pair; import org.apache.doris.nereids.glue.translator.PlanTranslatorContext; import org.apache.doris.planner.LocalExchangeNode.LocalExchangeType; @@ -59,6 +60,8 @@ public class ExchangeNode extends PlanNode { private boolean isRightChildOfBroadcastHashJoin = false; private TPartitionType partitionType; + // storage bucketing hash carried for BUCKET_SHFFULE_HASH_PARTITIONED; defaults to CRC32 (legacy) + private HashDistributionInfo.HashType distributionHashType = HashDistributionInfo.HashType.CRC32; /** * use for Nereids only. @@ -81,6 +84,26 @@ public void setPartitionType(TPartitionType partitionType) { this.partitionType = partitionType; } + public HashDistributionInfo.HashType getDistributionHashType() { + return distributionHashType; + } + + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return distributionHashType; + } + + @Override + protected HashDistributionInfo.HashType getOwnStorageHashType() { + return distributionHashType; + } + + public void setDistributionHashType(HashDistributionInfo.HashType distributionHashType) { + this.distributionHashType = distributionHashType == null + ? HashDistributionInfo.HashType.CRC32 + : distributionHashType; + } + public void updateTupleIds(TupleDescriptor outputTupleDesc) { if (outputTupleDesc != null) { clearTupleIds(); diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/HashDistributionPruner.java b/fe/fe-core/src/main/java/org/apache/doris/planner/HashDistributionPruner.java index 747bb17a6ad6bf..0a2705f77f9e39 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/HashDistributionPruner.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/HashDistributionPruner.java @@ -21,6 +21,7 @@ import org.apache.doris.analysis.LiteralExpr; import org.apache.doris.analysis.SlotRef; import org.apache.doris.catalog.Column; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.catalog.MaterializedIndex; import org.apache.doris.catalog.PartitionKey; import org.apache.doris.catalog.Tablet; @@ -64,12 +65,20 @@ public class HashDistributionPruner implements DistributionPruner { private final Map distributionColumnFilters; private final int hashMod; + private final HashType hashType; + public HashDistributionPruner(List schema, MaterializedIndex materializedIndex, List columns, Map filters, int hashMod, boolean isBaseIndexSelected) { + this(schema, materializedIndex, columns, filters, hashMod, isBaseIndexSelected, HashType.CRC32); + } + + public HashDistributionPruner(List schema, MaterializedIndex materializedIndex, List columns, + Map filters, int hashMod, boolean isBaseIndexSelected, HashType hashType) { this.tablets = materializedIndex.getTablets(); this.bucketNum = tablets.size(); this.distributionColumns = columns; this.hashMod = hashMod; + this.hashType = hashType; if (isBaseIndexSelected) { this.distributionColumnFilters = filters; } else { @@ -92,8 +101,14 @@ public HashDistributionPruner(List schema, MaterializedIndex materialize public Collection prune(int columnId, PartitionKey hashKey, int complex) { if (columnId == distributionColumns.size()) { // compute Hash Key - long hashValue = hashKey.getHashValue(); - return Lists.newArrayList(getTabletId((int) ((hashValue & 0xffffffff) % hashMod))); + int bucket; + if (hashType == HashType.IDENTITY) { + bucket = hashKey.getIdentityHashValue(hashMod); + } else { + long hashValue = hashKey.getHashValue(); + bucket = (int) ((hashValue & 0xffffffff) % hashMod); + } + return Lists.newArrayList(getTabletId(bucket)); } Column keyColumn = distributionColumns.get(columnId); PartitionColumnFilter filter = distributionColumnFilters.get(keyColumn.getName()); diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/HashJoinNode.java b/fe/fe-core/src/main/java/org/apache/doris/planner/HashJoinNode.java index 1fc7880e9b28ca..61fcffd9793d34 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/HashJoinNode.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/HashJoinNode.java @@ -28,6 +28,7 @@ import org.apache.doris.analysis.SlotId; import org.apache.doris.analysis.ToSqlParams; import org.apache.doris.analysis.TupleDescriptor; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.common.Pair; import org.apache.doris.nereids.glue.translator.PlanTranslatorContext; import org.apache.doris.nereids.trees.expressions.ExprId; @@ -132,6 +133,17 @@ public boolean isColocate() { return isColocate; } + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + if (distrMode == DistributionMode.BROADCAST) { + // A broadcast join does not repartition the probe side. Its output therefore keeps the + // probe child's storage bucket layout; the replicated build side must not participate + // in layout inference. + return children.get(0).getStorageDistributionHashType(); + } + return super.getStorageDistributionHashType(); + } + @Override public boolean requiresShuffleForCorrectness() { // BE: HashJoinBuild/Probe.is_shuffled_operator() = PARTITIONED || BUCKET_SHUFFLE || COLOCATE. diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/LocalExchangeNode.java b/fe/fe-core/src/main/java/org/apache/doris/planner/LocalExchangeNode.java index 66eda40079952f..a9738ec6ff4e04 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/LocalExchangeNode.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/LocalExchangeNode.java @@ -23,6 +23,7 @@ import org.apache.doris.analysis.Expr; import org.apache.doris.analysis.ExprToThriftVisitor; import org.apache.doris.analysis.TupleDescriptor; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.thrift.TExplainLevel; import org.apache.doris.thrift.TExpr; import org.apache.doris.thrift.TLocalExchangeNode; @@ -39,6 +40,9 @@ public class LocalExchangeNode extends PlanNode { public static final String EXCHANGE_NODE = "LOCAL-EXCHANGE"; private LocalExchangeType exchangeType; + // storage bucketing hash for BUCKET_HASH_SHUFFLE; inherited from the upstream ExchangeNode's + // bucket-shuffle distribution. Defaults to CRC32 (legacy behavior). + private HashDistributionInfo.HashType distributionHashType = HashDistributionInfo.HashType.CRC32; /** * use for Nereids only. @@ -56,6 +60,12 @@ public LocalExchangeNode(PlanNodeId id, PlanNode inputNode, LocalExchangeType ex this.children.add(inputNode); this.exchangeType = exchangeType; this.fragment = inputNode.getFragment(); + // Preserve the effective storage layout through passthrough/unary nodes as well as direct + // ExchangeNode and OlapScanNode children. + HashDistributionInfo.HashType childHashType = inputNode.getStorageDistributionHashType(); + if (childHashType != null) { + this.distributionHashType = childHashType; + } List hashExprs = distributeExprs; boolean isHashShuffle = (exchangeType == LocalExchangeType.BUCKET_HASH_SHUFFLE @@ -97,6 +107,19 @@ protected void toThrift(TPlanNode msg) { } msg.local_exchange_node.setDistributeExprLists(thriftDistributeExprLists); } + if (exchangeType == LocalExchangeType.BUCKET_HASH_SHUFFLE) { + msg.local_exchange_node.setDistributionHashType(DataPartition.toTHashType(distributionHashType)); + } + } + + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return distributionHashType; + } + + @Override + protected HashDistributionInfo.HashType getOwnStorageHashType() { + return distributionHashType; } private List distributeExprLists() { diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/NestedLoopJoinNode.java b/fe/fe-core/src/main/java/org/apache/doris/planner/NestedLoopJoinNode.java index 1f280dfc25efb2..3612c05aa90e18 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/NestedLoopJoinNode.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/NestedLoopJoinNode.java @@ -23,6 +23,7 @@ import org.apache.doris.analysis.SlotId; import org.apache.doris.analysis.TupleDescriptor; import org.apache.doris.analysis.TupleId; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.common.Pair; import org.apache.doris.nereids.glue.translator.PlanTranslatorContext; import org.apache.doris.nereids.trees.expressions.ExprId; @@ -105,6 +106,13 @@ public NestedLoopJoinNode(PlanNodeId id, PlanNode outer, PlanNode inner, List distributionPrune( info.getDistributionColumns(), columnFilters, info.getBucketNum(), - getSelectedIndexId() == olapTable.getBaseIndexId()); + getSelectedIndexId() == olapTable.getBaseIndexId(), + info.getHashType()); return new ArrayList<>(distributionPruner.prune()); } case RANDOM: { diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/OlapTableSink.java b/fe/fe-core/src/main/java/org/apache/doris/planner/OlapTableSink.java index d799a5f0557035..1df83be85cdfc3 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/OlapTableSink.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/OlapTableSink.java @@ -71,6 +71,7 @@ import org.apache.doris.thrift.TColumn; import org.apache.doris.thrift.TDataSink; import org.apache.doris.thrift.TDataSinkType; +import org.apache.doris.thrift.TDistributionHashType; import org.apache.doris.thrift.TExplainLevel; import org.apache.doris.thrift.TExprNode; import org.apache.doris.thrift.TNodeInfo; @@ -502,6 +503,13 @@ private void setPartialUpdateInfoForParam(TOlapTableSchemaParam schemaParam, Ola } } + TDistributionHashType getTDistributionHashType(DistributionInfo distInfo) { + if (distInfo instanceof HashDistributionInfo) { + return DataPartition.toTHashType(((HashDistributionInfo) distInfo).getHashType()); + } + return TDistributionHashType.CRC32; + } + private List getDistColumns(DistributionInfo distInfo) throws UserException { List distColumns = Lists.newArrayList(); switch (distInfo.getType()) { @@ -990,6 +998,7 @@ private TOlapTablePartitionParam createPartition(long dbId, OlapTable table) partitionParam.setTableId(table.getId()); partitionParam.setVersion(0); partitionParam.setPartitionType(partType.toThrift()); + partitionParam.setDistributionHashType(getTDistributionHashType(table.getDefaultDistributionInfo())); // create shadow partition for empty auto partition table. only use in this load. if (enableAutomaticPartition && partitionIds.isEmpty()) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/PlanFragment.java b/fe/fe-core/src/main/java/org/apache/doris/planner/PlanFragment.java index 98621ccc4f6636..a8fd8e77a18282 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/PlanFragment.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/PlanFragment.java @@ -26,6 +26,7 @@ import org.apache.doris.analysis.JoinOperator; import org.apache.doris.analysis.StatementBase; import org.apache.doris.analysis.ToSqlParams; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.common.TreeNode; import org.apache.doris.nereids.trees.plans.distribute.NereidsSpecifyInstances; import org.apache.doris.nereids.trees.plans.distribute.worker.job.ScanSource; @@ -321,6 +322,19 @@ public int getParallelExecNum() { public TPlanFragment toThrift() { TPlanFragment result = new TPlanFragment(); if (planRoot != null) { + // Reject a genuinely mixed-layout fragment before serializing its plan: a null root + // derivation can mean either "no bucketed storage here" (schema scans, empty sets) + // or "both CRC32 and IDENTITY under this root". Only the mixed case is dangerous: + // without a fragment-level hash type, BE-native bucket local exchanges fall back to + // CRC32 and would silently mis-bucket identity-routed rows. + if (planRoot.getStorageDistributionHashType() == null) { + Set declared = new HashSet<>(); + planRoot.collectStorageHashTypes(declared); + Preconditions.checkState(declared.size() <= 1, + "fragment mixes distribution hash types %s; bucket local exchanges need one " + + "unambiguous storage layout, re-align the inputs explicitly", + declared); + } result.setPlan(planRoot.treeToThrift()); } if (outputExprs != null) { @@ -334,6 +348,11 @@ public TPlanFragment toThrift() { } else { result.setPartition(dataPartitionForThrift.toThrift()); } + HashDistributionInfo.HashType hashType = planRoot == null + ? null : planRoot.getStorageDistributionHashType(); + if (hashType != null) { + result.setDistributionHashType(DataPartition.toTHashType(hashType)); + } // TODO chenhao , calculated by cost result.setMinReservationBytes(0); diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/PlanNode.java b/fe/fe-core/src/main/java/org/apache/doris/planner/PlanNode.java index 580dc9a4eaaaef..c024160ae297dc 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/PlanNode.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/PlanNode.java @@ -31,6 +31,7 @@ import org.apache.doris.analysis.ToSqlParams; import org.apache.doris.analysis.TupleDescriptor; import org.apache.doris.analysis.TupleId; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.common.Id; import org.apache.doris.common.Pair; import org.apache.doris.common.TreeNode; @@ -1182,6 +1183,50 @@ protected Pair enforceRequire( return Pair.of(leNode, preferType); } + /** + * Return the effective storage hash type when this subtree has one unambiguous bucket layout. + * Unary nodes preserve their child's layout; multi-input nodes preserve it only when every + * child reports the same layout. + */ + public HashDistributionInfo.HashType getStorageDistributionHashType() { + HashDistributionInfo.HashType hashType = null; + for (PlanNode child : children) { + HashDistributionInfo.HashType childHashType = child.getStorageDistributionHashType(); + if (childHashType == null) { + return null; + } + if (hashType != null && hashType != childHashType) { + return null; + } + hashType = childHashType; + } + return hashType; + } + + /** + * Collect every distinct storage hash type declared by nodes in this subtree that have a + * definite layout opinion (OLAP scans, exchanges, local exchanges; nodes without one, like + * schema scans or empty-set nodes, stay silent). Used to distinguish a genuinely mixed + * subtree (both CRC32 and IDENTITY) from one that simply has no bucketed storage at all. + */ + public void collectStorageHashTypes(Set hashTypes) { + HashDistributionInfo.HashType own = getOwnStorageHashType(); + if (own != null) { + hashTypes.add(own); + } + for (PlanNode child : children) { + child.collectStorageHashTypes(hashTypes); + } + } + + /** + * The layout this node itself contributes, or null when the node only aggregates its + * children's layouts (the default) or has no bucket layout at all. + */ + protected HashDistributionInfo.HashType getOwnStorageHashType() { + return null; + } + /** * Create a LocalExchangeNode wrapping child with the given exchange type. * No child-type skip — matches BE's _add_local_exchange which inserts LE for any child diff --git a/fe/fe-core/src/main/java/org/apache/doris/planner/RuntimeFilter.java b/fe/fe-core/src/main/java/org/apache/doris/planner/RuntimeFilter.java index d98224d141acbf..d4071039a7ce7a 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/planner/RuntimeFilter.java +++ b/fe/fe-core/src/main/java/org/apache/doris/planner/RuntimeFilter.java @@ -24,10 +24,12 @@ import org.apache.doris.analysis.SlotId; import org.apache.doris.analysis.ToSqlParams; import org.apache.doris.analysis.TupleId; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.common.FeConstants; import org.apache.doris.foundation.util.BitUtil; import org.apache.doris.qe.ConnectContext; import org.apache.doris.qe.SessionVariable; +import org.apache.doris.thrift.TDistributionHashType; import org.apache.doris.thrift.TMinMaxRuntimeFilterType; import org.apache.doris.thrift.TPartitionTargetExprMonotonicity; import org.apache.doris.thrift.TRuntimeFilterDesc; @@ -146,6 +148,7 @@ public FilterSizeLimits(SessionVariable sessionVariable) { = new HashMap<>(); private final Set partitionPruningTargetScanIds = new HashSet<>(); private final Set bucketPruningTargetScanIds = new HashSet<>(); + private final Map bucketPruningTargetHashTypes = new HashMap<>(); /** * Internal representation of a runtime filter target. @@ -389,6 +392,10 @@ public TRuntimeFilterDesc toThrift() { tFilter.setBucketPruningTargetIds(bucketPruningTargetScanIds.stream() .map(PlanNodeId::asInt) .collect(Collectors.toSet())); + Map hashTypes = new HashMap<>(); + bucketPruningTargetHashTypes.forEach((nodeId, hashType) -> + hashTypes.put(nodeId.asInt(), DataPartition.toTHashType(hashType))); + tFilter.setBucketPruningTargetHashTypes(hashTypes); } return tFilter; @@ -422,8 +429,10 @@ public boolean canPrunePartitionsFor(PlanNodeId scanNodeId) { return partitionPruningTargetScanIds.contains(scanNodeId); } - public void markTargetCanPruneBuckets(PlanNodeId scanNodeId) { + public void markTargetCanPruneBuckets(PlanNodeId scanNodeId, + HashDistributionInfo.HashType hashType) { bucketPruningTargetScanIds.add(scanNodeId); + bucketPruningTargetHashTypes.put(scanNodeId, hashType); } public boolean canPruneBucketsFor(PlanNodeId scanNodeId) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/service/FrontendServiceImpl.java b/fe/fe-core/src/main/java/org/apache/doris/service/FrontendServiceImpl.java index 2f98052fd9a321..746907bed89a6c 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/service/FrontendServiceImpl.java +++ b/fe/fe-core/src/main/java/org/apache/doris/service/FrontendServiceImpl.java @@ -36,6 +36,7 @@ import org.apache.doris.catalog.DatabaseIf; import org.apache.doris.catalog.DistributionInfo; import org.apache.doris.catalog.Env; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.catalog.InfoSchemaDb; import org.apache.doris.catalog.MaterializedIndex; import org.apache.doris.catalog.OlapTable; @@ -69,6 +70,7 @@ import org.apache.doris.common.DdlException; import org.apache.doris.common.DuplicatedRequestException; import org.apache.doris.common.FeConstants; +import org.apache.doris.common.FeMetaVersion; import org.apache.doris.common.IncrWindowNotReadyException; import org.apache.doris.common.InternalErrorCode; import org.apache.doris.common.LabelAlreadyUsedException; @@ -5971,6 +5973,21 @@ public TGetOlapTableMetaResult getOlapTableMeta(TGetOlapTableMetaRequest request Map tempPartitionChecksums = Maps.newHashMap(); table.readLock(); try { + // Older remote-Doris clients ignore hashType in Gson metadata and prune every + // HASH table with CRC32. Reject before exporting any metadata: the local BE + // execution-version gate cannot protect a plan produced by an older remote FE. + DistributionInfo distributionInfo = table.getDefaultDistributionInfo(); + if (distributionInfo instanceof HashDistributionInfo + && ((HashDistributionInfo) distributionInfo).getHashType() + == HashDistributionInfo.HashType.IDENTITY + && (!request.isSetVersion() || request.getVersion() < FeMetaVersion.VERSION_141)) { + // table_meta is required by Thrift even when the RPC returns an error. + result.setTableMeta(new byte[0]); + throw new UserException("IDENTITY distribution requires client metadata version " + + FeMetaVersion.VERSION_141 + " or newer for table " + dbName + "." + table.getName() + + "; client version: " + + (request.isSetVersion() ? request.getVersion() : "unspecified")); + } OlapTable copyTable = table.copyTableMeta(); try (DataOutputStream out = new DataOutputStream(bOutputStream)) { copyTable.write(out); diff --git a/fe/fe-core/src/test/java/org/apache/doris/backup/RestoreJobTest.java b/fe/fe-core/src/test/java/org/apache/doris/backup/RestoreJobTest.java index 00d49f6489b464..b36e06987c691e 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/backup/RestoreJobTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/backup/RestoreJobTest.java @@ -17,19 +17,28 @@ package org.apache.doris.backup; +import org.apache.doris.analysis.PartitionValue; import org.apache.doris.backup.BackupJobInfo.BackupIndexInfo; import org.apache.doris.backup.BackupJobInfo.BackupOlapTableInfo; import org.apache.doris.backup.BackupJobInfo.BackupPartitionInfo; import org.apache.doris.backup.BackupJobInfo.BackupTabletInfo; +import org.apache.doris.catalog.Column; +import org.apache.doris.catalog.DataProperty; import org.apache.doris.catalog.Database; import org.apache.doris.catalog.Env; import org.apache.doris.catalog.HashDistributionInfo; +import org.apache.doris.catalog.HashDistributionInfo.HashType; +import org.apache.doris.catalog.KeysType; import org.apache.doris.catalog.MaterializedIndex; import org.apache.doris.catalog.MaterializedIndex.IndexExtState; import org.apache.doris.catalog.OlapTable; import org.apache.doris.catalog.Partition; import org.apache.doris.catalog.PartitionInfo; +import org.apache.doris.catalog.PartitionKey; import org.apache.doris.catalog.PartitionType; +import org.apache.doris.catalog.PrimitiveType; +import org.apache.doris.catalog.RangePartitionInfo; +import org.apache.doris.catalog.RangePartitionItem; import org.apache.doris.catalog.ReplicaAllocation; import org.apache.doris.catalog.Resource; import org.apache.doris.catalog.Table; @@ -41,16 +50,20 @@ import org.apache.doris.common.jmockit.Deencapsulation; import org.apache.doris.datasource.InternalCatalog; import org.apache.doris.datasource.storage.StorageAdapter; +import org.apache.doris.nereids.trees.plans.commands.BackupCommand.BackupContent; import org.apache.doris.persist.EditLog; import org.apache.doris.system.SystemInfoService; import org.apache.doris.thrift.TStorageMedium; import com.google.common.collect.Lists; import com.google.common.collect.Maps; +import com.google.common.collect.Range; import org.junit.jupiter.api.AfterEach; import org.junit.jupiter.api.Assertions; import org.junit.jupiter.api.BeforeEach; import org.junit.jupiter.api.Test; +import org.junit.jupiter.params.ParameterizedTest; +import org.junit.jupiter.params.provider.EnumSource; import org.mockito.MockedConstruction; import org.mockito.MockedStatic; import org.mockito.Mockito; @@ -250,6 +263,91 @@ public void testSignature() throws AnalysisException { System.out.println("tbl signature: " + tbl.getSignature(BackupHandler.SIGNATURE_VERSION, partNames)); } + @ParameterizedTest + @EnumSource(HashType.class) + public void testRestoreDisjointPartitionWithSameHash(HashType hashType) throws Exception { + // Bucket counts may differ between partitions; only the table-wide hash must match. + checkRestoreDisjointPartition(hashType, hashType, 5); + } + + @ParameterizedTest + @EnumSource(HashType.class) + public void testRestoreDisjointPartitionWithDifferentHash(HashType localHash) throws Exception { + HashType remoteHash = localHash == HashType.CRC32 ? HashType.IDENTITY : HashType.CRC32; + checkRestoreDisjointPartition(localHash, remoteHash, 8); + } + + private void checkRestoreDisjointPartition(HashType localHash, HashType remoteHash, int remoteBuckets) + throws Exception { + OlapTable local = createHashPartitionedTable(30003L, "p1", 40003L, 0, 10, localHash, 8); + OlapTable remote = createHashPartitionedTable(30004L, "p2", 40004L, 10, 20, remoteHash, remoteBuckets); + db.registerTable(local); + List intersectPartNames = Lists.newArrayList(); + Assertions.assertTrue(local.getIntersectPartNamesWith(remote, intersectPartNames).ok()); + Assertions.assertTrue(intersectPartNames.isEmpty()); + + jobInfo.backupOlapTableObjects.clear(); + jobInfo.content = BackupContent.METADATA_ONLY; + BackupOlapTableInfo tableInfo = new BackupOlapTableInfo(); + tableInfo.id = remote.getId(); + BackupPartitionInfo partitionInfo = new BackupPartitionInfo(); + partitionInfo.id = remote.getPartition("p2").getId(); + tableInfo.partitions.put("p2", partitionInfo); + jobInfo.backupOlapTableObjects.put(remote.getName(), tableInfo); + BackupMeta meta = new BackupMeta(Lists.newArrayList(remote), Lists.newArrayList()); + mockedEnvStatic.when(Env::getCurrentEnv).thenReturn(env); + RestoreJob restore = Mockito.spy(new RestoreJob(label, "2018-01-01 01:01:01", + db.getId(), db.getFullName(), jobInfo, false, + new ReplicaAllocation((short) 1), 100000, -1, + false, false, false, false, false, false, false, false, + env, Repository.KEEP_ON_LOCAL_REPO_ID, meta)); + // Exercise real metadata validation, partition resetting and attachment. Only skip BE tasks + // and file mappings: this test has no physical tablets to create or restore. + Mockito.doNothing().when(restore).createReplicas(Mockito.any(), Mockito.any(), Mockito.any()); + Mockito.doNothing().when(restore).genFileMapping(Mockito.any(), Mockito.any(), + Mockito.anyLong(), Mockito.any(), Mockito.anyBoolean()); + Mockito.doNothing().when(restore).doCreateReplicas(); + Deencapsulation.invoke(restore, "checkAndPrepareMeta"); + if (localHash == remoteHash) { + Assertions.assertTrue(restore.getStatus().ok(), restore.getStatus().toString()); + Assertions.assertEquals(RestoreJob.RestoreJobState.CREATING, restore.getState()); + Assertions.assertEquals(1, restore.restoredPartitions.size()); + restore.allReplicasCreated(); + Assertions.assertSame(remote.getPartition("p2"), local.getPartition("p2")); + HashDistributionInfo restoredDistribution = + (HashDistributionInfo) local.getPartition("p2").getDistributionInfo(); + Assertions.assertEquals(remoteHash, restoredDistribution.getHashType()); + Assertions.assertEquals(remoteBuckets, restoredDistribution.getBucketNum()); + } else { + Assertions.assertFalse(restore.getStatus().ok(), + "Mixed hash restore passed metadata validation: " + localHash + " <- " + remoteHash); + Assertions.assertTrue(restore.getStatus().getErrMsg().contains("different schema")); + Assertions.assertTrue(restore.restoredPartitions.isEmpty()); + Assertions.assertNull(local.getPartition("p2")); + } + } + + private OlapTable createHashPartitionedTable(long tableId, String partitionName, long partitionId, + int lower, int upper, HashType hashType, int buckets) throws AnalysisException { + Column key = new Column("id", PrimitiveType.BIGINT, true); + Column date = new Column("dt", PrimitiveType.INT, true); + List partitionColumns = Lists.newArrayList(date); + RangePartitionInfo partitionInfo = new RangePartitionInfo(partitionColumns); + PartitionKey lowerKey = PartitionKey.createPartitionKey( + Lists.newArrayList(new PartitionValue(Integer.toString(lower))), partitionColumns); + PartitionKey upperKey = PartitionKey.createPartitionKey( + Lists.newArrayList(new PartitionValue(Integer.toString(upper))), partitionColumns); + partitionInfo.addPartition(partitionId, false, new RangePartitionItem(Range.closedOpen(lowerKey, upperKey)), + new DataProperty(TStorageMedium.HDD), new ReplicaAllocation((short) 1), false, true); + HashDistributionInfo distribution = new HashDistributionInfo( + buckets, false, Lists.newArrayList(key), hashType); + OlapTable table = new OlapTable(tableId, "restore_hash_table", Lists.newArrayList(key, date), KeysType.DUP_KEYS, + partitionInfo, distribution); + table.addPartition(new Partition(partitionId, partitionName, + new MaterializedIndex(tableId, MaterializedIndex.IndexState.NORMAL), distribution)); + return table; + } + @Test public void testSerialization() throws IOException, AnalysisException { // 1. Write objects to file diff --git a/fe/fe-core/src/test/java/org/apache/doris/catalog/DistributionHashTypeTest.java b/fe/fe-core/src/test/java/org/apache/doris/catalog/DistributionHashTypeTest.java new file mode 100644 index 00000000000000..58c38f5bbeee02 --- /dev/null +++ b/fe/fe-core/src/test/java/org/apache/doris/catalog/DistributionHashTypeTest.java @@ -0,0 +1,341 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +package org.apache.doris.catalog; + +import org.apache.doris.analysis.DistributionDesc; +import org.apache.doris.analysis.HashDistributionDesc; +import org.apache.doris.catalog.ColocateTableIndex.GroupId; +import org.apache.doris.catalog.HashDistributionInfo.HashType; +import org.apache.doris.common.AnalysisException; +import org.apache.doris.common.DdlException; +import org.apache.doris.common.FeMetaVersion; +import org.apache.doris.common.util.PropertyAnalyzer; +import org.apache.doris.meta.MetaContext; +import org.apache.doris.persist.gson.GsonUtils; + +import com.google.common.collect.Lists; +import com.google.common.collect.Maps; +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +import java.io.ByteArrayInputStream; +import java.io.ByteArrayOutputStream; +import java.io.DataInputStream; +import java.io.DataOutputStream; +import java.util.List; +import java.util.Map; + +// Tests for the pluggable bucketing hash function carried by the `distribution_hash_type` table +// property. Today HashType has CRC32 (default/legacy) and IDENTITY; more types will be added later, +// so the framework-level cases (gson round-trip, equals, property parse) iterate over +// HashType.values() and stay correct as new constants appear. Identity-specific cases verify that +// canonical bytes from every valid distribution-column type and multiple columns are accepted. +public class DistributionHashTypeTest { + + private Column intCol(String name) { + return new Column(name, PrimitiveType.BIGINT, true); + } + + // ------------------------------------------------------------------ + // Metadata / backward compatibility + // ------------------------------------------------------------------ + + @Test + public void testLegacyConstructorsDefaultToCrc32() { + Assertions.assertEquals(HashType.CRC32, new HashDistributionInfo().getHashType()); + Assertions.assertEquals(HashType.CRC32, + new HashDistributionInfo(8, Lists.newArrayList(intCol("id"))).getHashType()); + Assertions.assertEquals(HashType.CRC32, + new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id"))).getHashType()); + } + + @Test + public void testLegacyMetadataWithoutHashTypeDeserializesToCrc32() { + // Metadata written before hashType existed has no "hashType" key; gson leaves it null and + // getHashType() must fall back to CRC32 so old tables keep their historical bucket layout. + HashDistributionInfo original + = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), HashType.CRC32); + String json = GsonUtils.GSON.toJson(original); + String legacyJson = json.replaceAll(",?\\s*\"hashType\"\\s*:\\s*\"[A-Z0-9_]+\"", ""); + Assertions.assertFalse(legacyJson.contains("hashType")); + HashDistributionInfo restored = GsonUtils.GSON.fromJson(legacyJson, HashDistributionInfo.class); + Assertions.assertEquals(HashType.CRC32, restored.getHashType()); + } + + @Test + public void testHashTypeSurvivesGsonRoundTrip() { + // Framework-level: every hash type must round-trip. Adding a new HashType automatically + // extends this coverage. + for (HashType type : HashType.values()) { + HashDistributionInfo original = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), type); + HashDistributionInfo restored + = GsonUtils.GSON.fromJson(GsonUtils.GSON.toJson(original), HashDistributionInfo.class); + Assertions.assertEquals(type, restored.getHashType(), "hashType lost in gson round trip: " + type); + } + } + + @Test + public void testEqualityAndHashCodeConsiderHashType() { + // Any two distinct hash types must make otherwise-identical infos unequal. + HashType[] types = HashType.values(); + for (int i = 0; i < types.length; i++) { + HashDistributionInfo a = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), types[i]); + HashDistributionInfo aSame = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), types[i]); + Assertions.assertEquals(a, aSame); + Assertions.assertEquals(a.hashCode(), aSame.hashCode()); + for (int j = i + 1; j < types.length; j++) { + HashDistributionInfo b = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), types[j]); + Assertions.assertNotEquals(a, b); + } + } + } + + @Test + public void testToDistributionDescCarriesHashType() throws DdlException { + // toDistributionDesc() is used when a partition deep-copies the table distribution + // (dynamic partition / addMultiPartitions); the hashType must ride along. Verify by + // round-tripping desc back to info (HashDistributionDesc has no getter). + for (HashType type : HashType.values()) { + List columns = Lists.newArrayList(intCol("id")); + HashDistributionInfo info = new HashDistributionInfo(8, false, columns, type); + DistributionDesc desc = info.toDistributionDesc(); + Assertions.assertTrue(desc instanceof HashDistributionDesc); + HashDistributionInfo rebuilt = (HashDistributionInfo) desc.toDistributionInfo(columns); + Assertions.assertEquals(type, rebuilt.getHashType()); + HashDistributionInfo descriptorRoundTrip = (HashDistributionInfo) desc.toDistributionDescriptor() + .translateToCatalogStyle().toDistributionInfo(columns); + Assertions.assertEquals(type, descriptorRoundTrip.getHashType()); + } + } + + // NOTE: there is deliberately no unit test for the setHashType call site in + // InternalCatalog.addPartition(): driving that path needs a full catalog (AddPartitionOp + // analysis, schema resolution, agent batches), and a setter-level test like the removed + // testSetHashTypeInheritedByAddPartition stays tautological - it would keep passing with + // the inheritance statement deleted. The behavior is guarded end-to-end by the regression + // suite (test_distribution_hash_type_identity.groovy section 6): a real ALTER TABLE ADD + // PARTITION on an identity table, followed by writes into the new partition and an + // equality-pruned read-back that would drop rows if the partition fell back to CRC32. + + // ------------------------------------------------------------------ + // Property parsing + // ------------------------------------------------------------------ + + @Test + public void testAnalyzeDistributionHashType() throws AnalysisException { + // missing property -> CRC32 + Assertions.assertEquals(HashType.CRC32, PropertyAnalyzer.analyzeDistributionHashType(null)); + Assertions.assertEquals(HashType.CRC32, PropertyAnalyzer.analyzeDistributionHashType(Maps.newHashMap())); + + // every hash type parses case-insensitively and the property is consumed (removed) so it is + // not later flagged as an unknown property. + for (HashType type : HashType.values()) { + Map props = Maps.newHashMap(); + props.put(PropertyAnalyzer.PROPERTIES_DISTRIBUTION_HASH_TYPE, mixCase(type.name())); + Assertions.assertEquals(type, PropertyAnalyzer.analyzeDistributionHashType(props)); + Assertions.assertFalse(props.containsKey(PropertyAnalyzer.PROPERTIES_DISTRIBUTION_HASH_TYPE)); + } + } + + @Test + public void testAnalyzeDistributionHashTypeInvalidValueThrows() { + Map bad = Maps.newHashMap(); + bad.put(PropertyAnalyzer.PROPERTIES_DISTRIBUTION_HASH_TYPE, "murmur3"); + AnalysisException e + = Assertions.assertThrows(AnalysisException.class, () -> PropertyAnalyzer.analyzeDistributionHashType(bad)); + Assertions.assertTrue(e.getMessage().contains(PropertyAnalyzer.PROPERTIES_DISTRIBUTION_HASH_TYPE)); + } + + // ------------------------------------------------------------------ + // identity accepts canonical bytes from all valid distribution columns + // ------------------------------------------------------------------ + + @Test + public void testToDistributionInfoIdentitySingleIntegerColumn() throws DdlException { + List schema = Lists.newArrayList(intCol("shard_num"), new Column("v", PrimitiveType.INT, false)); + HashDistributionDesc desc + = new HashDistributionDesc(8, false, Lists.newArrayList("shard_num"), HashType.IDENTITY); + HashDistributionInfo info = (HashDistributionInfo) desc.toDistributionInfo(schema); + Assertions.assertEquals(HashType.IDENTITY, info.getHashType()); + Assertions.assertEquals(1, info.getDistributionColumns().size()); + } + + @Test + public void testToDistributionInfoIdentityAllowsLargeInt() throws DdlException { + List schema = Lists.newArrayList(new Column("big_id", PrimitiveType.LARGEINT, true)); + HashDistributionDesc desc = new HashDistributionDesc(8, false, Lists.newArrayList("big_id"), HashType.IDENTITY); + HashDistributionInfo info = (HashDistributionInfo) desc.toDistributionInfo(schema); + Assertions.assertEquals(HashType.IDENTITY, info.getHashType()); + } + + @Test + public void testToDistributionInfoIdentityAllowsNonIntegerColumn() throws DdlException { + List schema = Lists.newArrayList(new Column("s", PrimitiveType.VARCHAR, true)); + HashDistributionDesc desc = new HashDistributionDesc(8, false, Lists.newArrayList("s"), HashType.IDENTITY); + HashDistributionInfo info = (HashDistributionInfo) desc.toDistributionInfo(schema); + Assertions.assertEquals(HashType.IDENTITY, info.getHashType()); + Assertions.assertEquals(PrimitiveType.VARCHAR, + info.getDistributionColumns().get(0).getType().getPrimitiveType()); + } + + @Test + public void testToDistributionInfoIdentityAllowsMultipleColumns() throws DdlException { + List schema = Lists.newArrayList(intCol("a"), new Column("b", PrimitiveType.VARCHAR, true)); + HashDistributionDesc desc = new HashDistributionDesc(8, false, Lists.newArrayList("a", "b"), + HashType.IDENTITY); + HashDistributionInfo info = (HashDistributionInfo) desc.toDistributionInfo(schema); + Assertions.assertEquals(HashType.IDENTITY, info.getHashType()); + Assertions.assertEquals(2, info.getDistributionColumns().size()); + } + + @Test + public void testToDistributionInfoCrc32AllowsNonIntegerAndMultiColumn() throws DdlException { + // crc32 (default) keeps its historical freedom: multi-column and non-integer are fine. + List schema = Lists.newArrayList(new Column("a", PrimitiveType.VARCHAR, true), intCol("b")); + HashDistributionDesc desc = new HashDistributionDesc(8, false, Lists.newArrayList("a", "b"), HashType.CRC32); + HashDistributionInfo info = (HashDistributionInfo) desc.toDistributionInfo(schema); + Assertions.assertEquals(HashType.CRC32, info.getHashType()); + Assertions.assertEquals(2, info.getDistributionColumns().size()); + } + + @Test + public void testTableSignatureConsidersHashType() { + Column key = new Column("id", PrimitiveType.INT, true); + HashDistributionInfo distributionInfo = new HashDistributionInfo(8, Lists.newArrayList(key)); + OlapTable table = new OlapTable(1L, "t", Lists.newArrayList(key), KeysType.DUP_KEYS, + new SinglePartitionInfo(), distributionInfo); + table.addPartition(new Partition(2L, "p", new MaterializedIndex(3L, + MaterializedIndex.IndexState.NORMAL), distributionInfo)); + + String crc32Signature = table.getSignature(1, Lists.newArrayList("p")); + distributionInfo.setHashType(HashType.IDENTITY); + String identitySignature = table.getSignature(1, Lists.newArrayList("p")); + Assertions.assertNotEquals(crc32Signature, identitySignature); + } + + // ------------------------------------------------------------------ + // ColocateGroupSchema: hashType participates in colocate compatibility and metadata + // ------------------------------------------------------------------ + + private ColocateGroupSchema schemaWith(HashType type) { + return new ColocateGroupSchema(new GroupId(1L, 2L), Lists.newArrayList(intCol("id")), 8, + new ReplicaAllocation((short) 1), type); + } + + @Test + public void testCheckDistributionAllowsSameHashType() throws DdlException { + // A table whose distribution hashType matches the group's must pass checkDistribution. + for (HashType type : HashType.values()) { + ColocateGroupSchema schema = schemaWith(type); + HashDistributionInfo info = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), type); + schema.checkDistribution(info); // should not throw + } + } + + @Test + public void testCheckDistributionRejectsDifferentHashType() { + // Mixing hash types inside one colocate group would break co-location, so it must be + // rejected before the buckets-num / column checks even when those are identical. + HashType[] types = HashType.values(); + for (int i = 0; i < types.length; i++) { + for (int j = 0; j < types.length; j++) { + if (i == j) { + continue; + } + ColocateGroupSchema schema = schemaWith(types[i]); + HashDistributionInfo info + = new HashDistributionInfo(8, false, Lists.newArrayList(intCol("id")), types[j]); + Assertions.assertThrows(DdlException.class, () -> schema.checkDistribution(info)); + } + } + } + + @Test + public void testWritableRoundTripPreservesHashType() throws Exception { + // With a current-version journal, write() appends the hashType name and readFields() must + // restore it verbatim for every hash type. + MetaContext metaContext = new MetaContext(); + metaContext.setMetaVersion(FeMetaVersion.VERSION_141); + metaContext.setThreadLocalInfo(); + try { + for (HashType type : HashType.values()) { + ColocateGroupSchema original = schemaWith(type); + ByteArrayOutputStream bos = new ByteArrayOutputStream(); + original.write(new DataOutputStream(bos)); + ColocateGroupSchema restored + = ColocateGroupSchema.read(new DataInputStream(new ByteArrayInputStream(bos.toByteArray()))); + Assertions.assertEquals(type, restored.getHashType(), "hashType lost in Writable round trip: " + type); + Assertions.assertEquals(8, restored.getBucketsNum()); + } + } finally { + MetaContext.remove(); + } + } + + @Test + public void testReadFieldsBeforeVersion141FallsBackToCrc32() throws Exception { + // Build the exact legacy stream, which ended after ReplicaAllocation and had no hash type. + ColocateGroupSchema original = schemaWith(HashType.CRC32); + ByteArrayOutputStream bos = new ByteArrayOutputStream(); + DataOutputStream out = new DataOutputStream(bos); + original.getGroupId().write(out); + out.writeInt(original.getDistributionColTypes().size()); + for (Type type : original.getDistributionColTypes()) { + ColumnType.write(out, type); + } + out.writeInt(original.getBucketsNum()); + original.getReplicaAlloc().write(out); + + MetaContext readContext = new MetaContext(); + readContext.setMetaVersion(FeMetaVersion.VERSION_140); + readContext.setThreadLocalInfo(); + try { + ByteArrayInputStream input = new ByteArrayInputStream(bos.toByteArray()); + ColocateGroupSchema restored = ColocateGroupSchema.read(new DataInputStream(input)); + Assertions.assertEquals(HashType.CRC32, restored.getHashType()); + Assertions.assertEquals(0, input.available()); + } finally { + MetaContext.remove(); + } + } + + @Test + public void testGetHashTypeNullFallsBackToCrc32() { + // Legacy gson metadata has no "hashType" field; getHashType() must not NPE and defaults to + // CRC32, matching HashDistributionInfo's fallback. + ColocateGroupSchema schema = schemaWith(HashType.IDENTITY); + String json = GsonUtils.GSON.toJson(schema); + String legacyJson = json.replaceAll(",?\\s*\"hashType\"\\s*:\\s*\"[A-Z0-9_]+\"", ""); + Assertions.assertFalse(legacyJson.contains("hashType")); + ColocateGroupSchema restored = GsonUtils.GSON.fromJson(legacyJson, ColocateGroupSchema.class); + Assertions.assertEquals(HashType.CRC32, restored.getHashType()); + } + + // Alternate the case of each character so the parse path is exercised case-insensitively + // regardless of which hash type name it is. + private String mixCase(String s) { + StringBuilder sb = new StringBuilder(s.length()); + for (int i = 0; i < s.length(); i++) { + char c = s.charAt(i); + sb.append((i & 1) == 0 + ? Character.toUpperCase(c) + : Character.toLowerCase(c)); + } + return sb.toString(); + } +} diff --git a/fe/fe-core/src/test/java/org/apache/doris/catalog/MaterializedIndexTest.java b/fe/fe-core/src/test/java/org/apache/doris/catalog/MaterializedIndexTest.java index 0fc2bab1b3d61d..45939f28486c3c 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/catalog/MaterializedIndexTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/catalog/MaterializedIndexTest.java @@ -194,6 +194,18 @@ public void testPartitionMetaChecksum() { Assertions.assertEquals(firstPartition.getMetaChecksum(), firstPartition.getRemoteMetaChecksum()); } + @Test + public void testPartitionMetaChecksumChangesOnDistributionHashType() { + MaterializedIndex baseIndex = new MaterializedIndex(1L, IndexState.NORMAL); + HashDistributionInfo distributionInfo = new HashDistributionInfo( + 3, List.of(new Column("k1", PrimitiveType.INT))); + Partition partition = new Partition(1L, "p1", baseIndex, distributionInfo); + String crc32Checksum = partition.getMetaChecksum(); + + distributionInfo.setHashType(HashDistributionInfo.HashType.IDENTITY); + Assertions.assertNotEquals(crc32Checksum, partition.getMetaChecksum()); + } + @Test public void testPartitionMetaChecksumChangesOnReplicaQueryFields() { // Build a partition with one tablet/replica. diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslatorBucketPruneTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslatorBucketPruneTest.java index 43cd6d7bae709d..7f9a7b3ac3892f 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslatorBucketPruneTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/glue/translator/RuntimeFilterTranslatorBucketPruneTest.java @@ -46,6 +46,7 @@ import org.apache.doris.planner.RuntimeFilterId; import org.apache.doris.qe.ConnectContext; import org.apache.doris.qe.SessionVariable; +import org.apache.doris.thrift.TDistributionHashType; import org.apache.doris.thrift.TExprNodeType; import org.apache.doris.thrift.TMinMaxRuntimeFilterType; import org.apache.doris.thrift.TRuntimeFilterDesc; @@ -98,6 +99,8 @@ void testGroupedSameTargetSerializesOneExpressionAndBucketTarget() { Assertions.assertEquals(firstLegacySlotId(harness, target), desc.planId_to_target_expr.get(SCAN_NODE_ID).nodes.get(0).slot_ref.slot_id); Assertions.assertTrue(desc.isSetBucketPruningTargetIds()); + Assertions.assertEquals(TDistributionHashType.CRC32, + desc.bucket_pruning_target_hash_types.get(SCAN_NODE_ID)); Assertions.assertEquals(ImmutableList.of(SCAN_NODE_ID), desc.bucket_pruning_target_ids.stream().sorted().collect(java.util.stream.Collectors.toList())); } diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifierTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifierTest.java index 02157543a6ccb7..3ad375179ed306 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifierTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/processor/post/RuntimeFilterPruneClassifierTest.java @@ -72,6 +72,21 @@ void testSingleColumnHashInSupported() { new HashDistributionInfo(8, ImmutableList.of(distributionColumn))); Assertions.assertTrue(classification.canPruneBuckets()); + Assertions.assertEquals(HashDistributionInfo.HashType.CRC32, + classification.getBucketHashType()); + } + + @Test + void testIdentityHashTypePropagatedForBucketPruning() { + Column distributionColumn = new Column("dist_col", PrimitiveType.INT); + RuntimeFilterPruneClassifier.Classification classification = classifyBucket( + TRuntimeFilterType.IN, distributionColumn, + new HashDistributionInfo(8, false, ImmutableList.of(distributionColumn), + HashDistributionInfo.HashType.IDENTITY)); + + Assertions.assertTrue(classification.canPruneBuckets()); + Assertions.assertEquals(HashDistributionInfo.HashType.IDENTITY, + classification.getBucketHashType()); } @Test diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriverTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriverTest.java index fd23de7a2f46d6..6198e53441a83e 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriverTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/ChildOutputPropertyDeriverTest.java @@ -19,6 +19,7 @@ import org.apache.doris.catalog.ColocateTableIndex; import org.apache.doris.catalog.Env; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.common.FeConstants; import org.apache.doris.common.IdGenerator; import org.apache.doris.nereids.hint.DistributeHint; @@ -44,6 +45,7 @@ import org.apache.doris.nereids.trees.plans.LimitPhase; import org.apache.doris.nereids.trees.plans.RelationId; import org.apache.doris.nereids.trees.plans.SortPhase; +import org.apache.doris.nereids.trees.plans.algebra.SetOperation.Qualifier; import org.apache.doris.nereids.trees.plans.logical.LogicalOneRowRelation; import org.apache.doris.nereids.trees.plans.physical.AbstractPhysicalPlan; import org.apache.doris.nereids.trees.plans.physical.PhysicalAssertNumRows; @@ -53,7 +55,9 @@ import org.apache.doris.nereids.trees.plans.physical.PhysicalNestedLoopJoin; import org.apache.doris.nereids.trees.plans.physical.PhysicalQuickSort; import org.apache.doris.nereids.trees.plans.physical.PhysicalRepeat; +import org.apache.doris.nereids.trees.plans.physical.PhysicalSetOperation; import org.apache.doris.nereids.trees.plans.physical.PhysicalTopN; +import org.apache.doris.nereids.trees.plans.physical.PhysicalUnion; import org.apache.doris.nereids.types.BigIntType; import org.apache.doris.nereids.types.IntegerType; import org.apache.doris.nereids.types.TinyIntType; @@ -1136,4 +1140,134 @@ void testComputeUniformAfterRecomputeLogicalProperties_AsOfRightOuter() { Assertions.assertTrue(result.isUniformAndNotNull(rightSlot)); } + + private SlotReference slot(String name, long uniqueId) { + return new SlotReference(new ExprId((int) uniqueId), name, IntegerType.INSTANCE, false, + Collections.emptyList()); + } + + private LogicalProperties setOpLogicalProperties(SlotReference out1, SlotReference out2) { + List outputs = Lists.newArrayList(out1, out2); + return new LogicalProperties(() -> outputs, () -> DataTrait.EMPTY_TRAIT); + } + + private PhysicalSetOperation unionOf(List leftOutput, List rightOutput, + SlotReference out1, SlotReference out2) { + LogicalProperties leftLogical = new LogicalProperties(() -> Lists.newArrayList(leftOutput), + () -> DataTrait.EMPTY_TRAIT); + LogicalProperties rightLogical = new LogicalProperties(() -> Lists.newArrayList(rightOutput), + () -> DataTrait.EMPTY_TRAIT); + IdGenerator idGenerator = GroupId.createGenerator(); + GroupPlan left = new GroupPlan(new Group(idGenerator.getNextId(), leftLogical)); + GroupPlan right = new GroupPlan(new Group(idGenerator.getNextId(), rightLogical)); + return new PhysicalUnion(Qualifier.ALL, Lists.newArrayList(out1, out2), + ImmutableList.of(leftOutput, rightOutput), ImmutableList.of(), + Optional.empty(), setOpLogicalProperties(out1, out2), Lists.newArrayList(left, right)); + } + + /** + * The generic EXECUTION_BUCKETED path must derive the set operation's output hash spec from + * the children's specs: same shuffle type, keys mapped to the set operation outputs, and the + * hash type EXECUTION_BUCKETED always carries (CRC32, per DistributionSpecHash's + * normalization). Each child's equivalence map must cover every regular child output, as a + * PhysicalDistribute-derived spec would. + */ + @Test + void testSetOperationExecutionOutputDerivesKeys() { + SlotReference left1 = slot("l1", 1); + SlotReference left2 = slot("l2", 2); + SlotReference right1 = slot("r1", 3); + SlotReference right2 = slot("r2", 4); + SlotReference out1 = slot("o1", 5); + SlotReference out2 = slot("o2", 6); + PhysicalSetOperation setOperation = unionOf(Lists.newArrayList(left1, left2), + Lists.newArrayList(right1, right2), out1, out2); + + // Each child shuffles on its first output column; its equivalence map must contain every + // regular child output (the deriver maps each output position through + // exprIdToEquivalenceSet and bails out with ANY when one is missing). + PhysicalProperties leftChild = new PhysicalProperties(new DistributionSpecHash( + Lists.newArrayList(left1.getExprId(), left2.getExprId()), + ShuffleType.EXECUTION_BUCKETED, -1L, -1L, Collections.emptySet(), HashType.CRC32)); + PhysicalProperties rightChild = new PhysicalProperties(new DistributionSpecHash( + Lists.newArrayList(right1.getExprId(), right2.getExprId()), + ShuffleType.EXECUTION_BUCKETED, -1L, -1L, Collections.emptySet(), HashType.CRC32)); + PhysicalProperties result = new ChildOutputPropertyDeriver(Lists.newArrayList(leftChild, rightChild)) + .getOutputProperties(null, new GroupExpression(setOperation)); + + DistributionSpecHash output = Assertions.assertInstanceOf(DistributionSpecHash.class, + result.getDistributionSpec()); + Assertions.assertEquals(ShuffleType.EXECUTION_BUCKETED, output.getShuffleType()); + Assertions.assertEquals(HashType.CRC32, output.getHashType()); + Assertions.assertEquals(Lists.newArrayList(out1.getExprId(), out2.getExprId()), + output.getOrderedShuffledColumns()); + } + + /** + * The storage-layout branch must keep advertising the basic child's layout for a NON-EMPTY key + * set (table id and partition ids ride along, keyed by set-operation outputs). + */ + @Test + void testSetOperationStorageLayoutOutputKeepsLayout() { + SlotReference left1 = slot("l1", 1); + SlotReference left2 = slot("l2", 2); + SlotReference right1 = slot("r1", 3); + SlotReference right2 = slot("r2", 4); + SlotReference out1 = slot("o1", 5); + SlotReference out2 = slot("o2", 6); + PhysicalSetOperation setOperation = unionOf(Lists.newArrayList(left1, left2), + Lists.newArrayList(right1, right2), out1, out2); + + PhysicalProperties leftChild = new PhysicalProperties(new DistributionSpecHash( + Lists.newArrayList(left1.getExprId()), ShuffleType.STORAGE_BUCKETED, 100L, 7L, + Collections.emptySet(), HashType.IDENTITY)); + PhysicalProperties rightChild = new PhysicalProperties(new DistributionSpecHash( + Lists.newArrayList(right1.getExprId()), ShuffleType.STORAGE_BUCKETED, 100L, 7L, + Collections.emptySet(), HashType.IDENTITY)); + PhysicalProperties result = new ChildOutputPropertyDeriver(Lists.newArrayList(leftChild, rightChild)) + .getOutputProperties(null, new GroupExpression(setOperation)); + + DistributionSpecHash output = Assertions.assertInstanceOf(DistributionSpecHash.class, + result.getDistributionSpec()); + Assertions.assertEquals(ShuffleType.STORAGE_BUCKETED, output.getShuffleType()); + Assertions.assertEquals(HashType.IDENTITY, output.getHashType()); + Assertions.assertEquals(100L, output.getTableId()); + Assertions.assertEquals(Lists.newArrayList(out1.getExprId()), output.getOrderedShuffledColumns()); + } + + /** + * A set operation whose basic child shuffles on ZERO columns must not advertise a zero-key + * hash spec: containsSatisfy() is vacuously true on the empty equivalence map, so such a + * spec satisfies any hash REQUIRE demand and suppresses the parent's exchange. The empty + * key set must fall through to the generic loop, whose offset mapping cannot resolve any + * child output and degrades to a non-hash property (ANY/STORAGE_ANY) instead. + */ + @Test + void testSetOperationZeroShuffleKeysNormalizesToGather() { + SlotReference left1 = slot("l1", 1); + SlotReference left2 = slot("l2", 2); + SlotReference right1 = slot("r1", 3); + SlotReference right2 = slot("r2", 4); + SlotReference out1 = slot("o1", 5); + SlotReference out2 = slot("o2", 6); + PhysicalSetOperation setOperation = unionOf(Lists.newArrayList(left1, left2), + Lists.newArrayList(right1, right2), out1, out2); + + // Zero shuffled columns on the basic child: the storage-layout branch used to accept + // 0 == 0 and return a zero-key DistributionSpecHash (the hazard); it must now fall + // through to the generic loop, which returns a non-hash property. + PhysicalProperties zeroKeyChild = new PhysicalProperties(new DistributionSpecHash( + Collections.emptyList(), ShuffleType.STORAGE_BUCKETED, 100L, 7L, + Collections.emptySet(), HashType.IDENTITY)); + PhysicalProperties otherChild = new PhysicalProperties(new DistributionSpecHash( + Lists.newArrayList(right1.getExprId()), ShuffleType.STORAGE_BUCKETED, 100L, 7L, + Collections.emptySet(), HashType.IDENTITY)); + PhysicalProperties result = new ChildOutputPropertyDeriver( + Lists.newArrayList(zeroKeyChild, otherChild)) + .getOutputProperties(null, new GroupExpression(setOperation)); + + Assertions.assertFalse(result.getDistributionSpec() instanceof DistributionSpecHash, + "zero shuffled columns must not produce a zero-key hash spec, got: " + + result.getDistributionSpec().getClass().getSimpleName()); + } } diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/DistributionSpecHashTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/DistributionSpecHashTest.java index da99ec15b6d624..16f9542bda906c 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/DistributionSpecHashTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/properties/DistributionSpecHashTest.java @@ -17,6 +17,7 @@ package org.apache.doris.nereids.properties; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.nereids.properties.DistributionSpecHash.ShuffleType; import org.apache.doris.nereids.trees.expressions.ExprId; @@ -50,12 +51,13 @@ public void testWithShuffleExprsSubset() { map.put(e6, 2); DistributionSpecHash origin = new DistributionSpecHash( Lists.newArrayList(e1, e2, e3), - ShuffleType.EXECUTION_BUCKETED, + ShuffleType.STORAGE_BUCKETED, 0L, -1L, Sets.newHashSet(0L), Lists.newArrayList(Sets.newHashSet(e1, e4), Sets.newHashSet(e2, e5), Sets.newHashSet(e3, e6)), - map + map, + HashType.IDENTITY ); // retain middle slot only (original index 1): map renumbered to 0 in the new spec @@ -67,6 +69,7 @@ public void testWithShuffleExprsSubset() { expectedMiddle.put(e2, 0); expectedMiddle.put(e5, 0); Assertions.assertEquals(expectedMiddle, middleOnly.getExprIdToEquivalenceSet()); + Assertions.assertEquals(HashType.IDENTITY, middleOnly.getHashType()); } @Test @@ -387,4 +390,122 @@ public void testHashEqualSatisfyWithDifferentLength() { Assertions.assertFalse(bucketed1.satisfy(bucketed2)); Assertions.assertFalse(bucketed2.satisfy(bucketed1)); } + + @Test + public void testExecutionBucketedNormalizesStorageHashType() { + DistributionSpecHash fromIdentity = new DistributionSpecHash( + Lists.newArrayList(new ExprId(1)), ShuffleType.EXECUTION_BUCKETED, + 1L, -1L, Sets.newHashSet(1L), HashType.IDENTITY); + DistributionSpecHash fromCrc32 = new DistributionSpecHash( + Lists.newArrayList(new ExprId(2)), ShuffleType.EXECUTION_BUCKETED, + 2L, -1L, Sets.newHashSet(2L), HashType.CRC32); + + Assertions.assertEquals(HashType.CRC32, fromIdentity.getHashType()); + Assertions.assertEquals(HashType.CRC32, fromCrc32.getHashType()); + Assertions.assertDoesNotThrow(() -> DistributionSpecHash.merge(fromIdentity, fromCrc32)); + } + + @Test + public void testMergeRejectsDifferentHashTypes() { + DistributionSpecHash crc32 = naturalSpec(HashType.CRC32); + DistributionSpecHash identity = naturalSpec(HashType.IDENTITY); + Assertions.assertThrows(IllegalStateException.class, () -> DistributionSpecHash.merge(crc32, identity)); + } + + // Two NATURAL specs identical except for hashType must be unequal and hash differently, so the + // memo (which keys PhysicalProperties on DistributionSpecHash) never collapses a crc32 and an + // identity distribution into the same group entry and mis-shares their enforcer/cost. The + // hashCode inequality is asserted via container behavior (HashSet keeps two entries) rather + // than assertNotEquals on the raw hashCodes: unequal objects are only contractually allowed to + // collide, so a direct comparison could fail for a correct implementation. + @Test + public void testEqualsAndHashCodeConsiderHashType() { + DistributionSpecHash crc32 = naturalSpec(HashType.CRC32); + DistributionSpecHash crc32Same = naturalSpec(HashType.CRC32); + DistributionSpecHash identity = naturalSpec(HashType.IDENTITY); + + Assertions.assertEquals(crc32, crc32Same); + Assertions.assertEquals(crc32.hashCode(), crc32Same.hashCode()); + Assertions.assertNotEquals(crc32, identity); + + java.util.Set distinct = new java.util.HashSet<>(); + distinct.add(crc32); + distinct.add(identity); + Assertions.assertEquals(2, distinct.size(), "distinct hash types must stay distinct keys"); + java.util.Map map = new java.util.HashMap<>(); + map.put(crc32, 1); + map.put(identity, 2); + Assertions.assertEquals(2, map.size()); + } + + // satisfy()'s equal branch (NATURAL/STORAGE_BUCKETED/EXECUTION_BUCKETED target) must reject a + // provider whose hashType differs, otherwise a crc32-bucketed child would be wrongly accepted as + // satisfying an identity NATURAL requirement (and vice versa) and skip the needed reshuffle. + @Test + public void testSatisfyEqualBranchChecksHashType() { + DistributionSpecHash crc32Provider = naturalSpec(HashType.CRC32); + DistributionSpecHash crc32Required = naturalSpec(HashType.CRC32); + DistributionSpecHash identityRequired = naturalSpec(HashType.IDENTITY); + + Assertions.assertTrue(crc32Provider.satisfy(crc32Required)); + Assertions.assertFalse(crc32Provider.satisfy(identityRequired)); + + DistributionSpecHash identityProvider = naturalSpec(HashType.IDENTITY); + Assertions.assertTrue(identityProvider.satisfy(identityRequired)); + Assertions.assertFalse(identityProvider.satisfy(crc32Required)); + } + + // The REQUIRE branch is hashType-agnostic: execution shuffle is always crc32, and a REQUIRE spec + // defaults to CRC32. An identity NATURAL/bucketed provider must still satisfy a plain REQUIRE so + // identity single-table plans are not broken. + @Test + public void testSatisfyRequireBranchIgnoresHashType() { + DistributionSpecHash require = new DistributionSpecHash( + Lists.newArrayList(new ExprId(1), new ExprId(2)), + ShuffleType.REQUIRE, + 1, + Sets.newHashSet(1L), + Lists.newArrayList(Sets.newHashSet(new ExprId(1)), Sets.newHashSet(new ExprId(2))), + requireMap() + ); + + DistributionSpecHash naturalIdentity = new DistributionSpecHash( + Lists.newArrayList(new ExprId(1), new ExprId(2)), + ShuffleType.NATURAL, + 1, + -1L, + Sets.newHashSet(1L), + Lists.newArrayList(Sets.newHashSet(new ExprId(1)), Sets.newHashSet(new ExprId(2))), + requireMap(), + HashType.IDENTITY + ); + + Assertions.assertTrue(naturalIdentity.satisfy(require)); + } + + private Map requireMap() { + Map map = Maps.newHashMap(); + map.put(new ExprId(1), 0); + map.put(new ExprId(2), 1); + return map; + } + + private DistributionSpecHash naturalSpec(HashType hashType) { + Map map = Maps.newHashMap(); + map.put(new ExprId(0), 0); + map.put(new ExprId(1), 0); + map.put(new ExprId(2), 1); + map.put(new ExprId(3), 1); + return new DistributionSpecHash( + Lists.newArrayList(new ExprId(0), new ExprId(2)), + ShuffleType.NATURAL, + 0, + -1L, + Sets.newHashSet(0L), + Lists.newArrayList(Sets.newHashSet(new ExprId(0), new ExprId(1)), + Sets.newHashSet(new ExprId(2), new ExprId(3))), + map, + hashType + ); + } } diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/trees/expressions/functions/scalar/IdentityHashInternalTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/trees/expressions/functions/scalar/IdentityHashInternalTest.java new file mode 100644 index 00000000000000..fc7c792db475c7 --- /dev/null +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/trees/expressions/functions/scalar/IdentityHashInternalTest.java @@ -0,0 +1,120 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +package org.apache.doris.nereids.trees.expressions.functions.scalar; + +import org.apache.doris.nereids.exceptions.AnalysisException; +import org.apache.doris.nereids.trees.expressions.Expression; +import org.apache.doris.nereids.trees.expressions.SlotReference; +import org.apache.doris.nereids.trees.expressions.literal.BigIntLiteral; +import org.apache.doris.nereids.trees.expressions.literal.IntegerLiteral; +import org.apache.doris.nereids.trees.expressions.literal.StringLiteral; +import org.apache.doris.nereids.types.IntegerType; +import org.apache.doris.nereids.types.StringType; + +import com.google.common.collect.ImmutableList; +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +/** + * The trailing bucket count of identity_hash_internal must be a positive integer literal. + * These checks reject malformed calls at analysis time, before the expression reaches BE, + * whose implementation expects the modulus as a constant column. Non-constant or non-positive + * counts previously crashed BE (DCHECK abort in debug, nullptr dereference / division by + * zero in release). + */ +public class IdentityHashInternalTest { + + private static final SlotReference INT_COLUMN = new SlotReference( + "c1", IntegerType.INSTANCE, false, ImmutableList.of()); + private static final SlotReference STRING_COLUMN = new SlotReference( + "s1", StringType.INSTANCE, false, ImmutableList.of()); + + @Test + public void testValidPositiveIntegerLiteralPasses() { + for (Expression bucketCount : ImmutableList.of(new IntegerLiteral(1), + new IntegerLiteral(8), new IntegerLiteral(Integer.MAX_VALUE))) { + Expression expr = new IdentityHashInternal(INT_COLUMN, bucketCount); + // positive integer literal must be accepted + Assertions.assertNotNull(expr.withChildren(ImmutableList.of(INT_COLUMN, bucketCount))); + } + } + + @Test + public void testValidMultiColumnWithLiteralPasses() { + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, STRING_COLUMN, + new IntegerLiteral(8)); + Assertions.assertNotNull(expr.withChildren( + ImmutableList.of(INT_COLUMN, STRING_COLUMN, new IntegerLiteral(8)))); + } + + @Test + public void testRejectsNonLiteralBucketCount() { + // the trailing argument is a column, not a literal + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, INT_COLUMN); + Assertions.assertThrows(AnalysisException.class, + () -> expr.withChildren(ImmutableList.of(INT_COLUMN, INT_COLUMN)), + "non-literal bucket count must be rejected"); + } + + @Test + public void testRejectsStringLiteralBucketCount() { + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, new StringLiteral("8")); + Assertions.assertThrows(AnalysisException.class, + () -> expr.withChildren(ImmutableList.of(INT_COLUMN, new StringLiteral("8"))), + "string literal bucket count must be rejected"); + } + + @Test + public void testRejectsZeroBucketCount() { + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, new IntegerLiteral(0)); + AnalysisException e = Assertions.assertThrows(AnalysisException.class, + () -> expr.withChildren(ImmutableList.of(INT_COLUMN, new IntegerLiteral(0))), + "zero bucket count must be rejected"); + Assertions.assertTrue(e.getMessage().contains("positive integer"), e.getMessage()); + } + + @Test + public void testRejectsNegativeBucketCount() { + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, new IntegerLiteral(-8)); + AnalysisException e = Assertions.assertThrows(AnalysisException.class, + () -> expr.withChildren(ImmutableList.of(INT_COLUMN, new IntegerLiteral(-8))), + "negative bucket count must be rejected"); + Assertions.assertTrue(e.getMessage().contains("positive integer"), e.getMessage()); + } + + @Test + public void testRejectsBucketCountOverflowingUint32() { + // BE's modulus is uint32_t; anything above Integer.MAX_VALUE cannot be a bucket count. + Expression overflow = new BigIntLiteral(Integer.MAX_VALUE + 1L); + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, overflow); + Assertions.assertThrows(AnalysisException.class, + () -> expr.withChildren(ImmutableList.of(INT_COLUMN, overflow)), + "bucket count above Integer.MAX_VALUE must be rejected"); + } + + @Test + public void testSignatureStillAcceptsAnyDataColumns() { + // The leading distribution columns keep AnyDataType varArgs: the check only constrains + // the trailing bucket count, so the function still parses with any typed columns. + IdentityHashInternal expr = new IdentityHashInternal(INT_COLUMN, new IntegerLiteral(8)); + Assertions.assertNotNull(expr.withChildren( + ImmutableList.of(INT_COLUMN, new IntegerLiteral(8)))); + Assertions.assertTrue(expr.getSignatures().get(0).hasVarArgs, + "identity_hash_internal must stay variadic"); + } +} diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/util/JoinUtilsTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/util/JoinUtilsTest.java index 1958d6d84a2b27..efc1db5870cbf6 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/util/JoinUtilsTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/util/JoinUtilsTest.java @@ -20,6 +20,7 @@ import org.apache.doris.catalog.ColocateTableIndex; import org.apache.doris.catalog.ColocateTableIndex.GroupId; import org.apache.doris.catalog.Env; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.nereids.properties.DistributionSpecHash; import org.apache.doris.nereids.properties.DistributionSpecHash.ShuffleType; import org.apache.doris.nereids.trees.expressions.Add; @@ -310,4 +311,47 @@ public void testCouldColocateJoinForDiffTableNotInSameGroup() { Assertions.assertFalse(JoinUtils.couldColocateJoin(left, right, conjuncts)); } } + + // Two NATURAL sides with the same non-crc32 hashType (IDENTITY) can still colocate: same storage + // hash means each side's bucket layout matches, so no reshuffle is needed. + @Test + public void testCouldColocateJoinForSameIdentityHashType() { + ConnectContext ctx = new ConnectContext(); + ctx.setThreadLocalInfo(); + + DistributionSpecHash left = new DistributionSpecHash(Lists.newArrayList(new ExprId(1)), + ShuffleType.NATURAL, 1L, 1L, Collections.emptySet(), HashType.IDENTITY); + DistributionSpecHash right = new DistributionSpecHash(Lists.newArrayList(new ExprId(2)), + ShuffleType.NATURAL, 1L, 1L, Collections.emptySet(), HashType.IDENTITY); + + Expression leftKey1 = new SlotReference(new ExprId(1), "c1", + TinyIntType.INSTANCE, false, Lists.newArrayList()); + Expression rightKey1 = new SlotReference(new ExprId(2), "c1", + TinyIntType.INSTANCE, false, Lists.newArrayList()); + + List conjuncts = Lists.newArrayList(new EqualTo(leftKey1, rightKey1)); + Assertions.assertTrue(JoinUtils.couldColocateJoin(left, right, conjuncts)); + } + + // Different hashType on the two NATURAL sides (crc32 vs identity) must NOT colocate: the storage + // bucket layouts differ, so a bucket-local join would read mismatched buckets — the correctness + // red line. Guarded by JoinUtils.couldColocateJoin's leftHashSpec/rightHashSpec hashType check. + @Test + public void testCouldNotColocateJoinForDifferentHashType() { + ConnectContext ctx = new ConnectContext(); + ctx.setThreadLocalInfo(); + + DistributionSpecHash left = new DistributionSpecHash(Lists.newArrayList(new ExprId(1)), + ShuffleType.NATURAL, 1L, 1L, Collections.emptySet(), HashType.CRC32); + DistributionSpecHash right = new DistributionSpecHash(Lists.newArrayList(new ExprId(2)), + ShuffleType.NATURAL, 1L, 1L, Collections.emptySet(), HashType.IDENTITY); + + Expression leftKey1 = new SlotReference(new ExprId(1), "c1", + TinyIntType.INSTANCE, false, Lists.newArrayList()); + Expression rightKey1 = new SlotReference(new ExprId(2), "c1", + TinyIntType.INSTANCE, false, Lists.newArrayList()); + + List conjuncts = Lists.newArrayList(new EqualTo(leftKey1, rightKey1)); + Assertions.assertFalse(JoinUtils.couldColocateJoin(left, right, conjuncts)); + } } diff --git a/fe/fe-core/src/test/java/org/apache/doris/planner/HashDistributionPrunerTest.java b/fe/fe-core/src/test/java/org/apache/doris/planner/HashDistributionPrunerTest.java index a475b923874553..683dcf0664dcb7 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/planner/HashDistributionPrunerTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/planner/HashDistributionPrunerTest.java @@ -18,15 +18,23 @@ package org.apache.doris.planner; import org.apache.doris.analysis.Expr; +import org.apache.doris.analysis.IPv4Literal; +import org.apache.doris.analysis.IPv6Literal; import org.apache.doris.analysis.InPredicate; +import org.apache.doris.analysis.IntLiteral; +import org.apache.doris.analysis.LargeIntLiteral; +import org.apache.doris.analysis.LiteralExpr; +import org.apache.doris.analysis.NullLiteral; import org.apache.doris.analysis.SlotRef; import org.apache.doris.analysis.StringLiteral; +import org.apache.doris.analysis.TimeStampNsLiteral; +import org.apache.doris.analysis.VarBinaryLiteral; import org.apache.doris.catalog.Column; +import org.apache.doris.catalog.HashDistributionInfo.HashType; import org.apache.doris.catalog.LocalTablet; import org.apache.doris.catalog.MaterializedIndex; import org.apache.doris.catalog.PartitionKey; import org.apache.doris.catalog.PrimitiveType; -import org.apache.doris.catalog.Tablet; import com.google.common.collect.Lists; import com.google.common.collect.Sets; @@ -34,6 +42,8 @@ import org.junit.jupiter.api.Assertions; import org.junit.jupiter.api.Test; +import java.math.BigInteger; +import java.time.LocalDateTime; import java.util.Collection; import java.util.List; import java.util.Map; @@ -44,13 +54,9 @@ public class HashDistributionPrunerTest { @Test public void test() { List tabletIds = Lists.newArrayListWithExpectedSize(300); - List indexTablets = Lists.newArrayListWithExpectedSize(300); for (long i = 0; i < 300; i++) { tabletIds.add(i); - indexTablets.add(new LocalTablet(i)); } - MaterializedIndex index = new MaterializedIndex(); - index.appendTablets(indexTablets); // distribution columns Column dealDate = new Column("dealDate", PrimitiveType.DATE, false); @@ -98,6 +104,7 @@ public void test() { filters.put("CHANNEL", channelFilter); filters.put("SHOP_TYPE", shopTypeFilter); + MaterializedIndex index = createMaterializedIndex(tabletIds); HashDistributionPruner pruner = new HashDistributionPruner(null, index, columns, filters, tabletIds.size(), true); @@ -146,6 +153,129 @@ public void test() { Assertions.assertEquals(39, tablets.size()); } + // Identity bucketing treats each value's canonical bytes as an unsigned integer with its first + // byte least significant, then appends multiple columns before taking the bucket modulus. This + // must remain bit-identical with BE tablet routing and bucket-shuffle partitioning. + @Test + public void testIdentityPrune() { + List tabletIds = Lists.newArrayListWithExpectedSize(512); + for (long i = 0; i < 512; i++) { + tabletIds.add(i); + } + Column shardNum = new Column("shard_num", PrimitiveType.BIGINT, false); + List columns = Lists.newArrayList(shardNum); + + // in-range: shard_num = 100 -> 100 % 512 = 100 + assertIdentityBucket(tabletIds, columns, "SHARD_NUM", new IntLiteral(100), 100L); + // wraps: 600 % 512 = 88 + assertIdentityBucket(tabletIds, columns, "SHARD_NUM", new IntLiteral(600), 88L); + // Two's-complement bytes are interpreted as unsigned. A power-of-two modulus therefore + // still maps -1 to the final bucket. + assertIdentityBucket(tabletIds, columns, "SHARD_NUM", new IntLiteral(-1), 511L); + + // LARGEINT uses all 128 bits of its canonical little-endian representation. Both moduli + // are needed: 512 mirrors the common power-of-two bucket count, and the odd modulus 251 + // (2^100 + 5) % 251 = 24 != 5 % 251, so an implementation that truncates to the low + // 32/64 bits or collapses to value % 2^k still fails. + Column bigId = new Column("big_id", PrimitiveType.LARGEINT, false); + List bigCols = Lists.newArrayList(bigId); + BigInteger huge = BigInteger.ONE.shiftLeft(100).add(BigInteger.valueOf(5)); + long expected = huge.mod(BigInteger.valueOf(512)).longValue(); + assertIdentityBucket(tabletIds, bigCols, "BIG_ID", new LargeIntLiteral(huge), expected); + List oddTablets = Lists.newArrayListWithExpectedSize(251); + for (long i = 0; i < 251; i++) { + oddTablets.add(i); + } + long oddExpected = huge.mod(BigInteger.valueOf(251)).longValue(); + Assertions.assertNotEquals(oddExpected, BigInteger.valueOf(5) + .mod(BigInteger.valueOf(251)).longValue(), + "odd-modulus vector must not collapse to the low bits"); + assertIdentityBucket(oddTablets, bigCols, "BIG_ID", new LargeIntLiteral(huge), oddExpected); + + // With a non-power-of-two bucket count, -1 is UINT32_MAX rather than signed -1. + List tenTablets = Lists.newArrayListWithExpectedSize(10); + for (long i = 0; i < 10; i++) { + tenTablets.add(i); + } + assertIdentityBucket(tenTablets, columns, "SHARD_NUM", new IntLiteral(-1), 5L); + } + + @Test + public void testIdentityPruneWithMultipleTypedColumns() { + List tabletIds = Lists.newArrayListWithExpectedSize(257); + for (long i = 0; i < 257; i++) { + tabletIds.add(i); + } + List columns = Lists.newArrayList( + new Column("id", PrimitiveType.INT, false), + new Column("name", PrimitiveType.VARCHAR, false)); + + Map filters = new CaseInsensitiveMap(); + PartitionColumnFilter idFilter = new PartitionColumnFilter(); + idFilter.setLowerBound(new IntLiteral(1), true); + idFilter.setUpperBound(new IntLiteral(1), true); + filters.put("ID", idFilter); + PartitionColumnFilter nameFilter = new PartitionColumnFilter(); + nameFilter.setLowerBound(new StringLiteral("A"), true); + nameFilter.setUpperBound(new StringLiteral("A"), true); + filters.put("NAME", nameFilter); + + MaterializedIndex index = createMaterializedIndex(tabletIds); + HashDistributionPruner pruner = new HashDistributionPruner(null, index, columns, filters, + tabletIds.size(), true, HashType.IDENTITY); + // append(uint32_le(1), bytes("A")) = 1 * 256 + 65; 321 % 257 = 64 + Assertions.assertEquals(Lists.newArrayList(64L), pruner.prune()); + } + + @Test + public void testIdentityNullCanonicalBytes() { + PartitionKey nullKey = new PartitionKey(); + nullKey.pushColumn(new NullLiteral(), PrimitiveType.INT); + Assertions.assertEquals(0, nullKey.getIdentityHashValue(257)); + + nullKey.pushColumn(new StringLiteral("A"), PrimitiveType.VARCHAR); + Assertions.assertEquals(65, nullKey.getIdentityHashValue(257)); + } + + @Test + public void testIdentityTimestampNsCanonicalBytes() { + PartitionKey timestamp = new PartitionKey(); + timestamp.pushColumn(new TimeStampNsLiteral( + LocalDateTime.of(1970, 1, 1, 0, 0, 0, 1)), PrimitiveType.TIMESTAMP_NS); + Assertions.assertEquals(1, timestamp.getIdentityHashValue(257)); + } + + @Test + public void testIdentityPruneWithIpAndVarBinaryCanonicalBytes() throws Exception { + PartitionKey ipv4 = new PartitionKey(); + ipv4.pushColumn(new IPv4Literal("1.2.3.4"), PrimitiveType.IPV4); + Assertions.assertEquals(2, ipv4.getIdentityHashValue(257)); + + PartitionKey ipv6 = new PartitionKey(); + ipv6.pushColumn(new IPv6Literal("::1"), PrimitiveType.IPV6); + Assertions.assertEquals(1, ipv6.getIdentityHashValue(257)); + + PartitionKey varBinary = new PartitionKey(); + varBinary.pushColumn(new VarBinaryLiteral(new byte[] {(byte) 0xff, 0}), PrimitiveType.VARBINARY); + Assertions.assertEquals(255, varBinary.getIdentityHashValue(257)); + } + + private void assertIdentityBucket(List tabletIds, List columns, String colName, Expr value, + long expectedBucket) { + PartitionColumnFilter filter = new PartitionColumnFilter(); + filter.setLowerBound((LiteralExpr) value, true); + filter.setUpperBound((LiteralExpr) value, true); + Map filters = new CaseInsensitiveMap(); + filters.put(colName, filter); + + MaterializedIndex index = createMaterializedIndex(tabletIds); + HashDistributionPruner pruner = new HashDistributionPruner(null, index, columns, filters, tabletIds.size(), + true, HashType.IDENTITY); + Collection results = pruner.prune(); + Assertions.assertEquals(1, results.size()); + Assertions.assertEquals(Long.valueOf(expectedBucket), results.iterator().next()); + } + @Test public void testPruneWithMaterializedIndex() { List tabletIds = Lists.newArrayListWithExpectedSize(8); @@ -185,4 +315,12 @@ public void testPruneWithMaterializedIndex() { Assertions.assertEquals(tabletIds, Lists.newArrayList(allIndexTablets)); } + private MaterializedIndex createMaterializedIndex(List tabletIds) { + MaterializedIndex index = new MaterializedIndex(); + for (long tabletId : tabletIds) { + index.addTablet(new LocalTablet(tabletId), null, true); + } + return index; + } + } diff --git a/fe/fe-core/src/test/java/org/apache/doris/planner/LocalShuffleNodeCoverageTest.java b/fe/fe-core/src/test/java/org/apache/doris/planner/LocalShuffleNodeCoverageTest.java index e8bcbb195aa1d3..0a19c97e6a378f 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/planner/LocalShuffleNodeCoverageTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/planner/LocalShuffleNodeCoverageTest.java @@ -23,6 +23,7 @@ import org.apache.doris.analysis.Expr; import org.apache.doris.analysis.FunctionCallExpr; import org.apache.doris.analysis.GroupingInfo; +import org.apache.doris.analysis.IntLiteral; import org.apache.doris.analysis.JoinOperator; import org.apache.doris.analysis.OrderByElement; import org.apache.doris.analysis.SlotDescriptor; @@ -31,6 +32,8 @@ import org.apache.doris.analysis.TupleDescriptor; import org.apache.doris.analysis.TupleId; import org.apache.doris.catalog.FunctionName; +import org.apache.doris.catalog.HashDistributionInfo; +import org.apache.doris.common.Config; import org.apache.doris.common.Pair; import org.apache.doris.common.UserException; import org.apache.doris.nereids.glue.translator.PlanTranslatorContext; @@ -40,6 +43,7 @@ import org.apache.doris.planner.LocalExchangeNode.LocalExchangeTypeRequire; import org.apache.doris.qe.ConnectContext; import org.apache.doris.qe.SessionVariable; +import org.apache.doris.thrift.TDistributionHashType; import org.apache.doris.thrift.TExplainLevel; import org.apache.doris.thrift.TPartitionType; import org.apache.doris.thrift.TPlanNode; @@ -58,6 +62,155 @@ public class LocalShuffleNodeCoverageTest { private static final AtomicInteger NEXT_ID = new AtomicInteger(1); + @Test + public void testIdentityHashTypeRequiresSupportedExecutionVersion() { + int originalVersion = Config.be_exec_version; + try { + Config.be_exec_version = Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION - 1; + DataPartition identityPartition = new DataPartition( + TPartitionType.BUCKET_SHFFULE_HASH_PARTITIONED, + Collections.singletonList(new IntLiteral(1)), HashDistributionInfo.HashType.IDENTITY); + IllegalStateException exception = Assertions.assertThrows(IllegalStateException.class, + identityPartition::toThrift); + Assertions.assertTrue(exception.getMessage().contains("IDENTITY distribution requires")); + + Config.be_exec_version = Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION; + Assertions.assertEquals(TDistributionHashType.IDENTITY, + identityPartition.toThrift().getDistributionHashType()); + Config.be_exec_version = Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION - 1; + Assertions.assertEquals(TDistributionHashType.CRC32, + DataPartition.toTHashType(HashDistributionInfo.HashType.CRC32)); + } finally { + Config.be_exec_version = originalVersion; + } + } + + @Test + public void testIdentityHashTypePropagatesThroughLocalExchangeAndFragment() { + TrackingPlanNode identityChild = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.IDENTITY; + } + }; + LocalExchangeNode passthrough = new LocalExchangeNode(nextPlanNodeId(), identityChild, + LocalExchangeType.PASSTHROUGH, null); + LocalExchangeNode bucket = new LocalExchangeNode(nextPlanNodeId(), passthrough, + LocalExchangeType.BUCKET_HASH_SHUFFLE, Collections.emptyList()); + Assertions.assertEquals(HashDistributionInfo.HashType.IDENTITY, + bucket.getStorageDistributionHashType()); + Assertions.assertEquals(TDistributionHashType.IDENTITY, + bucket.treeToThrift().getNodes().get(0).getLocalExchangeNode().getDistributionHashType()); + + PlanFragment fragment = new PlanFragment(new PlanFragmentId(1), bucket, DataPartition.UNPARTITIONED); + Assertions.assertEquals(TDistributionHashType.IDENTITY, fragment.toThrift().getDistributionHashType()); + } + + /** + * A fragment whose subtree mixes IDENTITY and CRC32 bucket layouts must be rejected instead + * of silently serializing without a hash type: BE-native bucket local exchanges would then + * fall back to CRC32 and mis-bucket the identity-routed rows. + */ + @Test + public void testFragmentRejectsMixedStorageHashTypes() { + TrackingPlanNode identityScan = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.IDENTITY; + } + + @Override + protected HashDistributionInfo.HashType getOwnStorageHashType() { + return HashDistributionInfo.HashType.IDENTITY; + } + }; + TrackingPlanNode crc32Scan = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.CRC32; + } + + @Override + protected HashDistributionInfo.HashType getOwnStorageHashType() { + return HashDistributionInfo.HashType.CRC32; + } + }; + // A set operation over two layouts: getStorageDistributionHashType() is null (mixed), and + // the collector sees both opinions, so serialization must refuse the fragment. + UnionNode mixed = new UnionNode(nextPlanNodeId(), new TupleId(123)); + mixed.addChild(identityScan); + mixed.addChild(crc32Scan); + Assertions.assertNull(mixed.getStorageDistributionHashType()); + + PlanFragment fragment = new PlanFragment(new PlanFragmentId(1), mixed, DataPartition.UNPARTITIONED); + IllegalStateException rejected = Assertions.assertThrows(IllegalStateException.class, + fragment::toThrift); + Assertions.assertTrue(rejected.getMessage().contains("mixes distribution hash types"), + rejected.getMessage()); + } + + /** + * A fragment without any bucketed storage (no node declares a layout) still serializes: the + * mixed-layout rejection must not fire for layout-less fragments, and the serialized hash + * type stays at the thrift default (CRC32). + */ + @Test + public void testFragmentWithoutStorageLayoutStillSerializes() { + TrackingPlanNode layoutless = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP); + Assertions.assertNull(layoutless.getStorageDistributionHashType()); + + PlanFragment fragment = new PlanFragment(new PlanFragmentId(1), layoutless, DataPartition.UNPARTITIONED); + Assertions.assertEquals(TDistributionHashType.CRC32, + fragment.toThrift().getDistributionHashType()); + } + + /** + * collectStorageHashTypes de-duplicates: a unary chain of passthrough local exchanges over + * one identity scan contributes exactly one opinion, so a null root derivation caused by a + * multi-input node with one silent child is not misreported as mixed when it is not. + */ + @Test + public void testCollectStorageHashTypesDeduplicatesUnaryChain() { + TrackingPlanNode identityScan = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.IDENTITY; + } + }; + LocalExchangeNode passthrough = new LocalExchangeNode(nextPlanNodeId(), identityScan, + LocalExchangeType.PASSTHROUGH, null); + java.util.Set collected = new java.util.HashSet<>(); + passthrough.collectStorageHashTypes(collected); + Assertions.assertEquals(Collections.singleton(HashDistributionInfo.HashType.IDENTITY), collected); + } + + @Test + public void testBroadcastJoinPreservesProbeStorageHashType() { + TrackingPlanNode identityProbe = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.IDENTITY; + } + }; + TrackingPlanNode crc32Build = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.CRC32; + } + }; + HashJoinNode broadcastJoin = new HashJoinNode(nextPlanNodeId(), identityProbe, crc32Build, + JoinOperator.INNER_JOIN, Collections.singletonList(Mockito.mock(BinaryPredicate.class)), + Collections.emptyList(), null, null, false); + broadcastJoin.setDistributionMode(DistributionMode.BROADCAST); + + Assertions.assertEquals(HashDistributionInfo.HashType.IDENTITY, + broadcastJoin.getStorageDistributionHashType()); + LocalExchangeNode bucketExchange = new LocalExchangeNode(nextPlanNodeId(), broadcastJoin, + LocalExchangeType.BUCKET_HASH_SHUFFLE, Collections.emptyList()); + Assertions.assertEquals(HashDistributionInfo.HashType.IDENTITY, + bucketExchange.getStorageDistributionHashType()); + } + @Test public void testRequireSpecificAutoRequireHashPreservesSpecificHash() { // Pass-through operators (union / streaming agg / sort) forward their parent's specific @@ -547,6 +700,31 @@ public void testLayer1SkipUsesIsSerialOperatorOnBeNotIsSerialNode() { + "even if isSerialNode()=true — BE treats the node as non-serial."); } + @Test + public void testNestedLoopJoinPreservesProbeStorageHashType() { + TrackingPlanNode identityProbe = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.IDENTITY; + } + }; + TrackingPlanNode crc32Build = new TrackingPlanNode(nextPlanNodeId(), LocalExchangeType.NOOP) { + @Override + public HashDistributionInfo.HashType getStorageDistributionHashType() { + return HashDistributionInfo.HashType.CRC32; + } + }; + NestedLoopJoinNode nestedLoopJoin = new NestedLoopJoinNode(nextPlanNodeId(), identityProbe, crc32Build, + Lists.newArrayList(new TupleId(NEXT_ID.getAndIncrement())), JoinOperator.CROSS_JOIN, false); + + Assertions.assertEquals(HashDistributionInfo.HashType.IDENTITY, + nestedLoopJoin.getStorageDistributionHashType()); + LocalExchangeNode bucketExchange = new LocalExchangeNode(nextPlanNodeId(), nestedLoopJoin, + LocalExchangeType.BUCKET_HASH_SHUFFLE, Collections.emptyList()); + Assertions.assertEquals(HashDistributionInfo.HashType.IDENTITY, + bucketExchange.getStorageDistributionHashType()); + } + @Test public void testPassToOneBoundaryKeepsParallelSubtreeLocalExchange() { PlanTranslatorContext ctx = new PlanTranslatorContext(); diff --git a/fe/fe-core/src/test/java/org/apache/doris/planner/OlapTableSinkTest.java b/fe/fe-core/src/test/java/org/apache/doris/planner/OlapTableSinkTest.java index 7713a687fa9c6a..7fd7808e2a4aa3 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/planner/OlapTableSinkTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/planner/OlapTableSinkTest.java @@ -17,12 +17,17 @@ package org.apache.doris.planner; +import org.apache.doris.catalog.Column; import org.apache.doris.catalog.Env; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.catalog.OlapTable; +import org.apache.doris.catalog.PrimitiveType; +import org.apache.doris.common.Config; import org.apache.doris.planner.OlapTableSink.AdaptiveBucketAssignment; import org.apache.doris.planner.OlapTableSink.AdaptiveIndexBucketAssignment; import org.apache.doris.system.Backend; import org.apache.doris.system.SystemInfoService; +import org.apache.doris.thrift.TDistributionHashType; import org.apache.doris.thrift.TOlapTableIndexTablets; import org.apache.doris.thrift.TOlapTableLocationParam; import org.apache.doris.thrift.TOlapTablePartition; @@ -40,6 +45,28 @@ import java.util.Map; public class OlapTableSinkTest { + @Test + public void testIdentityDistributionRequiresSupportedExecutionVersion() { + OlapTable table = Mockito.mock(OlapTable.class); + OlapTableSink sink = new OlapTableSink(table, null, Collections.emptyList()); + HashDistributionInfo identity = new HashDistributionInfo(8, + Collections.singletonList(new Column("id", PrimitiveType.BIGINT))); + identity.setHashType(HashDistributionInfo.HashType.IDENTITY); + + int originalVersion = Config.be_exec_version; + try { + Config.be_exec_version = Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION - 1; + Assertions.assertThrows(IllegalStateException.class, + () -> sink.getTDistributionHashType(identity)); + + Config.be_exec_version = Config.DISTRIBUTION_HASH_TYPE_MIN_BE_EXEC_VERSION; + Assertions.assertEquals(TDistributionHashType.IDENTITY, + sink.getTDistributionHashType(identity)); + } finally { + Config.be_exec_version = originalVersion; + } + } + @Test public void testCreateDummyLocationUsesLoadAvailableBackendInCurrentComputeGroup() throws Exception { SystemInfoService systemInfoService = Mockito.mock(SystemInfoService.class); diff --git a/fe/fe-core/src/test/java/org/apache/doris/qe/IdentitySetOperationTest.java b/fe/fe-core/src/test/java/org/apache/doris/qe/IdentitySetOperationTest.java new file mode 100644 index 00000000000000..c8928c6be556ad --- /dev/null +++ b/fe/fe-core/src/test/java/org/apache/doris/qe/IdentitySetOperationTest.java @@ -0,0 +1,248 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +package org.apache.doris.qe; + +import org.apache.doris.catalog.Env; +import org.apache.doris.catalog.HashDistributionInfo; +import org.apache.doris.catalog.HashDistributionInfo.HashType; +import org.apache.doris.catalog.OlapTable; +import org.apache.doris.catalog.Partition; +import org.apache.doris.nereids.NereidsPlanner; +import org.apache.doris.nereids.properties.DistributionSpecHash; +import org.apache.doris.nereids.properties.DistributionSpecHash.ShuffleType; +import org.apache.doris.nereids.trees.plans.Plan; +import org.apache.doris.nereids.trees.plans.physical.PhysicalDistribute; +import org.apache.doris.nereids.trees.plans.physical.PhysicalOlapScan; +import org.apache.doris.nereids.trees.plans.physical.PhysicalPlan; +import org.apache.doris.nereids.trees.plans.physical.PhysicalSetOperation; +import org.apache.doris.planner.ExchangeNode; +import org.apache.doris.planner.LocalExchangeNode; +import org.apache.doris.planner.LocalExchangeNode.LocalExchangeType; +import org.apache.doris.planner.PlanFragment; +import org.apache.doris.planner.PlanNode; +import org.apache.doris.planner.SetOperationNode; +import org.apache.doris.thrift.TDistributionHashType; +import org.apache.doris.thrift.TPartitionType; +import org.apache.doris.utframe.TestWithFeService; + +import org.junit.jupiter.api.Assertions; +import org.junit.jupiter.api.Test; + +import java.util.ArrayList; +import java.util.List; + +/** SQL-to-thrift coverage of the storage layout INSIDE a set operation, not its parent join. */ +public class IdentitySetOperationTest extends TestWithFeService { + @Override + protected int backendNum() { + return 3; + } + + @Override + protected void runBeforeAll() throws Exception { + createDatabase("identity_set_operation"); + useDatabase("identity_set_operation"); + createTable("CREATE TABLE identity8(id BIGINT NOT NULL) DISTRIBUTED BY HASH(id) BUCKETS 8 " + + "PROPERTIES('replication_num'='1', 'distribution_hash_type'='identity')"); + createTable("CREATE TABLE identity5(id BIGINT NOT NULL) DISTRIBUTED BY HASH(id) BUCKETS 5 " + + "PROPERTIES('replication_num'='1', 'distribution_hash_type'='identity')"); + createTable("CREATE TABLE crc7(id BIGINT NOT NULL) DISTRIBUTED BY HASH(id) BUCKETS 7 " + + "PROPERTIES('replication_num'='1')"); + SessionVariable sv = connectContext.getSessionVariable(); + sv.setEnableLocalShufflePlanner(true); + sv.setEnableLocalShuffle(true); + sv.setEnableNereidsDistributePlanner(true); + sv.setPipelineTaskNum("4"); + sv.setBucketShuffleDowngradeRatio(0); + // Force a serial basic scan so a BUCKET_HASH_SHUFFLE local exchange must align it. + sv.setForceToLocalShuffle(true); + // Keep the SQL child order so the matrix really tests both left and right basics. + sv.setDisableNereidsRules("REORDER_INTERSECT"); + } + + @Test + public void testIdentityLeftBasic() throws Exception { + checkSetOperations("identity8", "identity5", 0, HashType.IDENTITY); + } + + @Test + public void testIdentityRightBasic() throws Exception { + checkSetOperations("identity5", "identity8", 1, HashType.IDENTITY); + } + + @Test + public void testMixedIdentityTarget() throws Exception { + checkSetOperations("identity8", "crc7", 0, HashType.IDENTITY); + } + + @Test + public void testMixedCrc32Target() throws Exception { + checkSetOperations("identity8", "crc7", 1, HashType.CRC32); + } + + private void checkSetOperations(String left, String right, int basicIndex, HashType hashType) + throws Exception { + String[] tables = {left, right}; + for (int i = 0; i < tables.length; i++) { + // Mock backend reports rather than ALTER STATS, which requires the internal + // statistics repository (not started by TestWithFeService). + OlapTable table = (OlapTable) Env.getCurrentInternalCatalog() + .getDbOrMetaException("identity_set_operation").getTableOrMetaException(tables[i]); + for (Partition partition : table.getPartitions()) { + partition.getBaseIndex().setRowCount(i == basicIndex ? 10000 : 100); + partition.getBaseIndex().setRowCountReported(true); + } + } + for (String op : new String[] {"UNION ALL", "INTERSECT", "EXCEPT"}) { + String sql = "SELECT id, row_number() OVER (PARTITION BY id ORDER BY id) rn FROM " + + "(SELECT id FROM " + left + " " + op + " SELECT id FROM " + right + ") u"; + NereidsPlanner planner = (NereidsPlanner) executeNereidsSql("explain distributed plan " + sql) + .planner(); + Assertions.assertTrue(SessionVariable.canUseNereidsDistributePlanner(connectContext)); + List sets = new ArrayList<>(); + collectPhysicalSets(planner.getOptimizedPlan(), sets); + Assertions.assertEquals(1, sets.size(), sql); + PhysicalSetOperation set = sets.get(0); + DistributionSpecHash output = hashSpec(set); + Assertions.assertEquals(ShuffleType.NATURAL, output.getShuffleType(), sql); + Assertions.assertEquals(hashType, output.getHashType(), sql); + Assertions.assertInstanceOf(PhysicalOlapScan.class, set.child(basicIndex), sql); + Assertions.assertEquals(tables[basicIndex], + ((PhysicalOlapScan) set.child(basicIndex)).getTable().getName(), sql); + DistributionSpecHash basic = hashSpec((PhysicalPlan) set.child(basicIndex)); + Assertions.assertEquals(basic.getTableId(), output.getTableId(), sql); + Assertions.assertEquals(basic.getPartitionIds(), output.getPartitionIds(), sql); + Assertions.assertEquals(set.getOutput().get(0).getExprId(), output.getOrderedShuffledColumns().get(0)); + PhysicalDistribute shuffled = Assertions.assertInstanceOf(PhysicalDistribute.class, + set.child(1 - basicIndex), sql); + DistributionSpecHash shuffledSpec = hashSpec(shuffled); + Assertions.assertEquals(ShuffleType.STORAGE_BUCKETED, shuffledSpec.getShuffleType(), sql); + Assertions.assertEquals(hashType, shuffledSpec.getHashType(), sql); + checkTranslatedSet(planner.getFragments(), hashType, sql); + } + } + + private static DistributionSpecHash hashSpec(PhysicalPlan plan) { + return Assertions.assertInstanceOf(DistributionSpecHash.class, + plan.getPhysicalProperties().getDistributionSpec()); + } + + private static void collectPhysicalSets(Plan plan, List sets) { + if (plan instanceof PhysicalSetOperation) { + sets.add((PhysicalSetOperation) plan); + } + for (Plan child : plan.children()) { + collectPhysicalSets(child, sets); + } + } + + private static void checkTranslatedSet(List fragments, HashType hashType, String sql) { + List sets = new ArrayList<>(); + for (PlanFragment fragment : fragments) { + collectSetNodes(fragment.getPlanRoot(), sets); + } + Assertions.assertEquals(1, sets.size(), sql); + SetOperationNode set = sets.get(0); + Assertions.assertTrue(set.isBucketShuffle(), sql); + Assertions.assertEquals(hashType, set.getStorageDistributionHashType(), sql); + List remotes = new ArrayList<>(); + List locals = new ArrayList<>(); + for (PlanNode child : set.getChildren()) { + collectExchanges(child, remotes, locals); + } + Assertions.assertEquals(1, remotes.size(), sql); + ExchangeNode remote = remotes.get(0); + Assertions.assertEquals(TPartitionType.BUCKET_SHFFULE_HASH_PARTITIONED, remote.getPartitionType(), sql); + Assertions.assertEquals(hashType, remote.getDistributionHashType(), sql); + TDistributionHashType thriftHash = hashType == HashType.IDENTITY + ? TDistributionHashType.IDENTITY : TDistributionHashType.CRC32; + PlanFragment sender = fragments.stream().filter(f -> f.getDestNode() == remote).findFirst().orElseThrow(); + Assertions.assertEquals(TPartitionType.BUCKET_SHFFULE_HASH_PARTITIONED, + sender.getOutputPartition().toThrift().getType(), sql); + Assertions.assertEquals(thriftHash, sender.getOutputPartition().toThrift().getDistributionHashType(), sql); + Assertions.assertFalse(locals.isEmpty(), "must exercise FE local bucket exchange: " + sql); + for (LocalExchangeNode local : locals) { + Assertions.assertEquals(LocalExchangeType.BUCKET_HASH_SHUFFLE, local.getExchangeType(), sql); + Assertions.assertEquals(thriftHash, local.treeToThrift().getNodes().get(0) + .getLocalExchangeNode().getDistributionHashType(), sql); + } + } + + private static void collectSetNodes(PlanNode node, List sets) { + if (node instanceof SetOperationNode) { + sets.add((SetOperationNode) node); + } + if (!(node instanceof ExchangeNode)) { + for (PlanNode child : node.getChildren()) { + collectSetNodes(child, sets); + } + } + } + + private static void collectExchanges(PlanNode node, List remotes, + List locals) { + if (node instanceof ExchangeNode) { + remotes.add((ExchangeNode) node); + return; + } + // PASSTHROUGH wrappers below the bucket exchange do not determine hash placement. + // Keep every hash exchange so an accidental execution-hash re-alignment still fails. + if (node instanceof LocalExchangeNode + && ((LocalExchangeNode) node).getExchangeType().isHashShuffle()) { + locals.add((LocalExchangeNode) node); + } + for (PlanNode child : node.getChildren()) { + collectExchanges(child, remotes, locals); + } + } + + /** + * ADD PARTITION on an identity-distributed table must inherit the table's hash type: + * DDL cannot carry distribution_hash_type, so InternalCatalog.addPartition overwrites the + * new partition's hash type with the table's. If the inheritance were dropped, BE would + * bucket rows in the new partition with one hash function while FE pruned with another, + * making the new partition's rows unreadable through equality pruning. This drives the + * real addPartition path (not the hash-type setter) and asserts the stored metadata. + */ + @Test + public void testAddPartitionInheritsIdentityHashType() throws Exception { + useDatabase("identity_set_operation"); + createTable("CREATE TABLE identity_add_partition (id BIGINT NOT NULL, dt INT NOT NULL) " + + "PARTITION BY RANGE(dt) ( PARTITION p1 values less than (10) ) " + + "DISTRIBUTED BY HASH(id) BUCKETS 5 " + + "PROPERTIES('replication_num'='1', 'distribution_hash_type'='identity')"); + + String addPartitionSql = "ALTER TABLE identity_add_partition ADD PARTITION p2 values less than (20) " + + "DISTRIBUTED BY HASH(id) BUCKETS 5"; + Assertions.assertNotNull(getSqlStmtExecutor(addPartitionSql)); + + OlapTable table = (OlapTable) Env.getCurrentInternalCatalog() + .getDbOrAnalysisException("identity_set_operation") + .getTableOrAnalysisException("identity_add_partition"); + Partition added = table.getPartition("p2"); + Assertions.assertNotNull(added, "ADD PARTITION must create p2"); + Assertions.assertTrue(added.getDistributionInfo() instanceof HashDistributionInfo, + "new partition must keep a hash distribution"); + Assertions.assertEquals(HashType.IDENTITY, + ((HashDistributionInfo) added.getDistributionInfo()).getHashType(), + "ADD PARTITION must inherit the table's identity hash type"); + // the initial partition keeps its type too + Assertions.assertEquals(HashType.IDENTITY, + ((HashDistributionInfo) table.getPartition("p1").getDistributionInfo()).getHashType()); + } +} diff --git a/fe/fe-core/src/test/java/org/apache/doris/service/FrontendServiceImplTest.java b/fe/fe-core/src/test/java/org/apache/doris/service/FrontendServiceImplTest.java index 25f0868b2af913..92d20445fa1e85 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/service/FrontendServiceImplTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/service/FrontendServiceImplTest.java @@ -20,6 +20,7 @@ import org.apache.doris.analysis.UserIdentity; import org.apache.doris.catalog.Database; import org.apache.doris.catalog.Env; +import org.apache.doris.catalog.HashDistributionInfo; import org.apache.doris.catalog.MaterializedIndex; import org.apache.doris.catalog.OlapTable; import org.apache.doris.catalog.Partition; @@ -29,6 +30,8 @@ import org.apache.doris.common.Config; import org.apache.doris.common.ErrorCode; import org.apache.doris.common.FeConstants; +import org.apache.doris.common.FeMetaVersion; +import org.apache.doris.common.io.Text; import org.apache.doris.common.util.DatasourcePrintableMap; import org.apache.doris.datasource.InternalCatalog; import org.apache.doris.mysql.authenticate.TestLogAppender; @@ -36,6 +39,7 @@ import org.apache.doris.nereids.trees.plans.commands.Command; import org.apache.doris.nereids.trees.plans.commands.CreateDatabaseCommand; import org.apache.doris.nereids.trees.plans.logical.LogicalPlan; +import org.apache.doris.persist.gson.GsonUtils; import org.apache.doris.qe.StmtExecutor; import org.apache.doris.tablefunction.BackendsTableValuedFunction; import org.apache.doris.thrift.TBackendsMetadataParams; @@ -48,6 +52,8 @@ import org.apache.doris.thrift.TFetchSchemaTableDataResult; import org.apache.doris.thrift.TGetDbsParams; import org.apache.doris.thrift.TGetDbsResult; +import org.apache.doris.thrift.TGetOlapTableMetaRequest; +import org.apache.doris.thrift.TGetOlapTableMetaResult; import org.apache.doris.thrift.TGetTablesParams; import org.apache.doris.thrift.TGetTablesResult; import org.apache.doris.thrift.TListTableStatusResult; @@ -77,15 +83,20 @@ import com.google.common.collect.Sets; import org.apache.logging.log4j.Level; +import org.apache.thrift.TDeserializer; import org.apache.thrift.TException; +import org.apache.thrift.TSerializer; import org.junit.jupiter.api.Assertions; import org.junit.jupiter.api.Test; import org.mockito.MockedStatic; import org.mockito.Mockito; +import java.io.ByteArrayInputStream; +import java.io.DataInputStream; import java.lang.reflect.Field; import java.lang.reflect.InvocationTargetException; import java.lang.reflect.Method; +import java.nio.ByteBuffer; import java.util.ArrayList; import java.util.Arrays; import java.util.Collections; @@ -129,6 +140,64 @@ private static void setPrivateField(Object target, String fieldName, Object valu field.set(target, value); } + @Test + public void testGetOlapTableMetaDistributionHashCompatibility() throws Exception { + FrontendServiceImpl impl = new FrontendServiceImpl(exeEnv); + for (String layout : Arrays.asList("crc32", "identity", "random")) { + String tableName = "remote_hash_" + layout; + createTable("CREATE TABLE test." + tableName + " (id BIGINT NOT NULL) DUPLICATE KEY(id) " + + "DISTRIBUTED BY " + (layout.equals("random") ? "RANDOM" : "HASH(id)") + + " BUCKETS 8 PROPERTIES('replication_num'='1'" + + (layout.equals("identity") ? ", 'distribution_hash_type'='identity'" : "") + ")"); + // An absent version is an old client too. Check the exact feature boundary as well + // as a newer client, without changing the established CRC32/RANDOM export behavior. + for (Integer version : Arrays.asList(null, FeMetaVersion.VERSION_140, + FeMetaVersion.VERSION_141, FeMetaVersion.VERSION_141 + 1)) { + TGetOlapTableMetaRequest request = new TGetOlapTableMetaRequest(); + request.setDb("test"); + request.setTable(tableName); + request.setTableId(-1L); + request.setUser("root"); + request.setPasswd(""); + if (version != null) { + request.setVersion(version); + } + // Exercise the wire contract too: table_meta is required even on an error response. + TGetOlapTableMetaResult result = new TGetOlapTableMetaResult(); + new TDeserializer().deserialize(result, new TSerializer().serialize(impl.getOlapTableMeta(request))); + String context = "layout=" + layout + ", client version=" + version; + if (layout.equals("identity") && (version == null || version < FeMetaVersion.VERSION_141)) { + Assertions.assertEquals(TStatusCode.ANALYSIS_ERROR, result.getStatus().getStatusCode(), context); + Assertions.assertTrue(result.getStatus().getErrorMsgs().get(0) + .contains("IDENTITY distribution requires client metadata version 141 or newer"), context); + Assertions.assertEquals(0, result.getTableMeta().length, context); + Assertions.assertFalse(result.isSetUpdatedPartitions(), context); + Assertions.assertFalse(result.isSetUpdatedTempPartitions(), context); + } else { + Assertions.assertEquals(TStatusCode.OK, result.getStatus().getStatusCode(), context); + try (DataInputStream in = new DataInputStream(new ByteArrayInputStream(result.getTableMeta()))) { + OlapTable exported = OlapTable.read(in); + if (!layout.equals("random")) { + HashDistributionInfo.HashType expected = layout.equals("identity") + ? HashDistributionInfo.HashType.IDENTITY : HashDistributionInfo.HashType.CRC32; + Assertions.assertEquals(expected, + ((HashDistributionInfo) exported.getDefaultDistributionInfo()).getHashType(), context); + Assertions.assertEquals(1, result.getUpdatedPartitionsSize(), context); + ByteBuffer buffer = result.getUpdatedPartitions().get(0); + try (DataInputStream partitionIn = new DataInputStream(new ByteArrayInputStream( + buffer.array(), buffer.position(), buffer.remaining()))) { + Partition partition = GsonUtils.GSON.fromJson(Text.readString(partitionIn), + Partition.class); + Assertions.assertEquals(expected, + ((HashDistributionInfo) partition.getDistributionInfo()).getHashType(), context); + } + } + } + } + } + } + } + @Test public void testCheckAuthDoesNotLogPassword() throws Exception { FrontendServiceImpl impl = new FrontendServiceImpl(exeEnv); diff --git a/gensrc/thrift/Descriptors.thrift b/gensrc/thrift/Descriptors.thrift index af2068ae72f9e6..513240696093ff 100644 --- a/gensrc/thrift/Descriptors.thrift +++ b/gensrc/thrift/Descriptors.thrift @@ -336,6 +336,8 @@ struct TOlapTablePartitionParam { 13: optional bool partitions_is_fake = false // remote insert fe master address 14: optional Types.TNetworkAddress master_address + // hash function type; CRC32 (legacy behavior) is the default for backward compatibility + 15: optional Types.TDistributionHashType distribution_hash_type = Types.TDistributionHashType.CRC32 } struct TOlapTableIndex { diff --git a/gensrc/thrift/Partitions.thrift b/gensrc/thrift/Partitions.thrift index b14e36a3f628ee..fa23e5d41ac085 100644 --- a/gensrc/thrift/Partitions.thrift +++ b/gensrc/thrift/Partitions.thrift @@ -203,4 +203,6 @@ struct TDataPartition { 2: optional list partition_exprs 3: optional list partition_infos 4: optional TMergePartitionInfo merge_partition_info + // storage bucketing hash for BUCKET_SHFFULE_HASH_PARTITIONED; !__isset means CRC32 (legacy) + 5: optional Types.TDistributionHashType distribution_hash_type = Types.TDistributionHashType.CRC32 } diff --git a/gensrc/thrift/PlanNodes.thrift b/gensrc/thrift/PlanNodes.thrift index 2bb119c149c386..b260f8a657415b 100644 --- a/gensrc/thrift/PlanNodes.thrift +++ b/gensrc/thrift/PlanNodes.thrift @@ -1514,6 +1514,8 @@ struct TLocalExchangeNode { // `TPipelineFragmentParams.total_instances`, and mapping global instance index to local instance by // `TPipelineFragmentParams.shuffle_idx_to_instance_idx` 2: optional list distribute_expr_lists + // storage bucketing hash for BUCKET_HASH_SHUFFLE; !__isset means CRC32 (legacy) + 3: optional Types.TDistributionHashType distribution_hash_type = Types.TDistributionHashType.CRC32 } struct TOlapRewriteNode { @@ -1690,6 +1692,9 @@ struct TRuntimeFilterDesc { // distribution column. BE still verifies that the delivered filter has an // exact IN set before using it for bucket pruning. 22: optional set bucket_pruning_target_ids; + + // Storage hash algorithm for each bucket-pruning target. Missing entries are legacy CRC32. + 23: optional map bucket_pruning_target_hash_types; } diff --git a/gensrc/thrift/Planner.thrift b/gensrc/thrift/Planner.thrift index 866d8d45320243..dd5a9cc9cfb60b 100644 --- a/gensrc/thrift/Planner.thrift +++ b/gensrc/thrift/Planner.thrift @@ -64,6 +64,10 @@ struct TPlanFragment { 8: optional i64 initial_reservation_total_claims 9: optional QueryCache.TQueryCacheParam query_cache_param + + // Effective storage bucketing hash used by BE-native bucket local exchanges. If absent, legacy + // fragments use CRC32. + 10: optional Types.TDistributionHashType distribution_hash_type = Types.TDistributionHashType.CRC32 } // location information for a single scan range diff --git a/gensrc/thrift/Types.thrift b/gensrc/thrift/Types.thrift index da2be22fdc982b..d185c1856fdcc0 100644 --- a/gensrc/thrift/Types.thrift +++ b/gensrc/thrift/Types.thrift @@ -694,6 +694,12 @@ struct TColumnGroup { 2: required list columns_in_group } +// hash function type used by HASH distribution to map rows to buckets. +enum TDistributionHashType { + CRC32 = 0, + IDENTITY = 1 +} + const i32 TSNAPSHOT_REQ_VERSION1 = 3; // corresponding to alpha rowset const i32 TSNAPSHOT_REQ_VERSION2 = 4; // corresponding to beta rowset // the snapshot request should always set prefer snapshot version to TPREFER_SNAPSHOT_REQ_VERSION diff --git a/regression-test/data/ddl_p0/test_distribution_hash_type_identity.out b/regression-test/data/ddl_p0/test_distribution_hash_type_identity.out new file mode 100644 index 00000000000000..e90cd12258412b --- /dev/null +++ b/regression-test/data/ddl_p0/test_distribution_hash_type_identity.out @@ -0,0 +1,416 @@ +-- This file is automatically generated. You should know what you did if you want to edit this +-- !identity_count -- +7 + +-- !identity_eq_0 -- +0 + +-- !identity_eq_1 -- +1 + +-- !identity_eq_7 -- +7 + +-- !identity_eq_8 -- +8 + +-- !identity_eq_513 -- +513 + +-- !identity_eq_negative_1 -- +-1 + +-- !identity_eq_1024 -- +1024 + +-- !identity_in -- +1024 +7 +8 + +-- !identity_string -- +beta 2 + +-- !identity_null -- +9 + +-- !identity_ipv4 -- +4 + +-- !identity_ipv6 -- +1 + +-- !identity_multi -- +-1 A 12 + +-- !identity_typed -- +7 + +-- !identity_matrix_row1 -- +1 + +-- !identity_matrix_row2 -- +2 + +-- !identity_matrix_count -- +2 + +-- !crc32_row_count -- +80 + +-- !identity_row_count -- +80 + +-- !crc32_rows_per_id -- +1 10 +2 10 +3 10 +4 10 +5 10 +6 10 +7 10 +8 10 + +-- !identity_rows_per_id -- +1 10 +2 10 +3 10 +4 10 +5 10 +6 10 +7 10 +8 10 + +-- !identity_added_partition -- +513 + +-- !identity_partition_count -- +4 + +-- !legacy_identity_integer -- +1 +1 +1 +1 +1 +1 +1 +1 +1 +1 + +-- !legacy_identity_string -- +beta 2 + +-- !legacy_identity_multi -- +-1 A 12 + +-- !identity_colocate_join -- +1 +1024 +7 +8 + +-- !mixed_hash_join -- +1 +1024 +7 +8 + +-- !mixed_hash_partitioned_multi_instance -- +0 241 987377 +1 242 987619 +10 241 985690 +11 241 985931 +12 241 986172 +13 241 986413 +14 241 986654 +15 241 986895 +16 241 987136 +2 241 987859 +3 241 988100 +4 242 989365 +5 241 988582 +6 241 988823 +7 241 982927 +8 242 985216 +9 241 985449 + +-- !identity_bucket_shuffle_native -- +-1 5 50 +-8 7 70 +1024 6 60 +513 4 40 +7 2 20 +8 3 30 + +-- !identity_bucket_shuffle_fe -- +-1 5 50 +-8 7 70 +1024 6 60 +513 4 40 +7 2 20 +8 3 30 + +-- !identity_broadcast_then_bucket_native -- +1024 6 60 +7 2 20 +8 3 30 + +-- !identity_broadcast_then_bucket_fe -- +1024 6 60 +7 2 20 +8 3 30 + +-- !identity_multi_bucket_shuffle -- +-1 A 12 22 +1 A 10 20 +2 BC 13 23 + +-- !identity_nullable_bucket_shuffle -- +10 100 +9 90 + +-- !set_identity_left_union_shape -- +PhysicalResultSink +--PhysicalWindow +----PhysicalQuickSort[LOCAL_SORT] +------PhysicalUnion[bucketShuffle] +--------PhysicalOlapScan[test_dist_hash_set_identity] +--------PhysicalDistribute[DistributionSpecHash] +----------PhysicalOlapScan[test_dist_hash_set_other_identity] + +-- !set_identity_left_union_result -- +-1 1 +-1 2 +-8 1 +0 1 +1 1 +1 2 +1 3 +1024 1 +1024 2 +2 1 +4294967297 1 +513 1 +513 2 +7 1 +7 2 +8 1 +8589934593 1 +9 1 + +-- !set_identity_left_intersect_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalIntersect[bucketShuffle] +------PhysicalOlapScan[test_dist_hash_set_identity] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_other_identity] + +-- !set_identity_left_intersect_result -- +-1 1 +1 1 +1024 1 +513 1 +7 1 + +-- !set_identity_left_except_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalExcept[bucketShuffle] +------PhysicalOlapScan[test_dist_hash_set_identity] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_other_identity] + +-- !set_identity_left_except_result -- +-8 1 +0 1 +2 1 +4294967297 1 +8 1 + +-- !set_identity_right_union_shape -- +PhysicalResultSink +--PhysicalWindow +----PhysicalQuickSort[LOCAL_SORT] +------PhysicalUnion[bucketShuffle] +--------PhysicalDistribute[DistributionSpecHash] +----------PhysicalOlapScan[test_dist_hash_set_other_identity] +--------PhysicalOlapScan[test_dist_hash_set_identity] + +-- !set_identity_right_union_result -- +-1 1 +-1 2 +-8 1 +0 1 +1 1 +1 2 +1 3 +1024 1 +1024 2 +2 1 +4294967297 1 +513 1 +513 2 +7 1 +7 2 +8 1 +8589934593 1 +9 1 + +-- !set_identity_right_intersect_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalIntersect[bucketShuffle] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_other_identity] +------PhysicalOlapScan[test_dist_hash_set_identity] + +-- !set_identity_right_intersect_result -- +-1 1 +1 1 +1024 1 +513 1 +7 1 + +-- !set_identity_right_except_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalExcept[bucketShuffle] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_other_identity] +------PhysicalOlapScan[test_dist_hash_set_identity] + +-- !set_identity_right_except_result -- +8589934593 1 +9 1 + +-- !set_mixed_identity_target_union_shape -- +PhysicalResultSink +--PhysicalWindow +----PhysicalQuickSort[LOCAL_SORT] +------PhysicalUnion[bucketShuffle] +--------PhysicalOlapScan[test_dist_hash_set_identity] +--------PhysicalDistribute[DistributionSpecHash] +----------PhysicalOlapScan[test_dist_hash_set_crc] + +-- !set_mixed_identity_target_union_result -- +-1 1 +-1 2 +-8 1 +0 1 +1 1 +1 2 +1 3 +1024 1 +1024 2 +2 1 +4294967297 1 +513 1 +513 2 +7 1 +7 2 +8 1 +8589934593 1 +9 1 + +-- !set_mixed_identity_target_intersect_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalIntersect[bucketShuffle] +------PhysicalOlapScan[test_dist_hash_set_identity] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_crc] + +-- !set_mixed_identity_target_intersect_result -- +-1 1 +1 1 +1024 1 +513 1 +7 1 + +-- !set_mixed_identity_target_except_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalExcept[bucketShuffle] +------PhysicalOlapScan[test_dist_hash_set_identity] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_crc] + +-- !set_mixed_identity_target_except_result -- +-8 1 +0 1 +2 1 +4294967297 1 +8 1 + +-- !set_mixed_crc_target_union_shape -- +PhysicalResultSink +--PhysicalWindow +----PhysicalQuickSort[LOCAL_SORT] +------PhysicalUnion[bucketShuffle] +--------PhysicalDistribute[DistributionSpecHash] +----------PhysicalOlapScan[test_dist_hash_set_identity] +--------PhysicalOlapScan[test_dist_hash_set_crc] + +-- !set_mixed_crc_target_union_result -- +-1 1 +-1 2 +-8 1 +0 1 +1 1 +1 2 +1 3 +1024 1 +1024 2 +2 1 +4294967297 1 +513 1 +513 2 +7 1 +7 2 +8 1 +8589934593 1 +9 1 + +-- !set_mixed_crc_target_intersect_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalIntersect[bucketShuffle] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_identity] +------PhysicalOlapScan[test_dist_hash_set_crc] + +-- !set_mixed_crc_target_intersect_result -- +-1 1 +1 1 +1024 1 +513 1 +7 1 + +-- !set_mixed_crc_target_except_shape -- +PhysicalResultSink +--PhysicalProject +----PhysicalExcept[bucketShuffle] +------PhysicalDistribute[DistributionSpecHash] +--------PhysicalOlapScan[test_dist_hash_set_identity] +------PhysicalOlapScan[test_dist_hash_set_crc] + +-- !set_mixed_crc_target_except_result -- +-8 1 +0 1 +2 1 +4294967297 1 +8 1 + +-- !identity_nlj_then_bucket_native -- +1 1 10 +7 7 70 +8 8 80 + +-- !identity_nlj_then_bucket_fe -- +1 1 10 +7 7 70 +8 8 80 + diff --git a/regression-test/data/nereids_p0/test_identity_bucket_prune_cache.out b/regression-test/data/nereids_p0/test_identity_bucket_prune_cache.out new file mode 100644 index 00000000000000..8c8da42884715f --- /dev/null +++ b/regression-test/data/nereids_p0/test_identity_bucket_prune_cache.out @@ -0,0 +1,29 @@ +-- This file is automatically generated. You should know what you did if you want to edit this +-- !full_without_prune -- +0 257 32639 +1 257 32639 +2 257 32639 +3 257 32639 +4 257 32639 + +-- !sparse_without_prune -- +0 3 127 +1 3 127 +2 3 127 +3 3 127 +4 3 127 + +-- !full_with_prune -- +0 257 32639 +1 257 32639 +2 257 32639 +3 257 32639 +4 257 32639 + +-- !sparse_with_prune -- +0 3 127 +1 3 127 +2 3 127 +3 3 127 +4 3 127 + diff --git a/regression-test/suites/check_hash_bucket_table/check_hash_bucket_table.groovy b/regression-test/suites/check_hash_bucket_table/check_hash_bucket_table.groovy index 3fe6713f66b5ee..963d2e9cabf9cc 100644 --- a/regression-test/suites/check_hash_bucket_table/check_hash_bucket_table.groovy +++ b/regression-test/suites/check_hash_bucket_table/check_hash_bucket_table.groovy @@ -30,7 +30,7 @@ suite("check_hash_bucket_table") { def excludedDbs = ["mysql", "information_schema", "__internal_schema"].toSet() logger.info("===== [check] begin to check hash bucket tables") - def checkPartition = { String db, String tblName, def info -> + def checkPartition = { String db, String tblName, def info, String hashType -> int bucketNum = info["Buckets"].toInteger() if (bucketNum <= 1) { return false} @@ -40,17 +40,24 @@ suite("check_hash_bucket_table") { def bucketCols = bucketColumns.split(",").collect { it.trim() } def bucketColsStr = bucketCols.collect { "`${it}`" }.join(",") def partitionName = info["PartitionName"] + // The per-tablet bucket-layout check mirrors the storage router: every row in one tablet + // must hash to one single bucket index. CRC32 tables use crc32_internal(cols) % num; + // IDENTITY tables use identity_hash_internal(cols, num), which applies the same + // canonical-bytes composition as VOlapTablePartitionParam with the bucket count built in. + def bucketHashExpr = hashType == "identity" + ? "identity_hash_internal(${bucketColsStr}, ${bucketNum})" + : "crc32_internal(${bucketColsStr}) % ${bucketNum}" try { def tabletIdList = sql_return_maparray(""" show replica status from `${tblName}` partition(`${partitionName}`); """).collect { it.TabletId }.toList() def tabletIds = tabletIdList.toSet() int replicaNum = tabletIdList.stream().filter { it == tabletIdList[0] }.count() - logger.info("""===== [check] Begin to check partition: ${db}.${tblName}, partition name: ${partitionName}, bucket num: ${bucketNum}, replica num: ${replicaNum}, bucket columns: ${bucketColsStr}""") + logger.info("""===== [check] Begin to check partition: ${db}.${tblName}, partition name: ${partitionName}, bucket num: ${bucketNum}, hash type: ${hashType}, replica num: ${replicaNum}, bucket columns: ${bucketColsStr}""") (0..replicaNum-1).each { replica -> sql "set use_fix_replica=${replica};" tabletIds.each { it2 -> def tabletId = it2 try { - def res = sql "select crc32_internal(${bucketColsStr}) % ${bucketNum} from `${db}`.`${tblName}` tablet(${tabletId}) group by crc32_internal(${bucketColsStr}) % ${bucketNum};" + def res = sql "select ${bucketHashExpr} from `${db}`.`${tblName}` tablet(${tabletId}) group by ${bucketHashExpr};" if (res.size() > 1) { logger.info("""===== [check] check failed: ${db}.${tblName}, partition name: ${partitionName}, tabletId: ${tabletId}, bucket columns: ${bucketColsStr}, res.size()=${res.size()}, res=${res}""") assert res.size() == 1 @@ -75,10 +82,21 @@ suite("check_hash_bucket_table") { def checkTable = { String db, String tblName -> sql "use `${db}`;" def showStmt = sql_return_maparray("show create table `${tblName}`")[0]["Create Table"] + // Pick the bucket algorithm from the declared distribution hash type instead of skipping + // the whole table: IDENTITY tables are checked with identity_hash_internal, everything + // else keeps the crc32_internal layout check. + String hashType = "crc32" + if (showStmt =~ /"distribution_hash_type"\s*=\s*"(\w+)"/) { + hashType = (showStmt =~ /"distribution_hash_type"\s*=\s*"(\w+)"/)[0][1].toLowerCase() + } + if (hashType != "crc32" && hashType != "identity") { + logger.info("===== [check] Skip unsupported hash table: ${db}.${tblName}, hash type: ${hashType}") + return false + } def partitionInfo = sql_return_maparray """ show partitions from `${tblName}`; """ int checkedPartition = 0 partitionInfo.each { - if (checkPartition(db, tblName, it)) { + if (checkPartition(db, tblName, it, hashType)) { ++checkedPartition } } diff --git a/regression-test/suites/ddl_p0/test_distribution_hash_type_identity.groovy b/regression-test/suites/ddl_p0/test_distribution_hash_type_identity.groovy new file mode 100644 index 00000000000000..022641e7684412 --- /dev/null +++ b/regression-test/suites/ddl_p0/test_distribution_hash_type_identity.groovy @@ -0,0 +1,841 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +import org.apache.doris.regression.suite.client.FrontendClientImpl +import org.apache.doris.thrift.TGetOlapTableMetaRequest +import org.apache.doris.thrift.TNetworkAddress +import org.apache.doris.thrift.TStatusCode + +suite("test_distribution_hash_type_identity") { + + // --------------------------------------------------------------------- + // 1. DDL: create table with distribution_hash_type = identity + // --------------------------------------------------------------------- + sql "DROP TABLE IF EXISTS test_dist_hash_identity" + sql """ + CREATE TABLE `test_dist_hash_identity` ( + `id` BIGINT NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + // SHOW CREATE TABLE round-trip: the property must be echoed back so the table can be rebuilt. + def createStmt = sql "SHOW CREATE TABLE test_dist_hash_identity" + assertTrue(createStmt[0][1].toString().toLowerCase() + .contains("\"distribution_hash_type\" = \"identity\"")) + + // default (property absent) is crc32: SHOW CREATE must NOT emit the property. + sql "DROP TABLE IF EXISTS test_dist_hash_default" + sql """ + CREATE TABLE `test_dist_hash_default` ( + `id` BIGINT NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1" + ); + """ + def defaultStmt = sql "SHOW CREATE TABLE test_dist_hash_default" + assertFalse(defaultStmt[0][1].toString().toLowerCase().contains("distribution_hash_type")) + + // Metadata version 141 introduced IDENTITY. Old remote-Doris FEs (including clients + // omitting the version) would ignore hashType and prune these physical buckets with CRC32. + // Exercise the actual RPC; CRC32 metadata must remain readable by those same clients. + def metadataClient = new FrontendClientImpl(new TNetworkAddress(getMasterIp(), getMasterPort("rpc"))) + try { + [null, 140, 141].each { version -> + ["test_dist_hash_default", "test_dist_hash_identity"].each { table -> + def request = new TGetOlapTableMetaRequest() + request.setDb(context.dbName) + request.setTable(table) + request.setTableId(-1L) + request.setUser(context.config.jdbcUser) + request.setPasswd(context.config.jdbcPassword) + if (version != null) { + request.setVersion(version) + } + def response = metadataClient.client.getOlapTableMeta(request) + if (table == "test_dist_hash_identity" && (version == null || version < 141)) { + assertEquals(TStatusCode.ANALYSIS_ERROR, response.status.statusCode) + assertTrue(response.status.errorMsgs[0].contains( + "IDENTITY distribution requires client metadata version 141 or newer")) + assertEquals(0, response.getTableMeta().length) + assertFalse(response.isSetUpdatedPartitions()) + } else { + assertEquals(TStatusCode.OK, response.status.statusCode) + assertTrue(response.isSetTableMeta()) + assertTrue(response.isSetUpdatedPartitions()) + } + } + } + } finally { + metadataClient.close() + } + + // --------------------------------------------------------------------- + // 2. identity accepts multiple distribution columns and all valid types + // --------------------------------------------------------------------- + sql "DROP TABLE IF EXISTS test_dist_hash_string" + sql """ + CREATE TABLE `test_dist_hash_string` ( + `name` VARCHAR(32) NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`name`) + DISTRIBUTED BY HASH(`name`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + sql "DROP TABLE IF EXISTS test_dist_hash_nullable" + sql """ + CREATE TABLE `test_dist_hash_nullable` ( + `name` VARCHAR(32) NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`name`) + DISTRIBUTED BY HASH(`name`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + sql "DROP TABLE IF EXISTS test_dist_hash_ipv4" + sql """ + CREATE TABLE `test_dist_hash_ipv4` ( + `addr` IPV4 NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`addr`) + DISTRIBUTED BY HASH(`addr`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + sql "DROP TABLE IF EXISTS test_dist_hash_ipv6" + sql """ + CREATE TABLE `test_dist_hash_ipv6` ( + `addr` IPV6 NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`addr`) + DISTRIBUTED BY HASH(`addr`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + sql "DROP TABLE IF EXISTS test_dist_hash_multi_col" + sql """ + CREATE TABLE `test_dist_hash_multi_col` ( + `id` INT NOT NULL, + `name` VARCHAR(32) NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`, `name`) + DISTRIBUTED BY HASH(`id`, `name`) BUCKETS 10 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + sql "DROP TABLE IF EXISTS test_dist_hash_typed_multi" + sql """ + CREATE TABLE `test_dist_hash_typed_multi` ( + `d` DATE NOT NULL, + `dt` DATETIMEV2(6) NOT NULL, + `amount` DECIMAL(18, 2) NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`d`, `dt`) + DISTRIBUTED BY HASH(`d`, `dt`, `amount`) BUCKETS 10 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + + // invalid hash type value rejected + sql "DROP TABLE IF EXISTS test_dist_hash_bad_value" + test { + sql """ + CREATE TABLE `test_dist_hash_bad_value` ( + `id` BIGINT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "murmur" + ); + """ + exception "Invalid distribution_hash_type" + } + + // --------------------------------------------------------------------- + // 3. colocate: same distribution_hash_type may share a group; different hash types may not. + // A colocate group keeps every table on its storage layout with no reshuffle, so all + // members must bucket rows with the same hash function. + // --------------------------------------------------------------------- + // 3a. two identity tables in the same colocate group -> allowed. + sql "DROP TABLE IF EXISTS test_dist_hash_colo_id1" + sql "DROP TABLE IF EXISTS test_dist_hash_colo_id2" + sql """ + CREATE TABLE `test_dist_hash_colo_id1` ( + `id` BIGINT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity", + "colocate_with" = "test_dist_hash_cg_identity" + ); + """ + sql """ + CREATE TABLE `test_dist_hash_colo_id2` ( + `id` BIGINT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity", + "colocate_with" = "test_dist_hash_cg_identity" + ); + """ + + // 3b. crc32 table joining an existing identity group -> rejected on hash type mismatch. + sql "DROP TABLE IF EXISTS test_dist_hash_colo_crc32" + test { + sql """ + CREATE TABLE `test_dist_hash_colo_crc32` ( + `id` BIGINT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "colocate_with" = "test_dist_hash_cg_identity" + ); + """ + exception "Colocate tables must have same distribution hash type" + } + + // --------------------------------------------------------------------- + // 4. read/write consistency: identity write then equality query must find the row. + // This is the core guarantee: BE buckets and FE prunes with the same hash function. + // Use explicit assertions (not qt_ recording) so a lost row fails loudly instead of + // silently recording an empty result set. + // --------------------------------------------------------------------- + sql """ INSERT INTO test_dist_hash_identity VALUES + (0, 100), (1, 101), (7, 107), (8, 108), (513, 613), (-1, 200), (1024, 300) """ + + qt_identity_count "SELECT COUNT(*) FROM test_dist_hash_identity" + + // Each equality query drives single-bucket pruning. + qt_identity_eq_0 "SELECT id FROM test_dist_hash_identity WHERE id = 0" + qt_identity_eq_1 "SELECT id FROM test_dist_hash_identity WHERE id = 1" + qt_identity_eq_7 "SELECT id FROM test_dist_hash_identity WHERE id = 7" + qt_identity_eq_8 "SELECT id FROM test_dist_hash_identity WHERE id = 8" + qt_identity_eq_513 "SELECT id FROM test_dist_hash_identity WHERE id = 513" + qt_identity_eq_negative_1 "SELECT id FROM test_dist_hash_identity WHERE id = -1" + qt_identity_eq_1024 "SELECT id FROM test_dist_hash_identity WHERE id = 1024" + + order_qt_identity_in "SELECT id FROM test_dist_hash_identity WHERE id IN (7, 8, 1024)" + + // Non-integer and multi-column identity layouts must use the same canonical bytes in BE + // writes, FE tablet pruning, and bucket shuffle. + sql "INSERT INTO test_dist_hash_string VALUES ('alpha', 1), ('beta', 2)" + qt_identity_string "SELECT name, v FROM test_dist_hash_string WHERE name = 'beta'" + + sql "INSERT INTO test_dist_hash_nullable VALUES (NULL, 9), ('x', 10)" + qt_identity_null "SELECT v FROM test_dist_hash_nullable WHERE name <=> NULL" + + sql "INSERT INTO test_dist_hash_ipv4 VALUES (to_ipv4('1.2.3.4'), 4), (to_ipv4('10.0.0.1'), 10)" + qt_identity_ipv4 "SELECT v FROM test_dist_hash_ipv4 WHERE addr = to_ipv4('1.2.3.4')" + + sql "INSERT INTO test_dist_hash_ipv6 VALUES (to_ipv6('::1'), 1), (to_ipv6('2001:db8::1'), 6)" + qt_identity_ipv6 "SELECT v FROM test_dist_hash_ipv6 WHERE addr = to_ipv6('::1')" + + sql """ INSERT INTO test_dist_hash_multi_col VALUES + (1, 'A', 10), (1, 'B', 11), (-1, 'A', 12), (2, 'BC', 13) """ + order_qt_identity_multi """ + SELECT id, name, v FROM test_dist_hash_multi_col + WHERE id = -1 AND name = 'A' + """ + + sql """ INSERT INTO test_dist_hash_typed_multi VALUES + ('2026-01-02', '2026-01-02 03:04:05.123456', 123.45, 7) """ + qt_identity_typed """ + SELECT v FROM test_dist_hash_typed_multi + WHERE d = '2026-01-02' + AND dt = '2026-01-02 03:04:05.123456' + AND amount = 123.45 + """ + + // Full type-matrix end-to-end coverage: every distribution-column type the FE/BE pair + // claims to support, written through the storage router and read back through equality + // pruning. A hash mismatch between FE pruning and BE routing drops the row, so each row + // below must come back from its equality query. Width-sensitive encodings are keyed to + // expose truncation: LARGEINT/DECIMAL256 use non-zero high bytes, CHAR pads, the legacy + // DATE/DATETIME pair exercises the string encoding rather than DATEV2/DATETIMEV2 binary, + // and TIMESTAMPTZ exercises its 8-byte encoding. TIME columns cannot be OLAP table columns + // and VARBINARY columns need an external catalog, so their encodings stay covered by the + // BE unit oracle (identity_partitioner_test.cpp). DECIMALV2 is out too: the column type + // is disabled by default (Config.disable_decimalv2). + sql "set enable_decimal256 = true" + sql "DROP TABLE IF EXISTS test_dist_hash_type_matrix" + sql """ + CREATE TABLE `test_dist_hash_type_matrix` ( + `b` BOOLEAN NOT NULL, + `ti` TINYINT NOT NULL, + `si` SMALLINT NOT NULL, + `li` LARGEINT NOT NULL, + `c` CHAR(8) NOT NULL, + `ld` DATE NOT NULL, + `ldt` DATETIME NOT NULL, + `tz` TIMESTAMPTZ(3) NOT NULL, + `d32` DECIMAL(9, 5) NOT NULL, + `d64` DECIMAL(18, 9) NOT NULL, + `d256` DECIMALV3(76, 40) NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`b`, `ti`, `si`, `li`) + DISTRIBUTED BY HASH(`b`, `ti`, `si`, `li`, `c`, `ld`, `ldt`, `tz`, `d32`, `d64`, `d256`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + sql """ INSERT INTO test_dist_hash_type_matrix VALUES + (true, -128, -32768, 170141183460469231731687303715884105727, 'mat', + '2026-01-02', '2026-01-02 03:04:05', '2026-01-02 03:04:05.123', + 1234.56789, 123456789.123456789, + 12345678901234567890123456789.123456789012345678901234567890123456789, 1), + (false, 127, 32767, -170141183460469231731687303715884105728, 'pad', + '2026-06-30', '2026-06-30 23:59:59', '2026-06-30 23:59:59.999', + -9999.99999, -999999999.999999999, + -12345678901234567890123456789.123456789012345678901234567890123456789, 2) """ + + qt_identity_matrix_row1 """ + SELECT v FROM test_dist_hash_type_matrix + WHERE b = true AND ti = -128 AND si = -32768 + AND li = 170141183460469231731687303715884105727 AND c = 'mat' + AND ld = '2026-01-02' AND ldt = '2026-01-02 03:04:05' AND tz = '2026-01-02 03:04:05.123' + AND d32 = 1234.56789 AND d64 = 123456789.123456789 + AND d256 = 12345678901234567890123456789.123456789012345678901234567890123456789 + """ + qt_identity_matrix_row2 """ + SELECT v FROM test_dist_hash_type_matrix + WHERE b = false AND ti = 127 AND si = 32767 + AND li = -170141183460469231731687303715884105728 AND c = 'pad' + AND ld = '2026-06-30' AND ldt = '2026-06-30 23:59:59' AND tz = '2026-06-30 23:59:59.999' + AND d32 = -9999.99999 AND d64 = -999999999.999999999 + AND d256 = -12345678901234567890123456789.123456789012345678901234567890123456789 + """ + qt_identity_matrix_count "SELECT COUNT(*) FROM test_dist_hash_type_matrix" + sql "set enable_decimal256 = false" + + // --------------------------------------------------------------------- + // 5. bucket data distribution: identity spreads rows evenly, crc32 does not. + // Insert ids 1..8 (10 rows each, 80 rows total) into a crc32 table and an identity + // table, both DISTRIBUTED BY HASH(id) BUCKETS 8. With this key set: + // - crc32(id)%8 folds ids 3 and 8 onto the same bucket and leaves one bucket empty, + // so the row distribution is skewed (one 20-row bucket, one 0-row bucket). + // - identity uses id%8 directly, mapping the 8 distinct ids onto 8 distinct buckets, + // so every bucket holds exactly 10 rows and none is empty. + // crc32(id)%8 for id=1..8 -> {1:7, 2:5, 3:3, 4:0, 5:6, 6:4, 7:2, 8:3}; + // bucket 1 receives no id (empty) while bucket 3 gets both 3 and 8. + // (verify with: select crc32(8)%8; -> same bucket as crc32(3)%8) + // id%8 for id=1..8 -> {1:1, 2:2, 3:3, 4:4, 5:5, 6:6, 7:7, 8:0}: 8 buckets, 10 rows each. + // --------------------------------------------------------------------- + // helper: read the per-bucket RowCount via SHOW TABLETS. Each tablet maps to one bucket and + // (single replica here) appears once, so the list of RowCounts is the per-bucket row spread. + // RowCount is reported asynchronously, so poll until the total matches the expected row count + // before trusting the layout. + def bucketRowCounts = { String tbl, int expectedTotal -> + def counts = null + for (int attempt = 0; attempt < 60; attempt++) { + def tablets = sql_return_maparray "SHOW TABLETS FROM ${tbl}" + def perBucket = tablets.collect { (it["RowCount"] as String) as long } + long total = perBucket.sum() as long + if (total == expectedTotal) { + counts = perBucket + break + } + sleep(5000) + } + assertNotNull(counts, "RowCount for ${tbl} never reached ${expectedTotal}".toString()) + return counts + } + + // truncate existing data + // identity table: even distribution, one row per bucket per id. + sql "TRUNCATE TABLE test_dist_hash_identity" + // crc32 (default) table: skewed distribution with an empty bucket. + sql "TRUNCATE TABLE test_dist_hash_default" + + // write ids 1..8, 10 rows each (v = 1..10) -> 80 rows total for both tables. + def bucketValues = [] + (1..8).each { id -> + (1..10).each { v -> bucketValues << "(${id}, ${v})" } + } + def bucketInsert = bucketValues.join(", ") + sql "INSERT INTO test_dist_hash_default VALUES ${bucketInsert}" + sql "INSERT INTO test_dist_hash_identity VALUES ${bucketInsert}" + + // sanity: both tables received all 80 rows with 10 rows per id (no rows dropped on write). + qt_crc32_row_count "SELECT COUNT(*) FROM test_dist_hash_default" + qt_identity_row_count "SELECT COUNT(*) FROM test_dist_hash_identity" + order_qt_crc32_rows_per_id "SELECT id, COUNT(*) FROM test_dist_hash_default GROUP BY id" + order_qt_identity_rows_per_id "SELECT id, COUNT(*) FROM test_dist_hash_identity GROUP BY id" + + // crc32: at least one bucket is empty and at least one bucket is overloaded (>10 rows), + // because crc32(id)%8 collides ids 3 and 8 and skips one bucket for ids 1..8. + def crc32Counts = bucketRowCounts("test_dist_hash_default", 80) + assertTrue(crc32Counts.any { it == 0L }, + "crc32 must leave at least one empty bucket, counts=${crc32Counts}".toString()) + assertTrue(crc32Counts.any { it > 10L }, + "crc32 must overload at least one bucket (>10), counts=${crc32Counts}".toString()) + + // identity: every bucket holds exactly 10 rows -> no empty bucket, perfectly even spread. + def identityCounts = bucketRowCounts("test_dist_hash_identity", 80) + assertEquals(8, identityCounts.size(), + "identity should fill all 8 buckets, counts=${identityCounts}".toString()) + assertFalse(identityCounts.any { it == 0L }, + "identity must NOT leave any empty bucket, counts=${identityCounts}".toString()) + identityCounts.each { c -> + assertEquals(10L, c as long, + "identity bucket must hold exactly 10 rows, counts=${identityCounts}".toString()) + } + + // --------------------------------------------------------------------- + // 6. ADD PARTITION inherits the table hash type (commit: inherit on ADD PARTITION). + // A partitioned identity table; manually added partitions must keep identity so + // writes/reads stay consistent. + // --------------------------------------------------------------------- + sql "DROP TABLE IF EXISTS test_dist_hash_identity_part" + sql """ + CREATE TABLE `test_dist_hash_identity_part` ( + `id` BIGINT NOT NULL, + `dt` INT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`, `dt`) + PARTITION BY RANGE(`dt`) ( + PARTITION p1 VALUES LESS THAN ("10") + ) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + // manual ADD PARTITION: DDL cannot carry distribution_hash_type, so it must be inherited. + sql """ ALTER TABLE test_dist_hash_identity_part ADD PARTITION p2 VALUES LESS THAN ("20") + DISTRIBUTED BY HASH(`id`) BUCKETS 8 """ + + sql """ INSERT INTO test_dist_hash_identity_part VALUES (5, 5), (513, 5), (5, 15), (513, 15) """ + // rows in the newly added partition p2 (dt=15) must be found by equality pruning too; + // if the new partition fell back to crc32, BE/FE hash mismatch would drop these rows. + qt_identity_added_partition "SELECT id FROM test_dist_hash_identity_part WHERE dt = 15 AND id = 513" + qt_identity_partition_count "SELECT COUNT(*) FROM test_dist_hash_identity_part" + + // Legacy planner must pass the table hash type to HashDistributionPruner as well. + sql "set enable_nereids_planner=false" + sql "set enable_fallback_to_original_planner=false" + qt_legacy_identity_integer "SELECT id FROM test_dist_hash_identity WHERE id = 1" + qt_legacy_identity_string "SELECT name, v FROM test_dist_hash_string WHERE name = 'beta'" + qt_legacy_identity_multi """ + SELECT id, name, v FROM test_dist_hash_multi_col WHERE id = -1 AND name = 'A' + """ + sql "set enable_nereids_planner=true" + + // --------------------------------------------------------------------- + // 7. colocate join: two identity tables in the same colocate group join with no reshuffle. + // Both sides keep their storage layout (same identity hash + same bucket count), so the + // plan must be a COLOCATE join and the result must match the non-optimized join. + // --------------------------------------------------------------------- + sql "set enable_nereids_planner=true" + sql "set disable_colocate_plan=false" + + waitForColocateGroupStable("test_dist_hash_cg_identity") + + sql "INSERT INTO test_dist_hash_colo_id1 VALUES (0), (1), (7), (8), (513), (-1), (1024)" + sql "INSERT INTO test_dist_hash_colo_id2 VALUES (1), (7), (8), (999), (1024)" + + explain { + sql("""SELECT a.id FROM test_dist_hash_colo_id1 a + JOIN test_dist_hash_colo_id2 b ON a.id = b.id""") + contains "HAS_COLO_PLAN_NODE: true" + } + + order_qt_identity_colocate_join """SELECT a.id FROM test_dist_hash_colo_id1 a + JOIN test_dist_hash_colo_id2 b ON a.id = b.id""" + + // a crc32 table joining an identity table must NOT colocate (different hash functions). + sql "DROP TABLE IF EXISTS test_dist_hash_join_crc32" + sql """ + CREATE TABLE `test_dist_hash_join_crc32` ( + `id` BIGINT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1" + ); + """ + sql "INSERT INTO test_dist_hash_join_crc32 VALUES (1), (7), (8), (1024)" + // Force both storage layouts through ordinary execution shuffle. Their source-table hash + // labels must be normalized because HASH_PARTITIONED uses the execution hash algorithm. + sql "set enable_bucket_shuffle_join = false" + sql "set parallel_pipeline_task_num = 4" + explain { + sql("""SELECT a.id FROM test_dist_hash_colo_id1 a + JOIN [shuffle] test_dist_hash_join_crc32 b ON a.id = b.id""") + contains "HAS_COLO_PLAN_NODE: false" + contains "INNER JOIN(PARTITIONED)" + } + order_qt_mixed_hash_join """SELECT a.id FROM test_dist_hash_colo_id1 a + JOIN [shuffle] test_dist_hash_join_crc32 b ON a.id = b.id""" + // Thousands of distinct keys feed all ordinary shuffle destinations, not just channel zero. + sql "INSERT INTO test_dist_hash_colo_id1 SELECT number + 2048 FROM numbers('number'='4096')" + sql "INSERT INTO test_dist_hash_join_crc32 SELECT number + 2048 FROM numbers('number'='4096')" + def mixedHashMultiInstanceSql = """ + SELECT a.id % 17 AS g, count(*), sum(a.id) + FROM test_dist_hash_colo_id1 a + JOIN [shuffle] test_dist_hash_join_crc32 b ON a.id = b.id + GROUP BY g + """ + explain { + sql(mixedHashMultiInstanceSql) + contains "INNER JOIN(PARTITIONED)" + contains "HAS_COLO_PLAN_NODE: false" + } + order_qt_mixed_hash_partitioned_multi_instance "${mixedHashMultiInstanceSql}" + sql "set enable_bucket_shuffle_join = true" + + // --------------------------------------------------------------------- + // 8. bucket-shuffle join: an identity table joins a table with a different bucket count. + // The optimizer keeps the identity side on its storage layout and reshuffles the other + // side to that layout. The reshuffle must use the identity hash on BE (not crc32), + // otherwise rows land on the wrong channel and the join result is wrong. + // --------------------------------------------------------------------- + // Force multiple local destinations so CRC32 and IDENTITY cannot both degenerate to channel 0. + sql "set parallel_pipeline_task_num = 4" + sql "set enable_nereids_planner=true" + sql "set enable_bucket_shuffle_join = true" + // Keep bucket shuffle deterministic across clusters: a positive downgrade ratio may replace it + // with a full PARTITIONED shuffle based on the bucket and parallel-instance counts. + sql "set bucket_shuffle_downgrade_ratio = 0" + + // [shuffle] prevents these tiny test tables from choosing a broadcast join. Together with the + // settings above, it exercises bucket shuffle without depending on table statistics. + def bucketShuffleJoinSql = """ + SELECT l.id, l.v, r.w FROM test_dist_hash_bs_left l + JOIN [shuffle] test_dist_hash_bs_right r ON l.id = r.id + """ + + // With this switch off, BE adds the required local exchange while building pipelines. + sql "set enable_local_shuffle_planner = false" + + sql "DROP TABLE IF EXISTS test_dist_hash_bs_left" + sql "DROP TABLE IF EXISTS test_dist_hash_bs_right" + sql """ + CREATE TABLE `test_dist_hash_bs_left` ( + `id` BIGINT NOT NULL, + `v` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + sql """ + CREATE TABLE `test_dist_hash_bs_right` ( + `id` BIGINT NOT NULL, + `w` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 5 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + // include negatives, out-of-range and boundary keys to exercise unsigned binary identity + // reshuffle across channels. + sql """INSERT INTO test_dist_hash_bs_left VALUES + (0, 1), (7, 2), (8, 3), (513, 4), (-1, 5), (1024, 6), (-8, 7)""" + sql """INSERT INTO test_dist_hash_bs_right VALUES + (7, 20), (8, 30), (513, 40), (-1, 50), (1024, 60), (-8, 70), (99, 80)""" + + // Standard EXPLAIN does not expose local-exchange placement, but it must retain the same + // bucket-shuffle join in both planning modes. Query results then validate the BE-native path. + explain { + sql(bucketShuffleJoinSql) + contains "INNER JOIN(BUCKET_SHUFFLE)" + } + order_qt_identity_bucket_shuffle_native "${bucketShuffleJoinSql}" + + // With this switch on, FE inserts explicit local-exchange nodes into the distributed plan. + sql "set enable_local_shuffle_planner = true" + explain { + sql(bucketShuffleJoinSql) + contains "INNER JOIN(BUCKET_SHUFFLE)" + } + order_qt_identity_bucket_shuffle_fe "${bucketShuffleJoinSql}" + sql "set enable_local_shuffle_planner = false" + + // Multi-column mixed-type identity bucket shuffle follows the same composition as storage. + // A broadcast join keeps its probe-side IDENTITY bucket layout. Its CRC32 build side must not + // erase that metadata before the result feeds another bucket-shuffle join. + def broadcastThenBucketSql = """ + SELECT p.id, p.v, r.w + FROM ( + SELECT a.id, a.v + FROM test_dist_hash_bs_left a + JOIN [broadcast] test_dist_hash_join_crc32 b ON a.id = b.id + ) p + JOIN [shuffle] test_dist_hash_bs_right r ON p.id = r.id + """ + explain { + sql(broadcastThenBucketSql) + contains "INNER JOIN(BROADCAST)" + contains "INNER JOIN(BUCKET_SHUFFLE)" + } + order_qt_identity_broadcast_then_bucket_native "${broadcastThenBucketSql}" + sql "set enable_local_shuffle_planner = true" + order_qt_identity_broadcast_then_bucket_fe "${broadcastThenBucketSql}" + sql "set enable_local_shuffle_planner = false" + + sql "DROP TABLE IF EXISTS test_dist_hash_bs_multi_right" + sql """ + CREATE TABLE `test_dist_hash_bs_multi_right` ( + `id` INT NOT NULL, + `name` VARCHAR(32) NOT NULL, + `w` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`id`, `name`) + DISTRIBUTED BY HASH(`id`, `name`) BUCKETS 7 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + sql """ INSERT INTO test_dist_hash_bs_multi_right VALUES + (1, 'A', 20), (-1, 'A', 22), (2, 'BC', 23), (9, 'missing', 24) """ + + explain { + sql("""SELECT l.id, l.name FROM test_dist_hash_multi_col l + JOIN [shuffle] test_dist_hash_bs_multi_right r + ON l.id = r.id AND l.name = r.name""") + contains "INNER JOIN(BUCKET_SHUFFLE)" + } + + order_qt_identity_multi_bucket_shuffle """SELECT l.id, l.name, l.v, r.w + FROM test_dist_hash_multi_col l + JOIN [shuffle] test_dist_hash_bs_multi_right r + ON l.id = r.id AND l.name = r.name""" + + sql "DROP TABLE IF EXISTS test_dist_hash_nullable_right" + sql """ + CREATE TABLE `test_dist_hash_nullable_right` ( + `name` VARCHAR(32) NULL, + `w` INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(`name`) + DISTRIBUTED BY HASH(`name`) BUCKETS 7 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ); + """ + sql "INSERT INTO test_dist_hash_nullable_right VALUES (NULL, 90), ('x', 100)" + explain { + sql("""SELECT l.v, r.w FROM test_dist_hash_nullable l + JOIN [shuffle] test_dist_hash_nullable_right r ON l.name <=> r.name""") + contains "INNER JOIN(BUCKET_SHUFFLE)" + } + order_qt_identity_nullable_bucket_shuffle """SELECT l.v, r.w FROM test_dist_hash_nullable l + JOIN [shuffle] test_dist_hash_nullable_right r + ON l.name <=> r.name""" + + // Set bucket shuffle requires BOTH FE local shuffle and Nereids distribute planning. + // The window is a hash consumer of the set output; asserting only a parent join's + // BUCKET_SHUFFLE would also pass if the whole set were re-bucketed after execution shuffle. + sql "set enable_local_shuffle_planner = true" + sql "set enable_nereids_distribute_planner = true" + sql "set enable_local_shuffle = true" + // Otherwise INTERSECT puts its smaller input first, hiding the left-basic arm. + sql "set disable_nereids_rules = 'REORDER_INTERSECT'" + // Keep every key reaching the exchanges; runtime-filter pruning is not the path under test. + sql "set runtime_filter_mode = 'OFF'" + sql "DROP TABLE IF EXISTS test_dist_hash_set_identity" + sql "DROP TABLE IF EXISTS test_dist_hash_set_other_identity" + sql "DROP TABLE IF EXISTS test_dist_hash_set_crc" + sql """CREATE TABLE test_dist_hash_set_identity(id BIGINT NOT NULL) + DISTRIBUTED BY HASH(id) BUCKETS 8 + PROPERTIES('replication_num'='1', 'distribution_hash_type'='identity')""" + sql """CREATE TABLE test_dist_hash_set_other_identity(id BIGINT NOT NULL) + DISTRIBUTED BY HASH(id) BUCKETS 5 + PROPERTIES('replication_num'='1', 'distribution_hash_type'='identity')""" + sql """CREATE TABLE test_dist_hash_set_crc(id BIGINT NOT NULL) + DISTRIBUTED BY HASH(id) BUCKETS 7 PROPERTIES('replication_num'='1')""" + sql """INSERT INTO test_dist_hash_set_identity VALUES + (-8), (-1), (0), (1), (2), (7), (8), (513), (1024), (4294967297)""" + sql """INSERT INTO test_dist_hash_set_other_identity VALUES + (-1), (1), (1), (7), (9), (513), (1024), (8589934593)""" + sql """INSERT INTO test_dist_hash_set_crc SELECT * FROM test_dist_hash_set_other_identity""" + + // Inject stats only to select the intended basic child, as in bucket_shuffle_set_operation. + // Check left and right basics, and both directions of mixed-hash target selection. + ["identity_left", "identity_right", "mixed_identity_target", "mixed_crc_target"].each { variant -> + def left = variant == "identity_right" ? "test_dist_hash_set_other_identity" + : "test_dist_hash_set_identity" + def right = variant == "identity_left" ? "test_dist_hash_set_other_identity" + : variant == "identity_right" ? "test_dist_hash_set_identity" : "test_dist_hash_set_crc" + def basic = variant == "mixed_crc_target" ? right : "test_dist_hash_set_identity" + [left, right].each { table -> + def rows = table == basic ? 10000 : 100 + // Also keep the estimated NDV large: otherwise pre-deduplication of INTERSECT / + // EXCEPT can shrink the intended basic below its sibling and reverse the target. + sql """ALTER TABLE ${table} MODIFY COLUMN id SET STATS + ('row_count'='${rows}', 'ndv'='${rows}', 'min_value'='-8', 'max_value'='8589934593')""" + } + ["union": "UNION ALL", "intersect": "INTERSECT", "except": "EXCEPT"].each { kind, op -> + def query = """SELECT id, row_number() OVER (PARTITION BY id ORDER BY id) rn + FROM (SELECT id FROM ${left} ${op} SELECT id FROM ${right}) u""" + // Golden subtree shows the set operator itself is bucketShuffle, its basic scan + // stays direct and its other child is PhysicalDistribute. Exact hash properties + // and remote/local thrift fields are checked by IdentitySetOperationTest in FE UT. + explain { + sql "shape plan " + query + check { String plan -> + def setNode = "Physical${kind.capitalize()}[bucketShuffle]" + assertTrue(plan.contains(setNode)) + def lines = plan.readLines() + def depth = { String line -> (line =~ /^-*/)[0].length() } + def parentOf = { int index -> + lines.take(index).reverse().find { depth(it) < depth(lines[index]) } ?: "" + } + int basicScan = lines.findIndexOf { it.contains("PhysicalOlapScan[${basic}]") } + assertTrue(basicScan >= 0) + assertTrue(parentOf(basicScan).contains(setNode), + "${basic} must stay the direct storage basic, not the exchanged side") + int exchange = lines.findIndexOf { it.contains("PhysicalDistribute[DistributionSpecHash]") } + assertTrue(exchange >= 0) + assertTrue(parentOf(exchange).contains(setNode), + "the set child, not the whole set output, must be bucket-shuffled") + } + } + quickTest("set_${variant}_${kind}_shape", "explain shape plan " + query) + quickTest("set_${variant}_${kind}_result", query, true) + } + } + + sql "set disable_nereids_rules = ''" + sql "set runtime_filter_mode = 'GLOBAL'" + + // --------------------------------------------------------------------- + // 9. Nested-loop join preserves the probe-side IDENTITY bucket layout. + // The CRC32 build side is broadcast and does not repartition probe rows. Feed the NLJ output + // into an IDENTITY bucket-shuffle join and verify both BE-native and FE-planned local exchange + // paths keep matching rows in the destination buckets. + // --------------------------------------------------------------------- + sql "DROP TABLE IF EXISTS test_identity_nlj_probe" + sql "DROP TABLE IF EXISTS test_identity_nlj_build" + sql "DROP TABLE IF EXISTS test_identity_nlj_bucket" + + sql """ + CREATE TABLE test_identity_nlj_probe ( + id BIGINT NOT NULL, + v INT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(id) + DISTRIBUTED BY HASH(id) BUCKETS 8 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ) + """ + sql """ + CREATE TABLE test_identity_nlj_build ( + threshold INT NOT NULL + ) ENGINE=OLAP + DUPLICATE KEY(threshold) + DISTRIBUTED BY HASH(threshold) BUCKETS 3 + PROPERTIES ("replication_allocation" = "tag.location.default: 1") + """ + sql """ + CREATE TABLE test_identity_nlj_bucket ( + id BIGINT NOT NULL, + w INT NULL + ) ENGINE=OLAP + DUPLICATE KEY(id) + DISTRIBUTED BY HASH(id) BUCKETS 7 + PROPERTIES ( + "replication_allocation" = "tag.location.default: 1", + "distribution_hash_type" = "identity" + ) + """ + + sql "INSERT INTO test_identity_nlj_probe VALUES (1, 1), (7, 7), (8, 8), (99, 99)" + sql "INSERT INTO test_identity_nlj_build VALUES (10)" + sql "INSERT INTO test_identity_nlj_bucket VALUES (1, 10), (7, 70), (8, 80), (100, 1000)" + + def nljThenBucketSql = """ + SELECT p.id, p.v, r.w + FROM ( + SELECT a.id, a.v + FROM test_identity_nlj_probe a + JOIN [broadcast] test_identity_nlj_build b ON a.v < b.threshold + ) p + JOIN [shuffle] test_identity_nlj_bucket r ON p.id = r.id + """ + + sql "set enable_nereids_planner = true" + sql "set enable_bucket_shuffle_join = true" + sql "set bucket_shuffle_downgrade_ratio = 0" + explain { + sql(nljThenBucketSql) + contains "NESTED LOOP JOIN" + contains "INNER JOIN(BUCKET_SHUFFLE)" + } + + sql "set enable_local_shuffle_planner = false" + order_qt_identity_nlj_then_bucket_native "${nljThenBucketSql}" + sql "set enable_local_shuffle_planner = true" + order_qt_identity_nlj_then_bucket_fe "${nljThenBucketSql}" +} diff --git a/regression-test/suites/nereids_p0/test_identity_bucket_prune_cache.groovy b/regression-test/suites/nereids_p0/test_identity_bucket_prune_cache.groovy new file mode 100644 index 00000000000000..312d56726f94e6 --- /dev/null +++ b/regression-test/suites/nereids_p0/test_identity_bucket_prune_cache.groovy @@ -0,0 +1,93 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +import org.apache.doris.regression.action.ProfileAction + +suite("test_identity_bucket_prune_cache") { + sql "DROP TABLE IF EXISTS test_identity_bucket_cache_probe" + sql "DROP TABLE IF EXISTS test_identity_bucket_cache_build" + sql """ + CREATE TABLE test_identity_bucket_cache_probe ( + id INT NULL, + p INT NOT NULL, + v INT NOT NULL + ) DUPLICATE KEY(id, p) + PARTITION BY RANGE(p) (PARTITION p0 VALUES LESS THAN ("1")) + DISTRIBUTED BY HASH(id) BUCKETS 1 + PROPERTIES ("replication_num" = "1", "distribution_hash_type" = "identity") + """ + // One scan sees several bucket counts. Each count needs its own membership set, + // but must not retain one cached bucket entry per runtime-filter value. + [3, 7, 16, 97].eachWithIndex { buckets, index -> + sql """ALTER TABLE test_identity_bucket_cache_probe + ADD PARTITION p${index + 1} VALUES LESS THAN ("${index + 2}") + DISTRIBUTED BY HASH(id) BUCKETS ${buckets}""" + } + sql """ + CREATE TABLE test_identity_bucket_cache_build (id INT NULL) + DUPLICATE KEY(id) + DISTRIBUTED BY HASH(id) BUCKETS 3 + PROPERTIES ("replication_num" = "1") + """ + sql """INSERT INTO test_identity_bucket_cache_probe + SELECT CAST(number % 256 AS INT), CAST(number DIV 256 AS INT), CAST(number % 256 AS INT) + FROM numbers("number" = "1280")""" + sql """INSERT INTO test_identity_bucket_cache_probe VALUES + (NULL, 0, -1), (NULL, 1, -1), (NULL, 2, -1), (NULL, 3, -1), (NULL, 4, -1)""" + sql """INSERT INTO test_identity_bucket_cache_build + SELECT CAST(number AS INT) FROM numbers("number" = "4096")""" + sql "INSERT INTO test_identity_bucket_cache_build VALUES (NULL)" + + // Keep the IDENTITY table on the probe side and retain the RF even when its large IN set + // covers every bucket. Waiting makes these queries exercise the cache before scanning. + sql "set disable_join_reorder = true" + sql "set enable_runtime_filter_prune = false" + sql "set runtime_filter_wait_infinitely = true" + sql "set runtime_filter_max_in_num = 40960" + sql "set runtime_filter_type = 1" + def fullCoverage = """SELECT p.p, count(*), sum(p.v) + FROM test_identity_bucket_cache_probe p + JOIN [shuffle] test_identity_bucket_cache_build b ON p.id <=> b.id + GROUP BY p.p ORDER BY p.p""" + def sparseCoverage = """SELECT p.p, count(*), sum(p.v) + FROM test_identity_bucket_cache_probe p + JOIN [shuffle] test_identity_bucket_cache_build b ON p.id <=> b.id + WHERE b.id % 128 = 0 OR b.id IS NULL + GROUP BY p.p ORDER BY p.p""" + + sql "set enable_runtime_filter_bucket_prune = false" + qt_full_without_prune fullCoverage + qt_sparse_without_prune sparseCoverage + sql "set enable_runtime_filter_bucket_prune = true" + qt_full_with_prune fullCoverage + qt_sparse_with_prune sparseCoverage + + // Result equality alone would also pass if no RF reached the probe. Check the existing + // profile counter on a uniquely tagged sparse query, as in rf_bucket_pruning. + sql "set enable_profile = true" + sql "set profile_level = 2" + def token = UUID.randomUUID().toString() + sql """SELECT "${token}", count(*) + FROM test_identity_bucket_cache_probe p + JOIN [shuffle] test_identity_bucket_cache_build b ON p.id <=> b.id + WHERE b.id % 128 = 0 OR b.id IS NULL""" + def profile = new ProfileAction(context).getProfileBySql(token, ["BucketsPrunedByRuntimeFilter"]) + def pruned = (profile =~ /-\s*BucketsPrunedByRuntimeFilter:\s*(\d+)/) + .collect { it[1].toLong() } + assertTrue(!pruned.isEmpty() && pruned.sum() > 0, + "The sparse query must exercise IDENTITY bucket pruning") +}