diff --git a/be/src/exec/operator/olap_scan_operator.cpp b/be/src/exec/operator/olap_scan_operator.cpp index 051a8b5348ad5a..3f6e62084cf596 100644 --- a/be/src/exec/operator/olap_scan_operator.cpp +++ b/be/src/exec/operator/olap_scan_operator.cpp @@ -217,6 +217,8 @@ Status OlapScanLocalState::_init_profile() { _lazy_read_seek_timer = ADD_TIMER(_segment_profile, "LazyReadSeekTime"); _lazy_read_seek_counter = ADD_COUNTER(_segment_profile, "LazyReadSeekCount", TUnit::UNIT); + _lazy_read_pruned_timer = ADD_TIMER(_segment_profile, "LazyReadPrunedTime"); + _output_col_timer = ADD_TIMER(_segment_profile, "OutputColumnTime"); _stats_filtered_counter = ADD_COUNTER(_segment_profile, "RowsStatsFiltered", TUnit::UNIT); diff --git a/be/src/exec/operator/olap_scan_operator.h b/be/src/exec/operator/olap_scan_operator.h index 57c4912b777f6b..aab56a084736ba 100644 --- a/be/src/exec/operator/olap_scan_operator.h +++ b/be/src/exec/operator/olap_scan_operator.h @@ -210,6 +210,7 @@ class OlapScanLocalState final : public ScanLocalState { RuntimeProfile::Counter* _lazy_read_timer = nullptr; RuntimeProfile::Counter* _lazy_read_seek_timer = nullptr; RuntimeProfile::Counter* _lazy_read_seek_counter = nullptr; + RuntimeProfile::Counter* _lazy_read_pruned_timer = nullptr; // total pages read // used by segment v2 diff --git a/be/src/exec/scan/olap_scanner.cpp b/be/src/exec/scan/olap_scanner.cpp index 04ede793ed579c..ef879b96582f62 100644 --- a/be/src/exec/scan/olap_scanner.cpp +++ b/be/src/exec/scan/olap_scanner.cpp @@ -762,6 +762,7 @@ void OlapScanner::_collect_profile_before_close() { COUNTER_UPDATE(local_state->_predicate_column_read_seek_counter, stats.predicate_column_read_seek_num); COUNTER_UPDATE(local_state->_lazy_read_timer, stats.lazy_read_ns); + COUNTER_UPDATE(local_state->_lazy_read_pruned_timer, stats.lazy_read_pruned_ns); COUNTER_UPDATE(local_state->_lazy_read_seek_timer, stats.block_lazy_read_seek_ns); COUNTER_UPDATE(local_state->_lazy_read_seek_counter, stats.block_lazy_read_seek_num); COUNTER_UPDATE(local_state->_output_col_timer, stats.output_col_ns); diff --git a/be/src/runtime/descriptors.cpp b/be/src/runtime/descriptors.cpp index edd1f8b99d3cc2..91756d5cec595b 100644 --- a/be/src/runtime/descriptors.cpp +++ b/be/src/runtime/descriptors.cpp @@ -105,6 +105,9 @@ SlotDescriptor::SlotDescriptor(const PSlotDescriptor& pdesc) auto convert_to_thrift_column_access_path = [](const PColumnAccessPath& pb_path) { TColumnAccessPath thrift_path; thrift_path.type = (TAccessPathType::type)pb_path.type(); + if (pb_path.has_version()) { + thrift_path.__set_version(pb_path.version()); + } if (pb_path.has_data_access_path()) { thrift_path.__isset.data_access_path = true; for (int i = 0; i < pb_path.data_access_path().path_size(); ++i) { @@ -161,6 +164,9 @@ void SlotDescriptor::to_protobuf(PSlotDescriptor* pslot) const { doris::PColumnAccessPath* pb_path) { pb_path->Clear(); pb_path->set_type((PAccessPathType)thrift_path.type); // 使用 reinterpret_cast 进行类型转换 + if (thrift_path.__isset.version) { + pb_path->set_version(thrift_path.version); + } if (thrift_path.__isset.data_access_path) { auto* pb_data = pb_path->mutable_data_access_path(); pb_data->Clear(); diff --git a/be/src/runtime/runtime_state.h b/be/src/runtime/runtime_state.h index 3184b74a7445c7..f264851b37715a 100644 --- a/be/src/runtime/runtime_state.h +++ b/be/src/runtime/runtime_state.h @@ -642,6 +642,11 @@ class RuntimeState { _query_options.enable_aggregate_function_null_v2; } + bool enable_prune_nested_column() const { + return _query_options.__isset.enable_prune_nested_column && + _query_options.enable_prune_nested_column; + } + bool is_read_csv_empty_line_as_null() const { return _query_options.__isset.read_csv_empty_line_as_null && _query_options.read_csv_empty_line_as_null; diff --git a/be/src/storage/olap_common.h b/be/src/storage/olap_common.h index ee200a23c88b79..4b76905d345839 100644 --- a/be/src/storage/olap_common.h +++ b/be/src/storage/olap_common.h @@ -335,6 +335,7 @@ struct OlapReaderStatistics { int64_t lazy_read_ns = 0; int64_t block_lazy_read_seek_num = 0; int64_t block_lazy_read_seek_ns = 0; + int64_t lazy_read_pruned_ns = 0; int64_t raw_rows_read = 0; diff --git a/be/src/storage/segment/column_reader.cpp b/be/src/storage/segment/column_reader.cpp index 7e2ee007f47474..696cb5d3694d8f 100644 --- a/be/src/storage/segment/column_reader.cpp +++ b/be/src/storage/segment/column_reader.cpp @@ -18,14 +18,17 @@ #include "storage/segment/column_reader.h" #include +#include #include #include #include #include #include +#include #include #include +#include #include #include "common/compiler_util.h" // IWYU pragma: keep @@ -90,6 +93,261 @@ inline bool read_as_string(PrimitiveType type) { type == PrimitiveType::TYPE_BITMAP || type == PrimitiveType::TYPE_FIXED_LENGTH_OBJECT; } +bool is_meta_access_path_component(const std::string& component) { + return StringCaseEqual()(component, ColumnIterator::ACCESS_OFFSET) || + StringCaseEqual()(component, ColumnIterator::ACCESS_NULL); +} + +bool uses_legacy_access_path_encoding(const TColumnAccessPath& path) { + return !path.__isset.version || + path.version == g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY; +} + +namespace { + +// Nested access paths are processed one container level at a time: +// +// 1. Each Map/Array/Struct iterator's set_access_paths() calls _prepare_nested_access_paths(). For +// the all-path and predicate-path channels independently, _split_access_paths() validates the +// wire encoding and type-selected payload, removes the current iterator name, and separates +// requests consumed by the current iterator from paths that still address a data descendant. A +// current DATA request requires all data children. Struct owns current-level NULL metadata; +// Map and Array own NULL and OFFSET metadata. A supported metadata-only request can stop before +// descendant routing and mark every data child SKIP. +// 2. This router interprets only the first remaining component and routes the path according to +// the container topology: +// - Struct components already name fields. Select the paths for each field without rewriting. +// - Array `*` names its only item. Retarget `*` to the item iterator name. +// - Map `KEYS` and `VALUES` directly name its logical children. Map `*` creates a complete DATA +// path for `KEYS` and routes any trailing components through `VALUES`. +// 3. The container forwards the routed all-path and predicate-path channels to each selected +// child's set_access_paths(), then finalizes that child's PREDICATE/LAZY_OUTPUT/SKIP +// requirement. The child repeats the same flow, which handles arbitrary Map/Array/Struct +// nesting. +// +// At this point versions have been validated, legacy paths have DATA type, and every payload +// selected by type is non-empty. Routing never interprets or rewrites an unselected compatibility +// payload. +class DescendantAccessPathRouter final { +public: + DescendantAccessPathRouter() = delete; + + // Preserve the two set_access_paths() input channels. all_paths originates from the + // all-access-path superset, while predicate_paths separately records predicate-phase paths. + // Routing may omit all_paths when the parent already requires complete child data. + struct ChildAccessPaths { + TColumnAccessPaths all_paths; + TColumnAccessPaths predicate_paths; + + bool empty() const { return all_paths.empty() && predicate_paths.empty(); } + }; + + struct MapChildAccessPaths { + ChildAccessPaths key; + ChildAccessPaths value; + }; + + // Map children use the logical KEYS/VALUES selectors as their access-path names, independent + // of physical child column names. Expand a wildcard to complete keys and route its trailing + // qualifiers to values. + static Result route_map_paths_to_children( + TColumnAccessPaths all_paths, TColumnAccessPaths predicate_paths) { + MapChildAccessPaths child_paths; + auto status = distribute_map_paths(std::move(all_paths), child_paths.key.all_paths, + child_paths.value.all_paths); + if (!status.ok()) { + return ResultError(std::move(status)); + } + status = distribute_map_paths(std::move(predicate_paths), child_paths.key.predicate_paths, + child_paths.value.predicate_paths); + if (!status.ok()) { + return ResultError(std::move(status)); + } + return child_paths; + } + + // Array has one data child. Retarget its logical wildcard selector to the item iterator name. + static Result route_array_paths_to_item(TColumnAccessPaths all_paths, + TColumnAccessPaths predicate_paths, + const std::string& item_name) { + ChildAccessPaths child_paths {.all_paths = std::move(all_paths), + .predicate_paths = std::move(predicate_paths)}; + auto retarget_wildcard_paths_to_item = [&](TColumnAccessPaths& paths) -> Status { + for (auto& path : paths) { + const bool is_wildcard = DORIS_TRY(selected_payload_head_matches( + path, ColumnIterator::ACCESS_ALL, PathHeadMatchMode::EXACT)); + if (is_wildcard) { + RETURN_IF_ERROR(replace_selected_payload_head(path, item_name)); + } + } + return Status::OK(); + }; + + auto status = retarget_wildcard_paths_to_item(child_paths.all_paths); + if (!status.ok()) { + return ResultError(std::move(status)); + } + status = retarget_wildcard_paths_to_item(child_paths.predicate_paths); + if (!status.ok()) { + return ResultError(std::move(status)); + } + return child_paths; + } + + // Struct selectors already use child field names. Select the paths for one child without + // rewriting them, so that the child can validate and strip its own name. + static Result select_struct_paths_for_child( + const TColumnAccessPaths& all_paths, const TColumnAccessPaths& predicate_paths, + const std::string& child_name, bool include_all_paths) { + ChildAccessPaths child_paths; + auto select_matching_paths = [&](const TColumnAccessPaths& source_paths, + TColumnAccessPaths& child_paths) -> Status { + for (const auto& path : source_paths) { + const bool matches_child = DORIS_TRY(selected_payload_head_matches( + path, child_name, PathHeadMatchMode::CASE_INSENSITIVE)); + if (matches_child) { + child_paths.emplace_back(path); + } + } + return Status::OK(); + }; + + if (include_all_paths) { + auto status = select_matching_paths(all_paths, child_paths.all_paths); + if (!status.ok()) { + return ResultError(std::move(status)); + } + } + auto status = select_matching_paths(predicate_paths, child_paths.predicate_paths); + if (!status.ok()) { + return ResultError(std::move(status)); + } + return child_paths; + } + +private: + enum class PathHeadMatchMode { EXACT, CASE_INSENSITIVE }; + enum class MapSelector { WILDCARD, KEYS, VALUES }; + + // TColumnAccessPath may carry both payload fields after compatibility forwarding. Selected + // means data_access_path for DATA and meta_access_path for META. Visit only that payload and + // leave the other payload untouched. + template + static Status visit_selected_payload(AccessPath& access_path, Visitor&& visitor) { + switch (access_path.type) { + case TAccessPathType::DATA: + if (!access_path.__isset.data_access_path) { + return Status::InternalError( + "Invalid DATA access path: data_access_path payload is not set"); + } + return std::forward(visitor)(access_path.data_access_path); + case TAccessPathType::META: + if (!access_path.__isset.meta_access_path) { + return Status::InternalError( + "Invalid META access path: meta_access_path payload is not set"); + } + return std::forward(visitor)(access_path.meta_access_path); + default: + return Status::InternalError("Invalid access path type: {}", + static_cast(access_path.type)); + } + } + + static Result selected_payload_head_matches(const TColumnAccessPath& access_path, + const std::string& expected_head, + PathHeadMatchMode match_mode) { + bool matches = false; + auto status = visit_selected_payload(access_path, [&](const auto& payload) { + DORIS_CHECK(!payload.path.empty()); + matches = match_mode == PathHeadMatchMode::EXACT + ? payload.path.front() == expected_head + : StringCaseEqual()(payload.path.front(), expected_head); + return Status::OK(); + }); + if (!status.ok()) { + return ResultError(std::move(status)); + } + return matches; + } + + static Status replace_selected_payload_head(TColumnAccessPath& access_path, + const std::string& child_name) { + return visit_selected_payload(access_path, [&](auto& payload) { + DORIS_CHECK(!payload.path.empty()); + payload.path.front() = child_name; + return Status::OK(); + }); + } + + // Map `*` applies trailing qualifiers only to values, while locating entries still requires + // complete KEYS as DATA. Construct a fresh key path because the source may be META, and + // preserve its version so child routing keeps the same legacy/typed encoding. + static TColumnAccessPath make_full_data_path_for_map_keys( + const TColumnAccessPath& version_source) { + TColumnAccessPath child_path; + child_path.__set_type(TAccessPathType::DATA); + TDataAccessPath data_path; + data_path.__set_path({ColumnIterator::ACCESS_MAP_KEYS}); + child_path.__set_data_access_path(data_path); + if (version_source.__isset.version) { + child_path.__set_version(version_source.version); + } + return child_path; + } + + static Result classify_map_selector(const TColumnAccessPath& access_path) { + std::optional selector; + auto status = visit_selected_payload(access_path, [&](const auto& payload) { + DORIS_CHECK(!payload.path.empty()); + const auto& head = payload.path.front(); + if (head == ColumnIterator::ACCESS_ALL) { + selector = MapSelector::WILDCARD; + } else if (head == ColumnIterator::ACCESS_MAP_KEYS) { + selector = MapSelector::KEYS; + } else if (head == ColumnIterator::ACCESS_MAP_VALUES) { + selector = MapSelector::VALUES; + } else { + return Status::InternalError( + "Invalid map access path selector '{}': expected '*', 'KEYS', or 'VALUES'", + head); + } + return Status::OK(); + }); + if (!status.ok()) { + return ResultError(std::move(status)); + } + DORIS_CHECK(selector.has_value()); + return *selector; + } + + static Status distribute_map_paths(TColumnAccessPaths source_paths, + TColumnAccessPaths& key_paths, + TColumnAccessPaths& value_paths) { + for (auto& path : source_paths) { + const auto selector = DORIS_TRY(classify_map_selector(path)); + switch (selector) { + case MapSelector::WILDCARD: + // A wildcard needs complete keys for runtime lookup, while any remaining + // qualifiers apply only to the value. The value keeps its type and version. + key_paths.emplace_back(make_full_data_path_for_map_keys(path)); + RETURN_IF_ERROR( + replace_selected_payload_head(path, ColumnIterator::ACCESS_MAP_VALUES)); + value_paths.emplace_back(std::move(path)); + break; + case MapSelector::KEYS: + key_paths.emplace_back(std::move(path)); + break; + case MapSelector::VALUES: + value_paths.emplace_back(std::move(path)); + break; + } + } + return Status::OK(); + } +}; + +} // namespace + Status ColumnReader::create_array(const ColumnReaderOptions& opts, const ColumnMetaPB& meta, const io::FileReaderSPtr& file_reader, std::shared_ptr* reader) { @@ -829,13 +1087,11 @@ Status ColumnReader::new_map_iterator(ColumnIteratorUPtr* iterator, &key_iterator, tablet_column && tablet_column->get_subtype_count() > 1 ? &tablet_column->get_sub_column(0) : nullptr)); - key_iterator->set_column_name(tablet_column ? tablet_column->get_sub_column(0).name() : ""); ColumnIteratorUPtr val_iterator; RETURN_IF_ERROR(_sub_readers[1]->new_iterator( &val_iterator, tablet_column && tablet_column->get_subtype_count() > 1 ? &tablet_column->get_sub_column(1) : nullptr)); - val_iterator->set_column_name(tablet_column ? tablet_column->get_sub_column(1).name() : ""); ColumnIteratorUPtr offsets_iterator; RETURN_IF_ERROR(_sub_readers[2]->new_iterator(&offsets_iterator, nullptr)); auto* file_iter = static_cast(offsets_iterator.release()); @@ -886,31 +1142,144 @@ Status ColumnReader::new_struct_iterator(ColumnIteratorUPtr* iterator, return Status::OK(); } -Result ColumnIterator::_get_sub_access_paths( - const TColumnAccessPaths& access_paths) { - TColumnAccessPaths sub_access_paths = access_paths; - for (auto it = sub_access_paths.begin(); it != sub_access_paths.end();) { - TColumnAccessPath& name_path = *it; - if (name_path.data_access_path.path.empty()) { +void ColumnIterator::_convert_to_place_holder_column(MutableColumnPtr& dst, size_t count) { + if (_read_phase == ReadPhase::LAZY) { + return; + } else if (_read_requirement == ReadRequirement::LAZY_OUTPUT && + _read_phase == ReadPhase::PREDICATE) { + // This branch is for non-predicate columns that still have to appear in the + // predicate-phase block so row filtering can keep all block columns aligned. + // Columns already marked PREDICATE are read normally, and SKIP/NORMAL + // columns do not participate in lazy materialization. + _has_place_holder_column = true; + } + + dst->insert_many_defaults(count); +} + +void ColumnIterator::_recovery_from_place_holder_column(MutableColumnPtr& dst) { + if (_read_phase == ReadPhase::LAZY && _has_place_holder_column) { + dst->clear(); + _has_place_holder_column = false; + } +} + +Result ColumnIterator::_split_access_paths( + TColumnAccessPaths access_paths) const { + AccessPathSplit split; + for (auto& path : access_paths) { + const bool uses_legacy_encoding = uses_legacy_access_path_encoding(path); + if (!uses_legacy_encoding && + path.version != g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED) { return ResultError( - Status::InternalError("Invalid access path for struct column: path is empty")); + Status::InternalError("Unsupported access path version: {}", path.version)); } - if (!StringCaseEqual()(name_path.data_access_path.path[0], _column_name)) { + std::vector* components = nullptr; + if (uses_legacy_encoding) { + if (path.type != TAccessPathType::DATA) { + return ResultError(Status::InternalError("Invalid legacy access path type: {}", + static_cast(path.type))); + } + if (!path.__isset.data_access_path) { + return ResultError(Status::InternalError( + "Invalid legacy access path: data_access_path payload is not set")); + } + components = &path.data_access_path.path; + } else { + switch (path.type) { + case TAccessPathType::DATA: + if (!path.__isset.data_access_path) { + return ResultError(Status::InternalError( + "Invalid DATA access path: data_access_path payload is not set")); + } + components = &path.data_access_path.path; + break; + case TAccessPathType::META: + if (!path.__isset.meta_access_path) { + return ResultError(Status::InternalError( + "Invalid META access path: meta_access_path payload is not set")); + } + components = &path.meta_access_path.path; + break; + default: + return ResultError(Status::InternalError("Invalid access path type: {}", + static_cast(path.type))); + } + } + + if (components->empty()) { + return ResultError(Status::InternalError( + "Invalid access path for column '{}': path is empty", _column_name)); + } + + if (!StringCaseEqual()((*components)[0], _column_name)) { return ResultError(Status::InternalError( R"(Invalid access path for column: expected name "{}", got "{}")", _column_name, - name_path.data_access_path.path[0])); + (*components)[0])); } - name_path.data_access_path.path.erase(name_path.data_access_path.path.begin()); - if (!name_path.data_access_path.path.empty()) { - ++it; - } else { - set_need_to_read(); - it = sub_access_paths.erase(it); + components->erase(components->begin()); + if (components->empty()) { + split.reads_current_data = true; + continue; + } + + const bool is_current_level_meta = + components->size() == 1 && is_meta_access_path_component((*components)[0]) && + (path.type == TAccessPathType::META || uses_legacy_encoding); + if (is_current_level_meta) { + if (StringCaseEqual()((*components)[0], ACCESS_OFFSET)) { + split.current_meta_mode = MetaReadMode::OFFSET_ONLY; + } else if (split.current_meta_mode == MetaReadMode::DEFAULT) { + split.current_meta_mode = MetaReadMode::NULL_MAP_ONLY; + } + continue; + } + + split.descendant_paths.emplace_back(std::move(path)); + } + return split; +} + +Result ColumnIterator::_prepare_nested_access_paths( + const TColumnAccessPaths& all_access_paths, + const TColumnAccessPaths& predicate_access_paths, NestedMetaSupport meta_support, + const std::function& set_all_data_descendants_read_requirement) { + const auto requirement_before = _read_requirement; + if (!predicate_access_paths.empty()) { + set_read_requirement_self(ReadRequirement::PREDICATE); + } + + NestedAccessPathPlan plan; + auto all_split = _split_access_paths(all_access_paths); + if (!all_split.has_value()) { + return ResultError(std::move(all_split).error()); + } + plan.all = std::move(all_split).value(); + + auto predicate_split = _split_access_paths(predicate_access_paths); + if (!predicate_split.has_value()) { + return ResultError(std::move(predicate_split).error()); + } + plan.predicate = std::move(predicate_split).value(); + if (plan.all.reads_current_data) { + set_lazy_output_requirement(); + } + if (plan.predicate.reads_current_data) { + set_read_requirement(ReadRequirement::PREDICATE); + } + + if (!plan.predicate.has_descendant_paths()) { + RETURN_IF_ERROR_RESULT(_check_and_set_meta_read_mode(requirement_before, plan.all)); + plan.skip_data_descendants = + read_null_map_only() || + (meta_support == NestedMetaSupport::NULL_MAP_AND_OFFSET && read_offset_only()); + if (plan.skip_data_descendants) { + set_all_data_descendants_read_requirement(ReadRequirement::SKIP); } } - return sub_access_paths; + return plan; } ///====================== MapFileColumnIterator ============================//// @@ -926,10 +1295,14 @@ MapFileColumnIterator::MapFileColumnIterator(std::shared_ptr reade if (_map_reader->is_nullable()) { _null_iterator = std::move(null_iterator); } + // Access paths identify map children by logical selectors rather than storage/schema child + // names. These names are consumed by each child's _split_access_paths(). + _key_iterator->set_column_name(ACCESS_MAP_KEYS); + _val_iterator->set_column_name(ACCESS_MAP_VALUES); } Status MapFileColumnIterator::init(const ColumnIteratorOptions& opts) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (_read_requirement == ReadRequirement::SKIP) { DLOG(INFO) << "Map column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -943,7 +1316,7 @@ Status MapFileColumnIterator::init(const ColumnIteratorOptions& opts) { } Status MapFileColumnIterator::seek_to_ordinal(ordinal_t ord) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Map column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -961,7 +1334,7 @@ Status MapFileColumnIterator::seek_to_ordinal(ordinal_t ord) { } RETURN_IF_ERROR(_offsets_iterator->seek_to_ordinal(ord)); if (read_offset_only()) { - // In OFFSET_ONLY mode, key/value iterators are SKIP_READING, no need to seek them + // In OFFSET_ONLY mode, key/value iterators are SKIP, no need to seek them return Status::OK(); } // here to use offset info @@ -985,23 +1358,37 @@ Status MapFileColumnIterator::init_prefetcher(const SegmentPrefetchParams& param void MapFileColumnIterator::collect_prefetchers( std::map>& prefetchers, PrefetcherInitMethod init_method) { - _offsets_iterator->collect_prefetchers(prefetchers, init_method); + if (!need_to_read()) { + return; + } + if (!read_null_map_only()) { + _offsets_iterator->collect_prefetchers(prefetchers, init_method); + } if (_map_reader->is_nullable()) { _null_iterator->collect_prefetchers(prefetchers, init_method); } + if (read_offset_only() || read_null_map_only()) { + return; + } // the actual data pages to read of key/value column depends on the read result of offset column, // so we can't init prefetch blocks according to rowids, just prefetch all data blocks here. - _key_iterator->collect_prefetchers(prefetchers, PrefetcherInitMethod::ALL_DATA_BLOCKS); - _val_iterator->collect_prefetchers(prefetchers, PrefetcherInitMethod::ALL_DATA_BLOCKS); + if (_key_iterator->need_to_read()) { + _key_iterator->collect_prefetchers(prefetchers, PrefetcherInitMethod::ALL_DATA_BLOCKS); + } + if (_val_iterator->need_to_read()) { + _val_iterator->collect_prefetchers(prefetchers, PrefetcherInitMethod::ALL_DATA_BLOCKS); + } } Status MapFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Map column iterator column " << _column_name << " skip reading."; - dst->resize(dst->size() + *n); + _convert_to_place_holder_column(dst, *n); return Status::OK(); } + _recovery_from_place_holder_column(dst); + if (read_null_map_only()) { // NULL_MAP_ONLY mode: read null map, fill nested ColumnMap with empty defaults DORIS_CHECK(dst->is_nullable()); @@ -1033,14 +1420,35 @@ Status MapFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* } auto& column_map = assert_cast( - dst->is_nullable() ? static_cast(*dst).get_nested_column() : *dst); - auto column_offsets_ptr = IColumn::mutate(std::move(column_map.get_offsets_ptr())); + is_column_nullable(*dst) ? static_cast(*dst).get_nested_column() + : *dst); + const bool read_meta_columns = need_to_read_meta_columns(); + MutableColumnPtr column_offsets_ptr; + if (read_meta_columns) { + column_offsets_ptr = IColumn::mutate(std::move(column_map.get_offsets_ptr())); + } else { + // The parent offsets were already materialized in the predicate phase, so + // they must not be appended to dst again. We still read offsets into a + // temporary column here: this sequential path may be serving a nested + // lazy read after seek_to_ordinal(), and the storage offsets are needed to + // compute how many key/value elements to read from the current source + // ordinal. The existing dst offsets only describe the filtered output + // shape and do not track the current source ordinal consumed by this + // iterator call. + const auto base_offset = + column_map.get_offsets().empty() ? 0 : column_map.get_offsets().back(); + column_offsets_ptr = ColumnMap::COffsets::create(); + assert_cast(*column_offsets_ptr) + .insert_value(base_offset); + } Defer defer_offsets {[&] { - auto typed_column_offsets_ptr = ColumnMap::COffsets::cast_to_column_mutptr( - assert_cast( - column_offsets_ptr.get())); - column_offsets_ptr = nullptr; - column_map.get_offsets_ptr() = std::move(typed_column_offsets_ptr); + if (read_meta_columns) { + auto typed_column_offsets_ptr = ColumnMap::COffsets::cast_to_column_mutptr( + assert_cast( + column_offsets_ptr.get())); + column_offsets_ptr = nullptr; + column_map.get_offsets_ptr() = std::move(typed_column_offsets_ptr); + } }}; bool offsets_has_null = false; ssize_t start = column_offsets_ptr->size(); @@ -1073,7 +1481,7 @@ Status MapFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* } } - if (dst->is_nullable()) { + if (is_column_nullable(*dst) && read_meta_columns) { size_t num_read = *n; auto null_map_ptr = static_cast(*dst).get_null_map_column_ptr(); // in not-null to null linked-schemachange mode, @@ -1095,12 +1503,14 @@ Status MapFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* Status MapFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t count, MutableColumnPtr& dst) { - if (_reading_flag == ReadingFlag::SKIP_READING) { - DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; - dst->resize(count); + if (!need_to_read()) { + DLOG(INFO) << "Map column iterator column " << _column_name << " skip reading."; + _convert_to_place_holder_column(dst, count); return Status::OK(); } + _recovery_from_place_holder_column(dst); + if (read_null_map_only()) { // NULL_MAP_ONLY mode: read null map by rowids, fill nested ColumnMap with empty defaults DORIS_CHECK(dst->is_nullable()); @@ -1128,47 +1538,80 @@ Status MapFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t if (count == 0) { return Status::OK(); } + // resolve ColumnMap and nullable wrapper auto& column_map = assert_cast( - dst->is_nullable() ? static_cast(*dst).get_nested_column() : *dst); - auto offsets_ptr = IColumn::mutate(std::move(column_map.get_offsets_ptr())); + is_column_nullable(*dst) ? static_cast(*dst).get_nested_column() + : *dst); + const bool read_meta_columns = need_to_read_meta_columns(); + MutableColumnPtr offsets_ptr; + if (read_meta_columns) { + offsets_ptr = IColumn::mutate(std::move(column_map.get_offsets_ptr())); + } else { + const auto base_offset = + column_map.get_offsets().empty() ? 0 : column_map.get_offsets().back(); + offsets_ptr = ColumnMap::COffsets::create(); + assert_cast(*offsets_ptr) + .insert_value(base_offset); + } Defer defer_offsets {[&] { - auto typed_offsets_ptr = ColumnMap::COffsets::cast_to_column_mutptr( - assert_cast(offsets_ptr.get())); - offsets_ptr = nullptr; - column_map.get_offsets_ptr() = std::move(typed_offsets_ptr); + if (read_meta_columns) { + auto typed_offsets_ptr = ColumnMap::COffsets::cast_to_column_mutptr( + assert_cast( + offsets_ptr.get())); + offsets_ptr = nullptr; + column_map.get_offsets_ptr() = std::move(typed_offsets_ptr); + } }}; auto& offsets = static_cast(*offsets_ptr); size_t base = offsets.get_data().empty() ? 0 : offsets.get_data().back(); // 1. bulk read null-map if nullable std::vector null_mask; // 0: not null, 1: null - if (_map_reader->is_nullable()) { - // For nullable map columns, the destination column must also be nullable. - if (UNLIKELY(!dst->is_nullable())) { + if (read_meta_columns) { + if (_map_reader->is_nullable()) { + // For nullable map columns, the destination column must also be nullable. + if (UNLIKELY(!is_column_nullable(*dst))) { + return Status::InternalError( + "unexpected non-nullable destination column for nullable map reader"); + } + MutableColumnPtr null_map_ptr = + static_cast(*dst).get_null_map_column_ptr(); + size_t null_before = null_map_ptr->size(); + RETURN_IF_ERROR(_null_iterator->read_by_rowids(rowids, count, null_map_ptr)); + // extract a light-weight view to decide element reads + auto& null_map_col = assert_cast(*null_map_ptr); + const auto* src = null_map_col.get_data().data() + null_before; + null_mask.assign(src, src + count); + } else if (is_column_nullable(*dst)) { + // in not-null to null linked-schemachange mode, + // actually we do not change dat data include meta in footer, + // so may dst from changed meta which is nullable but old data is not nullable, + // if so, we should set null_map to all null by default + MutableColumnPtr null_map_ptr = + static_cast(*dst).get_null_map_column_ptr(); + auto& null_map = assert_cast(*null_map_ptr); + null_map.insert_many_vals(0, count); + } + } else if (_map_reader->is_nullable()) { + // In lazy mode the parent null map has already been materialized during + // predicate read and filtered together with the block. Reuse that dst + // null map to avoid re-reading the same meta column from storage. + if (UNLIKELY(!is_column_nullable(*dst))) { return Status::InternalError( "unexpected non-nullable destination column for nullable map reader"); } - auto null_map_ptr = static_cast(*dst).get_null_map_column_ptr(); - size_t null_before = null_map_ptr->size(); - auto* null_map_col = null_map_ptr.get(); - MutableColumnPtr null_map_column = std::move(null_map_ptr); - RETURN_IF_ERROR(_null_iterator->read_by_rowids(rowids, count, null_map_column)); - // extract a light-weight view to decide element reads - null_mask.reserve(count); - for (size_t i = 0; i < count; ++i) { - null_mask.push_back(null_map_col->get_element(null_before + i)); - } - } else if (dst->is_nullable()) { - // in not-null to null linked-schemachange mode, - // actually we do not change dat data include meta in footer, - // so may dst from changed meta which is nullable but old data is not nullable, - // if so, we should set null_map to all null by default - auto null_map_ptr = static_cast(*dst).get_null_map_column_ptr(); - null_map_ptr->insert_many_vals(0, count); + const auto& null_map_col = static_cast(*dst).get_null_map_column(); + DORIS_CHECK(null_map_col.size() == count); + const auto* src = null_map_col.get_data().data(); + null_mask.assign(src, src + count); } - // 2. bulk read start ordinals for requested rows + // 2. Bulk read source start ordinals for requested rows. The offsets stored + // in dst already describe the filtered output shape when read_meta_columns is + // false, but they do not contain the source key/value ordinal for each + // selected rowid. We still need the storage offsets here to seek child + // iterators to the correct source element ranges. MutableColumnPtr starts_col = ColumnOffset64::create(); starts_col->reserve(count); RETURN_IF_ERROR(_offsets_iterator->read_by_rowids(rowids, count, starts_col)); @@ -1208,16 +1651,19 @@ Status MapFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t auto& next_starts_data = assert_cast(*next_starts_col).get_data(); std::vector sizes(count, 0); size_t acc = base; - const auto original_size = offsets.get_data().back(); - offsets.get_data().reserve(offsets.get_data().size() + count); + if (read_meta_columns) { + offsets.get_data().reserve(offsets.get_data().size() + count); + } for (size_t i = 0; i < count; ++i) { - size_t sz = static_cast(next_starts_data[i] - starts_data[i]); + auto sz = static_cast(next_starts_data[i] - starts_data[i]); if (_map_reader->is_nullable() && !null_mask.empty() && null_mask[i]) { sz = 0; // null rows do not consume elements } sizes[i] = sz; acc += sz; - offsets.get_data().push_back(acc); + if (read_meta_columns) { + offsets.get_data().push_back(acc); + } } // 6. read key/value elements for non-empty sizes @@ -1240,18 +1686,14 @@ Status MapFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t bool dummy_has_null = false; if (this_run != 0) { - if (_key_iterator->reading_flag() != ReadingFlag::SKIP_READING) { - RETURN_IF_ERROR(_key_iterator->seek_to_ordinal(start_idx)); - RETURN_IF_ERROR(_key_iterator->next_batch(&n, keys_ptr, &dummy_has_null)); - DCHECK(n == this_run); - } - - if (_val_iterator->reading_flag() != ReadingFlag::SKIP_READING) { - n = this_run; - RETURN_IF_ERROR(_val_iterator->seek_to_ordinal(start_idx)); - RETURN_IF_ERROR(_val_iterator->next_batch(&n, vals_ptr, &dummy_has_null)); - DCHECK(n == this_run); - } + RETURN_IF_ERROR(_key_iterator->seek_to_ordinal(start_idx)); + RETURN_IF_ERROR(_key_iterator->next_batch(&n, keys_ptr, &dummy_has_null)); + DCHECK(n == this_run); + + n = this_run; + RETURN_IF_ERROR(_val_iterator->seek_to_ordinal(start_idx)); + RETURN_IF_ERROR(_val_iterator->next_batch(&n, vals_ptr, &dummy_has_null)); + DCHECK(n == this_run); } start_idx = start; this_run = sz; @@ -1264,36 +1706,24 @@ Status MapFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t } size_t n = this_run; - const size_t total_count = offsets.get_data().back() - original_size; bool dummy_has_null = false; - if (_key_iterator->reading_flag() != ReadingFlag::SKIP_READING) { - if (this_run != 0) { - RETURN_IF_ERROR(_key_iterator->seek_to_ordinal(start_idx)); - RETURN_IF_ERROR(_key_iterator->next_batch(&n, keys_ptr, &dummy_has_null)); - DCHECK(n == this_run); - } - } else { - keys_ptr->insert_many_defaults(total_count); - } + if (this_run != 0) { + RETURN_IF_ERROR(_key_iterator->seek_to_ordinal(start_idx)); + RETURN_IF_ERROR(_key_iterator->next_batch(&n, keys_ptr, &dummy_has_null)); + DCHECK(n == this_run); - if (_val_iterator->reading_flag() != ReadingFlag::SKIP_READING) { - if (this_run != 0) { - n = this_run; - RETURN_IF_ERROR(_val_iterator->seek_to_ordinal(start_idx)); - RETURN_IF_ERROR(_val_iterator->next_batch(&n, vals_ptr, &dummy_has_null)); - DCHECK(n == this_run); - } - } else { - vals_ptr->insert_many_defaults(total_count); + n = this_run; + RETURN_IF_ERROR(_val_iterator->seek_to_ordinal(start_idx)); + RETURN_IF_ERROR(_val_iterator->next_batch(&n, vals_ptr, &dummy_has_null)); + DCHECK(n == this_run); } - return Status::OK(); } -void MapFileColumnIterator::set_need_to_read() { - set_reading_flag(ReadingFlag::NEED_TO_READ); - _key_iterator->set_need_to_read(); - _val_iterator->set_need_to_read(); +void MapFileColumnIterator::set_lazy_output_requirement() { + set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); + _key_iterator->set_lazy_output_requirement(); + _val_iterator->set_lazy_output_requirement(); } void MapFileColumnIterator::remove_pruned_sub_iterators() { @@ -1303,111 +1733,81 @@ void MapFileColumnIterator::remove_pruned_sub_iterators() { Status MapFileColumnIterator::set_access_paths(const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) { - if (all_access_paths.empty()) { + if (all_access_paths.empty() && predicate_access_paths.empty()) { return Status::OK(); } - if (!predicate_access_paths.empty()) { - set_reading_flag(ReadingFlag::READING_FOR_PREDICATE); - DLOG(INFO) << "Map column iterator set sub-column " << _column_name - << " to READING_FOR_PREDICATE"; - } - - auto sub_all_access_paths = DORIS_TRY(_get_sub_access_paths(all_access_paths)); - auto sub_predicate_access_paths = DORIS_TRY(_get_sub_access_paths(predicate_access_paths)); - - if (sub_all_access_paths.empty()) { + auto plan = DORIS_TRY(_prepare_nested_access_paths( + all_access_paths, predicate_access_paths, NestedMetaSupport::NULL_MAP_AND_OFFSET, + [this](ReadRequirement requirement) { + _key_iterator->set_read_requirement(requirement); + _val_iterator->set_read_requirement(requirement); + })); + if (plan.skip_data_descendants) { return Status::OK(); } - // Check for meta-only modes (OFFSET_ONLY or NULL_MAP_ONLY) - _check_and_set_meta_read_mode(sub_all_access_paths); - if (read_offset_only()) { - _key_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - _val_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Map column iterator set column " << _column_name - << " to OFFSET_ONLY reading mode, key/value columns set to SKIP_READING"; - return Status::OK(); - } - if (read_null_map_only()) { - _key_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - _val_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Map column iterator set column " << _column_name - << " to NULL_MAP_ONLY reading mode, key/value columns set to SKIP_READING"; + if (!plan.all.has_descendant_paths() && !plan.predicate.has_descendant_paths()) { return Status::OK(); } - TColumnAccessPaths key_all_access_paths; - TColumnAccessPaths val_all_access_paths; - TColumnAccessPaths key_predicate_access_paths; - TColumnAccessPaths val_predicate_access_paths; - - for (auto paths : sub_all_access_paths) { - if (paths.data_access_path.path[0] == ACCESS_ALL) { - // ACCESS_ALL means element_at(map, key) style access: the key column must be - // fully read so that the runtime can match the requested key, while any sub-path - // qualifiers (e.g. OFFSET) apply only to the value column. - // For key: create a path with just the column name (= full data access). - TColumnAccessPath key_path; - key_path.__set_type(paths.type); - TDataAccessPath key_data_path; - key_data_path.__set_path({_key_iterator->column_name()}); - key_path.__set_data_access_path(key_data_path); - key_all_access_paths.emplace_back(std::move(key_path)); - // For value: pass the full sub-path so qualifiers like OFFSET propagate. - paths.data_access_path.path[0] = _val_iterator->column_name(); - val_all_access_paths.emplace_back(paths); - } else if (paths.data_access_path.path[0] == ACCESS_MAP_KEYS) { - paths.data_access_path.path[0] = _key_iterator->column_name(); - key_all_access_paths.emplace_back(paths); - } else if (paths.data_access_path.path[0] == ACCESS_MAP_VALUES) { - paths.data_access_path.path[0] = _val_iterator->column_name(); - val_all_access_paths.emplace_back(paths); - } - } - const auto need_read_keys = !key_all_access_paths.empty(); - const auto need_read_values = !val_all_access_paths.empty(); - - for (auto paths : sub_predicate_access_paths) { - if (paths.data_access_path.path[0] == ACCESS_ALL) { - // Same logic as above: key needs full data, value gets the sub-path. - TColumnAccessPath key_path; - key_path.__set_type(paths.type); - TDataAccessPath key_data_path; - key_data_path.__set_path({_key_iterator->column_name()}); - key_path.__set_data_access_path(key_data_path); - key_predicate_access_paths.emplace_back(std::move(key_path)); - paths.data_access_path.path[0] = _val_iterator->column_name(); - val_predicate_access_paths.emplace_back(paths); - } else if (paths.data_access_path.path[0] == ACCESS_MAP_KEYS) { - paths.data_access_path.path[0] = _key_iterator->column_name(); - key_predicate_access_paths.emplace_back(paths); - } else if (paths.data_access_path.path[0] == ACCESS_MAP_VALUES) { - paths.data_access_path.path[0] = _val_iterator->column_name(); - val_predicate_access_paths.emplace_back(paths); - } - } - - if (need_read_keys) { - _key_iterator->set_reading_flag(ReadingFlag::NEED_TO_READ); - RETURN_IF_ERROR( - _key_iterator->set_access_paths(key_all_access_paths, key_predicate_access_paths)); + auto child_paths = DORIS_TRY(DescendantAccessPathRouter::route_map_paths_to_children( + std::move(plan.all.descendant_paths), std::move(plan.predicate.descendant_paths))); + + if (!child_paths.key.empty()) { + RETURN_IF_ERROR(_key_iterator->set_access_paths(child_paths.key.all_paths, + child_paths.key.predicate_paths)); + // Apply LAZY_OUTPUT after child predicate paths have been handled. Read requirements are + // monotonic, so a predicate-only child already promoted to PREDICATE will not + // be downgraded, while a non-predicate child becomes a lazy materialization target. + _key_iterator->set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); } else { - _key_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Map column iterator set key column to SKIP_READING"; + _key_iterator->set_read_requirement(ReadRequirement::SKIP); + DLOG(INFO) << "Map column iterator set key column to SKIP"; } - if (need_read_values) { - _val_iterator->set_reading_flag(ReadingFlag::NEED_TO_READ); - RETURN_IF_ERROR( - _val_iterator->set_access_paths(val_all_access_paths, val_predicate_access_paths)); + if (!child_paths.value.empty()) { + RETURN_IF_ERROR(_val_iterator->set_access_paths(child_paths.value.all_paths, + child_paths.value.predicate_paths)); + // Same as keys: predicate-only value paths stay PREDICATE because this + // post-processing update cannot lower a stronger child requirement. + _val_iterator->set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); } else { - _val_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Map column iterator set value column to SKIP_READING"; + _val_iterator->set_read_requirement(ReadRequirement::SKIP); + DLOG(INFO) << "Map column iterator set value column to SKIP"; } return Status::OK(); } +void MapFileColumnIterator::set_read_phase(ReadPhase mode) { + ColumnIterator::set_read_phase(mode); + _key_iterator->set_read_phase(mode); + _val_iterator->set_read_phase(mode); +} + +void MapFileColumnIterator::finalize_lazy_phase(MutableColumnPtr& dst) { + _recovery_from_place_holder_column(dst); + auto& map_column = assert_cast( + dst->is_nullable() ? static_cast(*dst).get_nested_column() : *dst); + auto keys_ptr = IColumn::mutate(std::move(map_column.get_keys_ptr())); + auto vals_ptr = IColumn::mutate(std::move(map_column.get_values_ptr())); + _key_iterator->finalize_lazy_phase(keys_ptr); + _val_iterator->finalize_lazy_phase(vals_ptr); + map_column.get_keys_ptr() = std::move(keys_ptr); + map_column.get_values_ptr() = std::move(vals_ptr); +} + +void MapFileColumnIterator::set_read_requirement(ReadRequirement requirement) { + set_read_requirement_self(requirement); + _key_iterator->set_read_requirement(requirement); + _val_iterator->set_read_requirement(requirement); +} + +bool MapFileColumnIterator::has_lazy_read_target() const { + return _read_requirement == ReadRequirement::LAZY_OUTPUT || + _key_iterator->has_lazy_read_target() || _val_iterator->has_lazy_read_target(); +} + //////////////////////////////////////////////////////////////////////////////// StructFileColumnIterator::StructFileColumnIterator( @@ -1420,7 +1820,7 @@ StructFileColumnIterator::StructFileColumnIterator( } Status StructFileColumnIterator::init(const ColumnIteratorOptions& opts) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (_read_requirement == ReadRequirement::SKIP) { DLOG(INFO) << "Struct column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -1435,12 +1835,14 @@ Status StructFileColumnIterator::init(const ColumnIteratorOptions& opts) { } Status StructFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Struct column iterator column " << _column_name << " skip reading."; - dst->resize(dst->size() + *n); + _convert_to_place_holder_column(dst, *n); return Status::OK(); } + _recovery_from_place_holder_column(dst); + if (read_null_map_only()) { // NULL_MAP_ONLY mode: read null map, fill nested ColumnStruct with empty defaults DORIS_CHECK(dst->is_nullable()); @@ -1484,7 +1886,7 @@ Status StructFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bo DCHECK(num_read == *n); } - if (dst->is_nullable()) { + if (is_column_nullable(*dst) && need_to_read_meta_columns()) { size_t num_read = *n; auto null_map_ptr = static_cast(*dst).get_null_map_column_ptr(); // in not-null to null linked-schemachange mode, @@ -1506,7 +1908,7 @@ Status StructFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bo } Status StructFileColumnIterator::seek_to_ordinal(ordinal_t ord) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Struct column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -1522,7 +1924,8 @@ Status StructFileColumnIterator::seek_to_ordinal(ordinal_t ord) { for (auto& column_iterator : _sub_column_iterators) { RETURN_IF_ERROR(column_iterator->seek_to_ordinal(ord)); } - if (_struct_reader->is_nullable()) { + + if (_struct_reader->is_nullable() && need_to_read_meta_columns()) { RETURN_IF_ERROR(_null_iterator->seek_to_ordinal(ord)); } return Status::OK(); @@ -1541,22 +1944,32 @@ Status StructFileColumnIterator::init_prefetcher(const SegmentPrefetchParams& pa void StructFileColumnIterator::collect_prefetchers( std::map>& prefetchers, PrefetcherInitMethod init_method) { - for (auto& column_iterator : _sub_column_iterators) { - column_iterator->collect_prefetchers(prefetchers, init_method); + if (!need_to_read()) { + return; } if (_struct_reader->is_nullable()) { _null_iterator->collect_prefetchers(prefetchers, init_method); } + if (read_null_map_only()) { + return; + } + for (auto& column_iterator : _sub_column_iterators) { + if (column_iterator->need_to_read()) { + column_iterator->collect_prefetchers(prefetchers, init_method); + } + } } Status StructFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t count, MutableColumnPtr& dst) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Struct column iterator column " << _column_name << " skip reading."; - dst->resize(count); + _convert_to_place_holder_column(dst, count); return Status::OK(); } + _recovery_from_place_holder_column(dst); + if (count == 0) { return Status::OK(); } @@ -1587,17 +2000,17 @@ Status StructFileColumnIterator::read_by_rowids(const rowid_t* rowids, const siz return Status::OK(); } -void StructFileColumnIterator::set_need_to_read() { - set_reading_flag(ReadingFlag::NEED_TO_READ); +void StructFileColumnIterator::set_lazy_output_requirement() { + set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); for (auto& sub_iterator : _sub_column_iterators) { - sub_iterator->set_need_to_read(); + sub_iterator->set_lazy_output_requirement(); } } void StructFileColumnIterator::remove_pruned_sub_iterators() { for (auto it = _sub_column_iterators.begin(); it != _sub_column_iterators.end();) { auto& sub_iterator = *it; - if (sub_iterator->reading_flag() == ReadingFlag::SKIP_READING) { + if (sub_iterator->read_requirement() == ReadRequirement::SKIP) { DLOG(INFO) << "Struct column iterator remove pruned sub-column " << sub_iterator->column_name(); it = _sub_column_iterators.erase(it); @@ -1611,67 +2024,87 @@ void StructFileColumnIterator::remove_pruned_sub_iterators() { Status StructFileColumnIterator::set_access_paths( const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) { - if (all_access_paths.empty()) { + if (all_access_paths.empty() && predicate_access_paths.empty()) { return Status::OK(); } - if (!predicate_access_paths.empty()) { - set_reading_flag(ReadingFlag::READING_FOR_PREDICATE); - DLOG(INFO) << "Struct column iterator set sub-column " << _column_name - << " to READING_FOR_PREDICATE"; - } - auto sub_all_access_paths = DORIS_TRY(_get_sub_access_paths(all_access_paths)); - auto sub_predicate_access_paths = DORIS_TRY(_get_sub_access_paths(predicate_access_paths)); + auto plan = DORIS_TRY(_prepare_nested_access_paths( + all_access_paths, predicate_access_paths, NestedMetaSupport::NULL_MAP, + [this](ReadRequirement requirement) { + for (auto& sub_iterator : _sub_column_iterators) { + sub_iterator->set_read_requirement(requirement); + } + })); - // Check for NULL_MAP_ONLY mode: only read null map, skip all sub-columns - _check_and_set_meta_read_mode(sub_all_access_paths); - if (read_null_map_only()) { - for (auto& sub_iterator : _sub_column_iterators) { - sub_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - } + if (plan.skip_data_descendants) { DLOG(INFO) << "Struct column iterator set column " << _column_name - << " to NULL_MAP_ONLY reading mode, all sub-columns set to SKIP_READING"; + << " to NULL_MAP_ONLY meta read mode, all sub-columns set to SKIP"; return Status::OK(); } - const auto no_sub_column_to_skip = sub_all_access_paths.empty(); - const auto no_predicate_sub_column = sub_predicate_access_paths.empty(); - + const bool reads_all_sub_columns = plan.all.reads_current_data; + const bool include_all_paths = !reads_all_sub_columns; for (auto& sub_iterator : _sub_column_iterators) { const auto name = sub_iterator->column_name(); - bool need_to_read = no_sub_column_to_skip; - TColumnAccessPaths sub_all_access_paths_of_this; - if (!need_to_read) { - for (const auto& paths : sub_all_access_paths) { - if (paths.data_access_path.path[0] == name) { - sub_all_access_paths_of_this.emplace_back(paths); - } - } - need_to_read = !sub_all_access_paths_of_this.empty(); - } + auto paths = DORIS_TRY(DescendantAccessPathRouter::select_struct_paths_for_child( + plan.all.descendant_paths, plan.predicate.descendant_paths, name, + include_all_paths)); + // Predicate paths must still reach the child even when no non-predicate path selects it. + const bool need_to_read = reads_all_sub_columns || !paths.empty(); if (!need_to_read) { - sub_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Struct column iterator set sub-column " << name << " to SKIP_READING"; + set_read_requirement_self(ReadRequirement::SKIP); + sub_iterator->set_read_requirement(ReadRequirement::SKIP); + DLOG(INFO) << "Struct column iterator set sub-column " << name << " to SKIP"; continue; } - set_reading_flag(ReadingFlag::NEED_TO_READ); - sub_iterator->set_reading_flag(ReadingFlag::NEED_TO_READ); - TColumnAccessPaths sub_predicate_access_paths_of_this; + RETURN_IF_ERROR(sub_iterator->set_access_paths(paths.all_paths, paths.predicate_paths)); + // Set LAZY_OUTPUT after routing child predicate paths. If the child was needed only for + // predicate evaluation, set_access_paths() has already promoted it to + // PREDICATE and this monotonic update will not downgrade it. Otherwise, this + // marks the child as a lazy materialization target. + set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); + sub_iterator->set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); + } + return Status::OK(); +} - if (!no_predicate_sub_column) { - for (const auto& paths : sub_predicate_access_paths) { - if (StringCaseEqual()(paths.data_access_path.path[0], name)) { - sub_predicate_access_paths_of_this.emplace_back(paths); - } - } - } +void StructFileColumnIterator::set_read_phase(ReadPhase mode) { + ColumnIterator::set_read_phase(mode); + for (auto& sub_iterator : _sub_column_iterators) { + sub_iterator->set_read_phase(mode); + } +} - RETURN_IF_ERROR(sub_iterator->set_access_paths(sub_all_access_paths_of_this, - sub_predicate_access_paths_of_this)); +void StructFileColumnIterator::finalize_lazy_phase(MutableColumnPtr& dst) { + _recovery_from_place_holder_column(dst); + auto& column_struct = assert_cast( + dst->is_nullable() ? static_cast(*dst).get_nested_column() : *dst); + + for (size_t i = 0; i < _sub_column_iterators.size(); ++i) { + auto& sub_column = column_struct.get_column_ptr(i); + MutableColumnPtr mutable_sub_column = IColumn::mutate(std::move(sub_column)); + _sub_column_iterators[i]->finalize_lazy_phase(mutable_sub_column); + sub_column = std::move(mutable_sub_column); } - return Status::OK(); +} + +void StructFileColumnIterator::set_read_requirement(ReadRequirement requirement) { + set_read_requirement_self(requirement); + for (const auto& sub_column_iterator : _sub_column_iterators) { + sub_column_iterator->set_read_requirement(requirement); + } +} + +bool StructFileColumnIterator::has_lazy_read_target() const { + if (_read_requirement == ReadRequirement::LAZY_OUTPUT) { + return true; + } + return std::any_of(_sub_column_iterators.begin(), _sub_column_iterators.end(), + [](const auto& sub_column_iterator) { + return sub_column_iterator->has_lazy_read_target(); + }); } //////////////////////////////////////////////////////////////////////////////// @@ -1758,8 +2191,8 @@ ArrayFileColumnIterator::ArrayFileColumnIterator(std::shared_ptr r } Status ArrayFileColumnIterator::init(const ColumnIteratorOptions& opts) { - if (_reading_flag == ReadingFlag::SKIP_READING) { - DLOG(INFO) << "Array column iterator column " << _column_name << " skip readking."; + if (_read_requirement == ReadRequirement::SKIP) { + DLOG(INFO) << "Array column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -1773,7 +2206,7 @@ Status ArrayFileColumnIterator::init(const ColumnIteratorOptions& opts) { Status ArrayFileColumnIterator::_seek_by_offsets(ordinal_t ord) { if (read_offset_only()) { - // In OFFSET_ONLY mode, item iterator is SKIP_READING, no need to seek it + // In OFFSET_ONLY mode, item iterator is SKIP, no need to seek it return Status::OK(); } // using offsets info @@ -1784,7 +2217,7 @@ Status ArrayFileColumnIterator::_seek_by_offsets(ordinal_t ord) { } Status ArrayFileColumnIterator::seek_to_ordinal(ordinal_t ord) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Array column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -1805,12 +2238,16 @@ Status ArrayFileColumnIterator::seek_to_ordinal(ordinal_t ord) { } Status ArrayFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) { - if (_reading_flag == ReadingFlag::SKIP_READING) { - DLOG(INFO) << "Array column iterator column " << _column_name << " skip reading."; - dst->resize(dst->size() + *n); + if (!need_to_read()) { + DLOG(INFO) << "Array column iterator column " << _column_name << " skip reading, read phase" + << static_cast(_read_phase) + << ", read requirement: " << static_cast(_read_requirement); + _convert_to_place_holder_column(dst, *n); return Status::OK(); } + _recovery_from_place_holder_column(dst); + if (read_null_map_only()) { // NULL_MAP_ONLY mode: read null map, fill nested ColumnArray with empty defaults DORIS_CHECK(dst->is_nullable()); @@ -1845,13 +2282,25 @@ Status ArrayFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, boo dst->is_nullable() ? static_cast(*dst).get_nested_column() : *dst); bool offsets_has_null = false; - auto column_offsets_ptr = std::move(*column_array.get_offsets_ptr()).mutate(); + const bool read_meta_columns = need_to_read_meta_columns(); + MutableColumnPtr column_offsets_ptr; + if (read_meta_columns) { + column_offsets_ptr = IColumn::mutate(std::move(column_array.get_offsets_ptr())); + } else { + const auto base_offset = + column_array.get_offsets().empty() ? 0 : column_array.get_offsets().back(); + column_offsets_ptr = ColumnArray::ColumnOffsets::create(); + assert_cast(*column_offsets_ptr) + .insert_value(base_offset); + } Defer defer_offsets {[&] { - auto typed_column_offsets_ptr = ColumnArray::ColumnOffsets::cast_to_column_mutptr( - assert_cast( - column_offsets_ptr.get())); - column_offsets_ptr = nullptr; - column_array.get_offsets_ptr() = std::move(typed_column_offsets_ptr); + if (read_meta_columns) { + auto typed_column_offsets_ptr = ColumnArray::ColumnOffsets::cast_to_column_mutptr( + assert_cast( + column_offsets_ptr.get())); + column_offsets_ptr = nullptr; + column_array.get_offsets_ptr() = std::move(typed_column_offsets_ptr); + } }}; ssize_t start = column_offsets_ptr->size(); RETURN_IF_ERROR(_offset_iterator->next_batch(n, column_offsets_ptr, &offsets_has_null)); @@ -1877,7 +2326,7 @@ Status ArrayFileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, boo } } - if (dst->is_nullable()) { + if (is_column_nullable(*dst) && read_meta_columns) { auto null_map_ptr = static_cast(*dst).get_null_map_column_ptr(); size_t num_read = *n; // in not-null to null linked-schemachange mode, @@ -1910,23 +2359,35 @@ Status ArrayFileColumnIterator::init_prefetcher(const SegmentPrefetchParams& par void ArrayFileColumnIterator::collect_prefetchers( std::map>& prefetchers, PrefetcherInitMethod init_method) { - _offset_iterator->collect_prefetchers(prefetchers, init_method); - // the actual data pages to read of item column depends on the read result of offset column, - // so we can't init prefetch blocks according to rowids, just prefetch all data blocks here. - _item_iterator->collect_prefetchers(prefetchers, PrefetcherInitMethod::ALL_DATA_BLOCKS); + if (!need_to_read()) { + return; + } + if (!read_null_map_only()) { + _offset_iterator->collect_prefetchers(prefetchers, init_method); + } if (_array_reader->is_nullable()) { _null_iterator->collect_prefetchers(prefetchers, init_method); } + if (read_offset_only() || read_null_map_only()) { + return; + } + // the actual data pages to read of item column depends on the read result of offset column, + // so we can't init prefetch blocks according to rowids, just prefetch all data blocks here. + if (_item_iterator->need_to_read()) { + _item_iterator->collect_prefetchers(prefetchers, PrefetcherInitMethod::ALL_DATA_BLOCKS); + } } Status ArrayFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t count, MutableColumnPtr& dst) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "Array column iterator column " << _column_name << " skip reading."; - dst->resize(count); + _convert_to_place_holder_column(dst, count); return Status::OK(); } + _recovery_from_place_holder_column(dst); + for (size_t i = 0; i < count; ++i) { // TODO(cambyszju): now read array one by one, need optimize later RETURN_IF_ERROR(seek_to_ordinal(rowids[i])); @@ -1936,68 +2397,68 @@ Status ArrayFileColumnIterator::read_by_rowids(const rowid_t* rowids, const size return Status::OK(); } -void ArrayFileColumnIterator::set_need_to_read() { - set_reading_flag(ReadingFlag::NEED_TO_READ); - _item_iterator->set_need_to_read(); +void ArrayFileColumnIterator::set_lazy_output_requirement() { + set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); + _item_iterator->set_lazy_output_requirement(); } void ArrayFileColumnIterator::remove_pruned_sub_iterators() { _item_iterator->remove_pruned_sub_iterators(); } -Status ArrayFileColumnIterator::set_access_paths(const TColumnAccessPaths& all_access_paths, - const TColumnAccessPaths& predicate_access_paths) { - if (all_access_paths.empty()) { - return Status::OK(); - } +void ArrayFileColumnIterator::set_read_phase(ReadPhase mode) { + ColumnIterator::set_read_phase(mode); + _item_iterator->set_read_phase(mode); +} - if (!predicate_access_paths.empty()) { - set_reading_flag(ReadingFlag::READING_FOR_PREDICATE); - DLOG(INFO) << "Array column iterator set sub-column " << _column_name - << " to READING_FOR_PREDICATE"; - } +void ArrayFileColumnIterator::finalize_lazy_phase(MutableColumnPtr& dst) { + _recovery_from_place_holder_column(dst); + auto& column_array = assert_cast( + dst->is_nullable() ? static_cast(*dst).get_nested_column() : *dst); + auto item_column_ptr = IColumn::mutate(std::move(column_array.get_data_ptr())); + _item_iterator->finalize_lazy_phase(item_column_ptr); + column_array.get_data_ptr() = std::move(item_column_ptr); +} - auto sub_all_access_paths = DORIS_TRY(_get_sub_access_paths(all_access_paths)); - auto sub_predicate_access_paths = DORIS_TRY(_get_sub_access_paths(predicate_access_paths)); +void ArrayFileColumnIterator::set_read_requirement(ReadRequirement requirement) { + set_read_requirement_self(requirement); + _item_iterator->set_read_requirement(requirement); +} - // Check for meta-only modes (OFFSET_ONLY or NULL_MAP_ONLY) - _check_and_set_meta_read_mode(sub_all_access_paths); - if (read_offset_only()) { - _item_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Array column iterator set column " << _column_name - << " to OFFSET_ONLY reading mode, item column set to SKIP_READING"; - return Status::OK(); - } - if (read_null_map_only()) { - _item_iterator->set_reading_flag(ReadingFlag::SKIP_READING); - DLOG(INFO) << "Array column iterator set column " << _column_name - << " to NULL_MAP_ONLY reading mode, item column set to SKIP_READING"; +bool ArrayFileColumnIterator::has_lazy_read_target() const { + return _read_requirement == ReadRequirement::LAZY_OUTPUT || + _item_iterator->has_lazy_read_target(); +} + +Status ArrayFileColumnIterator::set_access_paths(const TColumnAccessPaths& all_access_paths, + const TColumnAccessPaths& predicate_access_paths) { + if (all_access_paths.empty() && predicate_access_paths.empty()) { return Status::OK(); } - const auto no_sub_column_to_skip = sub_all_access_paths.empty(); - const auto no_predicate_sub_column = sub_predicate_access_paths.empty(); + auto plan = DORIS_TRY(_prepare_nested_access_paths( + all_access_paths, predicate_access_paths, NestedMetaSupport::NULL_MAP_AND_OFFSET, + [this](ReadRequirement requirement) { + _item_iterator->set_read_requirement(requirement); + })); - if (!no_sub_column_to_skip) { - for (auto& path : sub_all_access_paths) { - if (path.data_access_path.path[0] == ACCESS_ALL) { - path.data_access_path.path[0] = _item_iterator->column_name(); - } - } + if (plan.skip_data_descendants) { + return Status::OK(); } - if (!no_predicate_sub_column) { - for (auto& path : sub_predicate_access_paths) { - if (path.data_access_path.path[0] == ACCESS_ALL) { - path.data_access_path.path[0] = _item_iterator->column_name(); - } - } - } + const bool has_all_descendant = plan.all.has_descendant_paths(); + const bool has_predicate_descendant = plan.predicate.has_descendant_paths(); + auto child_paths = DORIS_TRY(DescendantAccessPathRouter::route_array_paths_to_item( + std::move(plan.all.descendant_paths), std::move(plan.predicate.descendant_paths), + _item_iterator->column_name())); - if (!no_sub_column_to_skip || !no_predicate_sub_column) { - _item_iterator->set_reading_flag(ReadingFlag::NEED_TO_READ); - RETURN_IF_ERROR( - _item_iterator->set_access_paths(sub_all_access_paths, sub_predicate_access_paths)); + if (has_all_descendant || has_predicate_descendant) { + RETURN_IF_ERROR(_item_iterator->set_access_paths(child_paths.all_paths, + child_paths.predicate_paths)); + // Predicate-only item paths stay PREDICATE because this update runs after + // child set_access_paths() and read requirements are monotonic. Non-predicate item paths are + // marked as lazy materialization targets. + _item_iterator->set_read_requirement_self(ReadRequirement::LAZY_OUTPUT); } return Status::OK(); } @@ -2022,24 +2483,35 @@ Status StringFileColumnIterator::init(const ColumnIteratorOptions& opts) { Status StringFileColumnIterator::set_access_paths( const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) { - if (all_access_paths.empty()) { - return Status::OK(); - } - - if (!predicate_access_paths.empty()) { - set_reading_flag(ReadingFlag::READING_FOR_PREDICATE); + RETURN_IF_ERROR(FileColumnIterator::set_access_paths(all_access_paths, predicate_access_paths)); + // OFFSET_ONLY mode is fundamentally incompatible with CHAR columns: + // CHAR is stored padded to its declared length (see + // OlapColumnDataConvertorChar::clone_and_padding), so the per-row length + // recorded in dict word info / page headers is always the padded length + // (e.g. 25 for CHAR(25)) — never the logical length expected by length(). + // Recovering the logical length requires scanning the chars buffer with + // strnlen() (shrink_padding_chars), which OFFSET_ONLY by definition skips. + // There is no partial-benefit path: any optimization that still produces + // the correct length() result must read the chars buffer in full. + // + // FE (NestedColumnPruning) already filters CHAR slots out of the + // OFFSET-only access plan, so reaching this branch means an FE/BE + // contract violation. Fail loudly instead of silently falling back. + if (read_offset_only() && get_reader() != nullptr && + get_reader()->get_meta_type() == FieldType::OLAP_FIELD_TYPE_CHAR) { + return Status::InternalError( + "OFFSET_ONLY access path is not supported on CHAR column '{}': CHAR is stored " + "padded so the per-row length information available without reading the chars " + "buffer is always the padded length, not the logical length. The FE planner " + "must not emit an OFFSET access path for CHAR columns.", + _column_name); } - - // Strip the column name from path[0] before checking for meta-only modes. - // Raw paths look like ["col_name", "OFFSET"] or ["col_name", "NULL"]. - auto sub_all_access_paths = DORIS_TRY(_get_sub_access_paths(all_access_paths)); - _check_and_set_meta_read_mode(sub_all_access_paths); if (read_offset_only()) { DLOG(INFO) << "String column iterator set column " << _column_name - << " to OFFSET_ONLY reading mode"; + << " to OFFSET_ONLY meta read mode"; } else if (read_null_map_only()) { DLOG(INFO) << "String column iterator set column " << _column_name - << " to NULL_MAP_ONLY reading mode"; + << " to NULL_MAP_ONLY meta read mode"; } return Status::OK(); @@ -2047,25 +2519,52 @@ Status StringFileColumnIterator::set_access_paths( //////////////////////////////////////////////////////////////////////////////// +Status FileColumnIterator::set_access_paths(const TColumnAccessPaths& all_access_paths, + const TColumnAccessPaths& predicate_access_paths) { + if (all_access_paths.empty() && predicate_access_paths.empty()) { + return Status::OK(); + } + + const auto requirement_before_access_path = _read_requirement; + if (!predicate_access_paths.empty()) { + set_read_requirement(ReadRequirement::PREDICATE); + } + + auto all_split = DORIS_TRY(_split_access_paths(all_access_paths)); + auto predicate_split = DORIS_TRY(_split_access_paths(predicate_access_paths)); + if (all_split.reads_current_data) { + set_lazy_output_requirement(); + } + if (predicate_split.reads_current_data) { + set_read_requirement(ReadRequirement::PREDICATE); + } + return _check_and_set_meta_read_mode(requirement_before_access_path, all_split); +} + FileColumnIterator::FileColumnIterator(std::shared_ptr reader) : _reader(reader) {} -void ColumnIterator::_check_and_set_meta_read_mode(const TColumnAccessPaths& sub_all_access_paths) { - for (const auto& path : sub_all_access_paths) { - if (!path.data_access_path.path.empty()) { - if (StringCaseEqual()(path.data_access_path.path[0], ACCESS_OFFSET)) { - _read_mode = ReadMode::OFFSET_ONLY; - return; - } else if (StringCaseEqual()(path.data_access_path.path[0], ACCESS_NULL)) { - _read_mode = ReadMode::NULL_MAP_ONLY; - return; - } - } +Status ColumnIterator::_check_and_set_meta_read_mode(ReadRequirement requirement_before_access_path, + const AccessPathSplit& all_access_paths) { + _meta_read_mode = MetaReadMode::DEFAULT; + if (requirement_before_access_path != ReadRequirement::NORMAL && + requirement_before_access_path != ReadRequirement::SKIP) { + // A stronger requirement means a parent/full-data path already required this iterator + // to materialize data. In that case a later predicate NULL/OFFSET path is only + // an additional predicate requirement and must not downgrade the read to + // meta-only. + return Status::OK(); } - _read_mode = ReadMode::DEFAULT; + + if (all_access_paths.reads_current_data || all_access_paths.has_descendant_paths()) { + return Status::OK(); + } + + _meta_read_mode = all_access_paths.current_meta_mode; + return Status::OK(); } Status FileColumnIterator::init(const ColumnIteratorOptions& opts) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (_read_requirement == ReadRequirement::SKIP) { DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -2110,7 +2609,7 @@ void FileColumnIterator::_trigger_prefetch_if_eligible(ordinal_t ord) { } Status FileColumnIterator::seek_to_ordinal(ordinal_t ord) { - if (_reading_flag == ReadingFlag::SKIP_READING) { + if (!need_to_read()) { DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; return Status::OK(); } @@ -2171,6 +2670,14 @@ Status FileColumnIterator::next_batch_of_zone_map(size_t* n, MutableColumnPtr& d } Status FileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) { + if (!need_to_read()) { + DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; + _convert_to_place_holder_column(dst, *n); + return Status::OK(); + } + + _recovery_from_place_holder_column(dst); + if (read_null_map_only()) { DLOG(INFO) << "File column iterator column " << _column_name << " in NULL_MAP_ONLY mode, reading only null map."; @@ -2219,12 +2726,6 @@ Status FileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* ha return Status::OK(); } - if (_reading_flag == ReadingFlag::SKIP_READING) { - DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; - dst->resize(dst->size() + *n); - return Status::OK(); - } - size_t curr_size = dst->byte_size(); dst->reserve(*n); size_t remaining = *n; @@ -2284,6 +2785,14 @@ Status FileColumnIterator::next_batch(size_t* n, MutableColumnPtr& dst, bool* ha Status FileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t count, MutableColumnPtr& dst) { + if (!need_to_read()) { + DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; + _convert_to_place_holder_column(dst, count); + return Status::OK(); + } + + _recovery_from_place_holder_column(dst); + if (read_null_map_only()) { DLOG(INFO) << "File column iterator column " << _column_name << " in NULL_MAP_ONLY mode, reading only null map by rowids."; @@ -2338,9 +2847,21 @@ Status FileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t co total_read_count += nrows_to_read; remaining -= nrows_to_read; } else { - memset(null_map_data.data() + base_size + total_read_count, 0, nrows_to_read); - total_read_count += nrows_to_read; - remaining -= nrows_to_read; + rowid_t current_ordinal_in_page = + cast_set(_page.offset_in_page + _page.first_ordinal); + size_t rows_in_current_page = 0; + for (size_t i = 0; i < nrows_to_read; ++i) { + if (rowids[total_read_count + i] - current_ordinal_in_page >= nrows_to_read) { + break; + } + ++rows_in_current_page; + } + DCHECK_GT(rows_in_current_page, 0); + memset(null_map_data.data() + base_size + total_read_count, 0, + rows_in_current_page); + _page.offset_in_page += rows_in_current_page; + total_read_count += rows_in_current_page; + remaining -= rows_in_current_page; } } @@ -2349,12 +2870,6 @@ Status FileColumnIterator::read_by_rowids(const rowid_t* rowids, const size_t co return Status::OK(); } - if (_reading_flag == ReadingFlag::SKIP_READING) { - DLOG(INFO) << "File column iterator column " << _column_name << " skip reading."; - dst->resize(count); - return Status::OK(); - } - size_t remaining = count; size_t total_read_count = 0; size_t nrows_to_read = 0; diff --git a/be/src/storage/segment/column_reader.h b/be/src/storage/segment/column_reader.h index 8c1831aa7cb9c9..a3bd46e5f701fb 100644 --- a/be/src/storage/segment/column_reader.h +++ b/be/src/storage/segment/column_reader.h @@ -19,10 +19,12 @@ #include #include +#include #include #include // for size_t #include // for uint32_t +#include #include #include // for unique_ptr #include @@ -374,7 +376,7 @@ class ColumnIterator { virtual Status set_access_paths(const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) { if (!predicate_access_paths.empty()) { - _reading_flag = ReadingFlag::READING_FOR_PREDICATE; + set_read_requirement_self(ReadRequirement::PREDICATE); } return Status::OK(); } @@ -383,33 +385,28 @@ class ColumnIterator { const std::string& column_name() const { return _column_name; } - // Since there may be multiple paths with conflicts or overlaps, - // we need to define several reading flags: + // Per-iterator read requirement derived from nested access paths. // - // NORMAL_READING — Default value, indicating that the column should be read. - // SKIP_READING — The column should not be read. - // NEED_TO_READ — The column must be read. - // READING_FOR_PREDICATE — The column is required for predicate evaluation. - // - // For example, suppose there are two paths: - // - Path 1 specifies that column A needs to be read, so it is marked as NEED_TO_READ. - // - Path 2 specifies that the column should not be read, but since it is already marked as NEED_TO_READ, - // it should not be changed to SKIP_READING. - enum class ReadingFlag : int { - NORMAL_READING, - SKIP_READING, - NEED_TO_READ, - READING_FOR_PREDICATE - }; - void set_reading_flag(ReadingFlag flag) { - if (static_cast(flag) > static_cast(_reading_flag)) { - _reading_flag = flag; - } + // The ordering is intentional and used by set_read_requirement_self(): requirements are + // monotonic and a weaker requirement must not downgrade a stronger one. + // - NORMAL: no pruning decision has been made yet. + // - SKIP: this iterator should not be read. + // - LAZY_OUTPUT: materialize this iterator in the lazy phase after predicate filtering. + // - PREDICATE: read this iterator in the predicate phase. This must stay stronger than + // LAZY_OUTPUT because parents may mark children as LAZY_OUTPUT after child set_access_paths() + // has already promoted predicate-only children to PREDICATE. + enum class ReadRequirement : int { NORMAL, SKIP, LAZY_OUTPUT, PREDICATE }; + + // Set the read requirement on this iterator and all nested child iterators. + virtual void set_read_requirement(ReadRequirement requirement) { + set_read_requirement_self(requirement); } - ReadingFlag reading_flag() const { return _reading_flag; } + ReadRequirement read_requirement() const { return _read_requirement; } - virtual void set_need_to_read() { set_reading_flag(ReadingFlag::NEED_TO_READ); } + virtual void set_lazy_output_requirement() { + set_read_requirement(ReadRequirement::LAZY_OUTPUT); + } virtual void remove_pruned_sub_iterators() {}; @@ -426,26 +423,126 @@ class ColumnIterator { static constexpr const char* ACCESS_NULL = "NULL"; // Meta-only read modes: - // - OFFSET_ONLY: only read offset information (e.g., for array_size/map_size/string_length) + // - OFFSET_ONLY: read offsets while skipping actual child/string data. For nullable + // complex columns, the parent null map is still materialized when needed. // - NULL_MAP_ONLY: only read null map (e.g., for IS NULL / IS NOT NULL predicates) // When these modes are enabled, actual content data is skipped. - enum class ReadMode : int { DEFAULT, OFFSET_ONLY, NULL_MAP_ONLY }; + enum class MetaReadMode : int { DEFAULT, OFFSET_ONLY, NULL_MAP_ONLY }; + + bool read_offset_only() const { return _meta_read_mode == MetaReadMode::OFFSET_ONLY; } + bool read_null_map_only() const { return _meta_read_mode == MetaReadMode::NULL_MAP_ONLY; } + + // The current scanner phase. This is intentionally separate from ReadRequirement + // (why this iterator is needed) and MetaReadMode (what physical metadata to read). + enum class ReadPhase : int { + NORMAL, // default full materialization without lazy read split + PREDICATE, // predicate evaluation before row filtering + LAZY // post-filter lazy materialization + }; - bool read_offset_only() const { return _read_mode == ReadMode::OFFSET_ONLY; } - bool read_null_map_only() const { return _read_mode == ReadMode::NULL_MAP_ONLY; } + virtual void set_read_phase(ReadPhase mode) { + _read_phase = mode; + if (mode == ReadPhase::PREDICATE) { + _has_place_holder_column = false; + } + } + + virtual bool need_to_read() const { + switch (_read_phase) { + case ReadPhase::NORMAL: + return _read_requirement != ReadRequirement::SKIP; + case ReadPhase::PREDICATE: + return _read_requirement == ReadRequirement::PREDICATE; + case ReadPhase::LAZY: + return _read_requirement == ReadRequirement::LAZY_OUTPUT; + default: + return false; + } + } + + // Whether the current iterator itself should materialize meta columns, such as + // the null-map column or the offset column, into the destination column. + // + // Do not use the virtual need_to_read() here. Complex iterators override + // need_to_read() in LAZY mode to keep the parent iterator active when only a + // nested child still has data to materialize. That parent-level control-flow + // decision is different from materializing the parent's own offsets/null-map: + // if the parent was already read for predicate evaluation, LAZY mode should + // only fill the missing children and must not append parent meta again. + bool need_to_read_meta_columns() const { return ColumnIterator::need_to_read(); } + + virtual void finalize_lazy_phase(MutableColumnPtr& dst) { + _recovery_from_place_holder_column(dst); + } + + // Set only this iterator's requirement without modifying requirements of any nested child + // iterators. Use this when the parent/wrapper state must be updated while child requirements + // are decided independently. + virtual void set_read_requirement_self(ReadRequirement requirement) { + if (static_cast(requirement) > static_cast(_read_requirement)) { + _read_requirement = requirement; + } + } + + // Whether this iterator or any nested iterator has data that must be materialized + // in lazy mode. Predicate-only branches are read before filtering and must not be + // re-read in the lazy phase. Meta-only access paths still become lazy targets when + // they appear only in all_access_paths, because OFFSET/NULL is the requested output. + virtual bool has_lazy_read_target() const { + return _read_requirement == ReadRequirement::LAZY_OUTPUT; + } protected: - // Checks sub access paths for OFFSET or NULL meta-only modes and - // updates _read_mode accordingly. Use the accessor helpers - // read_offset_only() / read_null_map_only() to query the current mode. - void _check_and_set_meta_read_mode(const TColumnAccessPaths& sub_all_access_paths); + struct AccessPathSplit { + TColumnAccessPaths descendant_paths; + bool reads_current_data = false; + MetaReadMode current_meta_mode = MetaReadMode::DEFAULT; + + bool has_descendant_paths() const { return !descendant_paths.empty(); } + }; + + // Nested columns share the same current-level access-path planning, while their data-child + // topology and descendant routing remain container-specific. + struct NestedAccessPathPlan { + AccessPathSplit all; + AccessPathSplit predicate; + bool skip_data_descendants = false; + }; + + // At their current level, Struct supports null-map metadata. Map and Array additionally + // support offsets. + enum class NestedMetaSupport { NULL_MAP, NULL_MAP_AND_OFFSET }; + + void _convert_to_place_holder_column(MutableColumnPtr& dst, size_t count); + + void _recovery_from_place_holder_column(MutableColumnPtr& dst); - Result _get_sub_access_paths(const TColumnAccessPaths& access_paths); + // Derive current-level meta-only read mode from an explicit access-path split. Meta-only is + // valid only when this iterator had no data-read requirement before applying the current paths, + // no current DATA path exists, and no path must be routed to a descendant iterator. + Status _check_and_set_meta_read_mode(ReadRequirement requirement_before_access_path, + const AccessPathSplit& all_access_paths); + + // Apply the common current-level access-path state transitions and select a supported + // parent-owned meta-only mode. When that mode skips data descendants, synchronously invoke the + // callback once with SKIP before returning the routing plan. The callback is never retained. + Result _prepare_nested_access_paths( + const TColumnAccessPaths& all_access_paths, + const TColumnAccessPaths& predicate_access_paths, NestedMetaSupport meta_support, + const std::function& set_all_data_descendants_read_requirement); + + // Normalize the wire encoding, strip this iterator's column name, and explicitly partition + // paths consumed by this iterator from paths that must be routed to descendants. This helper is + // intentionally side-effect free; callers apply DATA/predicate read requirements explicitly. + Result _split_access_paths(TColumnAccessPaths access_paths) const; ColumnIteratorOptions _opts; - ReadingFlag _reading_flag {ReadingFlag::NORMAL_READING}; - ReadMode _read_mode = ReadMode::DEFAULT; + ReadRequirement _read_requirement {ReadRequirement::NORMAL}; + MetaReadMode _meta_read_mode = MetaReadMode::DEFAULT; + ReadPhase _read_phase {ReadPhase::NORMAL}; std::string _column_name; + + bool _has_place_holder_column {false}; }; // This iterator is used to read column data from file @@ -468,6 +565,9 @@ class FileColumnIterator : public ColumnIterator { Status read_by_rowids(const rowid_t* rowids, const size_t count, MutableColumnPtr& dst) override; + Status set_access_paths(const TColumnAccessPaths& all_access_paths, + const TColumnAccessPaths& predicate_access_paths) override; + ordinal_t get_current_ordinal() const override { return _current_ordinal; } // get row ranges by zone map @@ -495,6 +595,11 @@ class FileColumnIterator : public ColumnIterator { std::map>& prefetchers, PrefetcherInitMethod init_method) override; +protected: + // Exposed to derived iterators (e.g. StringFileColumnIterator) so they can + // query column metadata such as the storage field type. + const std::shared_ptr& get_reader() const { return _reader; } + private: Status _seek_to_pos_in_page(ParsedPage* page, ordinal_t offset_in_page) const; Status _load_next_page(bool* eos); @@ -539,10 +644,10 @@ class EmptyFileColumnIterator final : public ColumnIterator { ordinal_t get_current_ordinal() const override { return 0; } }; -// StringFileColumnIterator extends FileColumnIterator with meta-only reading -// support for string/binary column types. When the OFFSET path is detected in -// set_access_paths, it sets only_read_offsets on the ColumnIteratorOptions so -// that the BinaryPlainPageDecoder skips chars memcpy and only fills offsets. +// StringFileColumnIterator extends FileColumnIterator's NULL metadata support with OFFSET-only +// reading for string/binary column types. When the OFFSET path is detected in set_access_paths, it +// sets only_read_offsets on the ColumnIteratorOptions so that the BinaryPlainPageDecoder skips +// chars memcpy and only fills offsets. class StringFileColumnIterator final : public FileColumnIterator { public: explicit StringFileColumnIterator(std::shared_ptr reader); @@ -589,6 +694,11 @@ class OffsetFileColumnIterator final : public ColumnIterator { return _offset_iterator->read_by_rowids(rowids, count, dst); } + void set_read_requirement(ReadRequirement requirement) override { + set_read_requirement_self(requirement); + _offset_iterator->set_read_requirement(requirement); + } + Status init_prefetcher(const SegmentPrefetchParams& params) override; void collect_prefetchers( std::map>& prefetchers, @@ -634,10 +744,33 @@ class MapFileColumnIterator final : public ColumnIterator { Status set_access_paths(const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) override; - void set_need_to_read() override; + void set_lazy_output_requirement() override; void remove_pruned_sub_iterators() override; + void set_read_phase(ReadPhase mode) override; + + bool need_to_read() const override { + switch (_read_phase) { + case ReadPhase::NORMAL: + return _read_requirement != ReadRequirement::SKIP; + case ReadPhase::PREDICATE: + return _read_requirement == ReadRequirement::PREDICATE; + case ReadPhase::LAZY: + // In lazy mode, read this map only when at least one key/value branch still + // has non-predicate data to materialize. + return has_lazy_read_target(); + default: + return false; + } + } + + void finalize_lazy_phase(MutableColumnPtr& dst) override; + + void set_read_requirement(ReadRequirement requirement) override; + + bool has_lazy_read_target() const override; + private: std::shared_ptr _map_reader = nullptr; ColumnIteratorUPtr _null_iterator; @@ -678,7 +811,7 @@ class StructFileColumnIterator final : public ColumnIterator { Status set_access_paths(const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) override; - void set_need_to_read() override; + void set_lazy_output_requirement() override; void remove_pruned_sub_iterators() override; @@ -687,6 +820,27 @@ class StructFileColumnIterator final : public ColumnIterator { std::map>& prefetchers, PrefetcherInitMethod init_method) override; + void set_read_phase(ReadPhase mode) override; + + bool need_to_read() const override { + switch (_read_phase) { + case ReadPhase::NORMAL: + return _read_requirement != ReadRequirement::SKIP; + case ReadPhase::PREDICATE: + return _read_requirement == ReadRequirement::PREDICATE; + case ReadPhase::LAZY: + // In lazy mode, read this struct only when at least one nested branch still + // has non-predicate data to materialize. + return has_lazy_read_target(); + default: + return false; + } + } + + void finalize_lazy_phase(MutableColumnPtr& dst) override; + void set_read_requirement(ReadRequirement requirement) override; + bool has_lazy_read_target() const override; + private: std::shared_ptr _struct_reader = nullptr; ColumnIteratorUPtr _null_iterator; @@ -725,7 +879,7 @@ class ArrayFileColumnIterator final : public ColumnIterator { Status set_access_paths(const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) override; - void set_need_to_read() override; + void set_lazy_output_requirement() override; void remove_pruned_sub_iterators() override; @@ -734,6 +888,29 @@ class ArrayFileColumnIterator final : public ColumnIterator { std::map>& prefetchers, PrefetcherInitMethod init_method) override; + void set_read_phase(ReadPhase mode) override; + + bool need_to_read() const override { + switch (_read_phase) { + case ReadPhase::NORMAL: + return _read_requirement != ReadRequirement::SKIP; + case ReadPhase::PREDICATE: + return _read_requirement == ReadRequirement::PREDICATE; + case ReadPhase::LAZY: + // In lazy mode, read this array only when its item branch still has + // non-predicate data to materialize. + return has_lazy_read_target(); + default: + return false; + } + } + + void finalize_lazy_phase(MutableColumnPtr& dst) override; + + void set_read_requirement(ReadRequirement requirement) override; + + bool has_lazy_read_target() const override; + private: std::shared_ptr _array_reader = nullptr; std::unique_ptr _offset_iterator; diff --git a/be/src/storage/segment/segment_iterator.cpp b/be/src/storage/segment/segment_iterator.cpp index 8198e00b674061..85f26f8c4f93d3 100644 --- a/be/src/storage/segment/segment_iterator.cpp +++ b/be/src/storage/segment/segment_iterator.cpp @@ -22,6 +22,7 @@ #include #include #include +#include #include #include @@ -119,6 +120,30 @@ namespace segment_v2 { #include "common/compile_check_begin.h" +class ScopedColumnIteratorReadPhase { +public: + ScopedColumnIteratorReadPhase(ColumnIterator* column_iter, ColumnIterator::ReadPhase mode) + : _column_iter(column_iter) { + DORIS_CHECK(_column_iter != nullptr); + _column_iter->set_read_phase(mode); + } + + ScopedColumnIteratorReadPhase(const ScopedColumnIteratorReadPhase&) = delete; + ScopedColumnIteratorReadPhase& operator=(const ScopedColumnIteratorReadPhase&) = delete; + + ~ScopedColumnIteratorReadPhase() { + // ReadPhase is a per-read phase knob. SegmentIterator only needs a + // temporary PREDICATE/LAZY mode while reading one column in one phase; it + // must be restored before the next column or later normal reads reuse the + // same ColumnIterator. Keep the restoration in one scoped helper instead + // of open-coding the same Defer block at every call site. + _column_iter->set_read_phase(ColumnIterator::ReadPhase::NORMAL); + } + +private: + ColumnIterator* _column_iter = nullptr; +}; + SegmentIterator::~SegmentIterator() = default; void SegmentIterator::_init_row_bitmap_by_condition_cache() { @@ -419,6 +444,10 @@ Status SegmentIterator::_init_impl(const StorageReadOptions& opts) { _score_runtime = _opts.score_runtime; _ann_topn_runtime = _opts.ann_topn_runtime; + _enable_prune_nested_column = _opts.io_ctx.reader_type == ReaderType::READER_QUERY && + _opts.runtime_state && + _opts.runtime_state->enable_prune_nested_column(); + if (opts.output_columns != nullptr) { _output_columns = *(opts.output_columns); } @@ -657,8 +686,14 @@ void SegmentIterator::_init_segment_prefetchers() { ? PrefetcherInitMethod::FROM_ROWIDS : PrefetcherInitMethod::ALL_DATA_BLOCKS; std::map> prefetchers; - for (const auto& column_iter : _column_iterators) { + for (size_t idx = 0; idx < _column_iterators.size(); ++idx) { + auto cid = cast_set(idx); + auto* column_iter = _column_iterators[cid].get(); if (column_iter != nullptr) { + ScopedColumnIteratorReadPhase scoped_read_phase { + column_iter, _support_lazy_read_pruned_columns.contains(cid) + ? ColumnIterator::ReadPhase::PREDICATE + : ColumnIterator::ReadPhase::NORMAL}; column_iter->collect_prefetchers(prefetchers, init_method); } } @@ -1984,6 +2019,25 @@ Status SegmentIterator::_vec_init_lazy_materialization() { if (_is_common_expr_column[cid] || _is_pred_column[cid]) { auto loc = _schema_block_id_map[cid]; _columns_to_filter.push_back(loc); + + const auto field_type = _schema->column(cid)->type(); + if (_is_common_expr_column[cid] && _enable_prune_nested_column && + (field_type == FieldType::OLAP_FIELD_TYPE_STRUCT || + field_type == FieldType::OLAP_FIELD_TYPE_ARRAY || + field_type == FieldType::OLAP_FIELD_TYPE_MAP)) { + DCHECK(_column_iterators[cid]); + if (_column_iterators[cid]->read_requirement() == + ColumnIterator::ReadRequirement::PREDICATE && + _column_iterators[cid]->has_lazy_read_target()) { + // Only split lazy recovery for complex common expr columns that have + // both predicate-only and non-predicate nested targets. The two requirement + // checks already imply that nested-column pruning happened: without an + // explicit predicate sub-path the parent would not be + // PREDICATE, and without a pruned non-predicate child there + // would be no lazy target to recover after filtering. + _support_lazy_read_pruned_columns.emplace(cid); + } + } } } @@ -2301,16 +2355,22 @@ Status SegmentIterator::_read_columns_by_index(uint32_t nrows_read_limit, uint16 }) } + auto* column_iter = _column_iterators[cid].get(); + ScopedColumnIteratorReadPhase scoped_read_phase { + column_iter, _support_lazy_read_pruned_columns.contains(cid) + ? ColumnIterator::ReadPhase::PREDICATE + : ColumnIterator::ReadPhase::NORMAL}; + if (is_continuous) { size_t rows_read = nrows_read; _opts.stats->predicate_column_read_seek_num += 1; if (_opts.runtime_state && _opts.runtime_state->enable_profile()) { SCOPED_RAW_TIMER(&_opts.stats->predicate_column_read_seek_ns); - RETURN_IF_ERROR(_column_iterators[cid]->seek_to_ordinal(_block_rowids[0])); + RETURN_IF_ERROR(column_iter->seek_to_ordinal(_block_rowids[0])); } else { - RETURN_IF_ERROR(_column_iterators[cid]->seek_to_ordinal(_block_rowids[0])); + RETURN_IF_ERROR(column_iter->seek_to_ordinal(_block_rowids[0])); } - RETURN_IF_ERROR(_column_iterators[cid]->next_batch(&rows_read, column)); + RETURN_IF_ERROR(column_iter->next_batch(&rows_read, column)); if (rows_read != nrows_read) { return Status::Error("nrows({}) != rows_read({})", nrows_read, rows_read); @@ -2330,20 +2390,18 @@ Status SegmentIterator::_read_columns_by_index(uint32_t nrows_read_limit, uint16 _opts.stats->predicate_column_read_seek_num += 1; if (_opts.runtime_state && _opts.runtime_state->enable_profile()) { SCOPED_RAW_TIMER(&_opts.stats->predicate_column_read_seek_ns); - RETURN_IF_ERROR( - _column_iterators[cid]->seek_to_ordinal(_block_rowids[processed])); + RETURN_IF_ERROR(column_iter->seek_to_ordinal(_block_rowids[processed])); } else { - RETURN_IF_ERROR( - _column_iterators[cid]->seek_to_ordinal(_block_rowids[processed])); + RETURN_IF_ERROR(column_iter->seek_to_ordinal(_block_rowids[processed])); } - RETURN_IF_ERROR(_column_iterators[cid]->next_batch(&rows_read, column)); + RETURN_IF_ERROR(column_iter->next_batch(&rows_read, column)); if (rows_read != current_batch_size) { return Status::Error( "batch nrows({}) != rows_read({})", current_batch_size, rows_read); } } else { - RETURN_IF_ERROR(_column_iterators[cid]->read_by_rowids( - &_block_rowids[processed], current_batch_size, column)); + RETURN_IF_ERROR(column_iter->read_by_rowids(&_block_rowids[processed], + current_batch_size, column)); } processed += current_batch_size; } @@ -2483,7 +2541,8 @@ Status SegmentIterator::_read_columns_by_rowids(std::vector& read_colu std::vector& rowid_vector, uint16_t* sel_rowid_idx, size_t select_size, MutableColumns* mutable_columns, - bool init_condition_cache) { + bool init_condition_cache, + bool read_for_predicate) { SCOPED_RAW_TIMER(&_opts.stats->lazy_read_ns); std::vector rowids(select_size); @@ -2527,10 +2586,45 @@ Status SegmentIterator::_read_columns_by_rowids(std::vector& read_colu "SegmentIterator meet invalid column, return columns size {}, cid {}", _current_return_columns.size(), cid); } - RETURN_IF_ERROR(_column_iterators[cid]->read_by_rowids(rowids.data(), select_size, - _current_return_columns[cid])); + + auto* column_iter = _column_iterators[cid].get(); + ScopedColumnIteratorReadPhase scoped_read_phase { + column_iter, read_for_predicate && _support_lazy_read_pruned_columns.contains(cid) + ? ColumnIterator::ReadPhase::PREDICATE + : ColumnIterator::ReadPhase::NORMAL}; + + RETURN_IF_ERROR(column_iter->read_by_rowids(rowids.data(), select_size, + _current_return_columns[cid])); + } + + return Status::OK(); +} + +Status SegmentIterator::_read_lazy_pruned_columns(Block* block) { + if (_support_lazy_read_pruned_columns.empty()) { + return Status::OK(); + } + + SCOPED_RAW_TIMER(&_opts.stats->lazy_read_pruned_ns); + DorisVector rowids(_selected_size); + for (size_t i = 0; i < _selected_size; ++i) { + rowids[i] = _block_rowids[_sel_rowid_idx[i]]; } + for (auto cid : _support_lazy_read_pruned_columns) { + // branch-4.2 keeps the column-id -> block-position map inside SegmentIterator; + // master's Schema::column_index() helper does not exist on this branch. + auto loc = cast_set(_schema_block_id_map[cid]); + auto column = IColumn::mutate(std::move(block->get_by_position(loc).column)); + auto* column_iter = _column_iterators[cid].get(); + ScopedColumnIteratorReadPhase scoped_read_phase {column_iter, + ColumnIterator::ReadPhase::LAZY}; + if (_selected_size > 0) { + RETURN_IF_ERROR(column_iter->read_by_rowids(rowids.data(), _selected_size, column)); + } + column_iter->finalize_lazy_phase(column); + block->get_by_position(loc).column = std::move(column); + } return Status::OK(); } @@ -2735,7 +2829,7 @@ Status SegmentIterator::_next_batch_internal(Block* block) { SCOPED_RAW_TIMER(&_opts.stats->non_predicate_read_ns); RETURN_IF_ERROR(_read_columns_by_rowids( _non_predicate_column_ids, _block_rowids, _sel_rowid_idx.data(), - _selected_size, &_current_return_columns)); + _selected_size, &_current_return_columns, false, true)); _replace_version_col_if_needed(_non_predicate_column_ids, _selected_size); RETURN_IF_ERROR(_process_columns(_non_predicate_column_ids, block)); } @@ -2766,7 +2860,7 @@ Status SegmentIterator::_next_batch_internal(Block* block) { RETURN_IF_ERROR(_read_columns_by_rowids( _non_predicate_columns, _block_rowids, _sel_rowid_idx.data(), _selected_size, &_current_return_columns, - _opts.condition_cache_digest && !_find_condition_cache)); + _opts.condition_cache_digest && !_find_condition_cache, false)); _replace_version_col_if_needed(_non_predicate_columns, _selected_size); } else { if (_opts.condition_cache_digest && !_find_condition_cache) { @@ -2778,6 +2872,8 @@ Status SegmentIterator::_next_batch_internal(Block* block) { } } } + + RETURN_IF_ERROR(_read_lazy_pruned_columns(block)); } // step5: output columns diff --git a/be/src/storage/segment/segment_iterator.h b/be/src/storage/segment/segment_iterator.h index d8f61daeba3553..5d849d438957f5 100644 --- a/be/src/storage/segment/segment_iterator.h +++ b/be/src/storage/segment/segment_iterator.h @@ -227,7 +227,9 @@ class SegmentIterator : public RowwiseIterator { std::vector& rowid_vector, uint16_t* sel_rowid_idx, size_t select_size, MutableColumns* mutable_columns, - bool init_condition_cache = false); + bool init_condition_cache = false, + bool read_for_predicate = false); + [[nodiscard]] Status _read_lazy_pruned_columns(Block* block); Status copy_column_data_by_selector(IColumn* input_col_ptr, MutableColumnPtr& output_col, uint16_t* sel_rowid_idx, uint16_t select_size, @@ -372,6 +374,9 @@ class SegmentIterator : public RowwiseIterator { bool _is_need_short_eval = false; bool _is_need_expr_eval = false; + std::set _support_lazy_read_pruned_columns; + bool _enable_prune_nested_column = false; + // fields for vectorization execution std::vector _vec_pred_column_ids; // keep columnId of columns for vectorized predicate evaluation diff --git a/be/src/storage/segment/variant/variant_column_reader.cpp b/be/src/storage/segment/variant/variant_column_reader.cpp index 8dca2d0034443a..6d0de2fb460732 100644 --- a/be/src/storage/segment/variant/variant_column_reader.cpp +++ b/be/src/storage/segment/variant/variant_column_reader.cpp @@ -86,7 +86,7 @@ class ReaderOwnedColumnIterator final : public ColumnIterator { : _inner(std::move(inner)), _owner(std::move(owner)) { DCHECK(_inner != nullptr); set_column_name(_inner->column_name()); - set_reading_flag(_inner->reading_flag()); + set_read_requirement(_inner->read_requirement()); } Status init(const ColumnIteratorOptions& opts) override { return _inner->init(opts); } @@ -130,15 +130,36 @@ class ReaderOwnedColumnIterator final : public ColumnIterator { Status set_access_paths(const TColumnAccessPaths& all_access_paths, const TColumnAccessPaths& predicate_access_paths) override { RETURN_IF_ERROR(_inner->set_access_paths(all_access_paths, predicate_access_paths)); - set_reading_flag(_inner->reading_flag()); + ColumnIterator::set_read_requirement_self(_inner->read_requirement()); return Status::OK(); } - void set_need_to_read() override { - _inner->set_need_to_read(); - set_reading_flag(_inner->reading_flag()); + void set_read_requirement(ReadRequirement requirement) override { + _inner->set_read_requirement(requirement); + ColumnIterator::set_read_requirement_self(_inner->read_requirement()); } + void set_read_requirement_self(ReadRequirement requirement) override { + _inner->set_read_requirement_self(requirement); + ColumnIterator::set_read_requirement_self(_inner->read_requirement()); + } + + void set_lazy_output_requirement() override { + _inner->set_lazy_output_requirement(); + ColumnIterator::set_read_requirement_self(_inner->read_requirement()); + } + + void set_read_phase(ReadPhase mode) override { + ColumnIterator::set_read_phase(mode); + _inner->set_read_phase(mode); + } + + void finalize_lazy_phase(MutableColumnPtr& dst) override { _inner->finalize_lazy_phase(dst); } + + bool has_lazy_read_target() const override { return _inner->has_lazy_read_target(); } + + bool need_to_read() const override { return _inner->need_to_read(); } + void remove_pruned_sub_iterators() override { _inner->remove_pruned_sub_iterators(); } Status init_prefetcher(const SegmentPrefetchParams& params) override { diff --git a/be/test/runtime/descriptor_test.cpp b/be/test/runtime/descriptor_test.cpp index 20c2b1000c8303..0a665cfc34c343 100644 --- a/be/test/runtime/descriptor_test.cpp +++ b/be/test/runtime/descriptor_test.cpp @@ -15,8 +15,10 @@ // specific language governing permissions and limitations // under the License. +#include #include #include +#include #include #include "common/exception.h" @@ -196,4 +198,59 @@ TEST_F(SlotDescriptorTest, DebugString) { EXPECT_TRUE(debug_str2.find("is_virtual=true") != std::string::npos); } -} // namespace doris \ No newline at end of file +TEST_F(SlotDescriptorTest, AccessPathsPreservedThroughProtobuf) { + TColumnAccessPath data_path; + data_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + data_path.__set_type(TAccessPathType::DATA); + TDataAccessPath data_payload; + data_payload.__set_path({"s", "field"}); + data_path.__set_data_access_path(data_payload); + + TColumnAccessPath meta_path; + meta_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + meta_path.__set_type(TAccessPathType::META); + TMetaAccessPath meta_payload; + meta_payload.__set_path({"s", "field", "NULL"}); + meta_path.__set_meta_access_path(meta_payload); + + TColumnAccessPath legacy_path; + legacy_path.__set_type(TAccessPathType::DATA); + data_payload.__set_path({"s", "legacy", "NULL"}); + legacy_path.__set_data_access_path(data_payload); + + TColumnAccessPath explicit_legacy_path; + explicit_legacy_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + explicit_legacy_path.__set_type(TAccessPathType::DATA); + data_payload.__set_path({"s", "legacy", "VALUES"}); + explicit_legacy_path.__set_data_access_path(data_payload); + + TSlotDescriptor thrift_descriptor = create_basic_slot_descriptor(); + thrift_descriptor.__set_all_access_paths( + {data_path, meta_path, legacy_path, explicit_legacy_path}); + thrift_descriptor.__set_predicate_access_paths({meta_path}); + + SlotDescriptor original(thrift_descriptor); + PSlotDescriptor protobuf_descriptor; + original.to_protobuf(&protobuf_descriptor); + ASSERT_EQ(protobuf_descriptor.all_access_paths_size(), 4); + EXPECT_TRUE(protobuf_descriptor.all_access_paths(0).has_version()); + EXPECT_EQ(protobuf_descriptor.all_access_paths(0).version(), + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + EXPECT_TRUE(protobuf_descriptor.all_access_paths(1).has_version()); + EXPECT_FALSE(protobuf_descriptor.all_access_paths(2).has_version()); + EXPECT_TRUE(protobuf_descriptor.all_access_paths(3).has_version()); + EXPECT_EQ(protobuf_descriptor.all_access_paths(3).version(), + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + SlotDescriptor round_trip(protobuf_descriptor); + + EXPECT_EQ(round_trip.all_access_paths(), thrift_descriptor.all_access_paths); + EXPECT_EQ(round_trip.predicate_access_paths(), thrift_descriptor.predicate_access_paths); + EXPECT_FALSE(round_trip.all_access_paths()[1].__isset.data_access_path); + EXPECT_FALSE(round_trip.predicate_access_paths()[0].__isset.data_access_path); + EXPECT_FALSE(round_trip.all_access_paths()[2].__isset.version); + EXPECT_TRUE(round_trip.all_access_paths()[3].__isset.version); + EXPECT_EQ(round_trip.all_access_paths()[3].version, + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); +} + +} // namespace doris diff --git a/be/test/storage/segment/column_reader_test.cpp b/be/test/storage/segment/column_reader_test.cpp index de8b5aadf1f3de..6bcdd148f8f766 100644 --- a/be/test/storage/segment/column_reader_test.cpp +++ b/be/test/storage/segment/column_reader_test.cpp @@ -16,6 +16,7 @@ // under the License. #include "storage/segment/column_reader.h" +#include #include #include #include @@ -23,27 +24,394 @@ #include #include +#include +#include #include +#include #include +#include #include #include "agent/be_exec_version_manager.h" #include "common/config.h" #include "io/fs/file_reader.h" +#include "io/fs/file_system.h" +#include "io/fs/file_writer.h" +#include "io/fs/local_file_system.h" +#include "storage/olap_common.h" #include "storage/segment/column_reader_cache.h" +#include "storage/segment/column_writer.h" #include "storage/segment/mock/mock_segment.h" #include "storage/segment/segment.h" #include "storage/segment/variant/variant_column_reader.h" #include "storage/tablet/tablet_schema.h" +#include "storage/types.h" #include "util/json/path_in_data.h" namespace doris::segment_v2 { +namespace { +class TestColumnIterator final : public ColumnIterator { +public: + Status seek_to_ordinal(ordinal_t /* ord */) override { return Status::OK(); } + + Status next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) override { + dst->insert_many_defaults(*n); + if (has_null != nullptr) { + *has_null = false; + } + return Status::OK(); + } + + Status read_by_rowids(const rowid_t* /* rowids */, const size_t count, + MutableColumnPtr& dst) override { + dst->insert_many_defaults(count); + return Status::OK(); + } + + ordinal_t get_current_ordinal() const override { return 0; } + + void force_set_read_requirement(ReadRequirement requirement) { + _read_requirement = requirement; + } + + using ColumnIterator::AccessPathSplit; + + Result split_access_paths(const TColumnAccessPaths& access_paths) const { + return _split_access_paths(access_paths); + } + + Status check_and_set_meta_read_mode(ReadRequirement requirement_before_access_path, + const TColumnAccessPaths& access_paths) { + auto split = DORIS_TRY(_split_access_paths(access_paths)); + return _check_and_set_meta_read_mode(requirement_before_access_path, split); + } + + void convert_to_place_holder_column(MutableColumnPtr& dst, size_t count) { + _convert_to_place_holder_column(dst, count); + } +}; + +TColumnAccessPath create_data_access_path(std::vector path) { + TColumnAccessPath access_path; + access_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + access_path.__set_type(TAccessPathType::DATA); + TDataAccessPath data_access_path; + data_access_path.__set_path(std::move(path)); + access_path.__set_data_access_path(std::move(data_access_path)); + return access_path; +} + +TColumnAccessPath create_legacy_data_access_path(std::vector path) { + TColumnAccessPath access_path; + access_path.__set_type(TAccessPathType::DATA); + TDataAccessPath data_access_path; + data_access_path.__set_path(std::move(path)); + access_path.__set_data_access_path(std::move(data_access_path)); + return access_path; +} + +TColumnAccessPath create_meta_access_path(std::vector path) { + TColumnAccessPath access_path; + access_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + access_path.__set_type(TAccessPathType::META); + TMetaAccessPath meta_access_path; + meta_access_path.__set_path(std::move(path)); + access_path.__set_meta_access_path(std::move(meta_access_path)); + return access_path; +} + +std::shared_ptr create_test_reader( + bool is_nullable = false, uint64_t num_rows = 0, + FieldType field_type = FieldType::OLAP_FIELD_TYPE_INT) { + auto reader = std::make_shared(); + reader->_meta_is_nullable = is_nullable; + reader->_num_rows = num_rows; + reader->_meta_type = field_type; + return reader; +} + +class TrackingColumnIterator final : public ColumnIterator { +public: + Status seek_to_ordinal(ordinal_t ord) override { + seek_ordinals.emplace_back(ord); + _current_ordinal = ord; + return Status::OK(); + } + + Status next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) override { + next_batch_sizes.emplace_back(*n); + if (!need_to_read()) { + _convert_to_place_holder_column(dst, *n); + if (has_null != nullptr) { + *has_null = false; + } + return Status::OK(); + } + + _recovery_from_place_holder_column(dst); + dst->insert_many_defaults(*n); + _current_ordinal += *n; + if (has_null != nullptr) { + *has_null = false; + } + return Status::OK(); + } + + Status read_by_rowids(const rowid_t* rowids, const size_t count, + MutableColumnPtr& dst) override { + read_by_rowids_batches.emplace_back(rowids, rowids + count); + if (!need_to_read()) { + _convert_to_place_holder_column(dst, count); + return Status::OK(); + } + + _recovery_from_place_holder_column(dst); + dst->insert_many_defaults(count); + return Status::OK(); + } + + ordinal_t get_current_ordinal() const override { return _current_ordinal; } + + Status set_access_paths(const TColumnAccessPaths& all_access_paths, + const TColumnAccessPaths& predicate_access_paths) override { + routed_all_access_paths = all_access_paths; + routed_predicate_access_paths = predicate_access_paths; + return ColumnIterator::set_access_paths(all_access_paths, predicate_access_paths); + } + + void collect_prefetchers( + std::map>& prefetchers, + PrefetcherInitMethod init_method) override { + record_collect_method(init_method); + prefetchers[init_method].emplace_back(prefetcher()); + } + + SegmentPrefetcher* prefetcher() const { + return reinterpret_cast(const_cast(this)); + } + + void clear_tracking() { + seek_ordinals.clear(); + next_batch_sizes.clear(); + read_by_rowids_batches.clear(); + collect_methods.clear(); + routed_all_access_paths.clear(); + routed_predicate_access_paths.clear(); + } + + std::vector seek_ordinals; + std::vector next_batch_sizes; + std::vector> read_by_rowids_batches; + std::vector collect_methods; + TColumnAccessPaths routed_all_access_paths; + TColumnAccessPaths routed_predicate_access_paths; + +private: + void record_collect_method(PrefetcherInitMethod init_method) { + collect_methods.emplace_back(init_method); + } + + ordinal_t _current_ordinal = 0; +}; + +class TrackingFileColumnIterator final : public FileColumnIterator { +public: + explicit TrackingFileColumnIterator(std::shared_ptr reader) + : FileColumnIterator(std::move(reader)) {} + + Status seek_to_ordinal(ordinal_t ord) override { + seek_ordinals.emplace_back(ord); + _current_ordinal = ord; + return Status::OK(); + } + + Status next_batch(size_t* n, MutableColumnPtr& dst, bool* has_null) override { + next_batch_sizes.emplace_back(*n); + dst->insert_many_defaults(*n); + _current_ordinal += *n; + if (has_null != nullptr) { + *has_null = false; + } + return Status::OK(); + } + + Status read_by_rowids(const rowid_t* rowids, const size_t count, + MutableColumnPtr& dst) override { + read_by_rowids_batches.emplace_back(rowids, rowids + count); + dst->insert_many_defaults(count); + return Status::OK(); + } + + ordinal_t get_current_ordinal() const override { return _current_ordinal; } + + void collect_prefetchers( + std::map>& prefetchers, + PrefetcherInitMethod init_method) override { + record_collect_method(init_method); + prefetchers[init_method].emplace_back(prefetcher()); + } + + SegmentPrefetcher* prefetcher() const { + return reinterpret_cast(const_cast(this)); + } + + std::vector seek_ordinals; + std::vector next_batch_sizes; + std::vector> read_by_rowids_batches; + std::vector collect_methods; + +private: + void record_collect_method(PrefetcherInitMethod init_method) { + collect_methods.emplace_back(init_method); + } + + ordinal_t _current_ordinal = 0; +}; + +class NullMapOnlyFileColumnIterator final : public FileColumnIterator { +public: + explicit NullMapOnlyFileColumnIterator(std::shared_ptr reader) + : FileColumnIterator(std::move(reader)) {} + + void force_null_map_only() { _meta_read_mode = MetaReadMode::NULL_MAP_ONLY; } +}; + +MutableColumnPtr create_int_struct_column(size_t field_count) { + Columns columns; + for (size_t i = 0; i < field_count; ++i) { + columns.emplace_back(ColumnInt32::create()); + } + return ColumnStruct::create(std::move(columns)); +} + +MutableColumnPtr create_nullable_int_struct_column(size_t field_count) { + return ColumnNullable::create(create_int_struct_column(field_count), ColumnUInt8::create()); +} + +MutableColumnPtr create_nullable_int_array_column() { + return ColumnNullable::create( + ColumnArray::create(ColumnInt32::create(), ColumnArray::ColumnOffsets::create()), + ColumnUInt8::create()); +} + +MutableColumnPtr create_nullable_int_map_column() { + return ColumnNullable::create(ColumnMap::create(ColumnInt32::create(), ColumnInt32::create(), + ColumnArray::ColumnOffsets::create()), + ColumnUInt8::create()); +} + +struct TrackingOffsetIterator { + OffsetFileColumnIteratorUPtr iterator; + TrackingFileColumnIterator* tracker = nullptr; +}; + +TrackingOffsetIterator create_tracking_offset_iterator() { + auto file_iterator = std::make_unique(create_test_reader()); + auto* tracker = file_iterator.get(); + return {std::make_unique(std::move(file_iterator)), tracker}; +} +} // namespace + +static const std::string COLUMN_READER_FILE_TEST_DIR = "./ut_dir/column_reader_test"; + class ColumnReaderTest : public ::testing::Test { protected: - void SetUp() override {} - void TearDown() override {} + void SetUp() override { + _old_disable_storage_page_cache = config::disable_storage_page_cache; + config::disable_storage_page_cache = true; + auto st = io::global_local_filesystem()->delete_directory(COLUMN_READER_FILE_TEST_DIR); + ASSERT_TRUE(st.ok()) << st.to_string(); + st = io::global_local_filesystem()->create_directory(COLUMN_READER_FILE_TEST_DIR); + ASSERT_TRUE(st.ok()) << st.to_string(); + } + + void TearDown() override { + EXPECT_TRUE( + io::global_local_filesystem()->delete_directory(COLUMN_READER_FILE_TEST_DIR).ok()); + config::disable_storage_page_cache = _old_disable_storage_page_cache; + } + +private: + bool _old_disable_storage_page_cache = false; }; +TEST_F(ColumnReaderTest, NullMapOnlyReadBySparseRowidsAcrossPages) { + ColumnMetaPB meta; + std::string fname = COLUMN_READER_FILE_TEST_DIR + "/null_map_only_sparse_rowids"; + auto fs = io::global_local_filesystem(); + + { + io::FileWriterPtr file_writer; + Status st = fs->create_file(fname, &file_writer); + ASSERT_TRUE(st.ok()) << st.to_string(); + + ColumnWriterOptions writer_opts; + writer_opts.meta = &meta; + writer_opts.meta->set_column_id(0); + writer_opts.meta->set_unique_id(0); + writer_opts.meta->set_type(static_cast(FieldType::OLAP_FIELD_TYPE_INT)); + writer_opts.meta->set_length(0); + writer_opts.meta->set_encoding(PLAIN_ENCODING); + writer_opts.meta->set_compression(segment_v2::CompressionTypePB::LZ4F); + writer_opts.meta->set_is_nullable(true); + writer_opts.data_page_size = sizeof(int32_t) * 2; + writer_opts.need_zone_map = false; + + TabletColumn column(FieldAggregationMethod::OLAP_FIELD_AGGREGATION_NONE, + FieldType::OLAP_FIELD_TYPE_INT); + std::unique_ptr writer; + st = ColumnWriter::create(writer_opts, &column, file_writer.get(), &writer); + ASSERT_TRUE(st.ok()) << st.to_string(); + st = writer->init(); + ASSERT_TRUE(st.ok()) << st.to_string(); + + for (int32_t i = 0; i < 6; ++i) { + st = writer->append(i == 2, &i); + ASSERT_TRUE(st.ok()) << st.to_string(); + } + + st = writer->finish(); + ASSERT_TRUE(st.ok()) << st.to_string(); + st = writer->write_data(); + ASSERT_TRUE(st.ok()) << st.to_string(); + st = writer->write_ordinal_index(); + ASSERT_TRUE(st.ok()) << st.to_string(); + st = file_writer->close(); + ASSERT_TRUE(st.ok()) << st.to_string(); + } + + io::FileReaderSPtr file_reader; + auto st = fs->open_file(fname, &file_reader); + ASSERT_TRUE(st.ok()) << st.to_string(); + + ColumnReaderOptions reader_opts; + std::shared_ptr reader; + st = ColumnReader::create(reader_opts, meta, 6, file_reader, &reader); + ASSERT_TRUE(st.ok()) << st.to_string(); + + NullMapOnlyFileColumnIterator iter(reader); + ColumnIteratorOptions iter_opts; + OlapReaderStatistics stats; + iter_opts.stats = &stats; + iter_opts.file_reader = file_reader.get(); + st = iter.init(iter_opts); + ASSERT_TRUE(st.ok()) << st.to_string(); + iter.force_null_map_only(); + + MutableColumnPtr dst = ColumnNullable::create(ColumnInt32::create(), ColumnUInt8::create()); + const rowid_t rowids[] = {0, 2}; + st = iter.read_by_rowids(rowids, std::size(rowids), dst); + ASSERT_TRUE(st.ok()) << st.to_string(); + + ASSERT_EQ(2, dst->size()); + const auto& nullable_col = assert_cast(*dst); + const auto& null_map = nullable_col.get_null_map_data(); + ASSERT_EQ(2, null_map.size()); + EXPECT_EQ(0, null_map[0]); + EXPECT_EQ(1, null_map[1]); + EXPECT_EQ(2, nullable_col.get_nested_column().size()); +} + TEST_F(ColumnReaderTest, StructAccessPaths) { auto create_struct_iterator = []() { auto null_reader = std::make_shared(); @@ -69,7 +437,7 @@ TEST_F(ColumnReaderTest, StructAccessPaths) { auto st = iterator->set_access_paths(TColumnAccessPaths {}, TColumnAccessPaths {}); ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); - ASSERT_EQ(iterator->_reading_flag, ColumnIterator::ReadingFlag::NORMAL_READING); + ASSERT_EQ(iterator->_read_requirement, ColumnIterator::ReadRequirement::NORMAL); TColumnAccessPaths all_access_paths; all_access_paths.emplace_back(); @@ -82,10 +450,10 @@ TEST_F(ColumnReaderTest, StructAccessPaths) { ASSERT_FALSE(st.ok()); // Only reading sub_col_1 - // sub_col_2 should be set to SKIP_READING - all_access_paths[0].data_access_path.path = {"self", "sub_col_1"}; + // sub_col_2 should be set to SKIP + all_access_paths[0] = create_data_access_path({"self", "sub_col_1"}); - predicate_access_paths[0].data_access_path.path = {"self", "sub_col_1"}; + predicate_access_paths[0] = create_data_access_path({"self", "sub_col_1"}); st = iterator->set_access_paths(all_access_paths, predicate_access_paths); // invalid name leads to error @@ -95,26 +463,2116 @@ TEST_F(ColumnReaderTest, StructAccessPaths) { // now column name is "self", should be ok st = iterator->set_access_paths(all_access_paths, predicate_access_paths); ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); - ASSERT_EQ(iterator->_reading_flag, ColumnIterator::ReadingFlag::READING_FOR_PREDICATE); + ASSERT_EQ(iterator->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); - ASSERT_EQ(iterator->_sub_column_iterators[0]->_reading_flag, - ColumnIterator::ReadingFlag::READING_FOR_PREDICATE); - ASSERT_EQ(iterator->_sub_column_iterators[1]->_reading_flag, - ColumnIterator::ReadingFlag::SKIP_READING); + ASSERT_EQ(iterator->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + ASSERT_EQ(iterator->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); // Reading all sub columns - all_access_paths[0].data_access_path.path = {"self"}; + all_access_paths[0] = create_data_access_path({"self"}); iterator = create_struct_iterator(); iterator->set_column_name("self"); st = iterator->set_access_paths(all_access_paths, predicate_access_paths); ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); - ASSERT_EQ(iterator->_reading_flag, ColumnIterator::ReadingFlag::READING_FOR_PREDICATE); + ASSERT_EQ(iterator->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + + ASSERT_EQ(iterator->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + ASSERT_EQ(iterator->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); +} + +TEST_F(ColumnReaderTest, ReadPhaseMatrix) { + TestColumnIterator iterator; + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::SKIP); + iterator.set_read_phase(ColumnIterator::ReadPhase::NORMAL); + EXPECT_FALSE(iterator.need_to_read()); + EXPECT_FALSE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_TRUE(iterator.need_to_read()); + EXPECT_TRUE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::NORMAL); + iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + EXPECT_FALSE(iterator.need_to_read()); + EXPECT_FALSE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_FALSE(iterator.need_to_read()); + EXPECT_FALSE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_TRUE(iterator.need_to_read()); + EXPECT_TRUE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + EXPECT_FALSE(iterator.need_to_read()); + EXPECT_FALSE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::NORMAL); + EXPECT_FALSE(iterator.need_to_read()); + EXPECT_FALSE(iterator.need_to_read_meta_columns()); + + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_TRUE(iterator.need_to_read()); + EXPECT_TRUE(iterator.need_to_read_meta_columns()); +} + +TEST_F(ColumnReaderTest, ReadRequirementPriorityAndLazyOutput) { + TestColumnIterator iterator; + + iterator.set_read_requirement(ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(iterator.read_requirement(), ColumnIterator::ReadRequirement::SKIP); + + iterator.set_lazy_output_requirement(); + EXPECT_EQ(iterator.read_requirement(), ColumnIterator::ReadRequirement::LAZY_OUTPUT); + + iterator.set_read_requirement(ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(iterator.read_requirement(), ColumnIterator::ReadRequirement::LAZY_OUTPUT); + + iterator.set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(iterator.read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + + iterator.set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(iterator.read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); +} + +TEST_F(ColumnReaderTest, MetaReadModePrefersOffsetOverNull) { + auto assert_meta_read_mode = [](TColumnAccessPaths access_paths, bool offset_only, + bool null_map_only) { + TestColumnIterator iterator; + iterator.set_column_name("self"); + auto st = iterator.check_and_set_meta_read_mode(ColumnIterator::ReadRequirement::NORMAL, + access_paths); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_EQ(iterator.read_offset_only(), offset_only); + EXPECT_EQ(iterator.read_null_map_only(), null_map_only); + }; + + assert_meta_read_mode(TColumnAccessPaths {}, false, false); + assert_meta_read_mode( + TColumnAccessPaths {create_meta_access_path({"self", ColumnIterator::ACCESS_OFFSET})}, + true, false); + assert_meta_read_mode( + TColumnAccessPaths {create_meta_access_path({"self", ColumnIterator::ACCESS_NULL})}, + false, true); + assert_meta_read_mode( + TColumnAccessPaths {create_meta_access_path({"self", ColumnIterator::ACCESS_OFFSET}), + create_meta_access_path({"self", ColumnIterator::ACCESS_NULL})}, + true, false); + assert_meta_read_mode(TColumnAccessPaths {create_data_access_path({"self", "child"})}, false, + false); + assert_meta_read_mode( + TColumnAccessPaths {create_data_access_path({"self", ColumnIterator::ACCESS_OFFSET})}, + false, false); + assert_meta_read_mode( + TColumnAccessPaths {create_data_access_path({"self", ColumnIterator::ACCESS_NULL})}, + false, false); + assert_meta_read_mode( + TColumnAccessPaths {create_meta_access_path({"self", ColumnIterator::ACCESS_OFFSET}), + create_data_access_path({"self", "child"})}, + false, false); + + { + TestColumnIterator iterator; + iterator.set_column_name("self"); + auto st = iterator.check_and_set_meta_read_mode( + ColumnIterator::ReadRequirement::LAZY_OUTPUT, + TColumnAccessPaths { + create_meta_access_path({"self", ColumnIterator::ACCESS_NULL})}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_FALSE(iterator.read_null_map_only()); + EXPECT_FALSE(iterator.read_offset_only()); + } +} + +TEST_F(ColumnReaderTest, TypedAccessPathRequiresMatchingPayload) { + FileColumnIterator iterator(create_test_reader(true)); + iterator.set_column_name("c"); + + TColumnAccessPath unknown_type_path; + unknown_type_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + TDataAccessPath data_access_path; + data_access_path.__set_path({"c"}); + unknown_type_path.__set_data_access_path(data_access_path); + auto st = iterator.set_access_paths({unknown_type_path}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("Invalid access path type")); + + TColumnAccessPath missing_meta_payload; + missing_meta_payload.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + missing_meta_payload.__set_type(TAccessPathType::META); + missing_meta_payload.__set_data_access_path(data_access_path); + st = iterator.set_access_paths({missing_meta_payload}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("meta_access_path payload is not set")); + + TColumnAccessPath missing_data_payload; + missing_data_payload.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + missing_data_payload.__set_type(TAccessPathType::DATA); + TMetaAccessPath meta_access_path; + meta_access_path.__set_path({"c", ColumnIterator::ACCESS_NULL}); + missing_data_payload.__set_meta_access_path(meta_access_path); + st = iterator.set_access_paths({missing_data_payload}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("data_access_path payload is not set")); + + TColumnAccessPath missing_legacy_payload; + missing_legacy_payload.__set_type(TAccessPathType::DATA); + st = iterator.set_access_paths({missing_legacy_payload}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), + ::testing::HasSubstr( + "Invalid legacy access path: data_access_path payload is not set")); + + for (const bool explicit_legacy_version : {false, true}) { + TColumnAccessPath invalid_legacy_type; + invalid_legacy_type.__set_type(TAccessPathType::META); + invalid_legacy_type.__set_data_access_path(data_access_path); + invalid_legacy_type.__set_meta_access_path(meta_access_path); + if (explicit_legacy_version) { + invalid_legacy_type.__set_version( + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + } + st = iterator.set_access_paths({invalid_legacy_type}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("Invalid legacy access path type")); + } + + FileColumnIterator data_precedence_iterator(create_test_reader(true)); + data_precedence_iterator.set_column_name("c"); + auto compatible_data_path = create_data_access_path({"c"}); + compatible_data_path.__set_meta_access_path(meta_access_path); + st = data_precedence_iterator.set_access_paths({compatible_data_path}, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_FALSE(data_precedence_iterator.read_null_map_only()); + EXPECT_FALSE(data_precedence_iterator.read_offset_only()); + + auto compatible_meta_path = create_meta_access_path({"c", ColumnIterator::ACCESS_NULL}); + compatible_meta_path.__set_data_access_path(data_access_path); + st = iterator.set_access_paths({compatible_meta_path}, {compatible_meta_path}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(iterator.read_null_map_only()); + + FileColumnIterator wrong_root_iterator(create_test_reader(true)); + wrong_root_iterator.set_column_name("c"); + auto wrong_root_path = create_meta_access_path({"other", ColumnIterator::ACCESS_NULL}); + st = wrong_root_iterator.set_access_paths({}, {wrong_root_path}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("expected name \"c\", got \"other\"")); + + auto unsupported_version_path = create_data_access_path({"c"}); + unsupported_version_path.__set_version( + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED + 1); + st = iterator.set_access_paths({unsupported_version_path}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("Unsupported access path version")); +} + +TEST_F(ColumnReaderTest, AccessPathVersionControlsLegacyMetaEncoding) { + for (const bool explicit_legacy_version : {false, true}) { + SCOPED_TRACE(explicit_legacy_version ? "explicit-version-0" : "missing-version"); + FileColumnIterator iterator(create_test_reader(true)); + iterator.set_column_name("c"); + auto legacy_null_path = create_legacy_data_access_path({"c", ColumnIterator::ACCESS_NULL}); + if (explicit_legacy_version) { + legacy_null_path.__set_version( + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + } + auto st = iterator.set_access_paths({legacy_null_path}, {legacy_null_path}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(iterator.read_null_map_only()); + } + + { + FileColumnIterator iterator(create_test_reader(true)); + iterator.set_column_name("c"); + auto typed_data_path = create_data_access_path({"c", ColumnIterator::ACCESS_NULL}); + auto st = iterator.set_access_paths({typed_data_path}, {typed_data_path}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_FALSE(iterator.read_null_map_only()); + EXPECT_FALSE(iterator.read_offset_only()); + } +} + +TEST_F(ColumnReaderTest, LegacyDataSpecialComponentsRemainDataSelectors) { + auto make_map_iterator = []() { + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + auto key_iterator = std::make_unique(create_test_reader()); + auto value_iterator = std::make_unique(create_test_reader()); + auto map_iterator = std::make_unique( + create_test_reader(), nullptr, std::move(offsets_iterator), std::move(key_iterator), + std::move(value_iterator)); + map_iterator->set_column_name("m"); + return map_iterator; + }; + + struct SelectorCase { + const char* component; + bool explicit_legacy_version; + ColumnIterator::ReadRequirement key_requirement; + ColumnIterator::ReadRequirement value_requirement; + }; + for (const auto& test_case : {SelectorCase {ColumnIterator::ACCESS_MAP_KEYS, false, + ColumnIterator::ReadRequirement::LAZY_OUTPUT, + ColumnIterator::ReadRequirement::SKIP}, + SelectorCase {ColumnIterator::ACCESS_MAP_VALUES, true, + ColumnIterator::ReadRequirement::SKIP, + ColumnIterator::ReadRequirement::LAZY_OUTPUT}}) { + SCOPED_TRACE(test_case.component); + auto map_iterator = make_map_iterator(); + auto path = create_legacy_data_access_path({"m", test_case.component}); + if (test_case.explicit_legacy_version) { + path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + } + + auto st = map_iterator->set_access_paths({path}, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_FALSE(map_iterator->read_null_map_only()); + EXPECT_FALSE(map_iterator->read_offset_only()); + EXPECT_EQ(map_iterator->_key_iterator->read_requirement(), test_case.key_requirement); + EXPECT_EQ(map_iterator->_val_iterator->read_requirement(), test_case.value_requirement); + } + + // `*` is also a legacy DATA selector. It keeps keys readable while OFFSET is promoted only + // after the remaining path reaches the value iterator. + auto map_iterator = make_map_iterator(); + auto offset_path = create_legacy_data_access_path( + {"m", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET}); + offset_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + auto st = map_iterator->set_access_paths({offset_path}, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_EQ(map_iterator->_key_iterator->read_requirement(), + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_TRUE(map_iterator->_val_iterator->read_offset_only()); +} - ASSERT_EQ(iterator->_sub_column_iterators[0]->_reading_flag, - ColumnIterator::ReadingFlag::READING_FOR_PREDICATE); - ASSERT_EQ(iterator->_sub_column_iterators[1]->_reading_flag, - ColumnIterator::ReadingFlag::NEED_TO_READ); +TEST_F(ColumnReaderTest, TypedMetaPathsRouteThroughScalarAndComplexIterators) { + { + FileColumnIterator scalar_iterator(create_test_reader(true)); + scalar_iterator.set_column_name("i"); + TColumnAccessPaths null_path {create_meta_access_path({"i", ColumnIterator::ACCESS_NULL})}; + auto st = scalar_iterator.set_access_paths(null_path, null_path); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(scalar_iterator.read_null_map_only()); + } + + { + auto item_iterator = std::make_unique(create_test_reader()); + item_iterator->set_column_name("item"); + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + ArrayFileColumnIterator array_iterator(create_test_reader(), std::move(offsets_iterator), + std::move(item_iterator), nullptr); + array_iterator.set_column_name("a"); + + TColumnAccessPaths offset_path { + create_meta_access_path({"a", ColumnIterator::ACCESS_OFFSET})}; + auto st = array_iterator.set_access_paths(offset_path, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(array_iterator.read_offset_only()); + EXPECT_EQ(array_iterator._item_iterator->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); + } + + { + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + auto key_iterator = std::make_unique(create_test_reader()); + auto value_iterator = std::make_unique(create_test_reader()); + MapFileColumnIterator map_iterator(create_test_reader(), nullptr, + std::move(offsets_iterator), std::move(key_iterator), + std::move(value_iterator)); + map_iterator.set_column_name("m"); + + TColumnAccessPaths offset_path { + create_meta_access_path({"m", ColumnIterator::ACCESS_OFFSET})}; + auto st = map_iterator.set_access_paths(offset_path, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(map_iterator.read_offset_only()); + EXPECT_EQ(map_iterator._key_iterator->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(map_iterator._val_iterator->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); + } + + { + std::vector sub_iterators; + auto string_iterator = std::make_unique(create_test_reader()); + string_iterator->set_column_name("text"); + sub_iterators.emplace_back(std::move(string_iterator)); + StructFileColumnIterator struct_iterator(create_test_reader(), nullptr, + std::move(sub_iterators)); + struct_iterator.set_column_name("s"); + + TColumnAccessPaths offset_path { + create_meta_access_path({"s", "text", ColumnIterator::ACCESS_OFFSET})}; + auto st = struct_iterator.set_access_paths(offset_path, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + auto* text_iterator = static_cast( + struct_iterator._sub_column_iterators[0].get()); + EXPECT_TRUE(text_iterator->read_offset_only()); + } + + { + auto item_iterator = std::make_unique(create_test_reader()); + item_iterator->set_column_name("item"); + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + ArrayFileColumnIterator array_iterator(create_test_reader(), std::move(offsets_iterator), + std::move(item_iterator), nullptr); + array_iterator.set_column_name("a"); + + TColumnAccessPaths offset_path {create_meta_access_path( + {"a", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET})}; + auto st = array_iterator.set_access_paths(offset_path, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + auto* typed_item_iterator = + static_cast(array_iterator._item_iterator.get()); + EXPECT_TRUE(typed_item_iterator->read_offset_only()); + } + + { + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + auto key_iterator = std::make_unique(create_test_reader()); + auto value_iterator = std::make_unique(create_test_reader()); + MapFileColumnIterator map_iterator(create_test_reader(), nullptr, + std::move(offsets_iterator), std::move(key_iterator), + std::move(value_iterator)); + map_iterator.set_column_name("m"); + + TColumnAccessPaths offset_path {create_meta_access_path( + {"m", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET})}; + auto st = map_iterator.set_access_paths(offset_path, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + auto* typed_key_iterator = + static_cast(map_iterator._key_iterator.get()); + auto* typed_value_iterator = + static_cast(map_iterator._val_iterator.get()); + EXPECT_FALSE(typed_key_iterator->read_offset_only()); + EXPECT_EQ(typed_key_iterator->read_requirement(), + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_TRUE(typed_value_iterator->read_offset_only()); + } +} + +TEST_F(ColumnReaderTest, TypedDataFieldsNamedMetaComponentsAreNotTreatedAsMetaPaths) { + auto make_struct_iterator = [](const std::string& field_name) { + std::vector sub_iterators; + auto field_iterator = std::make_unique(create_test_reader()); + field_iterator->set_column_name(field_name); + sub_iterators.emplace_back(std::move(field_iterator)); + auto struct_iterator = std::make_unique( + create_test_reader(), nullptr, std::move(sub_iterators)); + struct_iterator->set_column_name("s"); + return struct_iterator; + }; + + for (const std::string field_name : + {ColumnIterator::ACCESS_NULL, ColumnIterator::ACCESS_OFFSET}) { + SCOPED_TRACE(field_name); + auto struct_iterator = make_struct_iterator(field_name); + TColumnAccessPaths data_path {create_data_access_path({"s", field_name})}; + auto st = struct_iterator->set_access_paths(data_path, data_path); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_FALSE(struct_iterator->read_null_map_only()); + EXPECT_FALSE(struct_iterator->read_offset_only()); + ASSERT_EQ(struct_iterator->_sub_column_iterators.size(), 1); + EXPECT_EQ(struct_iterator->_sub_column_iterators[0]->read_requirement(), + ColumnIterator::ReadRequirement::PREDICATE); + } + + auto struct_iterator = make_struct_iterator(ColumnIterator::ACCESS_NULL); + TColumnAccessPaths meta_path {create_meta_access_path({"s", ColumnIterator::ACCESS_NULL})}; + auto st = struct_iterator->set_access_paths(meta_path, meta_path); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(struct_iterator->read_null_map_only()); + EXPECT_EQ(struct_iterator->_sub_column_iterators[0]->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); +} + +TEST_F(ColumnReaderTest, LegacyStructMetaComponentsRemainSentinels) { + auto null_iterator = std::make_unique(create_test_reader()); + std::vector sub_iterators; + auto field_iterator = std::make_unique(create_test_reader()); + field_iterator->set_column_name(ColumnIterator::ACCESS_NULL); + sub_iterators.emplace_back(std::move(field_iterator)); + StructFileColumnIterator struct_iterator(create_test_reader(), std::move(null_iterator), + std::move(sub_iterators)); + struct_iterator.set_column_name("s"); + + auto legacy_path = create_legacy_data_access_path({"s", ColumnIterator::ACCESS_NULL}); + legacy_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + auto st = struct_iterator.set_access_paths({legacy_path}, {legacy_path}); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(struct_iterator.read_null_map_only()); + EXPECT_EQ(struct_iterator._sub_column_iterators[0]->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); +} + +TEST_F(ColumnReaderTest, PlaceHolderLifecycleInLazyMode) { + TestColumnIterator iterator; + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + + MutableColumnPtr dst = ColumnInt32::create(); + iterator.convert_to_place_holder_column(dst, 3); + + EXPECT_EQ(3, dst->size()); + EXPECT_TRUE(iterator._has_place_holder_column); + + iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + iterator.finalize_lazy_phase(dst); + EXPECT_EQ(0, dst->size()); + EXPECT_FALSE(iterator._has_place_holder_column); + + MutableColumnPtr lazy_dst = ColumnInt32::create(); + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + iterator.convert_to_place_holder_column(lazy_dst, 4); + EXPECT_EQ(0, lazy_dst->size()); +} + +TEST_F(ColumnReaderTest, PlaceHolderRecoveryAfterColumnReplacement) { + TestColumnIterator iterator; + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + + MutableColumnPtr dst = ColumnInt32::create(); + iterator.convert_to_place_holder_column(dst, 3); + EXPECT_TRUE(iterator._has_place_holder_column); + + IColumn::Filter filter; + filter.resize(3); + filter[0] = 1; + filter[1] = 0; + filter[2] = 1; + dst = IColumn::mutate(dst->filter(filter, 2)); + EXPECT_EQ(2, dst->size()); + + iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + iterator.finalize_lazy_phase(dst); + EXPECT_EQ(0, dst->size()); + EXPECT_FALSE(iterator._has_place_holder_column); + + dst->insert_many_defaults(2); + iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + iterator.finalize_lazy_phase(dst); + EXPECT_EQ(2, dst->size()); +} + +TEST_F(ColumnReaderTest, SetReadRequirementPropagatesToNestedIterators) { + auto null_iter = std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + + auto sub_col = std::make_unique(std::make_shared()); + sub_col->set_column_name("sub_col"); + struct_sub_iters.emplace_back(std::move(sub_col)); + + auto array_item = std::make_unique(std::make_shared()); + array_item->set_column_name("item"); + auto array_offsets = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null = std::make_unique(std::make_shared()); + auto array_iter = std::make_unique( + std::make_shared(), std::move(array_offsets), std::move(array_item), + std::move(array_null)); + array_iter->set_column_name("arr"); + struct_sub_iters.emplace_back(std::move(array_iter)); + + StructFileColumnIterator struct_iter(std::make_shared(), std::move(null_iter), + std::move(struct_sub_iters)); + struct_iter.set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + + EXPECT_EQ(struct_iter.read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iter._sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* nested_array = + static_cast(struct_iter._sub_column_iterators[1].get()); + EXPECT_EQ(nested_array->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(nested_array->_item_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = std::make_unique(std::make_shared()); + MapFileColumnIterator map_iter(std::make_shared(), std::move(map_null_iter), + std::move(map_offsets_iter), std::move(map_key_iter), + std::move(map_val_iter)); + map_iter.set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iter._key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iter._val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); +} + +TEST_F(ColumnReaderTest, SetReadRequirementSelfKeepsNestedIteratorRequirements) { + auto null_iter = std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + auto sub_iter = std::make_unique(std::make_shared()); + sub_iter->set_column_name("sub_col"); + struct_sub_iters.emplace_back(std::move(sub_iter)); + + StructFileColumnIterator struct_iter(std::make_shared(), std::move(null_iter), + std::move(struct_sub_iters)); + struct_iter.set_read_requirement_self(ColumnIterator::ReadRequirement::PREDICATE); + + EXPECT_EQ(struct_iter.read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iter._sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::NORMAL); +} + +TEST_F(ColumnReaderTest, RemovePrunedSubIterators) { + auto struct_null_iter = std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + auto sub_keep = std::make_unique(std::make_shared()); + sub_keep->set_column_name("keep"); + auto sub_prune = std::make_unique(std::make_shared()); + sub_prune->set_column_name("prune"); + sub_prune->set_read_requirement(ColumnIterator::ReadRequirement::SKIP); + struct_sub_iters.emplace_back(std::move(sub_keep)); + struct_sub_iters.emplace_back(std::move(sub_prune)); + + auto array_item_null = std::make_unique(std::make_shared()); + std::vector item_struct_sub_iters; + auto item_keep = std::make_unique(std::make_shared()); + item_keep->set_column_name("keep"); + auto item_prune = std::make_unique(std::make_shared()); + item_prune->set_column_name("prune"); + item_prune->set_read_requirement(ColumnIterator::ReadRequirement::SKIP); + item_struct_sub_iters.emplace_back(std::move(item_keep)); + item_struct_sub_iters.emplace_back(std::move(item_prune)); + auto item_struct = std::make_unique(std::make_shared(), + std::move(array_item_null), + std::move(item_struct_sub_iters)); + + auto array_offsets = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null = std::make_unique(std::make_shared()); + auto array_iter = std::make_unique( + std::make_shared(), std::move(array_offsets), std::move(item_struct), + std::move(array_null)); + struct_sub_iters.emplace_back(std::move(array_iter)); + + StructFileColumnIterator struct_iter(std::make_shared(), + std::move(struct_null_iter), std::move(struct_sub_iters)); + ASSERT_EQ(3, struct_iter._sub_column_iterators.size()); + struct_iter.remove_pruned_sub_iterators(); + ASSERT_EQ(2, struct_iter._sub_column_iterators.size()); + + auto* nested_array = + static_cast(struct_iter._sub_column_iterators[1].get()); + auto* nested_struct = + static_cast(nested_array->_item_iterator.get()); + ASSERT_EQ(1, nested_struct->_sub_column_iterators.size()); + EXPECT_EQ(nested_struct->_sub_column_iterators[0]->column_name(), "keep"); +} + +TEST_F(ColumnReaderTest, FinalizeLazyModeOnNestedStruct) { + auto sub_iter = std::make_unique(); + auto* sub_iter_ptr = sub_iter.get(); + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + sub_iters.emplace_back(std::move(sub_iter)); + + StructFileColumnIterator struct_iter(std::make_shared(), std::move(null_iter), + std::move(sub_iters)); + sub_iter_ptr->set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + struct_iter.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + sub_iter_ptr->set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + + MutableColumnPtr nested_column = ColumnInt32::create(); + MutableColumnPtr nested_mut = IColumn::mutate(std::move(nested_column)); + sub_iter_ptr->convert_to_place_holder_column(nested_mut, 5); + EXPECT_EQ(5, nested_mut->size()); + + Columns struct_columns; + struct_columns.emplace_back(std::move(nested_mut)); + auto struct_column = ColumnStruct::create(struct_columns); + MutableColumnPtr struct_mut = std::move(struct_column); + struct_iter.set_read_phase(ColumnIterator::ReadPhase::LAZY); + sub_iter_ptr->set_read_phase(ColumnIterator::ReadPhase::LAZY); + struct_iter.finalize_lazy_phase(struct_mut); + + auto& column_struct = assert_cast(*struct_mut); + auto nested_after = column_struct.get_column_ptr(0); + EXPECT_EQ(0, nested_after->size()); +} + +TEST_F(ColumnReaderTest, SplitAccessPathsClassifiesCurrentAndDescendantPaths) { + TestColumnIterator iterator; + iterator.set_column_name("self"); + iterator.force_set_read_requirement(ColumnIterator::ReadRequirement::NORMAL); + + { + auto split = TEST_TRY(iterator.split_access_paths( + TColumnAccessPaths {create_data_access_path({"self"})})); + EXPECT_TRUE(split.reads_current_data); + EXPECT_EQ(split.current_meta_mode, ColumnIterator::MetaReadMode::DEFAULT); + EXPECT_TRUE(split.descendant_paths.empty()); + } + + { + auto split = TEST_TRY(iterator.split_access_paths(TColumnAccessPaths { + create_meta_access_path({"self", ColumnIterator::ACCESS_NULL}), + create_meta_access_path({"self", ColumnIterator::ACCESS_OFFSET})})); + EXPECT_FALSE(split.reads_current_data); + EXPECT_EQ(split.current_meta_mode, ColumnIterator::MetaReadMode::OFFSET_ONLY); + EXPECT_TRUE(split.descendant_paths.empty()); + } + + { + auto split = TEST_TRY(iterator.split_access_paths(TColumnAccessPaths { + create_data_access_path({"self", "child"}), + create_meta_access_path({"self", "child", ColumnIterator::ACCESS_NULL})})); + EXPECT_FALSE(split.reads_current_data); + EXPECT_EQ(split.current_meta_mode, ColumnIterator::MetaReadMode::DEFAULT); + ASSERT_EQ(split.descendant_paths.size(), 2); + EXPECT_EQ(split.descendant_paths[0].type, TAccessPathType::DATA); + EXPECT_EQ(split.descendant_paths[0].data_access_path.path, + (std::vector {"child"})); + EXPECT_EQ(split.descendant_paths[1].type, TAccessPathType::META); + EXPECT_EQ(split.descendant_paths[1].meta_access_path.path, + (std::vector {"child", ColumnIterator::ACCESS_NULL})); + } + + { + auto legacy_null = create_legacy_data_access_path({"self", ColumnIterator::ACCESS_NULL}); + auto legacy_offset = + create_legacy_data_access_path({"self", ColumnIterator::ACCESS_OFFSET}); + legacy_offset.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + auto split = TEST_TRY(iterator.split_access_paths( + TColumnAccessPaths {std::move(legacy_null), std::move(legacy_offset)})); + EXPECT_FALSE(split.reads_current_data); + EXPECT_EQ(split.current_meta_mode, ColumnIterator::MetaReadMode::OFFSET_ONLY); + EXPECT_TRUE(split.descendant_paths.empty()); + } + + { + auto legacy_keys = + create_legacy_data_access_path({"self", ColumnIterator::ACCESS_MAP_KEYS}); + auto split = TEST_TRY(iterator.split_access_paths( + TColumnAccessPaths {create_data_access_path({"self", ColumnIterator::ACCESS_NULL}), + std::move(legacy_keys)})); + EXPECT_FALSE(split.reads_current_data); + EXPECT_EQ(split.current_meta_mode, ColumnIterator::MetaReadMode::DEFAULT); + ASSERT_EQ(split.descendant_paths.size(), 2); + EXPECT_EQ(split.descendant_paths[0].data_access_path.path, + (std::vector {ColumnIterator::ACCESS_NULL})); + EXPECT_EQ(split.descendant_paths[1].data_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_KEYS})); + } + + // Splitting is a pure classification step. Callers explicitly apply lazy/predicate state. + EXPECT_EQ(iterator._read_requirement, ColumnIterator::ReadRequirement::NORMAL); +} + +TEST_F(ColumnReaderTest, MapRejectsUnrecognizedDescendantSelectors) { + auto make_map_iterator = []() { + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + auto map_iterator = std::make_unique( + create_test_reader(), nullptr, std::move(offsets_iterator), + std::make_unique(), + std::make_unique()); + map_iterator->set_column_name("m"); + return map_iterator; + }; + + std::vector invalid_paths; + invalid_paths.emplace_back(create_data_access_path({"m", "UNKNOWN"})); + invalid_paths.emplace_back( + create_meta_access_path({"m", "UNKNOWN", ColumnIterator::ACCESS_NULL})); + invalid_paths.emplace_back(create_legacy_data_access_path({"m", "UNKNOWN"})); + + for (const auto& path : invalid_paths) { + auto map_iterator = make_map_iterator(); + auto st = map_iterator->set_access_paths({path}, {}); + EXPECT_FALSE(st.ok()); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("Invalid map access path selector")); + EXPECT_THAT(st.to_string(), ::testing::HasSubstr("UNKNOWN")); + } + + auto path = create_meta_access_path( + {"m", ColumnIterator::ACCESS_MAP_VALUES, ColumnIterator::ACCESS_OFFSET}); + TDataAccessPath unselected_data_payload; + unselected_data_payload.__set_path({"m", "UNKNOWN"}); + path.__set_data_access_path(std::move(unselected_data_payload)); + auto map_iterator = make_map_iterator(); + auto st = map_iterator->set_access_paths({path}, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); +} + +TEST_F(ColumnReaderTest, MapChildPathRoutingUsesLogicalSelectorsAndPreservesVersion) { + auto key_iterator = std::make_unique(); + key_iterator->set_column_name("physical_key"); + auto* key = key_iterator.get(); + auto value_iterator = std::make_unique(); + value_iterator->set_column_name("physical_value"); + auto* value = value_iterator.get(); + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + MapFileColumnIterator map_iterator(create_test_reader(), nullptr, std::move(offsets_iterator), + std::move(key_iterator), std::move(value_iterator)); + map_iterator.set_column_name("m"); + EXPECT_EQ(key->column_name(), ColumnIterator::ACCESS_MAP_KEYS); + EXPECT_EQ(value->column_name(), ColumnIterator::ACCESS_MAP_VALUES); + + auto path = create_meta_access_path( + {"m", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET}); + TDataAccessPath unselected_data_payload; + unselected_data_payload.__set_path(path.meta_access_path.path); + path.__set_data_access_path(std::move(unselected_data_payload)); + auto legacy_path = create_data_access_path( + {"m", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET}); + legacy_path.__set_version(g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + + auto st = map_iterator.set_access_paths({path}, {legacy_path}); + ASSERT_TRUE(st.ok()) << st.to_string(); + + ASSERT_EQ(key->routed_all_access_paths.size(), 1); + const auto& key_path = key->routed_all_access_paths[0]; + EXPECT_EQ(key_path.type, TAccessPathType::DATA); + ASSERT_TRUE(key_path.__isset.version); + EXPECT_EQ(key_path.version, g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + EXPECT_EQ(key_path.data_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_KEYS})); + ASSERT_EQ(key->routed_predicate_access_paths.size(), 1); + const auto& legacy_key_path = key->routed_predicate_access_paths[0]; + EXPECT_EQ(legacy_key_path.type, TAccessPathType::DATA); + ASSERT_TRUE(legacy_key_path.__isset.version); + EXPECT_EQ(legacy_key_path.version, g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + EXPECT_EQ(legacy_key_path.data_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_KEYS})); + + ASSERT_EQ(value->routed_all_access_paths.size(), 1); + const auto& value_path = value->routed_all_access_paths[0]; + EXPECT_EQ(value_path.type, TAccessPathType::META); + ASSERT_TRUE(value_path.__isset.version); + EXPECT_EQ(value_path.version, g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + EXPECT_EQ(value_path.meta_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_VALUES, + ColumnIterator::ACCESS_OFFSET})); + EXPECT_EQ(value_path.data_access_path.path, + (std::vector {"m", ColumnIterator::ACCESS_ALL, + ColumnIterator::ACCESS_OFFSET})); + ASSERT_EQ(value->routed_predicate_access_paths.size(), 1); + const auto& legacy_value_path = value->routed_predicate_access_paths[0]; + EXPECT_EQ(legacy_value_path.type, TAccessPathType::DATA); + ASSERT_TRUE(legacy_value_path.__isset.version); + EXPECT_EQ(legacy_value_path.version, + g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_LEGACY); + EXPECT_EQ(legacy_value_path.data_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_VALUES, + ColumnIterator::ACCESS_OFFSET})); +} + +TEST_F(ColumnReaderTest, NestedMapWildcardRoutingUsesLogicalSelectorsWithoutPhysicalNames) { + auto inner_key_iterator = std::make_unique(); + auto* inner_key = inner_key_iterator.get(); + auto inner_value_iterator = std::make_unique(); + auto* inner_value = inner_value_iterator.get(); + auto inner_offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + auto inner_map_iterator = std::make_unique( + create_test_reader(), nullptr, std::move(inner_offsets_iterator), + std::move(inner_key_iterator), std::move(inner_value_iterator)); + + auto outer_key_iterator = std::make_unique(); + auto* outer_key = outer_key_iterator.get(); + auto outer_offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + MapFileColumnIterator outer_map_iterator( + create_test_reader(), nullptr, std::move(outer_offsets_iterator), + std::move(outer_key_iterator), std::move(inner_map_iterator)); + outer_map_iterator.set_column_name("m"); + + auto path = + create_meta_access_path({"m", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_ALL, + ColumnIterator::ACCESS_OFFSET}); + auto st = outer_map_iterator.set_access_paths({path}, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + + ASSERT_EQ(outer_key->routed_all_access_paths.size(), 1); + EXPECT_EQ(outer_key->routed_all_access_paths[0].data_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_KEYS})); + ASSERT_EQ(inner_key->routed_all_access_paths.size(), 1); + EXPECT_EQ(inner_key->routed_all_access_paths[0].data_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_KEYS})); + ASSERT_EQ(inner_value->routed_all_access_paths.size(), 1); + EXPECT_EQ(inner_value->routed_all_access_paths[0].meta_access_path.path, + (std::vector {ColumnIterator::ACCESS_MAP_VALUES, + ColumnIterator::ACCESS_OFFSET})); +} + +TEST_F(ColumnReaderTest, ArrayItemPathRoutingRewritesOnlySelectedPayload) { + auto item_iterator = std::make_unique(); + item_iterator->set_column_name("item"); + auto* item = item_iterator.get(); + auto offsets_iterator = std::make_unique( + std::make_unique(create_test_reader())); + ArrayFileColumnIterator array_iterator(create_test_reader(), std::move(offsets_iterator), + std::move(item_iterator), nullptr); + array_iterator.set_column_name("a"); + + auto path = create_meta_access_path( + {"a", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET}); + TDataAccessPath unselected_data_payload; + unselected_data_payload.__set_path(path.meta_access_path.path); + path.__set_data_access_path(std::move(unselected_data_payload)); + + auto st = array_iterator.set_access_paths({path}, {}); + ASSERT_TRUE(st.ok()) << st.to_string(); + + ASSERT_EQ(item->routed_all_access_paths.size(), 1); + const auto& item_path = item->routed_all_access_paths[0]; + EXPECT_EQ(item_path.type, TAccessPathType::META); + ASSERT_TRUE(item_path.__isset.version); + EXPECT_EQ(item_path.version, g_Descriptors_constants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + EXPECT_EQ(item_path.meta_access_path.path, + (std::vector {"item", ColumnIterator::ACCESS_OFFSET})); + EXPECT_EQ(item_path.data_access_path.path, + (std::vector {"a", ColumnIterator::ACCESS_ALL, + ColumnIterator::ACCESS_OFFSET})); +} + +TEST_F(ColumnReaderTest, StructCurrentMetaDoesNotRouteToDataFieldWithSameName) { + std::vector sub_iterators; + auto null_field_iterator = std::make_unique(); + null_field_iterator->set_column_name(ColumnIterator::ACCESS_NULL); + auto* null_field = null_field_iterator.get(); + sub_iterators.emplace_back(std::move(null_field_iterator)); + auto data_field_iterator = std::make_unique(); + data_field_iterator->set_column_name("data"); + auto* data_field = data_field_iterator.get(); + sub_iterators.emplace_back(std::move(data_field_iterator)); + + StructFileColumnIterator struct_iterator(create_test_reader(), nullptr, + std::move(sub_iterators)); + struct_iterator.set_column_name("s"); + TColumnAccessPaths all_access_paths { + create_meta_access_path({"s", ColumnIterator::ACCESS_NULL}), + create_data_access_path({"s", "data"})}; + TColumnAccessPaths predicate_access_paths {create_data_access_path({"s", "data"})}; + + auto st = struct_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << st.to_string(); + EXPECT_TRUE(null_field->routed_all_access_paths.empty()); + EXPECT_TRUE(null_field->routed_predicate_access_paths.empty()); + EXPECT_EQ(null_field->read_requirement(), ColumnIterator::ReadRequirement::SKIP); + ASSERT_EQ(data_field->routed_all_access_paths.size(), 1); + ASSERT_EQ(data_field->routed_predicate_access_paths.size(), 1); + EXPECT_EQ(data_field->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); +} + +TEST_F(ColumnReaderTest, NestedIteratorsPropagateReadPhase) { + auto struct_null_iterator = + std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + struct_sub_iters.emplace_back( + std::make_unique(std::make_shared())); + struct_sub_iters.emplace_back( + std::make_unique(std::make_shared())); + auto struct_iterator = std::make_unique( + std::make_shared(), std::move(struct_null_iterator), + std::move(struct_sub_iters)); + + struct_iterator->set_read_phase(ColumnIterator::ReadPhase::LAZY); + EXPECT_EQ(struct_iterator->_sub_column_iterators[0]->_read_phase, + ColumnIterator::ReadPhase::LAZY); + EXPECT_EQ(struct_iterator->_sub_column_iterators[1]->_read_phase, + ColumnIterator::ReadPhase::LAZY); + + auto array_item_iterator = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null_iter = std::make_unique(std::make_shared()); + ArrayFileColumnIterator array_iterator( + std::make_shared(), std::move(array_offsets_iter), + std::move(array_item_iterator), std::move(array_null_iter)); + array_iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + EXPECT_EQ(array_iterator._item_iterator->_read_phase, ColumnIterator::ReadPhase::PREDICATE); + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = std::make_unique(std::make_shared()); + MapFileColumnIterator map_iterator(std::make_shared(), std::move(map_null_iter), + std::move(map_offsets_iter), std::move(map_key_iter), + std::move(map_val_iter)); + map_iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + EXPECT_EQ(map_iterator._key_iterator->_read_phase, ColumnIterator::ReadPhase::LAZY); + EXPECT_EQ(map_iterator._val_iterator->_read_phase, ColumnIterator::ReadPhase::LAZY); +} + +TEST_F(ColumnReaderTest, AccessPathsPropagatePredicateToChildren) { + auto struct_null_iterator = + std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + struct_sub_iters.emplace_back( + std::make_unique(std::make_shared())); + struct_sub_iters.emplace_back( + std::make_unique(std::make_shared())); + auto struct_iterator = std::make_unique( + std::make_shared(), std::move(struct_null_iterator), + std::move(struct_sub_iters)); + struct_iterator->set_column_name("s"); + + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"s"}); + TColumnAccessPaths predicate_access_paths = all_access_paths; + + auto st = struct_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set struct access paths: " << st.to_string(); + EXPECT_EQ(struct_iterator->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iterator->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iterator->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto array_item_iterator = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null_iter = std::make_unique(std::make_shared()); + ArrayFileColumnIterator array_iterator( + std::make_shared(), std::move(array_offsets_iter), + std::move(array_item_iterator), std::move(array_null_iter)); + array_iterator.set_column_name("a"); + TColumnAccessPaths array_access_paths; + array_access_paths.emplace_back(); + array_access_paths[0] = create_data_access_path({"a"}); + TColumnAccessPaths array_predicate_paths = array_access_paths; + st = array_iterator.set_access_paths(array_access_paths, array_predicate_paths); + ASSERT_TRUE(st.ok()) << "failed to set array access paths: " << st.to_string(); + EXPECT_EQ(array_iterator._read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(array_iterator._item_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = std::make_unique(std::make_shared()); + MapFileColumnIterator map_iterator(std::make_shared(), std::move(map_null_iter), + std::move(map_offsets_iter), std::move(map_key_iter), + std::move(map_val_iter)); + map_iterator.set_column_name("m"); + TColumnAccessPaths map_access_paths; + map_access_paths.emplace_back(); + map_access_paths[0] = create_data_access_path({"m"}); + TColumnAccessPaths map_predicate_paths = map_access_paths; + st = map_iterator.set_access_paths(map_access_paths, map_predicate_paths); + ASSERT_TRUE(st.ok()) << "failed to set map access paths: " << st.to_string(); + EXPECT_EQ(map_iterator._read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iterator._key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iterator._val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); +} + +TEST_F(ColumnReaderTest, StructPredicateOnlyChildPathStillRoutesToChild) { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_a = std::make_unique(std::make_shared()); + sub_a->set_column_name("a"); + auto sub_b = std::make_unique(std::make_shared()); + sub_b->set_column_name("b"); + sub_iters.emplace_back(std::move(sub_a)); + sub_iters.emplace_back(std::move(sub_b)); + + StructFileColumnIterator struct_iterator(std::make_shared(), std::move(null_iter), + std::move(sub_iters)); + struct_iterator.set_column_name("s"); + + TColumnAccessPaths all_access_paths {create_data_access_path({"s", "a"})}; + TColumnAccessPaths predicate_access_paths {create_data_access_path({"s", "b"})}; + + auto st = struct_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set struct access paths: " << st.to_string(); + + EXPECT_EQ(struct_iterator._read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iterator._sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(struct_iterator._sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + struct_iterator.remove_pruned_sub_iterators(); + ASSERT_EQ(struct_iterator._sub_column_iterators.size(), 2); + EXPECT_EQ(struct_iterator._sub_column_iterators[0]->column_name(), "a"); + EXPECT_EQ(struct_iterator._sub_column_iterators[1]->column_name(), "b"); +} + +TEST_F(ColumnReaderTest, LegacyCurrentLevelPredicateNullPathUsesMetaOnlyMode) { + auto make_struct_iterator = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_a = std::make_unique(std::make_shared()); + sub_a->set_column_name("a"); + auto sub_b = std::make_unique(std::make_shared()); + sub_b->set_column_name("b"); + sub_iters.emplace_back(std::move(sub_a)); + sub_iters.emplace_back(std::move(sub_b)); + + auto struct_iterator = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + struct_iterator->set_column_name("s"); + return struct_iterator; + }; + + { + auto struct_iterator = make_struct_iterator(); + TColumnAccessPaths all_access_paths { + create_legacy_data_access_path({"s", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_legacy_data_access_path({"s", ColumnIterator::ACCESS_NULL})}; + + auto st = struct_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set struct access paths: " << st.to_string(); + + EXPECT_TRUE(struct_iterator->read_null_map_only()); + EXPECT_EQ(struct_iterator->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(struct_iterator->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + + struct_iterator->remove_pruned_sub_iterators(); + EXPECT_TRUE(struct_iterator->_sub_column_iterators.empty()); + } + + { + auto struct_iterator = make_struct_iterator(); + TColumnAccessPaths all_access_paths { + create_legacy_data_access_path({"s"}), + create_legacy_data_access_path({"s", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_legacy_data_access_path({"s", ColumnIterator::ACCESS_NULL})}; + + auto st = struct_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set struct access paths: " << st.to_string(); + + EXPECT_FALSE(struct_iterator->read_null_map_only()); + EXPECT_EQ(struct_iterator->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(struct_iterator->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + } + + { + auto array_item_iterator = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null_iter = + std::make_unique(std::make_shared()); + ArrayFileColumnIterator array_iterator( + std::make_shared(), std::move(array_offsets_iter), + std::move(array_item_iterator), std::move(array_null_iter)); + array_iterator.set_column_name("a"); + + TColumnAccessPaths all_access_paths { + create_legacy_data_access_path({"a", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_legacy_data_access_path({"a", ColumnIterator::ACCESS_NULL})}; + + auto st = array_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set array access paths: " << st.to_string(); + EXPECT_TRUE(array_iterator.read_null_map_only()); + EXPECT_EQ(array_iterator._item_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } + + { + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = std::make_unique(std::make_shared()); + MapFileColumnIterator map_iterator(std::make_shared(), + std::move(map_null_iter), std::move(map_offsets_iter), + std::move(map_key_iter), std::move(map_val_iter)); + map_iterator.set_column_name("m"); + + TColumnAccessPaths all_access_paths { + create_legacy_data_access_path({"m", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_legacy_data_access_path({"m", ColumnIterator::ACCESS_NULL})}; + + auto st = map_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set map access paths: " << st.to_string(); + EXPECT_TRUE(map_iterator.read_null_map_only()); + EXPECT_EQ(map_iterator._key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(map_iterator._val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } +} + +TEST_F(ColumnReaderTest, StructPredicateMetaPathDoesNotOverrideExistingDataNeed) { + auto make_struct_iterator = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto city_iter = + std::make_unique(std::make_shared()); + city_iter->set_column_name("city"); + auto data_iter = std::make_unique(std::make_shared()); + data_iter->set_column_name("data"); + sub_iters.emplace_back(std::move(city_iter)); + sub_iters.emplace_back(std::move(data_iter)); + auto struct_iterator = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + struct_iterator->set_column_name("s"); + return struct_iterator; + }; + + auto struct_iterator = make_struct_iterator(); + TColumnAccessPaths all_access_paths { + create_data_access_path({"s"}), + create_meta_access_path({"s", "city", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_meta_access_path({"s", "city", ColumnIterator::ACCESS_NULL})}; + + auto st = struct_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set struct access paths: " << st.to_string(); + + auto* city_iter = + static_cast(struct_iterator->_sub_column_iterators[0].get()); + EXPECT_FALSE(city_iter->read_null_map_only()); + EXPECT_EQ(city_iter->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iterator->_sub_column_iterators[1]->read_requirement(), + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + + struct_iterator = make_struct_iterator(); + all_access_paths = {create_meta_access_path({"s", "city", ColumnIterator::ACCESS_NULL})}; + predicate_access_paths = all_access_paths; + st = struct_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set predicate-only struct access paths: " << st.to_string(); + city_iter = + static_cast(struct_iterator->_sub_column_iterators[0].get()); + EXPECT_TRUE(city_iter->read_null_map_only()); + EXPECT_EQ(struct_iterator->_sub_column_iterators[1]->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); +} + +TEST_F(ColumnReaderTest, StructSiblingDataPathDoesNotDisablePredicateMetaOnlyRead) { + auto city_iterator = std::make_unique( + create_test_reader(true, 0, FieldType::OLAP_FIELD_TYPE_STRING)); + city_iterator->set_column_name("city"); + auto* city_iterator_ptr = city_iterator.get(); + + auto data_iterator = std::make_unique(create_test_reader()); + data_iterator->set_column_name("data"); + auto* data_iterator_ptr = data_iterator.get(); + + std::vector sub_iterators; + sub_iterators.emplace_back(std::move(city_iterator)); + sub_iterators.emplace_back(std::move(data_iterator)); + StructFileColumnIterator struct_iterator( + create_test_reader(false, 0, FieldType::OLAP_FIELD_TYPE_STRUCT), nullptr, + std::move(sub_iterators)); + struct_iterator.set_column_name("s"); + + TColumnAccessPaths all_access_paths { + create_data_access_path({"s", "data"}), + create_meta_access_path({"s", "city", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_meta_access_path({"s", "city", ColumnIterator::ACCESS_NULL})}; + + auto st = struct_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << st.to_string(); + + EXPECT_EQ(struct_iterator.read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(data_iterator_ptr->read_requirement(), ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_TRUE(city_iterator_ptr->read_null_map_only()); + EXPECT_EQ(city_iterator_ptr->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); +} + +TEST_F(ColumnReaderTest, ArrayPredicateMetaPathDoesNotOverrideExistingDataNeed) { + auto make_array_iterator = []() { + auto item_iter = + std::make_unique(std::make_shared()); + item_iter->set_column_name("item"); + auto offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto null_iter = std::make_unique(std::make_shared()); + auto array_iterator = std::make_unique( + std::make_shared(), std::move(offsets_iter), std::move(item_iter), + std::move(null_iter)); + array_iterator->set_column_name("a"); + return array_iterator; + }; + + auto array_iterator = make_array_iterator(); + TColumnAccessPaths all_access_paths {create_data_access_path({"a"}), + create_meta_access_path({"a", ColumnIterator::ACCESS_ALL, + ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths {create_meta_access_path( + {"a", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_NULL})}; + + auto st = array_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set array access paths: " << st.to_string(); + auto* item_iter = static_cast(array_iterator->_item_iterator.get()); + EXPECT_FALSE(item_iter->read_null_map_only()); + EXPECT_EQ(item_iter->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + + array_iterator = make_array_iterator(); + all_access_paths = {create_meta_access_path( + {"a", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_NULL})}; + predicate_access_paths = all_access_paths; + st = array_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set predicate-only array access paths: " << st.to_string(); + item_iter = static_cast(array_iterator->_item_iterator.get()); + EXPECT_TRUE(item_iter->read_null_map_only()); +} + +TEST_F(ColumnReaderTest, MapPredicateMetaPathDoesNotOverrideExistingDataNeed) { + auto make_map_iterator = []() { + auto null_iter = std::make_unique(std::make_shared()); + auto offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto key_iter = std::make_unique(std::make_shared()); + auto value_iter = + std::make_unique(std::make_shared()); + auto map_iterator = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(offsets_iter), + std::move(key_iter), std::move(value_iter)); + map_iterator->set_column_name("m"); + return map_iterator; + }; + + auto map_iterator = make_map_iterator(); + TColumnAccessPaths all_access_paths { + create_data_access_path({"m"}), + create_meta_access_path( + {"m", ColumnIterator::ACCESS_MAP_VALUES, ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths {create_meta_access_path( + {"m", ColumnIterator::ACCESS_MAP_VALUES, ColumnIterator::ACCESS_NULL})}; + + auto st = map_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set map access paths: " << st.to_string(); + auto* value_iter = static_cast(map_iterator->_val_iterator.get()); + EXPECT_FALSE(value_iter->read_null_map_only()); + EXPECT_EQ(map_iterator->_key_iterator->read_requirement(), + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(value_iter->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + + map_iterator = make_map_iterator(); + all_access_paths = {create_meta_access_path( + {"m", ColumnIterator::ACCESS_MAP_VALUES, ColumnIterator::ACCESS_NULL})}; + predicate_access_paths = all_access_paths; + st = map_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set predicate-only map access paths: " << st.to_string(); + value_iter = static_cast(map_iterator->_val_iterator.get()); + EXPECT_TRUE(value_iter->read_null_map_only()); + EXPECT_EQ(map_iterator->_key_iterator->read_requirement(), + ColumnIterator::ReadRequirement::SKIP); +} + +TEST_F(ColumnReaderTest, MapFullProjectionStillRoutesPredicateSubPaths) { + auto make_value_struct = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_a = std::make_unique(std::make_shared()); + sub_a->set_column_name("a"); + auto sub_b = std::make_unique(std::make_shared()); + sub_b->set_column_name("b"); + sub_iters.emplace_back(std::move(sub_a)); + sub_iters.emplace_back(std::move(sub_b)); + + auto value_struct = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + return value_struct; + }; + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_iterator = std::make_unique( + std::make_shared(), std::move(map_null_iter), std::move(map_offsets_iter), + std::move(map_key_iter), make_value_struct()); + map_iterator->set_column_name("m"); + + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"m"}); + + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"m", "KEYS"}); + predicate_access_paths.emplace_back(); + predicate_access_paths[1] = create_data_access_path({"m", "VALUES", "a"}); + + auto st = map_iterator->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set map access paths: " << st.to_string(); + + EXPECT_EQ(map_iterator->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iterator->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iterator->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* value_struct = static_cast(map_iterator->_val_iterator.get()); + EXPECT_EQ(value_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(value_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); +} + +TEST_F(ColumnReaderTest, MetaOnlyAllPathsStillRoutePredicateSubPaths) { + { + auto struct_null_iter = + std::make_unique(std::make_shared()); + std::vector sub_iters; + auto selected_iter = std::make_unique(std::make_shared()); + selected_iter->set_column_name("selected"); + auto skipped_iter = std::make_unique(std::make_shared()); + skipped_iter->set_column_name("skipped"); + sub_iters.emplace_back(std::move(selected_iter)); + sub_iters.emplace_back(std::move(skipped_iter)); + StructFileColumnIterator struct_iterator(std::make_shared(), + std::move(struct_null_iter), std::move(sub_iters)); + struct_iterator.set_column_name("s"); + + TColumnAccessPaths all_access_paths { + create_meta_access_path({"s", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths {create_data_access_path({"s", "selected"})}; + + auto st = struct_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set struct access paths: " << st.to_string(); + EXPECT_FALSE(struct_iterator.read_null_map_only()); + EXPECT_EQ(struct_iterator._sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(struct_iterator._sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } + + { + auto array_item_iterator = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null_iter = + std::make_unique(std::make_shared()); + ArrayFileColumnIterator array_iterator( + std::make_shared(), std::move(array_offsets_iter), + std::move(array_item_iterator), std::move(array_null_iter)); + array_iterator.set_column_name("a"); + + TColumnAccessPaths all_access_paths { + create_meta_access_path({"a", ColumnIterator::ACCESS_OFFSET}), + create_meta_access_path({"a", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_data_access_path({"a", ColumnIterator::ACCESS_ALL})}; + + auto st = array_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set array access paths: " << st.to_string(); + EXPECT_FALSE(array_iterator.read_offset_only()); + EXPECT_EQ(array_iterator._item_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + } + + { + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = std::make_unique(std::make_shared()); + MapFileColumnIterator map_iterator(std::make_shared(), + std::move(map_null_iter), std::move(map_offsets_iter), + std::move(map_key_iter), std::move(map_val_iter)); + map_iterator.set_column_name("m"); + + TColumnAccessPaths all_access_paths { + create_meta_access_path({"m", ColumnIterator::ACCESS_OFFSET}), + create_meta_access_path({"m", ColumnIterator::ACCESS_NULL})}; + TColumnAccessPaths predicate_access_paths { + create_data_access_path({"m", ColumnIterator::ACCESS_MAP_KEYS})}; + + auto st = map_iterator.set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set map access paths: " << st.to_string(); + EXPECT_FALSE(map_iterator.read_offset_only()); + EXPECT_EQ(map_iterator._key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iterator._val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } +} + +TEST_F(ColumnReaderTest, NestedStructArrayMapStructAccessPaths) { + auto make_value_struct = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_a = std::make_unique(std::make_shared()); + sub_a->set_column_name("a"); + auto sub_b = std::make_unique(std::make_shared()); + sub_b->set_column_name("b"); + sub_iters.emplace_back(std::move(sub_a)); + sub_iters.emplace_back(std::move(sub_b)); + + auto value_struct = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + return value_struct; + }; + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = make_value_struct(); + auto map_iterator = std::make_unique( + std::make_shared(), std::move(map_null_iter), std::move(map_offsets_iter), + std::move(map_key_iter), std::move(map_val_iter)); + map_iterator->set_column_name("item"); + + auto array_null_iter = std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_iterator = std::make_unique( + std::make_shared(), std::move(array_offsets_iter), + std::move(map_iterator), std::move(array_null_iter)); + array_iterator->set_column_name("col2"); + + auto struct_null_iter = std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + auto sub_col1 = std::make_unique(std::make_shared()); + sub_col1->set_column_name("col1"); + struct_sub_iters.emplace_back(std::move(sub_col1)); + struct_sub_iters.emplace_back(std::move(array_iterator)); + auto top_struct = std::make_unique(std::make_shared(), + std::move(struct_null_iter), + std::move(struct_sub_iters)); + top_struct->set_column_name("root"); + + TColumnAccessPaths access_paths; + access_paths.emplace_back(); + access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES", "a"}); + TColumnAccessPaths predicate_access_paths = access_paths; + + auto st = top_struct->set_access_paths(access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set nested access paths: " << st.to_string(); + + EXPECT_EQ(top_struct->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(top_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(top_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* value_struct = static_cast(map_iter->_val_iterator.get()); + EXPECT_EQ(value_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(value_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); +} + +TEST_F(ColumnReaderTest, NestedStructArrayMapStructAccessPathsVariants) { + auto build_nested_iterator = []() { + auto make_value_struct = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_a = std::make_unique(std::make_shared()); + sub_a->set_column_name("a"); + auto sub_b = std::make_unique(std::make_shared()); + sub_b->set_column_name("b"); + sub_iters.emplace_back(std::move(sub_a)); + sub_iters.emplace_back(std::move(sub_b)); + + auto value_struct = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + return value_struct; + }; + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = make_value_struct(); + auto map_iterator = std::make_unique( + std::make_shared(), std::move(map_null_iter), + std::move(map_offsets_iter), std::move(map_key_iter), std::move(map_val_iter)); + map_iterator->set_column_name("item"); + + auto array_null_iter = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_iterator = std::make_unique( + std::make_shared(), std::move(array_offsets_iter), + std::move(map_iterator), std::move(array_null_iter)); + array_iterator->set_column_name("col2"); + + auto struct_null_iter = + std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + auto sub_col1 = std::make_unique(std::make_shared()); + sub_col1->set_column_name("col1"); + struct_sub_iters.emplace_back(std::move(sub_col1)); + struct_sub_iters.emplace_back(std::move(array_iterator)); + auto top_struct = std::make_unique( + std::make_shared(), std::move(struct_null_iter), + std::move(struct_sub_iters)); + top_struct->set_column_name("root"); + return top_struct; + }; + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col1"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + EXPECT_EQ(top_struct->_read_requirement, ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(top_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(top_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "KEYS"}); + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES", "b"}); + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + EXPECT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2"}); + TColumnAccessPaths predicate_access_paths = all_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + EXPECT_EQ(top_struct->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(top_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* value_struct = static_cast(map_iter->_val_iterator.get()); + EXPECT_EQ(value_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(value_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES", "a"}); + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + EXPECT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + } + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "KEYS"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + + auto* value_struct = static_cast(map_iter->_val_iterator.get()); + EXPECT_EQ(value_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(value_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + + auto* value_struct = static_cast(map_iter->_val_iterator.get()); + EXPECT_EQ(value_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(value_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES", "a"}); + TColumnAccessPaths predicate_access_paths = all_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + EXPECT_EQ(top_struct->_read_requirement, ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(top_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* value_struct = static_cast(map_iter->_val_iterator.get()); + EXPECT_EQ(value_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_EQ(value_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*"}); + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES"}); + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(map_iter->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + EXPECT_FALSE(st.ok()); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"wrong_root", "col2"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + EXPECT_FALSE(st.ok()); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "wrong_item"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + EXPECT_FALSE(st.ok()); + } +} + +TEST_F(ColumnReaderTest, DeepNestedAccessPathsFiveLevels) { + auto make_item_struct = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_p = std::make_unique(std::make_shared()); + sub_p->set_column_name("p"); + auto sub_q = std::make_unique(std::make_shared()); + sub_q->set_column_name("q"); + sub_iters.emplace_back(std::move(sub_p)); + sub_iters.emplace_back(std::move(sub_q)); + + auto item_struct = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + item_struct->set_column_name("item"); + return item_struct; + }; + + auto make_value_struct = [make_item_struct]() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto array_offsets = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null = std::make_unique(std::make_shared()); + auto array_iter = std::make_unique( + std::make_shared(), std::move(array_offsets), make_item_struct(), + std::move(array_null)); + array_iter->set_column_name("arr"); + sub_iters.emplace_back(std::move(array_iter)); + + auto sub_z = std::make_unique(std::make_shared()); + sub_z->set_column_name("z"); + sub_iters.emplace_back(std::move(sub_z)); + + auto value_struct = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + return value_struct; + }; + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = make_value_struct(); + auto map_iter = std::make_unique( + std::make_shared(), std::move(map_null_iter), std::move(map_offsets_iter), + std::move(map_key_iter), std::move(map_val_iter)); + map_iter->set_column_name("m"); + + auto struct_null_iter = std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + auto sub_x = std::make_unique(std::make_shared()); + sub_x->set_column_name("x"); + struct_sub_iters.emplace_back(std::move(sub_x)); + struct_sub_iters.emplace_back(std::move(map_iter)); + auto top_struct = std::make_unique(std::make_shared(), + std::move(struct_null_iter), + std::move(struct_sub_iters)); + top_struct->set_column_name("root"); + + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "m", "VALUES", "arr", "*"}); + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"root", "m", "VALUES", "arr", "*", "q"}); + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set deep access paths: " << st.to_string(); + + auto* map_ptr = static_cast(top_struct->_sub_column_iterators[1].get()); + EXPECT_EQ(map_ptr->_key_iterator->_read_requirement, ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(map_ptr->_val_iterator->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); + + auto* value_struct = static_cast(map_ptr->_val_iterator.get()); + auto* array_iter = + static_cast(value_struct->_sub_column_iterators[0].get()); + auto* item_struct = static_cast(array_iter->_item_iterator.get()); + EXPECT_EQ(item_struct->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + EXPECT_EQ(item_struct->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::PREDICATE); +} + +TEST_F(ColumnReaderTest, NestedLazyOutputInLazyPredicatePhase) { + auto struct_null_iterator = + std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + struct_sub_iters.emplace_back( + std::make_unique(std::make_shared())); + struct_sub_iters.emplace_back( + std::make_unique(std::make_shared())); + StructFileColumnIterator struct_iterator(std::make_shared(), + std::move(struct_null_iterator), + std::move(struct_sub_iters)); + struct_iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + struct_iterator.set_read_requirement_self(ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_FALSE(struct_iterator.has_lazy_read_target()); + EXPECT_FALSE(struct_iterator.need_to_read()); + struct_iterator._sub_column_iterators[0]->set_lazy_output_requirement(); + EXPECT_TRUE(struct_iterator.has_lazy_read_target()); + EXPECT_TRUE(struct_iterator.need_to_read()); + + auto array_item_iterator = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_null_iter = std::make_unique(std::make_shared()); + ArrayFileColumnIterator array_iterator( + std::make_shared(), std::move(array_offsets_iter), + std::move(array_item_iterator), std::move(array_null_iter)); + array_iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + array_iterator.set_read_requirement_self(ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_FALSE(array_iterator.has_lazy_read_target()); + EXPECT_FALSE(array_iterator.need_to_read()); + array_iterator._item_iterator->set_lazy_output_requirement(); + EXPECT_TRUE(array_iterator.has_lazy_read_target()); + EXPECT_TRUE(array_iterator.need_to_read()); + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = std::make_unique(std::make_shared()); + MapFileColumnIterator map_iterator(std::make_shared(), std::move(map_null_iter), + std::move(map_offsets_iter), std::move(map_key_iter), + std::move(map_val_iter)); + map_iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + map_iterator.set_read_requirement_self(ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_FALSE(map_iterator.has_lazy_read_target()); + EXPECT_FALSE(map_iterator.need_to_read()); + map_iterator._val_iterator->set_lazy_output_requirement(); + EXPECT_TRUE(map_iterator.has_lazy_read_target()); + EXPECT_TRUE(map_iterator.need_to_read()); +} + +TEST_F(ColumnReaderTest, NestedReadPhaseLazyOutputMatrix) { + auto build_nested_iterator = []() { + auto make_value_struct = []() { + auto null_iter = std::make_unique(std::make_shared()); + std::vector sub_iters; + auto sub_a = std::make_unique(std::make_shared()); + sub_a->set_column_name("a"); + auto sub_b = std::make_unique(std::make_shared()); + sub_b->set_column_name("b"); + sub_iters.emplace_back(std::move(sub_a)); + sub_iters.emplace_back(std::move(sub_b)); + + auto value_struct = std::make_unique( + std::make_shared(), std::move(null_iter), std::move(sub_iters)); + return value_struct; + }; + + auto map_null_iter = std::make_unique(std::make_shared()); + auto map_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto map_key_iter = std::make_unique(std::make_shared()); + auto map_val_iter = make_value_struct(); + auto map_iterator = std::make_unique( + std::make_shared(), std::move(map_null_iter), + std::move(map_offsets_iter), std::move(map_key_iter), std::move(map_val_iter)); + map_iterator->set_column_name("item"); + + auto array_null_iter = + std::make_unique(std::make_shared()); + auto array_offsets_iter = std::make_unique( + std::make_unique(std::make_shared())); + auto array_iterator = std::make_unique( + std::make_shared(), std::move(array_offsets_iter), + std::move(map_iterator), std::move(array_null_iter)); + array_iterator->set_column_name("col2"); + + auto struct_null_iter = + std::make_unique(std::make_shared()); + std::vector struct_sub_iters; + auto sub_col1 = std::make_unique(std::make_shared()); + sub_col1->set_column_name("col1"); + struct_sub_iters.emplace_back(std::move(sub_col1)); + struct_sub_iters.emplace_back(std::move(array_iterator)); + auto top_struct = std::make_unique( + std::make_shared(), std::move(struct_null_iter), + std::move(struct_sub_iters)); + top_struct->set_column_name("root"); + return top_struct; + }; + + auto assert_need_to_read = [](StructFileColumnIterator* top_struct) { + auto* array_iter = + static_cast(top_struct->_sub_column_iterators[1].get()); + auto* map_iter = static_cast(array_iter->_item_iterator.get()); + auto* value_struct = static_cast(map_iter->_val_iterator.get()); + auto expect_scalar = [](ColumnIterator::ReadRequirement requirement, + ColumnIterator::ReadPhase mode) { + switch (mode) { + case ColumnIterator::ReadPhase::NORMAL: + return requirement != ColumnIterator::ReadRequirement::SKIP; + case ColumnIterator::ReadPhase::PREDICATE: + return requirement == ColumnIterator::ReadRequirement::PREDICATE; + case ColumnIterator::ReadPhase::LAZY: + return requirement == ColumnIterator::ReadRequirement::LAZY_OUTPUT; + default: + return false; + } + }; + auto expect_nested = [](ColumnIterator::ReadRequirement requirement, + ColumnIterator::ReadPhase mode) { + switch (mode) { + case ColumnIterator::ReadPhase::NORMAL: + return requirement != ColumnIterator::ReadRequirement::SKIP; + case ColumnIterator::ReadPhase::PREDICATE: + return requirement == ColumnIterator::ReadRequirement::PREDICATE; + default: + return false; + } + }; + + top_struct->set_read_phase(ColumnIterator::ReadPhase::NORMAL); + EXPECT_EQ(expect_nested(top_struct->read_requirement(), ColumnIterator::ReadPhase::NORMAL), + top_struct->need_to_read()); + EXPECT_EQ(expect_nested(array_iter->read_requirement(), ColumnIterator::ReadPhase::NORMAL), + array_iter->need_to_read()); + EXPECT_EQ(expect_nested(map_iter->read_requirement(), ColumnIterator::ReadPhase::NORMAL), + map_iter->need_to_read()); + EXPECT_EQ( + expect_nested(value_struct->read_requirement(), ColumnIterator::ReadPhase::NORMAL), + value_struct->need_to_read()); + EXPECT_EQ(expect_scalar(map_iter->_key_iterator->read_requirement(), + ColumnIterator::ReadPhase::NORMAL), + map_iter->_key_iterator->need_to_read()); + EXPECT_EQ(expect_nested(map_iter->_val_iterator->read_requirement(), + ColumnIterator::ReadPhase::NORMAL), + map_iter->_val_iterator->need_to_read()); + + top_struct->set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + EXPECT_EQ( + expect_nested(top_struct->read_requirement(), ColumnIterator::ReadPhase::PREDICATE), + top_struct->need_to_read()); + EXPECT_EQ( + expect_nested(array_iter->read_requirement(), ColumnIterator::ReadPhase::PREDICATE), + array_iter->need_to_read()); + EXPECT_EQ(expect_nested(map_iter->read_requirement(), ColumnIterator::ReadPhase::PREDICATE), + map_iter->need_to_read()); + EXPECT_EQ(expect_nested(value_struct->read_requirement(), + ColumnIterator::ReadPhase::PREDICATE), + value_struct->need_to_read()); + EXPECT_EQ(expect_scalar(map_iter->_key_iterator->read_requirement(), + ColumnIterator::ReadPhase::PREDICATE), + map_iter->_key_iterator->need_to_read()); + EXPECT_EQ(expect_nested(map_iter->_val_iterator->read_requirement(), + ColumnIterator::ReadPhase::PREDICATE), + map_iter->_val_iterator->need_to_read()); + + top_struct->set_read_phase(ColumnIterator::ReadPhase::LAZY); + EXPECT_EQ(top_struct->has_lazy_read_target(), top_struct->need_to_read()); + EXPECT_EQ(array_iter->has_lazy_read_target(), array_iter->need_to_read()); + EXPECT_EQ(map_iter->has_lazy_read_target(), map_iter->need_to_read()); + EXPECT_EQ(value_struct->has_lazy_read_target(), value_struct->need_to_read()); + EXPECT_EQ(expect_scalar(map_iter->_key_iterator->read_requirement(), + ColumnIterator::ReadPhase::LAZY), + map_iter->_key_iterator->need_to_read()); + EXPECT_EQ(map_iter->_val_iterator->has_lazy_read_target(), + map_iter->_val_iterator->need_to_read()); + }; + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES", "a"}); + TColumnAccessPaths predicate_access_paths = all_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + assert_need_to_read(top_struct.get()); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "KEYS"}); + TColumnAccessPaths predicate_access_paths; + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + assert_need_to_read(top_struct.get()); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*"}); + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES"}); + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); + assert_need_to_read(top_struct.get()); + } + + { + auto top_struct = build_nested_iterator(); + TColumnAccessPaths all_access_paths; + all_access_paths.emplace_back(); + all_access_paths[0] = create_data_access_path({"root", "col2", "*", "VALUES"}); + TColumnAccessPaths predicate_access_paths; + predicate_access_paths.emplace_back(); + predicate_access_paths[0] = create_data_access_path({"root", "col2", "*", "KEYS"}); + + auto st = top_struct->set_access_paths(all_access_paths, predicate_access_paths); + EXPECT_TRUE(st.ok()); + } } TEST_F(ColumnReaderTest, MultiAccessPaths) { @@ -194,7 +2652,7 @@ TEST_F(ColumnReaderTest, MultiAccessPaths) { // all access paths: // self.sub_col_2.*.KEYS // predicates paths empty - all_access_paths[0].data_access_path.path = {"self", "sub_col_2", "*", "KEYS"}; + all_access_paths[0] = create_data_access_path({"self", "sub_col_2", "*", "KEYS"}); TColumnAccessPaths predicate_access_paths; @@ -202,20 +2660,316 @@ TEST_F(ColumnReaderTest, MultiAccessPaths) { auto st = iterator->set_access_paths(all_access_paths, predicate_access_paths); ASSERT_TRUE(st.ok()) << "failed to set access paths: " << st.to_string(); - ASSERT_EQ(iterator->_reading_flag, ColumnIterator::ReadingFlag::NEED_TO_READ); + ASSERT_EQ(iterator->_read_requirement, ColumnIterator::ReadRequirement::LAZY_OUTPUT); - ASSERT_EQ(iterator->_sub_column_iterators[0]->_reading_flag, - ColumnIterator::ReadingFlag::SKIP_READING); - ASSERT_EQ(iterator->_sub_column_iterators[1]->_reading_flag, - ColumnIterator::ReadingFlag::NEED_TO_READ); + ASSERT_EQ(iterator->_sub_column_iterators[0]->_read_requirement, + ColumnIterator::ReadRequirement::SKIP); + ASSERT_EQ(iterator->_sub_column_iterators[1]->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); auto* array_iter = static_cast(iterator->_sub_column_iterators[1].get()); - ASSERT_EQ(array_iter->_item_iterator->_reading_flag, ColumnIterator::ReadingFlag::NEED_TO_READ); + ASSERT_EQ(array_iter->_item_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); auto* map_iter = static_cast(array_iter->_item_iterator.get()); - ASSERT_EQ(map_iter->_key_iterator->_reading_flag, ColumnIterator::ReadingFlag::NEED_TO_READ); - ASSERT_EQ(map_iter->_val_iterator->_reading_flag, ColumnIterator::ReadingFlag::SKIP_READING); + ASSERT_EQ(map_iter->_key_iterator->_read_requirement, + ColumnIterator::ReadRequirement::LAZY_OUTPUT); + ASSERT_EQ(map_iter->_val_iterator->_read_requirement, ColumnIterator::ReadRequirement::SKIP); +} + +TEST_F(ColumnReaderTest, StructNextBatchAndReadByRowidsUseSequentialChildReads) { + std::vector sub_column_iterators; + auto first_child = std::make_unique(); + auto* first_child_ptr = first_child.get(); + auto second_child = std::make_unique(); + auto* second_child_ptr = second_child.get(); + sub_column_iterators.emplace_back(std::move(first_child)); + sub_column_iterators.emplace_back(std::move(second_child)); + + StructFileColumnIterator struct_iterator(create_test_reader(), nullptr, + std::move(sub_column_iterators)); + + MutableColumnPtr dst = create_int_struct_column(2); + size_t rows = 3; + bool has_null = false; + auto st = struct_iterator.next_batch(&rows, dst, &has_null); + ASSERT_TRUE(st.ok()) << "struct next_batch failed: " << st.to_string(); + EXPECT_EQ(3, rows); + EXPECT_EQ(3, dst->size()); + EXPECT_THAT(first_child_ptr->next_batch_sizes, ::testing::ElementsAre(3)); + EXPECT_THAT(second_child_ptr->next_batch_sizes, ::testing::ElementsAre(3)); + + first_child_ptr->clear_tracking(); + second_child_ptr->clear_tracking(); + + const rowid_t rowids[] = {0, 1, 4, 5, 6}; + st = struct_iterator.read_by_rowids(rowids, std::size(rowids), dst); + ASSERT_TRUE(st.ok()) << "struct read_by_rowids failed: " << st.to_string(); + EXPECT_EQ(8, dst->size()); + EXPECT_THAT(first_child_ptr->seek_ordinals, ::testing::ElementsAre(0, 4)); + EXPECT_THAT(second_child_ptr->seek_ordinals, ::testing::ElementsAre(0, 4)); + EXPECT_THAT(first_child_ptr->next_batch_sizes, ::testing::ElementsAre(2, 3)); + EXPECT_THAT(second_child_ptr->next_batch_sizes, ::testing::ElementsAre(2, 3)); + EXPECT_TRUE(first_child_ptr->read_by_rowids_batches.empty()); + EXPECT_TRUE(second_child_ptr->read_by_rowids_batches.empty()); +} + +TEST_F(ColumnReaderTest, StructNullMapOnlyNextBatchSkipsSubColumns) { + auto null_iterator = std::make_unique(); + auto* null_iterator_ptr = null_iterator.get(); + std::vector sub_column_iterators; + auto child_iterator = std::make_unique(); + auto* child_iterator_ptr = child_iterator.get(); + child_iterator->set_column_name("field"); + sub_column_iterators.emplace_back(std::move(child_iterator)); + + StructFileColumnIterator struct_iterator(create_test_reader(true), std::move(null_iterator), + std::move(sub_column_iterators)); + struct_iterator.set_column_name("s"); + + TColumnAccessPaths null_path {create_meta_access_path({"s", ColumnIterator::ACCESS_NULL})}; + auto st = struct_iterator.set_access_paths(null_path, null_path); + ASSERT_TRUE(st.ok()) << "set_access_paths failed: " << st.to_string(); + EXPECT_TRUE(struct_iterator.read_null_map_only()); + EXPECT_EQ(child_iterator_ptr->read_requirement(), ColumnIterator::ReadRequirement::SKIP); + + MutableColumnPtr dst = create_nullable_int_struct_column(1); + size_t rows = 2; + bool has_null = false; + st = struct_iterator.next_batch(&rows, dst, &has_null); + ASSERT_TRUE(st.ok()) << "struct null-map-only next_batch failed: " << st.to_string(); + EXPECT_TRUE(has_null); + EXPECT_EQ(2, dst->size()); + EXPECT_THAT(null_iterator_ptr->next_batch_sizes, ::testing::ElementsAre(2)); + EXPECT_TRUE(child_iterator_ptr->next_batch_sizes.empty()); + + const auto& nullable_column = assert_cast(*dst); + EXPECT_EQ(2, nullable_column.get_null_map_column().size()); + const auto& nested_struct = assert_cast( + nullable_column.get_nested_column()); + EXPECT_EQ(2, nested_struct.get_column(0).size()); +} + +TEST_F(ColumnReaderTest, ArrayNullMapOnlyNextBatchAndReadByRowidsSkipItems) { + auto null_iterator = std::make_unique(); + auto* null_iterator_ptr = null_iterator.get(); + auto item_iterator = std::make_unique(); + auto* item_iterator_ptr = item_iterator.get(); + auto offset_iterator = create_tracking_offset_iterator(); + + ArrayFileColumnIterator array_iterator(create_test_reader(true), + std::move(offset_iterator.iterator), + std::move(item_iterator), std::move(null_iterator)); + array_iterator.set_column_name("a"); + + TColumnAccessPaths null_path {create_meta_access_path({"a", ColumnIterator::ACCESS_NULL})}; + auto st = array_iterator.set_access_paths(null_path, null_path); + ASSERT_TRUE(st.ok()) << "set_access_paths failed: " << st.to_string(); + EXPECT_TRUE(array_iterator.read_null_map_only()); + EXPECT_EQ(item_iterator_ptr->read_requirement(), ColumnIterator::ReadRequirement::SKIP); + + MutableColumnPtr dst = create_nullable_int_array_column(); + size_t rows = 3; + bool has_null = false; + st = array_iterator.next_batch(&rows, dst, &has_null); + ASSERT_TRUE(st.ok()) << "array null-map-only next_batch failed: " << st.to_string(); + EXPECT_TRUE(has_null); + EXPECT_EQ(3, dst->size()); + EXPECT_THAT(null_iterator_ptr->next_batch_sizes, ::testing::ElementsAre(3)); + EXPECT_TRUE(item_iterator_ptr->next_batch_sizes.empty()); + EXPECT_TRUE(offset_iterator.tracker->next_batch_sizes.empty()); + + null_iterator_ptr->clear_tracking(); + item_iterator_ptr->clear_tracking(); + + const rowid_t rowids[] = {1, 3}; + st = array_iterator.read_by_rowids(rowids, std::size(rowids), dst); + ASSERT_TRUE(st.ok()) << "array null-map-only read_by_rowids failed: " << st.to_string(); + EXPECT_EQ(5, dst->size()); + EXPECT_THAT(null_iterator_ptr->seek_ordinals, ::testing::ElementsAre(1, 3)); + EXPECT_THAT(null_iterator_ptr->next_batch_sizes, ::testing::ElementsAre(1, 1)); + EXPECT_TRUE(item_iterator_ptr->next_batch_sizes.empty()); + EXPECT_TRUE(offset_iterator.tracker->next_batch_sizes.empty()); +} + +TEST_F(ColumnReaderTest, MapNullMapOnlyNextBatchAndReadByRowidsSkipKeysAndValues) { + auto null_iterator = std::make_unique(); + auto* null_iterator_ptr = null_iterator.get(); + auto key_iterator = std::make_unique(); + auto* key_iterator_ptr = key_iterator.get(); + auto value_iterator = std::make_unique(); + auto* value_iterator_ptr = value_iterator.get(); + auto offset_iterator = create_tracking_offset_iterator(); + + MapFileColumnIterator map_iterator(create_test_reader(true, 4), std::move(null_iterator), + std::move(offset_iterator.iterator), std::move(key_iterator), + std::move(value_iterator)); + map_iterator.set_column_name("m"); + + TColumnAccessPaths null_path {create_meta_access_path({"m", ColumnIterator::ACCESS_NULL})}; + auto st = map_iterator.set_access_paths(null_path, null_path); + ASSERT_TRUE(st.ok()) << "set_access_paths failed: " << st.to_string(); + EXPECT_TRUE(map_iterator.read_null_map_only()); + EXPECT_EQ(key_iterator_ptr->read_requirement(), ColumnIterator::ReadRequirement::SKIP); + EXPECT_EQ(value_iterator_ptr->read_requirement(), ColumnIterator::ReadRequirement::SKIP); + + MutableColumnPtr dst = create_nullable_int_map_column(); + size_t rows = 3; + bool has_null = false; + st = map_iterator.next_batch(&rows, dst, &has_null); + ASSERT_TRUE(st.ok()) << "map null-map-only next_batch failed: " << st.to_string(); + EXPECT_TRUE(has_null); + EXPECT_EQ(3, dst->size()); + EXPECT_THAT(null_iterator_ptr->next_batch_sizes, ::testing::ElementsAre(3)); + EXPECT_TRUE(key_iterator_ptr->next_batch_sizes.empty()); + EXPECT_TRUE(value_iterator_ptr->next_batch_sizes.empty()); + EXPECT_TRUE(offset_iterator.tracker->next_batch_sizes.empty()); + + null_iterator_ptr->clear_tracking(); + key_iterator_ptr->clear_tracking(); + value_iterator_ptr->clear_tracking(); + + const rowid_t rowids[] = {1, 3}; + st = map_iterator.read_by_rowids(rowids, std::size(rowids), dst); + ASSERT_TRUE(st.ok()) << "map null-map-only read_by_rowids failed: " << st.to_string(); + EXPECT_EQ(5, dst->size()); + ASSERT_EQ(1, null_iterator_ptr->read_by_rowids_batches.size()); + EXPECT_THAT(null_iterator_ptr->read_by_rowids_batches[0], ::testing::ElementsAre(1, 3)); + EXPECT_TRUE(key_iterator_ptr->next_batch_sizes.empty()); + EXPECT_TRUE(value_iterator_ptr->next_batch_sizes.empty()); + EXPECT_TRUE(offset_iterator.tracker->next_batch_sizes.empty()); +} + +TEST_F(ColumnReaderTest, CollectPrefetchersHonorsNestedReadRequirements) { + auto null_iterator = std::make_unique(); + auto* null_iterator_ptr = null_iterator.get(); + std::vector sub_column_iterators; + auto predicate_child = std::make_unique(); + auto* predicate_child_ptr = predicate_child.get(); + auto lazy_child = std::make_unique(); + auto* lazy_child_ptr = lazy_child.get(); + sub_column_iterators.emplace_back(std::move(predicate_child)); + sub_column_iterators.emplace_back(std::move(lazy_child)); + + StructFileColumnIterator struct_iterator(create_test_reader(true), std::move(null_iterator), + std::move(sub_column_iterators)); + struct_iterator.set_read_requirement_self(ColumnIterator::ReadRequirement::PREDICATE); + predicate_child_ptr->set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + lazy_child_ptr->set_read_requirement(ColumnIterator::ReadRequirement::LAZY_OUTPUT); + struct_iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + + std::map> prefetchers; + struct_iterator.collect_prefetchers(prefetchers, PrefetcherInitMethod::FROM_ROWIDS); + + EXPECT_THAT(null_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_THAT(predicate_child_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_TRUE(lazy_child_ptr->collect_methods.empty()); + EXPECT_THAT(prefetchers[PrefetcherInitMethod::FROM_ROWIDS], + ::testing::ElementsAre(null_iterator_ptr->prefetcher(), + predicate_child_ptr->prefetcher())); + + null_iterator_ptr->clear_tracking(); + predicate_child_ptr->clear_tracking(); + lazy_child_ptr->clear_tracking(); + prefetchers.clear(); + + struct_iterator.set_read_phase(ColumnIterator::ReadPhase::LAZY); + struct_iterator.collect_prefetchers(prefetchers, PrefetcherInitMethod::FROM_ROWIDS); + + EXPECT_THAT(null_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_TRUE(predicate_child_ptr->collect_methods.empty()); + EXPECT_THAT(lazy_child_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); +} + +TEST_F(ColumnReaderTest, ArrayAndMapCollectPrefetchersUseAllDataBlocksForNestedData) { + { + auto null_iterator = std::make_unique(); + auto* null_iterator_ptr = null_iterator.get(); + auto item_iterator = std::make_unique(); + auto* item_iterator_ptr = item_iterator.get(); + auto offset_iterator = create_tracking_offset_iterator(); + auto* offset_iterator_ptr = offset_iterator.tracker; + + ArrayFileColumnIterator array_iterator(create_test_reader(true), + std::move(offset_iterator.iterator), + std::move(item_iterator), std::move(null_iterator)); + array_iterator.set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + array_iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + + std::map> prefetchers; + array_iterator.collect_prefetchers(prefetchers, PrefetcherInitMethod::FROM_ROWIDS); + + EXPECT_THAT(offset_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_THAT(null_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_THAT(item_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::ALL_DATA_BLOCKS)); + EXPECT_THAT(prefetchers[PrefetcherInitMethod::ALL_DATA_BLOCKS], + ::testing::ElementsAre(item_iterator_ptr->prefetcher())); + } + + { + auto null_iterator = std::make_unique(); + auto* null_iterator_ptr = null_iterator.get(); + auto key_iterator = std::make_unique(); + auto* key_iterator_ptr = key_iterator.get(); + auto value_iterator = std::make_unique(); + auto* value_iterator_ptr = value_iterator.get(); + auto offset_iterator = create_tracking_offset_iterator(); + auto* offset_iterator_ptr = offset_iterator.tracker; + + MapFileColumnIterator map_iterator(create_test_reader(true), std::move(null_iterator), + std::move(offset_iterator.iterator), + std::move(key_iterator), std::move(value_iterator)); + map_iterator.set_read_requirement(ColumnIterator::ReadRequirement::PREDICATE); + map_iterator.set_read_phase(ColumnIterator::ReadPhase::PREDICATE); + + std::map> prefetchers; + map_iterator.collect_prefetchers(prefetchers, PrefetcherInitMethod::FROM_ROWIDS); + + EXPECT_THAT(offset_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_THAT(null_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::FROM_ROWIDS)); + EXPECT_THAT(key_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::ALL_DATA_BLOCKS)); + EXPECT_THAT(value_iterator_ptr->collect_methods, + ::testing::ElementsAre(PrefetcherInitMethod::ALL_DATA_BLOCKS)); + EXPECT_THAT(prefetchers[PrefetcherInitMethod::ALL_DATA_BLOCKS], + ::testing::ElementsAre(key_iterator_ptr->prefetcher(), + value_iterator_ptr->prefetcher())); + } +} + +TEST_F(ColumnReaderTest, MapPredicateAccessAllWithOffsetKeepsKeysReadable) { + auto map_reader = create_test_reader(false, 0, FieldType::OLAP_FIELD_TYPE_MAP); + auto key_iter = std::make_unique( + create_test_reader(false, 0, FieldType::OLAP_FIELD_TYPE_STRING)); + auto* key_ptr = key_iter.get(); + auto val_iter = std::make_unique( + create_test_reader(false, 0, FieldType::OLAP_FIELD_TYPE_STRING)); + auto* val_ptr = val_iter.get(); + auto offset_iterator = create_tracking_offset_iterator(); + + MapFileColumnIterator map_iter(map_reader, nullptr, std::move(offset_iterator.iterator), + std::move(key_iter), std::move(val_iter)); + map_iter.set_column_name("map_col"); + + TColumnAccessPaths access_paths {create_meta_access_path( + {"map_col", ColumnIterator::ACCESS_ALL, ColumnIterator::ACCESS_OFFSET})}; + auto st = map_iter.set_access_paths(access_paths, access_paths); + ASSERT_TRUE(st.ok()) << "set_access_paths failed: " << st.to_string(); + + EXPECT_EQ(key_ptr->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_FALSE(key_ptr->read_offset_only()); + EXPECT_EQ(val_ptr->read_requirement(), ColumnIterator::ReadRequirement::PREDICATE); + EXPECT_TRUE(val_ptr->read_offset_only()); } TEST_F(ColumnReaderTest, OffsetPeekUsesPageSentinelWhenNoRemaining) { @@ -285,7 +3039,7 @@ TEST_F(ColumnReaderTest, MapReadByRowidsSkipReadingResizesDestination) { MapFileColumnIterator map_iter(map_reader, std::move(null_iter), std::move(offsets_iter), std::move(key_iter), std::move(val_iter)); map_iter.set_column_name("map_col"); - map_iter.set_reading_flag(ColumnIterator::ReadingFlag::SKIP_READING); + map_iter.set_read_requirement(ColumnIterator::ReadRequirement::SKIP); // prepare an empty ColumnMap as destination auto keys = ColumnInt32::create(); @@ -312,9 +3066,7 @@ TEST_F(ColumnReaderTest, MapAccessAllWithOffsetDoesNotPropagateOffsetToKey) { auto offsets_iter = std::make_unique( std::make_unique(std::make_shared())); auto key_iter = std::make_unique(std::make_shared()); - key_iter->set_column_name("key"); auto val_iter = std::make_unique(std::make_shared()); - val_iter->set_column_name("value"); MapFileColumnIterator map_iter(map_reader, std::move(null_iter), std::move(offsets_iter), std::move(key_iter), std::move(val_iter)); @@ -323,20 +3075,20 @@ TEST_F(ColumnReaderTest, MapAccessAllWithOffsetDoesNotPropagateOffsetToKey) { // path: [map_col, *, OFFSET] — simulates length(map_col['c_phone']) TColumnAccessPaths all_access_paths; all_access_paths.emplace_back(); - all_access_paths[0].data_access_path.path = {"map_col", "*", "OFFSET"}; + all_access_paths[0] = create_meta_access_path({"map_col", "*", "OFFSET"}); TColumnAccessPaths predicate_access_paths; auto st = map_iter.set_access_paths(all_access_paths, predicate_access_paths); ASSERT_TRUE(st.ok()) << "set_access_paths failed: " << st.to_string(); - // Key must be fully readable (NEED_TO_READ), NOT in OFFSET_ONLY mode. + // Key must be fully readable (LAZY_OUTPUT), NOT in OFFSET_ONLY mode. auto* key_ptr = static_cast(map_iter._key_iterator.get()); - ASSERT_EQ(key_ptr->_reading_flag, ColumnIterator::ReadingFlag::NEED_TO_READ); + ASSERT_EQ(key_ptr->_read_requirement, ColumnIterator::ReadRequirement::LAZY_OUTPUT); ASSERT_FALSE(key_ptr->read_offset_only()); // Value should be in OFFSET_ONLY mode since we only need string lengths. auto* val_ptr = static_cast(map_iter._val_iterator.get()); - ASSERT_EQ(val_ptr->_reading_flag, ColumnIterator::ReadingFlag::NEED_TO_READ); + ASSERT_EQ(val_ptr->_read_requirement, ColumnIterator::ReadRequirement::LAZY_OUTPUT); ASSERT_TRUE(val_ptr->read_offset_only()); } diff --git a/be/test/storage/segment/segment_iterator_lazy_pruned_test.cpp b/be/test/storage/segment/segment_iterator_lazy_pruned_test.cpp new file mode 100644 index 00000000000000..b990201b49544c --- /dev/null +++ b/be/test/storage/segment/segment_iterator_lazy_pruned_test.cpp @@ -0,0 +1,186 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +#include + +#include +#include + +#include "common/cast_set.h" +#include "core/assert_cast.h" +#include "core/block/block.h" +#include "core/column/column_vector.h" +#include "core/data_type/data_type_number.h" +#include "storage/olap_common.h" +#include "storage/segment/column_reader.h" +#include "storage/tablet/tablet_schema.h" + +// Use #define private public to access SegmentIterator::_read_lazy_pruned_columns() +// and the small amount of state it consumes. This mirrors the existing +// segment_iterator_* white-box tests. +#if defined(__clang__) +#pragma clang diagnostic push +#pragma clang diagnostic ignored "-Wkeyword-macro" +#endif +#define private public +#include "storage/segment/segment_iterator.h" +#undef private +#if defined(__clang__) +#pragma clang diagnostic pop +#endif + +namespace doris::segment_v2 { +namespace { + +class TrackingLazyColumnIterator final : public ColumnIterator { +public: + Status seek_to_ordinal(ordinal_t ord) override { + seek_ordinals.push_back(ord); + return Status::OK(); + } + + Status read_by_rowids(const rowid_t* rowids, const size_t count, + MutableColumnPtr& dst) override { + read_phases.push_back(_read_phase); + read_rowids.assign(rowids, rowids + count); + ++read_by_rowids_count; + + auto& int_column = assert_cast&>(*dst); + for (size_t i = 0; i < count; ++i) { + int_column.insert_value(cast_set(rowids[i])); + } + return Status::OK(); + } + + void finalize_lazy_phase(MutableColumnPtr& dst) override { + finalize_phases.push_back(_read_phase); + ++finalize_count; + } + + ordinal_t get_current_ordinal() const override { return 0; } + + ReadPhase phase() const { return _read_phase; } + + std::vector seek_ordinals; + std::vector read_rowids; + std::vector read_phases; + std::vector finalize_phases; + int read_by_rowids_count = 0; + int finalize_count = 0; +}; + +TabletSchemaSPtr make_tablet_schema() { + TabletSchemaPB schema_pb; + schema_pb.set_keys_type(KeysType::DUP_KEYS); + auto* col = schema_pb.add_column(); + col->set_unique_id(0); + col->set_name("c0"); + col->set_type("INT"); + col->set_is_key(true); + col->set_is_nullable(false); + + auto tablet_schema = std::make_shared(); + tablet_schema->init_from_pb(schema_pb); + return tablet_schema; +} + +SchemaSPtr make_read_schema(const TabletSchemaSPtr& tablet_schema) { + std::vector read_column_ids(tablet_schema->num_columns()); + for (uint32_t cid = 0; cid < read_column_ids.size(); ++cid) { + read_column_ids[cid] = cid; + } + return std::make_shared(tablet_schema->columns(), read_column_ids); +} + +Block make_int_block() { + Block block; + block.insert({ColumnInt32::create(), std::make_shared(), "c0"}); + return block; +} + +} // namespace + +class SegmentIteratorLazyPrunedTest : public ::testing::Test { +protected: + void SetUp() override { + _tablet_schema = make_tablet_schema(); + _read_schema = make_read_schema(_tablet_schema); + } + + std::unique_ptr make_iter(TrackingLazyColumnIterator** tracking_iter) { + auto iter = std::make_unique(nullptr, _read_schema); + iter->_opts.tablet_schema = _tablet_schema; + iter->_opts.stats = &_stats; + iter->_support_lazy_read_pruned_columns.insert(0); + iter->_column_iterators.resize(1); + + auto column_iter = std::make_unique(); + *tracking_iter = column_iter.get(); + iter->_column_iterators[0] = std::move(column_iter); + return iter; + } + + TabletSchemaSPtr _tablet_schema; + SchemaSPtr _read_schema; + OlapReaderStatistics _stats; +}; + +TEST_F(SegmentIteratorLazyPrunedTest, readsSelectedRowidsInLazyPhaseAndRestoresPhase) { + TrackingLazyColumnIterator* tracking_iter = nullptr; + auto iter = make_iter(&tracking_iter); + iter->_selected_size = 2; + iter->_block_rowids = {10, 20, 30, 40}; + iter->_sel_rowid_idx = {2, 0}; + + auto block = make_int_block(); + auto st = iter->_read_lazy_pruned_columns(&block); + ASSERT_TRUE(st.ok()) << st.to_string(); + + EXPECT_EQ(tracking_iter->read_by_rowids_count, 1); + EXPECT_EQ(tracking_iter->finalize_count, 1); + EXPECT_EQ(tracking_iter->read_rowids, (std::vector {30, 10})); + EXPECT_EQ(tracking_iter->read_phases, + (std::vector {ColumnIterator::ReadPhase::LAZY})); + EXPECT_EQ(tracking_iter->finalize_phases, + (std::vector {ColumnIterator::ReadPhase::LAZY})); + EXPECT_EQ(tracking_iter->phase(), ColumnIterator::ReadPhase::NORMAL); + + const auto& result = + assert_cast&>(*block.get_by_position(0).column); + ASSERT_EQ(result.size(), 2); + EXPECT_EQ(result.get_data()[0], 30); + EXPECT_EQ(result.get_data()[1], 10); +} + +TEST_F(SegmentIteratorLazyPrunedTest, emptySelectionStillFinalizesLazyPlaceholders) { + TrackingLazyColumnIterator* tracking_iter = nullptr; + auto iter = make_iter(&tracking_iter); + iter->_selected_size = 0; + + auto block = make_int_block(); + auto st = iter->_read_lazy_pruned_columns(&block); + ASSERT_TRUE(st.ok()) << st.to_string(); + + EXPECT_EQ(tracking_iter->read_by_rowids_count, 0); + EXPECT_EQ(tracking_iter->finalize_count, 1); + EXPECT_EQ(tracking_iter->finalize_phases, + (std::vector {ColumnIterator::ReadPhase::LAZY})); + EXPECT_EQ(tracking_iter->phase(), ColumnIterator::ReadPhase::NORMAL); + EXPECT_EQ(block.get_by_position(0).column->size(), 0); +} + +} // namespace doris::segment_v2 diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/TopNScanOpt.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/TopNScanOpt.java index 9e6807cc0b2905..9dcbfea57dd497 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/TopNScanOpt.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/processor/post/TopNScanOpt.java @@ -19,11 +19,11 @@ import org.apache.doris.nereids.CascadesContext; import org.apache.doris.nereids.processor.post.TopnFilterPushDownVisitor.PushDownContext; -import org.apache.doris.nereids.trees.expressions.Expression; import org.apache.doris.nereids.trees.plans.Plan; import org.apache.doris.nereids.trees.plans.SortPhase; import org.apache.doris.nereids.trees.plans.algebra.TopN; import org.apache.doris.nereids.trees.plans.physical.PhysicalTopN; +import org.apache.doris.nereids.types.DataType; /** * topN opt @@ -60,13 +60,20 @@ boolean checkTopN(TopN topN) { return false; } - Expression firstKey = topN.getOrderKeys().get(0).getExpr(); + DataType firstKeyType = topN.getOrderKeys().get(0).getExpr().getDataType(); - if (firstKey.getDataType().isFloatType() - || firstKey.getDataType().isDoubleType()) { - return false; - } - return true; + return isSupportedTopNRuntimeFilterType(firstKeyType); + } + + private boolean isSupportedTopNRuntimeFilterType(DataType dataType) { + return dataType.isBooleanType() + || dataType.isIntegralType() + || dataType.isDecimalLikeType() + || dataType.isStringLikeType() + || dataType.isDateLikeType() + || dataType.isTimeType() + || dataType.isIPType() + || dataType.isVarBinaryType(); } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathExpressionCollector.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathExpressionCollector.java index 74a9934e249d7e..a814617fbecccb 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathExpressionCollector.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathExpressionCollector.java @@ -18,6 +18,7 @@ package org.apache.doris.nereids.rules.rewrite; import org.apache.doris.analysis.AccessPathInfo; +import org.apache.doris.catalog.Column; import org.apache.doris.nereids.StatementContext; import org.apache.doris.nereids.rules.rewrite.AccessPathExpressionCollector.CollectorContext; import org.apache.doris.nereids.rules.rewrite.NestedColumnPruning.DataTypeAccessTree; @@ -84,15 +85,17 @@ public class AccessPathExpressionCollector extends DefaultExpressionVisitor { private StatementContext statementContext; private boolean bottomPredicate; + private boolean skipMetaPath; private Multimap slotToAccessPaths; private Stack> nameToLambdaArguments = new Stack<>(); public AccessPathExpressionCollector( StatementContext statementContext, Multimap slotToAccessPaths, - boolean bottomPredicate) { + boolean bottomPredicate, boolean skipMetaPath) { this.statementContext = statementContext; this.slotToAccessPaths = slotToAccessPaths; this.bottomPredicate = bottomPredicate; + this.skipMetaPath = skipMetaPath; } public void collect(Expression expression) { @@ -121,17 +124,17 @@ public Void visitSlotReference(SlotReference slotReference, CollectorContext con if (slotReference.hasSubColPath()) { path.addAll(slotReference.getSubPath()); } - // Strip NULL suffix for variant sub-column access — null-flag-only optimization - // does not apply to variant sub-column data layout. List builderPath = context.accessPathBuilder.getPathList(); - if (builderPath.size() > 1 - && AccessPathInfo.ACCESS_NULL.equals(builderPath.get(builderPath.size() - 1))) { + TAccessPathType pathType = context.type; + if (pathType == TAccessPathType.META) { + // Variant readers do not support metadata-only access paths. Drop the synthetic + // NULL/OFFSET component and read the referenced variant path as ordinary data. builderPath = new ArrayList<>(builderPath.subList(0, builderPath.size() - 1)); + pathType = TAccessPathType.DATA; } path.addAll(builderPath); int slotId = slotReference.getExprId().asInt(); - slotToAccessPaths.put(slotId, new CollectAccessPathResult( - path, context.bottomFilter, TAccessPathType.DATA)); + slotToAccessPaths.put(slotId, new CollectAccessPathResult(path, context.bottomFilter, pathType)); return null; } if (dataType instanceof VariantType) { @@ -144,20 +147,37 @@ public Void visitSlotReference(SlotReference slotReference, CollectorContext con return null; } if (dataType instanceof NestedColumnPrunable) { + // A META path ending in NULL directly on the slot means "read the slot's null map". + // Check the physical column's nullability (via getOriginalColumn), NOT the slot's + // nullability which may be synthetic (e.g. from outer join). If the physical column + // has no null map or is unknown, suppress this path. + // (Field-level null paths like [s, field, NULL] were already validated upstream.) + if (context.type == TAccessPathType.META + && isFunctionNullCheckPath(context.accessPathBuilder.accessPath) + && !hasPhysicalNullMap(slotReference)) { + return null; + } context.accessPathBuilder.addPrefix(slotReference.getName().toLowerCase()); ImmutableList path = Utils.fastToImmutableList(context.accessPathBuilder.accessPath); int slotId = slotReference.getExprId().asInt(); slotToAccessPaths.put(slotId, new CollectAccessPathResult(path, context.bottomFilter, context.type)); + return null; } if (dataType.isStringLikeType()) { int slotId = slotReference.getExprId().asInt(); if (!context.accessPathBuilder.isEmpty()) { // Accessed via an offset-only function (e.g. length()) or null-check (IS NULL). // Builder already has "OFFSET"/"NULL" at the tail; add the column name as prefix. + // For META NULL paths, suppress when the physical column has no null map. + if (context.type == TAccessPathType.META + && isFunctionNullCheckPath(context.accessPathBuilder.accessPath) + && !hasPhysicalNullMap(slotReference)) { + return null; + } context.accessPathBuilder.addPrefix(slotReference.getName()); ImmutableList path = ImmutableList.copyOf(context.accessPathBuilder.accessPath); slotToAccessPaths.put(slotId, - new CollectAccessPathResult(path, context.bottomFilter, TAccessPathType.DATA)); + new CollectAccessPathResult(path, context.bottomFilter, context.type)); } else { // Direct access to the string column → record a DATA path so that any // concurrent offset-only path for the same slot is suppressed. @@ -170,17 +190,23 @@ public Void visitSlotReference(SlotReference slotReference, CollectorContext con // For any other nullable column type (e.g. INT, BIGINT) accessed via IS NULL / IS NOT NULL: // record the [col_name, NULL] path so NestedColumnPruning can emit null-only access paths. // Skip NestedColumnPrunable types (already handled above) and string types (handled above). + // Check getOriginalColumn() rather than slotReference.nullable(): the latter may be + // inflated by outer join, while the former reflects the physical column's null map. + // Only NULL paths need the null-map check; OFFSET paths don't depend on nullability. if (!(dataType instanceof NestedColumnPrunable) && !dataType.isStringLikeType() - && !context.accessPathBuilder.isEmpty() && slotReference.nullable()) { + && isFunctionNullCheckPath(context.accessPathBuilder.accessPath) + && hasPhysicalNullMap(slotReference)) { context.accessPathBuilder.addPrefix(slotReference.getName()); ImmutableList path = ImmutableList.copyOf(context.accessPathBuilder.accessPath); int slotId = slotReference.getExprId().asInt(); slotToAccessPaths.put(slotId, - new CollectAccessPathResult(path, context.bottomFilter, TAccessPathType.DATA)); + new CollectAccessPathResult(path, context.bottomFilter, context.type)); } // For any other nullable column type accessed directly (not via IS NULL / length / etc.): - // record a [col_name] full-access path so that when the column is also used via IS NULL, - // stripNullSuffixPaths correctly suppresses the null-only optimization. + // record a [col_name] full-access path. When the same column also has a META NULL + // path (e.g. from IS NULL), both paths are sent to BE. The presence of a DATA path + // signals that full column data is needed, preventing the BE from entering + // NULL_MAP_ONLY mode which would read only the null bitmap. if (!(dataType instanceof NestedColumnPrunable) && !dataType.isStringLikeType() && !(dataType instanceof VariantType) && context.accessPathBuilder.isEmpty() && slotReference.nullable()) { @@ -199,10 +225,28 @@ public Void visitLength(Length length, CollectorContext context) { // length() only needs the offset array, not the chars data. // Add ACCESS_STRING_OFFSET as a suffix so the path builder accumulates // e.g. ["str_col", "OFFSET"] or ["c_struct", "f3", "OFFSET"]. - if (arg.getDataType().isStringLikeType() && context.accessPathBuilder.isEmpty()) { + // + // CHAR is excluded: CHAR(N) is stored padded to N bytes per row (see BE + // OlapColumnDataConvertorChar::clone_and_padding), so the per-row length + // information available without reading the chars buffer is the padded + // length (always N), not the logical post-trim length expected by + // length(). There is no way to recover the logical length from offsets + // alone — the chars buffer must be scanned with strnlen() (BE + // shrink_padding_chars). Falling through to the default visit causes + // length() to read the column normally, which is correct. + // NOTE: arg.getDataType() is the resolved type at the leaf of any + // chained access (struct field, map subscript, array index), so this + // single check covers nested CHAR cases too. + if (arg.getDataType().isStringLikeType() && !arg.getDataType().isCharType() + && context.accessPathBuilder.isEmpty()) { + if (skipMetaPath) { + return arg.accept(this, + new CollectorContext(context.statementContext, false)); + } CollectorContext offsetContext = new CollectorContext(context.statementContext, context.bottomFilter); - offsetContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_STRING_OFFSET); + offsetContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_OFFSET); + offsetContext.setType(TAccessPathType.META); return arg.accept(this, offsetContext); } // fall through to default (recurse into children with fresh contexts) @@ -214,9 +258,14 @@ public Void visitMapSize(MapSize mapSize, CollectorContext context) { Expression arg = mapSize.child(); DataType argType = arg.getDataType(); if (argType.isMapType() && context.accessPathBuilder.isEmpty()) { + if (skipMetaPath) { + return arg.accept(this, + new CollectorContext(context.statementContext, false)); + } CollectorContext offsetContext = new CollectorContext(context.statementContext, context.bottomFilter); - offsetContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_STRING_OFFSET); + offsetContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_OFFSET); + offsetContext.setType(TAccessPathType.META); return arg.accept(this, offsetContext); } return visit(mapSize, context); @@ -229,9 +278,14 @@ public Void visitCardinality(Cardinality cardinality, CollectorContext context) // Arrays and maps share the same offset-array + data storage layout as strings on the BE. DataType argType = arg.getDataType(); if ((argType.isArrayType() || argType.isMapType()) && context.accessPathBuilder.isEmpty()) { + if (skipMetaPath) { + return arg.accept(this, + new CollectorContext(context.statementContext, false)); + } CollectorContext offsetContext = new CollectorContext(context.statementContext, context.bottomFilter); - offsetContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_STRING_OFFSET); + offsetContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_OFFSET); + offsetContext.setType(TAccessPathType.META); // cardinality(map_keys(m)) == cardinality(m) == cardinality(map_values(m)): // all three count map entries, so emit the same [map_col, OFFSET] path. Expression effectiveArg = (arg instanceof MapKeys || arg instanceof MapValues) @@ -271,9 +325,10 @@ public Void visitCast(Cast cast, CollectorContext context) { DataTypeAccessTree originTree = DataTypeAccessTree.of(cast.child().getDataType(), TAccessPathType.DATA); List replacePath = new ArrayList<>(context.accessPathBuilder.getPathList()); - if (originTree.replacePathByAnotherTree(castTree, replacePath, 0)) { + if (originTree.replacePathByAnotherTree(castTree, replacePath, 0, context.type)) { CollectorContext castContext = new CollectorContext(context.statementContext, context.bottomFilter); castContext.accessPathBuilder.accessPath.addAll(replacePath); + castContext.setType(context.type); return continueCollectAccessPath(cast.child(), castContext); } } @@ -313,6 +368,23 @@ public Void visitElementAt(ElementAt elementAt, CollectorContext context) { Expression fieldName = arguments.get(1); DataType fieldType = fieldName.getDataType(); if (fieldName.isLiteral() && (fieldType.isIntegerLikeType() || fieldType.isStringLikeType())) { + // Only emit META [s, field, NULL] when the selected field itself is nullable. + if (context.type == TAccessPathType.META + && isFunctionNullCheckPath(context.accessPathBuilder.getPathList())) { + StructField field = resolveStructField( + (StructType) first.getDataType(), fieldName); + if (field == null || !field.isNullable()) { + // Non-nullable leaf: no META NULL path for the field is needed. + // However the field must still appear in the type so pruneDataType + // preserves it — the filter expression still references + // element_at(s, 'f') IS NULL and won't be rewritten to s IS NULL. + // Fall through with a fresh DATA context to emit [s, f] DATA. + context = new CollectorContext( + context.statementContext, context.bottomFilter); + } + // Nullable leaf: fall through to add field prefix → META [s, field, NULL] + } + if (fieldType.isIntegerLikeType()) { int fieldIndex = ((Number) ((Literal) fieldName).getValue()).intValue(); List fields = ((StructType) first.getDataType()).getFields(); @@ -344,6 +416,7 @@ public Void visitMapKeys(MapKeys mapKeys, CollectorContext context) { = new CollectorContext(context.statementContext, context.bottomFilter); removeStarContext.accessPathBuilder.accessPath.addAll(suffixPath.subList(1, suffixPath.size())); removeStarContext.accessPathBuilder.addPrefix(AccessPathInfo.ACCESS_MAP_KEYS); + removeStarContext.setType(context.type); return continueCollectAccessPath(mapKeys.getArgument(0), removeStarContext); } context.accessPathBuilder.addPrefix(AccessPathInfo.ACCESS_MAP_KEYS); @@ -363,6 +436,7 @@ public Void visitMapValues(MapValues mapValues, CollectorContext context) { = new CollectorContext(context.statementContext, context.bottomFilter); removeStarContext.accessPathBuilder.accessPath.addAll(suffixPath.subList(1, suffixPath.size())); removeStarContext.accessPathBuilder.addPrefix(AccessPathInfo.ACCESS_MAP_VALUES); + removeStarContext.setType(context.type); return continueCollectAccessPath(mapValues.getArgument(0), removeStarContext); } context.accessPathBuilder.addPrefix(AccessPathInfo.ACCESS_MAP_VALUES); @@ -558,14 +632,43 @@ public Void visitIsNull(IsNull isNull, CollectorContext context) { // and nested access (struct_element(s, 'city') IS NULL → [s, city, NULL]). // For unrecognized expressions, the default visitor resets context, safely discarding NULL. if (arg.nullable() && context.accessPathBuilder.isEmpty()) { + if (skipMetaPath) { + return arg.accept(this, + new CollectorContext(context.statementContext, false)); + } CollectorContext nullContext = new CollectorContext(context.statementContext, context.bottomFilter); nullContext.accessPathBuilder.addSuffix(AccessPathInfo.ACCESS_NULL); + nullContext.setType(TAccessPathType.META); return continueCollectAccessPath(arg, nullContext); } return visit(isNull, context); } + /** + * Resolve the StructField selected by a constant int/string literal. + * Returns null when the selector is not a recognized literal or index is out of bounds. + */ + // VisibleForTesting + static StructField resolveStructField(StructType structType, Expression fieldExpr) { + if (!fieldExpr.isLiteral()) { + return null; + } + if (fieldExpr.getDataType().isIntegerLikeType()) { + int index = ((Number) ((Literal) fieldExpr).getValue()).intValue(); + List fields = structType.getFields(); + if (index >= 1 && index <= fields.size()) { + return fields.get(index - 1); + } + return null; + } + if (fieldExpr.getDataType().isStringLikeType()) { + String name = ((Literal) fieldExpr).getStringValue().toLowerCase(); + return structType.getField(name); + } + return null; + } + @Override public Void visitIf(If ifExpr, CollectorContext context) { if (isFunctionNullCheckPath(context.accessPathBuilder.accessPath)) { @@ -720,15 +823,29 @@ public boolean equals(Object o) { return false; } CollectAccessPathResult that = (CollectAccessPathResult) o; - return isPredicate == that.isPredicate && Objects.equals(path, that.path); + return isPredicate == that.isPredicate + && type == that.type + && Objects.equals(path, that.path); } @Override public int hashCode() { - return path.hashCode(); + return Objects.hash(path, isPredicate, type); } } + /** + * Check whether the physical column backing a SlotReference has a null map on disk. + * Uses the catalog Column's isAllowNull, not the slot's nullable() which may be + * inflated by outer join (via withNullable(true)). Returns false when the slot has + * no physical column (conservative: assume no null map). + */ + private static boolean hasPhysicalNullMap(SlotReference slot) { + return slot.getOriginalColumn() + .map(Column::isAllowNull) + .orElse(false); + } + // if the map type is changed, we can not prune the type, because the map type need distinct the keys, // e.g. select map_values(cast(map(3.0, 1, 3.1, 2) as map)); // the result is [2] because the keys: 3.0 and 3.1 will cast to 3 and the second entry remained. diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathPlanCollector.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathPlanCollector.java index f3e92216afc57f..6427e5c5e82917 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathPlanCollector.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/AccessPathPlanCollector.java @@ -67,6 +67,11 @@ public class AccessPathPlanCollector extends DefaultPlanVisitor { private Multimap allSlotToAccessPaths = LinkedHashMultimap.create(); private Map> scanSlotToAccessPaths = new LinkedHashMap<>(); + private boolean skipMetaPath; + + public void setSkipMetaPath(boolean skipMetaPath) { + this.skipMetaPath = skipMetaPath; + } public Map> collect(Plan root, StatementContext context) { root.accept(this, context); @@ -86,7 +91,7 @@ public Void visitLogicalGenerate(LogicalGenerate generate, State List output = generate.getGeneratorOutput(); AccessPathExpressionCollector exprCollector - = new AccessPathExpressionCollector(context, allSlotToAccessPaths, false); + = new AccessPathExpressionCollector(context, allSlotToAccessPaths, false, skipMetaPath); for (int i = 0; i < output.size(); i++) { Slot generatorOutput = output.get(i); Function function = generators.get(i); @@ -240,7 +245,7 @@ public Void visitLogicalGenerate(LogicalGenerate generate, State @Override public Void visitLogicalProject(LogicalProject project, StatementContext context) { AccessPathExpressionCollector exprCollector - = new AccessPathExpressionCollector(context, allSlotToAccessPaths, false); + = new AccessPathExpressionCollector(context, allSlotToAccessPaths, false, skipMetaPath); for (NamedExpression output : project.getProjects()) { // e.g. select element_at(s, 'city') from (select s from tbl)a; // we will not treat the inner `s` access all path @@ -366,7 +371,8 @@ public Void visitLogicalFileScan(LogicalFileScan fileScan, StatementContext cont } Collection accessPaths = allSlotToAccessPaths.get(slot.getExprId().asInt()); if (!accessPaths.isEmpty()) { - scanSlotToAccessPaths.put(slot, normalizeDataSkippingOnlyAccessPaths(accessPaths)); + scanSlotToAccessPaths.put( + slot, normalizeDataSkippingOnlyAccessPaths(accessPaths)); } } return null; @@ -380,7 +386,8 @@ public Void visitLogicalTVFRelation(LogicalTVFRelation tvfRelation, StatementCon } Collection accessPaths = allSlotToAccessPaths.get(slot.getExprId().asInt()); if (!accessPaths.isEmpty()) { - scanSlotToAccessPaths.put(slot, normalizeDataSkippingOnlyAccessPaths(accessPaths)); + scanSlotToAccessPaths.put( + slot, normalizeDataSkippingOnlyAccessPaths(accessPaths)); } } return null; @@ -402,7 +409,7 @@ private void collectByExpressions(Plan plan, StatementContext context) { private void collectByExpressions(Plan plan, StatementContext context, boolean bottomPredicate) { AccessPathExpressionCollector exprCollector - = new AccessPathExpressionCollector(context, allSlotToAccessPaths, bottomPredicate); + = new AccessPathExpressionCollector(context, allSlotToAccessPaths, bottomPredicate, skipMetaPath); for (Expression expression : plan.getExpressions()) { exprCollector.collect(expression); } @@ -413,7 +420,7 @@ static List normalizeDataSkippingOnlyAccessPaths( List normalizedAccessPaths = new ArrayList<>(); for (CollectAccessPathResult accessPath : accessPaths) { List path = accessPath.getPath(); - if (isDataSkippingOnlyAccessPath(path) && path.size() > 1) { + if (path.size() > 1 && accessPath.getType() == TAccessPathType.META) { // NULL/OFFSET suffixes are OLAP segment-reader-only optimizations. External // table and TVF readers use access paths as real nested field paths, so read // the referenced column/sub-column normally instead of sending a pseudo field. @@ -427,13 +434,4 @@ static List normalizeDataSkippingOnlyAccessPaths( } return normalizedAccessPaths; } - - private static boolean isDataSkippingOnlyAccessPath(List path) { - if (path.isEmpty()) { - return false; - } - String lastComponent = path.get(path.size() - 1); - return AccessPathInfo.ACCESS_NULL.equals(lastComponent) - || AccessPathInfo.ACCESS_STRING_OFFSET.equals(lastComponent); - } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/NestedColumnPruning.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/NestedColumnPruning.java index 8056ebd747c214..5086325d66a0c2 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/NestedColumnPruning.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/NestedColumnPruning.java @@ -41,6 +41,7 @@ import org.apache.doris.nereids.types.StructType; import org.apache.doris.nereids.types.VariantType; import org.apache.doris.qe.SessionVariable; +import org.apache.doris.thrift.DescriptorsConstants; import org.apache.doris.thrift.TAccessPathType; import org.apache.doris.thrift.TColumnAccessPath; import org.apache.doris.thrift.TDataAccessPath; @@ -93,9 +94,13 @@ public Plan rewriteRoot(Plan plan, JobContext jobContext) { return plan; } AccessPathPlanCollector collector = new AccessPathPlanCollector(); - Map> slotToAccessPaths = collector.collect(plan, statementContext); - Map slotToResult = pruneDataType(slotToAccessPaths, + // MV rewrite fragments must not be perturbed by metadata-only paths + // (NULL/OFFSET) that would otherwise prune the slot to a narrower type. + // Skip collecting them at the source instead of working around them downstream. + collector.setSkipMetaPath( jobContext.getCascadesContext().isMaterializedViewRewritePlanFragment()); + Map> slotToAccessPaths = collector.collect(plan, statementContext); + Map slotToResult = pruneDataType(slotToAccessPaths); if (!slotToResult.isEmpty()) { Map slotIdToPruneType = Maps.newLinkedHashMap(); @@ -205,21 +210,15 @@ private static boolean expressionContainsNullCheck(Expression expr) { } private static Map pruneDataType( - Map> slotToAccessPaths, - boolean skipDataSkippingOnlyAccessPath) { + Map> slotToAccessPaths) { Map result = new LinkedHashMap<>(); Map slotIdToAllAccessTree = new LinkedHashMap<>(); Map slotIdToPredicateAccessTree = new LinkedHashMap<>(); Map variantSlots = new LinkedHashMap<>(); - // Segment-wise comparison preserves the distinction between the key "a.b" and path a/b. - Comparator>> pathComparator = (left, right) -> { - int comparison = comparePathSegments(left.second, right.second); - if (comparison != 0) { - return comparison; - } - return left.first.compareTo(right.first); - }; + Comparator>> pathComparator = Comparator + .comparing((Pair> path) -> path.first) + .thenComparing(path -> path.second, NestedColumnPruning::comparePathComponents); Multimap>> allAccessPaths = TreeMultimap.create( Comparator.naturalOrder(), pathComparator); @@ -235,7 +234,8 @@ private static Map pruneDataType( for (CollectAccessPathResult collectAccessPathResult : collectAccessPathResults) { List path = collectAccessPathResult.getPath(); TAccessPathType pathType = collectAccessPathResult.getType(); - allAccessPaths.put(slot.getExprId().asInt(), Pair.of(pathType, path)); + Pair> allPath = Pair.of(pathType, path); + allAccessPaths.put(slot.getExprId().asInt(), allPath); if (collectAccessPathResult.isPredicate()) { predicateAccessPaths.put( slot.getExprId().asInt(), Pair.of(pathType, path) @@ -244,27 +244,15 @@ private static Map pruneDataType( } continue; } - if (skipDataSkippingOnlyAccessPath - && containsDataSkippingOnlyAccessPath(collectAccessPathResults)) { - // An MV rewrite child context optimizes a temporary plan fragment rather - // than the final plan. A nested metadata-only path such as - // [s, city, NULL] or [s, city, OFFSET] would otherwise prune the scan slot - // to only that nested field, while the final MV rewritten plan may still - // reuse the same slot as a full complex value or need another child. Drop - // access-info for the whole slot instead of just removing that path: - // predicate expressions inside this fragment still reference the original - // slot shape, so partial pruning after deleting the predicate-only path - // could make the fragment itself inconsistent. - continue; - } for (CollectAccessPathResult collectAccessPathResult : collectAccessPathResults) { List path = collectAccessPathResult.getPath(); TAccessPathType pathType = collectAccessPathResult.getType(); + Pair> allPath = Pair.of(pathType, path); DataTypeAccessTree allAccessTree = slotIdToAllAccessTree.computeIfAbsent( - slot, i -> DataTypeAccessTree.ofRoot(slot, pathType) + slot, i -> DataTypeAccessTree.ofRoot(slot, allPath.first) ); - allAccessTree.setAccessByPath(path, 0, pathType); - allAccessPaths.put(slot.getExprId().asInt(), Pair.of(pathType, path)); + allAccessTree.setAccessByPath(allPath.second, 0, allPath.first); + allAccessPaths.put(slot.getExprId().asInt(), allPath); if (collectAccessPathResult.isPredicate()) { DataTypeAccessTree predicateAccessTree = slotIdToPredicateAccessTree.computeIfAbsent( @@ -278,105 +266,27 @@ && containsDataSkippingOnlyAccessPath(collectAccessPathResults)) { } } - // second: build non-predicate access paths + // phase 1.5: for slots with meta paths, expand map-star paths before building final + // access path lists. This turns map.*.OFFSET into precise map.KEYS + + // map.VALUES.OFFSET paths instead of broad map.* access. for (Entry kv : slotIdToAllAccessTree.entrySet()) { Slot slot = kv.getKey(); DataTypeAccessTree accessTree = kv.getValue(); - DataType prunedDataType = accessTree.pruneDataType().orElse(slot.getDataType()); - - if (slot.getDataType().isStringLikeType()) { - if (accessTree.hasStringOffsetOnlyAccess()) { - if (skipDataSkippingOnlyAccessPath) { - continue; - } - // Offset-only access (e.g. length(str_col)): type stays varchar, - // but we must still send the access path to BE so it skips the char data. - stripExactCoveredDataSkippingSuffixPaths(slot, allAccessPaths, allAccessPaths); - stripNullSuffixPaths(slot, allAccessPaths); - List allPaths = buildColumnAccessPaths(slot, allAccessPaths); - result.put(slot.getExprId().asInt(), - new AccessPathInfo(slot.getDataType(), allPaths, new ArrayList<>())); - } else if (accessTree.hasNullCheckOnlyAccess()) { - if (skipDataSkippingOnlyAccessPath) { - continue; - } - // Null-check-only access (e.g. str_col IS NULL): type stays varchar, - // but we send [col, NULL] access path so BE only reads the null flag. - List allPaths = buildColumnAccessPaths(slot, allAccessPaths); - result.put(slot.getExprId().asInt(), - new AccessPathInfo(slot.getDataType(), allPaths, new ArrayList<>())); - } - // direct access (accessAll=true) or other: skip — no type change, no access paths needed. + if (!accessTree.hasOffsetPath() && !accessTree.hasNullPath()) { continue; } + // Expand both sets so all-path and predicate-path routing stay consistent. + expandMapStarPaths(slot, allAccessPaths); + expandMapStarPaths(slot, predicateAccessPaths); + } - if ((slot.getDataType().isArrayType() || slot.getDataType().isMapType()) - && accessTree.hasStringOffsetOnlyAccess()) { - if (skipDataSkippingOnlyAccessPath) { - continue; - } - // Offset-only access (e.g. length(arr_col) / length(map_col)): type stays unchanged, - // but we must send the OFFSET access path to BE so it skips element/key-value data. - List allPaths = buildColumnAccessPaths(slot, allAccessPaths); - result.put(slot.getExprId().asInt(), - new AccessPathInfo(slot.getDataType(), allPaths, new ArrayList<>())); - continue; - } - - // Null-check-only access (e.g. col IS NULL / col IS NOT NULL): type stays unchanged, - // but we must send the [col, NULL] access path to BE so it only reads the null flag. - if (accessTree.hasNullCheckOnlyAccess()) { - if (skipDataSkippingOnlyAccessPath) { - continue; - } - List allPaths = buildColumnAccessPaths(slot, allAccessPaths); - result.put(slot.getExprId().asInt(), - new AccessPathInfo(slot.getDataType(), allPaths, new ArrayList<>())); - continue; - } - - if (slot.getDataType().isMapType() && accessTree.hasMapValueOffsetOnlyAccess()) { - if (skipDataSkippingOnlyAccessPath) { - continue; - } - // length(map_col['key']): keys read in full (element lookup) + values offset-only. - // Emit [col, KEYS] and [col, VALUES, OFFSET] directly instead of the collected - // [col, *, OFFSET] path which the BE cannot interpret for split key/value access. - String colName = slot.getName().toLowerCase(); - TDataAccessPath keysDataPath = new TDataAccessPath(); - keysDataPath.setPath( - new ArrayList<>(ImmutableList.of(colName, AccessPathInfo.ACCESS_MAP_KEYS))); - TColumnAccessPath keysColumnPath = new TColumnAccessPath(TAccessPathType.DATA); - keysColumnPath.setDataAccessPath(keysDataPath); - - TDataAccessPath valsOffsetDataPath = new TDataAccessPath(); - valsOffsetDataPath.setPath(new ArrayList<>(ImmutableList.of( - colName, AccessPathInfo.ACCESS_MAP_VALUES, AccessPathInfo.ACCESS_STRING_OFFSET))); - TColumnAccessPath valsOffsetColumnPath = new TColumnAccessPath(TAccessPathType.DATA); - valsOffsetColumnPath.setDataAccessPath(valsOffsetDataPath); - - result.put(slot.getExprId().asInt(), new AccessPathInfo( - slot.getDataType(), - ImmutableList.of(keysColumnPath, valsOffsetColumnPath), - new ArrayList<>())); - continue; - } - - // If a field is read in full, its metadata-only NULL/OFFSET access is redundant - // for any data type: e.g. [s] covers both [s.NULL] and [s.OFFSET]. - stripExactCoveredDataSkippingSuffixPaths(slot, allAccessPaths, allAccessPaths); - - // Strip OFFSET-suffix paths when a non-OFFSET path covers the same nested field or - // container. The overlapping array/map container may live under the root slot itself - // or under a nested struct field, so compare against the actual nested prefix instead - // of gating this logic on the root slot type. - stripCoveredOffsetSuffixPaths(slot, allAccessPaths, allAccessPaths); - - // Strip NULL-suffix paths when a non-NULL path also exists for the same slot. - // E.g. `SELECT col FROM t WHERE col IS NULL` — full data is needed, NULL path is redundant. - stripNullSuffixPaths(slot, allAccessPaths); + // second: build non-predicate access paths + for (Entry kv : slotIdToAllAccessTree.entrySet()) { + Slot slot = kv.getKey(); + DataTypeAccessTree accessTree = kv.getValue(); + DataType prunedDataType = accessTree.pruneDataType().orElse(slot.getDataType()); List allPaths = buildColumnAccessPaths(slot, allAccessPaths); - if (shouldSkipAccessInfo(slot, prunedDataType, allPaths, predicateAccessPaths)) { + if (shouldSkipAccessInfo(slot, prunedDataType, allPaths)) { continue; } result.put(slot.getExprId().asInt(), @@ -390,18 +300,14 @@ && containsDataSkippingOnlyAccessPath(collectAccessPathResults)) { new AccessPathInfo(slot.getDataType(), allPaths, new ArrayList<>())); } - // third: build predicate access path + // third: build predicate access paths for (Entry kv : slotIdToPredicateAccessTree.entrySet()) { Slot slot = kv.getKey(); - stripExactCoveredDataSkippingSuffixPaths(slot, predicateAccessPaths, allAccessPaths); - stripCoveredOffsetSuffixPaths(slot, predicateAccessPaths, allAccessPaths); - stripCoveredArrayNullSuffixPaths(slot, predicateAccessPaths, allAccessPaths); - stripNullSuffixPaths(slot, predicateAccessPaths); List predicatePaths = buildColumnAccessPaths(slot, predicateAccessPaths); AccessPathInfo accessPathInfo = result.get(slot.getExprId().asInt()); if (accessPathInfo != null) { - retainPredicatePathsInFinalAllAccessPaths( + addPredicatePathsToFinalAllAccessPaths( predicatePaths, accessPathInfo.getAllAccessPaths()); accessPathInfo.getPredicateAccessPaths().addAll(predicatePaths); } @@ -413,7 +319,7 @@ && containsDataSkippingOnlyAccessPath(collectAccessPathResults)) { buildColumnAccessPaths(slot, predicateAccessPaths); AccessPathInfo accessPathInfo = result.get(slot.getExprId().asInt()); if (accessPathInfo != null) { - retainPredicatePathsInFinalAllAccessPaths( + addPredicatePathsToFinalAllAccessPaths( predicatePaths, accessPathInfo.getAllAccessPaths()); accessPathInfo.getPredicateAccessPaths().addAll(predicatePaths); } @@ -422,479 +328,60 @@ && containsDataSkippingOnlyAccessPath(collectAccessPathResults)) { return result; } - private static boolean containsDataSkippingOnlyAccessPath( - List collectAccessPathResults) { - for (CollectAccessPathResult collectAccessPathResult : collectAccessPathResults) { - if (isDataSkippingOnlyAccessPath(collectAccessPathResult.getPath())) { - return true; - } - } - return false; - } - - private static boolean isDataSkippingOnlyAccessPath(List path) { - if (path.isEmpty()) { - return false; - } - String lastComponent = path.get(path.size() - 1); - return AccessPathInfo.ACCESS_NULL.equals(lastComponent) - || AccessPathInfo.ACCESS_STRING_OFFSET.equals(lastComponent); - } - - /** - * Decide whether an OFFSET-suffix path can be removed because another non-OFFSET path - * already covers the same container. - * - *

For map element_at paths, {@code *} means "read keys fully, then follow the rest of - * the path on the value side". So a VALUES path can cover the value-side OFFSET access, - * but it does NOT cover the key lookup requirement. In that case we remove the OFFSET path - * and add a KEYS-only path instead. - */ - private static OffsetPathRewrite analyzeOffsetPathRewrite( - DataType slotType, List path, List> nonOffsetPaths) { - if (path.isEmpty() - || !AccessPathInfo.ACCESS_STRING_OFFSET.equals(path.get(path.size() - 1))) { - return OffsetPathRewrite.keep(); - } - List prefix = path.subList(0, path.size() - 1); - return analyzePrefixCoverage(slotType, prefix, nonOffsetPaths); - } - - private static OffsetPathRewrite analyzePrefixCoverage( - DataType slotType, List prefix, List> nonOffsetPaths) { - List> supplementalPaths = new ArrayList<>(); - for (List nonOffset : nonOffsetPaths) { - OffsetPathRewrite candidate = compareOffsetPrefixCoverage(slotType, prefix, nonOffset); - if (!candidate.shouldRemoveOffsetPath()) { - continue; - } - if (candidate.getSupplementalPaths().isEmpty()) { - return OffsetPathRewrite.remove(); - } - supplementalPaths.addAll(candidate.getSupplementalPaths()); - } - if (supplementalPaths.isEmpty()) { - return OffsetPathRewrite.keep(); - } - return OffsetPathRewrite.rewriteWithSupplementalPaths(supplementalPaths); - } - - /** - * Remove OFFSET-only paths from {@code targetAccessPaths} when data paths in - * {@code coveringAccessPaths} already read the same array/map/string container or a child - * under it. - * - *

Examples: - *

    - *
  • {@code [arr.OFFSET, arr.*.field]} becomes {@code [arr.*.field]} because the array - * child read must keep BE on the normal data iterator path.
  • - *
  • {@code [map.*.OFFSET, map.VALUES]} becomes {@code [map.KEYS, map.VALUES]} because - * {@code map['k']} still needs full keys for lookup, while values cover the offset.
  • - *
- */ - private static void stripCoveredOffsetSuffixPaths( - Slot slot, Multimap>> targetAccessPaths, - Multimap>> coveringAccessPaths) { - int slotId = slot.getExprId().asInt(); - Collection>> targetPaths = targetAccessPaths.get(slotId); - if (targetPaths.isEmpty()) { - return; - } - - List> nonOffsetPaths = new ArrayList<>(); - for (Pair> p : coveringAccessPaths.get(slotId)) { - List path = p.second; - if (path.isEmpty() - || !AccessPathInfo.ACCESS_STRING_OFFSET.equals(path.get(path.size() - 1))) { - nonOffsetPaths.add(path); - } - } - for (Pair> p : targetPaths) { - List path = p.second; - if (path.isEmpty() - || !AccessPathInfo.ACCESS_STRING_OFFSET.equals(path.get(path.size() - 1))) { - nonOffsetPaths.add(path); - } - } - - List>> pathsToRemove = new ArrayList<>(); - List>> pathsToAdd = new ArrayList<>(); - for (Pair> p : new ArrayList<>(targetPaths)) { - OffsetPathRewrite rewrite = analyzeOffsetPathRewrite( - slot.getDataType(), p.second, nonOffsetPaths); - if (!rewrite.shouldRemoveOffsetPath()) { - continue; - } - pathsToRemove.add(p); - for (List supplementalPath : rewrite.getSupplementalPaths()) { - pathsToAdd.add(Pair.of(p.first, supplementalPath)); - } - } - targetPaths.removeAll(pathsToRemove); - targetPaths.addAll(pathsToAdd); - } - - /** - * Remove array NULL-only paths from {@code targetAccessPaths} when another path already reads - * the same array container or data under it. This mirrors OFFSET coverage because an array - * element/data read must not be combined with an array NULL_MAP_ONLY read for the same prefix. - * - *

Examples: - *

    - *
  • {@code [map.VALUES.NULL, map.VALUES.*.field]} becomes - * {@code [map.VALUES.*.field]}.
  • - *
  • {@code [map.*.NULL, map.VALUES.*.field]} becomes - * {@code [map.KEYS, map.VALUES.*.field]} so map lookup keys are still available.
  • - *
- */ - private static void stripCoveredArrayNullSuffixPaths( - Slot slot, Multimap>> targetAccessPaths, - Multimap>> coveringAccessPaths) { - int slotId = slot.getExprId().asInt(); - Collection>> targetPaths = targetAccessPaths.get(slotId); - if (targetPaths.isEmpty()) { - return; - } - - List> nonNullPaths = new ArrayList<>(); - for (Pair> p : coveringAccessPaths.get(slotId)) { - List path = p.second; - if (path.isEmpty() || !AccessPathInfo.ACCESS_NULL.equals(path.get(path.size() - 1))) { - nonNullPaths.add(path); + private static int comparePathComponents(List left, List right) { + int commonSize = Math.min(left.size(), right.size()); + for (int i = 0; i < commonSize; i++) { + int result = left.get(i).compareTo(right.get(i)); + if (result != 0) { + return result; } } - for (Pair> p : targetPaths) { - List path = p.second; - if (path.isEmpty() || !AccessPathInfo.ACCESS_NULL.equals(path.get(path.size() - 1))) { - nonNullPaths.add(path); - } - } - - List>> pathsToRemove = new ArrayList<>(); - List>> pathsToAdd = new ArrayList<>(); - for (Pair> p : new ArrayList<>(targetPaths)) { - List path = p.second; - if (path.isEmpty() || !AccessPathInfo.ACCESS_NULL.equals(path.get(path.size() - 1))) { - continue; - } - List prefix = path.subList(0, path.size() - 1); - Optional prefixType = dataTypeAtPath(slot.getDataType(), prefix); - if (!prefixType.isPresent() || !prefixType.get().isArrayType()) { - continue; - } - OffsetPathRewrite rewrite = analyzePrefixCoverage(slot.getDataType(), prefix, nonNullPaths); - if (!rewrite.shouldRemoveOffsetPath()) { - continue; - } - pathsToRemove.add(p); - for (List supplementalPath : rewrite.getSupplementalPaths()) { - pathsToAdd.add(Pair.of(p.first, supplementalPath)); - } - } - targetPaths.removeAll(pathsToRemove); - targetPaths.addAll(pathsToAdd); + return Integer.compare(left.size(), right.size()); } /** - * Remove exact metadata-only NULL/OFFSET paths when the same field is read in full. - * This rule is type-agnostic: once {@code s} itself is accessed, {@code s.NULL} and - * {@code s.OFFSET} are redundant and unsafe to keep with the full data path. + * Keep final allAccessPaths as a superset of non-metadata predicateAccessPaths. Predicate + * paths are collected from filter expressions first, but final all-path construction may later + * collapse ordinary paths to whole-column access. BE complex readers use allAccessPaths to + * decide which data sub-iterators can be pruned; any predicate data path missing from + * allAccessPaths can therefore make predicate reads disagree with pruning. * - *

Examples: - *

    - *
  • {@code [str_col, str_col.NULL]} becomes {@code [str_col]}.
  • - *
  • {@code [arr, arr.OFFSET]} becomes {@code [arr]}.
  • - *
  • {@code [map.*, map.*.OFFSET]} becomes {@code [map.*]}.
  • - *
+ * Metadata predicate paths are already retained in allAccessPaths while the access trees are + * built, so only missing data paths need to be added here. */ - private static void stripExactCoveredDataSkippingSuffixPaths( - Slot slot, Multimap>> targetAccessPaths, - Multimap>> coveringAccessPaths) { - int slotId = slot.getExprId().asInt(); - Collection>> targetPaths = targetAccessPaths.get(slotId); - if (targetPaths.isEmpty()) { - return; - } - - List> fullAccessPaths = new ArrayList<>(); - for (Pair> p : coveringAccessPaths.get(slotId)) { - if (!isDataSkippingOnlyAccessPath(p.second)) { - fullAccessPaths.add(p.second); - } - } - for (Pair> p : targetPaths) { - if (!isDataSkippingOnlyAccessPath(p.second)) { - fullAccessPaths.add(p.second); - } - } - - List>> pathsToRemove = new ArrayList<>(); - for (Pair> p : targetPaths) { - List path = p.second; - if (!isDataSkippingOnlyAccessPath(path)) { - continue; - } - List prefix = path.subList(0, path.size() - 1); - for (List fullAccessPath : fullAccessPaths) { - if (pathCoversPrefix(fullAccessPath, prefix)) { - pathsToRemove.add(p); - break; - } - } - } - targetPaths.removeAll(pathsToRemove); - } - - private static Optional dataTypeAtPath(DataType slotType, List path) { - if (path.isEmpty()) { - return Optional.empty(); - } - DataType currentType = slotType; - for (int i = 1; i < path.size(); i++) { - String component = path.get(i); - if (currentType.isStructType()) { - StructField field = ((StructType) currentType).getField(component); - if (field == null) { - return Optional.empty(); - } - currentType = field.getDataType(); - } else if (currentType.isArrayType()) { - if (!AccessPathInfo.ACCESS_ALL.equals(component)) { - return Optional.empty(); - } - currentType = ((ArrayType) currentType).getItemType(); - } else if (currentType.isMapType()) { - currentType = descendMapType((MapType) currentType, component); - } else { - return Optional.empty(); - } - } - return Optional.of(currentType); - } - - private static OffsetPathRewrite compareOffsetPrefixCoverage( - DataType slotType, List prefix, List nonOffset) { - if (nonOffset.isEmpty()) { - return OffsetPathRewrite.remove(); - } - int minLen = Math.min(prefix.size(), nonOffset.size()); - List> supplementalPaths = new ArrayList<>(); - DataType currentType = slotType; - for (int i = 0; i < minLen; i++) { - String prefixComponent = prefix.get(i); - String nonOffsetComponent = nonOffset.get(i); - if (i == 0) { - if (!prefixComponent.equals(nonOffsetComponent)) { - return OffsetPathRewrite.keep(); - } - continue; - } - if (currentType.isStructType()) { - if (!prefixComponent.equals(nonOffsetComponent)) { - return OffsetPathRewrite.keep(); - } - StructField field = ((StructType) currentType).getField(prefixComponent); - if (field == null) { - return OffsetPathRewrite.keep(); - } - currentType = field.getDataType(); - continue; - } - if (currentType.isArrayType()) { - if (!prefixComponent.equals(nonOffsetComponent) - || !AccessPathInfo.ACCESS_ALL.equals(prefixComponent)) { - return OffsetPathRewrite.keep(); - } - currentType = ((ArrayType) currentType).getItemType(); - continue; - } - if (currentType.isMapType()) { - MapType mapType = (MapType) currentType; - if (prefixComponent.equals(nonOffsetComponent)) { - currentType = descendMapType(mapType, prefixComponent); - continue; - } - if (AccessPathInfo.ACCESS_ALL.equals(prefixComponent) - && AccessPathInfo.ACCESS_MAP_VALUES.equals(nonOffsetComponent)) { - supplementalPaths.add(buildMapKeysOnlyPath(prefix, i)); - currentType = mapType.getValueType(); - continue; - } - if (AccessPathInfo.ACCESS_MAP_VALUES.equals(prefixComponent) - && AccessPathInfo.ACCESS_ALL.equals(nonOffsetComponent)) { - currentType = mapType.getValueType(); - continue; - } - if (AccessPathInfo.ACCESS_MAP_KEYS.equals(prefixComponent) - && AccessPathInfo.ACCESS_ALL.equals(nonOffsetComponent)) { - currentType = mapType.getKeyType(); - continue; - } - return OffsetPathRewrite.keep(); - } - if (!prefixComponent.equals(nonOffsetComponent)) { - return OffsetPathRewrite.keep(); + private static void addPredicatePathsToFinalAllAccessPaths( + List predicatePaths, List allPaths) { + for (TColumnAccessPath predicatePath : predicatePaths) { + if (!isMetaPath(predicatePath) && !isCoveredByAllPath(predicatePath, allPaths)) { + allPaths.add(predicatePath); } } - if (supplementalPaths.isEmpty()) { - return OffsetPathRewrite.remove(); - } - return OffsetPathRewrite.rewriteWithSupplementalPaths(supplementalPaths); - } - - private static DataType descendMapType(MapType mapType, String component) { - if (AccessPathInfo.ACCESS_MAP_KEYS.equals(component)) { - return mapType.getKeyType(); - } - return mapType.getValueType(); - } - - private static List buildMapKeysOnlyPath(List prefix, int mapTokenIndex) { - List keyPath = new ArrayList<>(prefix.subList(0, mapTokenIndex)); - keyPath.add(AccessPathInfo.ACCESS_MAP_KEYS); - return keyPath; } - private static final class OffsetPathRewrite { - private static final OffsetPathRewrite KEEP = new OffsetPathRewrite(false, ImmutableList.of()); - private static final OffsetPathRewrite REMOVE = new OffsetPathRewrite(true, ImmutableList.of()); - - private final boolean removeOffsetPath; - private final List> supplementalPaths; - - private OffsetPathRewrite(boolean removeOffsetPath, List> supplementalPaths) { - this.removeOffsetPath = removeOffsetPath; - this.supplementalPaths = supplementalPaths; - } - - private static OffsetPathRewrite keep() { - return KEEP; - } - - private static OffsetPathRewrite remove() { - return REMOVE; - } - - private static OffsetPathRewrite rewriteWithSupplementalPaths(List> supplementalPaths) { - return new OffsetPathRewrite(true, ImmutableList.copyOf(supplementalPaths)); - } - - private boolean shouldRemoveOffsetPath() { - return removeOffsetPath; - } - - private List> getSupplementalPaths() { - return supplementalPaths; - } - } - - /** - * Strip NULL-suffix paths that are redundant because a non-NULL path reads child - * data below the same prefix or reads an OFFSET path over the same prefix. - * - *

Examples: - *

    - *
  • {@code [struct_col.NULL, struct_col.city]} becomes {@code [struct_col.city]}.
  • - *
  • {@code [str_col.NULL, str_col.OFFSET]} becomes {@code [str_col.OFFSET]} because - * the offset read can provide nullness for variable-length columns.
  • - *
- * - *

A parent NULL path must also be removed when any child path is required under the - * same prefix, e.g. [struct_col, NULL] with [struct_col, city]. This looks like the - * parent null map may still be useful for predicates, but it cannot be kept in - * allAccessPaths with the current BE iterator contract: Struct/Array/Map iterators - * treat a leading NULL sub-path as NULL_MAP_ONLY and skip all children. If FE kept - * [struct_col.NULL, struct_col.city] in allAccessPaths, BE would read only the - * struct null map and default-fill city instead of routing the city child iterator. - * When the NULL path is removed from allAccessPaths, it must also be removed from - * predicateAccessPaths so the BE can rely on predicate paths being a subset of all - * paths. The normal nullable container read materializes the parent null map - * together with required children. - */ - private static void stripNullSuffixPaths( - Slot slot, Multimap>> allAccessPaths) { - int slotId = slot.getExprId().asInt(); - Collection>> slotPaths = allAccessPaths.get(slotId); - - List>> toRemove = new ArrayList<>(); - for (Pair> p : slotPaths) { - List path = p.second; - if (path.isEmpty() || !AccessPathInfo.ACCESS_NULL.equals(path.get(path.size() - 1))) { - continue; - } - // Prefix is the column/subcolumn path without the trailing NULL suffix. - // A non-NULL path that equals this prefix means the same column/subcolumn - // is read in full, making the NULL-only path redundant. - // An OFFSET-suffix path over the same prefix is also enough for the BE to - // derive null-ness for variable-length columns, so [col.NULL] is redundant - // when [col.OFFSET] already exists. - List prefix = path.subList(0, path.size() - 1); - boolean covered = false; - for (Pair> q : slotPaths) { - List other = q.second; - if (other.isEmpty() - || AccessPathInfo.ACCESS_NULL.equals(other.get(other.size() - 1))) { - continue; - } - if (other.equals(prefix)) { - covered = true; - break; - } - if (hasStrictPrefix(other, prefix)) { - covered = true; - break; - } - if (other.size() == prefix.size() + 1 - && AccessPathInfo.ACCESS_STRING_OFFSET.equals(other.get(other.size() - 1)) - && other.subList(0, prefix.size()).equals(prefix)) { - covered = true; - break; - } - } - if (covered) { - toRemove.add(p); + private static boolean isCoveredByAllPath(TColumnAccessPath predicatePath, List allPaths) { + for (TColumnAccessPath allPath : allPaths) { + if (allPath.getType() == predicatePath.getType() + && isPrefixPath(getAccessPathList(allPath), getAccessPathList(predicatePath))) { + return true; } } - for (Pair> r : toRemove) { - allAccessPaths.remove(slotId, r); - } + return false; } - /** - * Keep predicate access paths as a subset of final all access paths after NULL/OFFSET cleanup. - * Predicate paths are built from filter expressions first, but later all-path rewrites may drop - * metadata-only paths or collapse paths to whole-column access. Any predicate path not present - * in final all paths must be removed before sending access info to BE. - * - *

Examples: - *

    - *
  • All paths {@code [s]}, predicate paths {@code [s.city.NULL]} becomes no predicate - * paths after parent NULL removal.
  • - *
  • All paths {@code [s.city.NULL, s.zip]}, predicate paths - * {@code [s.NULL, s.city.NULL]} becomes {@code [s.city.NULL]}.
  • - *
- */ - private static void retainPredicatePathsInFinalAllAccessPaths( - List predicatePaths, List allPaths) { - if (predicatePaths.isEmpty()) { - return; + private static boolean isPrefixPath(List prefix, List path) { + if (prefix.size() > path.size()) { + return false; } - - List toRemove = new ArrayList<>(); - for (TColumnAccessPath predicatePath : predicatePaths) { - if (!allPaths.contains(predicatePath)) { - toRemove.add(predicatePath); + for (int i = 0; i < prefix.size(); ++i) { + if (!prefix.get(i).equals(path.get(i))) { + return false; } } - predicatePaths.removeAll(toRemove); + return true; } - private static boolean hasStrictPrefix(List path, List prefix) { - return path.size() > prefix.size() && path.subList(0, prefix.size()).equals(prefix); - } - - private static boolean pathCoversPrefix(List path, List prefix) { - return prefix.size() >= path.size() && prefix.subList(0, path.size()).equals(path); + private static boolean isMetaPath(TColumnAccessPath path) { + return path.getType() == TAccessPathType.META; } private static List buildColumnAccessPaths( @@ -911,12 +398,14 @@ private static List buildColumnAccessPaths( dataAccessPath.setPath(new ArrayList<>(pathInfo.second)); TColumnAccessPath accessPath = new TColumnAccessPath(TAccessPathType.DATA); accessPath.setDataAccessPath(dataAccessPath); + accessPath.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); paths.add(accessPath); } else { TMetaAccessPath dataAccessPath = new TMetaAccessPath(); dataAccessPath.setPath(new ArrayList<>(pathInfo.second)); TColumnAccessPath accessPath = new TColumnAccessPath(TAccessPathType.META); accessPath.setMetaAccessPath(dataAccessPath); + accessPath.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); paths.add(accessPath); } // only retain access the whole root @@ -938,34 +427,24 @@ private static List buildColumnAccessPaths( } else { accessPath.setMetaAccessPath(new TMetaAccessPath(ImmutableList.of(wholeColumnName))); } + accessPath.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); return new ArrayList<>(ImmutableList.of(accessPath)); } return paths; } - private static int comparePathSegments(List left, List right) { - int commonLength = Math.min(left.size(), right.size()); - for (int index = 0; index < commonLength; index++) { - int comparison = left.get(index).compareTo(right.get(index)); - if (comparison != 0) { - return comparison; - } - } - return Integer.compare(left.size(), right.size()); - } - private static boolean shouldSkipAccessInfo( - Slot slot, DataType prunedDataType, List allPaths, - Multimap>> predicateAccessPaths) { + Slot slot, DataType prunedDataType, List allPaths) { if (!prunedDataType.equals(slot.getDataType())) { return false; } if (slot.getDataType() instanceof NestedColumnPrunable || slot.getDataType().isVariantType()) { return false; } - if (!predicateAccessPaths.get(slot.getExprId().asInt()).isEmpty()) { - return false; - } + + // Only scalar / string-like types reach here (NestedColumnPrunable and Variant + // returned false above). A single [col] path means the entire column is read; + // no access info needs to be sent to BE. if (allPaths.size() != 1) { return false; } @@ -989,15 +468,18 @@ public static class DataTypeAccessTree { // if access 's.a.b' the node 's' and 'a' has accessPartialChild, and node 'b' has accessAll private boolean accessPartialChild; private boolean accessAll; - // True when this string-typed node is accessed ONLY via the offset array - // (e.g. length(str_col) or length(element_at(c_struct,'f3'))). - // When this flag is set and accessAll is NOT set, pruneDataType() returns BigIntType - // to signal that the BE only needs to read the offset array, not the chars data. - private boolean isStringOffsetOnly; - // True when this column node is accessed ONLY via IS NULL / IS NOT NULL. - // When this flag is set and accessAll is NOT set, the BE only needs to read the null flag, - // not the actual column data. - private boolean isNullCheckOnly; + // Cached marker set by setAccessByPath() when a path component is OFFSET. + // Avoids scanning the multimap: hasStringOffsetOnlyAccess() reads this flag in + // O(1) instead of checking every path for an OFFSET suffix. Used for array, map, + // and string-like types (they share offset-based storage in BE). + // When set without accessAll, pruneDataType() keeps the node's type so that BE + // reads only the offset structure, skipping element / key-value / chars data. + private boolean hasOffsetPath; + // Cached marker set by setAccessByPath() when a path component is NULL. + // Same purpose as hasOffsetPath — O(1) flag read instead of multimap scan + // in hasNullCheckOnlyAccess(). When set without accessAll, BE reads only the + // null bitmap, skipping actual column data. + private boolean hasNullPath; // for the future, only access the meta of the column, // e.g. `is not null` can only access the column's offset, not need to read the data private TAccessPathType pathType; @@ -1040,80 +522,33 @@ public Map getChildren() { } /** - * True when a MAP column is accessed as {@code length(map_col['key'])}: the keys must - * be read in full (for the element lookup) while the values only need the offset array - * (since only their length, not their content, is used). - * Expected access paths: [col, KEYS] and [col, VALUES, OFFSET]. + * recursively search in the tree, if any node hasOffsetPath */ - public boolean hasMapValueOffsetOnlyAccess() { - if (!isRoot) { - return false; - } - DataTypeAccessTree child = children.values().iterator().next(); - if (!child.type.isMapType() || child.accessAll) { - return false; - } - DataTypeAccessTree keysChild = child.children.get(AccessPathInfo.ACCESS_MAP_KEYS); - DataTypeAccessTree valsChild = child.children.get(AccessPathInfo.ACCESS_MAP_VALUES); - // Keys must be fully accessed (element-at lookup). - if (!keysChild.accessAll) { - return false; - } - // Values must be accessed offset-only (no deeper element reads). - if (!valsChild.isStringOffsetOnly || valsChild.accessAll) { - return false; - } - if (valsChild.type.isStringLikeType()) { - // String value: accessAll check above is sufficient. + public boolean hasOffsetPath() { + if (hasOffsetPath) { return true; } - if (valsChild.type.isArrayType()) { - // Array value (e.g. MAP>): verify no element was read directly - // (e.g. map_col['k'][0] would set allChild.accessAll=true). - DataTypeAccessTree allChild = valsChild.children.get(AccessPathInfo.ACCESS_ALL); - return !allChild.accessAll && !allChild.accessPartialChild; - } - return true; - } - - /** True when the column is accessed ONLY via the offset array (e.g. length(str_col), - * length(arr_col), length(map_col)), meaning the type must not change but an access - * path still needs to be sent to BE so it can skip the char/element data. */ - public boolean hasStringOffsetOnlyAccess() { - if (isRoot) { - DataTypeAccessTree child = children.values().iterator().next(); - if (!child.isStringOffsetOnly || child.accessAll) { - return false; - } - if (child.type.isStringLikeType()) { + for (DataTypeAccessTree child : children.values()) { + if (child.hasOffsetPath()) { return true; } - if (child.type.isArrayType()) { - // True only if no element was accessed (element_at / explode etc.) - DataTypeAccessTree allChild = child.children.get(AccessPathInfo.ACCESS_ALL); - return !allChild.accessAll && !allChild.accessPartialChild; - } - if (child.type.isMapType()) { - // True only if neither keys nor values were accessed directly - DataTypeAccessTree keysChild = child.children.get(AccessPathInfo.ACCESS_MAP_KEYS); - DataTypeAccessTree valsChild = child.children.get(AccessPathInfo.ACCESS_MAP_VALUES); - return !keysChild.accessAll && !keysChild.accessPartialChild - && !valsChild.accessAll && !valsChild.accessPartialChild; - } - return false; } - return type.isStringLikeType() && isStringOffsetOnly && !accessAll; + return false; } - /** True when the column is accessed ONLY via IS NULL / IS NOT NULL, - * meaning the BE only needs to read the null flag, not the actual data. */ - public boolean hasNullCheckOnlyAccess() { - if (isRoot) { - DataTypeAccessTree child = children.values().iterator().next(); - return child.isNullCheckOnly && !child.accessAll - && !child.isStringOffsetOnly && !child.accessPartialChild; + /** + * recursively search in the tree, if any node hasNullPath + */ + public boolean hasNullPath() { + if (hasNullPath) { + return true; + } + for (DataTypeAccessTree child : children.values()) { + if (child.hasNullPath()) { + return true; + } } - return isNullCheckOnly && !accessAll && !isStringOffsetOnly && !accessPartialChild; + return false; } /** pruneCastType */ @@ -1168,10 +603,23 @@ public DataType pruneCastType(DataTypeAccessTree origin, DataTypeAccessTree cast } /** replacePathByAnotherTree */ - public boolean replacePathByAnotherTree(DataTypeAccessTree cast, List path, int index) { + public boolean replacePathByAnotherTree( + DataTypeAccessTree cast, List path, int index, TAccessPathType pathType) { if (index >= path.size()) { return true; } + if (pathType == TAccessPathType.META && index == path.size() - 1) { + String metaPath = path.get(index); + if (!metaPath.equals(AccessPathInfo.ACCESS_NULL) + && !metaPath.equals(AccessPathInfo.ACCESS_OFFSET)) { + throw new AnalysisException("unsupported metadata access path: " + metaPath); + } + // A type-changing cast can change both the physical OFFSET representation and + // nullness (for example, a failed string-to-number cast). Only translate the + // metadata suffix when the source and target node types are identical; otherwise + // visitCast falls back to reading the original value data before evaluating cast. + return type.equals(cast.type); + } if (cast.type instanceof StructType) { List fields = ((StructType) cast.type).getFields(); for (int i = 0; i < fields.size(); i++) { @@ -1180,17 +628,17 @@ public boolean replacePathByAnotherTree(DataTypeAccessTree cast, List pa String originFieldName = ((StructType) type).getFields().get(i).getName(); path.set(index, originFieldName); return children.get(originFieldName).replacePathByAnotherTree( - cast.children.get(castFieldName), path, index + 1 + cast.children.get(castFieldName), path, index + 1, pathType ); } } } else if (cast.type instanceof ArrayType) { return children.values().iterator().next().replacePathByAnotherTree( - cast.children.values().iterator().next(), path, index + 1); + cast.children.values().iterator().next(), path, index + 1, pathType); } else if (cast.type instanceof MapType) { String fieldName = path.get(index); return children.get(AccessPathInfo.ACCESS_MAP_VALUES).replacePathByAnotherTree( - cast.children.get(fieldName), path, index + 1 + cast.children.get(fieldName), path, index + 1, pathType ); } return false; @@ -1207,11 +655,12 @@ public void setAccessByPath(List path, int accessIndex, TAccessPathType this.pathType = TAccessPathType.DATA; } - // NULL path component: the column is accessed only via IS NULL / IS NOT NULL. + // Terminal NULL component: the column is accessed only via IS NULL / IS NOT NULL. // Mark null-check-only and return without setting accessAll or accessPartialChild, // so that parent nodes can distinguish "null-only leaf" from "has real sub-access". - if (path.get(accessIndex).equals(AccessPathInfo.ACCESS_NULL)) { - isNullCheckOnly = true; + if (pathType == TAccessPathType.META && accessIndex == path.size() - 1 + && path.get(accessIndex).equals(AccessPathInfo.ACCESS_NULL)) { + hasNullPath = true; return; } @@ -1228,9 +677,10 @@ public void setAccessByPath(List path, int accessIndex, TAccessPathType } return; } else if (this.type.isArrayType()) { - if (path.get(accessIndex).equals(AccessPathInfo.ACCESS_STRING_OFFSET)) { - // length(array_col) — only the offset array is needed, not element data. - isStringOffsetOnly = true; + if (pathType == TAccessPathType.META && accessIndex == path.size() - 1 + && path.get(accessIndex).equals(AccessPathInfo.ACCESS_OFFSET)) { + // cardinality(array_col) — only the offset array is needed, not element data. + hasOffsetPath = true; return; } DataTypeAccessTree child = children.get(AccessPathInfo.ACCESS_ALL); @@ -1241,9 +691,10 @@ public void setAccessByPath(List path, int accessIndex, TAccessPathType return; } else if (this.type.isMapType()) { String fieldName = path.get(accessIndex); - if (fieldName.equals(AccessPathInfo.ACCESS_STRING_OFFSET)) { - // length(map_col) — only the offset array is needed, not key/value data. - isStringOffsetOnly = true; + if (pathType == TAccessPathType.META && accessIndex == path.size() - 1 + && fieldName.equals(AccessPathInfo.ACCESS_OFFSET)) { + // cardinality(map_col) — only the offset array is needed, not key/value data. + hasOffsetPath = true; return; } if (fieldName.equals(AccessPathInfo.ACCESS_ALL)) { @@ -1279,8 +730,9 @@ public void setAccessByPath(List path, int accessIndex, TAccessPathType } else if (type.isStringLikeType()) { // String leaf accessed via the offset array (e.g. path ends in "offset"). // Mark offset-only so pruneDataType() can return BigIntType instead of full data. - if (path.get(accessIndex).equals(AccessPathInfo.ACCESS_STRING_OFFSET)) { - isStringOffsetOnly = true; + if (pathType == TAccessPathType.META && accessIndex == path.size() - 1 + && path.get(accessIndex).equals(AccessPathInfo.ACCESS_OFFSET)) { + hasOffsetPath = true; return; // do NOT set accessAll — offset-only is distinguishable from full access } // Any other sub-path on a string column means full data is needed. @@ -1323,12 +775,11 @@ public Optional pruneDataType() { return children.values().iterator().next().pruneDataType(); } else if (accessAll) { return Optional.of(type); - } else if (isStringOffsetOnly && !accessPartialChild) { + } else if (hasOffsetPath && !accessPartialChild) { // Only the offset array is accessed (e.g. length(str_col)). - // The slot type stays unchanged (varchar); the access path tells BE to skip char data. - return Optional.empty(); - } else if (isNullCheckOnly && !accessPartialChild) { - // Only the null flag is accessed (e.g. col IS NULL / struct_element(s,'f') IS NULL). + return Optional.of(type); + } else if (hasNullPath && !accessPartialChild) { + // Only the null flag is accessed (e.g. col IS NULL / element_at(s,'f') IS NULL). // Return the node's type so that parent nodes include this child in their pruned type, // while the access path (ending in NULL) tells BE to skip actual data reading. return Optional.of(type); @@ -1390,4 +841,105 @@ private DataType pruneDataType(DataType dataType, List> n } } } + + /** + * Expand map-level {@code *} wildcards into {@code KEYS} + {@code VALUES} + * variants. For n map-level stars in a single path, n+1 paths are + * emitted: one all-VALUES path plus one KEYS-terminating path per star + * position. Array-level stars are left unchanged. + * + *

Paths with no map-level star are kept as-is. + */ + private static void expandMapStarPaths( + Slot slot, + Multimap>> accessPaths) { + int slotId = slot.getExprId().asInt(); + Collection>> slotPaths = accessPaths.get(slotId); + if (slotPaths.isEmpty()) { + return; + } + DataType slotType = slot.getDataType(); + + List>> toAdd = new ArrayList<>(); + List>> toRemove = new ArrayList<>(); + + for (Pair> p : slotPaths) { + List path = p.second; + List positions = new ArrayList<>(); + findMapStarPositions(path, slotType, positions); + if (positions.isEmpty()) { + continue; + } + toRemove.add(p); + toAdd.addAll(expandOnePath(p.first, path, positions)); + } + + slotPaths.removeAll(toRemove); + slotPaths.addAll(toAdd); + } + + private static void findMapStarPositions( + List path, DataType slotType, List positions) { + DataType current = slotType; + for (int i = 1; i < path.size(); i++) { + String component = path.get(i); + if (current.isStructType()) { + StructField field = ((StructType) current).getField(component); + if (field == null) { + break; + } + current = field.getDataType(); + } else if (current.isArrayType()) { + if (!AccessPathInfo.ACCESS_ALL.equals(component)) { + break; + } + current = ((ArrayType) current).getItemType(); + } else if (current.isMapType()) { + MapType mapType = (MapType) current; + if (AccessPathInfo.ACCESS_ALL.equals(component)) { + positions.add(i); + current = mapType.getValueType(); + } else if (AccessPathInfo.ACCESS_MAP_KEYS.equals(component)) { + current = mapType.getKeyType(); + } else if (AccessPathInfo.ACCESS_MAP_VALUES.equals(component)) { + current = mapType.getValueType(); + } else { + current = mapType.getValueType(); + } + } else { + break; + } + } + } + + private static List>> expandOnePath( + TAccessPathType type, List path, List positions) { + int n = positions.size(); + List>> result = new ArrayList<>(n + 1); + + // All-VALUES path: replace every map * with VALUES + List allValues = new ArrayList<>(path); + for (int pos : positions) { + allValues.set(pos, AccessPathInfo.ACCESS_MAP_VALUES); + } + result.add(Pair.of(type, allValues)); + + // KEYS-terminating path for each position. Map lookup still reads key data even when + // the values path only reads metadata. + for (int i = 0; i < n; i++) { + int keysPos = positions.get(i); + List keysPath = new ArrayList<>(); + for (int j = 0; j < keysPos; j++) { + String component = path.get(j); + if (positions.contains(j)) { + component = AccessPathInfo.ACCESS_MAP_VALUES; + } + keysPath.add(component); + } + keysPath.add(AccessPathInfo.ACCESS_MAP_KEYS); + result.add(Pair.of(TAccessPathType.DATA, keysPath)); + } + + return result; + } } diff --git a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/SlotTypeReplacer.java b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/SlotTypeReplacer.java index c9cf0dfc259210..e4ca2a096007a2 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/SlotTypeReplacer.java +++ b/fe/fe-core/src/main/java/org/apache/doris/nereids/rules/rewrite/SlotTypeReplacer.java @@ -65,6 +65,7 @@ import org.apache.doris.nereids.types.StructType; import org.apache.doris.nereids.types.VariantType; import org.apache.doris.nereids.util.MoreFieldsThread; +import org.apache.doris.thrift.DescriptorsConstants; import org.apache.doris.thrift.TAccessPathType; import org.apache.doris.thrift.TColumnAccessPath; import org.apache.doris.thrift.TDataAccessPath; @@ -635,6 +636,7 @@ private List replaceIcebergAccessPathToId( ); TColumnAccessPath newAccessPath = new TColumnAccessPath(TAccessPathType.DATA); newAccessPath.data_access_path = new TDataAccessPath(icebergColumnAccessPath); + copyAccessPathVersion(accessPath, newAccessPath); replacedAccessPaths.add(newAccessPath); } else { icebergColumnAccessPath.addAll(accessPath.meta_access_path.path); @@ -643,6 +645,7 @@ private List replaceIcebergAccessPathToId( ); TColumnAccessPath newAccessPath = new TColumnAccessPath(TAccessPathType.META); newAccessPath.meta_access_path = new TMetaAccessPath(icebergColumnAccessPath); + copyAccessPathVersion(accessPath, newAccessPath); replacedAccessPaths.add(newAccessPath); } } @@ -688,6 +691,15 @@ private void replaceIcebergAccessPathToId(List originPath, int index, Da } } + /** Keep the FE/BE access-path wire contract when a path is rebuilt with Iceberg field ids. */ + private static void copyAccessPathVersion(TColumnAccessPath source, TColumnAccessPath target) { + if (source.isSetVersion()) { + target.setVersion(source.getVersion()); + } else { + target.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); + } + } + private void tryRecordReplaceSlots(Plan plan, Object checkObj, Set shouldReplaceSlots) { if (checkObj instanceof SupportPruneNestedColumn && ((SupportPruneNestedColumn) checkObj).supportPruneNestedColumn()) { diff --git a/fe/fe-core/src/main/java/org/apache/doris/qe/SessionVariable.java b/fe/fe-core/src/main/java/org/apache/doris/qe/SessionVariable.java index 1bf84014bbc984..5a0b1112785ca5 100644 --- a/fe/fe-core/src/main/java/org/apache/doris/qe/SessionVariable.java +++ b/fe/fe-core/src/main/java/org/apache/doris/qe/SessionVariable.java @@ -5893,6 +5893,8 @@ public TQueryOptions toThrift() { tResult.setReadCsvEmptyLineAsNull(readCsvEmptyLineAsNull); tResult.setSerdeDialect(getSerdeDialect()); + tResult.setEnablePruneNestedColumn(enablePruneNestedColumns); + tResult.setEnableMatchWithoutInvertedIndex(enableMatchWithoutInvertedIndex); tResult.setEnableFallbackOnMissingInvertedIndex(enableFallbackOnMissingInvertedIndex); tResult.setEnableInvertedIndexSearcherCache(enableInvertedIndexSearcherCache); diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/postprocess/TopNRuntimeFilterTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/postprocess/TopNRuntimeFilterTest.java index f66a284121ca58..84868c93dd7226 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/postprocess/TopNRuntimeFilterTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/postprocess/TopNRuntimeFilterTest.java @@ -42,6 +42,7 @@ import org.junit.jupiter.api.Test; import org.mockito.Mockito; +import java.util.List; import java.util.Map; public class TopNRuntimeFilterTest extends SSBTestBase implements MemoPatternMatchSupported { @@ -125,6 +126,25 @@ public void testNotUseTopNRfOnWindow() { Assertions.assertFalse(checker.getCascadesContext().getTopnFilterContext().isTopnFilterSource(localTopN)); } + @Test + public void testNotUseTopNRfForUnsupportedComplexOrderKey() { + String sql = "select c_custkey from customer order by array(c_custkey) limit 5"; + PlanChecker checker = PlanChecker.from(connectContext).analyze(sql) + .rewrite() + .implement(); + PhysicalPlan plan = checker.getPhysicalPlan(); + plan = new PlanPostProcessors(checker.getCascadesContext()).process(plan); + + List> localTopNs = plan.collectToList( + node -> node instanceof PhysicalTopN + && ((PhysicalTopN) node).getSortPhase().isLocal()); + Assertions.assertFalse(localTopNs.isEmpty(), plan.treeString()); + for (PhysicalTopN localTopN : localTopNs) { + Assertions.assertFalse( + checker.getCascadesContext().getTopnFilterContext().isTopnFilterSource(localTopN)); + } + } + @Test public void testProbeExprNullableThroughRightOuterJoin() { // topn node push down filter value to scan node. diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/PruneNestedColumnTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/PruneNestedColumnTest.java index cd0e08327b29cc..3922522b47f584 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/PruneNestedColumnTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/PruneNestedColumnTest.java @@ -28,6 +28,7 @@ import org.apache.doris.nereids.trees.expressions.Alias; import org.apache.doris.nereids.trees.expressions.ArrayItemReference; import org.apache.doris.nereids.trees.expressions.Expression; +import org.apache.doris.nereids.trees.expressions.IsNull; import org.apache.doris.nereids.trees.expressions.NamedExpression; import org.apache.doris.nereids.trees.expressions.Slot; import org.apache.doris.nereids.trees.expressions.SlotReference; @@ -41,8 +42,10 @@ import org.apache.doris.nereids.trees.plans.physical.PhysicalUnion; import org.apache.doris.nereids.types.BigIntType; import org.apache.doris.nereids.types.DataType; +import org.apache.doris.nereids.types.IntegerType; import org.apache.doris.nereids.types.NestedColumnPrunable; import org.apache.doris.nereids.types.NullType; +import org.apache.doris.nereids.types.StringType; import org.apache.doris.nereids.types.StructField; import org.apache.doris.nereids.types.StructType; import org.apache.doris.nereids.types.VariantType; @@ -50,19 +53,23 @@ import org.apache.doris.nereids.util.PlanChecker; import org.apache.doris.planner.OlapScanNode; import org.apache.doris.planner.PlanFragment; +import org.apache.doris.thrift.DescriptorsConstants; import org.apache.doris.thrift.TAccessPathType; import org.apache.doris.thrift.TColumnAccessPath; import org.apache.doris.thrift.TDataAccessPath; import org.apache.doris.thrift.TMetaAccessPath; import org.apache.doris.utframe.TestWithFeService; +import com.google.common.collect.ArrayListMultimap; import com.google.common.collect.ImmutableList; +import com.google.common.collect.Multimap; import org.junit.jupiter.api.Assertions; import org.junit.jupiter.api.BeforeAll; import org.junit.jupiter.api.Test; import java.util.ArrayList; import java.util.Collection; +import java.util.Comparator; import java.util.LinkedHashMap; import java.util.List; import java.util.Map; @@ -139,6 +146,39 @@ public void createTable() throws Exception { + " >\n" + ") properties ('replication_num'='1')"); + createTable("create table meta_name_tbl(\n" + + " id int,\n" + + " s struct<`NULL`: string, `OFFSET`: string>\n" + + ") properties ('replication_num'='1')"); + + // Nested struct for verifying multi-level META NULL path generation. + // All fields are nullable by default (no NOT NULL in DDL). + createTable("create table nested_struct_tbl(\n" + + " id int,\n" + + " s struct<`outer`: struct<`a`: string, `inner_f`: string>>\n" + + ") properties ('replication_num'='1')"); + + // Tables for outer-join nullability test: verifying that synthetic nullability + // from outer join does NOT cause META NULL paths on physically NOT NULL columns. + createTable("create table driving_tbl(\n" + + " id int\n" + + ") properties ('replication_num'='1')"); + + createTable("create table not_null_struct_tbl(\n" + + " id int,\n" + + " s struct not null\n" + + ") properties ('replication_num'='1')"); + + // Table for verifying that NOT NULL struct FIELD is preserved in pruned type + // when a sibling field is also accessed. + // Doris DDL does not support NOT NULL on individual struct fields, so the + // NOT NULL branch is tested via testNotNullFieldPreservedInAccessPaths + // which constructs the StructType programmatically. + createTable("create table nullable_struct_tbl_two_fields(\n" + + " id int,\n" + + " s struct\n" + + ") properties ('replication_num'='1')"); + connectContext.getSessionVariable().setDisableNereidsRules(RuleType.PRUNE_EMPTY_PARTITION.name()); connectContext.getSessionVariable().enableNereidsTimeout = false; } @@ -171,36 +211,19 @@ public void testMap() throws Exception { public void testMapElementLengthWithMapValuesKeepsKeysPath() throws Exception { assertColumn("select length(map_col['a']), map_values(map_col)[1] from str_tbl", "map", - ImmutableList.of(path("map_col", "KEYS"), path("map_col", "VALUES")), + ImmutableList.of( + path("map_col", "KEYS"), + path("map_col", "VALUES"), + metaPath("map_col", "VALUES", "OFFSET")), ImmutableList.of() ); } - @Test - public void testStructRootArrayMixedAccessSuppressesOffsetPath() throws Exception { - assertAllAccessPathsContain( - "select cardinality(struct_element(s, 'arr')), " - + "struct_element(element_at(struct_element(s, 'arr'), 1), 'int_field') " - + "from nested_container_tbl", - ImmutableList.of(path("s", "arr", "*", "int_field")), - ImmutableList.of(path("s", "arr", "OFFSET"))); - } - - @Test - public void testStructRootMapMixedAccessKeepsKeysPath() throws Exception { - assertAllAccessPathsContain( - "select length(element_at(struct_element(s, 'm'), 'a')), " - + "element_at(map_values(struct_element(s, 'm')), 1) " - + "from nested_container_tbl", - ImmutableList.of(path("s", "m", "KEYS"), path("s", "m", "VALUES")), - ImmutableList.of(path("s", "m", "*", "OFFSET"), path("s", "m", "VALUES", "OFFSET"))); - } - @Test public void testCardinalityArrayElementKeepsOffsetPath() throws Exception { assertAllAccessPathsContain( "select cardinality(element_at(a, 1)) from nested_array_tbl", - ImmutableList.of(path("a", "*", "OFFSET")), + ImmutableList.of(metaPath("a", "*", "OFFSET")), ImmutableList.of(path("a", "*"))); } @@ -208,21 +231,21 @@ public void testCardinalityArrayElementKeepsOffsetPath() throws Exception { public void testCardinalityMapElementKeepsValueOffsetPath() throws Exception { assertColumn("select cardinality(map_arr_col['a']) from map_array_tbl", "map>", - ImmutableList.of(path("map_arr_col", "KEYS"), path("map_arr_col", "VALUES", "OFFSET")), + ImmutableList.of(path("map_arr_col", "KEYS"), metaPath("map_arr_col", "VALUES", "OFFSET")), ImmutableList.of()); } @Test - public void testFullFieldAccessStripsExactDataSkippingPath() throws Exception { + public void testFullFieldAccessKeepsExactMetadataPath() throws Exception { assertColumn("select struct_element(s, 'city') from tbl " + "where struct_element(s, 'city') is null", "struct", - ImmutableList.of(path("s", "city")), - ImmutableList.of()); + ImmutableList.of(path("s", "city"), metaPath("s", "city", "NULL")), + ImmutableList.of(metaPath("s", "city", "NULL"))); assertColumn("select cardinality(struct_element(s, 'data')), struct_element(s, 'data') from tbl", "struct>>>", - ImmutableList.of(path("s", "data")), + ImmutableList.of(path("s", "data"), metaPath("s", "data", "OFFSET")), ImmutableList.of()); assertColumn("select cardinality(a), a from nested_array_tbl", @@ -232,12 +255,15 @@ public void testFullFieldAccessStripsExactDataSkippingPath() throws Exception { assertColumn("select cardinality(map_arr_col['a']), map_arr_col['a'] from map_array_tbl", "map>", - ImmutableList.of(path("map_arr_col", "*")), + ImmutableList.of( + path("map_arr_col", "KEYS"), + path("map_arr_col", "VALUES"), + metaPath("map_arr_col", "VALUES", "OFFSET")), ImmutableList.of()); } @Test - public void testCardinalityMapElementOffsetCoveredByValueFieldAccess() throws Exception { + public void testCardinalityMapElementOffsetPredicateStaysOutOfAllAccessPaths() throws Exception { Pair> result = collectComplexSlots( "select struct_element(element_at(element_at(struct_element(s, 'm'), 'null'), 1), 'verified') " + "from map_array_value_tbl " @@ -248,13 +274,23 @@ public void testCardinalityMapElementOffsetCoveredByValueFieldAccess() throws Ex allAccessPaths.addAll(slotDescriptor.getAllAccessPaths()); predicateAccessPaths.addAll(slotDescriptor.getPredicateAccessPaths()); } - Assertions.assertTrue(allAccessPaths.contains(path("s", "m", "*", "*", "verified"))); - Assertions.assertFalse(allAccessPaths.contains(path("s", "m", "*", "OFFSET"))); - Assertions.assertFalse(predicateAccessPaths.contains(path("s", "m", "*", "OFFSET"))); + Assertions.assertFalse(allAccessPaths.contains(metaPath("s", "m", "*", "OFFSET")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(allAccessPaths.contains(path("s", "m", "KEYS")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(allAccessPaths.contains(path("s", "m", "VALUES", "*", "verified")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(allAccessPaths.contains(metaPath("s", "m", "VALUES", "OFFSET")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(predicateAccessPaths.contains(path("s", "m", "KEYS")), + "predicateAccessPaths=" + predicateAccessPaths); + Assertions.assertTrue(predicateAccessPaths.contains(metaPath("s", "m", "VALUES", "OFFSET")), + "predicateAccessPaths=" + predicateAccessPaths); } @Test - public void testMapElementArrayNullPathCoveredByValueFieldAccess() throws Exception { + public void testMapElementArrayNullPredicateStaysOutOfAllAccessPaths() throws Exception { + // The map-star NULL path expands to precise KEYS/VALUES paths instead of broad s.m.*. Pair> result = collectComplexSlots( "select struct_element(element_at(element_at(struct_element(s, 'm'), 'null'), 1), 'verified') " + "from map_array_value_tbl " @@ -265,9 +301,18 @@ public void testMapElementArrayNullPathCoveredByValueFieldAccess() throws Except allAccessPaths.addAll(slotDescriptor.getAllAccessPaths()); predicateAccessPaths.addAll(slotDescriptor.getPredicateAccessPaths()); } - Assertions.assertTrue(allAccessPaths.contains(path("s", "m", "*", "*", "verified"))); - Assertions.assertFalse(allAccessPaths.contains(path("s", "m", "*", "NULL"))); - Assertions.assertFalse(predicateAccessPaths.contains(path("s", "m", "*", "NULL"))); + Assertions.assertFalse(allAccessPaths.contains(metaPath("s", "m", "*", "NULL")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(allAccessPaths.contains(path("s", "m", "KEYS")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(allAccessPaths.contains(path("s", "m", "VALUES", "*", "verified")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(allAccessPaths.contains(metaPath("s", "m", "VALUES", "NULL")), + "allAccessPaths=" + allAccessPaths); + Assertions.assertTrue(predicateAccessPaths.contains(path("s", "m", "KEYS")), + "predicateAccessPaths=" + predicateAccessPaths); + Assertions.assertTrue(predicateAccessPaths.contains(metaPath("s", "m", "VALUES", "NULL")), + "predicateAccessPaths=" + predicateAccessPaths); } @Test @@ -351,6 +396,18 @@ public void testVariantPredicateAccessPath() throws Exception { ); } + @Test + public void testVariantRootNullCheckFallsBackToData() { + SlotReference slot = rewriteAndFindScanSlot( + "select 1 from variant_tbl where v is null", "v", false); + Assertions.assertEquals( + new TreeSet<>(ImmutableList.of(path("v"))), + new TreeSet<>(slot.getAllAccessPaths().get())); + Assertions.assertEquals( + new TreeSet<>(ImmutableList.of(path("v"))), + new TreeSet<>(slot.getPredicateAccessPaths().get())); + } + @Test public void testVariantProjectAndPredicateAccessPaths() throws Exception { assertVariantSubColumnSlots("select v['a'] from variant_tbl where v['b']['c'] = 1", @@ -491,6 +548,25 @@ public void testPruneCast() throws Exception { ImmutableList.of() ); + assertColumn("select length(element_at(cast(c_struct as struct), 'v')) from str_tbl", + "struct", + ImmutableList.of(metaPath("c_struct", "f3", "OFFSET")), + ImmutableList.of() + ); + + assertColumn("select length(element_at(cast(c_struct as struct), 'k')) from str_tbl", + "struct", + ImmutableList.of(path("c_struct")), + ImmutableList.of() + ); + + assertColumn("select 1 from str_tbl " + + "where element_at(cast(c_struct as struct), 'v') is null", + "struct", + ImmutableList.of(path("c_struct")), + ImmutableList.of(path("c_struct")) + ); + assertColumns("select element_at(s, 'city') from (select * from tbl union all select * from tbl2)t", ImmutableList.of( Triple.of( @@ -646,63 +722,67 @@ public void testProject() throws Exception { public void testFilter() throws Throwable { assertColumn("select 100 from tbl where s is not null", "struct>>>", - ImmutableList.of(path("s", "NULL")), - ImmutableList.of(path("s", "NULL")) + ImmutableList.of(metaPath("s", "NULL")), + ImmutableList.of(metaPath("s", "NULL")) ); - // The IF expression itself is not collected as a null-only parent access here; the - // struct_element predicate still lets NCP prune the scan slot to the city field. - assertColumn("select 100 from tbl where if(id = 1, null, s) is not null or struct_element(s, 'city') = 'beijing'", + // The IF expression contributes a parent metadata path, while the struct_element + // predicate keeps its independent data path. + assertColumn("select 100 from tbl where if(id = 1, null, s) is not null or element_at(s, 'city') = 'beijing'", "struct", - ImmutableList.of(path("s", "city")), - ImmutableList.of(path("s", "city")) + ImmutableList.of(metaPath("s", "NULL"), path("s", "city")), + ImmutableList.of(metaPath("s", "NULL"), path("s", "city")) ); assertColumn("select 100 from tbl where struct_element(s, 'city') is not null", "struct", - ImmutableList.of(path("s", "city", "NULL")), - ImmutableList.of(path("s", "city", "NULL")) + ImmutableList.of(metaPath("s", "city", "NULL")), + ImmutableList.of(metaPath("s", "city", "NULL")) ); assertColumn("select 100 from tbl where struct_element(s, 'data') is not null", "struct>>>", - ImmutableList.of(path("s", "data", "NULL")), - ImmutableList.of(path("s", "data", "NULL")) + ImmutableList.of(metaPath("s", "data", "NULL")), + ImmutableList.of(metaPath("s", "data", "NULL")) ); assertColumn("select 100 from tbl where element_at(s, 'data')[1] is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "NULL")), - ImmutableList.of(path("s", "data", "*", "NULL")) + ImmutableList.of(metaPath("s", "data", "*", "NULL")), + ImmutableList.of(metaPath("s", "data", "*", "NULL")) ); assertColumn("select 100 from tbl where map_keys(element_at(s, 'data')[1]) is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "NULL")), - ImmutableList.of(path("s", "data", "*", "NULL")) + ImmutableList.of(metaPath("s", "data", "*", "NULL")), + ImmutableList.of(metaPath("s", "data", "*", "NULL")) ); assertColumn("select 100 from tbl where map_values(element_at(s, 'data')[1]) is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "NULL")), - ImmutableList.of(path("s", "data", "*", "NULL")) + ImmutableList.of(metaPath("s", "data", "*", "NULL")), + ImmutableList.of(metaPath("s", "data", "*", "NULL")) ); assertColumn("select 100 from tbl where element_at(map_values(element_at(s, 'data')[1])[1], 'a') is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "VALUES", "a", "NULL")), - ImmutableList.of(path("s", "data", "*", "VALUES", "a", "NULL")) + ImmutableList.of(metaPath("s", "data", "*", "VALUES", "a", "NULL")), + ImmutableList.of(metaPath("s", "data", "*", "VALUES", "a", "NULL")) ); assertColumn("select 100 from tbl where element_at(s, 'data')[1][1] is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "*", "NULL")), - ImmutableList.of(path("s", "data", "*", "*", "NULL")) + ImmutableList.of(path("s", "data", "*", "KEYS"), metaPath("s", "data", "*", "VALUES", "NULL")), + ImmutableList.of(path("s", "data", "*", "KEYS"), metaPath("s", "data", "*", "VALUES", "NULL")) ); assertColumn("select 100 from tbl where element_at(element_at(s, 'data')[1][1], 'a') is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "*", "a", "NULL")), - ImmutableList.of(path("s", "data", "*", "*", "a", "NULL")) + ImmutableList.of(path("s", "data", "*", "KEYS"), + metaPath("s", "data", "*", "VALUES", "a", "NULL")), + ImmutableList.of(path("s", "data", "*", "KEYS"), + metaPath("s", "data", "*", "VALUES", "a", "NULL")) ); assertColumn("select 100 from tbl where element_at(element_at(s, 'data')[1][1], 'b') is not null", "struct>>>", - ImmutableList.of(path("s", "data", "*", "*", "b", "NULL")), - ImmutableList.of(path("s", "data", "*", "*", "b", "NULL")) + ImmutableList.of(path("s", "data", "*", "KEYS"), + metaPath("s", "data", "*", "VALUES", "b", "NULL")), + ImmutableList.of(path("s", "data", "*", "KEYS"), + metaPath("s", "data", "*", "VALUES", "b", "NULL")) ); } @@ -713,50 +793,44 @@ public void testMapKeysAndValuesFunctionNullCheckUseParentMapNullPath() throws E // read the parent map null map, not the KEYS/VALUES child null maps. assertColumn("select 100 from str_tbl where map_keys(map_col) is null", "map", - ImmutableList.of(path("map_col", "NULL")), - ImmutableList.of(path("map_col", "NULL")) + ImmutableList.of(metaPath("map_col", "NULL")), + ImmutableList.of(metaPath("map_col", "NULL")) ); assertColumn("select 100 from str_tbl where map_values(map_col) is null", "map", - ImmutableList.of(path("map_col", "NULL")), - ImmutableList.of(path("map_col", "NULL")) + ImmutableList.of(metaPath("map_col", "NULL")), + ImmutableList.of(metaPath("map_col", "NULL")) ); assertColumn("select map_keys(map_col) from str_tbl where map_keys(map_col) is null", "map", - ImmutableList.of(path("map_col", "KEYS")), - ImmutableList.of() + ImmutableList.of(path("map_col", "KEYS"), metaPath("map_col", "NULL")), + ImmutableList.of(metaPath("map_col", "NULL")) ); assertColumn("select map_values(map_col) from str_tbl where map_values(map_col) is null", "map", - ImmutableList.of(path("map_col", "VALUES")), - ImmutableList.of() + ImmutableList.of(path("map_col", "VALUES"), metaPath("map_col", "NULL")), + ImmutableList.of(metaPath("map_col", "NULL")) ); } @Test public void testProjectFilter() throws Throwable { - assertColumn("select s from tbl where element_at(s, 'city') is not null", - "struct>>>", - ImmutableList.of(path("s")), - ImmutableList.of() - ); - assertColumn("select s from tbl where struct_element(s, 'city') is null", - "struct>>>", - ImmutableList.of(path("s")), - ImmutableList.of() - ); - assertColumn("select element_at(s, 'data') from tbl where element_at(s, 'city') is not null", "struct>>>", - ImmutableList.of(path("s", "city", "NULL"), path("s", "data")), - ImmutableList.of(path("s", "city", "NULL")) + ImmutableList.of( + path("s", "data"), + metaPath("s", "city", "NULL")), + ImmutableList.of(metaPath("s", "city", "NULL")) ); assertColumn("select element_at(s, 'data') from tbl where element_at(s, 'city') is not null and element_at(s, 'data') is not null", "struct>>>", - ImmutableList.of(path("s", "city", "NULL"), path("s", "data")), - ImmutableList.of(path("s", "city", "NULL")) + ImmutableList.of( + path("s", "data"), + metaPath("s", "city", "NULL"), + metaPath("s", "data", "NULL")), + ImmutableList.of(metaPath("s", "city", "NULL"), metaPath("s", "data", "NULL")) ); } @@ -1162,6 +1236,51 @@ public void testDataTypeAccessTreeKeepsVariantTerminalPath() { ((StructType) prunedType).getFields().get(0).getDataType()); } + @Test + public void testDataPathNamedLikeMetadataComponent() { + StructType structType = new StructType(ImmutableList.of( + new StructField("NULL", StringType.INSTANCE, true, ""), + new StructField("OFFSET", StringType.INSTANCE, true, ""))); + SlotReference slot = new SlotReference("s", structType); + + DataTypeAccessTree nullFieldTree = DataTypeAccessTree.ofRoot(slot, TAccessPathType.DATA); + nullFieldTree.setAccessByPath(ImmutableList.of("s", "NULL"), 0, TAccessPathType.DATA); + Assertions.assertEquals("STRUCT<`null`:TEXT>", nullFieldTree.pruneDataType().get().toSql()); + + DataTypeAccessTree offsetFieldTree = DataTypeAccessTree.ofRoot(slot, TAccessPathType.DATA); + offsetFieldTree.setAccessByPath(ImmutableList.of("s", "OFFSET"), 0, TAccessPathType.DATA); + Assertions.assertEquals("STRUCT", offsetFieldTree.pruneDataType().get().toSql()); + + DataTypeAccessTree nullMetadataTree = DataTypeAccessTree.ofRoot(slot, TAccessPathType.META); + nullMetadataTree.setAccessByPath( + ImmutableList.of("s", "NULL", "NULL"), 0, TAccessPathType.META); + Assertions.assertEquals("STRUCT<`null`:TEXT>", nullMetadataTree.pruneDataType().get().toSql()); + + DataTypeAccessTree offsetMetadataTree = DataTypeAccessTree.ofRoot(slot, TAccessPathType.META); + offsetMetadataTree.setAccessByPath( + ImmutableList.of("s", "OFFSET", "OFFSET"), 0, TAccessPathType.META); + Assertions.assertEquals("STRUCT", offsetMetadataTree.pruneDataType().get().toSql()); + + CollectAccessPathResult dataPath = new CollectAccessPathResult( + ImmutableList.of("s", "NULL"), false, TAccessPathType.DATA); + CollectAccessPathResult metadataPath = new CollectAccessPathResult( + ImmutableList.of("s", "NULL"), false, TAccessPathType.META); + Assertions.assertNotEquals(dataPath, metadataPath); + } + + @Test + public void testMetadataPathBelowSameNamedStructField() throws Exception { + assertColumn("select 1 from meta_name_tbl where element_at(s, 'NULL') is null", + "struct", + ImmutableList.of(metaPath("s", "null", "NULL")), + ImmutableList.of(metaPath("s", "null", "NULL"))); + + assertColumn("select length(element_at(s, 'OFFSET')) from meta_name_tbl", + "struct", + ImmutableList.of(metaPath("s", "offset", "OFFSET")), + ImmutableList.of()); + } + @Test public void testWithVariant() throws Exception { connectContext.getSessionVariable().enableDecimal256 = true; @@ -1373,10 +1492,12 @@ private void assertAllAccessPathsContain(String sql, List exp allAccessPaths.addAll(slotDescriptor.getAllAccessPaths()); } for (TColumnAccessPath accessPath : expectContainAllAccessPaths) { - Assertions.assertTrue(allAccessPaths.contains(accessPath)); + Assertions.assertTrue(allAccessPaths.contains(accessPath), + "expected " + accessPath + " but allAccessPaths=" + allAccessPaths); } for (TColumnAccessPath accessPath : expectNotContainAllAccessPaths) { - Assertions.assertFalse(allAccessPaths.contains(accessPath)); + Assertions.assertFalse(allAccessPaths.contains(accessPath), + "expected NOT " + accessPath + " but allAccessPaths=" + allAccessPaths); } } @@ -1408,7 +1529,6 @@ private void assertColumns(String sql, TreeSet actualPredicateAccessPaths = new TreeSet<>(slotDescriptor.getPredicateAccessPaths()); Assertions.assertEquals(expectPredicateAccessPathSet, actualPredicateAccessPaths); - Assertions.assertTrue(actualAllAccessPaths.containsAll(actualPredicateAccessPaths)); Map slotIdToDataTypes = new LinkedHashMap<>(); Consumer assertHasSameType = e -> { @@ -1462,8 +1582,8 @@ public void testStructIsNullPruning() throws Exception { // struct column IS NULL → null-only access, emit [s, NULL] path, type stays struct assertColumn("select 1 from tbl where s is null", "struct>>>", - ImmutableList.of(path("s", "NULL")), - ImmutableList.of(path("s", "NULL"))); + ImmutableList.of(metaPath("s", "NULL")), + ImmutableList.of(metaPath("s", "NULL"))); } @Test @@ -1471,31 +1591,80 @@ public void testStructIsNotNullPruning() throws Exception { // struct column IS NOT NULL → same null-only access pattern assertColumn("select 1 from tbl where s is not null", "struct>>>", - ImmutableList.of(path("s", "NULL")), - ImmutableList.of(path("s", "NULL"))); + ImmutableList.of(metaPath("s", "NULL")), + ImmutableList.of(metaPath("s", "NULL"))); + } + + @Test + public void testResolveStructFieldNullable() { + StructType type = new StructType(ImmutableList.of( + new StructField("not_null_f", StringType.INSTANCE, false, ""), + new StructField("nullable_f", StringType.INSTANCE, true, "") + )); + // String-like literal: select by name + StructField notNullField = AccessPathExpressionCollector.resolveStructField( + type, new org.apache.doris.nereids.trees.expressions.literal.VarcharLiteral("not_null_f")); + Assertions.assertNotNull(notNullField); + Assertions.assertFalse(notNullField.isNullable()); + + StructField nullableField = AccessPathExpressionCollector.resolveStructField( + type, new org.apache.doris.nereids.trees.expressions.literal.VarcharLiteral("nullable_f")); + Assertions.assertNotNull(nullableField); + Assertions.assertTrue(nullableField.isNullable()); + + // Integer-like literal: select by 1-based index + StructField fieldByIndex = AccessPathExpressionCollector.resolveStructField( + type, new org.apache.doris.nereids.trees.expressions.literal.IntegerLiteral(1)); + Assertions.assertNotNull(fieldByIndex); + Assertions.assertEquals("not_null_f", fieldByIndex.getName()); + Assertions.assertFalse(fieldByIndex.isNullable()); + + // Out-of-bounds index returns null + StructField outOfBounds = AccessPathExpressionCollector.resolveStructField( + type, new org.apache.doris.nereids.trees.expressions.literal.IntegerLiteral(99)); + Assertions.assertNull(outOfBounds); + + // Non-literal returns null + StructField nonLiteral = AccessPathExpressionCollector.resolveStructField( + type, NullLiteral.INSTANCE); + Assertions.assertNull(nonLiteral); + } + + @Test + public void testScalarIsNullProducesMetaPath() { + SlotReference slot = rewriteAndFindScanSlot("select 1 from tbl where id is null", "id", false); + Assertions.assertEquals( + new TreeSet<>(ImmutableList.of(metaPath("id", "NULL"))), + new TreeSet<>(slot.getAllAccessPaths().get())); + Assertions.assertEquals( + new TreeSet<>(ImmutableList.of(metaPath("id", "NULL"))), + new TreeSet<>(slot.getPredicateAccessPaths().get())); } @Test public void testStructIsNullMixedAccess() throws Exception { - // Parent NULL path must be stripped from allPaths when a child path is also required. - // Otherwise BE StructFileColumnIterator sees the parent NULL sub-path first, switches - // the whole struct iterator to NULL_MAP_ONLY, and skips the child iterator. - // predicateAccessPaths drops [s, NULL] too, keeping it a subset of allAccessPaths. - assertColumn("select struct_element(s, 'city') from tbl where s is null", + // Predicate metadata paths stay typed in allPaths even when another child needs data. + assertColumn("select element_at(s, 'city') from tbl where s is null", "struct", - ImmutableList.of(path("s", "city")), - ImmutableList.of()); + ImmutableList.of(path("s", "city"), metaPath("s", "NULL")), + ImmutableList.of(metaPath("s", "NULL"))); + + assertColumn("select s from tbl where element_at(s, 'city') is null", + "struct>>>", + ImmutableList.of(path("s")), + ImmutableList.of(metaPath("s", "city", "NULL"))); // This shape is closer to the production bug: one predicate needs the parent // null map, another predicate needs a child null map, and the projection needs - // a different child data path. The parent [s.NULL] cannot remain in allPaths - // with [s.data], so it is also removed from predicate paths; [s.city.NULL] stays - // because it is still present in allPaths. - assertColumn("select struct_element(s, 'data') from tbl " - + "where s is null or struct_element(s, 'city') is null", + // a different child data path. Keep all three requirements independent. + assertColumn("select element_at(s, 'data') from tbl " + + "where s is null or element_at(s, 'city') is null", "struct>>>", - ImmutableList.of(path("s", "city", "NULL"), path("s", "data")), - ImmutableList.of(path("s", "city", "NULL"))); + ImmutableList.of( + path("s", "data"), + metaPath("s", "NULL"), + metaPath("s", "city", "NULL")), + ImmutableList.of(metaPath("s", "NULL"), metaPath("s", "city", "NULL"))); } @Test @@ -1505,7 +1674,7 @@ public void testStringLengthPruning() throws Exception { "select length(str_col) from str_tbl", "str_col", true, - ImmutableList.of(path("str_col", "OFFSET"))); + ImmutableList.of(metaPath("str_col", "OFFSET"))); // ── Case 2: length(str_col) + direct projection of str_col ─ suppressed ───── assertStringColumn( @@ -1521,13 +1690,13 @@ public void testStringLengthPruning() throws Exception { false, ImmutableList.of()); - // ── Case 4: length applied to a struct field ─ struct pruned to bigint field ─ + // ── Case 4: length applied to a struct field ─ metadata-only field read ──── // c_struct has {f1:int, f3:string}; only f3 accessed offset-only → - // pruned type is struct, access path is DATA(["c_struct","f3","offset"]) + // pruned type is struct, access path is META(["c_struct","f3","OFFSET"]) assertColumn( "select length(struct_element(c_struct, 'f3')) from str_tbl", "struct", - ImmutableList.of(path("c_struct", "f3", "OFFSET")), + ImmutableList.of(metaPath("c_struct", "f3", "OFFSET")), ImmutableList.of()); // ── Case 5: length(struct field) + direct read of same field ─ suppressed ─── @@ -1536,29 +1705,47 @@ public void testStringLengthPruning() throws Exception { assertColumn( "select length(struct_element(c_struct, 'f3')), struct_element(c_struct, 'f3') from str_tbl", "struct", - ImmutableList.of(path("c_struct", "f3")), + ImmutableList.of(path("c_struct", "f3"), metaPath("c_struct", "f3", "OFFSET")), + ImmutableList.of()); + + assertColumn( + "select length(map_keys(map_col)[1]) from str_tbl", + "map", + ImmutableList.of(metaPath("map_col", "KEYS", "OFFSET")), + ImmutableList.of()); + + assertColumn( + "select length(map_values(map_col)[1]) from str_tbl", + "map", + ImmutableList.of(metaPath("map_col", "VALUES", "OFFSET")), ImmutableList.of()); } @Test - public void testNonOlapDataSkippingOnlyAccessPathFallback() { + public void testNonOlapMetadataAccessPathFallback() { List normalizedAccessPaths = AccessPathPlanCollector.normalizeDataSkippingOnlyAccessPaths(ImmutableList.of( + new CollectAccessPathResult( + ImmutableList.of("s", "city", "NULL"), true, TAccessPathType.META), new CollectAccessPathResult( ImmutableList.of("s", "city", "NULL"), true, TAccessPathType.DATA), new CollectAccessPathResult( - ImmutableList.of("array_column", "OFFSET"), false, TAccessPathType.DATA), + ImmutableList.of("s", "NULL"), false, TAccessPathType.DATA), + new CollectAccessPathResult( + ImmutableList.of("s", "OFFSET"), false, TAccessPathType.DATA), new CollectAccessPathResult( ImmutableList.of("s", "city"), false, TAccessPathType.DATA))); - Assertions.assertEquals(3, normalizedAccessPaths.size()); + Assertions.assertEquals(5, normalizedAccessPaths.size()); Assertions.assertEquals(ImmutableList.of("s", "city"), normalizedAccessPaths.get(0).getPath()); Assertions.assertTrue(normalizedAccessPaths.get(0).isPredicate()); Assertions.assertEquals(TAccessPathType.DATA, normalizedAccessPaths.get(0).getType()); - Assertions.assertEquals(ImmutableList.of("array_column"), normalizedAccessPaths.get(1).getPath()); - Assertions.assertFalse(normalizedAccessPaths.get(1).isPredicate()); + Assertions.assertEquals(ImmutableList.of("s", "city", "NULL"), normalizedAccessPaths.get(1).getPath()); + Assertions.assertTrue(normalizedAccessPaths.get(1).isPredicate()); Assertions.assertEquals(TAccessPathType.DATA, normalizedAccessPaths.get(1).getType()); - Assertions.assertEquals(ImmutableList.of("s", "city"), normalizedAccessPaths.get(2).getPath()); + Assertions.assertEquals(ImmutableList.of("s", "NULL"), normalizedAccessPaths.get(2).getPath()); + Assertions.assertEquals(ImmutableList.of("s", "OFFSET"), normalizedAccessPaths.get(3).getPath()); + Assertions.assertEquals(ImmutableList.of("s", "city"), normalizedAccessPaths.get(4).getPath()); } @Test @@ -1566,12 +1753,14 @@ public void testMvRewritePlanFragmentSkipsNullOnlyAccessPath() { SlotReference normalSlot = rewriteAndFindScanSlot( "select 1 from str_tbl where str_col is not null", "str_col", false); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("str_col", "NULL"))), + new TreeSet<>(ImmutableList.of(metaPath("str_col", "NULL"))), new TreeSet<>(normalSlot.getAllAccessPaths().get())); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("str_col", "NULL"))), + new TreeSet<>(ImmutableList.of(metaPath("str_col", "NULL"))), new TreeSet<>(normalSlot.getPredicateAccessPaths().get())); + // MV fragment: IS NULL degrades to full column read via default visitor. + // [str_col] full-access path passes shouldSkipAccessInfo → no pruning. SlotReference fragmentSlot = rewriteAndFindScanSlot( "select 1 from str_tbl where str_col is not null", "str_col", true); assertNoAccessPaths(fragmentSlot); @@ -1579,20 +1768,22 @@ public void testMvRewritePlanFragmentSkipsNullOnlyAccessPath() { SlotReference nestedNormalSlot = rewriteAndFindScanSlot( "select 1 from tbl where struct_element(s, 'city') is not null", "s", false); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("s", "city", "NULL"))), + new TreeSet<>(ImmutableList.of(metaPath("s", "city", "NULL"))), new TreeSet<>(nestedNormalSlot.getAllAccessPaths().get())); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("s", "city", "NULL"))), + new TreeSet<>(ImmutableList.of(metaPath("s", "city", "NULL"))), new TreeSet<>(nestedNormalSlot.getPredicateAccessPaths().get())); - // MV rewrite optimizes temporary fragments whose later consumers are not visible. - // If the fragment only needs nested null metadata, e.g. [s.city.NULL], pruning the - // scan slot to struct can break the final rewritten MV plan when it still - // needs the full struct or another child. The fragment marker therefore suppresses - // nested null-only access info too, not just top-level [col.NULL]. + // MV fragment: IS NULL degrades to element_at via default visitor, + // producing [s, city] data path. struct is NestedColumnPrunable so + // pruning to struct is safe — no meta suffix remains. SlotReference nestedFragmentSlot = rewriteAndFindScanSlot( "select 1 from tbl where struct_element(s, 'city') is not null", "s", true); - assertNoAccessPaths(nestedFragmentSlot); + Assertions.assertEquals( + new TreeSet<>(ImmutableList.of(path("s", "city"))), + new TreeSet<>(nestedFragmentSlot.getAllAccessPaths().get())); + Assertions.assertTrue(!nestedFragmentSlot.getPredicateAccessPaths().isPresent() + || nestedFragmentSlot.getPredicateAccessPaths().get().isEmpty()); } @Test @@ -1600,10 +1791,10 @@ public void testMvRewritePlanFragmentSkipsOffsetOnlyAccessPath() { SlotReference normalSlot = rewriteAndFindScanSlot( "select 1 from str_tbl where length(str_col) > 0", "str_col", false); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("str_col", "OFFSET"))), + new TreeSet<>(ImmutableList.of(metaPath("str_col", "OFFSET"))), new TreeSet<>(normalSlot.getAllAccessPaths().get())); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("str_col", "OFFSET"))), + new TreeSet<>(ImmutableList.of(metaPath("str_col", "OFFSET"))), new TreeSet<>(normalSlot.getPredicateAccessPaths().get())); SlotReference fragmentSlot = rewriteAndFindScanSlot( @@ -1614,16 +1805,22 @@ public void testMvRewritePlanFragmentSkipsOffsetOnlyAccessPath() { "select 1 from str_tbl where length(struct_element(c_struct, 'f3')) > 0", "c_struct", false); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("c_struct", "f3", "OFFSET"))), + new TreeSet<>(ImmutableList.of(metaPath("c_struct", "f3", "OFFSET"))), new TreeSet<>(nestedNormalSlot.getAllAccessPaths().get())); Assertions.assertEquals( - new TreeSet<>(ImmutableList.of(path("c_struct", "f3", "OFFSET"))), + new TreeSet<>(ImmutableList.of(metaPath("c_struct", "f3", "OFFSET"))), new TreeSet<>(nestedNormalSlot.getPredicateAccessPaths().get())); + // MV fragment: length() degrades to element_at via default visitor, + // producing [c_struct, f3] data path without OFFSET suffix. SlotReference nestedFragmentSlot = rewriteAndFindScanSlot( "select 1 from str_tbl where length(struct_element(c_struct, 'f3')) > 0", "c_struct", true); - assertNoAccessPaths(nestedFragmentSlot); + Assertions.assertEquals( + new TreeSet<>(ImmutableList.of(path("c_struct", "f3"))), + new TreeSet<>(nestedFragmentSlot.getAllAccessPaths().get())); + Assertions.assertTrue(!nestedFragmentSlot.getPredicateAccessPaths().isPresent() + || nestedFragmentSlot.getPredicateAccessPaths().get().isEmpty()); } /** @@ -1690,9 +1887,19 @@ private SlotReference rewriteAndFindScanSlot(String sql, String columnName, } private void assertNoAccessPaths(SlotReference slot) { - Assertions.assertTrue(!slot.getAllAccessPaths().isPresent() || slot.getAllAccessPaths().get().isEmpty()); + String slotDebugInfo = String.format( + "slot=%s, name=%s, exprId=%s, qualifier=%s, dataType=%s, nullable=%s, " + + "subPath=%s, originalColumn=%s, allAccessPaths=%s, " + + "predicateAccessPaths=%s, displayAllAccessPaths=%s, " + + "displayPredicateAccessPaths=%s", + slot, slot.getName(), slot.getExprId(), slot.getQualifier(), slot.getDataType(), + slot.nullable(), slot.getSubPath(), slot.getOriginalColumn().map(Object::toString), + slot.getAllAccessPaths(), slot.getPredicateAccessPaths(), + slot.getDisplayAllAccessPaths(), slot.getDisplayPredicateAccessPaths()); + Assertions.assertTrue(!slot.getAllAccessPaths().isPresent() || slot.getAllAccessPaths().get().isEmpty(), + slotDebugInfo); Assertions.assertTrue(!slot.getPredicateAccessPaths().isPresent() - || slot.getPredicateAccessPaths().get().isEmpty()); + || slot.getPredicateAccessPaths().get().isEmpty(), slotDebugInfo); } private Pair> collectComplexSlots(String sql) throws Exception { @@ -1717,6 +1924,7 @@ private Pair> collectComplexSlots(String sql) private TColumnAccessPath path(String... path) { TColumnAccessPath accessPath = new TColumnAccessPath(TAccessPathType.DATA); accessPath.data_access_path = new TDataAccessPath(ImmutableList.copyOf(path)); + accessPath.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); return accessPath; } @@ -1742,6 +1950,7 @@ private interface CheckedRunnable { private TColumnAccessPath metaPath(String... path) { TColumnAccessPath accessPath = new TColumnAccessPath(TAccessPathType.META); accessPath.meta_access_path = new TMetaAccessPath(ImmutableList.copyOf(path)); + accessPath.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); return accessPath; } @@ -1791,4 +2000,152 @@ private void assertVariantSubColumnSlotCount(String sql, List expectedSu Assertions.assertEquals(expectedCount, actualCount); } + + /** + * Verify that synthetic nullability from outer join does NOT cause META NULL paths + * on physically NOT NULL columns. When a NOT NULL struct sits on the nullable side + * of a LEFT JOIN, the slot's {@code nullable()} returns true (from outer join + * semantics), but {@code getOriginalColumn().isAllowNull()} returns false + * (physical column has no null map). The fix in AccessPathExpressionCollector + * should suppress the {@code [s, NULL]} META path in this case. + */ + @Test + public void testNotNullStructOnOuterJoinNullableSide() throws Exception { + // driving_tbl LEFT JOIN not_null_struct_tbl: + // not_null_struct_tbl.s is NOT NULL in the schema, but after LEFT JOIN the + // slot becomes nullable (right side of LEFT JOIN → withNullable(true)). + // element_at(s, 'f') IS NULL in WHERE: + // - s.nullable() = true (synthetic, from outer join) + // - s.getOriginalColumn().isAllowNull() = false (physical, no null map) + // Expected: [s, f] DATA is present (field is read for IS NULL evaluation), + // [s, NULL] META must NOT be present (no physical null map). + assertAllAccessPathsContain( + "select driving_tbl.id from driving_tbl" + + " left join not_null_struct_tbl" + + " on driving_tbl.id = not_null_struct_tbl.id" + + " where element_at(not_null_struct_tbl.s, 'f') is null", + // expect-contain: field is read (DATA path) + ImmutableList.of(path("s", "f")), + // expect-NOT-contain: struct-level NULL path must be suppressed + ImmutableList.of(metaPath("s", "NULL"))); + } + + /** + * Verifies that a NOT NULL struct field is preserved in the pruned type when a + * sibling field is accessed via SELECT and the IS NULL check emits struct-level + * META NULL. Before the fix, the early return at visitElementAt dropped the NOT NULL + * field's path, causing pruneDataType to remove it from the struct type. The filter + * expression still referenced element_at(s, 'f') IS NULL and could not be rebuilt. + */ + @Test + public void testNullableFieldPreservedWithSiblingProjection() throws Exception { + // s STRUCT NULL (both fields nullable by default) + // SELECT element_at(s, 'g') → [s, g] DATA + // WHERE element_at(s, 'f') IS NULL → [s, NULL] META + [s, f, NULL] META + // Both fields preserved in pruned type. + assertColumn( + "select element_at(s, 'g') from nullable_struct_tbl_two_fields" + + " where element_at(s, 'f') is null", + "struct", + ImmutableList.of( + path("s", "g"), + metaPath("s", "f", "NULL")), + ImmutableList.of( + metaPath("s", "f", "NULL"))); + } + + /** + * Tests the NOT NULL struct field branch in visitElementAt line 375-384, + * complementing {@link #testNotNullStructOnOuterJoinNullableSide()} and + * {@link #testNullableFieldPreservedWithSiblingProjection()}. + * + *

Relationship with other tests

+ *
    + *
  • {@code testNotNullStructOnOuterJoinNullableSide}: the struct itself is + * NOT NULL → no struct null map → [s, NULL] META must be suppressed.
  • + *
  • {@code testNullableFieldPreservedWithSiblingProjection}: the struct IS + * nullable AND the field IS nullable → [s, NULL] META + [s, f, NULL] META + * both emitted, field preserved via the META path.
  • + *
  • This test: the struct IS nullable BUT the field is NOT NULL → + * [s, NULL] META is emitted for the struct null map, but [s, f, NULL] + * META is NOT emitted (no field-level null map). The fix must still emit + * [s, f] DATA so pruneDataType preserves the field in the struct type — + * the filter expression {@code element_at(s, 'f') IS NULL} still + * references 'f' and won't be rewritten to {@code s IS NULL}.
  • + *
+ * + *

Why manual construction instead of SQL

+ * Doris DDL does not support {@code NOT NULL} on individual struct fields + * ({@code struct} triggers a syntax error). This test + * therefore constructs the {@link StructType} with a nullable=false field + * programmatically and calls {@link AccessPathExpressionCollector} directly. + * Because the {@link SlotReference} lacks an {@code originalColumn}, + * {@code hasPhysicalNullMap} returns false, so [s, NULL] META is suppressed + * by the slot-level guard — that path is covered by the SQL-based + * {@code testNullableFieldPreservedWithSiblingProjection}. + */ + @Test + public void testNotNullFieldPreservedInAccessPaths() { + // Scenario from Review 2: + // Project(element_at(s, 'g')) + // Filter(element_at(s, 'f') IS NULL) + // Scan(s STRUCT NULL) + // + // Before fix: visitElementAt emitted [s, NULL] META then returned null, + // dropping field 'f'. pruneDataType removed 'f' from the struct type + // because no path referenced it. The filter still referenced + // element_at(s, 'f') IS NULL → rebuild failed. + // + // After fix: visitElementAt emits [s, NULL] META, then falls through + // with a fresh DATA context to emit [s, f] DATA, preserving 'f'. + + // s STRUCT + StructType structType = new StructType(ImmutableList.of( + new StructField("f", IntegerType.INSTANCE, false, ""), // NOT NULL + new StructField("g", IntegerType.INSTANCE, true, ""))); // nullable + SlotReference slot = new SlotReference("s", structType, true); // struct is nullable + + // SELECT element_at(s, 'g') → collector emits [s, g] DATA + ElementAt selectG = new ElementAt(slot, new org.apache.doris.nereids.trees.expressions.literal.VarcharLiteral("g")); + Multimap paths1 = ArrayListMultimap.create(); + new AccessPathExpressionCollector( + connectContext.getStatementContext(), paths1, false, false) + .collect(selectG); + + // WHERE element_at(s, 'f') IS NULL → should emit [s, NULL] META + [s, f] DATA + ElementAt whereF = new ElementAt(slot, new org.apache.doris.nereids.trees.expressions.literal.VarcharLiteral("f")); + IsNull isNull = new IsNull(whereF); + Multimap paths2 = ArrayListMultimap.create(); + new AccessPathExpressionCollector( + connectContext.getStatementContext(), paths2, true, false) + .collect(isNull); + + // Merge results + TreeSet allPaths = new TreeSet<>( + Comparator.comparing(CollectAccessPathResult::toString)); + allPaths.addAll(paths1.get(slot.getExprId().asInt())); + allPaths.addAll(paths2.get(slot.getExprId().asInt())); + + // [s, g] DATA must exist (from SELECT) + Assertions.assertTrue(allPaths.contains( + new CollectAccessPathResult( + ImmutableList.of("s", "g"), false, TAccessPathType.DATA)), + "expected [s,g] DATA in: " + allPaths); + // [s, f] DATA must exist (NOT NULL field preserved — the fix) + // isPredicate=true because it comes from the WHERE clause. + Assertions.assertTrue(allPaths.contains( + new CollectAccessPathResult( + ImmutableList.of("s", "f"), true, TAccessPathType.DATA)), + "expected [s,f] DATA (NOT NULL field preserved) in: " + allPaths); + // NOTE: [s, NULL] META is not asserted here because the manually constructed + // SlotReference has no originalColumn, so hasPhysicalNullMap returns false and + // the META NULL path is suppressed at visitSlotReference. The [s, NULL] META + // behavior is covered by testNullableFieldPreservedWithSiblingProjection. + // + // [s, f, NULL] META must NOT exist (f is NOT NULL, no field-level null map) + Assertions.assertFalse(allPaths.contains( + new CollectAccessPathResult( + ImmutableList.of("s", "f", "NULL"), true, TAccessPathType.META)), + "f is NOT NULL, expected NO [s,f,NULL] META in: " + allPaths); + } } diff --git a/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/VariantPruningLogicTest.java b/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/VariantPruningLogicTest.java index 2b17d366bbda96..62cac47d7fb5d6 100644 --- a/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/VariantPruningLogicTest.java +++ b/fe/fe-core/src/test/java/org/apache/doris/nereids/rules/rewrite/VariantPruningLogicTest.java @@ -28,6 +28,7 @@ import org.apache.doris.nereids.trees.plans.physical.PhysicalPlan; import org.apache.doris.planner.OlapScanNode; import org.apache.doris.planner.PlanFragment; +import org.apache.doris.thrift.DescriptorsConstants; import org.apache.doris.thrift.TAccessPathType; import org.apache.doris.thrift.TColumnAccessPath; import org.apache.doris.thrift.TDataAccessPath; @@ -302,6 +303,7 @@ private void assertAllAccessPathsContain( private TColumnAccessPath path(String... path) { TColumnAccessPath accessPath = new TColumnAccessPath(TAccessPathType.DATA); accessPath.data_access_path = new TDataAccessPath(ImmutableList.copyOf(path)); + accessPath.setVersion(DescriptorsConstants.TCOLUMN_ACCESS_PATH_VERSION_TYPED); return accessPath; } diff --git a/gensrc/proto/descriptors.proto b/gensrc/proto/descriptors.proto index f62204ac6f6094..b3080b5cbb6c9e 100644 --- a/gensrc/proto/descriptors.proto +++ b/gensrc/proto/descriptors.proto @@ -64,6 +64,9 @@ message PColumnAccessPath { required PAccessPathType type = 1; optional PDataAccessPath data_access_path = 2; optional PMetaAccessPath meta_access_path = 3; + // 0/absent is the legacy all-DATA encoding with special path components. Version 1 uses + // typed payloads and may carry meta_access_path. + optional int32 version = 4; } message PSlotDescriptor { diff --git a/gensrc/thrift/Descriptors.thrift b/gensrc/thrift/Descriptors.thrift index 3bce52b2f87fed..3472a8a659058c 100644 --- a/gensrc/thrift/Descriptors.thrift +++ b/gensrc/thrift/Descriptors.thrift @@ -64,10 +64,17 @@ struct TMetaAccessPath { 1: required list path } +const i32 TCOLUMN_ACCESS_PATH_VERSION_LEGACY = 0 +const i32 TCOLUMN_ACCESS_PATH_VERSION_TYPED = 1 + struct TColumnAccessPath { 1: required TAccessPathType type 2: optional TDataAccessPath data_access_path 3: optional TMetaAccessPath meta_access_path + // The version is absent for legacy senders. Legacy paths all have DATA type and encode + // KEYS/VALUES/* selectors plus NULL/OFFSET metadata in data_access_path. Starting from the + // typed version, type selects the authoritative payload and META may use meta_access_path. + 4: optional i32 version } struct TColumn { diff --git a/gensrc/thrift/PaloInternalService.thrift b/gensrc/thrift/PaloInternalService.thrift index 9e6d34c7c39621..8a19feb4b7642c 100644 --- a/gensrc/thrift/PaloInternalService.thrift +++ b/gensrc/thrift/PaloInternalService.thrift @@ -511,6 +511,8 @@ struct TQueryOptions { // Fall back to RE2 when Hyperscan cannot compile a regular expression. 228: optional bool enable_hyperscan_fallback = true; + // Master uses 226 for this option; branch-4.2 already uses 226-228, so the id was moved. + 229: optional bool enable_prune_nested_column = false; // For cloud, to control if the content would be written into file cache // In write path, to control if the content would be written into file cache. // In read path, read from file cache or remote storage when execute query. diff --git a/regression-test/data/datatype_p0/complex_types/test_pruned_columns.out b/regression-test/data/datatype_p0/complex_types/test_pruned_columns.out index b3312aa670c066..60d6198b597288 100644 --- a/regression-test/data/datatype_p0/complex_types/test_pruned_columns.out +++ b/regression-test/data/datatype_p0/complex_types/test_pruned_columns.out @@ -1,86 +1,389 @@ -- This file is automatically generated. You should know what you did if you want to edit this -- !sql -- -1 {"city":"beijing", "data":[{1:{"a":10, "b":20}, 2:{"a":30, "b":40}}], "value":1} -2 {"city":"shanghai", "data":[{2:{"a":50, "b":40}, 1:{"a":70, "b":80}}], "value":2} -3 {"city":"guangzhou", "data":[{1:{"a":90, "b":60}, 2:{"a":110, "b":40}}], "value":3} -4 {"city":"shenzhen", "data":[{2:{"a":130, "b":20}, 1:{"a":150, "b":40}}], "value":4} -5 {"city":"hangzhou", "data":[{1:{"a":170, "b":80}, 2:{"a":190, "b":40}}], "value":5} -6 {"city":"nanjing", "data":[{2:{"a":210, "b":60}, 1:{"a":230, "b":40}}], "value":6} -7 {"city":"tianjin", "data":[{1:{"a":250, "b":20}, 2:{"a":270, "b":40}}], "value":7} -8 {"city":"chongqing", "data":[{2:{"a":290, "b":80}, 1:{"a":310, "b":40}}], "value":8} -9 {"city":"wuhan", "data":[{1:{"a":330, "b":60}, 2:{"a":350, "b":40}}], "value":9} -10 {"city":"xian", "data":[{2:{"a":370, "b":20}, 1:{"a":390, "b":40}}], "value":10} -11 {"city":"changsha", "data":[{1:{"a":410, "b":80}, 2:{"a":430, "b":40}}], "value":11} -12 {"city":"qingdao", "data":[{2:{"a":450, "b":60}, 1:{"a":470, "b":40}}], "value":12} -13 {"city":"dalian", "data":[{1:{"a":490, "b":20}, 2:{"a":510, "b":40}}], "value":13} +\N 300 +beijing 300 +chengdu 300 +guangzhou 300 +hangzhou 300 +nanjing 300 +shanghai 300 +shenzhen 300 +wuhan 300 +xian 300 -- !sql1 -- -1 [10] +1 [10, 5] + +-- !sql1_1 -- + +-- !sql1_2 -- -- !sql2 -- -1 beijing -2 shanghai +0 beijing +1 shanghai +2 shenzhen 3 guangzhou -4 shenzhen -5 hangzhou -6 nanjing -7 tianjin -8 chongqing -9 wuhan -10 xian -11 changsha -12 qingdao -13 dalian +4 hangzhou +5 chengdu +6 wuhan +7 xian +8 nanjing +9 \N +10 beijing +11 shanghai +12 shenzhen +13 guangzhou +14 hangzhou +15 chengdu +16 wuhan +17 xian +18 nanjing +19 \N + +-- !sql2_1 -- +100 beijing +101 shanghai +102 shenzhen +103 guangzhou +104 hangzhou +105 chengdu +106 wuhan +107 xian +108 nanjing +109 \N +110 beijing +111 shanghai +112 shenzhen +113 guangzhou +114 hangzhou +115 chengdu +116 wuhan +117 xian +118 nanjing +119 \N + +-- !sql2_2 -- +2999 \N +2998 nanjing +2997 xian +2996 wuhan +2995 chengdu +2994 hangzhou +2993 guangzhou +2992 shenzhen +2991 shanghai +2990 beijing +2989 \N +2988 nanjing +2987 xian +2986 wuhan +2985 chengdu +2984 hangzhou +2983 guangzhou +2982 shenzhen +2981 shanghai +2980 beijing -- !sql3 -- -1 [{1:{"a":10, "b":20}, 2:{"a":30, "b":40}}] -2 [{2:{"a":50, "b":40}, 1:{"a":70, "b":80}}] -3 [{1:{"a":90, "b":60}, 2:{"a":110, "b":40}}] -4 [{2:{"a":130, "b":20}, 1:{"a":150, "b":40}}] -5 [{1:{"a":170, "b":80}, 2:{"a":190, "b":40}}] -6 [{2:{"a":210, "b":60}, 1:{"a":230, "b":40}}] -7 [{1:{"a":250, "b":20}, 2:{"a":270, "b":40}}] -8 [{2:{"a":290, "b":80}, 1:{"a":310, "b":40}}] -9 [{1:{"a":330, "b":60}, 2:{"a":350, "b":40}}] -10 [{2:{"a":370, "b":20}, 1:{"a":390, "b":40}}] -11 [{1:{"a":410, "b":80}, 2:{"a":430, "b":40}}] -12 [{2:{"a":450, "b":60}, 1:{"a":470, "b":40}}] -13 [{1:{"a":490, "b":20}, 2:{"a":510, "b":40}}] +0 [{1:{"a":0, "b":0}, 2:{"a":20, "b":10}}, {1:{"a":0, "b":0}, 2:{"a":0, "b":0}}] +1 [{1:{"a":10, "b":11}, 2:{"a":30, "b":20}}, {2:{"a":5, "b":2.5}, 3:{"a":3, "b":1.5}}] +2 [{1:{"a":20, "b":22}, 2:{"a":40, "b":30}}, {3:{"a":10, "b":5}, 4:{"a":6, "b":3}}] +3 [{1:{"a":30, "b":33}, 2:{"a":50, "b":40}}, {1:{"a":15, "b":7.5}, 5:{"a":9, "b":4.5}}] +4 [{1:{"a":40, "b":44}, 2:{"a":60, "b":50}}, {2:{"a":20, "b":10}, 6:{"a":12, "b":6}}] +5 [{1:{"a":50, "b":50}, 2:{"a":70, "b":60}}, {3:{"a":25, "b":12.5}, 2:{"a":15, "b":7.5}}] +6 [{1:{"a":60, "b":61}, 2:{"a":80, "b":70}}, {1:{"a":30, "b":15}, 3:{"a":18, "b":9}}] +7 [{1:{"a":70, "b":72}, 2:{"a":90, "b":80}}, {2:{"a":35, "b":17.5}, 4:{"a":21, "b":10.5}}] +8 [{1:{"a":80, "b":83}, 2:{"a":100, "b":90}}, {3:{"a":40, "b":20}, 5:{"a":24, "b":12}}] +9 [{1:{"a":90, "b":94}, 2:{"a":110, "b":100}}, {1:{"a":45, "b":22.5}, 6:{"a":27, "b":13.5}}] +10 [{1:{"a":100, "b":100}, 2:{"a":120, "b":10}}, {2:{"a":30, "b":15}}] +11 [{1:{"a":110, "b":111}, 2:{"a":130, "b":20}}, {3:{"a":33, "b":16.5}}] +12 [{1:{"a":120, "b":122}, 2:{"a":140, "b":30}}, {1:{"a":60, "b":30}, 4:{"a":36, "b":18}}] +13 [{1:{"a":130, "b":133}, 2:{"a":150, "b":40}}, {2:{"a":65, "b":32.5}, 5:{"a":39, "b":19.5}}] +14 [{1:{"a":140, "b":144}, 2:{"a":160, "b":50}}, {3:{"a":70, "b":35}, 6:{"a":42, "b":21}}] +15 [{1:{"a":150, "b":150}, 2:{"a":170, "b":60}}, {1:{"a":75, "b":37.5}, 2:{"a":45, "b":22.5}}] +16 [{1:{"a":160, "b":161}, 2:{"a":180, "b":70}}, {2:{"a":80, "b":40}, 3:{"a":48, "b":24}}] +17 [{1:{"a":170, "b":172}, 2:{"a":190, "b":80}}, {3:{"a":85, "b":42.5}, 4:{"a":51, "b":25.5}}] +18 [{1:{"a":180, "b":183}, 2:{"a":200, "b":90}}, {1:{"a":90, "b":45}, 5:{"a":54, "b":27}}] +19 [{1:{"a":190, "b":194}, 2:{"a":210, "b":100}}, {2:{"a":95, "b":47.5}, 6:{"a":57, "b":28.5}}] + +-- !sql3_1 -- +200 [{1:{"a":2000, "b":2000}, 2:{"a":2020, "b":10}}, {3:{"a":1000, "b":500}, 2:{"a":600, "b":300}}] +201 [{1:{"a":2010, "b":2011}, 2:{"a":2030, "b":20}}, {1:{"a":1005, "b":502.5}, 3:{"a":603, "b":301.5}}] +202 [{1:{"a":2020, "b":2022}, 2:{"a":2040, "b":30}}, {2:{"a":1010, "b":505}, 4:{"a":606, "b":303}}] +203 [{1:{"a":2030, "b":2033}, 2:{"a":2050, "b":40}}, {3:{"a":1015, "b":507.5}, 5:{"a":609, "b":304.5}}] +204 [{1:{"a":2040, "b":2044}, 2:{"a":2060, "b":50}}, {1:{"a":1020, "b":510}, 6:{"a":612, "b":306}}] +205 [{1:{"a":2050, "b":2050}, 2:{"a":2070, "b":60}}, {2:{"a":615, "b":307.5}}] +206 [{1:{"a":2060, "b":2061}, 2:{"a":2080, "b":70}}, {3:{"a":618, "b":309}}] +207 [{1:{"a":2070, "b":2072}, 2:{"a":2090, "b":80}}, {1:{"a":1035, "b":517.5}, 4:{"a":621, "b":310.5}}] +208 [{1:{"a":2080, "b":2083}, 2:{"a":2100, "b":90}}, {2:{"a":1040, "b":520}, 5:{"a":624, "b":312}}] +209 [{1:{"a":2090, "b":2094}, 2:{"a":2110, "b":100}}, {3:{"a":1045, "b":522.5}, 6:{"a":627, "b":313.5}}] +210 [{1:{"a":2100, "b":2100}, 2:{"a":2120, "b":10}}, {1:{"a":1050, "b":525}, 2:{"a":630, "b":315}}] +211 [{1:{"a":2110, "b":2111}, 2:{"a":2130, "b":20}}, {2:{"a":1055, "b":527.5}, 3:{"a":633, "b":316.5}}] +212 [{1:{"a":2120, "b":2122}, 2:{"a":2140, "b":30}}, {3:{"a":1060, "b":530}, 4:{"a":636, "b":318}}] +213 [{1:{"a":2130, "b":2133}, 2:{"a":2150, "b":40}}, {1:{"a":1065, "b":532.5}, 5:{"a":639, "b":319.5}}] +214 [{1:{"a":2140, "b":2144}, 2:{"a":2160, "b":50}}, {2:{"a":1070, "b":535}, 6:{"a":642, "b":321}}] +215 [{1:{"a":2150, "b":2150}, 2:{"a":2170, "b":60}}, {3:{"a":1075, "b":537.5}, 2:{"a":645, "b":322.5}}] +216 [{1:{"a":2160, "b":2161}, 2:{"a":2180, "b":70}}, {1:{"a":1080, "b":540}, 3:{"a":648, "b":324}}] +217 [{1:{"a":2170, "b":2172}, 2:{"a":2190, "b":80}}, {2:{"a":1085, "b":542.5}, 4:{"a":651, "b":325.5}}] +218 [{1:{"a":2180, "b":2183}, 2:{"a":2200, "b":90}}, {3:{"a":1090, "b":545}, 5:{"a":654, "b":327}}] +219 [{1:{"a":2190, "b":2194}, 2:{"a":2210, "b":100}}, {1:{"a":1095, "b":547.5}, 6:{"a":657, "b":328.5}}] + +-- !sql3_2 -- +2999 [{1:{"a":29990, "b":29994}, 2:{"a":30010, "b":100}}, {3:{"a":14995, "b":7497.5}, 6:{"a":8997, "b":4498.5}}] +2998 [{1:{"a":29980, "b":29983}, 2:{"a":30000, "b":90}}, {2:{"a":14990, "b":7495}, 5:{"a":8994, "b":4497}}] +2997 [{1:{"a":29970, "b":29972}, 2:{"a":29990, "b":80}}, {1:{"a":14985, "b":7492.5}, 4:{"a":8991, "b":4495.5}}] +2996 [{1:{"a":29960, "b":29961}, 2:{"a":29980, "b":70}}, {3:{"a":8988, "b":4494}}] +2995 [{1:{"a":29950, "b":29950}, 2:{"a":29970, "b":60}}, {2:{"a":8985, "b":4492.5}}] +2994 [{1:{"a":29940, "b":29944}, 2:{"a":29960, "b":50}}, {1:{"a":14970, "b":7485}, 6:{"a":8982, "b":4491}}] +2993 [{1:{"a":29930, "b":29933}, 2:{"a":29950, "b":40}}, {3:{"a":14965, "b":7482.5}, 5:{"a":8979, "b":4489.5}}] +2992 [{1:{"a":29920, "b":29922}, 2:{"a":29940, "b":30}}, {2:{"a":14960, "b":7480}, 4:{"a":8976, "b":4488}}] +2991 [{1:{"a":29910, "b":29911}, 2:{"a":29930, "b":20}}, {1:{"a":14955, "b":7477.5}, 3:{"a":8973, "b":4486.5}}] +2990 [{1:{"a":29900, "b":29900}, 2:{"a":29920, "b":10}}, {3:{"a":14950, "b":7475}, 2:{"a":8970, "b":4485}}] +2989 [{1:{"a":29890, "b":29894}, 2:{"a":29910, "b":100}}, {2:{"a":14945, "b":7472.5}, 6:{"a":8967, "b":4483.5}}] +2988 [{1:{"a":29880, "b":29883}, 2:{"a":29900, "b":90}}, {1:{"a":14940, "b":7470}, 5:{"a":8964, "b":4482}}] +2987 [{1:{"a":29870, "b":29872}, 2:{"a":29890, "b":80}}, {3:{"a":14935, "b":7467.5}, 4:{"a":8961, "b":4480.5}}] +2986 [{1:{"a":29860, "b":29861}, 2:{"a":29880, "b":70}}, {2:{"a":14930, "b":7465}, 3:{"a":8958, "b":4479}}] +2985 [{1:{"a":29850, "b":29850}, 2:{"a":29870, "b":60}}, {1:{"a":14925, "b":7462.5}, 2:{"a":8955, "b":4477.5}}] +2984 [{1:{"a":29840, "b":29844}, 2:{"a":29860, "b":50}}, {3:{"a":14920, "b":7460}, 6:{"a":8952, "b":4476}}] +2983 [{1:{"a":29830, "b":29833}, 2:{"a":29850, "b":40}}, {2:{"a":14915, "b":7457.5}, 5:{"a":8949, "b":4474.5}}] +2982 [{1:{"a":29820, "b":29822}, 2:{"a":29840, "b":30}}, {1:{"a":14910, "b":7455}, 4:{"a":8946, "b":4473}}] +2981 [{1:{"a":29810, "b":29811}, 2:{"a":29830, "b":20}}, {3:{"a":8943, "b":4471.5}}] +2980 [{1:{"a":29800, "b":29800}, 2:{"a":29820, "b":10}}, {2:{"a":8940, "b":4470}}] -- !sql4 -- -1 [{1:{"a":10, "b":20}, 2:{"a":30, "b":40}}] -2 [{2:{"a":50, "b":40}, 1:{"a":70, "b":80}}] -3 [{1:{"a":90, "b":60}, 2:{"a":110, "b":40}}] -5 [{1:{"a":170, "b":80}, 2:{"a":190, "b":40}}] -7 [{1:{"a":250, "b":20}, 2:{"a":270, "b":40}}] -9 [{1:{"a":330, "b":60}, 2:{"a":350, "b":40}}] -11 [{1:{"a":410, "b":80}, 2:{"a":430, "b":40}}] -13 [{1:{"a":490, "b":20}, 2:{"a":510, "b":40}}] +3 [{1:{"a":30, "b":33}, 2:{"a":50, "b":40}}, {1:{"a":15, "b":7.5}, 5:{"a":9, "b":4.5}}] +13 [{1:{"a":130, "b":133}, 2:{"a":150, "b":40}}, {2:{"a":65, "b":32.5}, 5:{"a":39, "b":19.5}}] +23 [{1:{"a":230, "b":233}, 2:{"a":250, "b":40}}, {3:{"a":115, "b":57.5}, 5:{"a":69, "b":34.5}}] +33 [{1:{"a":330, "b":333}, 2:{"a":350, "b":40}}, {1:{"a":165, "b":82.5}, 5:{"a":99, "b":49.5}}] +43 [{1:{"a":430, "b":433}, 2:{"a":450, "b":40}}, {2:{"a":215, "b":107.5}, 5:{"a":129, "b":64.5}}] +53 [{1:{"a":530, "b":533}, 2:{"a":550, "b":40}}, {3:{"a":265, "b":132.5}, 5:{"a":159, "b":79.5}}] +63 [{1:{"a":630, "b":633}, 2:{"a":650, "b":40}}, {1:{"a":315, "b":157.5}, 5:{"a":189, "b":94.5}}] +73 [{1:{"a":730, "b":733}, 2:{"a":750, "b":40}}, {2:{"a":365, "b":182.5}, 5:{"a":219, "b":109.5}}] +83 [{1:{"a":830, "b":833}, 2:{"a":850, "b":40}}, {3:{"a":415, "b":207.5}, 5:{"a":249, "b":124.5}}] +93 [{1:{"a":930, "b":933}, 2:{"a":950, "b":40}}, {1:{"a":465, "b":232.5}, 5:{"a":279, "b":139.5}}] +103 [{1:{"a":1030, "b":1033}, 2:{"a":1050, "b":40}}, {2:{"a":515, "b":257.5}, 5:{"a":309, "b":154.5}}] +113 [{1:{"a":1130, "b":1133}, 2:{"a":1150, "b":40}}, {3:{"a":565, "b":282.5}, 5:{"a":339, "b":169.5}}] +123 [{1:{"a":1230, "b":1233}, 2:{"a":1250, "b":40}}, {1:{"a":615, "b":307.5}, 5:{"a":369, "b":184.5}}] +133 [{1:{"a":1330, "b":1333}, 2:{"a":1350, "b":40}}, {2:{"a":665, "b":332.5}, 5:{"a":399, "b":199.5}}] +143 [{1:{"a":1430, "b":1433}, 2:{"a":1450, "b":40}}, {3:{"a":715, "b":357.5}, 5:{"a":429, "b":214.5}}] +153 [{1:{"a":1530, "b":1533}, 2:{"a":1550, "b":40}}, {1:{"a":765, "b":382.5}, 5:{"a":459, "b":229.5}}] +163 [{1:{"a":1630, "b":1633}, 2:{"a":1650, "b":40}}, {2:{"a":815, "b":407.5}, 5:{"a":489, "b":244.5}}] +173 [{1:{"a":1730, "b":1733}, 2:{"a":1750, "b":40}}, {3:{"a":865, "b":432.5}, 5:{"a":519, "b":259.5}}] +183 [{1:{"a":1830, "b":1833}, 2:{"a":1850, "b":40}}, {1:{"a":915, "b":457.5}, 5:{"a":549, "b":274.5}}] +193 [{1:{"a":1930, "b":1933}, 2:{"a":1950, "b":40}}, {2:{"a":965, "b":482.5}, 5:{"a":579, "b":289.5}}] + +-- !sql4_1 -- +1003 [{1:{"a":10030, "b":10033}, 2:{"a":10050, "b":40}}, {2:{"a":5015, "b":2507.5}, 5:{"a":3009, "b":1504.5}}] +1013 [{1:{"a":10130, "b":10133}, 2:{"a":10150, "b":40}}, {3:{"a":5065, "b":2532.5}, 5:{"a":3039, "b":1519.5}}] +1023 [{1:{"a":10230, "b":10233}, 2:{"a":10250, "b":40}}, {1:{"a":5115, "b":2557.5}, 5:{"a":3069, "b":1534.5}}] +1033 [{1:{"a":10330, "b":10333}, 2:{"a":10350, "b":40}}, {2:{"a":5165, "b":2582.5}, 5:{"a":3099, "b":1549.5}}] +1043 [{1:{"a":10430, "b":10433}, 2:{"a":10450, "b":40}}, {3:{"a":5215, "b":2607.5}, 5:{"a":3129, "b":1564.5}}] +1053 [{1:{"a":10530, "b":10533}, 2:{"a":10550, "b":40}}, {1:{"a":5265, "b":2632.5}, 5:{"a":3159, "b":1579.5}}] +1063 [{1:{"a":10630, "b":10633}, 2:{"a":10650, "b":40}}, {2:{"a":5315, "b":2657.5}, 5:{"a":3189, "b":1594.5}}] +1073 [{1:{"a":10730, "b":10733}, 2:{"a":10750, "b":40}}, {3:{"a":5365, "b":2682.5}, 5:{"a":3219, "b":1609.5}}] +1083 [{1:{"a":10830, "b":10833}, 2:{"a":10850, "b":40}}, {1:{"a":5415, "b":2707.5}, 5:{"a":3249, "b":1624.5}}] +1093 [{1:{"a":10930, "b":10933}, 2:{"a":10950, "b":40}}, {2:{"a":5465, "b":2732.5}, 5:{"a":3279, "b":1639.5}}] +1103 [{1:{"a":11030, "b":11033}, 2:{"a":11050, "b":40}}, {3:{"a":5515, "b":2757.5}, 5:{"a":3309, "b":1654.5}}] +1113 [{1:{"a":11130, "b":11133}, 2:{"a":11150, "b":40}}, {1:{"a":5565, "b":2782.5}, 5:{"a":3339, "b":1669.5}}] +1123 [{1:{"a":11230, "b":11233}, 2:{"a":11250, "b":40}}, {2:{"a":5615, "b":2807.5}, 5:{"a":3369, "b":1684.5}}] +1133 [{1:{"a":11330, "b":11333}, 2:{"a":11350, "b":40}}, {3:{"a":5665, "b":2832.5}, 5:{"a":3399, "b":1699.5}}] +1143 [{1:{"a":11430, "b":11433}, 2:{"a":11450, "b":40}}, {1:{"a":5715, "b":2857.5}, 5:{"a":3429, "b":1714.5}}] +1153 [{1:{"a":11530, "b":11533}, 2:{"a":11550, "b":40}}, {2:{"a":5765, "b":2882.5}, 5:{"a":3459, "b":1729.5}}] +1163 [{1:{"a":11630, "b":11633}, 2:{"a":11650, "b":40}}, {3:{"a":5815, "b":2907.5}, 5:{"a":3489, "b":1744.5}}] +1173 [{1:{"a":11730, "b":11733}, 2:{"a":11750, "b":40}}, {1:{"a":5865, "b":2932.5}, 5:{"a":3519, "b":1759.5}}] +1183 [{1:{"a":11830, "b":11833}, 2:{"a":11850, "b":40}}, {2:{"a":5915, "b":2957.5}, 5:{"a":3549, "b":1774.5}}] +1193 [{1:{"a":11930, "b":11933}, 2:{"a":11950, "b":40}}, {3:{"a":5965, "b":2982.5}, 5:{"a":3579, "b":1789.5}}] + +-- !sql4_2 -- +2993 [{1:{"a":29930, "b":29933}, 2:{"a":29950, "b":40}}, {3:{"a":14965, "b":7482.5}, 5:{"a":8979, "b":4489.5}}] +2983 [{1:{"a":29830, "b":29833}, 2:{"a":29850, "b":40}}, {2:{"a":14915, "b":7457.5}, 5:{"a":8949, "b":4474.5}}] +2973 [{1:{"a":29730, "b":29733}, 2:{"a":29750, "b":40}}, {1:{"a":14865, "b":7432.5}, 5:{"a":8919, "b":4459.5}}] +2963 [{1:{"a":29630, "b":29633}, 2:{"a":29650, "b":40}}, {3:{"a":14815, "b":7407.5}, 5:{"a":8889, "b":4444.5}}] +2953 [{1:{"a":29530, "b":29533}, 2:{"a":29550, "b":40}}, {2:{"a":14765, "b":7382.5}, 5:{"a":8859, "b":4429.5}}] +2943 [{1:{"a":29430, "b":29433}, 2:{"a":29450, "b":40}}, {1:{"a":14715, "b":7357.5}, 5:{"a":8829, "b":4414.5}}] +2933 [{1:{"a":29330, "b":29333}, 2:{"a":29350, "b":40}}, {3:{"a":14665, "b":7332.5}, 5:{"a":8799, "b":4399.5}}] +2923 [{1:{"a":29230, "b":29233}, 2:{"a":29250, "b":40}}, {2:{"a":14615, "b":7307.5}, 5:{"a":8769, "b":4384.5}}] +2913 [{1:{"a":29130, "b":29133}, 2:{"a":29150, "b":40}}, {1:{"a":14565, "b":7282.5}, 5:{"a":8739, "b":4369.5}}] +2903 [{1:{"a":29030, "b":29033}, 2:{"a":29050, "b":40}}, {3:{"a":14515, "b":7257.5}, 5:{"a":8709, "b":4354.5}}] +2893 [{1:{"a":28930, "b":28933}, 2:{"a":28950, "b":40}}, {2:{"a":14465, "b":7232.5}, 5:{"a":8679, "b":4339.5}}] +2883 [{1:{"a":28830, "b":28833}, 2:{"a":28850, "b":40}}, {1:{"a":14415, "b":7207.5}, 5:{"a":8649, "b":4324.5}}] +2873 [{1:{"a":28730, "b":28733}, 2:{"a":28750, "b":40}}, {3:{"a":14365, "b":7182.5}, 5:{"a":8619, "b":4309.5}}] +2863 [{1:{"a":28630, "b":28633}, 2:{"a":28650, "b":40}}, {2:{"a":14315, "b":7157.5}, 5:{"a":8589, "b":4294.5}}] +2853 [{1:{"a":28530, "b":28533}, 2:{"a":28550, "b":40}}, {1:{"a":14265, "b":7132.5}, 5:{"a":8559, "b":4279.5}}] +2843 [{1:{"a":28430, "b":28433}, 2:{"a":28450, "b":40}}, {3:{"a":14215, "b":7107.5}, 5:{"a":8529, "b":4264.5}}] +2833 [{1:{"a":28330, "b":28333}, 2:{"a":28350, "b":40}}, {2:{"a":14165, "b":7082.5}, 5:{"a":8499, "b":4249.5}}] +2823 [{1:{"a":28230, "b":28233}, 2:{"a":28250, "b":40}}, {1:{"a":14115, "b":7057.5}, 5:{"a":8469, "b":4234.5}}] +2813 [{1:{"a":28130, "b":28133}, 2:{"a":28150, "b":40}}, {3:{"a":14065, "b":7032.5}, 5:{"a":8439, "b":4219.5}}] +2803 [{1:{"a":28030, "b":28033}, 2:{"a":28050, "b":40}}, {2:{"a":14015, "b":7007.5}, 5:{"a":8409, "b":4204.5}}] -- !sql5 -- -1 beijing -2 shanghai 3 guangzhou -5 hangzhou -7 tianjin -9 wuhan -11 changsha -13 dalian +13 guangzhou +23 guangzhou +33 guangzhou +43 guangzhou +53 guangzhou +63 guangzhou +73 guangzhou +83 guangzhou +93 guangzhou +103 guangzhou +113 guangzhou +123 guangzhou +133 guangzhou +143 guangzhou +153 guangzhou +163 guangzhou +173 guangzhou +183 guangzhou +193 guangzhou -- !sql5_1 -- -61 +1003 guangzhou +1013 guangzhou +1023 guangzhou +1033 guangzhou +1043 guangzhou +1053 guangzhou +1063 guangzhou +1073 guangzhou +1083 guangzhou +1093 guangzhou +1103 guangzhou +1113 guangzhou +1123 guangzhou +1133 guangzhou +1143 guangzhou +1153 guangzhou +1163 guangzhou +1173 guangzhou +1183 guangzhou +1193 guangzhou -- !sql5_2 -- +2993 guangzhou +2983 guangzhou +2973 guangzhou +2963 guangzhou +2953 guangzhou +2943 guangzhou +2933 guangzhou +2923 guangzhou +2913 guangzhou +2903 guangzhou +2893 guangzhou +2883 guangzhou +2873 guangzhou +2863 guangzhou +2853 guangzhou +2843 guangzhou +2833 guangzhou +2823 guangzhou +2813 guangzhou +2803 guangzhou + +-- !sql5_3 -- 61 +-- !sql5_4 -- +61 + +-- !sql5_5 -- +9 {"city":null, "data":[{1:{"a":90, "b":94}, 2:{"a":110, "b":100}}, {1:{"a":45, "b":22.5}, 6:{"a":27, "b":13.5}}], "value":9} +19 {"city":null, "data":[{1:{"a":190, "b":194}, 2:{"a":210, "b":100}}, {2:{"a":95, "b":47.5}, 6:{"a":57, "b":28.5}}], "value":19} +29 {"city":null, "data":[{1:{"a":290, "b":294}, 2:{"a":310, "b":100}}, {3:{"a":145, "b":72.5}, 6:{"a":87, "b":43.5}}], "value":29} +39 {"city":null, "data":[{1:{"a":390, "b":394}, 2:{"a":410, "b":100}}, {1:{"a":195, "b":97.5}, 6:{"a":117, "b":58.5}}], "value":39} +49 {"city":null, "data":[{1:{"a":490, "b":494}, 2:{"a":510, "b":100}}, {2:{"a":245, "b":122.5}, 6:{"a":147, "b":73.5}}], "value":49} +59 {"city":null, "data":[{1:{"a":590, "b":594}, 2:{"a":610, "b":100}}, {3:{"a":295, "b":147.5}, 6:{"a":177, "b":88.5}}], "value":59} +69 {"city":null, "data":[{1:{"a":690, "b":694}, 2:{"a":710, "b":100}}, {1:{"a":345, "b":172.5}, 6:{"a":207, "b":103.5}}], "value":69} +79 {"city":null, "data":[{1:{"a":790, "b":794}, 2:{"a":810, "b":100}}, {2:{"a":395, "b":197.5}, 6:{"a":237, "b":118.5}}], "value":79} +89 {"city":null, "data":[{1:{"a":890, "b":894}, 2:{"a":910, "b":100}}, {3:{"a":445, "b":222.5}, 6:{"a":267, "b":133.5}}], "value":89} +99 {"city":null, "data":[{1:{"a":990, "b":994}, 2:{"a":1010, "b":100}}, {1:{"a":495, "b":247.5}, 6:{"a":297, "b":148.5}}], "value":99} +109 {"city":null, "data":[{1:{"a":1090, "b":1094}, 2:{"a":1110, "b":100}}, {2:{"a":545, "b":272.5}, 6:{"a":327, "b":163.5}}], "value":109} +119 {"city":null, "data":[{1:{"a":1190, "b":1194}, 2:{"a":1210, "b":100}}, {3:{"a":595, "b":297.5}, 6:{"a":357, "b":178.5}}], "value":119} +129 {"city":null, "data":[{1:{"a":1290, "b":1294}, 2:{"a":1310, "b":100}}, {1:{"a":645, "b":322.5}, 6:{"a":387, "b":193.5}}], "value":129} +139 {"city":null, "data":[{1:{"a":1390, "b":1394}, 2:{"a":1410, "b":100}}, {2:{"a":695, "b":347.5}, 6:{"a":417, "b":208.5}}], "value":139} +149 {"city":null, "data":[{1:{"a":1490, "b":1494}, 2:{"a":1510, "b":100}}, {3:{"a":745, "b":372.5}, 6:{"a":447, "b":223.5}}], "value":149} +159 {"city":null, "data":[{1:{"a":1590, "b":1594}, 2:{"a":1610, "b":100}}, {1:{"a":795, "b":397.5}, 6:{"a":477, "b":238.5}}], "value":159} +169 {"city":null, "data":[{1:{"a":1690, "b":1694}, 2:{"a":1710, "b":100}}, {2:{"a":845, "b":422.5}, 6:{"a":507, "b":253.5}}], "value":169} +179 {"city":null, "data":[{1:{"a":1790, "b":1794}, 2:{"a":1810, "b":100}}, {3:{"a":895, "b":447.5}, 6:{"a":537, "b":268.5}}], "value":179} +189 {"city":null, "data":[{1:{"a":1890, "b":1894}, 2:{"a":1910, "b":100}}, {1:{"a":945, "b":472.5}, 6:{"a":567, "b":283.5}}], "value":189} +199 {"city":null, "data":[{1:{"a":1990, "b":1994}, 2:{"a":2010, "b":100}}, {2:{"a":995, "b":497.5}, 6:{"a":597, "b":298.5}}], "value":199} + -- !sql6 -- -2 +5 12.5 +15 \N +25 \N +35 87.5 +45 \N +55 \N +65 162.5 +75 \N +85 \N +95 237.5 +105 \N +115 \N +125 312.5 +135 \N +145 \N +155 387.5 +165 \N +175 \N +185 462.5 +195 \N + +-- !sql6_1 -- +1005 \N +1015 \N +1025 2562.5 +1035 \N +1045 \N +1055 2637.5 +1065 \N +1075 \N +1085 2712.5 +1095 \N +1105 \N +1115 2787.5 +1125 \N +1135 \N +1145 2862.5 +1155 \N +1165 \N +1175 2937.5 +1185 \N +1195 \N + +-- !sql6_2 -- +2995 \N +2985 \N +2975 7437.5 +2965 \N +2955 \N +2945 7362.5 +2935 \N +2925 \N +2915 7287.5 +2905 \N +2895 \N +2885 7212.5 +2875 \N +2865 \N +2855 7137.5 +2845 \N +2835 \N +2825 7062.5 +2815 \N +2805 \N -- !sql7 -- +2 + +-- !sql8 -- 0.41 0.99 --- !sql8 -- +-- !sql9 -- \N added_z diff --git a/regression-test/data/nereids_rules_p0/column_pruning/left_join_not_null_column.out b/regression-test/data/nereids_rules_p0/column_pruning/left_join_not_null_column.out new file mode 100644 index 00000000000000..8a98c58828acaa --- /dev/null +++ b/regression-test/data/nereids_rules_p0/column_pruning/left_join_not_null_column.out @@ -0,0 +1,4 @@ +-- This file is automatically generated. You should know what you did if you want to edit this +-- !left_join_not_null_column -- +3 6 60 600.00 1 + diff --git a/regression-test/data/nereids_rules_p0/column_pruning/nested_container_offset_pruning.out b/regression-test/data/nereids_rules_p0/column_pruning/nested_container_offset_pruning.out index 8b2d33fd6e3e32..e4dc236aa24916 100644 --- a/regression-test/data/nereids_rules_p0/column_pruning/nested_container_offset_pruning.out +++ b/regression-test/data/nereids_rules_p0/column_pruning/nested_container_offset_pruning.out @@ -1,7 +1,19 @@ -- This file is automatically generated. You should know what you did if you want to edit this -- !struct_root_arr_mixed -- 1 2 10 +2 0 \N +3 1 30 -- !struct_root_map_mixed -- 1 1 x +2 6 longer +3 \N only-b + +-- !struct_root_arr_predicate_mixed -- +1 hello +3 empty + +-- !struct_root_map_predicate_mixed -- +1 x +2 longer diff --git a/regression-test/suites/datatype_p0/complex_types/test_pruned_columns.groovy b/regression-test/suites/datatype_p0/complex_types/test_pruned_columns.groovy index 4e70c26819f482..be802dc52ee9bc 100644 --- a/regression-test/suites/datatype_p0/complex_types/test_pruned_columns.groovy +++ b/regression-test/suites/datatype_p0/complex_types/test_pruned_columns.groovy @@ -15,7 +15,11 @@ // specific language governing permissions and limitations // under the License. +import org.apache.doris.regression.action.ProfileAction + suite("test_pruned_columns") { + sql "set batch_size = 32;" + sql "set enable_prune_nested_column = true" sql """DROP TABLE IF EXISTS `tbl_test_pruned_columns`""" sql """ CREATE TABLE `tbl_test_pruned_columns` ( @@ -23,59 +27,171 @@ suite("test_pruned_columns") { `s` struct>>, value:int> NULL ) ENGINE=OLAP DUPLICATE KEY(`id`) - DISTRIBUTED BY RANDOM BUCKETS AUTO + DISTRIBUTED BY RANDOM BUCKETS 2 PROPERTIES ( "replication_allocation" = "tag.location.default: 1" ); """ sql """ - insert into `tbl_test_pruned_columns` values - (1, named_struct('city', 'beijing', 'data', array(map(1, named_struct('a', 10, 'b', 20.0), 2, named_struct('a', 30, 'b', 40))), 'value', 1)), - (2, named_struct('city', 'shanghai', 'data', array(map(2, named_struct('a', 50, 'b', 40.0), 1, named_struct('a', 70, 'b', 80))), 'value', 2)), - (3, named_struct('city', 'guangzhou', 'data', array(map(1, named_struct('a', 90, 'b', 60.0), 2, named_struct('a', 110, 'b', 40))), 'value', 3)), - (4, named_struct('city', 'shenzhen', 'data', array(map(2, named_struct('a', 130, 'b', 20.0), 1, named_struct('a', 150, 'b', 40))), 'value', 4)), - (5, named_struct('city', 'hangzhou', 'data', array(map(1, named_struct('a', 170, 'b', 80.0), 2, named_struct('a', 190, 'b', 40))), 'value', 5)), - (6, named_struct('city', 'nanjing', 'data', array(map(2, named_struct('a', 210, 'b', 60.0), 1, named_struct('a', 230, 'b', 40))), 'value', 6)), - (7, named_struct('city', 'tianjin', 'data', array(map(1, named_struct('a', 250, 'b', 20.0), 2, named_struct('a', 270, 'b', 40))), 'value', 7)), - (8, named_struct('city', 'chongqing', 'data', array(map(2, named_struct('a', 290, 'b', 80.0), 1, named_struct('a', 310, 'b', 40))), 'value', 8)), - (9, named_struct('city', 'wuhan', 'data', array(map(1, named_struct('a', 330, 'b', 60.0), 2, named_struct('a', 350, 'b', 40))), 'value', 9)), - (10, named_struct('city', 'xian', 'data', array(map(2, named_struct('a', 370, 'b', 20.0), 1, named_struct('a', 390, 'b', 40))), 'value', 10)), - (11, named_struct('city', 'changsha', 'data', array(map(1, named_struct('a', 410, 'b', 80.0), 2, named_struct('a', 430, 'b', 40))), 'value', 11)), - (12, named_struct('city', 'qingdao', 'data', array(map(2, named_struct('a', 450, 'b', 60.0), 1, named_struct('a', 470, 'b', 40))), 'value', 12)), - (13, named_struct('city', 'dalian', 'data', array(map(1, named_struct('a', 490, 'b', 20.0), 2, named_struct('a', 510, 'b', 40))), 'value', 13)); + insert into `tbl_test_pruned_columns` + select + number as id, + named_struct( + 'city', + case (number % 10) + when 0 then 'beijing' + when 1 then 'shanghai' + when 2 then 'shenzhen' + when 3 then 'guangzhou' + when 4 then 'hangzhou' + when 5 then 'chengdu' + when 6 then 'wuhan' + when 7 then 'xian' + when 8 then 'nanjing' + else null + end, + 'data', + array( + map( + 1, named_struct('a', number * 10, 'b', (number * 10 + number % 5) * 1.0), + 2, named_struct('a', number * 10 + 20, 'b', (number % 10 + 1) * 10.0) + ), + map( + (number % 3 + 1), named_struct('a', number * 5, 'b', number * 2.5), + (number % 5 + 2), named_struct('a', number * 3, 'b', number * 1.5) + ) + ), + 'value', + number + ) as s + from numbers("number" = "3000"); """ qt_sql """ - select * from `tbl_test_pruned_columns` order by 1; + select element_at(s, 'city'), count() from `tbl_test_pruned_columns` group by element_at(s, 'city') order by 1, 2; """ qt_sql1 """ - select b.id, array_map(x -> element_at(map_values(x)[1], 'a'), element_at(s, 'data')) from `tbl_test_pruned_columns` t join (select 1 id) b on t.id = b.id order by 1; + select + b.id + , array_map(x -> element_at(map_values(x)[1], 'a') + , element_at(s, 'data')) + from `tbl_test_pruned_columns` t join (select 1 id) b on t.id = b.id + order by 1, 2 limit 0, 20; + """ + + qt_sql1_1 """ + select + b.id + , array_map(x -> element_at(map_values(x)[1], 'a') + , element_at(s, 'data')) + from `tbl_test_pruned_columns` t join (select 1 id) b on t.id = b.id + order by 1, 2 limit 100, 20; + """ + + qt_sql1_2 """ + select + b.id + , array_map(x -> element_at(map_values(x)[1], 'a') + , element_at(s, 'data')) + from `tbl_test_pruned_columns` t join (select 1 id) b on t.id = b.id + order by 1 desc, 2 limit 100, 20; """ qt_sql2 """ - select id, element_at(s, 'city') from `tbl_test_pruned_columns` order by 1; + select id, element_at(s, 'city') from `tbl_test_pruned_columns` order by 1 limit 0, 20; + """ + + qt_sql2_1 """ + select id, element_at(s, 'city') from `tbl_test_pruned_columns` order by 1 limit 100, 20; + """ + + qt_sql2_2 """ + select id, element_at(s, 'city') from `tbl_test_pruned_columns` order by 1 desc limit 0, 20; """ qt_sql3 """ - select id, element_at(s, 'data') from `tbl_test_pruned_columns` order by 1; + select id, element_at(s, 'data') from `tbl_test_pruned_columns` order by 1 limit 0, 20; + """ + + qt_sql3_1 """ + select id, element_at(s, 'data') from `tbl_test_pruned_columns` order by 1 limit 200, 20; + """ + + qt_sql3_2 """ + select id, element_at(s, 'data') from `tbl_test_pruned_columns` order by 1 desc limit 0, 20; """ qt_sql4 """ - select id, element_at(s, 'data') from `tbl_test_pruned_columns` where element_at(element_at(s, 'data')[1][2], 'b') = 40 order by 1; + select + id + , element_at(s, 'data') + from `tbl_test_pruned_columns` + where element_at(element_at(s, 'data')[1][2], 'b') = 40 + order by 1 limit 0, 20; + """ + + qt_sql4_1 """ + select + id + , element_at(s, 'data') + from `tbl_test_pruned_columns` + where element_at(element_at(s, 'data')[1][2], 'b') = 40 + order by 1 limit 100, 20; + """ + + qt_sql4_2 """ + select + id + , element_at(s, 'data') + from `tbl_test_pruned_columns` + where element_at(element_at(s, 'data')[1][2], 'b') = 40 + order by 1 desc limit 0, 20; """ qt_sql5 """ - select id, element_at(s, 'city') from `tbl_test_pruned_columns` where element_at(element_at(s, 'data')[1][2], 'b') = 40 order by 1; + select + id + , element_at(s, 'city') + from `tbl_test_pruned_columns` + where element_at(element_at(s, 'data')[1][2], 'b') = 40 + order by 1, 2 limit 0, 20; """ qt_sql5_1 """ - select /*+ set enable_prune_nested_column = 1; */ sum(s.value) from `tbl_test_pruned_columns` where id in(1,2,3,4,8,9,10,11,13); + select + id + , element_at(s, 'city') + from `tbl_test_pruned_columns` + where element_at(element_at(s, 'data')[1][2], 'b') = 40 + order by 1, 2 limit 100, 20; """ qt_sql5_2 """ - select /*+ set enable_prune_nested_column = 0; */ sum(s.value) from `tbl_test_pruned_columns` where id in(1,2,3,4,8,9,10,11,13); + select + id + , element_at(s, 'city') + from `tbl_test_pruned_columns` + where element_at(element_at(s, 'data')[1][2], 'b') = 40 + order by 1 desc, 2 limit 0, 20; + """ + + qt_sql5_3 """ + select /*+ SET_VAR(enable_prune_nested_column=true) */ sum(s.value) from `tbl_test_pruned_columns` where id in(1,2,3,4,8,9,10,11,13); + """ + + qt_sql5_4 """ + select /*+ SET_VAR(enable_prune_nested_column=false) */ sum(s.value) from `tbl_test_pruned_columns` where id in(1,2,3,4,8,9,10,11,13); + """ + + qt_sql5_5 """ + select + id + , s + from `tbl_test_pruned_columns` + where element_at(s, 'city') is null + order by 1 limit 0, 20; """ sql """DROP TABLE IF EXISTS `tbl_test_pruned_columns_map`""" @@ -98,10 +214,100 @@ suite("test_pruned_columns") { """ qt_sql6 """ - select count(element_at(dynamic_attributes['theme_preference'], 'confidence_score')) from `tbl_test_pruned_columns_map`; + select + id + , element_at(element_at(s, 'data')[2][3], 'b') + from `tbl_test_pruned_columns` + where element_at(s, 'city') = 'chengdu' + order by 1, 2 limit 0, 20; + """ + + sql "set enable_profile = true" + sql "set profile_level = 2" + sql "set enable_common_expr_pushdown = true" + + def lazyPrunedToken = "lazy_pruned_column_recovery_" + UUID.randomUUID().toString() + sql """ + select + "${lazyPrunedToken}" + , id + , element_at(s, 'data') + from `tbl_test_pruned_columns` + where element_at(s, 'city') = 'chengdu' + order by 1 limit 0, 20; + """ + + def profileAction = new ProfileAction(context) + def profileCompletionStateName = "Profile Completion State" + def profileCompletionStateComplete = "COMPLETE" + def lazyPrunedCounterName = "LazyReadPrunedTime" + def lazyPrunedProfile = "" + def lazyPrunedProfileState = "" + for (int attempt = 0; attempt < 60; attempt++) { + for (def profileItem : profileAction.getProfileList()) { + if (profileItem["Sql Statement"].toString().contains(lazyPrunedToken)) { + lazyPrunedProfileState = profileItem[profileCompletionStateName]?.toString() + def currentProfile = profileAction.getProfile(profileItem["Profile ID"].toString()) + if (currentProfile != null && !currentProfile.isEmpty()) { + lazyPrunedProfile = currentProfile + } + break + } + } + if (lazyPrunedProfileState == profileCompletionStateComplete + && lazyPrunedProfile.contains(lazyPrunedCounterName)) { + break + } + Thread.sleep(500) + } + assertTrue(lazyPrunedProfile != null && !lazyPrunedProfile.isEmpty(), + "profile not found for ${lazyPrunedToken}") + assertTrue(lazyPrunedProfileState == profileCompletionStateComplete, + "profile is not complete for ${lazyPrunedToken}, state: ${lazyPrunedProfileState}") + logger.info("${lazyPrunedToken} profile: ${lazyPrunedProfile}") + + def lazyPrunedTimer = (lazyPrunedProfile =~ /${lazyPrunedCounterName}:\s*([0-9.]+)(ns|us|ms|s)/) + boolean foundLazyPrunedTimer = false + boolean nonZeroLazyPrunedTimer = false + while (lazyPrunedTimer.find()) { + foundLazyPrunedTimer = true + if ((lazyPrunedTimer.group(1) as BigDecimal) > 0) { + nonZeroLazyPrunedTimer = true + break + } + } + assertTrue(foundLazyPrunedTimer, + "LazyReadPrunedTime not found in profile for ${lazyPrunedToken}") + assertTrue(nonZeroLazyPrunedTimer, + "LazyReadPrunedTime is zero in profile for ${lazyPrunedToken}: ${lazyPrunedProfile}") + + qt_sql6_1 """ + select + id + , element_at(element_at(s, 'data')[2][3], 'b') + from `tbl_test_pruned_columns` + where element_at(s, 'city') = 'chengdu' + order by 1, 2 limit 100, 20; """ + qt_sql6_2 """ + select + id + , element_at(element_at(s, 'data')[2][3], 'b') + from `tbl_test_pruned_columns` + where element_at(s, 'city') = 'chengdu' + order by 1 desc, 2 limit 0, 20; + """ + + sql "set enable_profile = false" + sql "unset variable profile_level" + sql "set enable_common_expr_pushdown = false" + qt_sql7 """ + select count(element_at(dynamic_attributes['theme_preference'], 'confidence_score')) from `tbl_test_pruned_columns_map`; + """ + + qt_sql8 """ select element_at(dynamic_attributes['theme_preference'], 'confidence_score') from `tbl_test_pruned_columns_map` order by id; """ @@ -113,12 +319,12 @@ suite("test_pruned_columns") { `s_info` STRUCT, `arr_s` ARRAY>, `map_s` MAP> - ) - UNIQUE KEY(`id`) - DISTRIBUTED BY HASH(`id`) BUCKETS 4 + ) + UNIQUE KEY(`id`) + DISTRIBUTED BY HASH(`id`) BUCKETS 4 PROPERTIES ( "replication_num" = "1", - "light_schema_change" = "true" + "light_schema_change" = "true" ); """ sql """ @@ -134,7 +340,7 @@ suite("test_pruned_columns") { INSERT INTO nested_sc_tbl VALUES (3, struct(30.5, 'v3', 888), array(struct(500, 600, 'added_z'), struct(501, 601, 'added_z_2')), map('k3', struct(3, 3.3))); """ - qt_sql8 """ + qt_sql9 """ select element_at(element_at(arr_s, 1), 'z') as inner_z FROM nested_sc_tbl ORDER BY id; """ -} \ No newline at end of file +} diff --git a/regression-test/suites/nereids_rules_p0/column_pruning/left_join_not_null_column.groovy b/regression-test/suites/nereids_rules_p0/column_pruning/left_join_not_null_column.groovy new file mode 100644 index 00000000000000..e7e9aa1b06de81 --- /dev/null +++ b/regression-test/suites/nereids_rules_p0/column_pruning/left_join_not_null_column.groovy @@ -0,0 +1,71 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +suite("left_join_not_null_column") { + sql "set enable_prune_nested_column = true" + + sql "drop table if exists left_join_not_null_dim" + sql "drop table if exists left_join_not_null_fact" + + sql """ + create table left_join_not_null_dim ( + id int not null, + segment varchar(32) not null + ) + unique key(id) + distributed by hash(id) buckets 1 + properties ( + "replication_num" = "1", + "enable_unique_key_merge_on_write" = "true" + ) + """ + + sql """ + create table left_join_not_null_fact ( + fact_id int not null, + dim_id int not null, + quantity int not null, + amount decimal(10, 2) not null + ) + unique key(fact_id) + distributed by hash(fact_id) buckets 1 + properties ( + "replication_num" = "1", + "enable_unique_key_merge_on_write" = "true" + ) + """ + + sql "insert into left_join_not_null_dim values (1, 'segment-a'), (2, 'segment-b')" + sql "insert into left_join_not_null_fact values (1, 1, 10, 100.00), (2, 2, 20, 200.00), (3, 3, 30, 300.00)" + + explain { + sql """ + select count(*), sum(f.fact_id), sum(f.quantity), sum(f.amount), + sum(case when d.segment is null then 1 else 0 end) + from left_join_not_null_fact f + left join left_join_not_null_dim d on f.dim_id = d.id + """ + notContains "segment.NULL" + } + + qt_left_join_not_null_column """ + select count(*), sum(f.fact_id), sum(f.quantity), sum(f.amount), + sum(case when d.segment is null then 1 else 0 end) + from left_join_not_null_fact f + left join left_join_not_null_dim d on f.dim_id = d.id + """ +} diff --git a/regression-test/suites/nereids_rules_p0/column_pruning/nested_container_offset_pruning.groovy b/regression-test/suites/nereids_rules_p0/column_pruning/nested_container_offset_pruning.groovy index 327925754e62d7..cb4baa6bdf74cb 100644 --- a/regression-test/suites/nereids_rules_p0/column_pruning/nested_container_offset_pruning.groovy +++ b/regression-test/suites/nereids_rules_p0/column_pruning/nested_container_offset_pruning.groovy @@ -30,21 +30,36 @@ suite("nested_container_offset_pruning") { PROPERTIES ("replication_allocation" = "tag.location.default: 1") """ sql """ - INSERT INTO nested_container_offset_pruning_tbl VALUES ( - 1, - named_struct( - 'arr', array( - named_struct('str_field', 'hello', 'int_field', 10), - named_struct('str_field', 'world', 'int_field', 20) - ), - 'm', {'a': 'x', 'b': 'y'} + INSERT INTO nested_container_offset_pruning_tbl VALUES + ( + 1, + named_struct( + 'arr', array( + named_struct('str_field', 'hello', 'int_field', 10), + named_struct('str_field', 'world', 'int_field', 20) + ), + 'm', {'a': 'x', 'b': 'y'} + ) + ), + ( + 2, + named_struct( + 'arr', array(), + 'm', {'a': 'longer', 'c': ''} + ) + ), + ( + 3, + named_struct( + 'arr', array(named_struct('str_field', 'empty', 'int_field', 30)), + 'm', {'b': 'only-b'} + ) ) - ) """ - // cardinality(s.arr) only needs array offsets, but element_at(...).int_field also needs - // array item data. The redundant s.arr.OFFSET path must be removed even though the root slot - // itself is STRUCT. + // cardinality(s.arr) needs array offsets, and element_at(...).int_field also needs array item + // data. Keep both paths: the BE consumes the current array-level metadata at the array + // iterator without forwarding it to the item iterator. order_qt_struct_root_arr_mixed """ SELECT id, cardinality(struct_element(s, 'arr')), @@ -52,9 +67,8 @@ suite("nested_container_offset_pruning") { FROM nested_container_offset_pruning_tbl ORDER BY id """ - // Same issue for nested maps: length(element_at(s.m, 'a')) needs the key lookup path, - // while map_values(s.m)[1] needs full value data. Dedup must therefore keep KEYS + VALUES - // and drop only the redundant value-side OFFSET path under the nested map container. + // Same issue for nested maps: length(element_at(s.m, 'a')) needs the key lookup path and + // value-string offsets, while map_values(s.m)[1] needs full value data. order_qt_struct_root_map_mixed """ SELECT id, length(element_at(struct_element(s, 'm'), 'a')), @@ -62,6 +76,26 @@ suite("nested_container_offset_pruning") { FROM nested_container_offset_pruning_tbl ORDER BY id """ + // Predicate on array offsets + output from array item should keep both current-level metadata + // and child data paths across FE access-path construction and BE iterator routing. + order_qt_struct_root_arr_predicate_mixed """ + SELECT id, + element_at(element_at(element_at(s, 'arr'), 1), 'str_field') + FROM nested_container_offset_pruning_tbl + WHERE cardinality(element_at(s, 'arr')) >= 1 + ORDER BY id + """ + + // Predicate length(map['a']) needs KEYS + VALUES.OFFSET, while projecting map['a'] still needs + // the full value data for the same map branch. + order_qt_struct_root_map_predicate_mixed """ + SELECT id, + element_at(element_at(s, 'm'), 'a') + FROM nested_container_offset_pruning_tbl + WHERE length(element_at(element_at(s, 'm'), 'a')) >= 1 + ORDER BY id + """ + explain { sql """ SELECT cardinality(struct_element(s, 'arr')), @@ -69,7 +103,7 @@ suite("nested_container_offset_pruning") { FROM nested_container_offset_pruning_tbl """ contains "s.arr.*.int_field" - notContains "s.arr.OFFSET" + contains "s.arr.OFFSET" } explain { @@ -80,6 +114,26 @@ suite("nested_container_offset_pruning") { """ contains "s.m.KEYS" contains "s.m.VALUES" - notContains "OFFSET" + contains "s.m.VALUES.OFFSET" + } + + explain { + sql """ + SELECT element_at(element_at(element_at(s, 'arr'), 1), 'str_field') + FROM nested_container_offset_pruning_tbl + WHERE cardinality(element_at(s, 'arr')) >= 1 + """ + contains "s.arr.*.str_field" + contains "s.arr.OFFSET" + } + + explain { + sql """ + SELECT element_at(element_at(s, 'm'), 'a') + FROM nested_container_offset_pruning_tbl + WHERE length(element_at(element_at(s, 'm'), 'a')) >= 1 + """ + contains "all access paths: [s.m.KEYS, s.m.VALUES, s.m.VALUES.OFFSET]" + contains "predicate access paths: [s.m.KEYS, s.m.VALUES.OFFSET]" } } diff --git a/regression-test/suites/nereids_rules_p0/column_pruning/null_column_pruning.groovy b/regression-test/suites/nereids_rules_p0/column_pruning/null_column_pruning.groovy index cbbeac2935701f..8c1b3883669548 100644 --- a/regression-test/suites/nereids_rules_p0/column_pruning/null_column_pruning.groovy +++ b/regression-test/suites/nereids_rules_p0/column_pruning/null_column_pruning.groovy @@ -18,14 +18,13 @@ // Regression tests for the IS NULL / IS NOT NULL column pruning optimization. // // When IS NULL (or IS NOT NULL) is the *only* use of a nullable column, the FE -// should emit a DATA access path with a "NULL" component so that the BE can +// should emit a META access path with a "NULL" component so that the BE can // satisfy the query by reading only the null flag instead of the full column data. // The EXPLAIN plan should show: // nested columns: : all access paths: [.NULL] // -// When the same column is also accessed for data (e.g., projected or used in -// struct_element), the NULL-only path must be stripped from allAccessPaths and -// predicateAccessPaths unless the same path is still present in allAccessPaths. +// When the same column is also accessed for data (e.g., projected or used in element_at), +// allAccessPaths keep the data path and predicateAccessPaths keep the predicate metadata path. suite("null_column_pruning") { sql """ DROP TABLE IF EXISTS ncp_tbl """ @@ -123,8 +122,7 @@ suite("null_column_pruning") { sql "select id, arr_col from ncp_tbl where arr_col is null" contains "nested columns" contains "all access paths: [arr_col]" - notContains "arr_col.NULL" - notContains "predicate access paths:" + contains "predicate access paths: [arr_col.NULL]" } order_qt_array_full_access_strips_null """ @@ -145,8 +143,7 @@ suite("null_column_pruning") { sql "select id, map_col from ncp_tbl where map_col is null" contains "nested columns" contains "all access paths: [map_col]" - notContains "map_col.NULL" - notContains "predicate access paths:" + contains "predicate access paths: [map_col.NULL]" } order_qt_map_full_access_strips_null """ @@ -180,66 +177,56 @@ suite("null_column_pruning") { // covers the same prefix and inherently includes the null flag. explain { sql "select int_col from ncp_tbl where int_col is null" - contains "nested columns" - contains "all access paths: [int_col]" - notContains "predicate access paths:" + notContains "nested columns" } order_qt_10 "select int_col from ncp_tbl where int_col is null"; // ─── Mixed: struct IS NULL + partial field access ─────────────────────────── - // struct_col IS NULL in WHERE + struct_element in SELECT → child data is also needed. - // The parent struct_col.NULL path must NOT stay in allAccessPaths with child paths. - // BE StructFileColumnIterator treats a leading NULL sub-path as NULL_MAP_ONLY; if - // allAccessPaths were [struct_col.NULL, struct_col.city], BE would skip the city - // child iterator and default-fill the projected value. The normal nullable struct - // read materializes the parent null map together with child data, and - // predicateAccessPaths is filtered so it remains a subset of allAccessPaths. + // struct_col IS NULL in WHERE + element_at in SELECT needs only the projected field data, + // while the predicate keeps the parent null map separately. explain { sql "select struct_element(struct_col, 'city') from ncp_tbl where struct_col is null" contains "nested columns" - contains "all access paths: [struct_col.city]" - notContains "predicate access paths:" + contains "all access paths: [struct_col.city, struct_col.NULL]" + contains "predicate access paths: [struct_col.NULL]" } order_qt_11 "select struct_element(struct_col, 'city') from ncp_tbl where struct_col is null"; - // This query verifies the real correctness risk: one branch needs the parent null - // map, another branch needs a child null map, and the projection needs another - // child data path. Keeping struct_col.NULL in allAccessPaths would put BE in - // NULL_MAP_ONLY mode for the whole struct and return the default zip value instead - // of reading the zip child column. + // This query verifies the real correctness risk: predicate paths need both parent and child + // null maps, while allAccessPaths keeps only the projected field data path. explain { sql "select struct_element(struct_col, 'zip') from ncp_tbl where struct_col is null or struct_element(struct_col, 'city') is null" contains "nested columns" - contains "all access paths: [struct_col.city.NULL, struct_col.zip]" - contains "predicate access paths: [struct_col.city.NULL]" + contains "all access paths: [struct_col.zip, struct_col.NULL, struct_col.city.NULL]" + contains "predicate access paths:" + contains "struct_col.NULL" + contains "struct_col.city.NULL" } order_qt_parent_null_with_child_data "select struct_element(struct_col, 'zip') from ncp_tbl where struct_col is null or struct_element(struct_col, 'city') is null"; // ─── Non-optimizable: struct IS NULL + full struct projected ──────────────── - // Full struct access covers its own null flag, so [struct_col.NULL] is stripped - // from allAccessPaths but kept in predicateAccessPaths. + // Full struct access stays in allAccessPaths. The predicate keeps [struct_col.NULL] + // in predicateAccessPaths. explain { sql "select struct_col from ncp_tbl where struct_col is null" contains "nested columns" contains "all access paths: [struct_col]" - notContains "predicate access paths:" + contains "predicate access paths: [struct_col.NULL]" } order_qt_12 "select struct_col from ncp_tbl where struct_col is null"; // ─── Nested struct field IS NULL ──────────────────────────────────────────── - // struct_element(struct_col, 'city') IS NULL should produce a null-flag-only - // predicate path [struct_col.city.NULL] while the projection reads city data. - // [struct_col.city.NULL] is stripped from allAccessPaths because [struct_col.city] - // covers the same prefix (full city data includes its null flag). + // struct_element(struct_col, 'city') IS NULL needs both the parent Struct null map and the + // selected field null map, while the projection reads city data. explain { sql "select struct_element(struct_col, 'city') from ncp_tbl where struct_element(struct_col, 'city') is null" contains "nested columns" - contains "all access paths: [struct_col.city]" - notContains "predicate access paths:" + contains "struct_col.city" + contains "predicate access paths: [struct_col.city.NULL]" } order_qt_13 "select struct_element(struct_col, 'city') from ncp_tbl where struct_element(struct_col, 'city') is null"; @@ -351,7 +338,15 @@ suite("null_column_pruning") { explain { sql "select count(1) from ncp_tbl where map_col['a'] is null" contains "nested columns" - contains "map_col.*.NULL" + contains "map_col.KEYS" + contains "map_col.VALUES.NULL" + // expectedPlan + // nested columns: + // map_col: + // origin type: map + // all access paths: [map_col.KEYS, map_col.VALUES.NULL] + // predicate access paths: [map_col.KEYS, map_col.VALUES.NULL] + } order_qt_20 "select count(1) from ncp_tbl where map_col['a'] is null"; @@ -360,7 +355,8 @@ suite("null_column_pruning") { explain { sql "select count(1) from ncp_tbl where map_col['a'] is not null" contains "nested columns" - contains "map_col.*.NULL" + contains "map_col.KEYS" + contains "map_col.VALUES.NULL" } order_qt_21 "select count(1) from ncp_tbl where map_col['a'] is not null"; @@ -393,26 +389,25 @@ suite("null_column_pruning") { order_qt_24 "select count(1) from ncp_tbl where struct_element(struct_col, 'city') is not null"; // ─── Mixed: map_keys IS NULL + map_keys projected ────────────────────────── - // Projection needs key data, while the predicate checks whether the parent map - // is NULL. The parent NULL path must not stay in either access path list, so BE - // does not switch the whole map iterator to NULL_MAP_ONLY and skip the keys child. + // Projection needs only map keys, while the predicate checks whether the parent map is NULL. + // Keep the parent NULL path independently in both path sets. explain { sql "select map_keys(map_col) from ncp_tbl where map_keys(map_col) is null" contains "nested columns" - contains "all access paths: [map_col.KEYS]" - notContains "predicate access paths:" + contains "all access paths: [map_col.KEYS, map_col.NULL]" + contains "predicate access paths: [map_col.NULL]" } order_qt_25 "select map_keys(map_col) from ncp_tbl where map_keys(map_col) is null"; // ─── Mixed: map_values IS NULL + map_values projected ────────────────────── - // Projection needs value data, while the predicate checks whether the parent + // Projection needs only map values, while the predicate checks whether the parent // map is NULL. A NULL value element does not make map_values(map_col) NULL. explain { sql "select map_values(map_col) from ncp_tbl where map_values(map_col) is null" contains "nested columns" - contains "all access paths: [map_col.VALUES]" - notContains "predicate access paths:" + contains "all access paths: [map_col.VALUES, map_col.NULL]" + contains "predicate access paths: [map_col.NULL]" } order_qt_26 "select map_values(map_col) from ncp_tbl where map_values(map_col) is null"; @@ -519,13 +514,13 @@ suite("null_column_pruning") { order_qt_33 "select 1 from ncp_tbl_nn where id is null"; // ─── length(str_col) = 0 OR str_col IS NULL ──────────────────────────────── - // length(str_col) already uses the OFFSET path, and BE can derive null-ness - // from that layout, so the extra NULL-only path is redundant. + // The OR predicate needs both pieces of metadata: length(str_col) uses OFFSET and + // str_col IS NULL uses NULL. Keep both paths in the predicate metadata set. explain { sql "select 1 from ncp_tbl where length(str_col) = 0 or str_col is null" contains "nested columns" contains "str_col.OFFSET" - notContains "str_col.NULL" + contains "str_col.NULL" } order_qt_34 "select 1 from ncp_tbl where length(str_col) = 0 or str_col is null"; diff --git a/regression-test/suites/nereids_rules_p0/column_pruning/string_length_column_pruning.groovy b/regression-test/suites/nereids_rules_p0/column_pruning/string_length_column_pruning.groovy index 8549d128ccd95f..aca98d9f817ad4 100644 --- a/regression-test/suites/nereids_rules_p0/column_pruning/string_length_column_pruning.groovy +++ b/regression-test/suites/nereids_rules_p0/column_pruning/string_length_column_pruning.groovy @@ -71,13 +71,12 @@ suite("string_length_column_pruning") { } sql "select length(str_col) from slcp_str_tbl" - // length(str_col) in IF plus ORDER BY on a plain primitive column: - // only str_col should appear in nested columns, and NULL is redundant when OFFSET exists. + // [str_col, OFFSET] strips [str_col, NULL]. explain { sql "select if(length(str_col) >= 5, true, false) a from slcp_str_tbl order by id" contains "nested columns" contains "str_col.OFFSET" - notContains "str_col.NULL" + contains "str_col.NULL" notContains "all access paths: [id]" } sql "select if(length(str_col) >= 5, true, false) a from slcp_str_tbl order by id" @@ -147,14 +146,16 @@ suite("string_length_column_pruning") { notContains "type=bigint" } sql "select sum(cardinality(arr_col)) from slcp_str_tbl" - // arr_col also accessed via element_at → full element data needed, OFFSET suppressed. + // arr_col is also accessed via element_at, so full element data is needed. Keep OFFSET + // as well because cardinality(arr_col) still needs array offset metadata. explain { sql "select cardinality(arr_col), arr_col[1] from slcp_str_tbl" - notContains "OFFSET" + contains "arr_col.*" + contains "arr_col.OFFSET" notContains "type=bigint" } - // Full access to the same array field covers its OFFSET metadata for any data type. + // [arr_col] strips [arr_col, OFFSET]. explain { sql "select id, cardinality(arr_col), arr_col from slcp_str_tbl" contains "nested columns" @@ -243,9 +244,7 @@ suite("string_length_column_pruning") { // ─── Map with complex value cases ──────────────────────────────────────────── - // cardinality(map_arr_col['a']): value is ARRAY. - // Keys read in full (element lookup); values need only the OFFSET array (array size). - // Expected paths: map_arr_col.KEYS + map_arr_col.VALUES.OFFSET + // Expected paths: [map_arr_col, KEYS] + [map_arr_col, VALUES, OFFSET] explain { sql "select cardinality(map_arr_col['a']) from slcp_str_tbl" contains "nested columns" @@ -265,27 +264,35 @@ suite("string_length_column_pruning") { notContains "type=bigint" } - // value also accessed directly (arr[0]) → full VALUES needed, OFFSET suppressed + // Value is also accessed directly (arr[0]), so full VALUES element data is needed. Keep + // VALUES.OFFSET as well because cardinality(map_arr_col['a']) still needs array offset metadata. explain { sql "select cardinality(map_arr_col['a']), map_arr_col['b'][0] from slcp_str_tbl" - notContains "OFFSET" + contains "map_arr_col.KEYS" + contains "map_arr_col.VALUES.*" + contains "map_arr_col.VALUES.OFFSET" notContains "type=bigint" } - // value array item also accessed directly → full VALUES item path covers value OFFSET. + // The struct field is also accessed directly, so keep only the required verified field data. + // VALUES.OFFSET is still needed by cardinality(map_arr_struct_col['a']). explain { sql "select cardinality(map_arr_struct_col['a']), map_arr_struct_col['a'][1].verified from slcp_str_tbl" contains "nested columns" - contains "map_arr_struct_col.*.*.verified" - notContains "map_arr_struct_col.*.OFFSET" + contains "map_arr_struct_col.KEYS" + contains "map_arr_struct_col.VALUES.*.verified" + contains "map_arr_struct_col.VALUES.OFFSET" notContains "type=bigint" } + // Returning map_arr_col['a'] needs full VALUES data. Keep VALUES.OFFSET too because + // cardinality(map_arr_col['a']) also reads the selected value-array offsets. explain { sql "select id, cardinality(map_arr_col['a']), map_arr_col['a'] from slcp_str_tbl" contains "nested columns" - contains "all access paths: [map_arr_col.*]" - notContains "map_arr_col.*.OFFSET" + contains "map_arr_col.KEYS" + contains "map_arr_col.VALUES" + contains "map_arr_col.VALUES.OFFSET" notContains "predicate access paths:" notContains "type=bigint" } @@ -295,15 +302,13 @@ suite("string_length_column_pruning") { order by id """ - // Predicate OFFSET path must also be removed when the projected value field already - // makes the corresponding array data path available. predicateAccessPaths remains a - // subset of allAccessPaths. + // Keep the value-array offset in predicate paths: predicate evaluation needs it for + // cardinality(), and lazy materialization still needs the verified field after filtering. explain { sql "select map_arr_struct_col['a'][1].verified from slcp_str_tbl where cardinality(map_arr_struct_col['a']) > 0" contains "nested columns" - contains "all access paths: [map_arr_struct_col.*.*.verified]" - notContains "map_arr_struct_col.*.OFFSET" - notContains "predicate access paths:" + contains "all access paths: [map_arr_struct_col.KEYS, map_arr_struct_col.VALUES.*.verified, map_arr_struct_col.VALUES.OFFSET]" + contains "predicate access paths: [map_arr_struct_col.KEYS, map_arr_struct_col.VALUES.OFFSET]" notContains "type=bigint" } @@ -313,13 +318,13 @@ suite("string_length_column_pruning") { order by 1 """ - // value array item also accessed directly → full VALUES item path covers value NULL. + // Predicate keeps the value-array NULL path so IS NULL can read the value-array null map + // during predicate evaluation, while allAccessPaths keeps the data paths used by projection. explain { sql "select map_arr_struct_col['a'][1].verified from slcp_str_tbl where map_arr_struct_col['a'] is null" contains "nested columns" - contains "map_arr_struct_col.*.*.verified" - notContains "map_arr_struct_col.*.NULL" - notContains "predicate access paths:" + contains "all access paths: [map_arr_struct_col.KEYS, map_arr_struct_col.VALUES.*.verified, map_arr_struct_col.VALUES.NULL]" + contains "predicate access paths: [map_arr_struct_col.KEYS, map_arr_struct_col.VALUES.NULL]" } // ─── Non-optimizable cases ────────────────────────────────────────────────── @@ -414,8 +419,7 @@ suite("string_length_column_pruning") { notContains "bigint" } - // length(map_col['a']): keys read fully for element lookup, values accessed offset-only. - // Expect access paths: map_col.KEYS (full) + map_col.VALUES.OFFSET + // Expected paths: [map_col, KEYS] + [map_col, VALUES, OFFSET] explain { sql "select length(map_col['a']) from slcp_str_tbl" contains "nested columns" @@ -435,10 +439,13 @@ suite("string_length_column_pruning") { notContains "bigint" } - // length(map_col['a']) + direct map access → OFFSET suppressed, full VALUES needed + // length(map_col['a']) + direct map value access still needs full VALUES data. + // Keep VALUES.OFFSET as the length() access on map value still depends on the map value offsets. explain { sql "select length(map_col['a']), map_col['b'] from slcp_str_tbl" - notContains "OFFSET" + contains "map_col.KEYS" + contains "map_col.VALUES" + contains "map_col.VALUES.OFFSET" notContains "bigint" } @@ -522,7 +529,9 @@ suite("string_length_column_pruning") { FROM slcp_struct_root_tbl """ contains "s.arr.*.int_field" - notContains "s.arr.OFFSET" + // cardinality(element_at(s, 'arr')) still needs the nested array offsets, + // while the selected element field needs the int_field data. + contains "s.arr.OFFSET" } explain { @@ -533,14 +542,15 @@ suite("string_length_column_pruning") { """ contains "s.m.KEYS" contains "s.m.VALUES" - notContains "OFFSET" + // The length() on a map value needs value-string offsets even though the query + // also directly reads map values. + contains "s.m.VALUES.OFFSET" } // ─── Map element_at + map_values mixed access ───────────────────────────────── - // length(map_col['a']) needs keys for the element_at lookup and value offsets for length(). - // map_values(map_col)[1] needs full value data. The mixed query must therefore keep a KEYS - // path for element_at lookup while dropping the redundant value-side OFFSET path. + // Full VALUES data is still needed for map_values(), while the length() access on a + // map value keeps VALUES.OFFSET. KEYS is kept for element_at lookup. order_qt_map_element_with_map_values """ select length(map_col['a']), map_values(map_col)[1] from slcp_str_tbl """ @@ -550,16 +560,17 @@ suite("string_length_column_pruning") { contains "nested columns" contains "KEYS" contains "VALUES" - notContains "OFFSET" + contains "map_col.VALUES.OFFSET" notContains "bigint" } - // Reverse direction: length(map_values(map_col)[1]) produces [map_col, VALUES, OFFSET] - // while map_col['a'] produces [map_col, *]. The * path reads full values, so OFFSET - // must be suppressed here as well. + // Full VALUES data is needed for map_col['a']; length(map_values(map_col)[1]) keeps + // VALUES.OFFSET for value-string offsets. explain { sql "select length(map_values(map_col)[1]), map_col['a'] from slcp_str_tbl" - notContains "OFFSET" + contains "map_col.KEYS" + contains "map_col.VALUES" + contains "map_col.VALUES.OFFSET" notContains "bigint" } }