From 2bacbce4a28a6ae5dc2c9d46847dd78d08704468 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 21 Sep 2026 16:01:11 +0000 Subject: [PATCH 001/185] Backport #116335 to 26.8: Check for timeouts and KILL QUERY while covering a polygon with H3 cells --- .../functions/regular-functions/geo/h3.mdx | 14 +- src/Functions/h3GeometryToCells.cpp | 432 ++++++++++++++++++ src/Functions/h3GeometryToCells.h | 27 ++ src/Functions/h3PolygonToCells.cpp | 178 +------- .../h3PolygonToCellsWithContainment.cpp | 153 +------ .../03754_h3_polygon_to_cells_const.reference | 18 +- ...ls_with_containment_cancellation.reference | 3 + ..._to_cells_with_containment_cancellation.sh | 40 ++ ...h3_polygon_to_cells_cancellation.reference | 8 + .../05043_h3_polygon_to_cells_cancellation.sh | 52 +++ ...gon_to_cells_sliver_cancellation.reference | 2 + ...h3_polygon_to_cells_sliver_cancellation.sh | 34 ++ 12 files changed, 643 insertions(+), 318 deletions(-) create mode 100644 src/Functions/h3GeometryToCells.cpp create mode 100644 src/Functions/h3GeometryToCells.h create mode 100644 tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.reference create mode 100755 tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.sh create mode 100644 tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.reference create mode 100755 tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.sh create mode 100644 tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.reference create mode 100755 tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.sh diff --git a/docs/reference/functions/regular-functions/geo/h3.mdx b/docs/reference/functions/regular-functions/geo/h3.mdx index 6130d22faba5..77016b0de76f 100644 --- a/docs/reference/functions/regular-functions/geo/h3.mdx +++ b/docs/reference/functions/regular-functions/geo/h3.mdx @@ -304,9 +304,11 @@ SELECT arrayJoin(h3kRing(644325529233966508, 1)) AS h3index; └────────────────────┘ ``` -## h3PolygonToCells +## h3PolygonToCells {#h3polygontocells} -Returns the hexagons (at specified resolution) contained by the provided geometry, either ring or (multi-)polygon. +Returns the hexagons (at specified resolution) contained by the provided geometry, either ring or (multi-)polygon. A hexagon is contained when the geometry contains its center, which is containment mode `0` of [h3PolygonToCellsWithContainment](#h3polygontocellswithcontainment). + +Every vertex of the geometry must be on the sphere: longitude in the range `[-180, 180]` and latitude in `[-90, 90]` degrees. The order of the returned cells is not guaranteed, so sort the array when the order matters. **Syntax** @@ -334,12 +336,12 @@ SELECT h3PolygonToCells([(-122.4089866999972145,37.813318999983238),(-122.354473 ``` text title="Response" ┌────────────h3index─┐ -│ 608692970769612799 │ -│ 608692971927240703 │ │ 608692970585063423 │ -│ 608692970819944447 │ │ 608692970719281151 │ │ 608692970752835583 │ +│ 608692970769612799 │ +│ 608692970819944447 │ +│ 608692971927240703 │ │ 608692972027903999 │ └────────────────────┘ ``` @@ -348,6 +350,8 @@ SELECT h3PolygonToCells([(-122.4089866999972145,37.813318999983238),(-122.354473 Returns the hexagons (at specified resolution) covering the provided geometry, either ring or (multi-)polygon, using H3 experimental containment modes. +Every vertex of the geometry must be on the sphere: longitude in the range `[-180, 180]` and latitude in `[-90, 90]` degrees. The order of the returned cells is not guaranteed, so sort the array when the order matters. + This function maps `flags` to H3 containment modes: - `0` — `CONTAINMENT_CENTER` diff --git a/src/Functions/h3GeometryToCells.cpp b/src/Functions/h3GeometryToCells.cpp new file mode 100644 index 000000000000..24d7ea0f3fd2 --- /dev/null +++ b/src/Functions/h3GeometryToCells.cpp @@ -0,0 +1,432 @@ +#include + +#if USE_H3 + +#include +#include +#include + +#include +#include +/// H3's polygon internals: private to the library, and not declared for C++. The candidate walk below +/// is H3's own, so it needs the same primitives. +extern "C" +{ +#include +#include +#include +#include +#include +#include +} + +static constexpr size_t MAX_ARRAY_SIZE = 1 << 30; + +namespace DB +{ +namespace ErrorCodes +{ + extern const int ARGUMENT_OUT_OF_BOUND; + extern const int TOO_LARGE_ARRAY_SIZE; + extern const int INCORRECT_DATA; +} + +namespace +{ + +/// Number of cells the size estimate aims to fit in the polygon's bounding box, `MAX_SIZE_CELL_THRESHOLD` +/// in H3's `polyfill.c`. +constexpr double MAX_SIZE_CELL_THRESHOLD = 10; +/// The coordinate range `latLngToCell` accepts, `VALID_RANGE_BBOX` in H3's `polyfill.c`. +constexpr BBox VALID_RANGE_BBOX = {M_PI_2, -M_PI_2, M_PI, -M_PI}; + +SphericalPointInRadians toRadianPoint(const SphericalPoint & degree_point) +{ + Float64 lon = get<0>(degree_point) * M_PI / 180.0; + Float64 lat = get<1>(degree_point) * M_PI / 180.0; + return SphericalPointInRadians(lon, lat); +} + +/// A vertex off the sphere makes H3's bounding box, and the area it derives from it, meaningless: the size +/// estimate then counts every cell of a polygon covering much of the globe, at the requested resolution. +/// NaN and infinity are already rejected when the geometry is converted; this is the range they leave open. +void validateCoordinates(const SphericalPoint & degree_point, std::string_view function_name) +{ + const Float64 lon = get<0>(degree_point); + const Float64 lat = get<1>(degree_point); + + if (!std::isfinite(lon) || !std::isfinite(lat) || std::abs(lon) > 180.0 || std::abs(lat) > 90.0) + throw Exception( + ErrorCodes::ARGUMENT_OUT_OF_BOUND, + "The vertex ({}, {}) of the geometry passed to function {} is out of bounds " + "(longitude must be -180..180 and latitude -90..90 degrees)", + toString(lon), + toString(lat), + function_name); +} + +void checkH3Error(H3Error error, std::string_view function_name) +{ + if (error != E_SUCCESS) + throw Exception(ErrorCodes::INCORRECT_DATA, + "Failed to compute H3 polygon to cells in function {}: {}", + function_name, describeH3Error(error)); +} + +/// The next cell in the sequence of all cells to check, `nextCell` in H3's `polyfill.c`. +H3Index nextCellInSequence(H3Index cell) +{ + int res = H3_GET_RESOLUTION(cell); + while (true) + { + /// A base cell has no parent, so continue with the next one. + if (res == 0) + return baseCellNumToCell(H3_GET_BASE_CELL(cell) + 1); + + H3Index parent = cell; + H3_SET_RESOLUTION(parent, res - 1); + H3_SET_INDEX_DIGIT(parent, res, H3_DIGIT_MASK); + + /// Not the last sibling: continue with the next one, skipping the missing child of a pentagon. + const int digit = static_cast(H3_GET_INDEX_DIGIT(cell, res)); + if (digit < INVALID_DIGIT - 1) + { + H3_SET_INDEX_DIGIT(cell, res, digit + ((isPentagon(parent) && digit == CENTER_DIGIT) ? 2 : 1)); + return cell; + } + + --res; + cell = parent; + } +} + +/// The candidate walk of `iterStepPolygonCompact` in H3's `polyfill.c`, over the same H3 primitives, with a +/// cancellation checkpoint per candidate. H3 wraps the walk in one call, which can reject millions of +/// candidates before it returns a cell, leaving no place to observe `max_execution_time` or `KILL QUERY`. +/// Keep in sync with H3 when the submodule is bumped. +class CompactCellScan +{ +public: + CompactCellScan( + const GeoPolygon * polygon_, + const BBox * bboxes_, + int res_, + UInt32 flags_, + std::string_view function_name_, + CancellationBudget & budget_, + size_t units_per_candidate_) + : polygon(polygon_) + , bboxes(bboxes_) + , res(res_) + , mode(static_cast(FLAG_GET_CONTAINMENT_MODE(flags_))) + , function_name(function_name_) + , budget(budget_) + , units_per_candidate(units_per_candidate_) + { + } + + /// The next cell covering the polygon, coarser than the target resolution when all of its children are + /// covered, or `H3_NULL` once the polygon is exhausted. + H3Index next() + { + if (cell == H3_NULL) + return H3_NULL; + + /// The first candidate is the cell the walk starts at; after that, the one after the last output. + if (started) + cell = nextCellInSequence(cell); + else + started = true; + + /// Nothing covers a polygon without vertices. + if (polygon->geoloop.numVerts == 0) + cell = H3_NULL; + + while (cell) + { + budget.chargeUnits(units_per_candidate); + + const int cell_res = H3_GET_RESOLUTION(cell); + + /// Target resolution: a fine-grained check. + if (cell_res == res && matchesContainmentMode(cell, cell_res)) + return cell; + + /// Coarser cell: check the bounding box of all of its children. + if (cell_res < res) + { + BBox bbox; + checkH3Error(cellToBBox(cell, &bbox, true), function_name); + + if (bboxOverlapsBBox(&bboxes[0], &bbox)) + { + /// Quick check for possible containment, then the expensive one on the polygon. + if (bboxContainsBBox(&bboxes[0], &bbox)) + { + CellBoundary bbox_boundary = bboxToCellBoundary(&bbox); + if (cellBoundaryInsidePolygon(polygon, bboxes, &bbox_boundary, &bbox)) + return cell; + } + + /// An intersecting bounding box means every child has to be tested. + H3Index child = H3_NULL; + checkH3Error(cellToCenterChild(cell, cell_res + 1, &child), function_name); + cell = child; + continue; + } + } + + cell = nextCellInSequence(cell); + } + + return H3_NULL; + } + +private: + bool matchesContainmentMode(H3Index candidate, int candidate_res) + { + if (mode == CONTAINMENT_CENTER || mode == CONTAINMENT_OVERLAPPING || mode == CONTAINMENT_OVERLAPPING_BBOX) + { + LatLng center; + checkH3Error(cellToLatLng(candidate, ¢er), function_name); + if (pointInsidePolygon(polygon, bboxes, ¢er)) + return true; + } + + if (mode == CONTAINMENT_OVERLAPPING || mode == CONTAINMENT_OVERLAPPING_BBOX) + { + /// The polygon may be wholly contained by the cell, which its first vertex tells us. That + /// vertex exists because the caller rejects a polygon without vertices. + const LatLng first_vertex = polygon->geoloop.verts[0]; + /// Out-of-range coordinates yield false positives with `latLngToCell`. + if (bboxContains(&VALID_RANGE_BBOX, &first_vertex)) + { + H3Index polygon_cell = H3_NULL; + checkH3Error(latLngToCell(&first_vertex, candidate_res, &polygon_cell), function_name); + if (polygon_cell == candidate) + return true; + } + } + + if (mode == CONTAINMENT_FULL || mode == CONTAINMENT_OVERLAPPING || mode == CONTAINMENT_OVERLAPPING_BBOX) + { + CellBoundary boundary; + checkH3Error(cellToBoundary(candidate, &boundary), function_name); + BBox bbox; + checkH3Error(cellToBBox(candidate, &bbox, false), function_name); + + if ((mode == CONTAINMENT_FULL || mode == CONTAINMENT_OVERLAPPING_BBOX) + && cellBoundaryInsidePolygon(polygon, bboxes, &boundary, &bbox)) + return true; + + /// Center point inclusion was checked above, so overlap only needs line intersection. + if ((mode == CONTAINMENT_OVERLAPPING || mode == CONTAINMENT_OVERLAPPING_BBOX) + && cellBoundaryCrossesPolygon(polygon, bboxes, &boundary, &bbox)) + return true; + } + + if (mode == CONTAINMENT_OVERLAPPING_BBOX) + { + /// A bounding box containing all the cell's children, so that this works for the size estimate. + BBox bbox; + checkH3Error(cellToBBox(candidate, &bbox, true), function_name); + + if (bboxOverlapsBBox(&bboxes[0], &bbox)) + { + CellBoundary bbox_boundary = bboxToCellBoundary(&bbox); + if (bboxContainsBBox(&bbox, &bboxes[0]) + || pointInsidePolygon(polygon, bboxes, &bbox_boundary.verts[0]) + || cellBoundaryCrossesPolygon(polygon, bboxes, &bbox_boundary, &bbox)) + return true; + } + } + + return false; + } + + const GeoPolygon * polygon; + const BBox * bboxes; + const int res; + const ContainmentMode mode; + const std::string_view function_name; + CancellationBudget & budget; + const size_t units_per_candidate; + + H3Index cell = baseCellNumToCell(0); + bool started = false; +}; + +/// `maxPolygonToCellsSizeExperimental` in H3's `polyfill.c`, over the walk above so that it is checkpointed +/// too: a rough upper bound from the polygon's bounding box, counted at a resolution coarse enough for the +/// box to hold about `MAX_SIZE_CELL_THRESHOLD` cells. +Int64 estimateCellCount( + const GeoPolygon * polygon, + const BBox * bboxes, + int res, + std::string_view function_name, + CancellationBudget & budget, + size_t units_per_candidate) +{ + if (polygon->geoloop.numVerts == 0) + return 0; + + /// A (very) rough area of the polygon's bounding box. + const BBox & polygon_bbox = bboxes[0]; + const double polygon_bbox_area_km2 = bboxHeightRads(&polygon_bbox) * bboxWidthRads(&polygon_bbox) + / cos(fmin(fabs(polygon_bbox.north), fabs(polygon_bbox.south))) * EARTH_RADIUS_KM * EARTH_RADIUS_KM; + + /// All this needs is a general order of magnitude of the number of cells that would fit. + int estimate_res = res; + while (estimate_res > 0) + { + double average_cell_area_km2 = 0; + checkH3Error(getHexagonAreaAvgKm2(estimate_res - 1, &average_cell_area_km2), function_name); + if (polygon_bbox_area_km2 / average_cell_area_km2 <= MAX_SIZE_CELL_THRESHOLD) + break; + --estimate_res; + } + + /// Ignore the requested containment mode and use the faster overlapping-bbox one. + CompactCellScan scan( + polygon, bboxes, estimate_res, CONTAINMENT_OVERLAPPING_BBOX, function_name, budget, units_per_candidate); + + Int64 estimate = 0; + while (H3Index cell = scan.next()) + { + Int64 children_size = 0; + checkH3Error(cellToChildrenSize(cell, res, &children_size), function_name); + estimate += children_size; + } + + return estimate; +} + +LatLng toH3LatLng(const SphericalPointInRadians & point) +{ + LatLng result; + result.lat = point.get<1>(); + result.lng = point.get<0>(); + return result; +} + +class GeoPolygonContainer +{ +private: + VectorWithMemoryTracking mainLoopVerts; + VectorWithMemoryTracking> holeVerts; + + mutable GeoLoop mutableMainLoop{}; + mutable GeoPolygon mutablePolygon{}; + mutable VectorWithMemoryTracking mutableHoles; + +public: + explicit GeoPolygonContainer(VectorWithMemoryTracking && mainLoop, + VectorWithMemoryTracking> && holes = {}) + : mainLoopVerts(std::move(mainLoop)), holeVerts(std::move(holes)) {} + + const GeoPolygon * unwrap() const + { + mutableMainLoop = {static_cast(mainLoopVerts.size()), + const_cast(mainLoopVerts.data())}; + + mutableHoles.clear(); + mutableHoles.reserve(holeVerts.size()); + for (const auto & hole : holeVerts) + mutableHoles.push_back({static_cast(hole.size()), + const_cast(hole.data())}); + + mutablePolygon = {mutableMainLoop, + static_cast(mutableHoles.size()), + mutableHoles.data()}; + return &mutablePolygon; + } +}; + +} + +void appendH3Cells( + const SphericalMultiPolygon & multi_polygon, + UInt8 resolution, + UInt32 flags, + std::string_view function_name, + CancellationBudget & budget, + ColumnUInt64 & dst_data) +{ + const size_t row_start_offset = dst_data.size(); + for (const auto & polygon : multi_polygon) + { + VectorWithMemoryTracking exterior; + exterior.reserve(polygon.outer().size()); + for (const auto & point : polygon.outer()) + { + validateCoordinates(point, function_name); + exterior.push_back(toH3LatLng(toRadianPoint(point))); + } + + size_t vertex_count = exterior.size(); + + VectorWithMemoryTracking> holes; + holes.reserve(polygon.inners().size()); + for (const auto & inner : polygon.inners()) + { + VectorWithMemoryTracking hole; + hole.reserve(inner.size()); + for (const auto & point : inner) + { + validateCoordinates(point, function_name); + hole.push_back(toH3LatLng(toRadianPoint(point))); + } + vertex_count += hole.size(); + holes.emplace_back(std::move(hole)); + } + + GeoPolygonContainer polygon_wrapper(std::move(exterior), std::move(holes)); + const GeoPolygon * geo_polygon = polygon_wrapper.unwrap(); + + checkH3Error(validatePolygonFlags(flags), function_name); + + /// Bounding boxes of the polygon and of each of its holes, which every candidate check needs. + VectorWithMemoryTracking bboxes(geo_polygon->numHoles + 1); + bboxesFromGeoPolygon(geo_polygon, bboxes.data()); + + /// A candidate at the target resolution is tested against every vertex of the polygon. + const size_t units_per_candidate = 1 + vertex_count; + + /// Cheap bbox-based estimate: rejects a polygon that cannot fit before enumerating it. Rough in + /// both directions, hence the exact check per cell below. + const size_t row_size_so_far = dst_data.size() - row_start_offset; + const size_t estimate = static_cast( + estimateCellCount(geo_polygon, bboxes.data(), resolution, function_name, budget, units_per_candidate)); + if (estimate > MAX_ARRAY_SIZE || row_size_so_far + estimate > MAX_ARRAY_SIZE) + throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, + "The result of function {} (array of {} elements) will be too large with resolution = {}", + function_name, row_size_so_far + estimate, toString(resolution)); + + CompactCellScan scan( + geo_polygon, bboxes.data(), resolution, flags, function_name, budget, units_per_candidate); + + while (H3Index compact_cell = scan.next()) + { + /// A cell coarser than the requested resolution stands for all of its children. + IterCellsChildren children = iterInitParent(compact_cell, resolution); + for (; children.h; iterStepChild(&children)) + { + /// The candidates rejected on the way here are charged by the walk; a child of a covered + /// cell only costs the index arithmetic that produced it. + budget.charge(sizeof(H3Index)); + + if (dst_data.size() - row_start_offset >= MAX_ARRAY_SIZE) + throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, + "The result of function {} (array of {} elements) will be too large with resolution = {}", + function_name, dst_data.size() - row_start_offset + 1, toString(resolution)); + + /// `PODArray::push_back` grows geometrically; `IColumn::reserve` would size it exactly. + dst_data.getData().push_back(children.h); + } + } + } +} + +} + +#endif diff --git a/src/Functions/h3GeometryToCells.h b/src/Functions/h3GeometryToCells.h new file mode 100644 index 000000000000..d0224ad6b20e --- /dev/null +++ b/src/Functions/h3GeometryToCells.h @@ -0,0 +1,27 @@ +#pragma once + +#include "config.h" + +#if USE_H3 + +#include +#include +#include + +namespace DB +{ + +/// Appends the H3 cells covering `multi_polygon` to `dst_data`, charging `budget` for every candidate cell +/// examined so that it can check for a timeout or `KILL QUERY` mid-row, including while the search rejects +/// candidates without producing any. `flags` is an H3 containment mode, 0 being CONTAINMENT_CENTER. +void appendH3Cells( + const SphericalMultiPolygon & multi_polygon, + UInt8 resolution, + UInt32 flags, + std::string_view function_name, + CancellationBudget & budget, + ColumnUInt64 & dst_data); + +} + +#endif diff --git a/src/Functions/h3PolygonToCells.cpp b/src/Functions/h3PolygonToCells.cpp index 40b0b9748141..3b8aabeca6e3 100644 --- a/src/Functions/h3PolygonToCells.cpp +++ b/src/Functions/h3PolygonToCells.cpp @@ -6,6 +6,7 @@ #include #include +#include #include #include @@ -13,16 +14,11 @@ #include #include +#include #include -#include -#include -#include -#include +#include #include -#include - -static constexpr size_t MAX_ARRAY_SIZE = 1 << 30; namespace DB { @@ -30,31 +26,7 @@ namespace ErrorCodes { extern const int ILLEGAL_TYPE_OF_ARGUMENT; extern const int ARGUMENT_OUT_OF_BOUND; - extern const int TOO_LARGE_ARRAY_SIZE; extern const int ILLEGAL_COLUMN; - extern const int QUERY_WAS_CANCELLED; - extern const int INCORRECT_DATA; -} - -namespace -{ - -SphericalPointInRadians toRadianPoint(const SphericalPoint & degree_point) -{ - Float64 lon = get<0>(degree_point) * M_PI / 180.0; - Float64 lat = get<1>(degree_point) * M_PI / 180.0; - - return SphericalPointInRadians(lon, lat); -} - -LatLng toH3LatLng(const SphericalPointInRadians & point) -{ - LatLng result; - result.lat = point.get<1>(); - result.lng = point.get<0>(); - return result; -} - } /// Takes a geometry (Ring, Polygon or MultiPolygon) and returns an array of H3 hexagons that cover this geometry. @@ -77,11 +49,11 @@ class FunctionH3PolygonToCells final : public IFunction ColumnPtr executeImpl(const ColumnsWithTypeAndName & arguments, const DataTypePtr &, size_t input_rows_count) const override { - /// Resolved from the executing thread rather than captured: this instance can be stored in table - /// metadata and then run by any later query. - QueryStatusPtr process_list_element; - if (auto query_context = CurrentThread::tryGetQueryContext()) - process_list_element = query_context->getProcessListElementSafe(); + /// One polygon can expand to hundreds of millions of cells, so the executor's between-blocks check + /// cannot bound this call. Resolved from the executing thread rather than captured: this instance + /// can be stored in table metadata and then run by any later query. + const std::function check_cancellation = makeCancellationCheck(name); + CancellationBudget budget(check_cancellation); const bool is_const_geometry = isColumnConst(*arguments[0].column); @@ -107,14 +79,13 @@ class FunctionH3PolygonToCells final : public IFunction arguments[1].column->getName(), getName()); const auto & data_resolution = col_resolution->getData(); + const String function_name = getName(); auto dst_data_column = ColumnUInt64::create(); auto dst_offsets_column = ColumnArray::ColumnOffsets::create(input_rows_count); auto & dst_data = *dst_data_column; auto & dst_offsets = dst_offsets_column->getData(); - size_t current_offset = 0; - callOnGeometryDataType(arguments[0].type, [&] (const auto & type) { using TypeConverter = std::decay_t; @@ -155,9 +126,6 @@ class FunctionH3PolygonToCells final : public IFunction if (is_const_geometry) const_multi_polygon = to_multi_polygon(std::move(geometries[0])); - /// Reuse buffer across rows to avoid repeated allocations - VectorWithMemoryTracking hindex_vec; - for (size_t row = 0; row < input_rows_count; ++row) { UInt8 resolution = data_resolution[row]; @@ -173,140 +141,20 @@ class FunctionH3PolygonToCells final : public IFunction row_multi_polygon = to_multi_polygon(std::move(geometries[row])); const SphericalMultiPolygon & multi_polygon = is_const_geometry ? const_multi_polygon : row_multi_polygon; - if (process_list_element && process_list_element->isKilled()) - throw Exception(ErrorCodes::QUERY_WAS_CANCELLED, "Query was cancelled"); - - const size_t row_start_offset = current_offset; - for (const auto & polygon : multi_polygon) - { - VectorWithMemoryTracking exterior; - exterior.reserve(polygon.outer().size()); - for (const auto & point : polygon.outer()) - exterior.push_back(toH3LatLng(toRadianPoint(point))); - - VectorWithMemoryTracking> holes; - holes.reserve(polygon.inners().size()); - for (const auto & inner : polygon.inners()) - { - VectorWithMemoryTracking hole; - hole.reserve(inner.size()); - for (const auto & point : inner) - hole.push_back(toH3LatLng(toRadianPoint(point))); - - holes.emplace_back(std::move(hole)); - } - - GeoPolygonContainer polygon_wrapper(std::move(exterior), std::move(holes)); - - int64_t polygon_size = 0; - H3Error size_err = maxPolygonToCellsSize(polygon_wrapper.unwrap(), resolution, 0, &polygon_size); - if (size_err != E_SUCCESS) - throw Exception( - ErrorCodes::INCORRECT_DATA, - "Failed to estimate H3 polygon to cells size in function {}: {}", - getName(), describeH3Error(size_err)); - - const size_t vec_size = static_cast(polygon_size); - const size_t row_size_so_far = current_offset - row_start_offset; - if (vec_size > MAX_ARRAY_SIZE || row_size_so_far + vec_size > MAX_ARRAY_SIZE) - throw Exception( - ErrorCodes::TOO_LARGE_ARRAY_SIZE, - "The result of function {} (array of {} elements) will be too large with resolution argument = {}", - getName(), row_size_so_far + vec_size, toString(resolution)); - - hindex_vec.assign(vec_size, 0); - H3Error fill_err = polygonToCells(polygon_wrapper.unwrap(), resolution, 0, hindex_vec.data()); - if (fill_err != E_SUCCESS) - throw Exception( - ErrorCodes::INCORRECT_DATA, - "Failed to compute H3 polygon to cells in function {}: {}", - getName(), describeH3Error(fill_err)); - - /// Go through PODArray::reserve: it grows capacity geometrically, IColumn::reserve sizes it exactly. - dst_data.getData().reserve(dst_data.size() + vec_size); - for (auto hindex : hindex_vec) - { - if (hindex != 0) - { - ++current_offset; - dst_data.insert(hindex); - } - } - } - - if (current_offset - row_start_offset > MAX_ARRAY_SIZE) - throw Exception( - ErrorCodes::TOO_LARGE_ARRAY_SIZE, - "The result of function {} (array of {} elements) will be too large with resolution argument = {}", - getName(), current_offset - row_start_offset, toString(resolution)); - - dst_offsets[row] = current_offset; + /// CONTAINMENT_CENTER: the rule `polygonToCells` documents for itself. + appendH3Cells(multi_polygon, resolution, 0, function_name, budget, dst_data); + dst_offsets[row] = dst_data.size(); } }); return ColumnArray::create(std::move(dst_data_column), std::move(dst_offsets_column)); } - -private: - class GeoPolygonContainer - { - private: - // Store the polygon data - VectorWithMemoryTracking mainLoopVerts; - VectorWithMemoryTracking> holeVerts; - - // Temporary storage for C-style structs - mutable GeoLoop mutableMainLoop{}; - mutable GeoPolygon mutablePolygon{}; - mutable VectorWithMemoryTracking mutableHoles; - - public: - // Constructor to create from C++ data - explicit GeoPolygonContainer( - VectorWithMemoryTracking && mainLoop, - VectorWithMemoryTracking> && holes = {}) - : mainLoopVerts(std::move(mainLoop)), holeVerts(std::move(holes)) {} - - // Method to get C-style GeoPolygon pointer - const GeoPolygon * unwrap() const - { - // Prepare main loop - mutableMainLoop = { - static_cast(mainLoopVerts.size()), - const_cast(mainLoopVerts.data()) - }; - - // Prepare holes - mutableHoles.clear(); - mutableHoles.reserve(holeVerts.size()); - for (const auto& hole : holeVerts) - { - mutableHoles.push_back({ - static_cast(hole.size()), - const_cast(hole.data()) - }); - } - - // Prepare full polygon - mutablePolygon = { - mutableMainLoop, - static_cast(mutableHoles.size()), - mutableHoles.data() - }; - - return &mutablePolygon; - } - - // Additional utility methods - size_t size() const { return mainLoopVerts.size(); } - bool empty() const { return mainLoopVerts.empty(); } - }; }; REGISTER_FUNCTION(H3PolygonToCells) { factory.registerFunction(FunctionDocumentation{ - .description="Returns the hexagons (at specified resolution) contained by the provided geometry, either ring or (multi-)polygon.", + .description="Returns the hexagons (at specified resolution) contained by the provided geometry, either ring or (multi-)polygon. Every vertex must be on the sphere: longitude in -180..180 and latitude in -90..90 degrees. The order of the returned cells is not guaranteed.", .syntax = "h3PolygonToCells(geometry, resolution)", .introduced_in = {25, 11}, .category = FunctionDocumentation::Category::Geo}); diff --git a/src/Functions/h3PolygonToCellsWithContainment.cpp b/src/Functions/h3PolygonToCellsWithContainment.cpp index 27969c18ab83..a2164d603376 100644 --- a/src/Functions/h3PolygonToCellsWithContainment.cpp +++ b/src/Functions/h3PolygonToCellsWithContainment.cpp @@ -8,6 +8,7 @@ #include #include +#include #include #include @@ -16,48 +17,26 @@ #include #include #include +#include #include -#include -#include +#include #include -#include -#include #include #include -static constexpr size_t MAX_ARRAY_SIZE = 1 << 30; - namespace DB { namespace ErrorCodes { extern const int ILLEGAL_TYPE_OF_ARGUMENT; extern const int ARGUMENT_OUT_OF_BOUND; - extern const int TOO_LARGE_ARRAY_SIZE; extern const int ILLEGAL_COLUMN; - extern const int QUERY_WAS_CANCELLED; - extern const int INCORRECT_DATA; } namespace { -SphericalPointInRadians toRadianPoint(const SphericalPoint & degree_point) -{ - Float64 lon = get<0>(degree_point) * M_PI / 180.0; - Float64 lat = get<1>(degree_point) * M_PI / 180.0; - return SphericalPointInRadians(lon, lat); -} - -LatLng toH3LatLng(const SphericalPointInRadians & point) -{ - LatLng result; - result.lat = point.get<1>(); - result.lng = point.get<0>(); - return result; -} - [[noreturn]] void throwInvalidContainmentFlag(Int64 flags, std::string_view function_name) { throw Exception( @@ -103,11 +82,11 @@ class Functionh3PolygonToCellsWithContainment : public IFunction ColumnPtr executeImpl(const ColumnsWithTypeAndName & arguments, const DataTypePtr &, size_t input_rows_count) const override { - /// Resolved from the executing thread rather than captured: this instance can be stored in table - /// metadata and then run by any later query. - QueryStatusPtr process_list_element; - if (auto query_context = CurrentThread::tryGetQueryContext()) - process_list_element = query_context->getProcessListElementSafe(); + /// One polygon can expand to hundreds of millions of cells, so the executor's between-blocks check + /// cannot bound this call. Resolved from the executing thread rather than captured: this instance + /// can be stored in table metadata and then run by any later query. + const std::function check_cancellation = makeCancellationCheck(name); + CancellationBudget budget(check_cancellation); const bool is_const_geometry = isColumnConst(*arguments[0].column); @@ -144,7 +123,7 @@ class Functionh3PolygonToCellsWithContainment : public IFunction toString(MAX_H3_RES)); } - /// H3 polygonToCellsExperimental expects uint32_t flags. + /// H3's containment mode is a `uint32_t` flag mask. /// Fast path: UInt8 literals (0..3) and UInt32 columns; otherwise cast to Int64 and validate 0..3. /// All flag values are validated before geometry conversion. auto col_flags_materialized = arguments[2].column->convertToFullColumnIfConst(); @@ -189,9 +168,6 @@ class Functionh3PolygonToCellsWithContainment : public IFunction auto & dst_data = *dst_data_column; auto & dst_offsets = dst_offsets_column->getData(); - size_t current_offset = 0; - - callOnGeometryDataType(arguments[0].type, [&] (const auto & type) { using TypeConverter = std::decay_t; @@ -230,8 +206,6 @@ class Functionh3PolygonToCellsWithContainment : public IFunction if (is_const_geometry) const_multi_polygon = to_multi_polygon(std::move(geometries[0])); - VectorWithMemoryTracking hindex_vec; - for (size_t row = 0; row < input_rows_count; ++row) { const UInt8 resolution = data_resolution[row]; @@ -247,114 +221,13 @@ class Functionh3PolygonToCellsWithContainment : public IFunction const SphericalMultiPolygon & multi_polygon = is_const_geometry ? const_multi_polygon : row_multi_polygon; - if (process_list_element && process_list_element->isKilled()) - throw Exception(ErrorCodes::QUERY_WAS_CANCELLED, "Query was cancelled"); - - const size_t row_start_offset = current_offset; - for (const auto & polygon : multi_polygon) - { - VectorWithMemoryTracking exterior; - exterior.reserve(polygon.outer().size()); - for (const auto & point : polygon.outer()) - exterior.push_back(toH3LatLng(toRadianPoint(point))); - - VectorWithMemoryTracking> holes; - holes.reserve(polygon.inners().size()); - for (const auto & inner : polygon.inners()) - { - VectorWithMemoryTracking hole; - hole.reserve(inner.size()); - for (const auto & point : inner) - hole.push_back(toH3LatLng(toRadianPoint(point))); - holes.emplace_back(std::move(hole)); - } - - GeoPolygonContainer polygon_wrapper(std::move(exterior), std::move(holes)); - - int64_t polygon_size = 0; - H3Error size_err = maxPolygonToCellsSizeExperimental( - polygon_wrapper.unwrap(), resolution, flags, &polygon_size); - if (size_err != E_SUCCESS) - throw Exception(ErrorCodes::INCORRECT_DATA, - "Failed to estimate H3 polygon to cells size in function {}: {}", - getName(), describeH3Error(size_err)); - - const size_t vec_size = static_cast(polygon_size); - const size_t row_size_so_far = current_offset - row_start_offset; - if (vec_size > MAX_ARRAY_SIZE || row_size_so_far + vec_size > MAX_ARRAY_SIZE) - throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, - "The result of function {} (array of {} elements) will be too large with resolution = {}", - getName(), row_size_so_far + vec_size, toString(resolution)); - - hindex_vec.assign(vec_size, 0); - H3Error fill_err = polygonToCellsExperimental( - polygon_wrapper.unwrap(), - resolution, - flags, - static_cast(vec_size), - hindex_vec.data()); - if (fill_err != E_SUCCESS) - throw Exception(ErrorCodes::INCORRECT_DATA, - "Failed to compute H3 polygon to cells in function {}: {}", - getName(), describeH3Error(fill_err)); - - /// Go through PODArray::reserve: it grows capacity geometrically, IColumn::reserve sizes it exactly. - dst_data.getData().reserve(dst_data.size() + vec_size); - for (auto hindex : hindex_vec) - { - if (hindex != 0) - { - ++current_offset; - dst_data.insert(hindex); - } - } - } - - if (current_offset - row_start_offset > MAX_ARRAY_SIZE) - throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, - "The result of function {} (array of {} elements) will be too large with resolution = {}", - getName(), current_offset - row_start_offset, toString(resolution)); - - dst_offsets[row] = current_offset; + appendH3Cells(multi_polygon, resolution, flags, function_name, budget, dst_data); + dst_offsets[row] = dst_data.size(); } }); return ColumnArray::create(std::move(dst_data_column), std::move(dst_offsets_column)); } - -private: - class GeoPolygonContainer - { - private: - VectorWithMemoryTracking mainLoopVerts; - VectorWithMemoryTracking> holeVerts; - - mutable GeoLoop mutableMainLoop{}; - mutable GeoPolygon mutablePolygon{}; - mutable VectorWithMemoryTracking mutableHoles; - - public: - explicit GeoPolygonContainer(VectorWithMemoryTracking && mainLoop, - VectorWithMemoryTracking> && holes = {}) - : mainLoopVerts(std::move(mainLoop)), holeVerts(std::move(holes)) {} - - const GeoPolygon * unwrap() const - { - mutableMainLoop = {static_cast(mainLoopVerts.size()), - const_cast(mainLoopVerts.data())}; - - mutableHoles.clear(); - mutableHoles.reserve(holeVerts.size()); - for (const auto & hole : holeVerts) - mutableHoles.push_back({static_cast(hole.size()), - const_cast(hole.data())}); - - mutablePolygon = {mutableMainLoop, - static_cast(mutableHoles.size()), - mutableHoles.data()}; - return &mutablePolygon; - } - }; }; REGISTER_FUNCTION(h3PolygonToCellsWithContainment) @@ -365,7 +238,9 @@ REGISTER_FUNCTION(h3PolygonToCellsWithContainment) "using H3's experimental algorithm with selectable containment mode. " "Flags: 0=CONTAINMENT_CENTER, 1=CONTAINMENT_FULL, 2=CONTAINMENT_OVERLAPPING, " "3=CONTAINMENT_OVERLAPPING_BBOX. The flags argument is passed to the H3 API as UInt32; " - "use integer literals (0..3) or toUInt32. Other native integer types are converted with an accurate cast. See H3 docs.", + "use integer literals (0..3) or toUInt32. Other native integer types are converted with an accurate cast. " + "Every vertex must be on the sphere: longitude in -180..180 and latitude in -90..90 degrees. " + "The order of the returned cells is not guaranteed. See H3 docs.", .syntax = "h3PolygonToCellsWithContainment(geometry, resolution, flags)", .introduced_in = {26, 6}, .category = FunctionDocumentation::Category::Geo}); diff --git a/tests/queries/0_stateless/03754_h3_polygon_to_cells_const.reference b/tests/queries/0_stateless/03754_h3_polygon_to_cells_const.reference index 9ef0d61ba356..2e477ec1feac 100644 --- a/tests/queries/0_stateless/03754_h3_polygon_to_cells_const.reference +++ b/tests/queries/0_stateless/03754_h3_polygon_to_cells_const.reference @@ -1,21 +1,21 @@ -[608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] 1 -[608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] 1 -[608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] 1 -[608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] 1 +[608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] 1 +[608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] 1 +[608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] 1 +[608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] 1 0 [] 1 [] 2 [] 3 [] 4 [] 5 [] -6 [604189372417310719,604189371209351167,604189371075133439] -7 [608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] -[608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] 1 +6 [604189371075133439,604189371209351167,604189372417310719] +7 [608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] +[608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] 1 0 [] 1 [] 2 [] 3 [] 4 [] 5 [] -6 [604189372417310719,604189371209351167,604189371075133439] -7 [608692970769612799,608692971927240703,608692970585063423,608692970819944447,608692970719281151,608692970752835583,608692972027903999] +6 [604189371075133439,604189371209351167,604189372417310719] +7 [608692970585063423,608692970719281151,608692970752835583,608692970769612799,608692970819944447,608692971927240703,608692972027903999] diff --git a/tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.reference b/tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.reference new file mode 100644 index 000000000000..a290b1c58163 --- /dev/null +++ b/tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.reference @@ -0,0 +1,3 @@ +out of bounds (longitude must be -180..180 and latitude -90..90 degrees) +stopped at the deadline +292 1 diff --git a/tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.sh b/tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.sh new file mode 100755 index 000000000000..ed18397917fd --- /dev/null +++ b/tests/queries/0_stateless/05042_h3_polygon_to_cells_with_containment_cancellation.sh @@ -0,0 +1,40 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: the H3 library is not built in fasttest + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# A vertex off the sphere used to reach H3 as it came, and its size estimate then counted every cell of a +# polygon covering much of the globe: about 45 minutes in one call, observing neither the deadline nor +# `KILL QUERY`. The oracle is the outer `timeout`. + +timeout 30 ${CLICKHOUSE_CLIENT} --query " + SELECT sum(length(h3PolygonToCellsWithContainment([(materialize(-122.40898669999721), -1.), (2., 100.0001), (-1., -1.7976931348623157e308)], 9, 0))) FROM numbers(200) + " 2>&1 | grep -o -m 1 -F 'out of bounds (longitude must be -180..180 and latitude -90..90 degrees)' + +# Valid vertices can still ask for more work than the deadline allows: every candidate cell is tested +# against every vertex, so the comb below - 20000 vertices inside half a degree - costs 7.5 s in a single +# row. The cells are now enumerated one at a time, so the deadline is observed mid-row. The oracle is the +# outer `timeout`, which is why the query must not depend on how fast the build is. + +COMB="arrayConcat([(0., 0.)], arrayMap(i -> (0.5 + 0.4 * (i % 2), i * 0.00002), range(20000)), [(0., 0.4)])" + +if timeout 30 ${CLICKHOUSE_CLIENT} --max_execution_time 1 --query " + SELECT length(h3PolygonToCellsWithContainment(${COMB}, 10, 0)) + " 2>&1 | grep -q -F 'TIMEOUT_EXCEEDED' +then + echo 'stopped at the deadline' +else + echo 'still running after 30 seconds' +fi + +# The checkpoint must not change what the function returns. + +${CLICKHOUSE_CLIENT} --query " + SELECT + length(h3PolygonToCellsWithContainment([(-122.4089866999972145, 37.813318999983238), (-122.3544736999993603, 37.7198061999978478), (-122.4798767000009008, 37.8151571999998453)], 9, 0)), + arraySort(h3PolygonToCellsWithContainment([(-122.4089866999972145, 37.813318999983238), (-122.3544736999993603, 37.7198061999978478), (-122.4798767000009008, 37.8151571999998453)], 7, 2)) + = arraySort(h3PolygonToCellsWithContainment([(materialize(-122.4089866999972145), 37.813318999983238), (-122.3544736999993603, 37.7198061999978478), (-122.4798767000009008, 37.8151571999998453)], 7, 2)) +" diff --git a/tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.reference b/tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.reference new file mode 100644 index 000000000000..a8426675f1ba --- /dev/null +++ b/tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.reference @@ -0,0 +1,8 @@ +stopped at the deadline +out of bounds (longitude must be -180..180 and latitude -90..90 degrees) +Point's component must not be NaN +Point's component must not be infinite +5 1 1 +7 1 1 +9 1 1 +11 1 1 diff --git a/tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.sh b/tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.sh new file mode 100755 index 000000000000..f6dfaed04d30 --- /dev/null +++ b/tests/queries/0_stateless/05043_h3_polygon_to_cells_cancellation.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: the H3 library is not built in fasttest + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Every candidate cell is tested against every vertex, so the comb below - 20000 valid vertices inside half +# a degree - took 29 s in a single row, with nothing observing the deadline until the row was done. The +# oracle is the outer `timeout`: a `TIMEOUT_EXCEEDED` arriving 29 s after a 1 s deadline still fails, and +# the margin holds on a sanitizer build because the unbounded version grows with it. + +COMB="arrayConcat([(0., 0.)], arrayMap(i -> (0.5 + 0.4 * (i % 2), i * 0.00002), range(20000)), [(0., 0.4)])" + +if timeout 30 ${CLICKHOUSE_CLIENT} --max_execution_time 1 --query " + SELECT length(h3PolygonToCells(${COMB}, 10)) + " 2>&1 | grep -q -F 'TIMEOUT_EXCEEDED' +then + echo 'stopped at the deadline' +else + echo 'still running after 30 seconds' +fi + +# A vertex off the sphere is rejected instead of reaching H3, which derives a meaningless bounding box. + +${CLICKHOUSE_CLIENT} --query " + SELECT length(h3PolygonToCells([(-122.40898669999721, -1.), (2., 100.0001), (-1., -1.7976931348623157e308)], 9)) + " 2>&1 | grep -o -m 1 -F 'out of bounds (longitude must be -180..180 and latitude -90..90 degrees)' + +# A NaN vertex never reaches that check: converting the geometry rejects it first, together with infinity. + +${CLICKHOUSE_CLIENT} --query "SELECT length(h3PolygonToCells([(nan, 0.), (1., 0.), (1., 1.)], 9))" 2>&1 | + grep -o -m 1 -F "Point's component must not be NaN" + +${CLICKHOUSE_CLIENT} --query "SELECT length(h3PolygonToCells([(inf, 0.), (1., 0.), (1., 1.)], 9))" 2>&1 | + grep -o -m 1 -F "Point's component must not be infinite" + +# The cells covering a geometry are those whose center it contains, which is containment mode 0, so the two +# functions agree. + +${CLICKHOUSE_CLIENT} --query " + WITH + [(-122.4089866999972145, 37.813318999983238), (-122.3544736999993603, 37.7198061999978478), (-122.4798767000009008, 37.8151571999998453)] AS sf, + [(55.66824, 12.595493), (55.667901, 12.593991), (55.667474, 12.595117), (55.66824, 12.595493)] AS cph + SELECT + resolution, + arraySort(h3PolygonToCells(sf, resolution)) = arraySort(h3PolygonToCellsWithContainment(sf, resolution, 0)), + arraySort(h3PolygonToCells(cph, resolution)) = arraySort(h3PolygonToCellsWithContainment(cph, resolution, 0)) + FROM (SELECT arrayJoin([toUInt8(5), 7, 9, 11]) AS resolution) + ORDER BY resolution +" diff --git a/tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.reference b/tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.reference new file mode 100644 index 000000000000..7350d390d432 --- /dev/null +++ b/tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.reference @@ -0,0 +1,2 @@ +stopped at the deadline +stopped at the deadline diff --git a/tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.sh b/tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.sh new file mode 100755 index 000000000000..3d8cca9ac769 --- /dev/null +++ b/tests/queries/0_stateless/05046_h3_polygon_to_cells_sliver_cancellation.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: the H3 library is not built in fasttest + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# A needle-thin polygon whose bounding box covers much of the globe. The search descends into every cell of +# that box and the needle contains the center of almost none of them, so a checkpoint per returned cell +# observes nothing and the query used to run for minutes past the deadline. The oracle is the outer +# `timeout`, whose margin holds on a sanitizer build because the unchecked version grows with it too. + +NEEDLE="[(-100., -60.), (100., 60.), (-99.9999999, -60.)]" + +if timeout 30 ${CLICKHOUSE_CLIENT} --max_execution_time 1 --query " + SELECT length(h3PolygonToCells(${NEEDLE}, 8)) + " 2>&1 | grep -q -F 'TIMEOUT_EXCEEDED' +then + echo 'stopped at the deadline' +else + echo 'still running after 30 seconds' +fi + +# The same shape with a containment mode, where a candidate cell is also tested for overlap. + +if timeout 30 ${CLICKHOUSE_CLIENT} --max_execution_time 1 --query " + SELECT length(h3PolygonToCellsWithContainment(${NEEDLE}, 8, 1)) + " 2>&1 | grep -q -F 'TIMEOUT_EXCEEDED' +then + echo 'stopped at the deadline' +else + echo 'still running after 30 seconds' +fi From c64d3c6c3aa995fae818db86a032c2f0a75cd0b7 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 21 Sep 2026 17:51:06 +0000 Subject: [PATCH 002/185] Backport #120764 to 26.8: Do not throw from `ReplxxLineReader` destructor when the terminal is gone --- src/Client/ReplxxLineReader.cpp | 30 ++++++-- src/Server/SSH/SSHPtyHandler.cpp | 31 +++++++- tests/integration/test_ssh/test.py | 115 +++++++++++++++++++++++++++++ 3 files changed, 169 insertions(+), 7 deletions(-) diff --git a/src/Client/ReplxxLineReader.cpp b/src/Client/ReplxxLineReader.cpp index 583883b5bd87..37af94675984 100644 --- a/src/Client/ReplxxLineReader.cpp +++ b/src/Client/ReplxxLineReader.cpp @@ -25,6 +25,7 @@ #include #include #include +#include #include #include "config.h" // USE_SKIM @@ -737,12 +738,31 @@ bool ReplxxLineReader::hintChosen() ReplxxLineReader::~ReplxxLineReader() { - if (history_file_fd >= 0 && close(history_file_fd)) - rx.print("Close of history file failed: %s\n", errnoToString().c_str()); + /// `Replxx::print` may fail with `std::runtime_error("write failed")` when e.g. the pty of the embedded + /// SSH client is gone already. A destructor is implicitly `noexcept`, so letting anything escape from + /// here would `std::terminate` the whole process. + try + { + if (history_file_fd >= 0 && close(history_file_fd)) + rx.print("Close of history file failed: %s\n", errnoToString().c_str()); - /// Reset cursor blinking - if (overwrite_mode) - rx.print("%s", "\033[0 q"); + /// Reset cursor blinking + if (overwrite_mode) + rx.print("%s", "\033[0 q"); + } + catch (...) + { + /// The reporting path must not be able to escape either: the `const char *` overload of + /// `tryLogCurrentException` builds a `String` for the logger name and calls `getLogger` + /// before it reaches its own `try`, so under memory pressure it can throw as well. + try + { + tryLogCurrentException(__PRETTY_FUNCTION__); + } + catch (...) // NOLINT(bugprone-empty-catch) Ok: reporting failed, nothing more to do + { + } + } } LineReader::InputStatus ReplxxLineReader::readOneLine(const String & prompt) diff --git a/src/Server/SSH/SSHPtyHandler.cpp b/src/Server/SSH/SSHPtyHandler.cpp index 02ecb110d3c7..8edc1d477100 100644 --- a/src/Server/SSH/SSHPtyHandler.cpp +++ b/src/Server/SSH/SSHPtyHandler.cpp @@ -5,6 +5,7 @@ #include #include #include +#include #include #include #include @@ -528,6 +529,18 @@ void SSHPtyHandler::run() } bool fds_set = false; + /// `ssh_event_add_fd` allocates a wrapper that is released only by `ssh_event_remove_fd`: + /// `ssh_event_free` does not know about it. The fds therefore have to be removed on every exit + /// path, including when `poll` throws because the connection with the client is already gone. + int registered_out_fd = -1; + int registered_err_fd = -1; + SCOPE_EXIT({ + if (registered_out_fd != -1) + event.removeFd(registered_out_fd); + if (registered_err_fd != -1) + event.removeFd(registered_err_fd); + }); + do { /* Poll the main event which takes care of the session, the channel and @@ -544,10 +557,16 @@ void SSHPtyHandler::run() /* If stdout valid, add stdout to be monitored by the poll event. */ if (sdata.channel_callback->client_input_output.out != -1) + { event.addFd(sdata.channel_callback->client_input_output.out, POLLIN, process_stdout, sdata.channel_callback->channel.getCChannelPtr()); + registered_out_fd = sdata.channel_callback->client_input_output.out; + } if (sdata.channel_callback->client_input_output.err != -1) + { event.addFd(sdata.channel_callback->client_input_output.err, POLLIN, process_stderr, sdata.channel_callback->channel.getCChannelPtr()); + registered_err_fd = sdata.channel_callback->client_input_output.err; + } } while (sdata.channel_callback->channel.isOpen() && !sdata.channel_callback->hasClientFinished() && !server.isCancelled()); @@ -558,8 +577,16 @@ void SSHPtyHandler::run() sdata.channel_callback->channel.isOpen(), sdata.channel_callback->hasClientFinished(), server.isCancelled() ); - event.removeFd(sdata.channel_callback->client_input_output.out); - event.removeFd(sdata.channel_callback->client_input_output.err); + if (registered_out_fd != -1) + { + event.removeFd(registered_out_fd); + registered_out_fd = -1; + } + if (registered_err_fd != -1) + { + event.removeFd(registered_err_fd); + registered_err_fd = -1; + } /// Drain any remaining data from stdout/stderr pipes before closing the channel. /// The client may have finished writing to the pipes before the event loop had a chance diff --git a/tests/integration/test_ssh/test.py b/tests/integration/test_ssh/test.py index 5fa570bfe6c0..350aa0023645 100644 --- a/tests/integration/test_ssh/test.py +++ b/tests/integration/test_ssh/test.py @@ -1,6 +1,7 @@ import os import re import socket +import struct import subprocess import time @@ -495,3 +496,117 @@ def test_interactive_tab_completion_respects_session_user(started_cluster): instance.query("DROP USER IF EXISTS completer") instance.query("DROP TABLE IF EXISTS default.visible_completion_target") instance.query("DROP TABLE IF EXISTS default.hidden_completion_target") + + +def test_interactive_session_torn_down_with_a_dead_pty(started_cluster): + """Losing the pty while the embedded client is shutting down must not kill the server. + + `ReplxxLineReader::~ReplxxLineReader` writes an escape sequence to the + terminal to reset cursor blinking when overwrite mode was ever enabled. + `Replxx::print` throws `std::runtime_error("write failed")` when that write + does not go through, and a destructor is implicitly `noexcept`, so the + exception used to `std::terminate` the whole server process. + + Reproduce it the way a real disconnect does: turn overwrite mode on with + the `Insert` key, then drop the TCP connection with a RST so that the + server's side of the pty is gone by the time the line reader is destroyed. + """ + # The daemon watchdog restarts the server after `std::terminate`, so "the + # server answers queries again" is not evidence of anything, and neither is + # a growing `uptime()`: on a fresh cluster the uptime before the disconnect + # is only a few seconds, so a restarted server reaches a larger value well + # within the sampling window below. Pin the process identity instead: the + # watchdog restarts the server by forking a new child, so the set of + # `clickhouse-server` pids in the container changes and cannot recover. + # The pattern is anchored at argv0 so that the shell running `pgrep` (whose + # own command line contains the pattern) does not match itself. + def server_pids(): + return instance.exec_in_container( + ["bash", "-c", "pgrep -f '^[^ ]*clickhouse(-| )server' | sort -n"], + nothrow=True, + ).split() + + pids_before = server_pids() + assert pids_before, "no `clickhouse-server` process in the container" + uptime_before = float(instance.query("SELECT uptime()").strip()) + + pkey = paramiko.Ed25519Key.from_private_key_file(f"{SCRIPT_DIR}/keys/lucy_ed25519") + client = paramiko.SSHClient() + client.set_missing_host_key_policy(paramiko.AutoAddPolicy()) + client.connect( + hostname=instance.ip_address, + port=9022, + username="lucy", + pkey=pkey, + timeout=30, + ) + try: + channel = client.invoke_shell(term="xterm", width=80, height=24) + channel.settimeout(20) + output = _read_channel_until(channel, timeout=20, marker=":) ") + assert ":) " in output, f"no prompt from the embedded client: {output!r}" + + # `Insert` toggles overwrite mode, which is what makes the destructor + # print the "reset cursor blinking" sequence in the first place. Wait + # for the raw `\033[5 q` ("blinking cursor") escape the key handler + # prints: it is the only observable proof that the server really + # consumed the key and that `overwrite_mode` became true. Without it + # the destructor writes nothing and the test would pass even unfixed. + channel.sendall("\x1b[2~") + raw = b"" + deadline = time.time() + 10 + while time.time() < deadline and b"\x1b[5 q" not in raw: + if channel.recv_ready(): + raw += channel.recv(65536) + else: + time.sleep(0.05) + assert ( + b"\x1b[5 q" in raw + ), f"overwrite mode was not enabled, the destructor would write nothing: {raw!r}" + + # Abort the connection with a RST instead of a graceful shutdown, so + # writes on the server side fail rather than being silently discarded. + sock = client.get_transport().sock + sock.setsockopt(socket.SOL_SOCKET, socket.SO_LINGER, struct.pack("ii", 1, 0)) + sock.close() + finally: + client.close() + + # The session teardown is asynchronous, so keep sampling for a while. A + # sample that fails only means the teardown is still in flight, but the + # *last* sample of the window must succeed and must still show the very + # same process: the server that survived the disconnect, not a fresh one. + last_failure = None + last_uptime = None + deadline = time.time() + 30 + while time.time() < deadline: + try: + uptime = float(instance.query("SELECT uptime()").strip()) + except Exception as e: # down or restarting — the next sample decides + last_failure = e + last_uptime = None + time.sleep(0.5) + continue + last_uptime = uptime + pids = server_pids() + assert pids == pids_before, ( + "the server process was replaced after the SSH disconnect, i.e. it " + "died while tearing the session down and was started again " + f"(pids {pids} != {pids_before})" + ) + assert uptime >= uptime_before, ( + "the server restarted after the SSH disconnect, i.e. it died while " + f"tearing the session down (uptime {uptime} < {uptime_before})" + ) + time.sleep(0.5) + + assert last_uptime is not None, ( + "the server did not answer a query at the end of the 30 s window after " + f"the SSH disconnect, i.e. it died while tearing the session down: {last_failure}" + ) + + # `from_host=True` also greps the rotated logs, so a restart cannot hide the + # fatal line by rotating it out of the active `clickhouse-server.log`. + assert not instance.contains_in_log( + "std::terminate", from_host=True + ), "the server called `std::terminate` while tearing down the SSH session" From 03c5fd680390bbc0d2f68dc66b682f25c0c53ea5 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 21 Sep 2026 20:27:35 +0000 Subject: [PATCH 003/185] Backport #120202 to 26.8: Retry `poll` on `EINTR` in the vendored `nats-io` client --- contrib/nats-io | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/contrib/nats-io b/contrib/nats-io index 3e3a3f10572e..b88704dfeb55 160000 --- a/contrib/nats-io +++ b/contrib/nats-io @@ -1 +1 @@ -Subproject commit 3e3a3f10572ed9ff1b7ad17873881877cb62f301 +Subproject commit b88704dfeb55e0f1dee165b9c1daccb90d5816d4 From 314d8c4c278dc987fc2444cd81f76424a88ee990 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 05:47:42 +0000 Subject: [PATCH 004/185] Backport #120182 to 26.8: Fix wrong results for `plus`/`minus` keys with a non-integer operand --- src/Functions/FunctionHelpers.cpp | 23 ++++ src/Functions/FunctionHelpers.h | 6 ++ .../Optimizations/actionsDAGUtils.cpp | 15 +-- ...3_injective_plus_minus_type_gate.reference | 95 +++++++++++++++++ .../04833_injective_plus_minus_type_gate.sql | 97 +++++++++++++++++ ...us_type_gate_sets_shards_windows.reference | 97 +++++++++++++++++ ...us_minus_type_gate_sets_shards_windows.sql | 97 +++++++++++++++++ ..._plus_minus_type_gate_plan_shape.reference | 98 +++++++++++++++++ ...ective_plus_minus_type_gate_plan_shape.sql | 100 ++++++++++++++++++ 9 files changed, 622 insertions(+), 6 deletions(-) create mode 100644 tests/queries/0_stateless/04833_injective_plus_minus_type_gate.reference create mode 100644 tests/queries/0_stateless/04833_injective_plus_minus_type_gate.sql create mode 100644 tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.reference create mode 100644 tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.sql create mode 100644 tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.reference create mode 100644 tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.sql diff --git a/src/Functions/FunctionHelpers.cpp b/src/Functions/FunctionHelpers.cpp index f5e46079539c..5eaa0bf567a1 100644 --- a/src/Functions/FunctionHelpers.cpp +++ b/src/Functions/FunctionHelpers.cpp @@ -498,6 +498,29 @@ bool allArgumentColumnsAreConstant(const ColumnsWithTypeAndName & args) return true; } +bool plusMinusWithConstantsIsInjective( + const ColumnWithTypeAndName & left, const ColumnWithTypeAndName & right, const DataTypePtr & return_type) +{ + /// Two varying operands are not injective (`x + y` maps many pairs to one sum), and with both + /// fixed there is no varying argument to be injective in. + const bool left_is_const = left.column && isColumnConst(*left.column); + const bool right_is_const = right.column && isColumnConst(*right.column); + if (left_is_const == right_is_const) + return false; + + auto is_integer_type = [](const DataTypePtr & type) + { return type && isInteger(*removeNullable(recursiveRemoveLowCardinality(type))); }; + + if (!is_integer_type(left.type) || !is_integer_type(right.type) || !is_integer_type(return_type)) + return false; + + /// A NULL among the varying argument's values maps to NULL one-to-one, so only the fixed operand + /// matters. Its `ColumnConst` nests a column of size 1, so the value is readable even when the + /// constant itself was materialized with size 0, as query-plan constants are. + const ColumnWithTypeAndName & constant = left_is_const ? left : right; + return !constant.column->onlyNull(); +} + bool convertLowCardinalityColumnsToFull(ColumnsWithTypeAndName & args) { bool converted = false; diff --git a/src/Functions/FunctionHelpers.h b/src/Functions/FunctionHelpers.h index 8a77b9a3eda2..9260ae9861d0 100644 --- a/src/Functions/FunctionHelpers.h +++ b/src/Functions/FunctionHelpers.h @@ -217,6 +217,12 @@ bool isLowCardinalityType(const IDataType & type); bool hasLowCardinalityTypes(const ColumnsWithTypeAndName & args); /// Returns true if all of the arguments have constant columns. bool allArgumentColumnsAreConstant(const ColumnsWithTypeAndName & args); +/// Whether `plus`/`minus` is injective in its varying argument, given the other one fixed. Only +/// integer arithmetic qualifies: integer wrap-around is a bijection, while every other operand class +/// collapses distinct arguments somewhere - an `Interval` at end-of-month days and DST transitions, a +/// float or `Decimal` by rounding or rescaling, a narrower date constant, a NULL constant. +bool plusMinusWithConstantsIsInjective( + const ColumnWithTypeAndName & left, const ColumnWithTypeAndName & right, const DataTypePtr & return_type); bool convertLowCardinalityColumnsToFull(ColumnsWithTypeAndName & args); void checkFunctionArgumentSizes(const ColumnsWithTypeAndName & arguments, size_t input_rows_count); diff --git a/src/Processors/QueryPlan/Optimizations/actionsDAGUtils.cpp b/src/Processors/QueryPlan/Optimizations/actionsDAGUtils.cpp index fa04c059cff6..a4462d6004c3 100644 --- a/src/Processors/QueryPlan/Optimizations/actionsDAGUtils.cpp +++ b/src/Processors/QueryPlan/Optimizations/actionsDAGUtils.cpp @@ -2,6 +2,7 @@ #include #include +#include #include #include #include @@ -555,12 +556,14 @@ bool isInjectiveFunction(const ActionsDAG::Node * node) if (node->function_base->isInjective({})) return true; - size_t fixed_args = 0; - for (const auto & child : node->children) - if (child->type == ActionsDAG::ActionType::COLUMN) - ++fixed_args; - static const std::vector injective = {"plus", "minus", "negate", "tuple"}; - return (fixed_args + 1 >= node->children.size()) && (std::ranges::find(injective, node->function_base->getName()) != injective.end()); + const auto & name = node->function_base->getName(); + if (node->children.size() != 2 || (name != "plus" && name != "minus")) + return false; + + const auto & left = *node->children[0]; + const auto & right = *node->children[1]; + return plusMinusWithConstantsIsInjective( + {left.column, left.result_type, left.result_name}, {right.column, right.result_type, right.result_name}, node->result_type); } NodeSet removeInjectiveFunctionsFromResultsRecursively(const ActionsDAG & actions) diff --git a/tests/queries/0_stateless/04833_injective_plus_minus_type_gate.reference b/tests/queries/0_stateless/04833_injective_plus_minus_type_gate.reference new file mode 100644 index 000000000000..48531a4780d3 --- /dev/null +++ b/tests/queries/0_stateless/04833_injective_plus_minus_type_gate.reference @@ -0,0 +1,95 @@ +-- { echo } + +-- --------------------------------------------------------------------------- +-- Correctness: per-partition evaluation must agree with the merged evaluation. Each arm +-- prints the forced count and then the merged count; they must be equal. The two settings +-- apply to the outer query, so they reach the aggregation under test. +-- --------------------------------------------------------------------------- + +-- Date + INTERVAL MONTH collapses the 29th, 30th and 31st of a month into one key. 2001 and +-- 2002 are not leap years, so all three days of each map to February 28. +DROP TABLE IF EXISTS t_month; +CREATE TABLE t_month (d Date, x UInt32) ENGINE = MergeTree ORDER BY d PARTITION BY d; +INSERT INTO t_month SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-01-', toString(29 + (intDiv(number, 10) % 3)))) AS d, number FROM numbers(60); +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM t_month GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +2 +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM t_month GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +2 +SELECT count() FROM (SELECT DISTINCT d + INTERVAL 1 MONTH AS k FROM t_month) SETTINGS force_distinct_partitions_independently = 1; +2 +SELECT count() FROM (SELECT DISTINCT d + INTERVAL 1 MONTH AS k FROM t_month) SETTINGS allow_distinct_partitions_independently = 0; +2 +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k FROM t_month LIMIT 1 BY k) SETTINGS allow_limit_by_partitions_independently = 1; +2 +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k FROM t_month LIMIT 1 BY k) SETTINGS allow_limit_by_partitions_independently = 0; +2 +DROP TABLE t_month; +-- the minus direction collapses the same way, going back into a shorter month +DROP TABLE IF EXISTS t_month_minus; +CREATE TABLE t_month_minus (d Date, x UInt32) ENGINE = MergeTree ORDER BY d PARTITION BY d; +INSERT INTO t_month_minus SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-03-', toString(29 + (intDiv(number, 10) % 3)))) AS d, number FROM numbers(60); +SELECT count() FROM (SELECT d - INTERVAL 1 MONTH AS k, count() FROM t_month_minus GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +2 +SELECT count() FROM (SELECT d - INTERVAL 1 MONTH AS k, count() FROM t_month_minus GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +2 +DROP TABLE t_month_minus; +-- DateTime + INTERVAL DAY collapses across a DST spring-forward, so the interval kind alone +-- does not decide safety +DROP TABLE IF EXISTS t_dst; +CREATE TABLE t_dst (t DateTime('Europe/Moscow'), x UInt32) ENGINE = MergeTree ORDER BY t PARTITION BY t; +INSERT INTO t_dst SELECT toDateTime('2010-03-27 00:00:00', 'Europe/Moscow') + (intDiv(number, 100) * 1800) AS t, number FROM numbers_mt(600); +SELECT count() FROM (SELECT t + INTERVAL 1 DAY AS k, count() FROM t_dst GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +4 +SELECT count() FROM (SELECT t + INTERVAL 1 DAY AS k, count() FROM t_dst GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +4 +DROP TABLE t_dst; +-- a Float64 addend collapses by mantissa rounding; a float column reaches the optimizer +-- through a non-float partition key +DROP TABLE IF EXISTS t_float; +CREATE TABLE t_float (f Float64, x UInt32) ENGINE = MergeTree ORDER BY f PARTITION BY toUInt64(f - 1e16); +INSERT INTO t_float SELECT 1e16 + intDiv(number, 100) AS f, number FROM numbers_mt(600); +SELECT count() FROM (SELECT f + 1.0 AS k, count() FROM t_float GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +2 +SELECT count() FROM (SELECT f + 1.0 AS k, count() FROM t_float GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +2 +-- the collapse the arm above depends on: fewer distinct sums than distinct addends +SELECT uniqExact(f), uniqExact(f + 1.0) FROM t_float; +3 2 +DROP TABLE t_float; +-- a constant of a narrower date type narrows the result: every multiple of 65536 maps to +-- 1970-01-01. A NULL constant maps every value to NULL. +DROP TABLE IF EXISTS t_narrow; +CREATE TABLE t_narrow (x UInt32, v UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x; +INSERT INTO t_narrow SELECT intDiv(number, 100) * 65536 AS x, number FROM numbers_mt(600); +SELECT count() FROM (SELECT x + toDate(0) AS k, count() FROM t_narrow GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +1 +SELECT count() FROM (SELECT x + toDate(0) AS k, count() FROM t_narrow GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +1 +SELECT count() FROM (SELECT DISTINCT x + CAST(NULL, 'Nullable(UInt32)') AS k FROM t_narrow) SETTINGS force_distinct_partitions_independently = 1; +1 +SELECT count() FROM (SELECT DISTINCT x + CAST(NULL, 'Nullable(UInt32)') AS k FROM t_narrow) SETTINGS allow_distinct_partitions_independently = 0; +1 +SELECT uniqExact(x), uniqExact(x + toDate(0)) FROM t_narrow; +6 1 +DROP TABLE t_narrow; +-- a Decimal constant rescales the varying operand: every multiple of 2^32 maps to one +-- Decimal(9, 1) +DROP TABLE IF EXISTS t_decimal; +CREATE TABLE t_decimal (x UInt64, v UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x; +INSERT INTO t_decimal SELECT intDiv(number, 100) * 4294967296 AS x, number FROM numbers_mt(400); +SELECT count() FROM (SELECT x + toDecimal32(0, 1) AS k, count() FROM t_decimal GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +1 +SELECT count() FROM (SELECT x + toDecimal32(0, 1) AS k, count() FROM t_decimal GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +1 +SELECT uniqExact(x), uniqExact(x + toDecimal32(0, 1)) FROM t_decimal; +4 1 +DROP TABLE t_decimal; +-- two varying operands are not injective either: many pairs share one sum +DROP TABLE IF EXISTS t_two_cols; +CREATE TABLE t_two_cols (x UInt32, y UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x; +INSERT INTO t_two_cols SELECT intDiv(number, 100) AS x, number % 100 AS y FROM numbers_mt(1000); +SELECT count() FROM (SELECT x + y AS k, count() FROM t_two_cols GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +109 +SELECT count() FROM (SELECT x + y AS k, count() FROM t_two_cols GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +109 +DROP TABLE t_two_cols; diff --git a/tests/queries/0_stateless/04833_injective_plus_minus_type_gate.sql b/tests/queries/0_stateless/04833_injective_plus_minus_type_gate.sql new file mode 100644 index 000000000000..cf5290884f8d --- /dev/null +++ b/tests/queries/0_stateless/04833_injective_plus_minus_type_gate.sql @@ -0,0 +1,97 @@ +-- Tags: no-random-settings, no-random-merge-tree-settings +-- no-random-settings, no-random-merge-tree-settings: randomized settings and part counts +-- change both the plans and the values these arms count. + +-- max_threads is pinned because the cost heuristic accepts a fixture only when its partition +-- count is at least max_threads / 2; arms that must not depend on the heuristic force it instead. +SET max_threads = 8; +SET enable_parallel_replicas = 0; +SET max_rows_in_distinct = 0; +SET max_bytes_in_distinct = 0; +-- The stateless CI profile sets these to 10G, and a nonzero limit disables per-partition +-- evaluation outright. +SET max_rows_to_group_by = 0; +SET max_rows_to_sort = 0; +SET max_bytes_to_sort = 0; +SET optimize_use_implicit_projections = 0; +-- The values and plans below are the analyzer's, so pin it. +SET enable_analyzer = 1; + +-- { echo } + +-- --------------------------------------------------------------------------- +-- Correctness: per-partition evaluation must agree with the merged evaluation. Each arm +-- prints the forced count and then the merged count; they must be equal. The two settings +-- apply to the outer query, so they reach the aggregation under test. +-- --------------------------------------------------------------------------- + +-- Date + INTERVAL MONTH collapses the 29th, 30th and 31st of a month into one key. 2001 and +-- 2002 are not leap years, so all three days of each map to February 28. +DROP TABLE IF EXISTS t_month; +CREATE TABLE t_month (d Date, x UInt32) ENGINE = MergeTree ORDER BY d PARTITION BY d; +INSERT INTO t_month SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-01-', toString(29 + (intDiv(number, 10) % 3)))) AS d, number FROM numbers(60); +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM t_month GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM t_month GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +SELECT count() FROM (SELECT DISTINCT d + INTERVAL 1 MONTH AS k FROM t_month) SETTINGS force_distinct_partitions_independently = 1; +SELECT count() FROM (SELECT DISTINCT d + INTERVAL 1 MONTH AS k FROM t_month) SETTINGS allow_distinct_partitions_independently = 0; +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k FROM t_month LIMIT 1 BY k) SETTINGS allow_limit_by_partitions_independently = 1; +SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k FROM t_month LIMIT 1 BY k) SETTINGS allow_limit_by_partitions_independently = 0; +DROP TABLE t_month; + +-- the minus direction collapses the same way, going back into a shorter month +DROP TABLE IF EXISTS t_month_minus; +CREATE TABLE t_month_minus (d Date, x UInt32) ENGINE = MergeTree ORDER BY d PARTITION BY d; +INSERT INTO t_month_minus SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-03-', toString(29 + (intDiv(number, 10) % 3)))) AS d, number FROM numbers(60); +SELECT count() FROM (SELECT d - INTERVAL 1 MONTH AS k, count() FROM t_month_minus GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT d - INTERVAL 1 MONTH AS k, count() FROM t_month_minus GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +DROP TABLE t_month_minus; + +-- DateTime + INTERVAL DAY collapses across a DST spring-forward, so the interval kind alone +-- does not decide safety +DROP TABLE IF EXISTS t_dst; +CREATE TABLE t_dst (t DateTime('Europe/Moscow'), x UInt32) ENGINE = MergeTree ORDER BY t PARTITION BY t; +INSERT INTO t_dst SELECT toDateTime('2010-03-27 00:00:00', 'Europe/Moscow') + (intDiv(number, 100) * 1800) AS t, number FROM numbers_mt(600); +SELECT count() FROM (SELECT t + INTERVAL 1 DAY AS k, count() FROM t_dst GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT t + INTERVAL 1 DAY AS k, count() FROM t_dst GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +DROP TABLE t_dst; + +-- a Float64 addend collapses by mantissa rounding; a float column reaches the optimizer +-- through a non-float partition key +DROP TABLE IF EXISTS t_float; +CREATE TABLE t_float (f Float64, x UInt32) ENGINE = MergeTree ORDER BY f PARTITION BY toUInt64(f - 1e16); +INSERT INTO t_float SELECT 1e16 + intDiv(number, 100) AS f, number FROM numbers_mt(600); +SELECT count() FROM (SELECT f + 1.0 AS k, count() FROM t_float GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT f + 1.0 AS k, count() FROM t_float GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +-- the collapse the arm above depends on: fewer distinct sums than distinct addends +SELECT uniqExact(f), uniqExact(f + 1.0) FROM t_float; +DROP TABLE t_float; + +-- a constant of a narrower date type narrows the result: every multiple of 65536 maps to +-- 1970-01-01. A NULL constant maps every value to NULL. +DROP TABLE IF EXISTS t_narrow; +CREATE TABLE t_narrow (x UInt32, v UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x; +INSERT INTO t_narrow SELECT intDiv(number, 100) * 65536 AS x, number FROM numbers_mt(600); +SELECT count() FROM (SELECT x + toDate(0) AS k, count() FROM t_narrow GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT x + toDate(0) AS k, count() FROM t_narrow GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +SELECT count() FROM (SELECT DISTINCT x + CAST(NULL, 'Nullable(UInt32)') AS k FROM t_narrow) SETTINGS force_distinct_partitions_independently = 1; +SELECT count() FROM (SELECT DISTINCT x + CAST(NULL, 'Nullable(UInt32)') AS k FROM t_narrow) SETTINGS allow_distinct_partitions_independently = 0; +SELECT uniqExact(x), uniqExact(x + toDate(0)) FROM t_narrow; +DROP TABLE t_narrow; + +-- a Decimal constant rescales the varying operand: every multiple of 2^32 maps to one +-- Decimal(9, 1) +DROP TABLE IF EXISTS t_decimal; +CREATE TABLE t_decimal (x UInt64, v UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x; +INSERT INTO t_decimal SELECT intDiv(number, 100) * 4294967296 AS x, number FROM numbers_mt(400); +SELECT count() FROM (SELECT x + toDecimal32(0, 1) AS k, count() FROM t_decimal GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT x + toDecimal32(0, 1) AS k, count() FROM t_decimal GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +SELECT uniqExact(x), uniqExact(x + toDecimal32(0, 1)) FROM t_decimal; +DROP TABLE t_decimal; + +-- two varying operands are not injective either: many pairs share one sum +DROP TABLE IF EXISTS t_two_cols; +CREATE TABLE t_two_cols (x UInt32, y UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x; +INSERT INTO t_two_cols SELECT intDiv(number, 100) AS x, number % 100 AS y FROM numbers_mt(1000); +SELECT count() FROM (SELECT x + y AS k, count() FROM t_two_cols GROUP BY k) SETTINGS force_aggregate_partitions_independently = 1; +SELECT count() FROM (SELECT x + y AS k, count() FROM t_two_cols GROUP BY k) SETTINGS allow_aggregate_partitions_independently = 0; +DROP TABLE t_two_cols; diff --git a/tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.reference b/tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.reference new file mode 100644 index 000000000000..d93a3042cfa8 --- /dev/null +++ b/tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.reference @@ -0,0 +1,97 @@ +-- { echo } + +-- per-partition set building reads the same predicate. The set fill deduplicates across +-- partitions anyway, so the merged answer stays correct and only the plan shape shows the +-- decline; the bare-key arm is the control that the fixture reaches the optimization. +-- 2001 and 2002 are not leap years, so all three days of each map to February 28. +DROP TABLE IF EXISTS t_set_month; +CREATE TABLE t_set_month (d Date, x UInt32) ENGINE = MergeTree ORDER BY d PARTITION BY d; +SYSTEM STOP MERGES t_set_month; +INSERT INTO t_set_month SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-01-', toString(29 + (intDiv(number, 10) % 3)))) AS d, number FROM numbers(60); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT count() FROM numbers(100) WHERE toDate('2001-02-28') + number IN (SELECT d + INTERVAL 1 MONTH FROM t_set_month) SETTINGS allow_creating_set_partitions_independently = 1) WHERE explain LIKE '%Pre-distinct%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT count() FROM numbers(100) WHERE toDate('2001-02-28') + number IN (SELECT d FROM t_set_month) SETTINGS allow_creating_set_partitions_independently = 1) WHERE explain LIKE '%Pre-distinct%' OR explain LIKE '%Read each partition through separate port%'; +Pre-distinct: 1 +Read each partition through separate port: 1 +DROP TABLE t_set_month; +-- an integer key keeps per-partition set building; the partition key is a function of the set's +-- own output column, which the interval arm above cannot use because its key is the collapsing one +DROP TABLE IF EXISTS t_set_int; +CREATE TABLE t_set_int (a UInt32, b UInt32) ENGINE = MergeTree ORDER BY tuple() PARTITION BY a % 8; +SYSTEM STOP MERGES t_set_int; +INSERT INTO t_set_int SELECT number % 64, number FROM numbers_mt(400); +INSERT INTO t_set_int SELECT number % 64, number FROM numbers_mt(400); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT count() FROM numbers(100) WHERE number IN (SELECT a + 1 FROM t_set_int) SETTINGS allow_creating_set_partitions_independently = 1) WHERE explain LIKE '%Pre-distinct%' OR explain LIKE '%Read each partition through separate port%'; +Pre-distinct: 1 +Read each partition through separate port: 1 +SELECT (SELECT count() FROM numbers(100) WHERE number IN (SELECT a + 1 FROM t_set_int) SETTINGS allow_creating_set_partitions_independently = 0) = (SELECT count() FROM numbers(100) WHERE number IN (SELECT a + 1 FROM t_set_int) SETTINGS allow_creating_set_partitions_independently = 1); +1 +DROP TABLE t_set_int; +-- --------------------------------------------------------------------------- +-- The distributed sharding-key consumer reaches the same predicate through its own rejection +-- loop and its own direct call, so it gets its own arms. Dropping the merge step is only +-- correct when the group key determines the shard: a key that collapses distinct shard-key +-- values leaves each shard's partial groups unmerged, so the same key is returned twice. +-- Each view filters itself by shardNum() so the two shards hold the disjoint rows the +-- declared key implies - a declared key alone does not redistribute rows on a read, and +-- without the filter every shard holds every row and even a sound merge drop doubles the +-- answer. The first arm of each pair counts merge steps (1 = kept, 0 = dropped) and the +-- second compares the answer against the unoptimized one; the integer pair is the control +-- that the optimization still fires where it is sound. +-- --------------------------------------------------------------------------- + +SELECT shardNum() AS s, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY s ORDER BY s; +1 10 +2 20 +SELECT count() FROM (EXPLAIN SELECT d + INTERVAL 1 MONTH AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1) WHERE explain ILIKE '%MergingAggregated%'; +1 +SELECT (SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 0)) = (SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1)); +1 +SELECT shardNum() AS s, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY s ORDER BY s; +1 15 +2 15 +SELECT count() FROM (EXPLAIN SELECT x + 1 AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1) WHERE explain ILIKE '%MergingAggregated%'; +0 +SELECT (SELECT count() FROM (SELECT x + 1 AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 0)) = (SELECT count() FROM (SELECT x + 1 AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1)); +1 +-- --------------------------------------------------------------------------- +-- The window consumer reaches the same predicate through the stream-disjointness +-- propagation, at two sites: the per-partition read request and the scatter skip above it. +-- INTERVAL MONTH collapses the 29th, 30th and 31st into one key, so one logical window +-- partition spans the table partitions those days live in and must not be evaluated per +-- table partition. The default arm carries no setting: the cost heuristic accepts this +-- fixture, so the answer has to be right without opting out. +-- --------------------------------------------------------------------------- + +DROP TABLE IF EXISTS t_win_month; +CREATE TABLE t_win_month (d Date) ENGINE = MergeTree ORDER BY d PARTITION BY d; +INSERT INTO t_win_month SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-01-', toString(29 + (intDiv(number, 10) % 3)))) AS d FROM numbers(60); +-- the collapse the arms below depend on: more table partitions than window keys +SELECT uniqExact(_partition_id), uniqExact(d + INTERVAL 1 MONTH) FROM t_win_month; +6 2 +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) AS c FROM t_win_month) ORDER BY c SETTINGS force_window_partitions_independently = 1; +30 +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) AS c FROM t_win_month) ORDER BY c SETTINGS allow_window_partitions_independently = 0; +30 +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) AS c FROM t_win_month) ORDER BY c; +30 +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) FROM t_win_month SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Read each partition through separate port: 1%'; +0 +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) FROM t_win_month SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Skip scatter by partition: 1%'; +0 +DROP TABLE t_win_month; +-- the integer control: an injective addend keeps both sites firing, so the arms above +-- attribute to the operand type and not to the window shape or the fixture +DROP TABLE IF EXISTS t_win_int; +CREATE TABLE t_win_int (x UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x % 8; +INSERT INTO t_win_int SELECT number % 8 FROM numbers_mt(800); +SELECT uniqExact(_partition_id), uniqExact(x + 1) FROM t_win_int; +8 8 +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY x + 1) AS c FROM t_win_int) ORDER BY c SETTINGS force_window_partitions_independently = 1; +100 +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY x + 1) AS c FROM t_win_int) ORDER BY c SETTINGS allow_window_partitions_independently = 0; +100 +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY x + 1) FROM t_win_int SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Read each partition through separate port: 1%'; +1 +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY x + 1) FROM t_win_int SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Skip scatter by partition: 1%'; +1 +DROP TABLE t_win_int; diff --git a/tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.sql b/tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.sql new file mode 100644 index 000000000000..509e8a3e3d5e --- /dev/null +++ b/tests/queries/0_stateless/05053_injective_plus_minus_type_gate_sets_shards_windows.sql @@ -0,0 +1,97 @@ +-- Tags: distributed, no-random-settings, no-random-merge-tree-settings +-- no-random-settings, no-random-merge-tree-settings: randomized settings and part counts +-- change both the plans and the values these arms count. + +SET explain_query_plan_default = 'legacy'; +-- max_threads is pinned because the cost heuristic accepts a fixture only when its partition +-- count is at least max_threads / 2; arms that must not depend on the heuristic force it instead. +SET max_threads = 8; +SET enable_parallel_replicas = 0; +SET max_rows_in_distinct = 0; +SET max_bytes_in_distinct = 0; +-- The stateless CI profile sets these to 10G, and a nonzero limit disables per-partition +-- evaluation outright. +SET max_rows_to_group_by = 0; +SET max_rows_to_sort = 0; +SET max_bytes_to_sort = 0; +SET optimize_use_implicit_projections = 0; +-- The values and plans below are the analyzer's, so pin it. +SET enable_analyzer = 1; + +-- { echo } + +-- per-partition set building reads the same predicate. The set fill deduplicates across +-- partitions anyway, so the merged answer stays correct and only the plan shape shows the +-- decline; the bare-key arm is the control that the fixture reaches the optimization. +-- 2001 and 2002 are not leap years, so all three days of each map to February 28. +DROP TABLE IF EXISTS t_set_month; +CREATE TABLE t_set_month (d Date, x UInt32) ENGINE = MergeTree ORDER BY d PARTITION BY d; +SYSTEM STOP MERGES t_set_month; +INSERT INTO t_set_month SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-01-', toString(29 + (intDiv(number, 10) % 3)))) AS d, number FROM numbers(60); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT count() FROM numbers(100) WHERE toDate('2001-02-28') + number IN (SELECT d + INTERVAL 1 MONTH FROM t_set_month) SETTINGS allow_creating_set_partitions_independently = 1) WHERE explain LIKE '%Pre-distinct%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT count() FROM numbers(100) WHERE toDate('2001-02-28') + number IN (SELECT d FROM t_set_month) SETTINGS allow_creating_set_partitions_independently = 1) WHERE explain LIKE '%Pre-distinct%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_set_month; + +-- an integer key keeps per-partition set building; the partition key is a function of the set's +-- own output column, which the interval arm above cannot use because its key is the collapsing one +DROP TABLE IF EXISTS t_set_int; +CREATE TABLE t_set_int (a UInt32, b UInt32) ENGINE = MergeTree ORDER BY tuple() PARTITION BY a % 8; +SYSTEM STOP MERGES t_set_int; +INSERT INTO t_set_int SELECT number % 64, number FROM numbers_mt(400); +INSERT INTO t_set_int SELECT number % 64, number FROM numbers_mt(400); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT count() FROM numbers(100) WHERE number IN (SELECT a + 1 FROM t_set_int) SETTINGS allow_creating_set_partitions_independently = 1) WHERE explain LIKE '%Pre-distinct%' OR explain LIKE '%Read each partition through separate port%'; +SELECT (SELECT count() FROM numbers(100) WHERE number IN (SELECT a + 1 FROM t_set_int) SETTINGS allow_creating_set_partitions_independently = 0) = (SELECT count() FROM numbers(100) WHERE number IN (SELECT a + 1 FROM t_set_int) SETTINGS allow_creating_set_partitions_independently = 1); +DROP TABLE t_set_int; + +-- --------------------------------------------------------------------------- +-- The distributed sharding-key consumer reaches the same predicate through its own rejection +-- loop and its own direct call, so it gets its own arms. Dropping the merge step is only +-- correct when the group key determines the shard: a key that collapses distinct shard-key +-- values leaves each shard's partial groups unmerged, so the same key is returned twice. +-- Each view filters itself by shardNum() so the two shards hold the disjoint rows the +-- declared key implies - a declared key alone does not redistribute rows on a read, and +-- without the filter every shard holds every row and even a sound merge drop doubles the +-- answer. The first arm of each pair counts merge steps (1 = kept, 0 = dropped) and the +-- second compares the answer against the unoptimized one; the integer pair is the control +-- that the optimization still fires where it is sound. +-- --------------------------------------------------------------------------- + +SELECT shardNum() AS s, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY s ORDER BY s; +SELECT count() FROM (EXPLAIN SELECT d + INTERVAL 1 MONTH AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1) WHERE explain ILIKE '%MergingAggregated%'; +SELECT (SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 0)) = (SELECT count() FROM (SELECT d + INTERVAL 1 MONTH AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE toYYYYMMDD(toDate('2001-01-29') + (number % 3)) % 2 = (shardNum() - 1)), toUInt64(toYYYYMMDD(d))) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1)); +SELECT shardNum() AS s, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY s ORDER BY s; +SELECT count() FROM (EXPLAIN SELECT x + 1 AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1) WHERE explain ILIKE '%MergingAggregated%'; +SELECT (SELECT count() FROM (SELECT x + 1 AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 0)) = (SELECT count() FROM (SELECT x + 1 AS k, count() FROM remote('127.{1,2}', view(SELECT toDate('2001-01-29') + (number % 3) AS d, number AS x FROM numbers(30) WHERE number % 2 = (shardNum() - 1)), toUInt64(x)) GROUP BY k SETTINGS optimize_skip_unused_shards = 1, optimize_distributed_group_by_sharding_key = 1)); + +-- --------------------------------------------------------------------------- +-- The window consumer reaches the same predicate through the stream-disjointness +-- propagation, at two sites: the per-partition read request and the scatter skip above it. +-- INTERVAL MONTH collapses the 29th, 30th and 31st into one key, so one logical window +-- partition spans the table partitions those days live in and must not be evaluated per +-- table partition. The default arm carries no setting: the cost heuristic accepts this +-- fixture, so the answer has to be right without opting out. +-- --------------------------------------------------------------------------- + +DROP TABLE IF EXISTS t_win_month; +CREATE TABLE t_win_month (d Date) ENGINE = MergeTree ORDER BY d PARTITION BY d; +INSERT INTO t_win_month SELECT toDate(concat(toString(2001 + intDiv(number, 30)), '-01-', toString(29 + (intDiv(number, 10) % 3)))) AS d FROM numbers(60); +-- the collapse the arms below depend on: more table partitions than window keys +SELECT uniqExact(_partition_id), uniqExact(d + INTERVAL 1 MONTH) FROM t_win_month; +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) AS c FROM t_win_month) ORDER BY c SETTINGS force_window_partitions_independently = 1; +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) AS c FROM t_win_month) ORDER BY c SETTINGS allow_window_partitions_independently = 0; +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) AS c FROM t_win_month) ORDER BY c; +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) FROM t_win_month SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Read each partition through separate port: 1%'; +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY d + INTERVAL 1 MONTH) FROM t_win_month SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Skip scatter by partition: 1%'; +DROP TABLE t_win_month; + +-- the integer control: an injective addend keeps both sites firing, so the arms above +-- attribute to the operand type and not to the window shape or the fixture +DROP TABLE IF EXISTS t_win_int; +CREATE TABLE t_win_int (x UInt32) ENGINE = MergeTree ORDER BY x PARTITION BY x % 8; +INSERT INTO t_win_int SELECT number % 8 FROM numbers_mt(800); +SELECT uniqExact(_partition_id), uniqExact(x + 1) FROM t_win_int; +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY x + 1) AS c FROM t_win_int) ORDER BY c SETTINGS force_window_partitions_independently = 1; +SELECT DISTINCT c FROM (SELECT count() OVER (PARTITION BY x + 1) AS c FROM t_win_int) ORDER BY c SETTINGS allow_window_partitions_independently = 0; +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY x + 1) FROM t_win_int SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Read each partition through separate port: 1%'; +SELECT count() > 0 FROM (EXPLAIN actions = 1 SELECT count() OVER (PARTITION BY x + 1) FROM t_win_int SETTINGS force_window_partitions_independently = 1) WHERE explain ILIKE '%Skip scatter by partition: 1%'; +DROP TABLE t_win_int; diff --git a/tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.reference b/tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.reference new file mode 100644 index 000000000000..d695e73641ae --- /dev/null +++ b/tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.reference @@ -0,0 +1,98 @@ +-- { echo } + +-- --------------------------------------------------------------------------- +-- Preservation: integer arithmetic with an integer constant stays injective, so the +-- optimization must keep firing. A correctness-only test would not notice this. Each arm +-- prints two plan lines; an arm printing nothing has lost the optimization. +-- --------------------------------------------------------------------------- + +DROP TABLE IF EXISTS t_int; +CREATE TABLE t_int (a UInt32) ENGINE = MergeTree ORDER BY a PARTITION BY intDiv(a, 2) * 2 + 1; +INSERT INTO t_int SELECT number FROM numbers_mt(32); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT intDiv(a, 2) + 1 AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +-- subtraction widens to a signed result, which is a different branch of the integer test +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT intDiv(a, 2) - 1 AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +-- the constant may be either operand, so both positions have to be recognized +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT 7 - intDiv(a, 2) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT 1 + intDiv(a, 2) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +-- negate and tuple are not named by the query-plan consumer: they answer isInjective themselves, so +-- the optimization has to keep firing for them. +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT -intDiv(a, 2) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT tuple(intDiv(a, 2)) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_int; +-- wide integers are exact, so the gate must not be narrowed to native widths +DROP TABLE IF EXISTS t_int128; +CREATE TABLE t_int128 (a Int128) ENGINE = MergeTree ORDER BY a PARTITION BY a % 8; +INSERT INTO t_int128 SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT a + 7 AS a1 FROM t_int128 SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_int128; +DROP TABLE IF EXISTS t_uint256; +CREATE TABLE t_uint256 (b UInt256) ENGINE = MergeTree ORDER BY b PARTITION BY b % 8; +INSERT INTO t_uint256 SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT b + 7 AS b1 FROM t_uint256 SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_uint256; +-- both type-wrapper orders must be stripped before the integer test. Contrast with the NULL +-- constant arm above: a Nullable column keeps the optimization, a NULL constant loses it. +DROP TABLE IF EXISTS t_nullable; +CREATE TABLE t_nullable (n Nullable(UInt32)) ENGINE = MergeTree ORDER BY n PARTITION BY n % 8 SETTINGS allow_nullable_key = 1; +INSERT INTO t_nullable SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT n + 1 AS n1 FROM t_nullable SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_nullable; +DROP TABLE IF EXISTS t_lc; +CREATE TABLE t_lc (l LowCardinality(UInt32)) ENGINE = MergeTree ORDER BY l PARTITION BY l % 8; +INSERT INTO t_lc SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT l + 1 AS l1 FROM t_lc SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_lc; +DROP TABLE IF EXISTS t_lc_nullable; +CREATE TABLE t_lc_nullable (ln LowCardinality(Nullable(UInt32))) ENGINE = MergeTree ORDER BY ln PARTITION BY ln % 8 SETTINGS allow_nullable_key = 1; +INSERT INTO t_lc_nullable SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT ln + 1 AS ln1 FROM t_lc_nullable SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_lc_nullable; +-- --------------------------------------------------------------------------- +-- Accepted narrowings. These three expressions are injective, but every type they would +-- admit also admits a non-injective case above that cannot be told apart: an Interval on +-- Date is not separable from one on DateTime by kind, a Date operand re-admits the +-- narrowing constant, and a Decimal operand re-admits the rescaling constant. Each arm +-- prints nothing; the bare-key arm after it is the control showing the fixture does reach +-- the optimization. +-- --------------------------------------------------------------------------- + +DROP TABLE IF EXISTS t_date; +CREATE TABLE t_date (d Date) ENGINE = MergeTree ORDER BY d PARTITION BY toDayOfWeek(d); +INSERT INTO t_date SELECT toDate('2001-01-01') + number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT d + INTERVAL 1 DAY AS d1 FROM t_date SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT d + 1 AS d1 FROM t_date SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT toDayOfWeek(d) AS d1 FROM t_date SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_date; +DROP TABLE IF EXISTS t_dec; +CREATE TABLE t_dec (dec Decimal64(2)) ENGINE = MergeTree ORDER BY dec PARTITION BY toUInt32(dec) % 8; +INSERT INTO t_dec SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT dec + 1 AS dec1 FROM t_dec SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT toUInt32(dec) AS dec1 FROM t_dec SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +Skip stream merging: 1 +Read each partition through separate port: 1 +DROP TABLE t_dec; diff --git a/tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.sql b/tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.sql new file mode 100644 index 000000000000..cb93bc55fbcc --- /dev/null +++ b/tests/queries/0_stateless/05054_injective_plus_minus_type_gate_plan_shape.sql @@ -0,0 +1,100 @@ +-- Tags: no-random-settings, no-random-merge-tree-settings +-- no-random-settings, no-random-merge-tree-settings: randomized settings and part counts +-- change both the plans and the values these arms count. + +SET explain_query_plan_default = 'legacy'; +-- max_threads is pinned because the cost heuristic accepts a fixture only when its partition +-- count is at least max_threads / 2; arms that must not depend on the heuristic force it instead. +SET max_threads = 8; +SET enable_parallel_replicas = 0; +SET max_rows_in_distinct = 0; +SET max_bytes_in_distinct = 0; +-- The stateless CI profile sets these to 10G, and a nonzero limit disables per-partition +-- evaluation outright. +SET max_rows_to_group_by = 0; +SET max_rows_to_sort = 0; +SET max_bytes_to_sort = 0; +SET optimize_use_implicit_projections = 0; +SET allow_suspicious_low_cardinality_types = 1; +-- The EXPLAIN QUERY TREE arms below are analyzer-only; old-analyzer jobs would error on them. +SET enable_analyzer = 1; + +-- { echo } + +-- --------------------------------------------------------------------------- +-- Preservation: integer arithmetic with an integer constant stays injective, so the +-- optimization must keep firing. A correctness-only test would not notice this. Each arm +-- prints two plan lines; an arm printing nothing has lost the optimization. +-- --------------------------------------------------------------------------- + +DROP TABLE IF EXISTS t_int; +CREATE TABLE t_int (a UInt32) ENGINE = MergeTree ORDER BY a PARTITION BY intDiv(a, 2) * 2 + 1; +INSERT INTO t_int SELECT number FROM numbers_mt(32); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT intDiv(a, 2) + 1 AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +-- subtraction widens to a signed result, which is a different branch of the integer test +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT intDiv(a, 2) - 1 AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +-- the constant may be either operand, so both positions have to be recognized +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT 7 - intDiv(a, 2) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT 1 + intDiv(a, 2) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +-- negate and tuple are not named by the query-plan consumer: they answer isInjective themselves, so +-- the optimization has to keep firing for them. +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT -intDiv(a, 2) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT tuple(intDiv(a, 2)) AS a1 FROM t_int SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_int; + +-- wide integers are exact, so the gate must not be narrowed to native widths +DROP TABLE IF EXISTS t_int128; +CREATE TABLE t_int128 (a Int128) ENGINE = MergeTree ORDER BY a PARTITION BY a % 8; +INSERT INTO t_int128 SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT a + 7 AS a1 FROM t_int128 SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_int128; + +DROP TABLE IF EXISTS t_uint256; +CREATE TABLE t_uint256 (b UInt256) ENGINE = MergeTree ORDER BY b PARTITION BY b % 8; +INSERT INTO t_uint256 SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT b + 7 AS b1 FROM t_uint256 SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_uint256; + +-- both type-wrapper orders must be stripped before the integer test. Contrast with the NULL +-- constant arm above: a Nullable column keeps the optimization, a NULL constant loses it. +DROP TABLE IF EXISTS t_nullable; +CREATE TABLE t_nullable (n Nullable(UInt32)) ENGINE = MergeTree ORDER BY n PARTITION BY n % 8 SETTINGS allow_nullable_key = 1; +INSERT INTO t_nullable SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT n + 1 AS n1 FROM t_nullable SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_nullable; + +DROP TABLE IF EXISTS t_lc; +CREATE TABLE t_lc (l LowCardinality(UInt32)) ENGINE = MergeTree ORDER BY l PARTITION BY l % 8; +INSERT INTO t_lc SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT l + 1 AS l1 FROM t_lc SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_lc; + +DROP TABLE IF EXISTS t_lc_nullable; +CREATE TABLE t_lc_nullable (ln LowCardinality(Nullable(UInt32))) ENGINE = MergeTree ORDER BY ln PARTITION BY ln % 8 SETTINGS allow_nullable_key = 1; +INSERT INTO t_lc_nullable SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT ln + 1 AS ln1 FROM t_lc_nullable SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_lc_nullable; + +-- --------------------------------------------------------------------------- +-- Accepted narrowings. These three expressions are injective, but every type they would +-- admit also admits a non-injective case above that cannot be told apart: an Interval on +-- Date is not separable from one on DateTime by kind, a Date operand re-admits the +-- narrowing constant, and a Decimal operand re-admits the rescaling constant. Each arm +-- prints nothing; the bare-key arm after it is the control showing the fixture does reach +-- the optimization. +-- --------------------------------------------------------------------------- + +DROP TABLE IF EXISTS t_date; +CREATE TABLE t_date (d Date) ENGINE = MergeTree ORDER BY d PARTITION BY toDayOfWeek(d); +INSERT INTO t_date SELECT toDate('2001-01-01') + number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT d + INTERVAL 1 DAY AS d1 FROM t_date SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT d + 1 AS d1 FROM t_date SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT toDayOfWeek(d) AS d1 FROM t_date SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_date; + +DROP TABLE IF EXISTS t_dec; +CREATE TABLE t_dec (dec Decimal64(2)) ENGINE = MergeTree ORDER BY dec PARTITION BY toUInt32(dec) % 8; +INSERT INTO t_dec SELECT number FROM numbers_mt(200); +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT dec + 1 AS dec1 FROM t_dec SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +SELECT replaceRegexpOne(explain, '^[ ]*(.*)', '\\1') FROM (EXPLAIN actions = 1 SELECT DISTINCT toUInt32(dec) AS dec1 FROM t_dec SETTINGS allow_distinct_partitions_independently = 1) WHERE explain LIKE '%Skip stream merging%' OR explain LIKE '%Read each partition through separate port%'; +DROP TABLE t_dec; From bb7091ad5ed09ac6151489008e575f43b3a80960 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 06:27:27 +0000 Subject: [PATCH 005/185] Backport #120972 to 26.8: MySQL dictionary source: accept `enable_local_infile` only in the configuration file --- src/Dictionaries/MySQLDictionarySource.cpp | 27 ++++++ .../dictionaries/mysql_dict_local_infile.xml | 30 +++++++ .../test_dictionaries_mysql/test.py | 72 +++++++++++++++ ...l_dictionary_enable_local_infile.reference | 12 +++ ...29_mysql_dictionary_enable_local_infile.sh | 90 +++++++++++++++++++ 5 files changed, 231 insertions(+) create mode 100644 tests/integration/test_dictionaries_mysql/configs/dictionaries/mysql_dict_local_infile.xml create mode 100644 tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.reference create mode 100755 tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.sh diff --git a/src/Dictionaries/MySQLDictionarySource.cpp b/src/Dictionaries/MySQLDictionarySource.cpp index 62965ab14b05..e18fdb935dee 100644 --- a/src/Dictionaries/MySQLDictionarySource.cpp +++ b/src/Dictionaries/MySQLDictionarySource.cpp @@ -88,6 +88,26 @@ static void checkNoSSLPaths(const Poco::Util::AbstractConfiguration & config, co key, contents_key); } } + +/// `enable_local_infile` sets `MYSQL_OPT_LOCAL_INFILE`, which lets the MySQL endpoint ask the client +/// for the contents of a file of its choosing, read with the server's own privileges: the option is +/// off by default because it is insecure (`mysqlxx/Connection.h`). Same reasoning as above. +/// `fallback_prefix` is the parent prefix a `` inherits the value from, resolved in the same +/// order as `Pool::Pool`, so what is checked is the value the connection will actually use. +static void checkNoLocalInfile( + const Poco::Util::AbstractConfiguration & config, + const std::string & prefix, + const std::string & fallback_prefix = {}) +{ + const bool inherited + = !fallback_prefix.empty() && config.getBool(fallback_prefix + ".enable_local_infile", false); + + if (config.getBool(prefix + ".enable_local_infile", inherited)) + throw Exception( + ErrorCodes::BAD_ARGUMENTS, + "`enable_local_infile` cannot be enabled in a dictionary created with a DDL query. " + "It is only accepted in a dictionary defined in a server configuration file"); +} #endif void registerDictionarySourceMysql(DictionarySourceFactory & factory); @@ -197,6 +217,7 @@ void registerDictionarySourceMysql(DictionarySourceFactory & factory) { const auto replica_prefix = settings_config_prefix + "." + replica_key; checkNoSSLPaths(config, replica_prefix); + checkNoLocalInfile(config, replica_prefix, settings_config_prefix); global_context->getRemoteHostFilter().checkHostAndPort( config.getString(replica_prefix + ".host"), toString(config.getInt(replica_prefix + ".port", 3306))); @@ -205,6 +226,7 @@ void registerDictionarySourceMysql(DictionarySourceFactory & factory) } else { + checkNoLocalInfile(config, settings_config_prefix); global_context->getRemoteHostFilter().checkHostAndPort( config.getString(settings_config_prefix + ".host"), toString(config.getInt(settings_config_prefix + ".port", 3306))); @@ -300,6 +322,7 @@ Setting fields: | `fail_on_connection_loss` | Controls behavior of the server on connection loss. If `true`, an exception is thrown immediately if the connection between client and server was lost. If `false`, the server retries to fetch data at least three times before reporting an error. Note that retrying leads to increased response times. Default value: `false`. | | `query` | The custom query. Optional. | | `enable_compression` | Enables zlib compression for the MySQL protocol connection. When set to `1`, ClickHouse requests protocol-level compression from the MySQL server. Can also be set per-replica inside ``. Default value: `0`. | +| `enable_local_infile` | Allows the MySQL server to ask ClickHouse for the contents of a local file (`LOAD DATA LOCAL INFILE`). Only a dictionary defined in a server configuration file may enable it; a `CREATE DICTIONARY` query may leave it unset or set it to `0`. Can also be set per-replica inside ``, and a `` that does not set it inherits the value above. Default value: `0`. | | `ssl_ca_pem` | Contents of the CA certificate that the MySQL server certificate is verified against. Optional. | | `ssl_cert_pem` | Contents of the client certificate, for certificate-based authentication. Optional. | | `ssl_key_pem` | Contents of the private key belonging to `ssl_cert_pem`. Optional. | @@ -313,6 +336,10 @@ The `table` or `where` fields cannot be used together with the `query` field. An `ssl_ca`, `ssl_cert` and `ssl_key` name files that the server opens with its own privileges, so they are only accepted for a dictionary defined in a server configuration file, or through a named collection defined there. A `CREATE DICTIONARY` query that specifies the TLS credentials directly must pass their contents instead, in `ssl_ca_pem`, `ssl_cert_pem` and `ssl_key_pem`. Those values are masked in logs and in `SHOW` queries, the same way passwords are. + +`enable_local_infile` lets the MySQL server ask ClickHouse for the contents of a file of its choosing, which the server reads with its own privileges, so it is only accepted for a dictionary defined in a server configuration file. A `CREATE DICTIONARY` query that enables it, at the source or at one of its `` entries, is rejected with `BAD_ARGUMENTS`. + + There is no explicit parameter `secure`. When establishing an SSL-connection security is mandatory. diff --git a/tests/integration/test_dictionaries_mysql/configs/dictionaries/mysql_dict_local_infile.xml b/tests/integration/test_dictionaries_mysql/configs/dictionaries/mysql_dict_local_infile.xml new file mode 100644 index 000000000000..9d5a3fe658f3 --- /dev/null +++ b/tests/integration/test_dictionaries_mysql/configs/dictionaries/mysql_dict_local_infile.xml @@ -0,0 +1,30 @@ + + + dict_local_infile + + + test + mysql80 + 3306 + root + ClickHouse_MySQL_P@ssw0rd + dict_local_infile_table
+ 1 +
+ + + + + + + id + + + value + String + + + + 0 +
+
diff --git a/tests/integration/test_dictionaries_mysql/test.py b/tests/integration/test_dictionaries_mysql/test.py index d2bd9de40401..b7be44a3ef57 100644 --- a/tests/integration/test_dictionaries_mysql/test.py +++ b/tests/integration/test_dictionaries_mysql/test.py @@ -19,6 +19,7 @@ "configs/dictionaries/mysql_dict_compression.xml", "configs/dictionaries/mysql_dict_compression_wire.xml", "configs/dictionaries/mysql_dict_no_compression_wire.xml", + "configs/dictionaries/mysql_dict_local_infile.xml", ] CONFIG_FILES = [ "configs/remote_servers.xml", @@ -1032,3 +1033,74 @@ def reload(name): mysql_connection, "DROP TABLE IF EXISTS test.inherit_pool_test;" ) mysql_connection.close() + + +def test_enable_local_infile_xml_dict(started_cluster): + """`enable_local_infile` is rejected in a dictionary created with a DDL query, but a dictionary + defined in a server configuration file is written by an operator and keeps working.""" + mysql_connection = get_mysql_conn(started_cluster) + + try: + execute_mysql_query( + mysql_connection, "DROP TABLE IF EXISTS test.dict_local_infile_table;" + ) + execute_mysql_query( + mysql_connection, + "CREATE TABLE test.dict_local_infile_table (id INT NOT NULL, value TEXT, PRIMARY KEY(id));", + ) + execute_mysql_query( + mysql_connection, + "INSERT INTO test.dict_local_infile_table VALUES (1, 'local_infile');", + ) + + # The dictionary is lazily loaded, so nothing has touched the source yet; reload it now that + # its table exists. + for _ in range(10): + try: + instance.query("SYSTEM RELOAD DICTIONARY dict_local_infile") + break + except Exception: + time.sleep(0.5) + + # Loading the dictionary means the source was instantiated and connected with the option on, + # which is what the guard must not prevent for this route. + value = instance.query( + "SELECT dictGet('dict_local_infile', 'value', toUInt64(1))" + ).strip() + last_exception = instance.query( + "SELECT last_exception FROM system.dictionaries WHERE name = 'dict_local_infile'" + ).strip() + assert value == "local_infile", ( + " was rejected in the XML dict config: " + f"{last_exception!r}" + ) + + # The same option from a DDL query is rejected, at the source and at a replica alike. The + # endpoint here is the working MySQL server, so a rejection cannot be a connection failure + # in disguise. + credentials = f"USER 'root' PASSWORD '{mysql_pass}' DB 'test' TABLE 'dict_local_infile_table'" + for source in ( + f"HOST 'mysql80' PORT 3306 {credentials} ENABLE_LOCAL_INFILE 1", + f"{credentials} REPLICA(PRIORITY 1 HOST 'mysql80' PORT 3306 ENABLE_LOCAL_INFILE 1)", + ): + instance.query("DROP DICTIONARY IF EXISTS dict_local_infile_ddl") + instance.query( + f""" + CREATE DICTIONARY dict_local_infile_ddl (id UInt64, value String) + PRIMARY KEY id + SOURCE(MYSQL({source})) + LAYOUT(FLAT()) + LIFETIME(0) + """ + ) + with pytest.raises(Exception) as exc: + instance.query("SYSTEM RELOAD DICTIONARY dict_local_infile_ddl") + assert "cannot be enabled in a dictionary created with a DDL query" in str( + exc.value + ), f"Unexpected error for {source!r}: {exc.value}" + finally: + instance.query("DROP DICTIONARY IF EXISTS dict_local_infile_ddl") + execute_mysql_query( + mysql_connection, "DROP TABLE IF EXISTS test.dict_local_infile_table;" + ) + mysql_connection.close() diff --git a/tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.reference b/tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.reference new file mode 100644 index 000000000000..b2e6ae32e037 --- /dev/null +++ b/tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.reference @@ -0,0 +1,12 @@ +--- enabled at the source +OK +--- enabled at a replica +OK +--- enabled at a later replica, disabled at the first +OK +--- inherited by a replica from the source +OK +--- disabled at the source +OK +--- disabled at every replica, enabled at the source +OK diff --git a/tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.sh b/tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.sh new file mode 100755 index 000000000000..38f1d9982888 --- /dev/null +++ b/tests/queries/0_stateless/05229_mysql_dictionary_enable_local_infile.sh @@ -0,0 +1,90 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# no-fasttest: the MySQL integration is not available in the fast test build. + +# `enable_local_infile` sets `MYSQL_OPT_LOCAL_INFILE` on the connection, which lets the MySQL endpoint +# ask the client for the contents of a file of its choosing: the server opens it with its own +# privileges. A dictionary created with a DDL query may not enable it, whether at the source itself or +# at one of its replicas. A dictionary defined in a server configuration file is written by an operator +# and keeps working; that route is covered by tests/integration/test_dictionaries_mysql. +# +# No MySQL server is needed: the rejection happens while the source is instantiated, before any +# connection. A source is instantiated when the dictionary is loaded, so the reload is what surfaces +# the rejection if the `CREATE` itself did not. +# +# The arms that must NOT be rejected point at `127.0.0.1` port 1, where nothing listens, so they end in +# a connection failure. That failure is asserted too: it is what proves the arm reached an actual +# connection attempt instead of being rejected for some unrelated reason, which is what keeps the +# "no rejection" oracle from passing vacuously. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +REJECTED="cannot be enabled in a dictionary created with a DDL query" +CONNECTION_FAILED="ALL_CONNECTION_TRIES_FAILED" + +create_and_load() { + local name="$1" + local source="$2" + $CLICKHOUSE_CLIENT --query "DROP DICTIONARY IF EXISTS ${name}" + $CLICKHOUSE_CLIENT --query " + CREATE DICTIONARY ${name} (id UInt32, name String) PRIMARY KEY id + SOURCE(MYSQL(${source})) + LAYOUT(FLAT()) LIFETIME(0)" + $CLICKHOUSE_CLIENT --query "SYSTEM RELOAD DICTIONARY ${name}" + $CLICKHOUSE_CLIENT --query "DROP DICTIONARY IF EXISTS ${name}" +} + +expect_rejected() { + echo "--- $1" + if create_and_load "$2" "$3" 2>&1 | grep -q "$REJECTED"; then + echo "OK" + else + echo "FAIL: expected an error matching: $REJECTED" + fi +} + +expect_not_rejected() { + echo "--- $1" + local output + output=$(create_and_load "$2" "$3" 2>&1) + if echo "$output" | grep -q "$REJECTED"; then + echo "FAIL: unexpected error matching: $REJECTED" + elif ! echo "$output" | grep -q "$CONNECTION_FAILED"; then + echo "FAIL: expected the source to reach a connection attempt" + else + echo "OK" + fi +} + +expect_rejected "enabled at the source" dict_infile_source \ + "HOST '127.0.0.1' PORT 1 USER 'u' PASSWORD 'p' DB 'd' TABLE 't' ENABLE_LOCAL_INFILE 1" + +expect_rejected "enabled at a replica" dict_infile_replica \ + "USER 'u' PASSWORD 'p' DB 'd' TABLE 't' + REPLICA(PRIORITY 1 HOST '127.0.0.1' PORT 1 ENABLE_LOCAL_INFILE 1)" + +# Every is checked, not just the first: the pool resolves the option per replica and +# fails over between them, so a later replica enabling it is enough to reach MYSQL_OPT_LOCAL_INFILE. +expect_rejected "enabled at a later replica, disabled at the first" dict_infile_replica_second \ + "USER 'u' PASSWORD 'p' DB 'd' TABLE 't' PORT 1 + REPLICA(PRIORITY 1 HOST '127.0.0.1' ENABLE_LOCAL_INFILE 0) + REPLICA(PRIORITY 2 HOST '127.0.0.1' ENABLE_LOCAL_INFILE 1)" + +# A replica that does not set the option inherits it from the source, so the source is where this one +# is enabled. The guard resolves the value the same way the connection pool does. +expect_rejected "inherited by a replica from the source" dict_infile_inherited \ + "USER 'u' PASSWORD 'p' DB 'd' TABLE 't' PORT 1 ENABLE_LOCAL_INFILE 1 + REPLICA(PRIORITY 1 HOST '127.0.0.1')" + +# Explicitly disabled is what the connection does by default, so it is accepted. +expect_not_rejected "disabled at the source" dict_infile_disabled \ + "HOST '127.0.0.1' PORT 1 USER 'u' PASSWORD 'p' DB 'd' TABLE 't' ENABLE_LOCAL_INFILE 0" + +# Only per-replica pools are built when the source has replicas, so a replica that turns the option off +# never enables it, whatever the source says. Two replicas, because the option is resolved per replica. +expect_not_rejected "disabled at every replica, enabled at the source" dict_infile_replicas_disabled \ + "USER 'u' PASSWORD 'p' DB 'd' TABLE 't' PORT 1 ENABLE_LOCAL_INFILE 1 + REPLICA(PRIORITY 1 HOST '127.0.0.1' ENABLE_LOCAL_INFILE 0) + REPLICA(PRIORITY 2 HOST '127.0.0.1' ENABLE_LOCAL_INFILE 0)" From 47a0b71b58f22f214c6f6d96371c61cfbb658713 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 09:40:47 +0000 Subject: [PATCH 006/185] Backport #120736 to 26.8: Update NuRaft to fix abandoned-peer rejoining --- contrib/NuRaft | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/contrib/NuRaft b/contrib/NuRaft index a0eda0069d2d..884294e9dbee 160000 --- a/contrib/NuRaft +++ b/contrib/NuRaft @@ -1 +1 @@ -Subproject commit a0eda0069d2dac0852c453cbe7b0285eda218479 +Subproject commit 884294e9dbee183b53601bbe2c8c5753cec1a3db From a377c2249b6bda1cd8291f34a8bded599bb506f4 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 10:03:20 +0000 Subject: [PATCH 007/185] Backport #112934 to 26.8: Keep Iceberg statistics for supported columns when a field type is unsupported --- .../DataLakes/Iceberg/IcebergWrites.cpp | 25 +++--- ...eberg_insert_per_file_statistics.reference | 4 +- ...4652_iceberg_insert_per_file_statistics.sh | 16 ++-- ...insert_partial_statistics_bounds.reference | 8 ++ ...ceberg_insert_partial_statistics_bounds.sh | 79 +++++++++++++++++++ 5 files changed, 110 insertions(+), 22 deletions(-) create mode 100644 tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.reference create mode 100755 tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.sh diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index a83d8b5f3278..27c5a81df1cb 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -296,21 +296,22 @@ std::vector dumpFieldToBytes(const Field & field, DataTypePtr type) } } -bool canWriteStatistics( +/// Retains only the bounds that can be serialized. A field left out is simply absent from the +/// manifest bounds map, which readers treat as "bound unknown" for that column. +std::vector> filterWritableStatistics( const std::vector> & statistics, const std::unordered_map & field_id_to_column_index, SharedHeader sample_block) { - if (statistics.empty()) - return false; - + std::vector> writable; + writable.reserve(statistics.size()); for (const auto & [field_id, stat] : statistics) { auto type = sample_block->getDataTypes()[field_id_to_column_index.at(field_id)]; - if (!canDumpIcebergStats(stat, type)) - return false; + if (canDumpIcebergStats(stat, type)) + writable.emplace_back(field_id, stat); } - return true; + return writable; } } @@ -610,13 +611,15 @@ void generateManifestFile( auto dump_fields = [&](size_t field_id, Field value) { return dumpFieldToBytes(value, sample_block->getDataTypes()[field_id_to_column_index.at(field_id)]); }; - auto lower_statistics = effective_statistics->getLowerBounds(); - if (canWriteStatistics(lower_statistics, field_id_to_column_index, sample_block)) + auto lower_statistics + = filterWritableStatistics(effective_statistics->getLowerBounds(), field_id_to_column_index, sample_block); + if (!lower_statistics.empty()) { set_fields(lower_statistics, Iceberg::f_lower_bounds, dump_fields); } - auto upper_statistics = effective_statistics->getUpperBounds(); - if (canWriteStatistics(upper_statistics, field_id_to_column_index, sample_block)) + auto upper_statistics + = filterWritableStatistics(effective_statistics->getUpperBounds(), field_id_to_column_index, sample_block); + if (!upper_statistics.empty()) { set_fields(upper_statistics, Iceberg::f_upper_bounds, dump_fields); } diff --git a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference index 515699bb5704..8f2287b860fa 100644 --- a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference +++ b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference @@ -34,8 +34,8 @@ partition_record_count_null_count: [('{\'eu\'}',1,0),('{\'eu\'}',1,1),('{\'us\'} 0 --- per-entry pairing against the referenced Parquet file --- id=1 rows=1/1 score_nulls=0/0 paired=yes lower=[(1,1),(2,10)] upper=[(1,1),(2,10)] -id=2 rows=1/1 score_nulls=1/1 paired=yes lower=[] upper=[] -id=3 rows=1/1 score_nulls=1/1 paired=yes lower=[] upper=[] +id=2 rows=1/1 score_nulls=1/1 paired=yes lower=[(1,2)] upper=[(1,2)] +id=3 rows=1/1 score_nulls=1/1 paired=yes lower=[(1,3)] upper=[(1,3)] --- per-entry bounds describe only their own file --- id=1 lower=[(1,1),(2,100)] upper=[(1,1),(2,100)] bounds_are_own_row=yes id=2 lower=[(1,2),(2,200)] upper=[(1,2),(2,200)] bounds_are_own_row=yes diff --git a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh index dc05b1e1ec7e..63458bc946f1 100755 --- a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh +++ b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh @@ -153,11 +153,10 @@ ${CLICKHOUSE_CLIENT} --query " echo '--- per-entry pairing against the referenced Parquet file ---' for manifest in $(find "${PAIRED_PATH}/metadata" -maxdepth 1 -name '*.avro' -not -name 'snap-*.avro' -type f | sort); do # `lower_bounds`/`upper_bounds` hold raw little-endian bytes (`dumpValue` in `IcebergWrites.cpp`), - # so they are decoded with `reinterpretAsInt32`; both key columns are `Int32` here. An entry whose - # `score` is all-NULL legitimately has NO bounds at all: `canWriteStatistics` is all-or-nothing - # across the entry's columns and `ColumnNullable::getExtremes` yields NULL extremes for an - # all-NULL column, which `canDumpIcebergStats` rejects. That is pre-existing behaviour, so it is - # asserted here rather than fixed. + # so they are decoded with `reinterpretAsInt32`; both key columns are `Int32` here. Bounds are + # per-field: an entry keeps a bound for every column whose extreme is serializable and omits the + # rest. `ColumnNullable::getExtremes` yields NULL extremes for an all-NULL column, which + # `canDumpIcebergStats` rejects, so an all-NULL `score` contributes no bound while `id` still does. ${CLICKHOUSE_CLIENT} --query " WITH entries AS ( SELECT @@ -190,10 +189,9 @@ for manifest in $(find "${PAIRED_PATH}/metadata" -maxdepth 1 -name '*.avro' -not " done -# The bounds above are present on one entry only, because the two all-NULL files legitimately carry no -# bounds at all. This companion has no nullable column, so every entry carries bounds and the decoded -# value of each is checked against the single row its own file holds. Without it the bounds half of -# this change would be pinned on a single entry. +# Only one entry above carries a `score` bound, because the other two files hold an all-NULL `score`. +# This companion has no nullable column, so every entry carries a bound for both of its columns and the +# decoded value of each is checked against the single row its own file holds. ${CLICKHOUSE_CLIENT} --query " ${ONE_ROW_PER_FILE} CREATE TABLE bounded (id Int32, v Int32) diff --git a/tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.reference b/tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.reference new file mode 100644 index 000000000000..415eb3c1f59a --- /dev/null +++ b/tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.reference @@ -0,0 +1,8 @@ +--- A: key Int32 + arr Array(Int32) +lower_ids=[1] upper_ids=[1] key_lower=[0] key_upper=[4] +--- B: key Int32 + opt Nullable(Int32) all NULL +lower_ids=[1] upper_ids=[1] key_lower=[0] key_upper=[4] +--- C: key Int32 + val String (control, all supported) +lower_ids=[1,2] upper_ids=[1,2] key_lower=[0] key_upper=[4] +--- D: key Int32 + f Nullable(Float64) with values +lower_ids=[1] upper_ids=[1] key_lower=[0] key_upper=[4] diff --git a/tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.sh b/tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.sh new file mode 100755 index 000000000000..8acdb1edb91b --- /dev/null +++ b/tests/queries/0_stateless/04739_iceberg_insert_partial_statistics_bounds.sh @@ -0,0 +1,79 @@ +#!/usr/bin/env bash +# Tags: no-fasttest + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +. "$CUR_DIR"/../shell_config.sh + +PREFIX="t_${CLICKHOUSE_DATABASE}_${RANDOM}" +ROOT="${USER_FILES_PATH}/${PREFIX}" + +cleanup() +{ + for suffix in a b c d; do + ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS ${PREFIX}_${suffix}" + done + rm -rf "${ROOT}" +} +trap cleanup EXIT + +# Prints the field ids present in lower_bounds / upper_bounds, plus the decoded Int32 bounds of +# field id 1 (the `key` column, always Int32 and always serializable here). +report() +{ + local table_dir="$1" + for manifest in $(find "${table_dir}/metadata" -maxdepth 1 -name '*.avro' \ + -not -name 'snap-*.avro' -type f | sort); do + ${CLICKHOUSE_CLIENT} --query " + SELECT + 'lower_ids=' || toString(arraySort(arrayMap(x -> x.1, tupleElement(data_file, 'lower_bounds')))), + 'upper_ids=' || toString(arraySort(arrayMap(x -> x.1, tupleElement(data_file, 'upper_bounds')))), + 'key_lower=' || toString(arrayMap(x -> reinterpretAsInt32(x.2), + arrayFilter(x -> x.1 = 1, tupleElement(data_file, 'lower_bounds')))), + 'key_upper=' || toString(arrayMap(x -> reinterpretAsInt32(x.2), + arrayFilter(x -> x.1 = 1, tupleElement(data_file, 'upper_bounds')))) + FROM file('${manifest}', Avro) + ORDER BY 1, 2, 3, 4 + " + done +} + +# Case A: an unsupported column type (Array(Int32)) must not suppress the bounds of `key`. +echo '--- A: key Int32 + arr Array(Int32)' +${CLICKHOUSE_CLIENT} --query " + CREATE TABLE ${PREFIX}_a (key Int32, arr Array(Int32)) + ENGINE = IcebergLocal('${ROOT}/a/') +" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query \ + "INSERT INTO ${PREFIX}_a SELECT number::Int32, [number, number] FROM numbers(5)" +report "${ROOT}/a" + +# Case B: an entirely-NULL Nullable column yields a Null bound, which must not suppress `key` either. +echo '--- B: key Int32 + opt Nullable(Int32) all NULL' +${CLICKHOUSE_CLIENT} --query " + CREATE TABLE ${PREFIX}_b (key Int32, opt Nullable(Int32)) + ENGINE = IcebergLocal('${ROOT}/b/') +" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query \ + "INSERT INTO ${PREFIX}_b SELECT number::Int32, NULL FROM numbers(5)" +report "${ROOT}/b" + +# Case C: control. All columns serializable, so both field ids keep their bounds exactly as before. +echo '--- C: key Int32 + val String (control, all supported)' +${CLICKHOUSE_CLIENT} --query " + CREATE TABLE ${PREFIX}_c (key Int32, val String) + ENGINE = IcebergLocal('${ROOT}/c/') +" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query \ + "INSERT INTO ${PREFIX}_c SELECT number::Int32, 'val-' || toString(number) FROM numbers(5)" +report "${ROOT}/c" + +# Case D: wrapper composition with a non-NULL value. Float64 is unsupported for bounds even though +# the byte dumper handles it, so the filter must gate on the bounds predicate and skip only `f`. +echo '--- D: key Int32 + f Nullable(Float64) with values' +${CLICKHOUSE_CLIENT} --query " + CREATE TABLE ${PREFIX}_d (key Int32, f Nullable(Float64)) + ENGINE = IcebergLocal('${ROOT}/d/') +" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query \ + "INSERT INTO ${PREFIX}_d SELECT number::Int32, number + 0.5 FROM numbers(5)" +report "${ROOT}/d" From 082700968837c479d07759aabc9e2ad890c486b0 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 10:12:08 +0000 Subject: [PATCH 008/185] Backport #121392 to 26.8: Add column size statistics in metadata files --- src/Processors/Formats/IOutputFormat.h | 3 + .../Formats/Impl/ParquetBlockOutputFormat.cpp | 58 +++++++++ .../Formats/Impl/ParquetBlockOutputFormat.h | 4 + .../DataLakes/Iceberg/Compaction.cpp | 1 + .../DataLakes/Iceberg/DataFileStatistics.cpp | 60 ++++++++-- .../DataLakes/Iceberg/DataFileStatistics.h | 2 + .../DataLakes/Iceberg/IcebergWrites.cpp | 7 +- .../DataLakes/Iceberg/MultipleFileWriter.cpp | 6 + .../DataLakes/Iceberg/Mutations.cpp | 5 + ...est_writes_statistics_by_minmax_pruning.py | 113 +++++++++++++++++- ...eberg_insert_per_file_statistics.reference | 6 +- ...4652_iceberg_insert_per_file_statistics.sh | 24 ++-- 12 files changed, 262 insertions(+), 27 deletions(-) diff --git a/src/Processors/Formats/IOutputFormat.h b/src/Processors/Formats/IOutputFormat.h index a86ca31cf01d..6edf87caef91 100644 --- a/src/Processors/Formats/IOutputFormat.h +++ b/src/Processors/Formats/IOutputFormat.h @@ -1,6 +1,7 @@ #pragma once #include +#include #include #include #include @@ -86,6 +87,8 @@ class IOutputFormat : public IProcessor virtual bool supportsWritingException() const { return false; } virtual void setException(const String & /*exception_message*/) {} + virtual std::unordered_map getColumnSizesOnDisk() const { return {}; } + /// A framing format (see IFramingFormat.h) multiplexes the formatted data along with auxiliary /// packets (progress, logs, profile events, exceptions) in the output stream. The format must /// have been created over the framing format's payload buffer. When set, the format notifies diff --git a/src/Processors/Formats/Impl/ParquetBlockOutputFormat.cpp b/src/Processors/Formats/Impl/ParquetBlockOutputFormat.cpp index 7d5ea6dd2941..28c3c0816d54 100644 --- a/src/Processors/Formats/Impl/ParquetBlockOutputFormat.cpp +++ b/src/Processors/Formats/Impl/ParquetBlockOutputFormat.cpp @@ -21,6 +21,11 @@ namespace CurrentMetrics namespace DB { +namespace ErrorCodes +{ + extern const int LOGICAL_ERROR; +} + using namespace Parquet; ParquetBlockOutputFormat::ParquetBlockOutputFormat(WriteBuffer & out_, SharedHeader header_, const FormatSettings & format_settings_, FormatFilterInfoPtr format_filter_info_) @@ -198,10 +203,62 @@ void ParquetBlockOutputFormat::finalizeImpl() writeFileHeader(file_state, out); } Block header = materializeBlock(getPort(PortKind::Main).getHeader()); + collectColumnSizesOnDisk(header); writeFileFooter(file_state, schema, options, out, header); chassert(out.count() - base_offset == file_state.offset); } +static size_t countSchemaLeaves(const SchemaElements & schema, size_t & index) +{ + if (index >= schema.size()) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Parquet schema of {} elements is truncated", schema.size()); + + const auto & element = schema[index]; + ++index; + + if (!element.__isset.num_children || element.num_children == 0) + return 1; + + size_t leaves = 0; + for (Int32 i = 0; i < element.num_children; ++i) + leaves += countSchemaLeaves(schema, index); + return leaves; +} + +void ParquetBlockOutputFormat::collectColumnSizesOnDisk(const Block & header) +{ + std::vector leaves_per_column; + leaves_per_column.reserve(header.columns()); + size_t schema_index = 1; + size_t num_leaves = 0; + for (size_t i = 0; i < header.columns(); ++i) + { + leaves_per_column.push_back(countSchemaLeaves(schema, schema_index)); + num_leaves += leaves_per_column.back(); + } + + column_sizes_on_disk.clear(); + for (const auto & row_group : file_state.completed_row_groups) + { + const auto & column_chunks = row_group.row_group.columns; + if (column_chunks.size() != num_leaves) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Parquet row group has {} column chunks while the schema has {} leaf columns", + column_chunks.size(), + num_leaves); + + size_t leaf_index = 0; + for (size_t i = 0; i < header.columns(); ++i) + { + size_t column_size = 0; + for (size_t j = 0; j < leaves_per_column[i]; ++j, ++leaf_index) + column_size += static_cast(column_chunks[leaf_index].meta_data.total_compressed_size); + column_sizes_on_disk[header.getByPosition(i).name] += column_size; + } + } +} + void ParquetBlockOutputFormat::resetFormatterImpl() { if (pool) @@ -216,6 +273,7 @@ void ParquetBlockOutputFormat::resetFormatterImpl() task_queue.clear(); row_groups.clear(); file_state = {}; + column_sizes_on_disk.clear(); staging_chunks.clear(); staging_rows = 0; staging_bytes = 0; diff --git a/src/Processors/Formats/Impl/ParquetBlockOutputFormat.h b/src/Processors/Formats/Impl/ParquetBlockOutputFormat.h index 07322de00c29..c38b1e5d0ce7 100644 --- a/src/Processors/Formats/Impl/ParquetBlockOutputFormat.h +++ b/src/Processors/Formats/Impl/ParquetBlockOutputFormat.h @@ -20,6 +20,8 @@ class ParquetBlockOutputFormat final : public IOutputFormat String getName() const override { return "ParquetBlockOutputFormat"; } + std::unordered_map getColumnSizesOnDisk() const override { return column_sizes_on_disk; } + private: struct MemoryToken { @@ -94,6 +96,7 @@ class ParquetBlockOutputFormat final : public IOutputFormat void consume(Chunk) override; void finalizeImpl() override; + void collectColumnSizesOnDisk(const Block & header); void resetFormatterImpl() override; void onCancel() noexcept override; @@ -119,6 +122,7 @@ class ParquetBlockOutputFormat final : public IOutputFormat Parquet::IcebergOptionality iceberg_optionality; Parquet::SchemaElements schema; Parquet::FileWriteState file_state; + std::unordered_map column_sizes_on_disk; size_t base_offset = 0; // initial out.count(), just for assert std::mutex mutex; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp index beb714c61a23..bb21135a0aab 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Compaction.cpp @@ -407,6 +407,7 @@ static void writeDataFiles( } output_format->flush(); output_format->finalize(); + data_file->manifest_list->statistics.addColumnSizesOnDisk(output_format->getColumnSizesOnDisk(), *sample_block); write_buffer->finalize(); auto file_bytes = write_buffer->count(); if (file_bytes == 0 && !data_file->patched_path.empty()) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.cpp index 4e6879496faf..95e3293aa664 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.cpp @@ -3,10 +3,16 @@ #include #include #include +#include namespace DB { +namespace ErrorCodes +{ + extern const int LOGICAL_ERROR; +} + #if USE_AVRO DataFileStatistics::DataFileStatistics(Poco::JSON::Array::Ptr schema_) @@ -33,9 +39,8 @@ void DataFileStatistics::update(const Chunk & chunk) if (!chunk.hasRows()) return; size_t num_columns = chunk.getNumColumns(); - if (column_sizes.empty()) + if (null_counts.empty()) { - column_sizes.resize(num_columns, 0); null_counts.resize(num_columns, 0); for (size_t i = 0; i < num_columns; ++i) { @@ -48,7 +53,6 @@ void DataFileStatistics::update(const Chunk & chunk) for (size_t i = 0; i < num_columns; ++i) { const auto & col = chunk.getColumns()[i]; - column_sizes[i] += col->byteSize(); if (const auto * nullable_col = checkAndGetColumn(col.get())) { for (UInt8 v : nullable_col->getNullMapData()) @@ -58,23 +62,61 @@ void DataFileStatistics::update(const Chunk & chunk) } } -void DataFileStatistics::merge(const DataFileStatistics & other) +void DataFileStatistics::addColumnSizesOnDisk(const std::unordered_map & sizes_by_column_name, const Block & sample_block) { - if (other.column_sizes.empty()) + if (sizes_by_column_name.empty()) return; + if (sample_block.columns() != field_ids.size()) + throw Exception( + ErrorCodes::LOGICAL_ERROR, + "Iceberg data file has {} columns while its schema has {} fields", + sample_block.columns(), + field_ids.size()); + if (column_sizes.empty()) + column_sizes.resize(field_ids.size(), 0); + + for (size_t i = 0; i < field_ids.size(); ++i) + { + const auto & column_name = sample_block.getByPosition(i).name; + auto it = sizes_by_column_name.find(column_name); + if (it == sizes_by_column_name.end()) + throw Exception( + ErrorCodes::LOGICAL_ERROR, "Written data file does not report the on-disk size of column {}", column_name); + column_sizes[i] += static_cast(it->second); + } +} + +void DataFileStatistics::merge(const DataFileStatistics & other) +{ + if (!other.column_sizes.empty()) + { + if (column_sizes.empty()) + { + column_sizes = other.column_sizes; + } + else + { + chassert(column_sizes.size() == other.column_sizes.size()); + for (size_t i = 0; i < column_sizes.size(); ++i) + column_sizes[i] += other.column_sizes[i]; + } + } + + if (other.null_counts.empty()) + return; + + if (null_counts.empty()) { - column_sizes = other.column_sizes; null_counts = other.null_counts; ranges = other.ranges; return; } - chassert(column_sizes.size() == other.column_sizes.size()); - for (size_t i = 0; i < column_sizes.size(); ++i) + chassert(null_counts.size() == other.null_counts.size()); + for (size_t i = 0; i < null_counts.size(); ++i) { - column_sizes[i] += other.column_sizes[i]; null_counts[i] += other.null_counts[i]; ranges[i] = uniteRanges(ranges[i], other.ranges[i]); } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.h index 680b283e0388..6352f2df9ad7 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/DataFileStatistics.h @@ -7,6 +7,7 @@ #include #include +#include #include #include @@ -25,6 +26,7 @@ class DataFileStatistics explicit DataFileStatistics(Poco::JSON::Array::Ptr schema_); void update(const Chunk & chunk); + void addColumnSizesOnDisk(const std::unordered_map & sizes_by_column_name, const Block & sample_block); void merge(const DataFileStatistics & other); std::vector> getColumnSizes() const; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index a83d8b5f3278..f9e3ef2fb69a 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -596,10 +596,11 @@ void generateManifestFile( } else if (effective_statistics) { - auto statistics = effective_statistics->getColumnSizes(); - set_fields(statistics, Iceberg::f_column_sizes, [](size_t, size_t value) { return static_cast(value); }); + auto column_sizes = effective_statistics->getColumnSizes(); + if (!column_sizes.empty()) + set_fields(column_sizes, Iceberg::f_column_sizes, [](size_t, size_t value) { return static_cast(value); }); - statistics = effective_statistics->getNullCounts(); + auto statistics = effective_statistics->getNullCounts(); set_fields(statistics, Iceberg::f_null_value_counts, [](size_t, size_t value) { return static_cast(value); }); std::unordered_map field_id_to_column_index; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp index 06f1ca6fcec6..1935421c3701 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/MultipleFileWriter.cpp @@ -85,6 +85,12 @@ void MultipleFileWriter::finalize() { output_format->flush(); output_format->finalize(); + + auto column_sizes_on_disk = output_format->getColumnSizesOnDisk(); + if (current_file_stats) + current_file_stats->addColumnSizesOnDisk(column_sizes_on_disk, *sample_block); + stats.addColumnSizesOnDisk(column_sizes_on_disk, *sample_block); + buffer->finalize(); UInt64 file_bytes = buffer->count(); total_bytes += file_bytes; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp index 3bb6890d7f48..6ffc50d493b8 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Mutations.cpp @@ -267,6 +267,8 @@ static std::optional writeDataFiles( { delete_data_writers[partition_key]->flush(); delete_data_writers[partition_key]->finalize(); + delete_data_statistics.at(partition_key).addColumnSizesOnDisk( + delete_data_writers[partition_key]->getColumnSizesOnDisk(), getPositionDeleteFileSampleBlock()); delete_data_write_buffers[partition_key]->finalize(); { auto delete_bytes = delete_data_write_buffers[partition_key]->count(); @@ -289,6 +291,7 @@ static std::optional writeDataFiles( PullingPipelineExecutor executor(pipeline); auto header = interpreter->getUpdatedHeader(); + auto update_sample_block = getNonVirtualColumns(header, /*remove_low_cardinality=*/ true); Block block; while (executor.pull(block)) @@ -338,6 +341,8 @@ static std::optional writeDataFiles( { update_data_writers[partition_key]->flush(); update_data_writers[partition_key]->finalize(); + update_data_statistics.at(partition_key).addColumnSizesOnDisk( + update_data_writers[partition_key]->getColumnSizesOnDisk(), update_sample_block); update_data_write_buffers[partition_key]->finalize(); { auto update_bytes = update_data_write_buffers[partition_key]->count(); diff --git a/tests/integration/test_storage_iceberg_no_spark/test_writes_statistics_by_minmax_pruning.py b/tests/integration/test_storage_iceberg_no_spark/test_writes_statistics_by_minmax_pruning.py index 2cc7d839778a..63ac6063258f 100644 --- a/tests/integration/test_storage_iceberg_no_spark/test_writes_statistics_by_minmax_pruning.py +++ b/tests/integration/test_storage_iceberg_no_spark/test_writes_statistics_by_minmax_pruning.py @@ -1,11 +1,22 @@ +import glob +import os + +import avro.datafile +import avro.io +import pyarrow.parquet as pq import pytest from helpers.iceberg_utils import ( check_validity_and_get_prunned_files_general, create_iceberg_table, - get_uuid_str + default_download_directory, + get_last_snapshot, + get_uuid_str, + unescape_path, ) +TABLE_ROOT = "/var/lib/clickhouse/user_files/iceberg_data/default" + @pytest.mark.parametrize("format_version", [1, 2]) @pytest.mark.parametrize("storage_type", ["s3", "azure", "local"]) @@ -133,4 +144,102 @@ def check_validity_and_get_prunned_files(select_expression): f"SELECT * FROM {TABLE_NAME} WHERE number <= 5 ORDER BY ALL" ) == 3 - ) \ No newline at end of file + ) + + +def _read_avro(path): + with open(path, "rb") as f: + return list(avro.datafile.DataFileReader(f, avro.io.DatumReader())) + + +def _data_files_of_last_snapshot(table_path): + """The `data_file` record of every manifest entry of the newest snapshot.""" + snapshot_id = get_last_snapshot(table_path) + manifest_lists = glob.glob(f"{table_path}/metadata/snap-{snapshot_id}-*.avro") + assert len(manifest_lists) == 1, manifest_lists + + data_files = [] + for list_entry in _read_avro(manifest_lists[0]): + manifest = os.path.join( + table_path, "metadata", os.path.basename(unescape_path(list_entry["manifest_path"])) + ) + data_files.extend(entry["data_file"] for entry in _read_avro(manifest)) + return data_files + + +def _column_sizes_from_parquet(path): + """Per-column compressed size as the Parquet footer of one data file reports it, keyed by the + Iceberg field id the column carries.""" + parquet_file = pq.ParquetFile(path) + field_ids = [ + int(field.metadata[b"PARQUET:field_id"]) for field in parquet_file.schema_arrow + ] + + sizes = dict.fromkeys(field_ids, 0) + metadata = parquet_file.metadata + for row_group_index in range(metadata.num_row_groups): + row_group = metadata.row_group(row_group_index) + assert row_group.num_columns == len(field_ids) + for column_index, field_id in enumerate(field_ids): + sizes[field_id] += row_group.column(column_index).total_compressed_size + return sizes + + +@pytest.mark.parametrize("format_version", [1, 2]) +@pytest.mark.parametrize("storage_type", ["s3", "local"]) +def test_writes_column_sizes_are_on_disk_sizes( + started_cluster_iceberg_no_spark, format_version, storage_type +): + """`data_file.column_sizes` is the size the column occupies inside the data file, so every + entry must repeat what the Parquet footer of the file it names says. An in-memory size passes + neither check below: `s` holds 1000000 bytes of one repeated character, which compresses to a + fraction of the file, and the sum over a file cannot exceed the file itself.""" + instance = started_cluster_iceberg_no_spark.instances["node1"] + TABLE_NAME = ( + "test_writes_column_sizes_are_on_disk_sizes_" + + storage_type + + "_" + + get_uuid_str() + ) + + create_iceberg_table( + storage_type, + instance, + TABLE_NAME, + started_cluster_iceberg_no_spark, + "(id Int32, s String)", + format_version, + order_by="id", + ) + # Blocks of 1000 rows that the insert pipeline does not squash, so the writer rolls over at + # exactly 4000 rows and the commit holds three data files of known sizes. + instance.query( + f"INSERT INTO {TABLE_NAME} SELECT number, repeat('x', 100) FROM numbers(10000)", + settings={ + "iceberg_insert_max_rows_in_data_file": 4000, + "max_block_size": 1000, + "max_insert_block_size": 1000, + "min_insert_block_size_rows": 0, + "min_insert_block_size_bytes": 0, + "max_insert_threads": 1, + }, + ) + + table_path = f"{TABLE_ROOT}/{TABLE_NAME}/" + default_download_directory( + started_cluster_iceberg_no_spark, storage_type, table_path, table_path + ) + + data_files = _data_files_of_last_snapshot(table_path) + assert len(data_files) == 3 + + for data_file in data_files: + data_path = os.path.join( + table_path, "data", os.path.basename(unescape_path(data_file["file_path"])) + ) + written = {pair["key"]: pair["value"] for pair in data_file["column_sizes"]} + + assert written == _column_sizes_from_parquet(data_path) + assert sum(written.values()) <= data_file["file_size_in_bytes"] + + assert instance.query(f"SELECT count(), uniqExact(s) FROM {TABLE_NAME}") == "10000\t1\n" diff --git a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference index 515699bb5704..09024e5f5d00 100644 --- a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference +++ b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.reference @@ -41,6 +41,6 @@ id=1 lower=[(1,1),(2,100)] upper=[(1,1),(2,100)] bounds_are_own_row=yes id=2 lower=[(1,2),(2,200)] upper=[(1,2),(2,200)] bounds_are_own_row=yes id=3 lower=[(1,3),(2,300)] upper=[(1,3),(2,300)] bounds_are_own_row=yes --- per-entry column_sizes describe only their own file --- -id=1 width=4 id_size=4 s_size=12 -id=2 width=8 id_size=4 s_size=16 -id=3 width=12 id_size=4 s_size=20 +id=1 width=10000 id_size_positive=yes s_size_follows_own_width=yes sizes_fit_own_file=yes +id=2 width=100000 id_size_positive=yes s_size_follows_own_width=yes sizes_fit_own_file=yes +id=3 width=1000000 id_size_positive=yes s_size_follows_own_width=yes sizes_fit_own_file=yes diff --git a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh index dc05b1e1ec7e..710ebee9df7d 100755 --- a/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh +++ b/tests/queries/0_stateless/04652_iceberg_insert_per_file_statistics.sh @@ -235,16 +235,16 @@ done # The two scenarios above pin `record_count`, `null_value_counts` and the bounds to the file each # entry names, but not `column_sizes`, whose only other assertion is a cross-table `max()` inequality # that constrains no individual entry. Both of their fixtures also give every file the same per-file -# sizes, so a permutation cannot move a size value there. This scenario gives each file a `String` of -# a different width, which makes the sizes distinct and therefore permutation-sensitive. The values -# are in-memory `IColumn::byteSize` sums, not Parquet file sizes, so they are deterministic: -# `ColumnString::byteSize` is `chars.size() + offsets.size() * sizeof(offsets[0])`, and `insertData` -# appends exactly `length` bytes with no terminator. +# sizes, so a permutation cannot move a size value there. This scenario gives each file an +# incompressible `String` of a different order of magnitude, which makes the sizes distinct and +# therefore permutation-sensitive. `column_sizes` holds the post-compression size of the column chunk +# inside the Parquet file, so the exact byte count is a property of the encoder; what is asserted is +# that each entry's size follows the width of the row its own file holds and fits inside that file. ${CLICKHOUSE_CLIENT} --query " ${ONE_ROW_PER_FILE} CREATE TABLE sized (id Int32, s String) ENGINE = IcebergLocal('${SIZED_PATH}', 'Parquet') ORDER BY (id); - INSERT INTO sized SELECT number + 1, repeat('x', (number + 1) * 4) FROM numbers(3); + INSERT INTO sized SELECT number + 1, randomPrintableASCII(toUInt32(pow(10, number + 4))) FROM numbers(3); " echo '--- per-entry column_sizes describe only their own file ---' @@ -252,8 +252,9 @@ for manifest in $(find "${SIZED_PATH}/metadata" -maxdepth 1 -name '*.avro' -not ${CLICKHOUSE_CLIENT} --query " WITH entries AS ( SELECT - replaceRegexpOne(tupleElement(data_file, 'file_path'), '^.*/', '') AS base, - CAST(tupleElement(data_file, 'column_sizes'), 'Map(Int32, Int64)') AS entry_sizes + replaceRegexpOne(tupleElement(data_file, 'file_path'), '^.*/', '') AS base, + CAST(tupleElement(data_file, 'column_sizes'), 'Map(Int32, Int64)') AS entry_sizes, + tupleElement(data_file, 'file_size_in_bytes') AS entry_file_size FROM file('${manifest}', Avro) ), files AS ( @@ -267,8 +268,11 @@ for manifest in $(find "${SIZED_PATH}/metadata" -maxdepth 1 -name '*.avro' -not SELECT 'id=' || toString(f.own_id) || ' width=' || toString(f.own_width) - || ' id_size=' || toString(e.entry_sizes[1]) - || ' s_size=' || toString(e.entry_sizes[2]) AS entry + || ' id_size_positive=' || if(e.entry_sizes[1] > 0, 'yes', 'no') + || ' s_size_follows_own_width=' || if( + e.entry_sizes[2] > f.own_width / 2 AND e.entry_sizes[2] < f.own_width * 2, 'yes', 'no') + || ' sizes_fit_own_file=' || if( + e.entry_sizes[1] + e.entry_sizes[2] <= e.entry_file_size, 'yes', 'no') AS entry FROM entries AS e INNER JOIN files AS f ON e.base = f.base ORDER BY f.own_id FORMAT TSV; From cd83bc785c31287f3feebfd6d54e66ec32e7c895 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 12:01:46 +0000 Subject: [PATCH 009/185] Backport #116099 to 26.8: Honour `date_time_overflow_behavior` on parsing Date[Time] from text --- src/Core/FormatFactorySettings.h | 2 +- .../Serializations/SerializationDate.cpp | 32 ++-- .../Serializations/SerializationDate32.cpp | 40 +++-- .../Serializations/SerializationDateTime.cpp | 34 ++-- src/Formats/FormatSettings.h | 2 + src/Formats/JSONExtractTree.cpp | 25 ++- src/Functions/FunctionsConversion.h | 63 +++---- src/IO/ReadHelpers.cpp | 74 ++++++--- src/IO/ReadHelpers.h | 156 +++++++++++------- src/IO/parseDateTimeBestEffort.cpp | 101 ++++++------ src/IO/parseDateTimeBestEffort.h | 12 +- ...me_overflow_behavior_from_string.reference | 42 +++++ ...ate_time_overflow_behavior_from_string.sql | 105 ++++++++++++ 13 files changed, 470 insertions(+), 218 deletions(-) create mode 100644 tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.reference create mode 100644 tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.sql diff --git a/src/Core/FormatFactorySettings.h b/src/Core/FormatFactorySettings.h index 5bc1e8766ec4..d48bdde8189e 100644 --- a/src/Core/FormatFactorySettings.h +++ b/src/Core/FormatFactorySettings.h @@ -1613,7 +1613,7 @@ Possible values: Use the precise float parsing algorithm, which always returns the closest representable value to the input. When disabled, a faster but less accurate algorithm is used that may differ from the precise result by the least significant bits. )", 0) \ DECLARE(DateTimeOverflowBehavior, date_time_overflow_behavior, "ignore", R"( -Defines the behavior when [Date](/reference/data-types/date), [Date32](/reference/data-types/date32), [DateTime](/reference/data-types/datetime), [DateTime64](/reference/data-types/datetime64) or integers are converted into Date, Date32, DateTime or DateTime64 but the value cannot be represented in the result type. +Defines the behavior when [Date](/reference/data-types/date), [Date32](/reference/data-types/date32), [DateTime](/reference/data-types/datetime), [DateTime64](/reference/data-types/datetime64) or integers are converted into Date, Date32, DateTime or DateTime64 but the value cannot be represented in the result type. It also applies when a `Date` or `DateTime` is parsed from text, including by an input format. Possible values: diff --git a/src/DataTypes/Serializations/SerializationDate.cpp b/src/DataTypes/Serializations/SerializationDate.cpp index 0552baf4fd04..8b33f626b671 100644 --- a/src/DataTypes/Serializations/SerializationDate.cpp +++ b/src/DataTypes/Serializations/SerializationDate.cpp @@ -38,26 +38,26 @@ void SerializationDate::deserializeWholeText(IColumn & column, ReadBuffer & istr throwUnexpectedDataAfterParsedValue(column, istr, settings, "Date"); } -bool SerializationDate::tryDeserializeWholeText(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate::tryDeserializeWholeText(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum x; - if (!tryReadDateText(x, istr, time_zone) || !istr.eof()) + if (!tryReadDateText(x, istr, time_zone, nullptr, !settings.throwOnDateTimeOverflow()) || !istr.eof()) return false; assert_cast(column).getData().push_back(x); return true; } -void SerializationDate::deserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +void SerializationDate::deserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum x; - readDateText(x, istr, time_zone); + readDateText(x, istr, time_zone, !settings.throwOnDateTimeOverflow()); assert_cast(column).getData().push_back(x); } -bool SerializationDate::tryDeserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate::tryDeserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum x; - if (!tryReadDateText(x, istr, time_zone)) + if (!tryReadDateText(x, istr, time_zone, nullptr, !settings.throwOnDateTimeOverflow())) return false; assert_cast(column).getData().push_back(x); return true; @@ -75,19 +75,19 @@ void SerializationDate::serializeTextQuoted(const IColumn & column, size_t row_n writeChar('\'', ostr); } -void SerializationDate::deserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +void SerializationDate::deserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum x; assertChar('\'', istr); - readDateText(x, istr, time_zone); + readDateText(x, istr, time_zone, !settings.throwOnDateTimeOverflow()); assertChar('\'', istr); assert_cast(column).getData().push_back(x); /// It's important to do this at the end - for exception safety. } -bool SerializationDate::tryDeserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate::tryDeserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum x; - if (!checkChar('\'', istr) || !tryReadDateText(x, istr, time_zone) || !checkChar('\'', istr)) + if (!checkChar('\'', istr) || !tryReadDateText(x, istr, time_zone, nullptr, !settings.throwOnDateTimeOverflow()) || !checkChar('\'', istr)) return false; assert_cast(column).getData().push_back(x); @@ -109,7 +109,7 @@ void SerializationDate::deserializeTextJSON(IColumn & column, ReadBuffer & istr, return; } DayNum x; - readDateText(x, istr, time_zone); + readDateText(x, istr, time_zone, !format_settings.throwOnDateTimeOverflow()); assertChar('"', istr); assert_cast(column).getData().push_back(x); } @@ -120,7 +120,7 @@ bool SerializationDate::tryDeserializeTextJSON(IColumn & column, ReadBuffer & is return SerializationNumber::tryDeserializeTextJSON(column, istr, format_settings); DayNum x; - if (!tryReadDateText(x, istr, time_zone) || !checkChar('"', istr)) + if (!tryReadDateText(x, istr, time_zone, nullptr, !format_settings.throwOnDateTimeOverflow()) || !checkChar('"', istr)) return false; assert_cast(column).getData().push_back(x); return true; @@ -133,17 +133,17 @@ void SerializationDate::serializeTextCSV(const IColumn & column, size_t row_num, writeChar('"', ostr); } -void SerializationDate::deserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +void SerializationDate::deserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum value; - readCSV(value, istr, time_zone); + readCSV(value, istr, time_zone, !settings.throwOnDateTimeOverflow()); assert_cast(column).getData().push_back(value); } -bool SerializationDate::tryDeserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate::tryDeserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { DayNum value; - if (!tryReadCSV(value, istr, time_zone)) + if (!tryReadCSV(value, istr, time_zone, !settings.throwOnDateTimeOverflow())) return false; assert_cast(column).getData().push_back(value); return true; diff --git a/src/DataTypes/Serializations/SerializationDate32.cpp b/src/DataTypes/Serializations/SerializationDate32.cpp index 0f17f1de8779..085bbf971765 100644 --- a/src/DataTypes/Serializations/SerializationDate32.cpp +++ b/src/DataTypes/Serializations/SerializationDate32.cpp @@ -10,6 +10,11 @@ namespace DB { +namespace ErrorCodes +{ + extern const int CANNOT_PARSE_DATE; +} + UInt128 SerializationDate32::getHash(const DateLUTImpl & time_zone_) { SipHash hash; @@ -37,26 +42,26 @@ void SerializationDate32::deserializeWholeText(IColumn & column, ReadBuffer & is throwUnexpectedDataAfterParsedValue(column, istr, settings, "Date32"); } -bool SerializationDate32::tryDeserializeWholeText(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate32::tryDeserializeWholeText(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { ExtendedDayNum x; - if (!tryReadDateText(x, istr, time_zone) || !istr.eof()) + if (!tryReadDateText(x, istr, time_zone, nullptr, !settings.throwOnDateTimeOverflow()) || !istr.eof()) return false; assert_cast(column).getData().push_back(x); return true; } -void SerializationDate32::deserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +void SerializationDate32::deserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { ExtendedDayNum x; - readDateText(x, istr, time_zone); + readDateText(x, istr, time_zone, !settings.throwOnDateTimeOverflow()); assert_cast(column).getData().push_back(x); } -bool SerializationDate32::tryDeserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate32::tryDeserializeTextEscaped(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { ExtendedDayNum x; - if (!tryReadDateText(x, istr, time_zone)) + if (!tryReadDateText(x, istr, time_zone, nullptr, !settings.throwOnDateTimeOverflow())) return false; assert_cast(column).getData().push_back(x); return true; @@ -74,19 +79,19 @@ void SerializationDate32::serializeTextQuoted(const IColumn & column, size_t row writeChar('\'', ostr); } -void SerializationDate32::deserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +void SerializationDate32::deserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { ExtendedDayNum x; assertChar('\'', istr); - readDateText(x, istr, time_zone); + readDateText(x, istr, time_zone, !settings.throwOnDateTimeOverflow()); assertChar('\'', istr); assert_cast(column).getData().push_back(x); /// It's important to do this at the end - for exception safety. } -bool SerializationDate32::tryDeserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate32::tryDeserializeTextQuoted(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { ExtendedDayNum x; - if (!checkChar('\'', istr) || !tryReadDateText(x, istr, time_zone) || !checkChar('\'', istr)) + if (!checkChar('\'', istr) || !tryReadDateText(x, istr, time_zone, nullptr, !settings.throwOnDateTimeOverflow()) || !checkChar('\'', istr)) return false; assert_cast(column).getData().push_back(x); /// It's important to do this at the end - for exception safety. return true; @@ -107,7 +112,7 @@ void SerializationDate32::deserializeTextJSON(IColumn & column, ReadBuffer & ist return; } ExtendedDayNum x; - readDateText(x, istr, time_zone); + readDateText(x, istr, time_zone, !format_settings.throwOnDateTimeOverflow()); assertChar('"', istr); assert_cast(column).getData().push_back(x); } @@ -118,7 +123,7 @@ bool SerializationDate32::tryDeserializeTextJSON(IColumn & column, ReadBuffer & return SerializationNumber::tryDeserializeTextJSON(column, istr, format_settings); ExtendedDayNum x; - if (!tryReadDateText(x, istr, time_zone) || !checkChar('"', istr)) + if (!tryReadDateText(x, istr, time_zone, nullptr, !format_settings.throwOnDateTimeOverflow()) || !checkChar('"', istr)) return false; assert_cast(column).getData().push_back(x); return true; @@ -131,18 +136,25 @@ void SerializationDate32::serializeTextCSV(const IColumn & column, size_t row_nu writeChar('"', ostr); } -void SerializationDate32::deserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +void SerializationDate32::deserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { LocalDate value; readCSV(value, istr); + /// This one goes through `LocalDate`, which accepts a calendar-invalid date and resolves it to a default + if (settings.throwOnDateTimeOverflow() + && !tryToMakeDayNum(time_zone, value.year(), value.month(), value.day())) + throw Exception(ErrorCodes::CANNOT_PARSE_DATE, "Cannot parse date"); assert_cast(column).getData().push_back(value.getExtenedDayNum()); } -bool SerializationDate32::tryDeserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings &) const +bool SerializationDate32::tryDeserializeTextCSV(IColumn & column, ReadBuffer & istr, const FormatSettings & settings) const { LocalDate value; if (!tryReadCSV(value, istr)) return false; + if (settings.throwOnDateTimeOverflow() + && !tryToMakeDayNum(time_zone, value.year(), value.month(), value.day())) + return false; assert_cast(column).getData().push_back(value.getExtenedDayNum()); return true; } diff --git a/src/DataTypes/Serializations/SerializationDateTime.cpp b/src/DataTypes/Serializations/SerializationDateTime.cpp index 042807705667..16df898076f8 100644 --- a/src/DataTypes/Serializations/SerializationDateTime.cpp +++ b/src/DataTypes/Serializations/SerializationDateTime.cpp @@ -43,16 +43,19 @@ namespace inline void readText(time_t & x, ReadBuffer & istr, const FormatSettings & settings, const DateLUTImpl & time_zone, const DateLUTImpl & utc_time_zone) { + const auto overflow = settings.throwOnDateTimeOverflow() + ? DateTimeOverflow::Report + : DateTimeOverflow::Saturate; switch (settings.date_time_input_format) { case FormatSettings::DateTimeInputFormat::Basic: - readDateTimeTextImpl<>(x, istr, time_zone); + readDateTimeTextImpl<>(x, istr, time_zone, nullptr, nullptr, overflow == DateTimeOverflow::Saturate); break; case FormatSettings::DateTimeInputFormat::BestEffort: - parseDateTimeBestEffort(x, istr, time_zone, utc_time_zone); + parseDateTimeBestEffort(x, istr, time_zone, utc_time_zone, overflow); break; case FormatSettings::DateTimeInputFormat::BestEffortUS: - parseDateTimeBestEffortUS(x, istr, time_zone, utc_time_zone); + parseDateTimeBestEffortUS(x, istr, time_zone, utc_time_zone, overflow); break; } @@ -62,17 +65,20 @@ readText(time_t & x, ReadBuffer & istr, const FormatSettings & settings, const D inline bool tryReadText( time_t & x, ReadBuffer & istr, const FormatSettings & settings, const DateLUTImpl & time_zone, const DateLUTImpl & utc_time_zone) { + const auto overflow = settings.throwOnDateTimeOverflow() + ? DateTimeOverflow::Report + : DateTimeOverflow::Saturate; bool res = false; switch (settings.date_time_input_format) { case FormatSettings::DateTimeInputFormat::Basic: - res = tryReadDateTimeText(x, istr, time_zone); + res = tryReadDateTimeText(x, istr, time_zone, nullptr, nullptr, overflow == DateTimeOverflow::Saturate); break; case FormatSettings::DateTimeInputFormat::BestEffort: - res = tryParseDateTimeBestEffort(x, istr, time_zone, utc_time_zone); + res = tryParseDateTimeBestEffort(x, istr, time_zone, utc_time_zone, overflow); break; case FormatSettings::DateTimeInputFormat::BestEffortUS: - res = tryParseDateTimeBestEffortUS(x, istr, time_zone, utc_time_zone); + res = tryParseDateTimeBestEffortUS(x, istr, time_zone, utc_time_zone, overflow); break; } @@ -181,11 +187,11 @@ void SerializationDateTime::deserializeTextQuoted(IColumn & column, ReadBuffer & } else if (settings.read_datetime_number_as_raw_value) /// Legacy: the raw value (seconds). { - readDateTimeAsRawValue(x, istr); + readDateTimeAsRawValue(x, istr, !settings.throwOnDateTimeOverflow()); } else /// Just 1504193808 or 1703363853.5 (a Unix timestamp, possibly with a sub-second part) { - readDateTimeAsNumber(x, istr); + readDateTimeAsNumber(x, istr, !settings.throwOnDateTimeOverflow()); } /// It's important to do this at the end - for exception safety. @@ -202,12 +208,12 @@ bool SerializationDateTime::tryDeserializeTextQuoted(IColumn & column, ReadBuffe } else if (settings.read_datetime_number_as_raw_value) /// Legacy: the raw value (seconds). { - if (!tryReadDateTimeAsRawValue(x, istr)) + if (!tryReadDateTimeAsRawValue(x, istr, !settings.throwOnDateTimeOverflow())) return false; } else /// Just 1504193808 or 1703363853.5 (a Unix timestamp, possibly with a sub-second part) { - if (!tryReadDateTimeAsNumber(x, istr)) + if (!tryReadDateTimeAsNumber(x, istr, !settings.throwOnDateTimeOverflow())) return false; } @@ -234,11 +240,11 @@ void SerializationDateTime::deserializeTextJSON(IColumn & column, ReadBuffer & i } else if (settings.read_datetime_number_as_raw_value) /// Legacy: the raw value (seconds). { - readDateTimeAsRawValue(x, istr); + readDateTimeAsRawValue(x, istr, !settings.throwOnDateTimeOverflow()); } else { - readDateTimeAsNumber(x, istr); + readDateTimeAsNumber(x, istr, !settings.throwOnDateTimeOverflow()); } assert_cast(column).getData().push_back(static_cast(x)); @@ -254,12 +260,12 @@ bool SerializationDateTime::tryDeserializeTextJSON(IColumn & column, ReadBuffer } else if (settings.read_datetime_number_as_raw_value) /// Legacy: the raw value (seconds). { - if (!tryReadDateTimeAsRawValue(x, istr)) + if (!tryReadDateTimeAsRawValue(x, istr, !settings.throwOnDateTimeOverflow())) return false; } else { - if (!tryReadDateTimeAsNumber(x, istr)) + if (!tryReadDateTimeAsNumber(x, istr, !settings.throwOnDateTimeOverflow())) return false; } diff --git a/src/Formats/FormatSettings.h b/src/Formats/FormatSettings.h index 42f62e34fb22..99385f454d62 100644 --- a/src/Formats/FormatSettings.h +++ b/src/Formats/FormatSettings.h @@ -133,6 +133,8 @@ struct FormatSettings DateTimeOverflowBehavior date_time_overflow_behavior = DateTimeOverflowBehavior::Ignore; + bool throwOnDateTimeOverflow() const { return date_time_overflow_behavior == DateTimeOverflowBehavior::Throw; } + bool input_format_ipv4_default_on_conversion_error = false; bool input_format_ipv6_default_on_conversion_error = false; bool check_conversion_from_numbers_to_enum = true; diff --git a/src/Formats/JSONExtractTree.cpp b/src/Formats/JSONExtractTree.cpp index c8e2d6c27bc0..c00dac73ec2b 100644 --- a/src/Formats/JSONExtractTree.cpp +++ b/src/Formats/JSONExtractTree.cpp @@ -701,7 +701,7 @@ class DateNode : public JSONExtractTreeNode auto data = element.getString(); ReadBufferFromMemory buf(data); DateType date; - if (!tryReadDateText(date, buf) || !buf.eof()) + if (!tryReadDateText(date, buf, DateLUT::instance(), nullptr, !format_settings.throwOnDateTimeOverflow()) || !buf.eof()) { error = fmt::format("cannot parse Date value here: {}", data); return false; @@ -737,7 +737,7 @@ class DateTimeNode : public JSONExtractTreeNode, public TimezoneMixi time_t value = 0; if (element.isString()) { - if (!tryParse(value, element.getString(), format_settings.date_time_input_format)) + if (!tryParse(value, element.getString(), format_settings.date_time_input_format, !format_settings.throwOnDateTimeOverflow())) { error = fmt::format("cannot parse DateTime value here: {}", element.getString()); return false; @@ -756,12 +756,22 @@ class DateTimeNode : public JSONExtractTreeNode, public TimezoneMixi return false; } value = element.getInt64(); + if (format_settings.throwOnDateTimeOverflow() && (value < 0 || value > 0xFFFFFFFF)) + { + error = fmt::format("value {} is out of bounds of type DateTime", value); + return false; + } } else { /// Clamp in the unsigned domain before narrowing to time_t, /// because values above INT64_MAX would wrap to negative on cast. UInt64 raw = element.getUInt64(); + if (format_settings.throwOnDateTimeOverflow() && raw > 0xFFFFFFFF) + { + error = fmt::format("value {} is out of bounds of type DateTime", raw); + return false; + } value = static_cast(std::min(raw, UInt64(0xFFFFFFFF))); } } @@ -774,7 +784,7 @@ class DateTimeNode : public JSONExtractTreeNode, public TimezoneMixi /// exactly can cross the second boundary (`1703363853.9999999` arrives here as `1703363854.0`). String str_value = jsonElementToString(element, format_settings); ReadBufferFromMemory buf(str_value); - if (!tryReadDateTimeAsNumber(value, buf) || !buf.eof()) + if (!tryReadDateTimeAsNumber(value, buf, !format_settings.throwOnDateTimeOverflow()) || !buf.eof()) { error = fmt::format("cannot read DateTime value from JSON element: {}", str_value); return false; @@ -790,21 +800,22 @@ class DateTimeNode : public JSONExtractTreeNode, public TimezoneMixi return true; } - bool tryParse(time_t & value, std::string_view data, FormatSettings::DateTimeInputFormat date_time_input_format) const + bool tryParse(time_t & value, std::string_view data, FormatSettings::DateTimeInputFormat date_time_input_format, bool saturate_on_overflow) const { + const auto overflow = saturate_on_overflow ? DateTimeOverflow::Saturate : DateTimeOverflow::Report; ReadBufferFromMemory buf(data); switch (date_time_input_format) { case FormatSettings::DateTimeInputFormat::Basic: - if (tryReadDateTimeText(value, buf, time_zone) && buf.eof()) + if (tryReadDateTimeText(value, buf, time_zone, nullptr, nullptr, saturate_on_overflow) && buf.eof()) return true; break; case FormatSettings::DateTimeInputFormat::BestEffort: - if (tryParseDateTimeBestEffort(value, buf, time_zone, utc_time_zone) && buf.eof()) + if (tryParseDateTimeBestEffort(value, buf, time_zone, utc_time_zone, overflow) && buf.eof()) return true; break; case FormatSettings::DateTimeInputFormat::BestEffortUS: - if (tryParseDateTimeBestEffortUS(value, buf, time_zone, utc_time_zone) && buf.eof()) + if (tryParseDateTimeBestEffortUS(value, buf, time_zone, utc_time_zone, overflow) && buf.eof()) return true; break; } diff --git a/src/Functions/FunctionsConversion.h b/src/Functions/FunctionsConversion.h index e3158387728c..79d6e6e8fe97 100644 --- a/src/Functions/FunctionsConversion.h +++ b/src/Functions/FunctionsConversion.h @@ -1084,7 +1084,7 @@ inline void convertFromTime(DataTypeTime::FieldType & x, time_t & /** Conversion of strings to numbers, dates, datetimes: through parsing. */ template -void parseImpl(typename DataType::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool precise_float_parsing) +void parseImpl(typename DataType::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool precise_float_parsing, bool) { if constexpr (is_floating_point) { @@ -1098,33 +1098,33 @@ void parseImpl(typename DataType::FieldType & x, ReadBuffer & rb, const DateLUTI } template <> -inline void parseImpl(DataTypeDate::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline void parseImpl(DataTypeDate::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool saturate_on_overflow) { DayNum tmp(0); - readDateText(tmp, rb, *time_zone); + readDateText(tmp, rb, *time_zone, saturate_on_overflow); x = tmp; } template <> -inline void parseImpl(DataTypeDate32::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline void parseImpl(DataTypeDate32::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool saturate_on_overflow) { ExtendedDayNum tmp(0); - readDateText(tmp, rb, *time_zone); + readDateText(tmp, rb, *time_zone, saturate_on_overflow); x = tmp; } // NOTE: no need of extra overload of DateTime64, since readDateTimeText64 has different signature and that case is explicitly handled in the calling code. template <> -inline void parseImpl(DataTypeDateTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline void parseImpl(DataTypeDateTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool saturate_on_overflow) { time_t time = 0; - readDateTimeText(time, rb, *time_zone); + readDateTimeText(time, rb, *time_zone, saturate_on_overflow); convertFromTime(x, time); } template <> -inline void parseImpl(DataTypeTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline void parseImpl(DataTypeTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool) { time_t time = 0; readTimeText(time, rb, *time_zone); @@ -1132,7 +1132,7 @@ inline void parseImpl(DataTypeTime::FieldType & x, ReadBuffer & rb } template <> -inline void parseImpl(DataTypeUUID::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool) +inline void parseImpl(DataTypeUUID::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool, bool) { UUID tmp; readUUIDText(tmp, rb); @@ -1140,7 +1140,7 @@ inline void parseImpl(DataTypeUUID::FieldType & x, ReadBuffer & rb } template <> -inline void parseImpl(DataTypeIPv4::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool) +inline void parseImpl(DataTypeIPv4::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool, bool) { IPv4 tmp; readIPv4Text(tmp, rb); @@ -1148,7 +1148,7 @@ inline void parseImpl(DataTypeIPv4::FieldType & x, ReadBuffer & rb } template <> -inline void parseImpl(DataTypeIPv6::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool) +inline void parseImpl(DataTypeIPv6::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool, bool) { IPv6 tmp; readIPv6Text(tmp, rb); @@ -1156,7 +1156,7 @@ inline void parseImpl(DataTypeIPv6::FieldType & x, ReadBuffer & rb } template -bool tryParseImpl(typename DataType::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool precise_float_parsing) +bool tryParseImpl(typename DataType::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool precise_float_parsing, bool) { if constexpr (is_floating_point) { @@ -1170,37 +1170,37 @@ bool tryParseImpl(typename DataType::FieldType & x, ReadBuffer & rb, const DateL } template <> -inline bool tryParseImpl(DataTypeDate::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline bool tryParseImpl(DataTypeDate::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool saturate_on_overflow) { DayNum tmp(0); - if (!tryReadDateText(tmp, rb, *time_zone)) + if (!tryReadDateText(tmp, rb, *time_zone, nullptr, saturate_on_overflow)) return false; x = tmp; return true; } template <> -inline bool tryParseImpl(DataTypeDate32::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline bool tryParseImpl(DataTypeDate32::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool saturate_on_overflow) { ExtendedDayNum tmp(0); - if (!tryReadDateText(tmp, rb, *time_zone)) + if (!tryReadDateText(tmp, rb, *time_zone, nullptr, saturate_on_overflow)) return false; x = tmp; return true; } template <> -inline bool tryParseImpl(DataTypeDateTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +inline bool tryParseImpl(DataTypeDateTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool saturate_on_overflow) { time_t time = 0; - if (!tryReadDateTimeText(time, rb, *time_zone)) + if (!tryReadDateTimeText(time, rb, *time_zone, nullptr, nullptr, saturate_on_overflow)) return false; convertFromTime(x, time); return true; } template <> -[[maybe_unused]]inline bool tryParseImpl(DataTypeTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool) +[[maybe_unused]]inline bool tryParseImpl(DataTypeTime::FieldType & x, ReadBuffer & rb, const DateLUTImpl * time_zone, bool, bool) { time_t time = 0; if (!tryReadTimeText(time, rb, *time_zone)) @@ -1210,7 +1210,7 @@ template <> } template <> -inline bool tryParseImpl(DataTypeUUID::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool) +inline bool tryParseImpl(DataTypeUUID::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool, bool) { UUID tmp; if (!tryReadUUIDText(tmp, rb)) @@ -1221,7 +1221,7 @@ inline bool tryParseImpl(DataTypeUUID::FieldType & x, ReadBuffer & } template <> -inline bool tryParseImpl(DataTypeIPv4::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool) +inline bool tryParseImpl(DataTypeIPv4::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool, bool) { IPv4 tmp; if (!tryReadIPv4Text(tmp, rb)) @@ -1232,7 +1232,7 @@ inline bool tryParseImpl(DataTypeIPv4::FieldType & x, ReadBuffer & } template <> -inline bool tryParseImpl(DataTypeIPv6::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool) +inline bool tryParseImpl(DataTypeIPv6::FieldType & x, ReadBuffer & rb, const DateLUTImpl *, bool, bool) { IPv6 tmp; if (!tryReadIPv6Text(tmp, rb)) @@ -1365,6 +1365,11 @@ struct ConvertThroughParsing utc_time_zone = &DateLUT::instance("UTC"); } + /// Nothing to wrap around when parsing text, so `ignore` behaves like `saturate` + const bool saturate_on_overflow [[maybe_unused]] + = settings.date_time_overflow_behavior != FormatSettings::DateTimeOverflowBehavior::Throw; + const auto overflow [[maybe_unused]] = saturate_on_overflow ? DateTimeOverflow::Saturate : DateTimeOverflow::Report; + const IColumn * col_from = arguments[0].column.get(); const ColumnString * col_from_string = checkAndGetColumn(col_from); const ColumnFixedString * col_from_fixed_string = checkAndGetColumn(col_from); @@ -1459,7 +1464,7 @@ struct ConvertThroughParsing else { time_t res = 0; - parseDateTimeBestEffort(res, read_buffer, *local_time_zone, *utc_time_zone); + parseDateTimeBestEffort(res, read_buffer, *local_time_zone, *utc_time_zone, overflow); convertFromTime(vec_to[i], res); } } @@ -1486,7 +1491,7 @@ struct ConvertThroughParsing else { time_t res = 0; - parseDateTimeBestEffortUS(res, read_buffer, *local_time_zone, *utc_time_zone); + parseDateTimeBestEffortUS(res, read_buffer, *local_time_zone, *utc_time_zone, overflow); convertFromTime(vec_to[i], res); } } @@ -1532,11 +1537,11 @@ struct ConvertThroughParsing } if constexpr (std::is_same_v) { - if (!tryParseImpl(vec_to[i], read_buffer, local_time_zone, settings.precise_float_parsing)) + if (!tryParseImpl(vec_to[i], read_buffer, local_time_zone, settings.precise_float_parsing, saturate_on_overflow)) throw Exception(ErrorCodes::CANNOT_PARSE_TEXT, "Cannot parse string to type {}", TypeName); } else - parseImpl(vec_to[i], read_buffer, local_time_zone, settings.precise_float_parsing); + parseImpl(vec_to[i], read_buffer, local_time_zone, settings.precise_float_parsing, saturate_on_overflow); } while (false); } } @@ -1571,7 +1576,7 @@ struct ConvertThroughParsing else { time_t res = 0; - parsed = tryParseDateTimeBestEffort(res, read_buffer, *local_time_zone, *utc_time_zone); + parsed = tryParseDateTimeBestEffort(res, read_buffer, *local_time_zone, *utc_time_zone, overflow); convertFromTime(vec_to[i],res); } } @@ -1598,7 +1603,7 @@ struct ConvertThroughParsing else { time_t res = 0; - parsed = tryParseDateTimeBestEffortUS(res, read_buffer, *local_time_zone, *utc_time_zone); + parsed = tryParseDateTimeBestEffortUS(res, read_buffer, *local_time_zone, *utc_time_zone, overflow); convertFromTime(vec_to[i],res); } } @@ -1631,7 +1636,7 @@ struct ConvertThroughParsing } else { - parsed = tryParseImpl(vec_to[i], read_buffer, local_time_zone, settings.precise_float_parsing); + parsed = tryParseImpl(vec_to[i], read_buffer, local_time_zone, settings.precise_float_parsing, saturate_on_overflow); } } diff --git a/src/IO/ReadHelpers.cpp b/src/IO/ReadHelpers.cpp index 7d0c89e43caa..8526315701bf 100644 --- a/src/IO/ReadHelpers.cpp +++ b/src/IO/ReadHelpers.cpp @@ -47,6 +47,7 @@ namespace ErrorCodes extern const int TOO_DEEP_RECURSION; extern const int TOO_LARGE_STRING_SIZE; extern const int SYNTAX_ERROR; + extern const int VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE; } /// Converts num_bytes hex-encoded bytes from src to dst in a single pass, folding validity into @@ -1695,8 +1696,9 @@ ReturnType readDateTimeTextFallback( second = (s[6] - '0') * 10 + (s[7] - '0'); } - if constexpr (throw_exception) + if (saturate_on_overflow) { + /// Use saturating version - makeDateTime saturates out-of-range years if (unlikely(year == 0)) datetime = 0; else @@ -1704,29 +1706,28 @@ ReturnType readDateTimeTextFallback( } else { - if (saturate_on_overflow) + /// Use non-saturating version - report out-of-range values instead of clamping them + auto datetime_maybe = tryToMakeDateTime(date_lut, year, month, day, hour, minute, second); + if (!datetime_maybe) { - /// Use saturating version - makeDateTime saturates out-of-range years - if (unlikely(year == 0)) - datetime = 0; + if constexpr (throw_exception) + throw Exception(ErrorCodes::CANNOT_PARSE_DATETIME, "Cannot parse DateTime"); else - datetime = makeDateTime(date_lut, year, month, day, hour, minute, second); - } - else - { - /// Use non-saturating version - return false for out-of-range values - auto datetime_maybe = tryToMakeDateTime(date_lut, year, month, day, hour, minute, second); - if (!datetime_maybe) return false; + } - if constexpr (!dt64_mode) + if constexpr (!dt64_mode) + { + if (*datetime_maybe < 0 || *datetime_maybe > static_cast(UINT32_MAX)) { - if (*datetime_maybe < 0 || *datetime_maybe > static_cast(UINT32_MAX)) + if constexpr (throw_exception) + throw Exception(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value {} is out of bounds of type DateTime", *datetime_maybe); + else return false; } - - datetime = *datetime_maybe; } + + datetime = *datetime_maybe; } } else @@ -1758,6 +1759,8 @@ ReturnType readDateTimeTextFallback( return false; } + return checkParsedDateTimeRange(datetime, saturate_on_overflow); + } return ReturnType(true); @@ -2595,6 +2598,15 @@ namespace /// Scale `value` to whole seconds and clamp it to the `DateTime` range. The multiplication is bound-checked /// with truncating division rather than `common::mulOverflow`, a no-op stub for big-int types. +/// Whether `value` scaled to whole seconds fits the DateTime range, using the same truncating bound +bool datetimeSecondsInRange(Int128 value, UInt32 unread_scale) +{ + static constexpr Int128 max_seconds = 0xFFFFFFFF; + if (value < 0) + return false; + return value <= max_seconds / DecimalUtils::scaleMultiplier(unread_scale); +} + time_t datetimeSecondsFromNumber(Int128 value, UInt32 unread_scale) { static constexpr Int128 max_seconds = 0xFFFFFFFF; @@ -2619,7 +2631,7 @@ bool datetime64TicksFromNumber(DateTime64 & x, Int128 value, UInt32 unread_scale } template -ReturnType readDateTimeAsNumberImpl(time_t & x, ReadBuffer & buf) +ReturnType readDateTimeAsNumberImpl(time_t & x, ReadBuffer & buf, bool saturate_on_overflow) { static constexpr bool throw_exception = std::is_same_v; Decimal128 tmp; @@ -2639,12 +2651,20 @@ ReturnType readDateTimeAsNumberImpl(time_t & x, ReadBuffer & buf) else return ReturnType(false); } + if (!saturate_on_overflow && !datetimeSecondsInRange(tmp.value, unread_scale)) + { + if constexpr (throw_exception) + throw Exception(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value is out of bounds of type DateTime"); + else + return ReturnType(false); + } + x = datetimeSecondsFromNumber(tmp.value, unread_scale); return ReturnType(true); } template -ReturnType readDateTimeAsRawValueImpl(time_t & x, ReadBuffer & buf) +ReturnType readDateTimeAsRawValueImpl(time_t & x, ReadBuffer & buf, bool saturate_on_overflow) { static constexpr bool throw_exception = std::is_same_v; /// Saturating 128-bit read: a plain `readIntText` does not check overflow, so an out-of-range value would @@ -2655,6 +2675,14 @@ ReturnType readDateTimeAsRawValueImpl(time_t & x, ReadBuffer & buf) else if (!readIntText128Saturating(tmp, buf)) return ReturnType(false); + if (!saturate_on_overflow && !datetimeSecondsInRange(tmp, 0)) + { + if constexpr (throw_exception) + throw Exception(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value is out of bounds of type DateTime"); + else + return ReturnType(false); + } + x = datetimeSecondsFromNumber(tmp, 0); return ReturnType(true); } @@ -2685,6 +2713,7 @@ ReturnType readDateTime64AsNumberImpl(DateTime64 & x, UInt32 scale, ReadBuffer & else return ReturnType(false); } + return ReturnType(true); } @@ -2705,15 +2734,16 @@ ReturnType readDateTime64AsRawValueImpl(DateTime64 & x, ReadBuffer & buf) else return ReturnType(false); } + return ReturnType(true); } } -void readDateTimeAsNumber(time_t & x, ReadBuffer & buf) { readDateTimeAsNumberImpl(x, buf); } -bool tryReadDateTimeAsNumber(time_t & x, ReadBuffer & buf) { return readDateTimeAsNumberImpl(x, buf); } -void readDateTimeAsRawValue(time_t & x, ReadBuffer & buf) { readDateTimeAsRawValueImpl(x, buf); } -bool tryReadDateTimeAsRawValue(time_t & x, ReadBuffer & buf) { return readDateTimeAsRawValueImpl(x, buf); } +void readDateTimeAsNumber(time_t & x, ReadBuffer & buf, bool saturate_on_overflow) { readDateTimeAsNumberImpl(x, buf, saturate_on_overflow); } +bool tryReadDateTimeAsNumber(time_t & x, ReadBuffer & buf, bool saturate_on_overflow) { return readDateTimeAsNumberImpl(x, buf, saturate_on_overflow); } +void readDateTimeAsRawValue(time_t & x, ReadBuffer & buf, bool saturate_on_overflow) { readDateTimeAsRawValueImpl(x, buf, saturate_on_overflow); } +bool tryReadDateTimeAsRawValue(time_t & x, ReadBuffer & buf, bool saturate_on_overflow) { return readDateTimeAsRawValueImpl(x, buf, saturate_on_overflow); } void readDateTime64AsNumber(DateTime64 & x, UInt32 scale, ReadBuffer & buf) { readDateTime64AsNumberImpl(x, scale, buf); } bool tryReadDateTime64AsNumber(DateTime64 & x, UInt32 scale, ReadBuffer & buf) { return readDateTime64AsNumberImpl(x, scale, buf); } diff --git a/src/IO/ReadHelpers.h b/src/IO/ReadHelpers.h index 2d216684a75b..173ebc3f9d6f 100644 --- a/src/IO/ReadHelpers.h +++ b/src/IO/ReadHelpers.h @@ -57,6 +57,7 @@ namespace ErrorCodes extern const int TOO_LARGE_STRING_SIZE; extern const int TOO_LARGE_ARRAY_SIZE; extern const int SIZE_OF_FIXED_STRING_DOESNT_MATCH; + extern const int VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE; } /// Helper functions for formatted input. @@ -645,39 +646,40 @@ inline ReturnType readDateTextImpl(DayNum & date, ReadBuffer & buf, const DateLU LocalDate local_date; if constexpr (throw_exception) - { readDateTextImpl(local_date, buf, allowed_delimiters); + else if (!readDateTextImpl(local_date, buf, allowed_delimiters)) + return false; + + if (saturate_on_overflow) + { + /// Use saturating versions - makeDayNum saturates out-of-range years, convertToDayNum saturates to 0 or 0xFFFF ExtendedDayNum ret = makeDayNum(date_lut, local_date.year(), local_date.month(), local_date.day()); convertToDayNum(date, ret); + return ReturnType(true); } - else + + auto ret = tryToMakeDayNum(date_lut, local_date.year(), local_date.month(), local_date.day()); + if (!ret) { - if (!readDateTextImpl(local_date, buf, allowed_delimiters)) + if constexpr (throw_exception) + throw Exception(ErrorCodes::CANNOT_PARSE_DATE, "Cannot parse date"); + else return false; + } - if (saturate_on_overflow) - { - /// Use saturating versions - makeDayNum saturates out-of-range years, convertToDayNum saturates to 0 or 0xFFFF - ExtendedDayNum ret = makeDayNum(date_lut, local_date.year(), local_date.month(), local_date.day()); - convertToDayNum(date, ret); - } + if (!tryToConvertToDayNum(date, *ret)) + { + if constexpr (throw_exception) + throw Exception(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value {} is out of bounds of type Date", ret->toUnderType()); else - { - /// Use non-saturating versions - return false for out-of-range values - auto ret = tryToMakeDayNum(date_lut, local_date.year(), local_date.month(), local_date.day()); - if (!ret) - return false; - - if (!tryToConvertToDayNum(date, *ret)) - return false; - } - - return true; + return false; } + + return ReturnType(true); } template -inline ReturnType readDateTextImpl(ExtendedDayNum & date, ReadBuffer & buf, const DateLUTImpl & date_lut, const char * allowed_delimiters = nullptr) +inline ReturnType readDateTextImpl(ExtendedDayNum & date, ReadBuffer & buf, const DateLUTImpl & date_lut, const char * allowed_delimiters = nullptr, bool saturate_on_overflow = true) { static constexpr bool throw_exception = std::is_same_v; @@ -688,6 +690,22 @@ inline ReturnType readDateTextImpl(ExtendedDayNum & date, ReadBuffer & buf, cons else if (!readDateTextImpl(local_date, buf, allowed_delimiters)) return false; + if (!saturate_on_overflow) + { + /// Every four-digit year fits into Date32, so only a calendar-invalid date can fail here + auto ret = tryToMakeDayNum(date_lut, local_date.year(), local_date.month(), local_date.day()); + if (!ret) + { + if constexpr (throw_exception) + throw Exception(ErrorCodes::CANNOT_PARSE_DATE, "Cannot parse date"); + else + return false; + } + + date = *ret; + return ReturnType(true); + } + /// A calendar-invalid date (e.g. month 13) yields 1900-01-01 (-getDayNumOffsetEpoch(), -25567) for Date32 and 1970-01-01 for Date. date = makeDayNum(date_lut, local_date.year(), local_date.month(), local_date.day(), -static_cast(getDayNumOffsetEpoch())); return ReturnType(true); @@ -699,14 +717,14 @@ inline void readDateText(LocalDate & date, ReadBuffer & buf) readDateTextImpl(date, buf); } -inline void readDateText(DayNum & date, ReadBuffer & buf, const DateLUTImpl & date_lut = DateLUT::instance()) +inline void readDateText(DayNum & date, ReadBuffer & buf, const DateLUTImpl & date_lut = DateLUT::instance(), bool saturate_on_overflow = true) { - readDateTextImpl(date, buf, date_lut); + readDateTextImpl(date, buf, date_lut, nullptr, saturate_on_overflow); } -inline void readDateText(ExtendedDayNum & date, ReadBuffer & buf, const DateLUTImpl & date_lut = DateLUT::instance()) +inline void readDateText(ExtendedDayNum & date, ReadBuffer & buf, const DateLUTImpl & date_lut = DateLUT::instance(), bool saturate_on_overflow = true) { - readDateTextImpl(date, buf, date_lut); + readDateTextImpl(date, buf, date_lut, nullptr, saturate_on_overflow); } inline bool tryReadDateText(LocalDate & date, ReadBuffer & buf, const char * allowed_delimiters = nullptr) @@ -719,9 +737,9 @@ inline bool tryReadDateText(DayNum & date, ReadBuffer & buf, const DateLUTImpl & return readDateTextImpl(date, buf, time_zone, allowed_delimiters, saturate_on_overflow); } -inline bool tryReadDateText(ExtendedDayNum & date, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance(), const char * allowed_delimiters = nullptr) +inline bool tryReadDateText(ExtendedDayNum & date, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance(), const char * allowed_delimiters = nullptr, bool saturate_on_overflow = true) { - return readDateTextImpl(date, buf, time_zone, allowed_delimiters); + return readDateTextImpl(date, buf, time_zone, allowed_delimiters, saturate_on_overflow); } UUID parseUUID(std::span src); @@ -856,6 +874,23 @@ inline T parseFromStringWithoutAssertEOF(std::string_view str) template ReturnType readDateTimeTextFallback(time_t & datetime, ReadBuffer & buf, const DateLUTImpl & date_lut, const char * allowed_date_delimiters = nullptr, const char * allowed_time_delimiters = nullptr, bool saturate_on_overflow = true); +/// A digit-only timestamp is read as a plain integer, so its range has to be checked separately +template +inline ReturnType checkParsedDateTimeRange(time_t datetime [[maybe_unused]], bool saturate_on_overflow [[maybe_unused]]) +{ + if constexpr (!dt64_mode) + { + if (!saturate_on_overflow && (datetime < 0 || datetime > static_cast(UINT32_MAX))) + { + if constexpr (std::is_same_v) + throw Exception(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value {} is out of bounds of type DateTime", datetime); + else + return false; + } + } + return ReturnType(true); +} + template ReturnType readTimeTextFallback(time_t & time, ReadBuffer & buf, const DateLUTImpl & date_lut, const char * allowed_date_delimiters = nullptr, const char * allowed_time_delimiters = nullptr); @@ -928,8 +963,9 @@ inline ReturnType readDateTimeTextImpl(time_t & datetime, ReadBuffer & buf, cons second = (s[17] - '0') * 10 + (s[18] - '0'); } - if constexpr (throw_exception) + if (saturate_on_overflow) { + /// Use saturating version - makeDateTime saturates out-of-range years if (unlikely(year == 0)) datetime = 0; else @@ -937,30 +973,29 @@ inline ReturnType readDateTimeTextImpl(time_t & datetime, ReadBuffer & buf, cons } else { - if (saturate_on_overflow) + /// Use non-saturating version - report out-of-range values instead of clamping them + auto datetime_maybe = tryToMakeDateTime(date_lut, year, month, day, hour, minute, second); + if (!datetime_maybe) { - /// Use saturating version - makeDateTime saturates out-of-range years - if (unlikely(year == 0)) - datetime = 0; + if constexpr (throw_exception) + throw Exception(ErrorCodes::CANNOT_PARSE_DATETIME, "Cannot parse datetime"); else - datetime = makeDateTime(date_lut, year, month, day, hour, minute, second); - } - else - { - /// Use non-saturating version - return false for out-of-range values - auto datetime_maybe = tryToMakeDateTime(date_lut, year, month, day, hour, minute, second); - if (!datetime_maybe) return false; + } - /// For usual DateTime check if value is within supported range - if constexpr (!dt64_mode) + /// For usual DateTime check if value is within supported range + if constexpr (!dt64_mode) + { + if (*datetime_maybe < 0 || *datetime_maybe > static_cast(UINT32_MAX)) { - if (*datetime_maybe < 0 || *datetime_maybe > static_cast(UINT32_MAX)) + if constexpr (throw_exception) + throw Exception(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value {} is out of bounds of type DateTime", *datetime_maybe); + else return false; } - - datetime = *datetime_maybe; } + + datetime = *datetime_maybe; } if (dt_long) @@ -971,7 +1006,12 @@ inline ReturnType readDateTimeTextImpl(time_t & datetime, ReadBuffer & buf, cons return ReturnType(true); } /// Why not readIntTextUnsafe? Because for needs of AdFox, parsing of unix timestamp with leading zeros is supported: 000...NNNN. - return readIntTextImpl(datetime, buf); + if constexpr (throw_exception) + readIntTextImpl(datetime, buf); + else if (!readIntTextImpl(datetime, buf)) + return false; + + return checkParsedDateTimeRange(datetime, saturate_on_overflow); } return readDateTimeTextFallback(datetime, buf, date_lut, allowed_date_delimiters, allowed_time_delimiters, saturate_on_overflow); } @@ -1396,9 +1436,9 @@ inline ReturnType readTimeTextImpl(Time64 & time64, UInt32 scale, ReadBuffer & b return ReturnType(is_ok); } -inline void readDateTimeText(time_t & datetime, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance()) +inline void readDateTimeText(time_t & datetime, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance(), bool saturate_on_overflow = true) { - readDateTimeTextImpl(datetime, buf, time_zone); + readDateTimeTextImpl(datetime, buf, time_zone, nullptr, nullptr, saturate_on_overflow); } inline void readTimeText(time_t & datetime, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance()) @@ -1442,10 +1482,10 @@ inline bool tryReadTime64Text(Time64 & time64, UInt32 scale, ReadBuffer & buf, c /// `toDateTime64` and the `Values` format. Parsing stops at the first character that is not part of the /// number (e.g. the `,` or `}` following the value in JSON). The `AsRawValue` variants implement the legacy /// behavior, where the number is the raw underlying value. -void readDateTimeAsNumber(time_t & x, ReadBuffer & buf); -bool tryReadDateTimeAsNumber(time_t & x, ReadBuffer & buf); -void readDateTimeAsRawValue(time_t & x, ReadBuffer & buf); -bool tryReadDateTimeAsRawValue(time_t & x, ReadBuffer & buf); +void readDateTimeAsNumber(time_t & x, ReadBuffer & buf, bool saturate_on_overflow = true); +bool tryReadDateTimeAsNumber(time_t & x, ReadBuffer & buf, bool saturate_on_overflow = true); +void readDateTimeAsRawValue(time_t & x, ReadBuffer & buf, bool saturate_on_overflow = true); +bool tryReadDateTimeAsRawValue(time_t & x, ReadBuffer & buf, bool saturate_on_overflow = true); void readDateTime64AsNumber(DateTime64 & x, UInt32 scale, ReadBuffer & buf); bool tryReadDateTime64AsNumber(DateTime64 & x, UInt32 scale, ReadBuffer & buf); @@ -1667,8 +1707,8 @@ inline void readText(T & x, ReadBuffer & buf) { readFloatTextPrecise(x, buf); } inline void readText(String & x, ReadBuffer & buf) { readEscapedString(x, buf); } -inline void readText(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance()) { readDateText(x, buf, time_zone); } -inline bool tryReadText(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance()) { return tryReadDateText(x, buf, time_zone); } +inline void readText(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance(), bool saturate_on_overflow = true) { readDateText(x, buf, time_zone, saturate_on_overflow); } +inline bool tryReadText(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone = DateLUT::instance(), bool saturate_on_overflow = true) { return tryReadDateText(x, buf, time_zone, nullptr, saturate_on_overflow); } inline void readText(LocalDate & x, ReadBuffer & buf) { readDateText(x, buf); } inline bool tryReadText(LocalDate & x, ReadBuffer & buf) { return tryReadDateText(x, buf); } @@ -1794,7 +1834,7 @@ inline ReturnType readCSVSimple(T & x, ReadBuffer & buf) // standalone overload for dates: to avoid instantiating DateLUTs while parsing other types template -inline ReturnType readCSVSimple(T & x, ReadBuffer & buf, const DateLUTImpl & time_zone) +inline ReturnType readCSVSimple(T & x, ReadBuffer & buf, const DateLUTImpl & time_zone, bool saturate_on_overflow = true) { static constexpr bool throw_exception = std::is_same_v; @@ -1811,8 +1851,8 @@ inline ReturnType readCSVSimple(T & x, ReadBuffer & buf, const DateLUTImpl & tim ++buf.position(); if constexpr (throw_exception) - readText(x, buf, time_zone); - else if (!tryReadText(x, buf, time_zone)) + readText(x, buf, time_zone, saturate_on_overflow); + else if (!tryReadText(x, buf, time_zone, saturate_on_overflow)) return ReturnType(false); if (maybe_quote == '\'' || maybe_quote == '\"') @@ -1853,8 +1893,8 @@ inline bool tryReadCSV(LocalDate & x, ReadBuffer & buf) { return readCSVSimple(x, buf); } -inline void readCSV(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone) { readCSVSimple(x, buf, time_zone); } -inline bool tryReadCSV(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone) { return readCSVSimple(x, buf, time_zone); } +inline void readCSV(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone, bool saturate_on_overflow = true) { readCSVSimple(x, buf, time_zone, saturate_on_overflow); } +inline bool tryReadCSV(DayNum & x, ReadBuffer & buf, const DateLUTImpl & time_zone, bool saturate_on_overflow = true) { return readCSVSimple(x, buf, time_zone, saturate_on_overflow); } inline void readCSV(LocalDateTime & x, ReadBuffer & buf) { readCSVSimple(x, buf); } inline bool tryReadCSV(LocalDateTime & x, ReadBuffer & buf) { return readCSVSimple(x, buf); } diff --git a/src/IO/parseDateTimeBestEffort.cpp b/src/IO/parseDateTimeBestEffort.cpp index 6a41e80d165d..1fb167e7267d 100644 --- a/src/IO/parseDateTimeBestEffort.cpp +++ b/src/IO/parseDateTimeBestEffort.cpp @@ -17,6 +17,7 @@ namespace ErrorCodes { extern const int LOGICAL_ERROR; extern const int CANNOT_PARSE_DATETIME; +extern const int VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE; } @@ -99,7 +100,8 @@ ReturnType parseDateTimeBestEffortImpl( const DateLUTImpl & utc_time_zone, DateTimeSubsecondPart * fractional, const char * allowed_date_delimiters = nullptr, - bool * has_explicit_zero_year = nullptr) + bool * has_explicit_zero_year = nullptr, + DateTimeOverflow overflow = DateTimeOverflow::Saturate) { auto on_error = [&]( int error_code [[maybe_unused]], @@ -323,7 +325,7 @@ ReturnType parseDateTimeBestEffortImpl( /// Fractional part is not allowed. return on_error(ErrorCodes::CANNOT_PARSE_DATETIME, "Cannot read DateTime: unexpected fractional part"); } - return ReturnType(true); + return checkParsedDateTimeRange(res, overflow == DateTimeOverflow::Saturate); } if (num_digits == 16 && !year && !has_time) { @@ -339,7 +341,7 @@ ReturnType parseDateTimeBestEffortImpl( /// Fractional part is not allowed. return on_error(ErrorCodes::CANNOT_PARSE_DATETIME, "Cannot read DateTime: unexpected fractional part"); } - return ReturnType(true); + return checkParsedDateTimeRange(res, overflow == DateTimeOverflow::Saturate); } if (num_digits == 19 && !year && !has_time) { @@ -355,7 +357,7 @@ ReturnType parseDateTimeBestEffortImpl( /// Fractional part is not allowed. return on_error(ErrorCodes::CANNOT_PARSE_DATETIME, "Cannot read DateTime: unexpected fractional part"); } - return ReturnType(true); + return checkParsedDateTimeRange(res, overflow == DateTimeOverflow::Saturate); } if (num_digits == 10 && !year && !has_time) { @@ -374,7 +376,7 @@ ReturnType parseDateTimeBestEffortImpl( readDigits(digits, sizeof(digits), in))); readDecimalNumber(fractional->value, fractional->digits, digits); } - return ReturnType(true); + return checkParsedDateTimeRange(res, overflow == DateTimeOverflow::Saturate); } if (num_digits == 9 && !year && !has_time) { @@ -393,7 +395,7 @@ ReturnType parseDateTimeBestEffortImpl( readDigits(digits, sizeof(digits), in))); readDecimalNumber(fractional->value, fractional->digits, digits); } - return ReturnType(true); + return checkParsedDateTimeRange(res, overflow == DateTimeOverflow::Saturate); } if (num_digits == 14 && !year && !has_time) { @@ -935,6 +937,10 @@ ReturnType parseDateTimeBestEffortImpl( if (has_explicit_zero_year) *has_explicit_zero_year = zero_year_was_read; + /// Year 0000 is outside DateTime and the substitution below would hide that, which `throw` forbids + if (!is_64 && zero_year_was_read && overflow == DateTimeOverflow::Report) + return on_error(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Year 0000 is out of bounds of type DateTime"); + if constexpr (strict) return on_error(ErrorCodes::CANNOT_PARSE_DATETIME, "Cannot read DateTime: year is required"); @@ -996,7 +1002,8 @@ ReturnType parseDateTimeBestEffortImpl( } }; - if constexpr (!strict || std::is_same_v) + /// `strict` always range-checks, otherwise only when the caller asked not to saturate + if (overflow == DateTimeOverflow::Saturate && !(strict && std::is_same_v)) { if (has_time_zone_offset) { @@ -1008,52 +1015,33 @@ ReturnType parseDateTimeBestEffortImpl( res = local_time_zone.makeDateTime(year, month, day_of_month, hour, minute, second); } - if constexpr (std::is_same_v) - return true; + return ReturnType(true); } - else - { - if (has_time_zone_offset) - { - auto res_maybe = utc_time_zone.tryToMakeDateTime(year, month, day_of_month, hour, minute, second); - if (!res_maybe) - return false; - /// For usual DateTime check if value is within supported range - if constexpr (!is_64) - { - if (*res_maybe < 0 || *res_maybe > UINT32_MAX) - return false; - } - res = *res_maybe; - adjust_time_zone(); + const DateLUTImpl & time_zone = has_time_zone_offset ? utc_time_zone : local_time_zone; + auto res_maybe = time_zone.tryToMakeDateTime(year, month, day_of_month, hour, minute, second); + if (!res_maybe) + return on_error( + ErrorCodes::CANNOT_PARSE_DATETIME, + "Cannot read DateTime: unexpected date: {}-{}-{}", + year, + static_cast(month), + static_cast(day_of_month)); - /// After timezone adjustment, the value may have shifted outside the valid range. - /// For example, "2106-02-07 06:28:15-01:00" is within range before adjustment, - /// but after converting to UTC it exceeds UINT32_MAX. - if constexpr (!is_64) - { - if (res < 0 || static_cast(res) > UINT32_MAX) - return false; - } - } - else - { - auto res_maybe = local_time_zone.tryToMakeDateTime(year, month, day_of_month, hour, minute, second); - if (!res_maybe) - return false; + res = *res_maybe; - /// For usual DateTime check if value is within supported range - if constexpr (!is_64) - { - if (*res_maybe < 0 || *res_maybe > UINT32_MAX) - return false; - } - res = *res_maybe; - } + if (has_time_zone_offset) + adjust_time_zone(); - return true; + /// Only the adjusted value has to be in range: "2106-02-07 07:28:15+01:00" is past the maximum before the + /// offset is applied and is exactly the maximum after it, and the same holds at the lower bound. + if constexpr (!is_64) + { + if (res < 0 || res > UINT32_MAX) + return on_error(ErrorCodes::VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE, "Value {} is out of bounds of type DateTime", res); } + + return ReturnType(true); } template @@ -1094,7 +1082,12 @@ ReturnType parseDateTime64BestEffortImpl(DateTime64 & res, UInt32 scale, ReadBuf void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone) { - parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr); + parseDateTimeBestEffort(res, in, local_time_zone, utc_time_zone, DateTimeOverflow::Saturate); +} + +void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow) +{ + parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr, nullptr, nullptr, overflow); } void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, bool & has_explicit_zero_year) @@ -1103,19 +1096,19 @@ void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr, nullptr, &has_explicit_zero_year); } -void parseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone) +void parseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow) { - parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr); + parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr, nullptr, nullptr, overflow); } -bool tryParseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone) +bool tryParseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow) { - return parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr); + return parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr, nullptr, nullptr, overflow); } -bool tryParseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone) +bool tryParseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow) { - return parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr); + return parseDateTimeBestEffortImpl(res, in, local_time_zone, utc_time_zone, nullptr, nullptr, nullptr, overflow); } void parseDateTime64BestEffort(DateTime64 & res, UInt32 scale, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone) diff --git a/src/IO/parseDateTimeBestEffort.h b/src/IO/parseDateTimeBestEffort.h index c42243b1efc0..a09d44de1482 100644 --- a/src/IO/parseDateTimeBestEffort.h +++ b/src/IO/parseDateTimeBestEffort.h @@ -55,8 +55,14 @@ class ReadBuffer; * Mon/Tue/Wed/Thu/Fri/Sat/Sun - simply ignored. */ +/// Whether an out-of-range result saturates to the bounds of the target type or is reported as an error +enum class DateTimeOverflow : uint8_t { Saturate, Report }; + void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); +/// Not defaulted, because the 4-argument form above would then be ambiguous with this one +void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow); + /// The same, but additionally reports whether the input contained an explicitly written year of `0000`. /// Such a year cannot be represented: internally a year field of `0` means "the year is not specified", /// so it is silently replaced with the current (or previous) year, and the returned value is then not the @@ -66,9 +72,9 @@ void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & void parseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, bool & has_explicit_zero_year); void parseTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); -bool tryParseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); -void parseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); -bool tryParseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); +bool tryParseDateTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow = DateTimeOverflow::Saturate); +void parseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow = DateTimeOverflow::Saturate); +bool tryParseDateTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone, DateTimeOverflow overflow = DateTimeOverflow::Saturate); bool tryParseTimeBestEffort(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); void parseTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); bool tryParseTimeBestEffortUS(time_t & res, ReadBuffer & in, const DateLUTImpl & local_time_zone, const DateLUTImpl & utc_time_zone); diff --git a/tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.reference b/tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.reference new file mode 100644 index 000000000000..a19952f23e73 --- /dev/null +++ b/tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.reference @@ -0,0 +1,42 @@ +saturate +2149-06-06 1970-01-01 2106-02-07 06:28:15 1970-01-01 00:00:00 +2149-06-06 2149-06-06 2149-06-06 +2149-06-06 2149-06-06 +ignore +2149-06-06 1970-01-01 2106-02-07 06:28:15 1970-01-01 00:00:00 +throw +throw, in range +2149-06-06 1970-01-01 2106-02-07 06:28:15 1970-01-01 00:00:00 +2299-12-31 1900-01-01 2299-12-31 23:59:59.999 +throw, OrNull and OrZero still fall back +\N 1970-01-01 \N 1970-01-01 00:00:00 +throw, input formats +2149-06-06 +2106-02-07 06:28:15 +throw, tentative parsers must not accept a clamped value +2150-12-31 \N 2150-12-31 +2149-06-06 +throw, a digit-only timestamp is text too +2106-02-07 06:28:15 2023-11-14 22:13:20 \N 1970-01-01 00:00:00 +throw, Date32 rejects what it cannot represent instead of substituting a default +\N 2299-12-31 1900-01-01 +2000-13-01 +throw, an unquoted numeric token is checked too +2023-11-14 22:13:20 +2106-02-07 06:28:15 +2023-12-23 20:37:33 +throw, the last second keeps its fractional ticks +9999-12-31 23:59:59.500 +9999-12-31 23:59:59.999 +9999-12-31 23:59:59.500 +throw, the range is checked after the timezone offset is applied +2106-02-07 06:28:15 1970-01-01 00:00:00 +2106-02-07 06:28:15 2106-02-07 06:28:15 +throw, typed JSON columns are checked like declared ones +2149-06-06 +2299-12-31 23:59:59.999 +throw, JSONExtract keeps returning a default or NULL +1970-01-01 \N 2149-06-06 +throw, an explicitly written year 0000 is not silently replaced +\N 1970-01-01 00:00:00 +3 diff --git a/tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.sql b/tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.sql new file mode 100644 index 000000000000..95e7f651a503 --- /dev/null +++ b/tests/queries/0_stateless/05019_date_time_overflow_behavior_from_string.sql @@ -0,0 +1,105 @@ +-- date_time_overflow_behavior was ignored when the value came from text instead of a typed column +SET session_timezone = 'UTC'; + +SELECT 'saturate'; +SET date_time_overflow_behavior = 'saturate'; +SELECT toDate('9999-12-31'), toDate('1969-12-31'), toDateTime('2106-02-07 06:28:16'), toDateTime('1969-12-31 23:59:59'); +SELECT CAST('9999-12-31' AS Date), CAST(materialize('9999-12-31') AS Date), CAST('9999-12-31'::FixedString(10) AS Date); +SELECT toDateOrNull('9999-12-31'), toDateOrZero('9999-12-31'); + +SELECT 'ignore'; +SET date_time_overflow_behavior = 'ignore'; +SELECT toDate('9999-12-31'), toDate('1969-12-31'), toDateTime('2106-02-07 06:28:16'), toDateTime('1969-12-31 23:59:59'); + +SELECT 'throw'; +SET date_time_overflow_behavior = 'throw'; +SELECT toDate('2149-06-07'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT toDate('1969-12-31'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT toDateTime('2106-02-07 06:28:16'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT toDateTime('1969-12-31 23:59:59'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT CAST('2149-06-07' AS Date); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT CAST(materialize('2149-06-07') AS Date); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT CAST('2149-06-07'::FixedString(10) AS Date); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT CAST(materialize('2106-02-07 06:28:16') AS DateTime); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } + +SELECT 'throw, in range'; +SELECT toDate('2149-06-06'), toDate('1970-01-01'), toDateTime('2106-02-07 06:28:15'), toDateTime('1970-01-01 00:00:00'); +-- Date32 and DateTime64 are not covered yet: a high-scale DateTime64 text parse still raises DECIMAL_OVERFLOW +-- in every mode, because the tick range is only checked in DecimalUtils +SELECT toDate32('2299-12-31'), toDate32('1900-01-01'), toDateTime64('2299-12-31 23:59:59.999', 3); + +SELECT 'throw, OrNull and OrZero still fall back'; +SELECT toDateOrNull('2149-06-07'), toDateOrZero('2149-06-07'), toDateTimeOrNull('2106-02-07 06:28:16'), toDateTimeOrZero('1969-12-31 23:59:59'); + +SELECT 'throw, input formats'; +SELECT * FROM format(CSV, 'v Date', '2150-12-31'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(CSV, 'v Date', '1960-01-01'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(TSV, 'v Date', '2150-12-31'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(JSONEachRow, 'v Date', '{"v":"2150-12-31"}'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(CSV, 'v DateTime', '2106-02-07 06:28:16'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(TSV, 'v DateTime', '1960-01-01 00:00:00'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":"2106-02-07 06:28:16"}'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(CSV, 'v Date', '2149-06-06'); +SELECT * FROM format(TSV, 'v DateTime', '2106-02-07 06:28:15'); + +SELECT 'throw, tentative parsers must not accept a clamped value'; +SELECT v, variantElement(v, 'Date') AS d, variantElement(v, 'String') AS s +FROM format(CSV, 'v Variant(Date, String)', '2150-12-31') SETTINGS allow_experimental_variant_type = 1; +SELECT v FROM format(CSV, 'v Variant(Date, String)', '2149-06-06') SETTINGS allow_experimental_variant_type = 1; + +SELECT 'throw, a digit-only timestamp is text too'; +SELECT toDateTime('4294967296'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT CAST('4294967296' AS DateTime); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT CAST(materialize('4294967296') AS DateTime); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":"4294967296"}'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(Values, 'v DateTime', '(\'4294967296\')'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT toDateTime('4294967295'), toDateTime('1700000000'), toDateTimeOrNull('4294967296'), toDateTimeOrZero('4294967296'); + +SELECT 'throw, Date32 rejects what it cannot represent instead of substituting a default'; +SELECT toDate32('2000-13-01'); -- { serverError CANNOT_PARSE_DATE } +SELECT toDate32('99999999'); -- { serverError CANNOT_PARSE_DATE } +SELECT CAST(materialize('2000-13-01') AS Date32); -- { serverError CANNOT_PARSE_DATE } +SELECT * FROM format(CSV, 'v Date32', '2000-13-01'); -- { serverError CANNOT_PARSE_DATE } +SELECT * FROM format(TSV, 'v Date32', '2000-13-01'); -- { serverError CANNOT_PARSE_DATE } +SELECT * FROM format(JSONEachRow, 'v Date32', '{"v":"2000-13-01"}'); -- { serverError CANNOT_PARSE_DATE } +SELECT toDate32OrNull('2000-13-01'), toDate32('2299-12-31'), toDate32('1900-01-01'); +SELECT v FROM format(CSV, 'v Variant(Date32, String)', '2000-13-01') SETTINGS allow_experimental_variant_type = 1; + +SELECT 'throw, an unquoted numeric token is checked too'; +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":4294967296}'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":-1}'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(Values, 'v DateTime', '(4294967296)'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":1700000000}'); +SELECT * FROM format(Values, 'v DateTime', '(4294967295)'); +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":1703363853.5}'); + +SELECT 'throw, the last second keeps its fractional ticks'; +SELECT * FROM format(JSONEachRow, 'v DateTime64(3)', '{"v":253402300799.5}'); +SELECT * FROM format(JSONEachRow, 'v DateTime64(3)', '{"v":253402300799.999}'); +SELECT * FROM format(Values, 'v DateTime64(3)', '(253402300799.5)'); + +SELECT 'throw, the range is checked after the timezone offset is applied'; +SELECT parseDateTimeBestEffort('2106-02-07 07:28:15+01:00', 'UTC'), parseDateTimeBestEffort('1969-12-31 23:00:00-01:00', 'UTC'); +SELECT parseDateTimeBestEffortOrNull('2106-02-07 07:28:15+01:00', 'UTC'), toDateTime('2106-02-07 07:28:15+01:00', 'UTC'); +SELECT parseDateTimeBestEffort('2106-02-07 08:28:15+01:00', 'UTC'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } + +SELECT 'throw, typed JSON columns are checked like declared ones'; +SELECT j.d FROM format(JSONEachRow, 'j JSON(d Date)', '{"j":{"d":"2150-12-31"}}'); -- { serverError INCORRECT_DATA } +SELECT j.d FROM format(JSONEachRow, 'j JSON(d DateTime)', '{"j":{"d":"2106-02-07 06:28:16"}}'); -- { serverError INCORRECT_DATA } +SELECT j.d FROM format(JSONEachRow, 'j JSON(d DateTime)', '{"j":{"d":"2106-02-07 06:28:16"}}') SETTINGS date_time_input_format = 'basic'; -- { serverError INCORRECT_DATA } +SELECT j.d FROM format(JSONEachRow, 'j JSON(d DateTime)', '{"j":{"d":4294967296}}'); -- { serverError INCORRECT_DATA } +SELECT j.d FROM format(JSONEachRow, 'j JSON(d Date)', '{"j":{"d":"2149-06-06"}}'); +SELECT j.d FROM format(JSONEachRow, 'j JSON(d DateTime64(3))', '{"j":{"d":"2299-12-31 23:59:59.999"}}'); + +SELECT 'throw, JSONExtract keeps returning a default or NULL'; +SELECT JSONExtract('{"d":"2150-12-31"}', 'd', 'Date'), JSONExtract('{"d":"2150-12-31"}', 'd', 'Nullable(Date)'), JSONExtract('{"d":"2149-06-06"}', 'd', 'Date'); + +SELECT 'throw, an explicitly written year 0000 is not silently replaced'; +SELECT parseDateTimeBestEffort('0000-01-01 00:00:00'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT parseDateTimeBestEffort('00000101'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT parseDateTimeBestEffortUS('01/01/0000'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT toDateTime('0000-01-01 00:00:00'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT * FROM format(JSONEachRow, 'v DateTime', '{"v":"0000-01-01 00:00:00"}'); -- { serverError VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE } +SELECT parseDateTimeBestEffortOrNull('0000-01-01 00:00:00'), parseDateTimeBestEffortOrZero('0000-01-01 00:00:00'); +-- An absent year is a documented best-effort feature, not an overflow +SELECT toMonth(parseDateTimeBestEffort('Mar 3 01:33:48')); From 2cbc5a63b0edc26b453bb46048a58c61e038fc50 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 13:32:16 +0000 Subject: [PATCH 010/185] Backport #120230 to 26.8: Fix a Redis full scan dropping rows, and an out-of-range access on a malformed MGET reply --- src/Storages/StorageRedis.cpp | 22 ++- .../test_storage_redis/fake_redis.py | 128 ++++++++++++++ tests/integration/test_storage_redis/test.py | 163 +++++++++++++++++- 3 files changed, 309 insertions(+), 4 deletions(-) create mode 100644 tests/integration/test_storage_redis/fake_redis.py diff --git a/src/Storages/StorageRedis.cpp b/src/Storages/StorageRedis.cpp index b29f3c83332b..04906d787887 100644 --- a/src/Storages/StorageRedis.cpp +++ b/src/Storages/StorageRedis.cpp @@ -123,8 +123,13 @@ class RedisDataSource final : public ISource MutableColumns columns = sample_block.cloneEmptyColumns(); RedisArray values = storage.multiGet(scan_keys); - for (size_t i = 0; i < scan_keys.size() && !values.get(i).isNull(); i++) + for (size_t i = 0; i < scan_keys.size(); ++i) { + /// MGET answers by position, and a scanned key can hold another Redis type or expire + /// before the MGET runs, so a nil marks one absent value, not the end of the batch. + if (values.get(i).isNull()) + continue; + fillColumns(scan_keys.get(i).value(), values.get(i).value(), primary_key_pos, sample_block, columns @@ -461,7 +466,7 @@ Chunk StorageRedis::getBySerializedKeys(const RedisArray & keys, PaddedPODArray< "StorageRedis::getBySerializedKeys: null_map size {} does not match keys size {}", null_map->size(), keys.size()); - for (size_t i = 0; i < values.size(); ++i) + for (size_t i = 0; i < keys.size(); ++i) { if (null_map && !(*null_map)[i]) { @@ -517,7 +522,18 @@ RedisArray StorageRedis::multiGet(const RedisArray & keys) const for (size_t i = 0; i < keys.size(); ++i) cmd_mget.add(keys.get(i)); - return connection->client->execute(cmd_mget); + RedisArray values = connection->client->execute(cmd_mget); + + /// Callers pair the reply with the request by position, into arrays sized from `keys`. + if (values.isNull() || values.size() != keys.size()) + throw Exception( + ErrorCodes::INTERNAL_REDIS_ERROR, + "Redis table {} returned {} values for MGET of {} keys", + getStorageID().getFullNameNotQuoted(), + values.isNull() ? 0 : values.size(), + keys.size()); + + return values; } void StorageRedis::multiSet(const RedisArray & data) const diff --git a/tests/integration/test_storage_redis/fake_redis.py b/tests/integration/test_storage_redis/fake_redis.py new file mode 100644 index 000000000000..240744332982 --- /dev/null +++ b/tests/integration/test_storage_redis/fake_redis.py @@ -0,0 +1,128 @@ +""" +A minimal RESP server that answers MGET with a reply of the test's choosing. + +`StorageRedis` pairs the elements of an MGET reply with the requested keys by position, so both +the element count and the position of a nil decide what the engine does. Usage: + + fake_redis.py [ ] + +Every MGET is answered with `max(0, len(keys) + delta)` elements: the requested keys echoed back +as their own values, followed by RESP nils. `` and `` are comma-separated +key names; when they are given, SCAN answers with those keys and MGET answers nil for the names +in ``, so a full scan meets a fixed valid/nil sequence rather than Redis's own key +order. Anything else is answered `+OK`. +""" + +import socket +import sys +import threading + + +def read_exactly(stream, size): + data = stream.read(size) + if len(data) != size: + raise EOFError + return data + + +def read_line(stream): + line = stream.readline() + if not line: + raise EOFError + return line.rstrip(b"\r\n") + + +def read_command(stream): + """Read one RESP array. + + Argument payloads are read by declared length, never up to a newline: the keys on the wire + are ClickHouse's serializeBinary output and may contain \\r, \\n or non-UTF-8 bytes. + """ + header = read_line(stream) + if not header.startswith(b"*"): + return None + args = [] + for _ in range(int(header[1:])): + arg_header = read_line(stream) + if not arg_header.startswith(b"$"): + raise EOFError + args.append(read_exactly(stream, int(arg_header[1:]))) + read_exactly(stream, 2) # trailing CRLF + return args + + +def serialize_string(name): + """A String shorter than 128 bytes as ClickHouse serializes it: one length byte, then bytes. + + The engine deserializes the keys SCAN reports and the values MGET returns, so both have to + arrive in that form. See serialize_binary_for_string in test.py. + """ + return bytes([len(name)]) + name.encode() + + +def mget_reply(keys, delta, nil_keys): + count = max(0, len(keys) + delta) + out = [b"*%d\r\n" % count] + for i in range(count): + if i < len(keys) and keys[i] not in nil_keys: + out.append(b"$%d\r\n%s\r\n" % (len(keys[i]), keys[i])) + else: + out.append(b"$-1\r\n") + return b"".join(out) + + +def scan_reply(scan_keys): + """Cursor 0 with every key, so the engine reads the whole keyspace in one batch.""" + out = [b"*2\r\n$1\r\n0\r\n", b"*%d\r\n" % len(scan_keys)] + for key in scan_keys: + out.append(b"$%d\r\n%s\r\n" % (len(key), key)) + return b"".join(out) + + +def handle(conn, delta, scan_keys, nil_keys): + try: + with conn.makefile("rb") as stream: + while True: + args = read_command(stream) + if not args: + break + command = args[0].upper() + if command == b"MGET": + conn.sendall(mget_reply(args[1:], delta, nil_keys)) + elif command == b"SCAN" and scan_keys: + conn.sendall(scan_reply(scan_keys)) + else: + conn.sendall(b"+OK\r\n") + except (EOFError, OSError, ValueError): + pass + finally: + conn.close() + + +def parse_names(argument): + return [serialize_string(name) for name in argument.split(",") if name] + + +def main(): + port = int(sys.argv[1]) + delta = int(sys.argv[2]) + scan_keys = parse_names(sys.argv[3]) if len(sys.argv) > 3 else [] + nil_keys = parse_names(sys.argv[4]) if len(sys.argv) > 4 else [] + + server = socket.socket(socket.AF_INET, socket.SOCK_STREAM) + server.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1) + server.bind(("0.0.0.0", port)) + # The engine holds a connection pool and a direct join reads on several pipeline threads, so + # connections have to be served concurrently. + server.listen(64) + print(f"listening on {port}, MGET reply length = requested + {delta}", flush=True) + + while True: + conn, _ = server.accept() + threading.Thread( + target=handle, args=(conn, delta, scan_keys, nil_keys), daemon=True + ).start() + + +if __name__ == "__main__": + main() diff --git a/tests/integration/test_storage_redis/test.py b/tests/integration/test_storage_redis/test.py index af2892125fce..8bf1f96526a7 100644 --- a/tests/integration/test_storage_redis/test.py +++ b/tests/integration/test_storage_redis/test.py @@ -1,5 +1,6 @@ ## sudo -H pip install redis import json +import os import struct import sys @@ -8,7 +9,9 @@ from helpers.client import QueryRuntimeException from helpers.cluster import ClickHouseCluster -from helpers.test_tools import TSV +from helpers.test_tools import TSV, wait_condition + +SCRIPT_DIR = os.path.dirname(os.path.realpath(__file__)) cluster = ClickHouseCluster(__file__) @@ -529,3 +532,161 @@ def check_query(query, read_type, keys_count, rows_read): plan = node.query("EXPLAIN actions=1, optimize=0 SELECT * FROM test_get_keys") assert 'ReadType: FullScan' in plan + + +# Ports of the three fake Redis endpoints, inside the ClickHouse container. +FAKE_REDIS_PORT_OK = 16379 +FAKE_REDIS_PORT_LONG = 16380 +FAKE_REDIS_PORT_SHORT = 16381 +FAKE_REDIS_PORT_SCAN = 16382 + + +def start_fake_redis(port, delta, keys=""): + node.exec_in_container( + [ + "bash", + "-c", + f"python3 /fake_redis.py {port} {delta} {keys}" + f" > /var/log/clickhouse-server/fake_redis_{port}.log 2>&1", + ], + detach=True, + user="root", + ) + wait_condition( + lambda: node.exec_in_container( + ["bash", "-c", f"exec 3<>/dev/tcp/127.0.0.1/{port} && echo OK"], + nothrow=True, + ), + lambda r: "OK" in r, + max_attempts=40, + delay=0.5, + ) + + +def test_malformed_mget_reply(started_cluster): + """An MGET reply whose element count differs from the request must be rejected. + + The result loop of StorageRedis was bounded by the reply length while the null map it + indexes is sized from the request, so an over-long reply read and wrote past the end of the + null map. A short reply produced fewer rows than keys, which breaks the row-per-key contract + that IKeyValueEntity::getByKeys promises to a direct join. + """ + tables = ("redis_fake_ok", "redis_fake_long", "redis_fake_short", "t_fake_left") + for table in tables: + drop_table(table) + + node.copy_file_to_container( + os.path.join(SCRIPT_DIR, "fake_redis.py"), "/fake_redis.py" + ) + start_fake_redis(FAKE_REDIS_PORT_OK, 0) + start_fake_redis(FAKE_REDIS_PORT_LONG, 1) + start_fake_redis(FAKE_REDIS_PORT_SHORT, -1) + + node.query( + f""" + CREATE TABLE redis_fake_ok (key String, value String) + Engine=Redis('127.0.0.1:{FAKE_REDIS_PORT_OK}') PRIMARY KEY (key); + + CREATE TABLE redis_fake_long (key String, value String) + Engine=Redis('127.0.0.1:{FAKE_REDIS_PORT_LONG}') PRIMARY KEY (key); + + CREATE TABLE redis_fake_short (key String, value String) + Engine=Redis('127.0.0.1:{FAKE_REDIS_PORT_SHORT}') PRIMARY KEY (key); + + CREATE TABLE t_fake_left (k String) ENGINE = TinyLog; + INSERT INTO t_fake_left VALUES ('a'), ('b'); + """ + ) + + def direct_join(table): + return node.query( + f"SELECT key, value FROM (SELECT k AS key FROM t_fake_left) AS t " + f"INNER JOIN {table} USING (key) ORDER BY key " + f"SETTINGS join_algorithm = 'direct' FORMAT TSV" + ) + + # An endpoint that answers with one element per key still works, so a mock that never + # listens or frames RESP wrongly reddens here instead of green-washing the arms below. + assert TSV.toMat(direct_join("redis_fake_ok")) == [["a", "a"], ["b", "b"]] + + with pytest.raises(QueryRuntimeException) as long_join: + direct_join("redis_fake_long") + assert "INTERNAL_REDIS_ERROR" in str(long_join.value) + assert "for MGET of" in str(long_join.value) + + with pytest.raises(QueryRuntimeException) as short_join: + direct_join("redis_fake_short") + assert "INTERNAL_REDIS_ERROR" in str(short_join.value) + + # The other caller of the reply reads it with no null map at all. + with pytest.raises(QueryRuntimeException) as long_in: + node.query("SELECT * FROM redis_fake_long WHERE key IN ('a', 'b')") + assert "INTERNAL_REDIS_ERROR" in str(long_in.value) + + # A zero-element reply is a null array in Poco, not an empty one, so it is the isNull() + # term of the guard that rejects it. One key against the delta = -1 endpoint produces it. + with pytest.raises(QueryRuntimeException) as zero_in: + node.query("SELECT * FROM redis_fake_short WHERE key IN ('a')") + assert "INTERNAL_REDIS_ERROR" in str(zero_in.value) + assert "returned 0 values" in str(zero_in.value) + + for table in tables: + drop_table(table) + + +def test_full_scan_skips_missing_values(started_cluster): + """A full scan must skip the keys MGET answers with nil, not stop at the first one. + + MGET answers by position, and a key SCAN listed can hold a non-string type or expire + before the MGET runs, so a nil marks one absent value and not the end of the batch. + """ + address = get_address_for_ch() + table = "test_full_scan_missing" + fake_table = "redis_fake_scan" + + client = get_redis_connection(db_id=4) + client.flushdb() + drop_table(table) + drop_table(fake_table) + + # Redis alone decides in what order SCAN reports keys, so the mock pins the one thing this + # test is about: a nil arriving before a key that still has a value. + node.copy_file_to_container( + os.path.join(SCRIPT_DIR, "fake_redis.py"), "/fake_redis.py" + ) + start_fake_redis(FAKE_REDIS_PORT_SCAN, 0, "a,b,c b") + + node.query( + f""" + CREATE TABLE {fake_table} (key String, value String) + Engine=Redis('127.0.0.1:{FAKE_REDIS_PORT_SCAN}') PRIMARY KEY (key); + """ + ) + rows = node.query(f"SELECT key, value FROM {fake_table} ORDER BY key FORMAT TSV") + assert TSV.toMat(rows) == [["a", "a"], ["c", "c"]] + + # The same thing on a real Redis, which answers nil for any key that holds another type. + node.query( + f""" + CREATE TABLE {table}(k String, v String) + Engine=Redis('{address}', 4, 'clickhouse') PRIMARY KEY (k); + + INSERT INTO {table} SELECT toString(number), toString(number) FROM numbers(16); + """ + ) + + # Control: every row is readable before the keys below exist, so a fixture that writes + # nothing reddens here instead of leaving the assertion after it vacuous. + assert int(node.query(f"SELECT uniqExact(k) FROM {table}")) == 16 + + for i in range(16): + client.rpush(f"list_{i}", "x") + + # SCAN can report a key twice while Redis rehashes and no read path dedupes the rows, so the + # oracle here is the key set: a lost key is the defect, a repeated one is not. + keys = node.query(f"SELECT DISTINCT k FROM {table} ORDER BY toUInt32(k) FORMAT TSV") + assert TSV.toMat(keys) == [[str(i)] for i in range(16)] + + client.flushdb() + drop_table(table) + drop_table(fake_table) From 5ea8ee9630fdbd7d5e4da40731805e735491fdfa Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 14:46:41 +0000 Subject: [PATCH 011/185] Backport #120713 to 26.8: Add a setting to disable reading through the executable table function and Executable/ExecutablePool tables --- .../experimental_settings_ignore.txt | 1 + src/Core/Settings.cpp | 5 ++ src/Core/SettingsChangesHistory.cpp | 1 + src/Storages/StorageExecutable.cpp | 8 ++ .../05218_allow_executable_tables.reference | 1 + .../05218_allow_executable_tables.sql | 35 +++++++++ ...e_tables_ddl_does_not_run_script.reference | 7 ++ ...ecutable_tables_ddl_does_not_run_script.sh | 73 +++++++++++++++++++ 8 files changed, 131 insertions(+) create mode 100644 tests/queries/0_stateless/05218_allow_executable_tables.reference create mode 100644 tests/queries/0_stateless/05218_allow_executable_tables.sql create mode 100644 tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.reference create mode 100755 tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.sh diff --git a/ci/jobs/scripts/check_style/experimental_settings_ignore.txt b/ci/jobs/scripts/check_style/experimental_settings_ignore.txt index 4c4a0ec48a10..0b63ce3eb10b 100644 --- a/ci/jobs/scripts/check_style/experimental_settings_ignore.txt +++ b/ci/jobs/scripts/check_style/experimental_settings_ignore.txt @@ -14,6 +14,7 @@ allow_deprecated_database_ordinary allow_deprecated_snowflake_conversion_functions allow_distributed_ddl allow_drop_detached +allow_executable_tables allow_execute_multiif_columnar allow_experimental_ai_functions allow_experimental_alter_materialized_view_structure diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 304d1177a179..d405f501fe15 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -4564,6 +4564,11 @@ Possible values: \ DECLARE(Bool, allow_execute_multiif_columnar, true, R"( Allow execute multiIf function columnar +)", 0) \ + DECLARE(Bool, allow_executable_tables, true, R"( +Allow reading through the `executable` table function and from `Executable` and `ExecutablePool` tables. + +Disabling this refuses reads only: creating, attaching, dropping and describing such tables still works. `ExecutablePool` processes that have already started are left running rather than terminated, but each read is still refused until the setting is enabled again. )", 0) \ DECLARE(Bool, formatdatetime_f_prints_single_zero, false, R"( Formatter '%f' in function 'formatDateTime' prints a single zero instead of six zeros if the formatted value has no fractional seconds. diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index a13257bba353..89e1cfb53456 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -44,6 +44,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() addSettingsChanges(settings_changes_history, "26.8", { {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7: the previous value is deliberately equal to the new one, because 26.7 already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, + {"allow_executable_tables", true, true, "New setting to disable reading through the `executable` table function and from `Executable` and `ExecutablePool` tables."}, {"allow_experimental_ai_functions", false, false, "The setting is obsolete, AI functions are beta now and enabled by default."}, {"ai_function_max_retries", 0, 1, "Retry a transient API error once by default, so a single 429 or 5xx from the provider does not fail the query."}, {"adaptive_aggregator_freeze_threshold_bytes", 4194304, 4194304, "New setting bounding the adaptive aggregator's frozen local tables in bytes, whichever of it and the key-count threshold is reached first; 0 disables the byte bound."}, diff --git a/src/Storages/StorageExecutable.cpp b/src/Storages/StorageExecutable.cpp index 08922daedfd6..cd6eaf90454b 100644 --- a/src/Storages/StorageExecutable.cpp +++ b/src/Storages/StorageExecutable.cpp @@ -37,6 +37,7 @@ namespace DB { namespace Setting { + extern const SettingsBool allow_executable_tables; extern const SettingsBool allow_experimental_analyzer; extern const SettingsSeconds max_execution_time; } @@ -58,6 +59,7 @@ namespace ErrorCodes extern const int BAD_ARGUMENTS; extern const int UNSUPPORTED_METHOD; extern const int NUMBER_OF_ARGUMENTS_DOESNT_MATCH; + extern const int SUPPORT_IS_DISABLED; } namespace @@ -163,6 +165,12 @@ void StorageExecutable::readImpl( size_t max_block_size, size_t /*threads*/) { + if (!context->getSettingsRef()[Setting::allow_executable_tables]) + throw Exception( + ErrorCodes::SUPPORT_IS_DISABLED, + "The `executable` table function and the `Executable` and `ExecutablePool` table " + "engines are disabled. Set `allow_executable_tables` setting to enable them"); + auto & script_name = settings->script_name; auto user_scripts_path = context->getUserScriptsPath(); diff --git a/tests/queries/0_stateless/05218_allow_executable_tables.reference b/tests/queries/0_stateless/05218_allow_executable_tables.reference new file mode 100644 index 000000000000..604b15353012 --- /dev/null +++ b/tests/queries/0_stateless/05218_allow_executable_tables.reference @@ -0,0 +1 @@ +x UInt32 diff --git a/tests/queries/0_stateless/05218_allow_executable_tables.sql b/tests/queries/0_stateless/05218_allow_executable_tables.sql new file mode 100644 index 000000000000..f56bfcf22caa --- /dev/null +++ b/tests/queries/0_stateless/05218_allow_executable_tables.sql @@ -0,0 +1,35 @@ +-- `allow_executable_tables` gates reading through the `executable` table function and from +-- `Executable` and `ExecutablePool` tables. + +SELECT * FROM executable('nonexist.sh', 'TSV', 'x UInt32'); -- { serverError UNSUPPORTED_METHOD } + +SET allow_executable_tables = 0; +SELECT * FROM executable('nonexist.sh', 'TSV', 'x UInt32'); -- { serverError SUPPORT_IS_DISABLED } +DESCRIBE executable('nonexist.sh', 'TSV', 'x UInt32'); + +-- A definition is still accepted and can be managed; only reading it is refused. +CREATE TABLE t_exec_gate (x UInt32) ENGINE = Executable('nonexist.sh', 'TSV'); +CREATE VIEW v_exec_gate AS SELECT * FROM t_exec_gate; +CREATE MATERIALIZED VIEW mv_exec_gate ENGINE = MergeTree ORDER BY x AS SELECT * FROM t_exec_gate; +CREATE TABLE t_exec_pool_gate (x UInt32) ENGINE = ExecutablePool('nonexist.sh', 'TSV'); +DETACH TABLE t_exec_gate; +ATTACH TABLE t_exec_gate; + +SELECT * FROM t_exec_gate; -- { serverError SUPPORT_IS_DISABLED } +SELECT * FROM v_exec_gate; -- { serverError SUPPORT_IS_DISABLED } +SELECT * FROM t_exec_pool_gate; -- { serverError SUPPORT_IS_DISABLED } + +-- The gate is evaluated on the node that runs the read, so a distributed wrapper does not lift it. +SELECT * FROM remote('127.0.0.1', executable('nonexist.sh', 'TSV', 'x UInt32')); -- { serverError SUPPORT_IS_DISABLED } + +-- A statement that reads as part of its own execution is refused too. +CREATE TABLE t_exec_gate_copy ENGINE = MergeTree ORDER BY x AS SELECT * FROM executable('nonexist.sh', 'TSV', 'x UInt32'); -- { serverError SUPPORT_IS_DISABLED } +CREATE MATERIALIZED VIEW mv_exec_gate_populate ENGINE = MergeTree ORDER BY x POPULATE AS SELECT * FROM t_exec_gate; -- { serverError SUPPORT_IS_DISABLED } + +SET allow_executable_tables = 1; +SELECT * FROM t_exec_gate; -- { serverError UNSUPPORTED_METHOD } + +SET allow_executable_tables = 0; +DROP VIEW mv_exec_gate; +DROP VIEW v_exec_gate; +DROP TABLE t_exec_gate; diff --git a/tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.reference b/tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.reference new file mode 100644 index 000000000000..b1b9467c5b0b --- /dev/null +++ b/tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.reference @@ -0,0 +1,7 @@ +SUPPORT_IS_DISABLED +script did not run +1 +script ran +7 +7 +pool worker starts: 1 diff --git a/tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.sh b/tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.sh new file mode 100755 index 000000000000..5f87c83191ce --- /dev/null +++ b/tests/queries/0_stateless/05227_allow_executable_tables_ddl_does_not_run_script.sh @@ -0,0 +1,73 @@ +#!/usr/bin/env bash +# Defining an `Executable` table must not run its script, and a closed `allow_executable_tables` +# must refuse a read before spawning the process. The script appends to a marker file, so any +# execution is observable. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +SCRIPTS_DIR=$(mktemp -d "${CLICKHOUSE_TMP}/exec_gate_scripts_XXXXXX") +trap 'rm -rf "${SCRIPTS_DIR}"' EXIT + +MARKER="${SCRIPTS_DIR}/executed" +POOL_MARKER="${SCRIPTS_DIR}/pool_started" + +cat > "${SCRIPTS_DIR}/marker.sh" << SCRIPT +#!/usr/bin/env bash +echo ran >> '${MARKER}' +printf '1\n' +SCRIPT +chmod +x "${SCRIPTS_DIR}/marker.sh" + +cat > "${SCRIPTS_DIR}/marker_pool.sh" << SCRIPT +#!/usr/bin/env bash +echo ran >> '${POOL_MARKER}' +while IFS= read -r n +do + printf '%s\n' "\${n}" + IFS= read -r x + printf '%s\n' "\${x}" +done +SCRIPT +chmod +x "${SCRIPTS_DIR}/marker_pool.sh" + +CONFIG_FILE="${SCRIPTS_DIR}/local_config.xml" +cat > "${CONFIG_FILE}" << CONFIG + + ${SCRIPTS_DIR}/ + +CONFIG + +# Gate closed: the definition is accepted and the refused read spawns nothing. +$CLICKHOUSE_LOCAL --config-file="${CONFIG_FILE}" --query " +SET allow_executable_tables = 0; +CREATE TABLE t_gate (x UInt32) ENGINE = Executable('marker.sh', 'TSV'); +SELECT * FROM t_gate; +" 2>&1 | grep -o -m1 'SUPPORT_IS_DISABLED' + +# Gate open: defining and wrapping still run nothing. +$CLICKHOUSE_LOCAL --config-file="${CONFIG_FILE}" --query " +CREATE TABLE t_gate (x UInt32) ENGINE = Executable('marker.sh', 'TSV'); +CREATE VIEW v_gate AS SELECT * FROM t_gate; +" +if [ -e "${MARKER}" ]; then echo "script ran"; else echo "script did not run"; fi + +# Positive control: proves the marker is observable, so the check above is not vacuous. +$CLICKHOUSE_LOCAL --config-file="${CONFIG_FILE}" --query " +CREATE TABLE t_gate (x UInt32) ENGINE = Executable('marker.sh', 'TSV'); +SELECT * FROM t_gate; +" +if [ -e "${MARKER}" ]; then echo "script ran"; else echo "script did not run"; fi + +# A pooled worker started by a permitted read survives the gate closing: the refused read neither +# reuses nor respawns it, and reopening the gate serves from the same worker. +$CLICKHOUSE_LOCAL --config-file="${CONFIG_FILE}" --ignore-error --query " +CREATE TABLE t_pool (x UInt32) ENGINE = ExecutablePool('marker_pool.sh', 'TSV', (SELECT 7)) SETTINGS send_chunk_header = 1, pool_size = 1; +SELECT * FROM t_pool; +SET allow_executable_tables = 0; +SELECT * FROM t_pool; +SET allow_executable_tables = 1; +SELECT * FROM t_pool; +" < /dev/null 2>/dev/null +echo "pool worker starts: $(wc -l < "${POOL_MARKER}")" From 0c0d3e0519130ca1b1aa521f5537a58bfa98adcb Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 15:56:57 +0000 Subject: [PATCH 012/185] Backport #121474 to 26.8: Hide the password of a cluster-name `remote()` call in the query log --- src/Parsers/FunctionSecretArgumentsFinder.cpp | 3 +- ...te_cluster_name_password_masking.reference | 13 ++++ ...7_remote_cluster_name_password_masking.sql | 59 +++++++++++++++++++ 3 files changed, 74 insertions(+), 1 deletion(-) create mode 100644 tests/queries/0_stateless/05237_remote_cluster_name_password_masking.reference create mode 100644 tests/queries/0_stateless/05237_remote_cluster_name_password_masking.sql diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 6215a2933f43..0275cd816508 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -680,7 +680,8 @@ void FunctionSecretArgumentsFinder::findRemoteFunctionSecretArguments() { /// remote(named_collection, ..., password = 'password', ...) findSecretNamedArgument("password", 1); - return; + /// An identifier is also a cluster name when no such collection exists, and that form keeps the + /// password in a positional slot, so the walk below has to run for it too. } /// We're going to replace 'password' with '[HIDDEN'] for the following signatures: diff --git a/tests/queries/0_stateless/05237_remote_cluster_name_password_masking.reference b/tests/queries/0_stateless/05237_remote_cluster_name_password_masking.reference new file mode 100644 index 000000000000..3b8835252a03 --- /dev/null +++ b/tests/queries/0_stateless/05237_remote_cluster_name_password_masking.reference @@ -0,0 +1,13 @@ +1 +1 +1 +1 1 1 +SELECT 1 FROM remote(\'\', \'db\', \'t\', \'usr\', \'[HIDDEN]\') +SELECT 1 FROM remote(\'\', \'db\', \'t\', \'usr\', \'[HIDDEN]\', rand()) +SELECT 1 FROM remote(\'\', db.t, \'usr\', \'[HIDDEN]\') +SELECT 1 FROM remote(\'\', numbers(10), \'usr\', \'[HIDDEN]\') +SELECT 1 FROM remote(cluster_that_does_not_exist_05237, system, one, \'usr\', \'[HIDDEN]\') +SELECT 1 FROM remote(test_shard_localhost, system, one, \'usr\', \'[HIDDEN]\') +SELECT 1 FROM remote(test_shard_localhost, system.one, \'usr\', \'[HIDDEN]\') +SELECT 1 FROM remoteSecure(cluster_that_does_not_exist_05237, system, one, \'usr\', \'[HIDDEN]\') +1 1 1 1 1 diff --git a/tests/queries/0_stateless/05237_remote_cluster_name_password_masking.sql b/tests/queries/0_stateless/05237_remote_cluster_name_password_masking.sql new file mode 100644 index 000000000000..c3deec41db3c --- /dev/null +++ b/tests/queries/0_stateless/05237_remote_cluster_name_password_masking.sql @@ -0,0 +1,59 @@ +-- An identifier as the first argument of `remote` names either a named collection or a cluster +-- from the server configuration. In the cluster-name form the user and the password are ordinary +-- positional arguments, so the secret arguments finder has to run its positional walk for it too. + +-- These run: test_shard_localhost is a configured cluster, so the positional credentials are +-- parsed and then unused. +SELECT 1 FROM remote(test_shard_localhost, system, one, 'usr', 'SECRETPW1'); +SELECT 1 FROM remote(test_shard_localhost, system.one, 'usr', 'SECRETPW2'); + +-- An unknown cluster is reported only after the whole argument list has been parsed and logged. +SELECT 1 FROM remote(cluster_that_does_not_exist_05237, system, one, 'usr', 'SECRETPW3'); -- { serverError CLUSTER_DOESNT_EXIST } +SELECT 1 FROM remoteSecure(cluster_that_does_not_exist_05237, system, one, 'usr', 'SECRETPW4'); -- { serverError CLUSTER_DOESNT_EXIST } + +-- The addresses form, for comparison: `remote('')` is rejected in parseAddress, also after the +-- whole argument list has been parsed. +SELECT 1 FROM remote('', 'db', 't', 'usr', 'SECRETPW5'); -- { serverError BAD_ARGUMENTS } +SELECT 1 FROM remote('', db.t, 'usr', 'SECRETPW6'); -- { serverError BAD_ARGUMENTS } +SELECT 1 FROM remote('', numbers(10), 'usr', 'SECRETPW7'); -- { serverError BAD_ARGUMENTS } +SELECT 1 FROM remote('', 'db', 't', 'usr', 'SECRETPW8', rand()); -- { serverError BAD_ARGUMENTS } + +-- Not credentials, and they have to stay readable: a user name with no password after it, a +-- sharding key spelled as an equality whose left operand happens to be called `password`, and a +-- SETTINGS clause. +SELECT 1 FROM remote(test_shard_localhost, system, one, 'usr'); +SELECT 1 FROM remote('', 'db', 't', 'usr', password = 'NOT_A_CREDENTIAL'); -- { serverError BAD_ARGUMENTS } +SELECT 1 FROM remote('', 'db', 't', 'usr', rand(), SETTINGS skip_unavailable_shards = 1); -- { serverError BAD_ARGUMENTS } + +-- The analyzer keeps its own finder over the query tree, and `EXPLAIN QUERY TREE` renders through +-- it, so the same statement is masked on that path too. +SELECT countIf(explain LIKE '%[HIDDEN]%') = 1, countIf(explain LIKE '%SECRETPW9%') = 0, countIf(explain LIKE '%usr%') = 1 +FROM (EXPLAIN QUERY TREE SELECT 1 FROM remote(test_shard_localhost, system, one, 'usr', 'SECRETPW9')); + +SYSTEM FLUSH LOGS query_log; + +-- `system.query_log` holds the same masked text as `system.text_log` and the server log. The two +-- queries below exclude themselves through `system.query_log` and `EXPLAIN`, and they must not +-- filter on the secret: once it is masked, a secret-keyed filter matches nothing and passes +-- vacuously. +SELECT DISTINCT query FROM system.query_log +WHERE current_database = currentDatabase() + AND event_date >= yesterday() + AND query LIKE '%FROM remote%' + AND query NOT LIKE '%system.query_log%' + AND query NOT LIKE '%EXPLAIN%' + AND query LIKE '%[HIDDEN]%' +ORDER BY query; + +SELECT + uniqExact(query) = 11, + countIf(query LIKE '%SECRETPW%') = 0, + uniqExactIf(query, query LIKE '%[HIDDEN]%') = 8, + uniqExactIf(query, query LIKE '%NOT_A_CREDENTIAL%') = 1, + uniqExactIf(query, query LIKE '%skip_unavailable_shards%') = 1 +FROM system.query_log +WHERE current_database = currentDatabase() + AND event_date >= yesterday() + AND query LIKE '%FROM remote%' + AND query NOT LIKE '%system.query_log%' + AND query NOT LIKE '%EXPLAIN%'; From 414f313387dec19bdf8afb87df428a0c3f4d9ae2 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 15:59:26 +0000 Subject: [PATCH 013/185] Backport #121134 to 26.8: Skip `Alias` tables in a database-wide `TRUNCATE` --- src/Storages/StorageAlias.h | 4 ++ ...s_skipped_by_truncate_all_tables.reference | 18 +++++++ ...3_alias_skipped_by_truncate_all_tables.sql | 51 +++++++++++++++++++ 3 files changed, 73 insertions(+) create mode 100644 tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.reference create mode 100644 tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.sql diff --git a/src/Storages/StorageAlias.h b/src/Storages/StorageAlias.h index fa0262c17b82..3fa188702e06 100644 --- a/src/Storages/StorageAlias.h +++ b/src/Storages/StorageAlias.h @@ -35,6 +35,10 @@ class StorageAlias final : public IStorage, WithContext bool readsFromOtherTables() const override { return true; } + /// An `Alias` has no data of its own, so a bulk `TRUNCATE ALL TABLES` must skip it. + /// Only the bulk paths consult this; an explicit `TRUNCATE TABLE ` still truncates the target. + bool supportsTruncate() const override { return false; } + /// Get the target storage this alias points to StoragePtr getTargetTable(std::optional access_check = std::nullopt) const; StoragePtr tryGetTargetTable() const { return DatabaseCatalog::instance().tryGetTable(StorageID(target_database, target_table), getContext()); } diff --git a/tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.reference b/tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.reference new file mode 100644 index 000000000000..922dbfa66508 --- /dev/null +++ b/tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.reference @@ -0,0 +1,18 @@ +-- fixture +al_1 Alias +al_2 Alias +al_3 Alias +al_4 Alias +al_5 Alias +al_6 Alias +al_7 Alias +al_8 Alias +j Join +-- truncate all tables: the target is emptied by its own entry +0 +-- an alias is not a data holder: truncating only the aliases truncates nothing +300 +-- an alias to a missing table no longer fails the whole statement +0 +-- the statement names one database, so a target in another one keeps its rows +300 diff --git a/tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.sql b/tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.sql new file mode 100644 index 000000000000..e1fed47f288b --- /dev/null +++ b/tests/queries/0_stateless/05233_alias_skipped_by_truncate_all_tables.sql @@ -0,0 +1,51 @@ +-- Tags: no-replicated-database +-- no-replicated-database: TRUNCATE ALL TABLES is not supported for Replicated databases. + +-- A database-wide TRUNCATE enqueues one task per table name, all sharing one query id. An `Alias` is +-- a name without data, so the alias entry and its target entry converge on one storage and both take +-- that storage's lock, which `RWLockImpl` rejects for a query id that already holds it. +-- +-- The target must stay non-MergeTree for this to be covered: two exclusive acquisitions abort, two +-- shared ones are ref-counted. `Join` is deliberate: the test runner rewrites +-- `Log`/`TinyLog`/`StripeLog`/`Memory` to `MergeTree` in some lanes, which would make this vacuous. + +CREATE TABLE j (k UInt64, v UInt64) ENGINE = Join(ANY, LEFT, k); +INSERT INTO j SELECT number, number FROM numbers(300); + +CREATE TABLE al_1 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_2 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_3 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_4 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_5 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_6 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_7 ENGINE = Alias(currentDatabase(), 'j'); +CREATE TABLE al_8 ENGINE = Alias(currentDatabase(), 'j'); + +-- Without this an arm could report the expected value while probing a plain table instead of an alias. +SELECT '-- fixture'; +SELECT name, engine FROM system.tables WHERE database = currentDatabase() ORDER BY name; + +SELECT '-- truncate all tables: the target is emptied by its own entry'; +TRUNCATE ALL TABLES FROM {CLICKHOUSE_DATABASE:Identifier}; +SELECT count() FROM j; + +SELECT '-- an alias is not a data holder: truncating only the aliases truncates nothing'; +INSERT INTO j SELECT number, number FROM numbers(300); +TRUNCATE TABLES FROM {CLICKHOUSE_DATABASE:Identifier} LIKE 'al%'; +SELECT count() FROM j; + +SELECT '-- an alias to a missing table no longer fails the whole statement'; +CREATE TABLE al_missing ENGINE = Alias(currentDatabase(), 'no_such_table'); +TRUNCATE ALL TABLES FROM {CLICKHOUSE_DATABASE:Identifier}; +SELECT count() FROM j; + +-- A `MergeTree` target on purpose: this arm's loss is silent, with no lock convergence and no error, +-- so it fails on its own rather than through the abort the arms above provoke. +SELECT '-- the statement names one database, so a target in another one keeps its rows'; +CREATE DATABASE {CLICKHOUSE_DATABASE_1:Identifier}; +CREATE TABLE {CLICKHOUSE_DATABASE_1:Identifier}.t (k UInt64) ENGINE = MergeTree ORDER BY k; +INSERT INTO {CLICKHOUSE_DATABASE_1:Identifier}.t SELECT number FROM numbers(300); +CREATE TABLE al_cross ENGINE = Alias({CLICKHOUSE_DATABASE_1:String}, 't'); +TRUNCATE ALL TABLES FROM {CLICKHOUSE_DATABASE:Identifier}; +SELECT count() FROM {CLICKHOUSE_DATABASE_1:Identifier}.t; +DROP DATABASE {CLICKHOUSE_DATABASE_1:Identifier}; From 2c89004b4d3d76173f07e5612f8f7abad86c9b27 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 16:37:26 +0000 Subject: [PATCH 014/185] Backport #118457 to 26.8: Fix Paimon incremental read failing with REPLICA_IS_ALREADY_ACTIVE after a Keeper session loss --- src/Common/FailPoint.cpp | 1 + .../DataLakes/Paimon/PaimonStreamState.cpp | 15 ++- .../test_paimon_incremental_read/test.py | 103 ++++++++++++++++++ 3 files changed, 117 insertions(+), 2 deletions(-) diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 5df9d5d8f3b1..1cc541d93098 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -161,6 +161,7 @@ static struct InitFiu PAUSEABLE_ONCE(kafka2_remove_zk_before_get_children) \ PAUSEABLE_ONCE(kafka2_remove_zk_before_final_multi) \ PAUSEABLE_ONCE(keeper_map_delete_pause_before_multi) \ + PAUSEABLE_ONCE(paimon_incremental_read_pause_before_is_active_remove) \ PAUSEABLE(dummy_pausable_failpoint) \ PAUSEABLE(paimon_incremental_read_pause_after_watermark_commit) \ ONCE(execute_query_calling_empty_set_result_func_on_exception) \ diff --git a/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonStreamState.cpp b/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonStreamState.cpp index bda89875692e..88ab26fb305f 100644 --- a/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonStreamState.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonStreamState.cpp @@ -10,6 +10,7 @@ #include #include #include +#include namespace DB { @@ -20,6 +21,11 @@ extern const int LOGICAL_ERROR; extern const int REPLICA_IS_ALREADY_ACTIVE; } +namespace FailPoints +{ +extern const char paimon_incremental_read_pause_before_is_active_remove[]; +} + PaimonStreamState::PaimonStreamState( zkutil::ZooKeeperPtr keeper_, const String & keeper_path_, @@ -162,14 +168,19 @@ bool PaimonStreamState::activate() { /// Stale node from our previous session — safe to reclaim. /// Use versioned delete (CAS) to guard against TOCTOU races. + FailPointInjection::pauseFailPoint( + FailPoints::paimon_incremental_read_pause_before_is_active_remove); auto remove_code = keeper->tryRemove(is_active_path, stat.version); - if (remove_code != Coordination::Error::ZOK) + /// ZNONODE is this removal's postcondition, not a failure: Keeper reaps the + /// ephemeral itself once the session that created it expires. + if (remove_code != Coordination::Error::ZOK && remove_code != Coordination::Error::ZNONODE) { LOG_WARNING(log, "Failed to remove stale is_active node at {} (code: {}). " "Will retry on next attempt.", is_active_path.string(), remove_code); return false; } - LOG_INFO(log, "Removed stale is_active node from previous session at {}", is_active_path.string()); + LOG_INFO(log, "Cleared stale is_active node from previous session at {} ({})", + is_active_path.string(), remove_code); } else { diff --git a/tests/integration/test_paimon_incremental_read/test.py b/tests/integration/test_paimon_incremental_read/test.py index 4e256a4a1de6..2ceb6c298730 100644 --- a/tests/integration/test_paimon_incremental_read/test.py +++ b/tests/integration/test_paimon_incremental_read/test.py @@ -21,6 +21,7 @@ CH_MV_PAIMON_TABLE = "paimon_mv_source" CH_MV_MERGETREE_TABLE = "paimon_mv_dest" CH_MV_NAME = "paimon_refresh_mv" +CH_TABLE_NAME_ACTIVATE_RECLAIM = "paimon_inc_read_activate_reclaim" cluster = ClickHouseCluster(__file__) node = cluster.add_instance( @@ -425,3 +426,105 @@ def test_paimon_to_mergetree_via_refresh_mv(started_cluster): node.query(f"DROP VIEW IF EXISTS {CH_MV_NAME} SYNC;") node.query(f"DROP TABLE IF EXISTS {CH_MV_MERGETREE_TABLE} SYNC;") node.query(f"DROP TABLE IF EXISTS {CH_MV_PAIMON_TABLE} SYNC;") + + +def test_paimon_incremental_read_activate_tolerates_reaped_is_active(started_cluster): + """A read after a Keeper session loss reclaims the `is_active` marker its own + previous session left behind, and must survive Keeper reaping that ephemeral + concurrently: the marker is visible to the reclaim path's `tryGet` and already + gone by its versioned `tryRemove`, which then reports ZNONODE. The marker here + is fabricated because it reproduces exactly that observable state, whereas + waiting for the real reap to land inside a two-round-trip window would make the + test nondeterministic.""" + writer_container_id = cluster.get_instance_docker_id("paimon-incremental-writer") + + warehouse_name = "warehouse_activate_reclaim" + warehouse_uri = f"file://{USER_FILES_PATH}/{warehouse_name}/" + warehouse_dir = f"{USER_FILES_PATH}/{warehouse_name}" + table_path = f"{USER_FILES_PATH}/{warehouse_name}/test.db/test_table" + # Unique per run: committed_snapshot persists in Keeper, so a rerun against + # the same cluster must not inherit an earlier run's watermark. + keeper_path = f"/clickhouse/paimon_activate_reclaim_{uuid.uuid4().hex}" + is_active_path = f"{keeper_path}/replicas/r1/is_active" + failpoint = "paimon_incremental_read_pause_before_is_active_remove" + + _clean_warehouse(writer_container_id, warehouse_dir) + + # Warm-up commit (snapshot 1), consumed to establish the baseline. + _run_writer(writer_container_id, warehouse_uri=warehouse_uri, start_id=0, rows_per_commit=1, commit_times=1) + _create_clickhouse_table_for_paimon_incremental_read( + CH_TABLE_NAME_ACTIVATE_RECLAIM, table_path, keeper_path=keeper_path + ) + count_query = f"SELECT count() FROM {CH_TABLE_NAME_ACTIVATE_RECLAIM}" + _wait_until_query_result(count_query, "1\n", database="default") + _wait_until_query_result(count_query, "0\n", database="default") + + zk = cluster.get_kazoo_client("zoo1") + reader = None + reader_result = {} + try: + # Read the server's own marker instead of hardcoding its payload format. + identifier = zk.get(is_active_path)[0] + + # Snapshot 2: the batch the post-reconnect read must deliver. + _run_writer(writer_container_id, warehouse_uri=warehouse_uri, start_id=1, rows_per_commit=10, commit_times=1) + + session_query = ( + "SELECT client_id FROM system.zookeeper_connection WHERE name = 'default'" + ) + old_session = node.query(session_query) + # Finalizes the shared session, so the Keeper handle latched in + # PaimonStreamState stays expired and the next read takes the + # needsNewKeeper() branch that calls activate(). + node.query("SYSTEM RECONNECT ZOOKEEPER") + assert node.query(session_query) != old_session, ( + f"the Keeper session was not replaced (still {old_session!r})" + ) + + deadline = time.monotonic() + 60 + while zk.exists(is_active_path) is not None: + assert time.monotonic() < deadline, "the old session's is_active was never reaped" + time.sleep(0.5) + + # Persistent, not ephemeral: activate() never inspects stat.ephemeralOwner, + # so persistence is invisible to the code under test while keeping a second + # Keeper session out of the test's failure modes. + zk.create(is_active_path, identifier) + + node.query(f"SYSTEM ENABLE FAILPOINT {failpoint}") + reader = threading.Thread( + target=lambda: reader_result.update( + zip(("out", "err"), node.query_and_get_answer_with_error(count_query)) + ) + ) + reader.start() + + # Returns only once a thread has parked at the failpoint. Since the + # failpoint sits inside the identifier-matched branch, that proves the read + # entered the reclaim path with a marker it recognises as its own. + node.query(f"SYSTEM WAIT FAILPOINT {failpoint} PAUSE", timeout=60) + assert reader.is_alive(), ( + f"the reader returned before parking at the failpoint: {reader_result!r}" + ) + + # The reap lands inside the read's tryGet/tryRemove window. + zk.delete(is_active_path) + node.query(f"SYSTEM NOTIFY FAILPOINT {failpoint}") + + reader.join(timeout=120) + assert not reader.is_alive(), "the reader thread never finished" + assert not reader_result.get("err"), f"the read failed: {reader_result!r}" + # Not just "did not throw": the pending snapshot must still be delivered. + assert reader_result.get("out") == "10\n", ( + f"the pending snapshot was not delivered: {reader_result!r}" + ) + assert zk.get(f"{keeper_path}/committed_snapshot")[0] == b"2" + _wait_until_query_result(count_query, "0\n", database="default") + finally: + # The failpoint is process-global: an early failure must leave neither it + # armed nor the reader thread parked on it. + node.query(f"SYSTEM DISABLE FAILPOINT {failpoint}") + zk.stop() + if reader is not None: + reader.join(timeout=60) + node.query(f"DROP TABLE IF EXISTS {CH_TABLE_NAME_ACTIVATE_RECLAIM} SYNC;") From 32d9c0dcf6f6655396e77a7a4caa97911b3dfc27 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 22 Sep 2026 17:15:36 +0000 Subject: [PATCH 015/185] Backport #121402 to 26.8: Fix OOB write in the stem function --- src/Functions/stem.cpp | 8 +++++-- .../queries/0_stateless/01890_stem.reference | 11 ++++++++++ tests/queries/0_stateless/01890_stem.sql | 21 +++++++++++++++++++ 3 files changed, 38 insertions(+), 2 deletions(-) diff --git a/src/Functions/stem.cpp b/src/Functions/stem.cpp index e732014562fb..c8784dfdebfe 100644 --- a/src/Functions/stem.cpp +++ b/src/Functions/stem.cpp @@ -99,9 +99,10 @@ class Stemmer /// Rows where null_map[i] != 0 are skipped and emitted as empty strings. /// For FixedString, getDataAt returns the value with null-byte padding; trimRight removes it. /// For String, trailing zero bytes are valid data and must not be trimmed. - /// Snowball stemming never lengthens a word, so upper_bound bytes is a safe pre-allocation. MutableColumnPtr stemColumn(const IColumn & col, size_t input_rows_count, const NullMap * null_map = nullptr) { + /// upper_bound is only an initial estimate: some stemmers lengthen a word (e.g. Turkish maps + /// the ASCII 'i' to the 2-byte 'ı'), so the loop grows res_data when the output overflows it. size_t upper_bound = 0; const bool is_fixed_string = checkAndGetColumn(&col) != nullptr; if (const auto * col_str = checkAndGetColumn(&col)) @@ -131,7 +132,10 @@ class Stemmer if (is_fixed_string) trimRight(word, '\0'); std::string_view stemmed = stem(word); - chassert(data_size + stemmed.size() <= res_data.size()); + + /// Stemming can lengthen the word, grow the output buffer. + if (data_size + stemmed.size() > res_data.size()) + res_data.resize(data_size + stemmed.size()); memcpy(res_data.data() + data_size, stemmed.data(), stemmed.size()); data_size += stemmed.size(); diff --git a/tests/queries/0_stateless/01890_stem.reference b/tests/queries/0_stateless/01890_stem.reference index 7b5f7af34489..6b0de99cce39 100644 --- a/tests/queries/0_stateless/01890_stem.reference +++ b/tests/queries/0_stateless/01890_stem.reference @@ -103,6 +103,17 @@ run bless bless disguis +- Lengthening stemmers. +-- Turkish stems a 3-byte word to a 5-byte word, so the output is longer than the input. +uagı +-- Over a multi-block scan the output must stay correct even when it overflows the input-sized estimate. +0 +-- Turkish also lengthens via a multi-byte substitution (3 bytes to 4 bytes). +aaç +-- Estonian lengthens by appending ASCII letters (4 bytes to 5 bytes), a different mechanism. +keesi +-- Estonian over a multi-block scan must also stay correct past the input-sized estimate. +0 - Negative tests. -- Whitespace in a String input raises BAD_ARGUMENTS. -- Whitespace in an Array element raises BAD_ARGUMENTS. diff --git a/tests/queries/0_stateless/01890_stem.sql b/tests/queries/0_stateless/01890_stem.sql index cc5dd09fac70..c678bc88fb96 100644 --- a/tests/queries/0_stateless/01890_stem.sql +++ b/tests/queries/0_stateless/01890_stem.sql @@ -144,6 +144,27 @@ INSERT INTO stem_test_lc VALUES ('blessing'), ('disguise'), ('blessing'); SELECT stem(word, 'en') FROM stem_test_lc ORDER BY word; DROP TABLE stem_test_lc; +SELECT '- Lengthening stemmers.'; + +SELECT '-- Turkish stems a 3-byte word to a 5-byte word, so the output is longer than the input.'; +SELECT stem('uag', 'tr'); + +SELECT '-- Over a multi-block scan the output must stay correct even when it overflows the input-sized estimate.'; +SELECT countIf(s != 'uagı') +FROM (SELECT stem(materialize('uag'), 'tr') AS s FROM numbers(6800)) +SETTINGS max_block_size = 1700; + +SELECT '-- Turkish also lengthens via a multi-byte substitution (3 bytes to 4 bytes).'; +SELECT stem('aac', 'tr'); + +SELECT '-- Estonian lengthens by appending ASCII letters (4 bytes to 5 bytes), a different mechanism.'; +SELECT stem('keeb', 'et'); + +SELECT '-- Estonian over a multi-block scan must also stay correct past the input-sized estimate.'; +SELECT countIf(s != 'keesi') +FROM (SELECT stem(materialize('keeb'), 'et') AS s FROM numbers(6800)) +SETTINGS max_block_size = 1700; + SELECT '- Negative tests.'; SELECT '-- Whitespace in a String input raises BAD_ARGUMENTS.'; From b4bef00062dbf355c3ffe753010b371b1fc148ef Mon Sep 17 00:00:00 2001 From: Nikita Fomichev Date: Tue, 22 Sep 2026 23:05:02 +0200 Subject: [PATCH 016/185] Keep external roles when copying a Context Since https://github.com/ClickHouse/ClickHouse/pull/110867 a remote node of a secret interserver query enables the initiator's current roles only as external roles and clears the user's default roles. The `ContextData` copy constructor did not copy `external_roles`, so any copied context that recalculated access (for example, a view with an access-related setting in its `SETTINGS` clause) lost all roles and a user whose grants come from roles got `ACCESS_DENIED` on the remote node. This is the relevant part of https://github.com/ClickHouse/ClickHouse/pull/110144 (master): copy `external_roles` and make `setExternalRolesWithLock` replace the list instead of appending, so that `setUser` on a copied context (e.g. `EXECUTE AS`) does not inherit the pushed roles. --- src/Interpreters/Context.cpp | 20 ++++---- ...ushed_roles_survive_context_copy.reference | 6 +++ ...erver_pushed_roles_survive_context_copy.sh | 51 +++++++++++++++++++ 3 files changed, 68 insertions(+), 9 deletions(-) create mode 100644 tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.reference create mode 100755 tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.sh diff --git a/src/Interpreters/Context.cpp b/src/Interpreters/Context.cpp index 756c5ee47bcd..5649d6f637b9 100644 --- a/src/Interpreters/Context.cpp +++ b/src/Interpreters/Context.cpp @@ -1369,6 +1369,7 @@ ContextData::ContextData(const ContextData &o) : input_blocks_reader(o.input_blocks_reader), user_id(o.user_id), current_roles(o.current_roles), + external_roles(o.external_roles), settings_constraints_and_current_profiles(o.settings_constraints_and_current_profiles), access(o.access), need_recalculate_access(o.need_recalculate_access), @@ -2231,15 +2232,16 @@ void Context::setCurrentRolesWithLock(const std::vector & new_current_role void Context::setExternalRolesWithLock(const std::vector & new_external_roles, const std::lock_guard &) { - // External roles are roles received from other node, current roles is a collection of roles that were assigned locally - if (!new_external_roles.empty()) - { - if (external_roles) - external_roles->insert(external_roles->end(), new_external_roles.begin(), new_external_roles.end()); - else - external_roles = std::make_shared>(new_external_roles); - need_recalculate_access = true; - } + // External roles are roles received from another node; current roles is a collection of roles that were assigned locally. + // Replace them unconditionally (rather than append) so that switching the principal via `setUser` clears any external + // roles carried over from a previous principal on the same or a copied context. `ContextData`'s copy constructor + // preserves `external_roles`, so without this reset a context authenticated with pushed roles would keep them after + // `setUser(target_user)` (e.g. `EXECUTE AS target_user`), silently widening the target's privileges. + if (new_external_roles.empty()) + external_roles = nullptr; + else + external_roles = std::make_shared>(new_external_roles); + need_recalculate_access = true; } void Context::setCurrentRolesImpl(const std::vector & new_current_roles, bool throw_if_not_granted, bool skip_if_not_granted, const std::shared_ptr & user) diff --git a/tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.reference b/tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.reference new file mode 100644 index 000000000000..a0fd5153ac79 --- /dev/null +++ b/tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.reference @@ -0,0 +1,6 @@ +-- remote read of a table +90 +-- remote read of a view with SETTINGS +90 +-- remote read of a view with SETTINGS, explicit SET ROLE +90 diff --git a/tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.sh b/tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.sh new file mode 100755 index 000000000000..fd1f2c002b05 --- /dev/null +++ b/tests/queries/0_stateless/05238_interserver_pushed_roles_survive_context_copy.sh @@ -0,0 +1,51 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# On a secret interserver query the remote node enables the initiator's current roles as external roles and +# drops the user's default roles. The external roles must survive `Context::createCopy`: a copied context that +# recalculates access (here, a view with an access-related setting in its `SETTINGS` clause) must not lose them, +# otherwise a user whose grants come only from roles gets `ACCESS_DENIED` on the remote node. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +USER="user_${CLICKHOUSE_DATABASE}" +ROLE="role_${CLICKHOUSE_DATABASE}" + +$CLICKHOUSE_CLIENT -m -q " +DROP TABLE IF EXISTS t; +CREATE TABLE t (x UInt32) ENGINE = MergeTree ORDER BY x; +INSERT INTO t SELECT number FROM numbers(10); +CREATE VIEW v AS SELECT x FROM t SETTINGS allow_introspection_functions = 1; +CREATE TABLE t_dist AS t ENGINE = Distributed(test_cluster_interserver_secret, ${CLICKHOUSE_DATABASE}, t); +CREATE TABLE v_dist AS t ENGINE = Distributed(test_cluster_interserver_secret, ${CLICKHOUSE_DATABASE}, v); + +DROP ROLE IF EXISTS ${ROLE}; +CREATE ROLE ${ROLE}; +GRANT SELECT ON ${CLICKHOUSE_DATABASE}.* TO ${ROLE}; +DROP USER IF EXISTS ${USER}; +CREATE USER ${USER} IDENTIFIED WITH no_password SETTINGS readonly = 0; +GRANT ${ROLE} TO ${USER}; +ALTER USER ${USER} DEFAULT ROLE ${ROLE}; +" + +echo "-- remote read of a table" +$CLICKHOUSE_CLIENT --user "${USER}" -q "SELECT sum(x) FROM t_dist SETTINGS prefer_localhost_replica = 0" + +echo "-- remote read of a view with SETTINGS" +$CLICKHOUSE_CLIENT --user "${USER}" -q "SELECT sum(x) FROM v_dist SETTINGS prefer_localhost_replica = 0" + +echo "-- remote read of a view with SETTINGS, explicit SET ROLE" +$CLICKHOUSE_CLIENT --user "${USER}" -m -q " +SET ROLE ${ROLE}; +SELECT sum(x) FROM v_dist SETTINGS prefer_localhost_replica = 0; +" + +$CLICKHOUSE_CLIENT -m -q " +DROP TABLE IF EXISTS v_dist; +DROP TABLE IF EXISTS t_dist; +DROP VIEW IF EXISTS v; +DROP TABLE IF EXISTS t; +DROP USER IF EXISTS ${USER}; +DROP ROLE IF EXISTS ${ROLE}; +" From c3b63d002dbd5dd2ae6ff82a728db1d63798a5f7 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 01:47:27 +0000 Subject: [PATCH 017/185] Backport #121424 to 26.8: Disable the AI function per-query quotas by default --- src/Core/Settings.cpp | 14 +-- src/Core/SettingsChangesHistory.cpp | 102 ++++++++++++++++++ tests/integration/test_ai_functions/test.py | 4 +- .../0_stateless/03300_ai_functions.reference | 6 +- ...functions_compatibility_defaults.reference | 2 + ...28_ai_functions_compatibility_defaults.sql | 15 ++- 6 files changed, 127 insertions(+), 16 deletions(-) diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 304d1177a179..b762ebe4ed93 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8736,21 +8736,21 @@ Initial delay in milliseconds before the first retry of a failed AI function API DECLARE(Bool, ai_function_throw_on_error, true, R"( If true (default), an AI function call that fails permanently after exhausting all retries aborts the query with an exception. If false, the failed row receives the default value for the column type (empty string for String) and processing continues. )", BETA) \ - DECLARE(UInt64, ai_function_max_input_tokens_per_query, 1000000, R"( -Maximum total input (prompt) tokens across all AI function API calls in a single query. Tracked cumulatively from provider responses. Note that this limit may be exceeded by up to one call's worth of input tokens per in-flight request, since a call's input tokens are not known until its response arrives. Like the other AI quotas, it is enforced per server / query fragment, not summed across a distributed query, and must be set in the top-level query - a sub-query `SETTINGS` override is ignored. Set to 0 to disable. + DECLARE(UInt64, ai_function_max_input_tokens_per_query, 0, R"( +Maximum total input (prompt) tokens across all AI function API calls in a single query. 0 (default) disables the limit. Tracked cumulatively from provider responses. Note that this limit may be exceeded by up to one call's worth of input tokens per in-flight request, since a call's input tokens are not known until its response arrives. Like the other AI quotas, it is enforced per server / query fragment, not summed across a distributed query, and must be set in the top-level query - a sub-query `SETTINGS` override is ignored. This limit is only enforced for providers that report a `usage` object in their response (OpenAI, Anthropic, vLLM). Providers that omit token usage (notably HuggingFace TEI) cause the counter to stay at 0 — use `ai_function_max_api_calls_per_query` instead to bound such calls. )", BETA) \ - DECLARE(UInt64, ai_function_max_output_tokens_per_query, 500000, R"( -Maximum total output (completion) tokens across all AI function API calls in a single query. Tracked cumulatively from provider responses. Note that this limit may be exceeded by up to one call's worth of output tokens per in-flight request, since a call's output tokens are not known until its response arrives. Like the other AI quotas, it is enforced per server / query fragment, not summed across a distributed query, and must be set in the top-level query - a sub-query `SETTINGS` override is ignored. Set to 0 to disable. + DECLARE(UInt64, ai_function_max_output_tokens_per_query, 0, R"( +Maximum total output (completion) tokens across all AI function API calls in a single query. 0 (default) disables the limit. Tracked cumulatively from provider responses. Note that this limit may be exceeded by up to one call's worth of output tokens per in-flight request, since a call's output tokens are not known until its response arrives. Like the other AI quotas, it is enforced per server / query fragment, not summed across a distributed query, and must be set in the top-level query - a sub-query `SETTINGS` override is ignored. This limit is only enforced for providers that report a `usage` object in their response (OpenAI, Anthropic, vLLM). It does not apply to the embedding functions (`aiEmbed`, `aiSimilarity`), which never produce output tokens. )", BETA) \ - DECLARE(UInt64, ai_function_max_api_calls_per_query, 1000, R"( -Maximum number of HTTP requests that AI functions may dispatch per query. Enforced independently by each server and query fragment: within one execution context it is an exact cap shared by every AI function, block, and thread there, but a distributed query (across shards or parallel-replica fragments) may dispatch up to this many requests per shard/fragment. It must be set in the top-level query - a sub-query `SETTINGS` override is ignored. Set to 0 to disable. + DECLARE(UInt64, ai_function_max_api_calls_per_query, 0, R"( +Maximum number of HTTP requests that AI functions may dispatch per query. 0 (default) disables the limit. Enforced independently by each server and query fragment: within one execution context it is an exact cap shared by every AI function, block, and thread there, but a distributed query (across shards or parallel-replica fragments) may dispatch up to this many requests per shard/fragment. It must be set in the top-level query - a sub-query `SETTINGS` override is ignored. )", BETA) \ DECLARE(Bool, ai_function_throw_on_quota_exceeded, true, R"( -If true (default), exceeding an AI function quota limit (`ai_function_max_input_tokens_per_query`, `ai_function_max_output_tokens_per_query`, or `ai_function_max_api_calls_per_query`) aborts the query with an exception. If false, remaining rows receive the default value for the column type (empty string for String). Like the quota limits, this must be set in the top-level query - a sub-query `SETTINGS` override is ignored. +If true (default), exceeding an AI function quota limit (`ai_function_max_input_tokens_per_query`, `ai_function_max_output_tokens_per_query`, or `ai_function_max_api_calls_per_query`) aborts the query with an exception. All three limits are disabled by default, so this has no effect until one of them is set. If false, remaining rows receive the default value for the column type (empty string for String). Like the quota limits, this must be set in the top-level query - a sub-query `SETTINGS` override is ignored. )", BETA) \ DECLARE(NonZeroUInt64, ai_function_embedding_max_batch_size, 100, R"( Maximum number of texts to include in a single HTTP request made by the embedding functions (`aiEmbed`, `aiSimilarity`). Texts are grouped into batches of this size to reduce API call overhead. For example, 500 unique texts with a batch size of 100 result in 5 HTTP requests. diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index a13257bba353..97a310296201 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -41,6 +41,108 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() /// controls new feature and it's 'true' by default, use 'false' as previous_value). /// It's used to implement `compatibility` setting (see https://github.com/ClickHouse/ClickHouse/issues/35972) /// Note: please check if the key already exists to prevent duplicate entries. + addSettingsChanges(settings_changes_history, "26.10", + { + {"allow_executable_tables", true, true, "New setting to disable reading through the `executable` table function and from `Executable` and `ExecutablePool` tables."}, + {"ai_function_max_input_tokens_per_query", 1000000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, + {"ai_function_max_output_tokens_per_query", 500000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, + {"ai_function_max_api_calls_per_query", 1000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, + {"iceberg_tolerate_conflicting_manifest_schemas", false, true, "New setting: when an Iceberg manifest file header carries a schema that conflicts with the schema registered for the same schema-id from metadata.json, prefer the metadata.json schema and log a warning instead of failing the query, matching the behavior of other query engines. `compatibility` below 26.10 restores the previous strict behavior."}, + {"legacy_join_size_limits_trigger_spilling", true, false, "`max_rows_in_join` / `max_bytes_in_join` are now hard caps for every hash join, including the ones that spill to disk, where they used to act as the spill trigger. Spilling is driven by `max_bytes_before_external_join` / `max_bytes_ratio_before_external_join` alone."}, + {"prefer_optimize_projection", false, false, "New setting: choose a usable projection regardless of its estimated cost, like `force_optimize_projection`, but without failing the query when no projection is used."}, + {"query_plan_optimize_join_order_conflict_detector", "", "", "New setting selecting the conflict detector that decides join reordering validity in the DPsub join order algorithm: `a` for the (correct but incomplete) CD-A, `c` for the (correct and complete) CD-C, empty for none."}, + {"use_text_index_postings_cache", false, true, "Enabled the text index posting lists cache globally. Previously each query used a small private cache, which caused posting lists and phrase search results to be recomputed within a single query on large tables."}, + {"output_format_arrow_unsupported_types", "binary", "binary", "New setting superseding `output_format_arrow_unsupported_types_as_binary`, adding a `text` mode. Its default matches the previous behavior, so `compatibility` must not change it."}, + }); + addSettingsChanges(settings_changes_history, "26.9", + { + {"max_bytes_before_external_distinct", 0, 0, "New setting to enable spilling of `DISTINCT` to disk when memory usage exceeds the given threshold in bytes. If 0, only `max_bytes_ratio_before_external_distinct` applies."}, + {"max_bytes_ratio_before_external_distinct", 0., 0.5, "New setting to enable spilling of `DISTINCT` to disk when memory usage exceeds the given ratio of available memory. If 0, only `max_bytes_before_external_distinct` applies."}, + {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7 or 26.8: the previous value is deliberately equal to the new one, because those versions already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, + {"allow_delta_lake_create_table", false, false, "New setting: allow creating a new DeltaLake table using delta-kernel-rs or registering an existing one into a catalog."}, + {"delta_lake_accurate_write_cast", false, true, "New setting: cast written values to the Delta write-schema type with an accurate cast that throws on a value that does not fit the target type instead of silently truncating; `compatibility` below 26.9 uses the plain, non-throwing cast."}, + {"allow_experimental_nullable_tuple_type", false, true, "`Nullable(Tuple)` is now GA"}, + {"enable_nullable_tuple_type", false, true, "`Nullable(Tuple)` is now GA"}, + {"allow_nullable_tuple_in_extracted_subcolumns", false, true, "`Nullable(Tuple)` is now GA: a `Tuple` subcolumn extracted from a `Tuple`, `Variant`, `Dynamic` or `JSON` column is `Nullable(Tuple)` and is NULL in the rows where the subcolumn is missing. The setting is read once at server startup, so `compatibility` restores the previous behavior only from the startup profile (for example, users.xml), not from a session-level `SET`."}, + {"workload_admission_timeout_ms", 0, 0, "New setting bounding how long a query waits to be admitted by workload scheduling (acquiring its query slot and memory reservation) before failing; 0 (default) preserves the previous unbounded wait."}, + {"s3_disable_checksum", false, false, "Obsolete setting: checksum calculation no longer re-reads the source"}, + {"session_query_ids_history_size", 0, 1000, "New setting limiting the size of the session-local query id history exposed through the new `system.session_query_ids` system table. The previous value `0` (recording disabled) reproduces the pre-26.9 behavior."}, + {"query_plan_optimize_join_order_conflict_detector", "", "", "New setting selecting the conflict detector that decides join reordering validity in the DPsub join order algorithm: `a` for the (correct but incomplete) CD-A, `c` for the (correct and complete) CD-C, empty for none."}, + {"reader_executor_window_size", 4194304, 8388608, "Raised the default read window of the experimental `ReaderExecutor` from 4 MiB to 8 MiB. Under memory pressure the window is reduced from this base, floored at 128 KiB."}, + {"webassembly_udf_input_split_memory_ratio", 0.0, 0.5, "New setting controlling the fraction of a WebAssembly UDF instance's linear memory that one call's serialized input may occupy, which also enables the dynamic splitting of that input by its serialized size; `compatibility` below 26.9 sets it to 0 and restores the previous behavior, where `webassembly_udf_max_input_block_size = 0` meant one call per pipeline block."}, + {"cascades_aggregation_pushdown", false, true, "New setting to consider pushing partial aggregation below a join (eager aggregation) in the Cascades optimizer."}, + {"optimize_read_in_reverse_order_final", false, true, "New setting to enable the read-in-order optimization when reading in reverse order of the sorting key with the `FINAL` modifier from `ReplacingMergeTree` tables."}, + {"load_marks_asynchronously", false, true, "Load marks of all streams in parallel by default. On remote disks, synchronous loading of marks of columns with many substreams (such as `JSON`) took one network round trip per stream."}, + {"ast_fuzzer_oracle", false, false, "New setting to enable correctness oracle checks in the server-side AST fuzzer."}, + {"create_token_default_ttl_seconds", 1800, 1800, "New setting giving a lifetime to a token created by `CREATE TOKEN` without an explicit `VALID UNTIL` or `VALID FOR` clause. The statement is new, so there is no earlier behavior to restore and the previous value is the default itself: a `compatibility` with an older version must not turn tokens into never-expiring ones."}, + {"enable_hash_join_row_store", false, true, "New setting to enable transforming the payload of a hash join into a row-major layout."}, + {"min_rows_ratio_for_hash_join_row_store", 5.0, 5.0, "New setting to control the minimum estimated ratio of join output rows to build-side rows to enable transforming hash join payload to row-major. 0 means the transformation is always allowed."}, + {"enable_sharding_aggregator", false, false, "Obsolete setting, the sharded aggregator has been removed in favor of the adaptive aggregator (`enable_adaptive_aggregator`)."}, + {"allow_preliminary_distinct_abandoning", false, true, "New setting that lets the preliminary `DISTINCT` give up deduplicating mostly-unique input, because the final `DISTINCT` deduplicates its output again."}, + {"query_plan_fuse_filter_into_array_join", false, true, "New optimization to fuse a filter on ARRAY JOINed columns into the ARRAY JOIN step, enabled by default."}, + {"iceberg_file_entries_queue_size", 100, 100, "New setting for the previously hardcoded capacity of the queue between the Iceberg data manifest decode tasks and the query."}, + {"iceberg_manifest_decode_concurrency", 2, 4, "New setting bounding how many Iceberg manifest files are decoded concurrently, for delete and data manifests alike. It replaces `iceberg_delete_manifest_decode_concurrency` (kept as an alias). `2` approximates the pre-26.9 data path, which decoded one manifest at a time with the next one's fetch already in flight; under `compatibility` at or below 26.8 the delete decode therefore also runs at 2 rather than its released default of 4, preserving the older data-path memory envelope at the cost of some delete-decode overlap."}, + {"query_plan_lower_array_join_function", false, false, "New optimization to lower an arrayJoin function into a real ARRAY JOIN step; disabled by default."}, + {"adaptive_aggregator_freeze_threshold_bytes", 4194304, 4194304, "New setting bounding the adaptive aggregator's frozen local tables in bytes, whichever of it and the key-count threshold is reached first; 0 disables the byte bound."}, + {"allow_experimental_ai_functions", false, false, "The setting is obsolete, AI functions are beta now and enabled by default."}, + {"allow_experimental_analyzer", true, true, "The setting is obsolete: the analyzer is mandatory and the old query analysis is no longer supported. Disabling it is refused instead of being ignored, and `compatibility` with a version below 24.3 no longer reverts it."}, + {"allow_url_wildcard_from_index_pages", false, false, "Added an alias for setting `allow_experimental_url_wildcard_from_index_pages`."}, + {"allow_kafka_offsets_storage_in_keeper", false, false, "Added an alias for setting `allow_experimental_kafka_offsets_storage_in_keeper`."}, + {"allow_correlated_subqueries", true, true, "Added an alias for setting `allow_experimental_correlated_subqueries`."}, + {"allow_geo_types_in_iceberg", false, false, "Added an alias for setting `allow_experimental_geo_types_in_iceberg`."}, + {"enable_materialized_postgresql_table", false, false, "Added an alias for setting `allow_experimental_materialized_postgresql_table`."}, + {"enable_funnel_functions", false, false, "Added an alias for setting `allow_experimental_funnel_functions`."}, + {"enable_unique_key", false, false, "Added an alias for setting `allow_experimental_unique_key`."}, + {"allow_join_right_table_sorting", false, false, "Added an alias for setting `allow_experimental_join_right_table_sorting`."}, + {"enable_json_lazy_type_hints", false, false, "Added an alias for setting `allow_experimental_json_lazy_type_hints`."}, + {"enable_join_runtime_filters", true, true, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_filter_exact_values_limit", 10000, 10000, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_bloom_filter_bytes", 512_KiB, 512_KiB, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_bloom_filter_hash_functions", 3, 3, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_filter_pass_ratio_threshold_for_disabling", 0.7, 0.7, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_filter_blocks_to_skip_before_reenabling", 30, 30, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_bloom_filter_max_ratio_of_set_bits", 0.7, 0.7, "The JOIN runtime filters became a Production tier feature."}, + {"join_runtime_filter_min_probe_rows", 1000, 1000, "The JOIN runtime filters became a Production tier feature."}, + {"enable_join_runtime_filters_index_analysis", false, false, "The JOIN runtime filters became a Production tier feature."}, + {"ai_function_max_retries", 0, 1, "Retry a transient API error once by default, so a single 429 or 5xx from the provider does not fail the query."}, + {"query_plan_aggregation_bucket_top_k", false, true, "New setting to toggle the plan optimization that materializes only each two-level bucket's best n groups when a final aggregation feeds ORDER BY over its outputs with LIMIT n and the per-bucket selection is provably exact."}, + {"enable_trino_dialect", false, false, "New setting to enable the `trino` value of the `dialect` setting, which translates Trino SQL syntax and maps Trino function names to ClickHouse equivalents."}, + {"enable_join_key_only_hash_tables", false, true, "New setting to store the join keys alone, without a reference to a right row, in the hash tables of joins whose result can never contain a value taken from a right row (`LEFT ANTI`, and `LEFT SEMI` when no right column is selected)."}, + {"distributed_plan_read_in_order", false, false, "New setting to allow the read-in-order optimization for `ORDER BY` in a distributed query plan, so a sorted read of the table's sorting key can skip the sort and stop early. Off by default: only shapes where no exchange survives between the read and the sort are safe today."}, + {"distributed_cache_client_id", "", "", "New setting (CI tests only) to override the distributed cache client id per query."}, + {"query_plan_propagate_predicate_across_join", false, true, "New setting that lifts filter conjuncts across equi-join keys so primary-key pruning fires on both sides."}, + {"read_through_distributed_cache", false, false, "The setting moved to the server configuration and is ignored as a profile setting: reading from the distributed cache is now switched by the server setting `enable_read_through_distributed_cache`, which is applied without a restart (so that merges, mutations and `Buffer` flushes follow it too, instead of being pinned to the value the server started with). Use `force_read_through_distributed_cache` to deviate from the server setting per query."}, + {"write_through_distributed_cache", false, false, "The setting moved to the server configuration and is ignored as a profile setting: writing to the distributed cache is now switched by the server setting `enable_write_through_distributed_cache`, which is applied without a restart (so that merges, mutations and `Buffer` flushes follow it too, instead of being pinned to the value the server started with). Use `force_write_through_distributed_cache` to deviate from the server setting per query."}, + {"force_read_through_distributed_cache", "auto", "auto", "New setting overriding the server setting `enable_read_through_distributed_cache` for a single query."}, + {"force_write_through_distributed_cache", "auto", "auto", "New setting overriding the server setting `enable_write_through_distributed_cache` for a single query."}, + {"distributed_cache_min_inflight_bytes_to_discard_connection_on_seek", 0, 4 * 1024 * 1024, "New setting to drop and reopen a distributed cache connection on a seek when too many in-flight bytes would otherwise be discarded. Defaults to 4 MiB; 0 restores the previous behavior (always reuse the connection via the read range id)."}, + {"distributed_plan_workers_provisioning_timeout_ms", 10000, 10000, "New setting bounding how long a query waits for leased stateless workers to become reachable before execution."}, + {"distributed_plan_fallback_to_local_execution", false, true, "New setting to fall back to local execution when a plan cannot be distributed (only takes effect under `make_distributed_plan`)."}, + {"query_plan_optimize_lazy_materialization_for_object_storage", false, true, "New setting to use lazy materialization for `ORDER BY ... LIMIT n` queries reading Parquet files from object storage (including Iceberg tables)."}, + {"iceberg_compaction_commit_batch_size", 100, 100, "New setting"}, + {"iceberg_compaction_max_rows_in_data_file", std::numeric_limits::max(), std::numeric_limits::max(), "New setting for the max rows of an iceberg data file produced by compaction, separate from the insert-time limit."}, + {"iceberg_compaction_max_bytes_in_data_file", std::numeric_limits::max(), std::numeric_limits::max(), "New setting for the max bytes of an iceberg data file produced by compaction, separate from the insert-time limit."}, + {"enable_json_lazy_type_hints", false, false, "Lazy JSON type hints are now Beta. An alias for setting 'allow_experimental_json_lazy_type_hints'."}, + {"s3_upload_checksum_algorithm", "", "", "New setting to choose the checksum algorithm for S3 uploads."}, + {"network_compression_method", "LZ4", "ZSTD", "Switched the default compression method for client/server and server/server communication from `LZ4` to `ZSTD` to reduce network traffic."}, + {"network_zstd_compression_level", 1, 3, "Aligned the default network `ZSTD` compression level with the new default on-disk `ZSTD(3)` compression."}, + {"use_statistics_for_min_max_aggregation", false, true, "New setting to answer `min`, `max` and `count` aggregations without `GROUP BY` and filters from per-part column statistics for parts that have them materialized, reading only the remaining parts. previous_value=false so `compatibility` with versions before 26.9 keeps the optimization disabled and restores the pre-existing plan."}, + {"parallel_replicas_allow_merge_tables", false, false, "New setting to allow reading from a `Merge` table with plan-based parallel replicas, by expanding the `Merge` read into a union of the reads from the underlying `MergeTree` tables. It only has an effect together with `parallel_replicas_plan_based`."}, + {"iceberg_compaction_max_bytes_in_data_file", std::numeric_limits::max(), 512 * 1024 * 1024, "New setting for the max bytes of an iceberg data file produced by compaction, separate from the insert-time limit. The default is aligned with the documented default of the Iceberg table property `write.target-file-size-bytes` (512 MiB), see https://iceberg.apache.org/docs/1.5.2/configuration/. Previously compaction merged all eligible files of a partition into a single output file."}, + {"iceberg_insert_max_bytes_in_data_file", 1024 * 1024 * 1024, 512 * 1024 * 1024, "Aligned with the documented default of the Iceberg table property `write.target-file-size-bytes` (512 MiB), see https://iceberg.apache.org/docs/1.5.2/configuration/."}, + {"iceberg_data_file_size_lower_threshold_compaction", 10 * 1024 * 1024, 384 * 1024 * 1024, "Aligned with how the Iceberg `rewrite_data_files` procedure derives `min-file-size-bytes`: 0.75 of the target file size (512 MiB). Compaction now selects files below 384 MiB instead of below 10 MiB."}, + {"iceberg_data_file_size_upper_threshold_compaction", 10ULL * 1024 * 1024 * 1024, 512ULL * 1024 * 1024 * 9 / 5, "Aligned with how the Iceberg `rewrite_data_files` procedure derives `max-file-size-bytes`: 1.8 of the target file size (512 MiB)."}, + {"iceberg_manifest_min_count_to_compact", 30, 100, "Aligned with the documented default of the Iceberg table property `commit.manifest.min-count-to-merge` (100), see https://iceberg.apache.org/docs/1.5.2/configuration/."}, + {"parallel_replicas_allow_merge_tables", false, false, "New setting to allow reading from a `Merge` table with plan-based parallel replicas, by expanding the `Merge` read into a union of the reads from the underlying `MergeTree` tables. It only has an effect together with `parallel_replicas_plan_based`."}, + {"optimize_mutations_with_partition_pruning", false, true, "New setting to automatically prune partitions for mutations based on WHERE clause"}, + {"statistics_max_set_size_for_exact_selectivity_estimation", 10000, 10000, "The bound on the cost of estimating the selectivity of `IN` with a large set is kept under `compatibility` with an earlier version: the previous value is deliberately equal to the new one, so that the uncapped estimation, which could add hundreds of milliseconds to the planning of a single query, is not restored."}, + {"type_json_skip_null_typed_paths", false, false, "New setting to treat NULL values in typed JSON paths as absent"}, + {"use_iceberg_manifest_list_partition_pruning", false, true, "New setting to skip Iceberg manifest files whose manifest-list partition summaries cannot match the query filter, without reading them."}, + {"enable_time_series_table", false, false, "The `TimeSeries` table engine and the `promql` dialect were moved to the private preview tier. Added an alias for setting `allow_experimental_time_series_table`."}, + {"enable_time_series_aggregate_functions", false, false, "The `timeSeries*` aggregate functions were moved to the private preview tier. Added an alias for setting `allow_experimental_time_series_aggregate_functions`."}, + {"output_format_arrow_record_batch_size", 0, 0, "New setting to combine small blocks in `Arrow` and `ArrowStream` output using a target row count. The default `0` preserves one record batch per block."}, + {"output_format_arrow_record_batch_size_bytes", 0, 0, "New setting to combine small blocks in `Arrow` and `ArrowStream` output using a target size in bytes of accumulated data. The default `0` preserves one record batch per block."}, + }); addSettingsChanges(settings_changes_history, "26.8", { {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7: the previous value is deliberately equal to the new one, because 26.7 already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, diff --git a/tests/integration/test_ai_functions/test.py b/tests/integration/test_ai_functions/test.py index 5c20520e2fb7..1c7e063b7f5d 100644 --- a/tests/integration/test_ai_functions/test.py +++ b/tests/integration/test_ai_functions/test.py @@ -2153,8 +2153,8 @@ def test_api_call_quota_ignores_subquery_settings(started_cluster): ) outer_wins = int(get_profile_events(qid)["api_calls"]) - # The quota is set only in the subquery; the outer query leaves it at the default (far - # above 64). The subquery cap is ignored, so all 64 rows run rather than stopping at 5 - + # The quota is set only in the subquery; the outer query leaves it at the default (0 - + # no limit). The subquery cap is ignored, so all 64 rows run rather than stopping at 5 - # a quota set only in a subquery has no effect. qid = unique_query_id("quota_levels_subquery_only") instance.query( diff --git a/tests/queries/0_stateless/03300_ai_functions.reference b/tests/queries/0_stateless/03300_ai_functions.reference index 2e1f69e2d613..7fd7da027037 100644 --- a/tests/queries/0_stateless/03300_ai_functions.reference +++ b/tests/queries/0_stateless/03300_ai_functions.reference @@ -44,9 +44,9 @@ String -- Setting defaults ai_function_embedding_default_credentials ai_function_embedding_max_batch_size 100 -ai_function_max_api_calls_per_query 1000 -ai_function_max_input_tokens_per_query 1000000 -ai_function_max_output_tokens_per_query 500000 +ai_function_max_api_calls_per_query 0 +ai_function_max_input_tokens_per_query 0 +ai_function_max_output_tokens_per_query 0 ai_function_max_retries 1 ai_function_request_timeout_sec 60 ai_function_retry_initial_delay_ms 1000 diff --git a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference index a6800ceefe7f..b4345fecb835 100644 --- a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference +++ b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference @@ -1,4 +1,6 @@ -- Current defaults +false 0 1 +-- compatibility = 26.9 restores the API call quota false 1000 1 -- compatibility = 26.6 restores the legacy defaults true 0 0 diff --git a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql index cf37018c792d..1881d8b85a6f 100644 --- a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql +++ b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql @@ -3,10 +3,12 @@ -- no-replicated-database: named collections are server-global, not database-scoped -- ============================================================================= --- Three AI function defaults were flipped: `ai_function_allow_insecure_endpoint` from 1 to 0 --- and `ai_function_max_api_calls_per_query` from 0 (unlimited) to 1000 in 26.8, and --- `ai_function_max_retries` from 0 to 1 in 26.9. `compatibility = 26.6` predates all three --- and restores them, which pins the previous_value/new_value pairs in `SettingsChangesHistory`. +-- Four AI function default flips: `ai_function_allow_insecure_endpoint` from 1 to 0 and +-- `ai_function_max_api_calls_per_query` from 0 (unlimited) to 1000 in 26.8, then +-- `ai_function_max_retries` from 0 to 1 in 26.9, and `ai_function_max_api_calls_per_query` +-- back to 0 (unlimited) in 26.10. `compatibility = 26.6` predates all of them and +-- `compatibility = 26.9` reverts only the last one, which pins the previous_value/new_value +-- pairs in `SettingsChangesHistory`. -- -- The endpoint check runs in `resolveAIParams`, before the zero-row early return -- in `executeImpl`, so an empty source table exercises it without any real HTTP @@ -24,6 +26,11 @@ SELECT '-- Current defaults'; SELECT getSetting('ai_function_allow_insecure_endpoint'), getSetting('ai_function_max_api_calls_per_query'), getSetting('ai_function_max_retries'); SELECT aiGenerate(x, map('credentials', 'ai_compat_remote_http')) FROM tab; -- { serverError BAD_ARGUMENTS } +SELECT '-- compatibility = 26.9 restores the API call quota'; +SET compatibility = '26.9'; +SELECT getSetting('ai_function_allow_insecure_endpoint'), getSetting('ai_function_max_api_calls_per_query'), getSetting('ai_function_max_retries'); +SELECT aiGenerate(x, map('credentials', 'ai_compat_remote_http')) FROM tab; -- { serverError BAD_ARGUMENTS } + SELECT '-- compatibility = 26.6 restores the legacy defaults'; SET compatibility = '26.6'; SELECT getSetting('ai_function_allow_insecure_endpoint'), getSetting('ai_function_max_api_calls_per_query'), getSetting('ai_function_max_retries'); From fc66ebbb23dd880e7954149bc263208a076344bc Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 01:52:41 +0000 Subject: [PATCH 018/185] Backport #121544 to 26.8: Cast Delta Lake partition columns to the declared type --- .../StorageObjectStorageSource.cpp | 6 +++ ...ake_legacy_partition_column_type.reference | 10 ++++ ...delta_lake_legacy_partition_column_type.sh | 52 +++++++++++++++++++ 3 files changed, 68 insertions(+) create mode 100644 tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.reference create mode 100755 tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.sh diff --git a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp index 42f4fcaa6d45..97dfc9cb5868 100644 --- a/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp +++ b/src/Storages/ObjectStorage/StorageObjectStorageSource.cpp @@ -30,6 +30,7 @@ #include #include #include +#include #include #include #include @@ -793,6 +794,11 @@ Chunk StorageObjectStorageSource::generate() const auto column_pos = read_from_format_info.source_header.getPositionByName(name_and_type.name); auto partition_column = name_and_type.type->createColumnConst(chunk.getNumRows(), value)->convertToFullColumnIfConst(); + /// The `_delta_log` type differs from the declared one when the columns were + /// specified rather than inferred, and the block follows the declared schema. + const auto & declared_type = read_from_format_info.source_header.getByPosition(column_pos).type; + if (!name_and_type.type->equals(*declared_type)) + partition_column = castColumn({partition_column, name_and_type.type, name_and_type.name}, declared_type); /// This column is filled with default value now, remove it. chunk.erase(column_pos); /// Add correct values. diff --git a/tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.reference b/tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.reference new file mode 100644 index 000000000000..f9023ab5da2c --- /dev/null +++ b/tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.reference @@ -0,0 +1,10 @@ +-- allow_delta_kernel_rs = 0 +DateTime 1 2026-09-21 09:00:00 +1 2026-09-21 09:00:00 +1 2026-09-21 09:00:00 +Nullable(DateTime) 1 2026-09-21 09:00:00 +-- allow_delta_kernel_rs = 1 +DateTime 1 2026-09-21 09:00:00 +1 2026-09-21 09:00:00 +1 2026-09-21 09:00:00 +Nullable(DateTime) 1 2026-09-21 09:00:00 diff --git a/tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.sh b/tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.sh new file mode 100755 index 000000000000..a57c95c56012 --- /dev/null +++ b/tests/queries/0_stateless/05237_delta_lake_legacy_partition_column_type.sh @@ -0,0 +1,52 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-msan +# Tag no-fasttest: delta-kernel pulls in extra dependencies. +# Tag no-msan: delta-kernel-rs (Rust) is not built under MSan, so DeltaLakeLocal is absent. + +# The legacy (non delta-kernel) DeltaLake reader inserts partition columns into the chunk with the +# type parsed from `_delta_log`, which is not the declared column type when the table function (or +# the insertion table, through `use_structure_from_insertion_table_in_table_functions`) asks for a +# different one. Here `_delta_log` says `Nullable(DateTime64(6))` and the query asks for `DateTime`. +# Without a conversion, a query with no filter reads the column through the wrong type and returns +# garbage, and a filter on it throws a `LOGICAL_ERROR` out of the expression it is compared with. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +ROOT="${USER_FILES_PATH}/${CLICKHOUSE_DATABASE}_delta_legacy_partition" +trap 'rm -rf "${ROOT}" 2>/dev/null' EXIT +rm -rf "${ROOT}" +mkdir -p "${ROOT}/_delta_log" + +# Data file holding only the non-partition column, as a partitioned Delta table stores it. +${CLICKHOUSE_CLIENT} --query " + INSERT INTO FUNCTION file('${ROOT}/process_time=2026-09-21 09%3A00%3A00/data.parquet', Parquet, 'id Int32') VALUES (1); +" < /dev/null + +# A v0 transaction log declaring `process_time` as a nullable `timestamp` partition column. +SCHEMA='{\"type\":\"struct\",\"fields\":[{\"name\":\"id\",\"type\":\"integer\",\"nullable\":false,\"metadata\":{}},{\"name\":\"process_time\",\"type\":\"timestamp\",\"nullable\":true,\"metadata\":{}}]}' +cat > "${ROOT}/_delta_log/00000000000000000000.json" < SELECT * FROM deltaLake(...)` +# passes down when the destination declares `process_time` as `DateTime`. +TABLE_FUNCTION="deltaLakeLocal('${ROOT}', 'Parquet', 'id Int32, process_time DateTime')" +NULLABLE_TABLE_FUNCTION="deltaLakeLocal('${ROOT}', 'Parquet', 'id Int32, process_time Nullable(DateTime)')" + +for kernel in 0 1 +do + echo "-- allow_delta_kernel_rs = ${kernel}" + # `session_timezone` is pinned because the two readers disagree on the zone a partition + # timestamp is parsed in, which is a separate question from the type this test is about. + ${CLICKHOUSE_CLIENT} --allow_delta_kernel_rs="${kernel}" --session_timezone UTC --query " + SELECT toTypeName(process_time), * FROM ${TABLE_FUNCTION}; + SELECT * FROM ${TABLE_FUNCTION} WHERE process_time = '2026-09-21 09:00:00'; + SELECT * FROM ${TABLE_FUNCTION} WHERE toDateTime(process_time) = toDateTime('2026-09-21 09:00:00'); + SELECT * FROM ${TABLE_FUNCTION} WHERE process_time = '2026-09-21 10:00:00'; + SELECT toTypeName(process_time), * FROM ${NULLABLE_TABLE_FUNCTION}; + " < /dev/null +done From 70ebdd9e8e1901efaa2455e8d77e6d0bb61f778b Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 10:18:08 +0000 Subject: [PATCH 019/185] Backport #121623 to 26.8: Add setting analyzer_compatibility_allow_cte_redefinition --- .../performance-and-monitoring/analyzer.mdx | 25 +++-- .../Resolve/IdentifierResolveScope.cpp | 8 +- src/Analyzer/Resolve/IdentifierResolveScope.h | 4 +- src/Analyzer/Resolve/QueryAnalyzer.cpp | 88 ++++++++++++++--- src/Analyzer/Resolve/QueryAnalyzer.h | 3 + src/Core/Settings.cpp | 8 ++ src/Core/SettingsChangesHistory.cpp | 1 + ...atibility_allow_cte_redefinition.reference | 34 +++++++ ...r_compatibility_allow_cte_redefinition.sql | 99 +++++++++++++++++++ 9 files changed, 246 insertions(+), 24 deletions(-) create mode 100644 tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.reference create mode 100644 tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.sql diff --git a/docs/guides/clickhouse/performance-and-monitoring/analyzer.mdx b/docs/guides/clickhouse/performance-and-monitoring/analyzer.mdx index 730c453a6957..2bf316f458c8 100644 --- a/docs/guides/clickhouse/performance-and-monitoring/analyzer.mdx +++ b/docs/guides/clickhouse/performance-and-monitoring/analyzer.mdx @@ -277,22 +277,31 @@ SELECT category, sum(value) FROM t WHERE service = 'svc1' GROUP BY category; Error: `CTE with name ... already exists (MULTIPLE_EXPRESSIONS_FOR_ALIAS)`. Exception code: 179 -Cause: The old analyzer permitted defining multiple Common Table Expressions (WITH ...) with the same name shadowing the earlier one. The analyzer forbids this ambiguity. +Cause: The old analyzer permitted defining multiple Common Table Expressions (WITH ...) with the same name, a later definition shadowing the earlier one. The analyzer rejects this ambiguity by default. -Solution: Rename duplicate CTEs to be unique. +Solution: Rename duplicate CTEs to be unique. As a migration aid, enable `analyzer_compatibility_allow_cte_redefinition = 1` (available since ClickHouse `26.10`) to restore the legacy behavior: a reference binds to the latest definition of the name that is not being resolved at that moment, so a redefinition can read the previous definition and the query body reads the last one. + +Limitations: a CTE declared as `MATERIALIZED` and a CTE in a `WITH RECURSIVE` clause cannot be redefined even with the setting enabled. One shape differs from the old analyzer: a CTE declared between two definitions of a name also binds to the last definition, where the old analyzer bound it to the definition visible at its declaration point. ```sql /* ORIGINAL QUERY */ -WITH - data AS (SELECT 1 AS id), - data AS (SELECT 2 AS id) -- Redefined +WITH + data AS (SELECT 1 AS id), + data AS (SELECT id + 1 AS id FROM data) -- Redefined, reads the previous definition SELECT * FROM data; /* FIXED QUERY */ -WITH - raw_data AS (SELECT 1 AS id), - processed_data AS (SELECT 2 AS id) +WITH + raw_data AS (SELECT 1 AS id), + processed_data AS (SELECT id + 1 AS id FROM raw_data) SELECT * FROM processed_data; + +/* LEGACY BEHAVIOR AS A MIGRATION AID */ +WITH + data AS (SELECT 1 AS id), + data AS (SELECT id + 1 AS id FROM data) +SELECT * FROM data +SETTINGS analyzer_compatibility_allow_cte_redefinition = 1; ``` ### Ambiguous column identifiers {#ambiguous-column-identifiers} diff --git a/src/Analyzer/Resolve/IdentifierResolveScope.cpp b/src/Analyzer/Resolve/IdentifierResolveScope.cpp index a169c74744eb..5fdb1eb7d443 100644 --- a/src/Analyzer/Resolve/IdentifierResolveScope.cpp +++ b/src/Analyzer/Resolve/IdentifierResolveScope.cpp @@ -321,7 +321,13 @@ void dump_list(WriteBuffer & buffer, const String & list_name, const std::ranges dump_mapping(buffer, "Alias name to expression node", aliases.alias_name_to_expression_node); dump_mapping(buffer, "Alias name to function node", aliases.alias_name_to_lambda_node); dump_mapping(buffer, "Alias name to table expression node", aliases.alias_name_to_table_expression_node); - dump_mapping(buffer, "CTE name to query node", cte_name_to_query_node); + if (!cte_name_to_query_node.empty()) + { + buffer << "CTE name to query node table size: " << cte_name_to_query_node.size() << '\n'; + for (const auto & [cte_name, cte_nodes] : cte_name_to_query_node) + for (const auto & cte_node : cte_nodes) + buffer << " { '" << cte_name << "' : " << cte_node->formatASTForErrorMessage() << " }\n"; + } dump_mapping(buffer, "WINDOW name to window node", window_name_to_window_node); dump_list(buffer, "Nodes with duplicated aliases size ", aliases.nodes_with_duplicated_aliases); diff --git a/src/Analyzer/Resolve/IdentifierResolveScope.h b/src/Analyzer/Resolve/IdentifierResolveScope.h index 2864b0b31168..33609d26a3ae 100644 --- a/src/Analyzer/Resolve/IdentifierResolveScope.h +++ b/src/Analyzer/Resolve/IdentifierResolveScope.h @@ -154,8 +154,8 @@ struct IdentifierResolveScope std::list *> join_using_columns; - /// CTE name to query node - std::unordered_map cte_name_to_query_node; + /// CTE name to its definitions in declaration order (several only with `analyzer_compatibility_allow_cte_redefinition`) + std::unordered_map cte_name_to_query_node; /// Window name to window node std::unordered_map window_name_to_window_node; diff --git a/src/Analyzer/Resolve/QueryAnalyzer.cpp b/src/Analyzer/Resolve/QueryAnalyzer.cpp index 3c684a85673b..71b19fafddb4 100644 --- a/src/Analyzer/Resolve/QueryAnalyzer.cpp +++ b/src/Analyzer/Resolve/QueryAnalyzer.cpp @@ -85,6 +85,7 @@ namespace Setting { extern const SettingsBool aggregate_functions_null_for_empty; extern const SettingsBool analyzer_compatibility_allow_non_aggregate_in_having; + extern const SettingsBool analyzer_compatibility_allow_cte_redefinition; extern const SettingsBool enable_streaming_queries; extern const SettingsBool analyzer_compatibility_join_using_top_level_identifier; extern const SettingsBool analyzer_compatibility_multiple_joins_qualify_column_names; @@ -149,6 +150,18 @@ namespace ErrorCodes namespace { +/// True for a `WITH` element declared `AS MATERIALIZED`, before or after its replacement by a `TableNode`. +bool isMaterializedCTEDefinition(const QueryTreeNodePtr & node) +{ + if (const auto * query_node = node->as()) + return query_node->isMaterialized(); + if (const auto * union_node = node->as()) + return union_node->isMaterialized(); + if (const auto * table_node = node->as()) + return table_node->isMaterializedCTE(); + return false; +} + /// Recursively clears aliases from `node` and all of its descendants, stopping at /// nested-scope boundaries (`QUERY`, `UNION`, `LAMBDA`). /// @@ -1306,7 +1319,9 @@ IdentifierResolveResult QueryAnalyzer::tryResolveIdentifierFromCTE( ) { auto full_name = identifier_lookup.identifier.getFullName(); - auto cte_query_node_it = scope.cte_name_to_query_node.find(full_name); + auto cte_nodes_it = scope.cte_name_to_query_node.find(full_name); + if (cte_nodes_it == scope.cte_name_to_query_node.end()) + return {}; /// CTE may reference table expressions with the same name, e.g.: /// @@ -1320,10 +1335,23 @@ IdentifierResolveResult QueryAnalyzer::tryResolveIdentifierFromCTE( /// /// To accomplish this behaviour it's not allowed to resolve identifiers to /// CTE that is being resolved. - if (cte_query_node_it == scope.cte_name_to_query_node.end() || ctes_in_resolve_process.contains(cte_query_node_it->second)) + /// + /// With `analyzer_compatibility_allow_cte_redefinition` a name can have several definitions; the latest one + /// not being resolved wins, so a redefinition reads the previous definition and the query body the last one. + auto & cte_nodes = cte_nodes_it->second; + /// Every site that marks a scope-map CTE node as being resolved updates both sets. With one definition keep the + /// structural check: the materialized-CTE expression site inserts a clone, which only identity would miss. + const bool several_definitions = cte_nodes.size() > 1; + auto cte_node_it = std::find_if(cte_nodes.rbegin(), cte_nodes.rend(), [this, several_definitions](const QueryTreeNodePtr & node) + { + if (several_definitions) + return !cte_definitions_in_resolve_process.contains(node.get()); + return !ctes_in_resolve_process.contains(node); + }); + if (cte_node_it == cte_nodes.rend()) return {}; - auto & cte_node = cte_query_node_it->second; + auto & cte_node = *cte_node_it; auto * query_node = cte_node->as(); auto * union_node = cte_node->as(); @@ -3299,6 +3327,7 @@ ProjectionNames QueryAnalyzer::resolveExpressionNode( /// In this example argument of function `in` is being resolve here. If CTE `test1` is not forbidden, /// `test1` is resolved to CTE (not to the table) in `initializeQueryJoinTreeNode` function. ctes_in_resolve_process.insert(original_cte_node); + cte_definitions_in_resolve_process.insert(original_cte_node.get()); if (subquery_node) resolveQuery(resolved_identifier_node, subquery_scope); @@ -3306,6 +3335,7 @@ ProjectionNames QueryAnalyzer::resolveExpressionNode( resolveUnion(resolved_identifier_node, subquery_scope); ctes_in_resolve_process.erase(original_cte_node); + cte_definitions_in_resolve_process.erase(original_cte_node.get()); } else if (table_node != nullptr && table_node->isMaterializedCTE()) { @@ -5986,12 +6016,18 @@ void QueryAnalyzer::resolveQueryJoinTreeNode(QueryTreeNodePtr & join_tree_node, QueryTreeNodePtr original_cte_node = try_get_original_cte_node(join_tree_node); if (original_cte_node) + { ctes_in_resolve_process.insert(original_cte_node); + cte_definitions_in_resolve_process.insert(original_cte_node.get()); + } resolveExpressionNode(join_tree_node, scope, false /*allow_lambda_expression*/, true /*allow_table_expression*/, true /*ignore_alias=*/); if (original_cte_node) + { ctes_in_resolve_process.erase(original_cte_node); + cte_definitions_in_resolve_process.erase(original_cte_node.get()); + } break; } case QueryTreeNodeType::TABLE_FUNCTION: @@ -6014,19 +6050,22 @@ void QueryAnalyzer::resolveQueryJoinTreeNode(QueryTreeNodePtr & join_tree_node, /// Prevent recursive CTE references during subquery resolution. const auto & cte_name = materialized_cte_ptr->cte_name; - QueryTreeNodePtr cte_map_node; + QueryTreeNodes cte_map_nodes; for (auto * s = &scope; s; s = s->parent_scope) { auto it = s->cte_name_to_query_node.find(cte_name); if (it != s->cte_name_to_query_node.end()) { - cte_map_node = it->second; + cte_map_nodes = it->second; break; } } - if (cte_map_node) + for (const auto & cte_map_node : cte_map_nodes) + { ctes_in_resolve_process.insert(cte_map_node); + cte_definitions_in_resolve_process.insert(cte_map_node.get()); + } IdentifierResolveScope & subquery_scope = createIdentifierResolveScope(subquery, &scope); subquery_scope.subquery_depth = scope.subquery_depth + 1; @@ -6036,8 +6075,11 @@ void QueryAnalyzer::resolveQueryJoinTreeNode(QueryTreeNodePtr & join_tree_node, else resolveUnion(subquery, subquery_scope); - if (cte_map_node) + for (const auto & cte_map_node : cte_map_nodes) + { ctes_in_resolve_process.erase(cte_map_node); + cte_definitions_in_resolve_process.erase(cte_map_node.get()); + } bool is_correlated = subquery->as() ? subquery->as()->isCorrelated() @@ -6490,12 +6532,32 @@ void QueryAnalyzer::resolveQuery(const QueryTreeNodePtr & query_node, Identifier continue; const auto & cte_name = subquery_node ? subquery_node->getCTEName() : union_node->getCTEName(); - auto [_, inserted] = scope.cte_name_to_query_node.emplace(cte_name, node); - if (!inserted) - throw Exception(ErrorCodes::MULTIPLE_EXPRESSIONS_FOR_ALIAS, - "CTE with name {} already exists. In scope {}", - cte_name, - scope.scope_node->formatASTForErrorMessage()); + auto & cte_nodes = scope.cte_name_to_query_node[cte_name]; + if (!cte_nodes.empty()) + { + if (query_node_typed.isRecursiveWith()) + throw Exception(ErrorCodes::MULTIPLE_EXPRESSIONS_FOR_ALIAS, + "CTE with name {} already exists and cannot be redefined in a recursive WITH clause. In scope {}", + cte_name, + scope.scope_node->formatASTForErrorMessage()); + + /// A redefinition is rejected on its second registration, so only the first node can be materialized. + if (isMaterializedCTEDefinition(node) || isMaterializedCTEDefinition(cte_nodes.front())) + throw Exception(ErrorCodes::MULTIPLE_EXPRESSIONS_FOR_ALIAS, + "CTE with name {} already exists and cannot be redefined because it is declared as MATERIALIZED. In scope {}", + cte_name, + scope.scope_node->formatASTForErrorMessage()); + + /// Checked last, so the hint is given only when enabling the setting would help. + if (!scope.context->getSettingsRef()[Setting::analyzer_compatibility_allow_cte_redefinition]) + throw Exception(ErrorCodes::MULTIPLE_EXPRESSIONS_FOR_ALIAS, + "CTE with name {} already exists. Enable the setting analyzer_compatibility_allow_cte_redefinition " + "to let a later definition shadow the earlier one. In scope {}", + cte_name, + scope.scope_node->formatASTForErrorMessage()); + } + + cte_nodes.push_back(node); } /** WITH section can be safely removed, because WITH section only can provide aliases to query expressions diff --git a/src/Analyzer/Resolve/QueryAnalyzer.h b/src/Analyzer/Resolve/QueryAnalyzer.h index 129c38311cd7..b478854440f2 100644 --- a/src/Analyzer/Resolve/QueryAnalyzer.h +++ b/src/Analyzer/Resolve/QueryAnalyzer.h @@ -312,6 +312,9 @@ class QueryAnalyzer /// CTEs that are currently in resolve process QueryTreeNodePtrWithHashSet ctes_in_resolve_process; + /// Same as `ctes_in_resolve_process` but by identity: structural comparison cannot tell identical redefinitions apart. + std::unordered_set cte_definitions_in_resolve_process; + /// Window definitions that are currently in resolve process std::unordered_set windows_in_resolve_process; diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 304d1177a179..372310666525 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8322,6 +8322,14 @@ SELECT ll.Date FROM (SELECT * FROM t AS ll LEFT JOIN t1 ON ll.k = t1.k LEFT JOIN ``` Takes effect only when the analyzer is enabled (`enable_analyzer = 1`). +)", 0) \ + DECLARE(Bool, analyzer_compatibility_allow_cte_redefinition, false, R"( +Allow a Common Table Expression name to be defined more than once in a single `WITH` clause. A reference to such a name binds to the latest definition that is not being resolved at that moment: a redefinition can read the previous definition of the same name, and the query body reads the last one. This matches the query analysis that ClickHouse used before v24.3, where a later definition silently shadowed the earlier ones. One shape differs from that analysis: a CTE declared between two definitions of a name also binds to the last definition, where the old analysis bound it to the definition visible at its declaration point. By default a redefinition is rejected with `MULTIPLE_EXPRESSIONS_FOR_ALIAS`. A CTE declared as `MATERIALIZED` and a CTE in a `WITH RECURSIVE` clause cannot be redefined even when the setting is enabled. + +Possible values: + +- 0 - A CTE name can be defined only once in a `WITH` clause. +- 1 - A later definition of a CTE name shadows the earlier ones. )", 0) \ DECLARE(Bool, enable_identifier_resolve_cache, true, R"( Enable the identifier resolution cache in the query analyzer. The cache shares resolved alias nodes to prevent AST explosion when the same alias is referenced multiple times. Set to false to disable caching if incorrect results are suspected. diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index a13257bba353..4e1f44a9c834 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -43,6 +43,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() /// Note: please check if the key already exists to prevent duplicate entries. addSettingsChanges(settings_changes_history, "26.8", { + {"analyzer_compatibility_allow_cte_redefinition", false, false, "New compatibility setting. When enabled, the analyzer accepts a CTE name defined more than once in a single `WITH` clause and lets a later definition shadow the earlier ones, as the query analysis before v24.3 did."}, {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7: the previous value is deliberately equal to the new one, because 26.7 already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, {"allow_experimental_ai_functions", false, false, "The setting is obsolete, AI functions are beta now and enabled by default."}, {"ai_function_max_retries", 0, 1, "Retry a transient API error once by default, so a single 429 or 5xx from the provider does not fail the query."}, diff --git a/tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.reference b/tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.reference new file mode 100644 index 000000000000..72c437349e4f --- /dev/null +++ b/tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.reference @@ -0,0 +1,34 @@ +-- 1 unreferenced duplicates +1 10 +2 20 +-- 2 redefinition reads the previous definition +40 +-- 3 the query body reads the last definition +2 +-- 4 chain of three definitions +20 +-- 5 two references to the redefined name +2 2 +-- 6 redefinition in a nested scope shadows the outer CTE +3 +-- 7 IN with a redefined CTE +2 +-- 8 the first definition reads the table of the same name +12 +-- 9 progressive redefinition +0 0 0 100 300 +1 10 20 120 360 +2 20 40 140 420 +-- 10 UNION CTE redefined +3 +-- 11 consumer declared between two definitions +2 2 +-- 12 view +1 80 high +2 30 low +-- 13 MATERIALIZED and RECURSIVE CTEs cannot be redefined +-- 14 identical redefinitions are distinct definitions +20 +6 +-- 15 all definitions in resolve process fall through to the outer CTE +102 diff --git a/tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.sql b/tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.sql new file mode 100644 index 000000000000..b32f65522af3 --- /dev/null +++ b/tests/queries/0_stateless/05238_analyzer_compatibility_allow_cte_redefinition.sql @@ -0,0 +1,99 @@ +-- Compatibility setting `analyzer_compatibility_allow_cte_redefinition`: a CTE name may be defined more than once +-- in one WITH clause, a later definition shadowing the earlier ones as the query analysis before v24.3 did. +-- Expected values come from that analysis (26.8 with enable_analyzer = 0), except where noted. + +DROP TABLE IF EXISTS t_cte_redefinition; +DROP VIEW IF EXISTS v_cte_redefinition; + +-- Default: a redefinition is rejected, referenced or not. +WITH d AS (SELECT 1 AS id), d AS (SELECT 2 AS id) SELECT * FROM d; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } +WITH d AS (SELECT 1 AS id), d AS (SELECT 2 AS id) SELECT 1; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } + +SET analyzer_compatibility_allow_cte_redefinition = 1; + +SELECT '-- 1 unreferenced duplicates'; +WITH qb AS (SELECT 1 AS Id, 10 AS Shop), qb AS (SELECT 2 AS Id, 20 AS Shop) +SELECT a.Id, a.Shop FROM (SELECT 1 AS Id, 10 AS Shop UNION ALL SELECT 2, 20) AS a ORDER BY Id; + +SELECT '-- 2 redefinition reads the previous definition'; +WITH + qb AS (SELECT Id, Shop FROM (SELECT 1 AS Id, 10 AS Shop UNION ALL SELECT 2, 20 UNION ALL SELECT 1, 30) WHERE Id = 1), + qb AS (SELECT sum(Shop) AS x FROM qb) +SELECT a.x FROM qb AS a; + +SELECT '-- 3 the query body reads the last definition'; +WITH d AS (SELECT 1 AS id), d AS (SELECT 2 AS id) SELECT * FROM d; + +SELECT '-- 4 chain of three definitions'; +WITH a AS (SELECT 1 AS x), a AS (SELECT x + 1 AS x FROM a), a AS (SELECT x * 10 AS x FROM a) SELECT x FROM a; + +SELECT '-- 5 two references to the redefined name'; +WITH d AS (SELECT 1 AS id), d AS (SELECT id + 1 AS id FROM d) SELECT l.id, r.id FROM d AS l CROSS JOIN d AS r; + +SELECT '-- 6 redefinition in a nested scope shadows the outer CTE'; +WITH a AS (SELECT 1 AS x) SELECT * FROM (WITH a AS (SELECT 2 AS x), a AS (SELECT x + 1 AS x FROM a) SELECT * FROM a); + +SELECT '-- 7 IN with a redefined CTE'; +WITH s AS (SELECT 1 AS x), s AS (SELECT x + 1 AS x FROM s) SELECT number FROM numbers(5) WHERE number IN s; + +SELECT '-- 8 the first definition reads the table of the same name'; +CREATE TABLE t_cte_redefinition (x UInt8) ENGINE = Memory; +INSERT INTO t_cte_redefinition VALUES (5); +WITH + t_cte_redefinition AS (SELECT x + 1 AS x FROM t_cte_redefinition), + t_cte_redefinition AS (SELECT x * 2 AS x FROM t_cte_redefinition) +SELECT x FROM t_cte_redefinition; + +SELECT '-- 9 progressive redefinition'; +WITH + base AS (SELECT number AS id, number * 10 AS val FROM numbers(3)), + joined AS (SELECT id, val, val * 2 AS doubled FROM base), + joined AS (SELECT *, doubled + 100 AS shifted FROM joined), + joined AS (SELECT *, shifted * 3 AS final_val FROM joined) +SELECT * FROM joined ORDER BY id; + +SELECT '-- 10 UNION CTE redefined'; +WITH u AS (SELECT 1 AS x UNION ALL SELECT 2), u AS (SELECT sum(x) AS x FROM u) SELECT x FROM u; + +SELECT '-- 11 consumer declared between two definitions'; +-- The analysis before v24.3 bound `b` to the first `a` (result 1, 2). Here a reference that is not +-- inside a definition of `a` binds to the last definition of `a`. +WITH a AS (SELECT 1 AS v), b AS (SELECT v FROM a), a AS (SELECT 2 AS v), c AS (SELECT v FROM a) +SELECT b.v, c.v FROM b CROSS JOIN c; + +SELECT '-- 12 view'; +CREATE VIEW v_cte_redefinition AS +WITH + raw AS (SELECT 1 AS id, 80 AS pct UNION ALL SELECT 2, 30), + enriched AS (SELECT id, pct FROM raw), + enriched AS (SELECT *, if(pct > 50, 'high', 'low') AS tier FROM enriched) +SELECT * FROM enriched ORDER BY id; +SELECT * FROM v_cte_redefinition; +SELECT * FROM v_cte_redefinition SETTINGS analyzer_compatibility_allow_cte_redefinition = 0; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } + +SELECT '-- 13 MATERIALIZED and RECURSIVE CTEs cannot be redefined'; +SET enable_materialized_cte = 1; +WITH m AS MATERIALIZED (SELECT 1 AS x), m AS (SELECT x + 1 AS x FROM m) SELECT * FROM m; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } +WITH m AS (SELECT 1 AS x), m AS MATERIALIZED (SELECT x + 1 AS x FROM m) SELECT * FROM m; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } +WITH m AS MATERIALIZED (SELECT 1 AS x), m AS MATERIALIZED (SELECT 2 AS x) SELECT 1; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } +SET enable_materialized_cte = 0; +WITH m AS MATERIALIZED (SELECT 1 AS x), m AS (SELECT 2 AS x) SELECT 1; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } +WITH RECURSIVE r AS (SELECT 1 AS n UNION ALL SELECT n + 1 FROM r WHERE n < 3), r AS (SELECT 10 AS n) SELECT * FROM r; -- { serverError MULTIPLE_EXPRESSIONS_FOR_ALIAS } + +SELECT '-- 14 identical redefinitions are distinct definitions'; +WITH + t_cte_redefinition AS (SELECT x * 2 AS x FROM t_cte_redefinition), + t_cte_redefinition AS (SELECT x * 2 AS x FROM t_cte_redefinition) +SELECT x FROM t_cte_redefinition; +WITH + d AS (SELECT 1 AS x), + d AS (SELECT x + 1 AS x FROM d), + d AS (SELECT x + 1 AS x FROM d), + d AS (SELECT l.x + r.x AS x FROM d AS l CROSS JOIN d AS r) +SELECT x FROM d; + +SELECT '-- 15 all definitions in resolve process fall through to the outer CTE'; +WITH a AS (SELECT 100 AS x) SELECT x FROM (WITH a AS (SELECT x + 1 AS x FROM a), a AS (SELECT x + 1 AS x FROM a) SELECT x FROM a); + +DROP VIEW v_cte_redefinition; +DROP TABLE t_cte_redefinition; From 85fe9c4f83a1c464dae46e8c15ae884ef2b486a0 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 11:14:09 +0000 Subject: [PATCH 020/185] Backport #111535 to 26.8: Fix logical error !raw_type->isNullable() in AND-compare chain optimization --- .../Passes/LogicalExpressionOptimizerPass.cpp | 12 +- ...pare_chain_nothing_collapsed_and.reference | 19 +++ ...nd_compare_chain_nothing_collapsed_and.sql | 131 ++++++++++++++++++ 3 files changed, 160 insertions(+), 2 deletions(-) create mode 100644 tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.reference create mode 100644 tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.sql diff --git a/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp b/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp index 793cfb6e6497..1f27cc986e5d 100644 --- a/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp +++ b/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp @@ -899,9 +899,17 @@ static AddComparisonFilterResult addComparisonFilter( return AddComparisonFilterResult::ADDED; } + /// A comparison with a nullable result is ambiguous under NULL and must not be pruned or folded; + /// keep it as-is. Test the comparison node's result type, not the raw operand type, so nested and + /// carrier-hidden nullability (e.g. `LowCardinality(Nullable)`, `Dynamic`, `Variant`) is caught. + if (isNullableOrLowCardinalityNullable(new_filter.original_node->getResultType())) + { + filter_map[expression].opaque_filters.push_back(std::move(new_filter)); + return AddComparisonFilterResult::ADDED; + } + /// Step 1: convert the constant to the column's type for uniform comparison. const auto & raw_type = expression->getResultType(); - chassert(!raw_type->isNullable()); auto expr_type = removeLowCardinality(raw_type); new_filter.converted_value = tryConvertToColumnType(new_filter.constant_node, expr_type); @@ -1961,7 +1969,7 @@ class LogicalExpressionOptimizerVisitor : public InDepthQueryTreeVisitorWithCont expression = lhs; } - /// Both sides are non-constant — keep as-is. + /// Both sides are non-constant (or the constant is NULL-valued) — keep as-is. if (!constant) { all_operands.emplace_back(argument_index, argument); diff --git a/tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.reference b/tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.reference new file mode 100644 index 000000000000..b170576ac2c0 --- /dev/null +++ b/tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.reference @@ -0,0 +1,19 @@ +1 +1 +1 +[10] +[10] +[10] +[10] +1 +1 +1 +1 +1 +1 +1 +1 +[10] +[10] +1 +1 diff --git a/tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.sql b/tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.sql new file mode 100644 index 000000000000..4bb4cb072734 --- /dev/null +++ b/tests/queries/0_stateless/04626_and_compare_chain_nothing_collapsed_and.sql @@ -0,0 +1,131 @@ +-- The `optimize_and_compare_chain` / `optimize_redundant_comparisons` passes prune and fold an +-- AND-chain of comparisons. Both skip a Nullable-typed AND, because a comparison over a Nullable +-- operand cannot participate in range pruning. That guard checks only the AND's own result type, +-- which is unsound: when one AND operand is `Nothing`-typed, the function resolver collapses the +-- whole AND's result type to bare `Nothing` (`Nothing::isNullable()` is false), so the guard passes +-- even though another operand is a comparison over a directly-Nullable expression. That operand then +-- reached the pruning path and hit `chassert(!raw_type->isNullable())`. + +SET enable_analyzer = 1; + +-- 1) No logical error. These queries are otherwise invalid (a `Nothing`-typed value cannot be +-- materialized), so they must fail with a normal handled exception, never a logical error. +-- optimize_and_compare_chain path (chassert reached via the seed loop of tryOptimizeAndCompareChain): +SELECT tuple((materialize(toNullable(NULL)) = 1) AND (assumeNotNull(materialize(toNullable(NULL))) = 2)) SETTINGS optimize_and_compare_chain = 1; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((materialize(toNullable(1::Int32)) = 1) AND (assumeNotNull(materialize(toNullable(NULL))) = 2)) SETTINGS optimize_and_compare_chain = 1; -- { serverError ILLEGAL_COLUMN } +-- optimize_redundant_comparisons path (tryOptimizeAndCompareNotEqualsChain), independent of the chain setting: +SELECT tuple((materialize(toNullable(1::Int32)) = 1) AND (assumeNotNull(materialize(toNullable(NULL))) = 2)) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((materialize(toNullable(1::Int32)) != 1) AND (assumeNotNull(materialize(toNullable(NULL))) != 2)) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } + +-- 2) The optimizer still runs on such a `Nothing`-collapsed AND: the directly-Nullable operand is +-- kept as-is (the new opaque-filter fallback) while a redundant non-null comparison in the SAME +-- AND is still folded. In `x > 3 AND x > 5`, `x > 3` is redundant and pruned, so exactly one +-- `greater` survives when enabled against both when disabled. Pin both counts: a relative +-- comparison also holds when the predicates disappear entirely. +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = 2)) SETTINGS optimize_redundant_comparisons = 1) WHERE explain LIKE '%function_name: greater,%'; +SELECT count() = 2 FROM (EXPLAIN QUERY TREE SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = 2)) SETTINGS optimize_redundant_comparisons = 0) WHERE explain LIKE '%function_name: greater,%'; +-- The exact predicate that used to hit the assertion is `equals(...) -> Nullable(Nothing)`; assert it +-- survives (exactly one such node) rather than being silently dropped. The chain's other `equals` +-- returns `Nothing`, so match the `Nullable(Nothing)` result type specifically. +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = 2)) SETTINGS optimize_redundant_comparisons = 1) WHERE explain LIKE '%function_name: equals, function_type: ordinary, result_type: Nullable(Nothing)%'; + +-- 3) Correctness guard: a genuine (non-collapsed) AND-compare chain over a Nullable column must still +-- be pruned/folded correctly, i.e. the optimization result matches the unoptimized one, and NULLs +-- are excluded by the comparisons. `optimize_and_compare_chain` / `optimize_redundant_comparisons` +-- must not change results. +DROP TABLE IF EXISTS t_and_chain_nullable; +CREATE TABLE t_and_chain_nullable (x Nullable(Int32)) ENGINE = Memory; +INSERT INTO t_and_chain_nullable VALUES (1), (5), (NULL), (10); +SELECT groupArray(x) FROM (SELECT x FROM t_and_chain_nullable WHERE (x > 3) AND (x > 5) ORDER BY x SETTINGS optimize_and_compare_chain = 1); +SELECT groupArray(x) FROM (SELECT x FROM t_and_chain_nullable WHERE (x > 3) AND (x > 5) ORDER BY x SETTINGS optimize_and_compare_chain = 0); +SELECT groupArray(x) FROM (SELECT x FROM t_and_chain_nullable WHERE (x != 1) AND (x != 5) ORDER BY x SETTINGS optimize_redundant_comparisons = 1); +SELECT groupArray(x) FROM (SELECT x FROM t_and_chain_nullable WHERE (x != 1) AND (x != 5) ORDER BY x SETTINGS optimize_redundant_comparisons = 0); +DROP TABLE t_and_chain_nullable; +-- The checks above compare results, which stay equal even if `optimize_and_compare_chain` stops +-- deriving anything. Pin the derivation itself with exact node counts: `a < b AND b < 5` gains the +-- transitive `a < 5`, so the enabled tree holds exactly 3 `less` nodes against 2 when disabled. +SELECT count() = 3 FROM (EXPLAIN QUERY TREE SELECT a, b FROM values('a Int32, b Int32', (1, 2), (4, 9)) WHERE (a < b) AND (b < 5) SETTINGS optimize_and_compare_chain = 1) WHERE explain ILIKE '%function_name: less,%'; +SELECT count() = 2 FROM (EXPLAIN QUERY TREE SELECT a, b FROM values('a Int32, b Int32', (1, 2), (4, 9)) WHERE (a < b) AND (b < 5) SETTINGS optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: less,%'; + +-- 4) In a `Nothing`-collapsed AND, an operand may be a comparison whose constant side is itself +-- NULL-valued (e.g. `expr = NULL`), found by the AST fuzzer mutating case 2's `= 2` to `= NULL`. +-- A NULL-valued constant carries no comparable value, so `tryOptimizeAndCompareNotEqualsChain` +-- must not treat it as the constant side (it used to hit `chassert(!literal->getValue().isNull())`). +-- The whole AND is still `Nothing`-typed, so these queries fail with a normal handled exception. +-- RHS-NULL constant: +SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = NULL)) SETTINGS optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (assumeNotNull(materialize(toNullable(NULL))) != NULL)) SETTINGS optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +-- LHS-NULL constant (the other assertion branch): +SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (NULL = assumeNotNull(materialize(toNullable(NULL))))) SETTINGS optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +-- Independent of the pruning setting (the classification loop runs unconditionally): +SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (assumeNotNull(materialize(toNullable(NULL))) = NULL)) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 0; -- { serverError ILLEGAL_COLUMN } + +-- 5) Liveness of the keep-as-is mechanism for the NULL-valued-constant operand: the optimizer must +-- still run on this `Nothing`-collapsed AND (so a redundant sibling is pruned) while keeping the +-- NULL-valued `equals` operands rather than skipping classification entirely. In the tree with +-- pruning on, `x > 3` is redundant given `x > 5`, so exactly one `greater` node survives; with +-- pruning off both survive; both NULL-valued `equals` operands are kept in either case. +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = NULL)) SETTINGS optimize_redundant_comparisons = 1) WHERE explain ILIKE '%function_name: greater,%'; +SELECT count() = 2 FROM (EXPLAIN QUERY TREE SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = NULL)) SETTINGS optimize_redundant_comparisons = 0) WHERE explain ILIKE '%function_name: greater,%'; +SELECT count() = 2 FROM (EXPLAIN QUERY TREE SELECT tuple((materialize(toNullable(NULL)) = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))) = NULL)) SETTINGS optimize_redundant_comparisons = 1) WHERE explain ILIKE '%function_name: equals,%'; + +-- 6) A comparison whose RESULT is nullable must be kept as-is too, even when the raw operand type does +-- not report `isNullable`: `LowCardinality(Nullable(T))` (nested nullability) and the NULL-capable +-- carriers `Dynamic` / `Variant` all yield a nullable comparison result. They used to slip past the +-- guard, fold the contradictory `x = 1 AND x = 2`, and change the handled exception depending on the +-- setting. The error must now be the same regardless of the pruning / chain settings. +SET allow_suspicious_low_cardinality_types = 1; +SET allow_experimental_dynamic_type = 1; +SET allow_experimental_variant_type = 1; +SELECT tuple((x = 1) AND (x = 2) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x LowCardinality(Nullable(Int32))', NULL) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 0; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((x = 1) AND (x = 2) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x LowCardinality(Nullable(Int32))', NULL) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((x = 1) AND (x = 2) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x Dynamic', NULL) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 0; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((x = 1) AND (x = 2) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x Dynamic', NULL) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((x = 1) AND (x = 2) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x Variant(Int32, String)', NULL) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 0; -- { serverError ILLEGAL_COLUMN } +SELECT tuple((x = 1) AND (x = 2) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x Variant(Int32, String)', NULL) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 1; -- { serverError ILLEGAL_COLUMN } +-- Liveness for the LC(Nullable) operand: a redundant NON-nullable sibling (`> 3` given `> 5`) is still +-- pruned to one `greater` node while the LC-nullable `equals` operand is kept, proving the optimizer +-- runs and keeps only that operand opaque rather than declining the whole collapsed AND. +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT tuple((x = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x LowCardinality(Nullable(Int32))', NULL) SETTINGS optimize_redundant_comparisons = 1) WHERE explain ILIKE '%function_name: greater,%'; +SELECT count() = 2 FROM (EXPLAIN QUERY TREE SELECT tuple((x = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x LowCardinality(Nullable(Int32))', NULL) SETTINGS optimize_redundant_comparisons = 0) WHERE explain ILIKE '%function_name: greater,%'; +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT tuple((x = 1) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) AND (assumeNotNull(materialize(toNullable(NULL))))) FROM values('x LowCardinality(Nullable(Int32))', NULL) SETTINGS optimize_redundant_comparisons = 1) WHERE explain ILIKE '%function_name: equals,%'; +-- A non-collapsed `LowCardinality(Nullable(T))` chain must still return correct results, unchanged by +-- the optimization (NULL excluded, `1`/`5` filtered out; only `10` remains). +SELECT groupArray(x) FROM (SELECT x FROM values('x LowCardinality(Nullable(Int32))', 1, 5, NULL, 10) WHERE (x != 1) AND (x != 5) ORDER BY x SETTINGS optimize_redundant_comparisons = 1); +SELECT groupArray(x) FROM (SELECT x FROM values('x LowCardinality(Nullable(Int32))', 1, 5, NULL, 10) WHERE (x != 1) AND (x != 5) ORDER BY x SETTINGS optimize_redundant_comparisons = 0); + +-- 7) The carrier the AST fuzzer keeps rediscovering on master (`Logical error: +-- '!raw_type->isNullable()'`, STID `2508-50fe`, e.g. +-- https://s3.amazonaws.com/clickhouse-test-reports/json.html?REF=master&sha=d469feea5f342065ffe8b2384d4ddf354dae3978&name_0=MasterCI&name_1=Stress%20test%20%28arm_asan_ubsan%29 ). +-- It reaches the same `addComparisonFilter` through a different route than the cases above: the +-- `Nothing`-typed operand comes from `ARRAY JOIN []` (whose element type is `Nothing`) instead of +-- `assumeNotNull(materialize(toNullable(NULL)))`, and the collapsed AND sits in a `JOIN ON` +-- section. `JOIN ON` is load-bearing - the same AND in a `WHERE` is rejected earlier, so only the +-- join expression lets a `Nothing`-typed AND reach the optimizer. +DROP TABLE IF EXISTS t_and_chain_array_join; +CREATE TABLE t_and_chain_array_join (c0 Int32) ENGINE = MergeTree() ORDER BY tuple(); +INSERT INTO t_and_chain_array_join VALUES (1), (2); + +-- `t2.c0 = a0` is `Nothing`-typed and collapses the AND's own result type, while +-- `toNullable(t2.c0) > 0` stays `Nullable(UInt8)` and reaches the pruning path. `ARRAY JOIN []` +-- produces no rows, so the query is valid and returns nothing; assert the empty result on every +-- combination of the two settings, since either entry point alone reaches the assertion. +SELECT 1 FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join AS t2 ON (t2.c0 = a0) AND (toNullable(t2.c0) > 0) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 1; +SELECT 1 FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join AS t2 ON (t2.c0 = a0) AND (toNullable(t2.c0) > 0) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 0; +SELECT 1 FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join AS t2 ON (t2.c0 = a0) AND (toNullable(t2.c0) > 0) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 1; +SELECT 1 FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join AS t2 ON (t2.c0 = a0) AND (toNullable(t2.c0) > 0) SETTINGS optimize_and_compare_chain = 0, optimize_redundant_comparisons = 0; +-- A result-only check stays green if the queries stop running the optimizer at all, so pin the +-- pruning as well: given `> 5`, the sibling `> 3` is redundant and folded away, leaving the +-- Nullable-result `greater` plus one surviving constant comparison. Both settings are pinned on +-- every query below because `clickhouse-test` randomizes `optimize_and_compare_chain`. +SELECT count() = 2 FROM (EXPLAIN QUERY TREE SELECT 1 FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join AS t2 ON (t2.c0 = a0) AND (toNullable(t2.c0) > 0) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 1) WHERE explain ILIKE '%function_name: greater,%'; +SELECT count() = 3 FROM (EXPLAIN QUERY TREE SELECT 1 FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join AS t2 ON (t2.c0 = a0) AND (toNullable(t2.c0) > 0) AND (materialize(toInt32(5)) > 3) AND (materialize(toInt32(5)) > 5) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 0) WHERE explain ILIKE '%function_name: greater,%'; +-- The exact fuzzer query: here the Nullable-result comparison is over a correlated scalar +-- subquery rather than a column. `optimize_and_compare_chain` does not gate this one - +-- `tryOptimizeAndCompareChain` skips a chain holding a correlated subquery, while +-- `tryOptimizeAndCompareNotEqualsChain` has no such guard - so it arrives only via +-- `optimize_redundant_comparisons`. A correlated subquery is not supported in a join expression, +-- so the query must report that handled exception instead of aborting. +SELECT 1 AS x FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join ON (t_and_chain_array_join.c0 = a0) AND (t_and_chain_array_join.c0 != a0) AND (0 > (SELECT t_and_chain_array_join.c0)) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 1; -- { serverError NOT_IMPLEMENTED } +SELECT 1 AS x FROM t_and_chain_array_join AS tx ARRAY JOIN [] AS a0 LEFT JOIN t_and_chain_array_join ON (t_and_chain_array_join.c0 = a0) AND (t_and_chain_array_join.c0 != a0) AND (0 > (SELECT t_and_chain_array_join.c0)) SETTINGS optimize_and_compare_chain = 1, optimize_redundant_comparisons = 0; -- { serverError NOT_IMPLEMENTED } +DROP TABLE t_and_chain_array_join; From d5e75c168af4cb73a26af21596a3292b39460406 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 12:13:34 +0000 Subject: [PATCH 021/185] Backport #121559 to 26.8: keeper: change request queue push timeout --- src/Common/ZooKeeper/ZooKeeperImpl.cpp | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/src/Common/ZooKeeper/ZooKeeperImpl.cpp b/src/Common/ZooKeeper/ZooKeeperImpl.cpp index 0e61bf81933b..039ff73ba69c 100644 --- a/src/Common/ZooKeeper/ZooKeeperImpl.cpp +++ b/src/Common/ZooKeeper/ZooKeeperImpl.cpp @@ -1785,14 +1785,17 @@ void ZooKeeper::pushRequest(RequestInfo && info) info.request->spans.maybeInitialize(KeeperSpan::ClientRequestsQueue, info.request->tracing_context.get()); - if (!requests_queue.tryPush(std::move(info), args.operation_timeout_ms)) + /// A failed push kills the session (the `catch` below calls `finalize`), so be patient here. + const UInt64 push_timeout_ms = 3 * static_cast(args.session_timeout_ms); + + if (!requests_queue.tryPush(std::move(info), push_timeout_ms)) { if (requests_queue.isFinished()) throw Exception::fromMessage(Error::ZSESSIONEXPIRED, "Session expired"); throw Exception(Error::ZOPERATIONTIMEOUT, - "Cannot push request to queue within operation timeout of {} ms", - args.operation_timeout_ms); + "Cannot push request to queue within {} ms", + push_timeout_ms); } } catch (...) From 3ebbecc896aac56072ed20b1e621ba9f8a241898 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 14:03:29 +0000 Subject: [PATCH 022/185] Backport #121065 to 26.8: Keep detecting an always-false equals chain when comparison pruning is off --- .../Passes/LogicalExpressionOptimizerPass.cpp | 40 +++++++---- .../Passes/LogicalExpressionOptimizerPass.h | 2 +- src/Core/Settings.cpp | 2 +- ...comparison_filter_optimization_1.reference | 8 +-- ...comparison_filter_optimization_2.reference | 6 +- ...omparison_filter_type_conversion.reference | 10 +-- ...lse_equals_chain_without_pruning.reference | 20 ++++++ ...ays_false_equals_chain_without_pruning.sql | 66 +++++++++++++++++++ 8 files changed, 128 insertions(+), 26 deletions(-) create mode 100644 tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.reference create mode 100644 tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.sql diff --git a/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp b/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp index 1f27cc986e5d..520da50c2124 100644 --- a/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp +++ b/src/Analyzer/Passes/LogicalExpressionOptimizerPass.cpp @@ -417,6 +417,8 @@ struct ExpressionFilters /// Excluded from the analysis: non-lossless conversions (they also veto the fold-to-false /// collapse), NaN constants, and everything when pruning is disabled. std::vector opaque_filters; + /// Index in `opaque_filters` of the first `equals` on this expression; set only when pruning is disabled. + std::optional first_equals_position; }; using ComparisonFilterMap = QueryTreeNodePtrWithHashMap; @@ -881,8 +883,8 @@ static void rebuildComparisonNode(ComparisonFilterInfo & filter, const ContextPt } /// Insert a new comparison filter for `expression` into `filter_map`. -/// When `enable_pruning` is true, performs type conversion, boundary folding, and -/// comparison against existing filters for the same expression. +/// Performs type conversion; with `enable_pruning` also boundary folding and comparison against +/// every existing filter for the expression, otherwise only against the first `equals` seen. /// Returns ALWAYS_FALSE if a contradiction is found, ALWAYS_TRUE if the condition holds /// for the column type or is implied by existing filters, or ADDED otherwise. static AddComparisonFilterResult addComparisonFilter( @@ -892,13 +894,6 @@ static AddComparisonFilterResult addComparisonFilter( bool enable_pruning, const ContextPtr & context) { - /// Pruning disabled — just store the filter without analysis. - if (!enable_pruning) - { - filter_map[expression].opaque_filters.push_back(std::move(new_filter)); - return AddComparisonFilterResult::ADDED; - } - /// A comparison with a nullable result is ambiguous under NULL and must not be pruned or folded; /// keep it as-is. Test the comparison node's result type, not the raw operand type, so nested and /// carrier-hidden nullability (e.g. `LowCardinality(Nullable)`, `Dynamic`, `Variant`) is caught. @@ -915,8 +910,11 @@ static AddComparisonFilterResult addComparisonFilter( new_filter.converted_value = tryConvertToColumnType(new_filter.constant_node, expr_type); /// Step 2: for integer columns, try boundary folding / float-literal rewriting. - if (auto result = tryFoldBoundaryOrRewriteFloatForIntColumn(new_filter, expr_type)) - return *result; + if (enable_pruning) + { + if (auto result = tryFoldBoundaryOrRewriteFloatForIntColumn(new_filter, expr_type)) + return *result; + } auto & filters = filter_map[expression]; @@ -931,6 +929,24 @@ static AddComparisonFilterResult addComparisonFilter( return AddComparisonFilterResult::ADDED; } + if (!enable_pruning) + { + auto result = AddComparisonFilterResult::ADDED; + if (new_filter.function == ComparisonFunction::EQUALS) + { + if (filters.first_equals_position) + { + if (compareComparisonFilters(filters.opaque_filters[*filters.first_equals_position], new_filter) + == ValueComparisonResult::ALWAYS_FALSE) + result = AddComparisonFilterResult::ALWAYS_FALSE; + } + else + filters.first_equals_position = filters.opaque_filters.size(); + } + filters.opaque_filters.push_back(std::move(new_filter)); + return result; + } + /// Step 3: compare against the existing equals/range filters. auto & range_filters = filters.range_filters; for (size_t i = 0; i < range_filters.size(); ++i) @@ -1889,7 +1905,7 @@ class LogicalExpressionOptimizerVisitor : public InDepthQueryTreeVisitorWithCont /** Optimize AND chains by analyzing comparison conditions on the same expression. * This method performs two things in a single pass: * - * (a) Comparison chain pruning (when `optimize_redundant_comparisons` is enabled): + * (a) Comparison chain pruning (when `optimize_redundant_comparisons` is enabled, except an always-false `equals` pair): * Given an AND expression where the same column appears in multiple comparisons * against constants (e.g. `a = 3 AND a < 5 AND a > 1`), we collect all conditions * on the same non-constant expression into a per-expression `ComparisonFilterMap`. diff --git a/src/Analyzer/Passes/LogicalExpressionOptimizerPass.h b/src/Analyzer/Passes/LogicalExpressionOptimizerPass.h index 8ffb09f5e17b..253a3614ac30 100644 --- a/src/Analyzer/Passes/LogicalExpressionOptimizerPass.h +++ b/src/Analyzer/Passes/LogicalExpressionOptimizerPass.h @@ -107,7 +107,7 @@ namespace DB * ------------------------------- * * 8. Prune redundant comparisons and detect conflicting comparison conditions on the same expression - * within AND chains. Controlled by setting `optimize_redundant_comparisons`. + * within AND chains. Controlled by `optimize_redundant_comparisons`, except an always-false `equals` pair. * Handles all six comparison operators (=, !=, <, <=, >, >=) and their combinations: * duplicate removal, contradiction detection, and range tightening. * ------------------------------- diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 8a97e1d50638..d8d0f8dd345c 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8453,7 +8453,7 @@ Allow extracting common expressions from disjunctions in WHERE, PREWHERE, ON, HA Populate constant comparison in AND chains to enhance filtering ability. Support operators `<`, `<=`, `>`, `>=`, `=` and mix of them. For example, `(a < b) AND (b < c) AND (c < 5)` would be `(a < b) AND (b < c) AND (c < 5) AND indexHint(b < 5) AND indexHint(a < 5)`. The derived comparisons are wrapped in `indexHint`: they participate in index analysis (primary key, partition key, skipping indexes) and prune the read set, but cost nothing per row and do not affect PREWHERE. A comparison derived through expressions of different tables stays executable (`(t1.a < t2.b) AND (t2.b < 5)` derives plain `t1.a < 5`): it is the only condition that can be pushed below the join, where it filters a join input the original chain cannot reach. Derived comparisons that contradict an existing condition are also added as plain conditions, so the `AND` folds to `false`. )", 0) \ DECLARE(Bool, optimize_redundant_comparisons, true, R"( -Detect conflicting and redundant comparison conditions on the same expression within AND chains. For example, `a < 1 AND a > 5` would be rewritten to `false`. +Detect conflicting and redundant comparison conditions on the same expression within AND chains. For example, `a < 1 AND a > 5` would be rewritten to `false`. A contradiction between two `equals` on the same expression (for example, `a = 1 AND a = 2`) is detected independently of this setting. )", 0) \ DECLARE(UInt64, optimize_and_compare_chain_max_hash_work, 5'000'000, R"( Work budget for the `optimize_and_compare_chain` optimization during query analysis, measured in the number of query-tree nodes hashed by `getTreeHash` (the dominant cost of this optimization). Once a query has hashed more than this many nodes while applying the optimization, it stops applying it for the rest of the query. This bounds analysis time for queries with very many or very large `AND`-chains of comparisons, where the optimization can otherwise dominate analysis while folding nothing. Stopping early is always safe: it only forgoes an optimization and never changes results. Set to `0` to disable the budget (unlimited). diff --git a/tests/queries/0_stateless/04032_and_comparison_filter_optimization_1.reference b/tests/queries/0_stateless/04032_and_comparison_filter_optimization_1.reference index a0f46fa1a31f..148993579eca 100644 --- a/tests/queries/0_stateless/04032_and_comparison_filter_optimization_1.reference +++ b/tests/queries/0_stateless/04032_and_comparison_filter_optimization_1.reference @@ -4,10 +4,10 @@ eq_eq_same SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, 3), equals(__table1.i, 3))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.i, 3)\nSETTINGS optimize_redundant_comparisons = 1 eq_eq_diff -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, 3), equals(__table1.i, 5))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 eq_eq_flip -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(3, __table1.i), equals(__table1.i, 5))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 eq_eq_cross_same 3 30 3 c y 2024-06-15 12:00:00 @@ -15,7 +15,7 @@ eq_eq_cross_same SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, 3), equals(__table1.i, toUInt8(3)))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.i, 3)\nSETTINGS optimize_redundant_comparisons = 1 eq_eq_cross_diff -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, 3), equals(__table1.i, toUInt8(5)))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 eq_eq_float_same 3 30 3 c y 2024-06-15 12:00:00 @@ -194,5 +194,5 @@ multi_expr SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(greater(__table1.i, 1), less(__table1.i, 5), greater(__table1.f, 2.), less(__table1.f, 6.))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(greater(__table1.i, 1), less(__table1.i, 5), greater(__table1.f, 2.), less(__table1.f, 6.))\nSETTINGS optimize_redundant_comparisons = 1 multi_expr_conflict -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, 3), equals(__table1.i, 5), greater(__table1.f, 1.))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 diff --git a/tests/queries/0_stateless/04032_and_comparison_filter_optimization_2.reference b/tests/queries/0_stateless/04032_and_comparison_filter_optimization_2.reference index 08f40bf45839..4e7ebdf27a6d 100644 --- a/tests/queries/0_stateless/04032_and_comparison_filter_optimization_2.reference +++ b/tests/queries/0_stateless/04032_and_comparison_filter_optimization_2.reference @@ -15,7 +15,7 @@ lc_eq_eq SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.lc, \'y\'), equals(__table1.lc, \'y\'))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE _CAST(equals(__table1.lc, \'y\'), \'UInt8\')\nSETTINGS optimize_redundant_comparisons = 1 lc_eq_conflict -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.lc, \'y\'), equals(__table1.lc, \'z\'))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 str_eq_eq 3 30 3 c y 2024-06-15 12:00:00 @@ -23,10 +23,10 @@ str_eq_eq SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.s, \'c\'), equals(__table1.s, \'c\'))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.s, \'c\')\nSETTINGS optimize_redundant_comparisons = 1 str_eq_conflict -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.s, \'c\'), equals(__table1.s, \'a\'))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 transitive_conflict -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, 3), equals(__table1.i, __table1.u), equals(__table1.u, 5), equals(__table1.i, 5), equals(__table1.u, 3))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 transitive_no_redundant SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(less(__table1.i, 3), greater(__table1.u, 3), less(__table1.u, 10))\nSETTINGS optimize_redundant_comparisons = 0 diff --git a/tests/queries/0_stateless/04032_and_comparison_filter_type_conversion.reference b/tests/queries/0_stateless/04032_and_comparison_filter_type_conversion.reference index bceb5ddeb7ed..76b4c36132c4 100644 --- a/tests/queries/0_stateless/04032_and_comparison_filter_type_conversion.reference +++ b/tests/queries/0_stateless/04032_and_comparison_filter_type_conversion.reference @@ -4,7 +4,7 @@ int_str_eq_same SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, \'3\'), equals(__table1.i, 3))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.i, \'3\')\nSETTINGS optimize_redundant_comparisons = 1 int_str_eq_diff -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.i, \'3\'), equals(__table1.i, \'5\'))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 int_str_range 3 30 3 c y 2024-06-15 12:00:00 @@ -17,7 +17,7 @@ float_int_eq_same SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.f, 3), equals(__table1.f, 3.))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.f, 3)\nSETTINGS optimize_redundant_comparisons = 1 float_int_eq_diff -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.f, 3), equals(__table1.f, 4))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 float_int_range 1 10 1.5 a x 2024-01-01 00:00:00 @@ -42,7 +42,7 @@ float_str_and_int SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.f, \'3.0\'), less(__table1.f, 5))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.f, \'3.0\')\nSETTINGS optimize_redundant_comparisons = 1 float_str_eq_diff -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.f, \'3.0\'), equals(__table1.f, \'5.0\'))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 float_str_range 1 10 1.5 a x 2024-01-01 00:00:00 @@ -57,7 +57,7 @@ dt_str_eq_same SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.dt, \'2024-06-15 12:00:00\'), equals(__table1.dt, \'2024-06-15 12:00:00\'))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.dt, \'2024-06-15 12:00:00\')\nSETTINGS optimize_redundant_comparisons = 1 dt_str_eq_diff -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.dt, \'2024-06-15 12:00:00\'), equals(__table1.dt, \'2025-01-01 00:00:00\'))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 dt_str_range 3 30 3 c y 2024-06-15 12:00:00 @@ -78,7 +78,7 @@ dt_int_eq_prune SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.dt, 1718452800), greater(__table1.dt, 1704067200))\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE equals(__table1.dt, 1718452800)\nSETTINGS optimize_redundant_comparisons = 1 dt_int_eq_conflict -SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE and(equals(__table1.dt, 1718452800), equals(__table1.dt, 1735689600))\nSETTINGS optimize_redundant_comparisons = 0 +SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 0 SELECT\n __table1.i AS i,\n __table1.u AS u,\n __table1.f AS f,\n __table1.s AS s,\n __table1.lc AS lc,\n __table1.dt AS dt\nFROM default.`04032_t` AS __table1\nWHERE 0\nSETTINGS optimize_redundant_comparisons = 1 dt_int_range 3 30 3 c y 2024-06-15 12:00:00 diff --git a/tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.reference b/tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.reference new file mode 100644 index 000000000000..880c4353ea11 --- /dev/null +++ b/tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.reference @@ -0,0 +1,20 @@ +1 +1 +1 +1 +0 +0 +[1] +[1] +1 +1 +1 +1 +0 +0 +0 +0 +1 +1 +1 +1 diff --git a/tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.sql b/tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.sql new file mode 100644 index 000000000000..4ad605f383b8 --- /dev/null +++ b/tests/queries/0_stateless/05230_always_false_equals_chain_without_pruning.sql @@ -0,0 +1,66 @@ +-- `optimize_redundant_comparisons` gates comparison-chain pruning, boundary folding and range +-- strengthening, all introduced together with the setting. Detecting that two `equals` on the same +-- expression carry different values, and folding the AND to `false`, is older: it ran unconditionally +-- before the setting existed. Leaving that fold behind the setting made `compatibility` below the +-- release that added it, and an explicit `optimize_redundant_comparisons = 0`, execute the full plan +-- of a query whose filter is always false. +-- Every query pins `optimize_and_compare_chain` (the test runner randomizes it, and it derives +-- transitive conjuncts that change node counts) and `enable_analyzer = 1` (the pass is analyzer-only). + +-- 1) The fold happens with pruning disabled. Counting `equals` nodes rather than matching a constant's +-- rendered value: 'Low' and 'Medium' also appear in the fixture's own arguments. +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT s FROM values('s String', ('Low'), ('Medium')) WHERE (s = 'Low') AND (s = 'Medium') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: equals,%'; +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT s FROM values('s String', ('Low'), ('Medium')) WHERE (s = 'Low') AND (s = 'Medium') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: equals,%'; +-- A run of identical `equals` before the conflicting one: the conflict is found against one stored +-- representative, whose slot must survive the growth of the filter list. +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT s FROM values('s String', ('Low'), ('Medium')) WHERE (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Low') AND (s = 'Medium') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: equals,%'; + +-- 2) The same fold through the reported carrier: `compatibility` below the release that added the +-- setting resolves it to `false`. +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT s FROM values('s String', ('Low'), ('Medium')) WHERE (s = 'Low') AND (s = 'Medium') SETTINGS enable_analyzer = 1, compatibility = '25.12', optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: equals,%'; + +-- 3) The reported shape: a LEFT JOIN onto an aggregated subquery, filtered by an always-false pair of +-- `equals` on the same expression. The fold must keep the aggregation from running at all, not +-- merely return no rows -- `throwIf` in the subquery's own WHERE cannot be pruned away, so the +-- query raises FUNCTION_THROW_IF_VALUE_IS_NON_ZERO if the right side is read. +DROP TABLE IF EXISTS t_left; +DROP TABLE IF EXISTS t_right; +CREATE TABLE t_left (k String, sev String) ENGINE = MergeTree ORDER BY k; +CREATE TABLE t_right (k String, sev String, v UInt32) ENGINE = MergeTree ORDER BY k; +INSERT INTO t_left SELECT toString(number), 'Low' FROM numbers(200); +INSERT INTO t_right SELECT toString(number % 100), 'Low', number FROM numbers(500); +SELECT count() FROM (SELECT m.k FROM t_left AS m LEFT JOIN (SELECT k, argMax(sev, v) AS sev FROM t_right WHERE throwIf(v >= 0, 'right side executed') GROUP BY k) AS p ON m.k = p.k WHERE (ifNull(p.sev, m.sev) = 'Low') AND (ifNull(p.sev, m.sev) = 'Medium') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0); +SELECT count() FROM (SELECT m.k FROM t_left AS m LEFT JOIN (SELECT k, argMax(sev, v) AS sev FROM t_right WHERE throwIf(v >= 0, 'right side executed') GROUP BY k) AS p ON m.k = p.k WHERE (ifNull(p.sev, m.sev) = 'Low') AND (ifNull(p.sev, m.sev) = 'Medium') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0); +DROP TABLE t_left; +DROP TABLE t_right; + +-- 4) Two `equals` conflict only when their values differ in the column's type, never merely because a +-- second `equals` arrives: `1` and `1.0` are the same Float64, so the row survives. +SELECT groupArray(x) FROM (SELECT x FROM values('x Float64', (1.0), (2.0)) WHERE (x = 1) AND (x = 1.0) ORDER BY x SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0); +SELECT groupArray(x) FROM (SELECT x FROM values('x Float64', (1.0), (2.0)) WHERE (x = 1) AND (x = 1.0) ORDER BY x SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0); + +-- 5) A conflict does not collapse an AND that also holds a comparison whose constant cannot be +-- converted: executing it raises TYPE_MISMATCH, and dropping it would make the error depend on the +-- setting. The convertible counterpart below shows the conflict is otherwise found on both settings. +SELECT count() FROM values('i Int32', (1)) WHERE (i = 1) AND (i = 2) AND (i > 'str') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0; -- { serverError TYPE_MISMATCH } +SELECT count() FROM values('i Int32', (1)) WHERE (i = 1) AND (i = 2) AND (i > 'str') SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0; -- { serverError TYPE_MISMATCH } +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT i FROM values('i Int32', (1)) WHERE (i = 1) AND (i = 2) AND (i > 0) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: equals,%'; +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT i FROM values('i Int32', (1)) WHERE (i = 1) AND (i = 2) AND (i > 0) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: equals,%'; + +-- 6) Only the equals/equals conflict is ungated. A contradiction between two ranges is part of what +-- the setting introduced and stays gated: one `less` survives with pruning off, none with it on. +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT a FROM values('a Int32', (3)) WHERE (a < 1) AND (a > 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: less,%'; +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT a FROM values('a Int32', (3)) WHERE (a < 1) AND (a > 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: less,%'; + +-- 7) `optimize_and_compare_chain` appends a contradiction it derives (`x = 3 AND x = y AND y = 5` +-- yields `x = 5`) as a plain conjunct, for this pass to fold. With pruning off that conjunct is now +-- folded too, so the derived contradiction collapses the AND; with the chain pass off nothing is +-- derived and nothing collapses. The result is empty in every combination either way. +SELECT count() FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0; +SELECT count() FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 1; +SELECT count() FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0; +SELECT count() FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 1; +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT x FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: and,%'; +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT x FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 0, optimize_and_compare_chain = 1) WHERE explain ILIKE '%function_name: and,%'; +SELECT count() = 1 FROM (EXPLAIN QUERY TREE SELECT x FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 0) WHERE explain ILIKE '%function_name: and,%'; +SELECT count() = 0 FROM (EXPLAIN QUERY TREE SELECT x FROM values('x Int32, y Int32', (3, 5), (3, 3)) WHERE (x = 3) AND (x = y) AND (y = 5) SETTINGS enable_analyzer = 1, optimize_redundant_comparisons = 1, optimize_and_compare_chain = 1) WHERE explain ILIKE '%function_name: and,%'; From 30f0fd25292a6b1c30a570a7cc7bbb660605e729 Mon Sep 17 00:00:00 2001 From: Elmi Ahmadov Date: Wed, 23 Sep 2026 14:42:48 +0000 Subject: [PATCH 023/185] Backport #114728 to 26.8: Text index: support '%' and '%' patterns for LIKE/ILIKE optimization --- .../mergetree-family/textindexes.mdx | 29 +- src/Core/Settings.cpp | 5 +- .../MergeTree/MergeTreeIndexConditionText.cpp | 142 +++++--- .../MergeTree/MergeTreeIndexConditionText.h | 1 + .../MergeTree/MergeTreeReaderTextIndex.cpp | 25 +- src/Storages/MergeTree/TextIndexAnalyzer.cpp | 23 +- src/Storages/MergeTree/TextIndexAnalyzer.h | 2 + tests/performance/text_index_like.xml | 18 +- ...6_text_index_function_like_affix.reference | 123 +++++++ .../02346_text_index_function_like_affix.sql | 338 ++++++++++++++++++ ...4050_text_index_starts_ends_with.reference | 28 +- ...04061_text_index_json_all_values.reference | 8 +- 12 files changed, 649 insertions(+), 93 deletions(-) create mode 100644 tests/queries/0_stateless/02346_text_index_function_like_affix.reference create mode 100644 tests/queries/0_stateless/02346_text_index_function_like_affix.sql diff --git a/docs/reference/engines/table-engines/mergetree-family/textindexes.mdx b/docs/reference/engines/table-engines/mergetree-family/textindexes.mdx index 2866642128ba..29a4dbc5d465 100644 --- a/docs/reference/engines/table-engines/mergetree-family/textindexes.mdx +++ b/docs/reference/engines/table-engines/mergetree-family/textindexes.mdx @@ -528,8 +528,8 @@ Search tokens that the postprocessor maps to an empty string are ignored, i.e. t | [hasAnyTokens(col, arr)](/reference/functions/regular-functions/string-search-functions#hasAnyTokens) | no (array elements are tokens as-is) | all | yes | | [hasAllTokens(col, arr)](/reference/functions/regular-functions/string-search-functions#hasAllTokens) | no (array elements are tokens as-is) | all | yes | | [hasPhrase](/reference/functions/regular-functions/string-search-functions#hasPhrase) | yes | `splitByNonAlpha`, `splitByString`, `splitByRegexp`³, `ngrams`, `asciiCJK`, `icu` | yes³ | -| [startsWith](/reference/functions/regular-functions/string-functions#startsWith) | yes | `splitByNonAlpha`, `ngrams`, `sparseGrams`, `asciiCJK` | yes | -| [endsWith](/reference/functions/regular-functions/string-functions#endsWith) | yes | `splitByNonAlpha`, `ngrams`, `sparseGrams`, `asciiCJK` | yes | +| [startsWith](/reference/functions/regular-functions/string-functions#startsWith) | yes⁴ | `splitByNonAlpha`, `ngrams`, `sparseGrams`, `asciiCJK`, `array`⁴ | yes⁴ | +| [endsWith](/reference/functions/regular-functions/string-functions#endsWith) | yes⁴ | `splitByNonAlpha`, `ngrams`, `sparseGrams`, `asciiCJK`, `array`⁴ | yes⁴ | | [like](/reference/functions/regular-functions/string-search-functions#like) | yes¹ | `splitByNonAlpha`, `ngrams`, `sparseGrams`, `asciiCJK`¹ | yes¹ | | [match](/reference/functions/regular-functions/string-search-functions#match) | yes¹ | `splitByNonAlpha`, `ngrams`, `sparseGrams`, `asciiCJK`¹ | yes¹ | | [ilike](/reference/functions/regular-functions/string-search-functions#like) | yes² (`lower`/`upper` only) | `splitByNonAlpha`, `array`² | no² | @@ -542,14 +542,19 @@ Search tokens that the postprocessor maps to an empty string are ignored, i.e. t | [hasAll](/reference/functions/regular-functions/array-functions#hasAll) | yes | `array` | yes | ¹ `LIKE` and `match` use direct read as a hint for the listed tokenizers, otherwise they fall back to brute-force scan. -`LIKE` additionally supports a *direct read (without hint)* (enabled via `use_text_index_like_evaluation_by_dictionary_scan`) for `splitByNonAlpha` and `array` tokenizers without preprocessor or postprocessor. +`LIKE` additionally supports evaluation by a dictionary scan (enabled via `use_text_index_like_evaluation_by_dictionary_scan`) for `splitByNonAlpha` and `array` tokenizers without preprocessor or postprocessor. +A `%value%` pattern is then a *direct read (without hint)*, while `value%` and `%value` patterns remain a hint, see [LIKE/ILIKE queries](#like-ilike-queries-perf). -² `ILIKE` is only supported via direct read (without hint) (`use_text_index_like_evaluation_by_dictionary_scan = 1`, `splitByNonAlpha` or `array` tokenizer). -There is no fallback to using the index as a hint: if the setting is disabled or the tokenizer is not in the supported set, the index is not used for `ILIKE`. +² `ILIKE` is only supported via evaluation by a dictionary scan (`use_text_index_like_evaluation_by_dictionary_scan = 1`, `splitByNonAlpha` or `array` tokenizer). +There is no fallback to using the index as a hint for patterns the dictionary scan does not support: if the setting is disabled or the tokenizer is not in the supported set, the index is not used for `ILIKE`. The preprocessor, if present, must be `lower` or `upper`; postprocessors are not supported. ³ `hasPhrase` on a `splitByRegexp` text index does **not** support a postprocessor: the combination is rejected with an exception, because the postprocessor row-level rewrite assumes whitespace-splitting `splitByNonAlpha`-style tokens. Without a postprocessor, `splitByRegexp` is fully supported by `hasPhrase`. +⁴ `startsWith` and `endsWith` search the complete tokens of the needle, and the token at the open end of the needle is incomplete because the value continues there: `startsWith(col, 'ClickHouse is')` searches the token `ClickHouse`, while `startsWith(col, 'ClickHouse')` has no complete token to search. +The latter is instead evaluated by a dictionary scan (`use_text_index_like_evaluation_by_dictionary_scan = 1`, `splitByNonAlpha` or `array` tokenizer, no preprocessor or postprocessor), which is also the path taken by `col LIKE 'ClickHouse%'` because the analyzer pass [optimize_rewrite_like_perfect_affix](/reference/settings/session-settings/optimize-rewrite#optimize_rewrite_like_perfect_affix) rewrites it into `startsWith`. +See [LIKE/ILIKE queries](#like-ilike-queries-perf). + **Experimental: Support phrase search argument (optional)**. Experimental parameter `support_phrase_search` (default: `0`) controls whether the index stores token positions. @@ -1393,10 +1398,20 @@ This ordering enables skipping even more data granules than the granules skipped ### LIKE/ILIKE queries {#like-ilike-queries-perf} -When a LIKE/ILIKE query pattern is `%%` and the text index tokenizer is `splitByNonAlpha` or `array`, ClickHouse leverages the inverted index to speed up LIKE/ILIKE queries significantly. To achieve that, ClickHouse scans the inverted index dictionary instead of a full-table scan to find the matching pattern. +When a LIKE/ILIKE query pattern is `%%`, `%` or `%` and the text index tokenizer is `splitByNonAlpha` or `array`, ClickHouse leverages the inverted index to speed up LIKE/ILIKE queries significantly. To achieve that, ClickHouse scans the inverted index dictionary instead of a full-table scan to find the matching pattern. + +How the result of the dictionary scan is used depends on where the needle is anchored: +- `%value%` matches a row if and only if it matches one of the row's tokens, so the index decides the query on its own: a [direct read (without hint)](#direct-read) that removes the original predicate. +- `value%` and `%value` anchor the needle at the whole value, whereas the dictionary scan can only anchor it at a token, so the scan returns a superset of the matching rows and is used as a [direct read as a hint](#direct-read). + +The same dictionary scan serves `startsWith(col, 'value')` and `endsWith(col, 'value')` when the needle has no complete token to search. +This is the path most `value%` and `%value` patterns actually take, because the analyzer pass [optimize_rewrite_like_perfect_affix](/reference/settings/session-settings/optimize-rewrite#optimize_rewrite_like_perfect_affix) (enabled by default) rewrites `col LIKE 'value%'` into `startsWith(col, 'value')` and `col LIKE '%value'` into `endsWith(col, 'value')`. +Needles that span several tokens keep using the complete tokens of the needle and do not need a dictionary scan. When the optimization is enabled, LIKE/ILIKE queries should be significantly faster than a full-table scan. However, when the pattern matches most dictionary tokens, the performance can be worse compared to a full-table scan. Luckily, there is a fallback mechanism to prevent that. +The speed-up of a `value%` or `%value` pattern comes from skipping granules, so it depends on how the matching rows are laid out. Needles that match rows in every granule prune nothing, and the query pays for the dictionary scan on top of the full scan it would have done anyway. Disable [use_text_index_like_evaluation_by_dictionary_scan](/reference/settings/session-settings/use-text#use_text_index_like_evaluation_by_dictionary_scan) for such workloads. + The optimization is controlled by a setting: - [use_text_index_like_evaluation_by_dictionary_scan](/reference/settings/session-settings/use-text#use_text_index_like_evaluation_by_dictionary_scan) @@ -1404,7 +1419,7 @@ The fallback mechanism is controlled by two settings: - [text_index_like_min_pattern_length](/reference/settings/session-settings/text-index#text_index_like_min_pattern_length) - [text_index_like_max_postings_to_read](/reference/settings/session-settings/text-index#text_index_like_max_postings_to_read) -This optimization supports only functions `like` and `ilike`. +This optimization supports only functions `like`, `ilike`, `startsWith`, and `endsWith`. ### Trivial count queries {#count-queries-perf} diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 8a97e1d50638..fa6107178cb1 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8899,9 +8899,12 @@ Maximal selectivity of the filter to use the hint built from the inverted text i )", 0) \ DECLARE(Bool, use_text_index_like_evaluation_by_dictionary_scan, true, R"( Enable evaluation of LIKE/ILIKE queries by scanning the inverted text index dictionary. + +The accelerated patterns are `%value%`, `value%` and `%value`, as well as the `startsWith` and `endsWith` calls that `optimize_rewrite_like_perfect_affix` rewrites into `value%` and `%value`. )", 0) \ DECLARE(UInt64, text_index_like_min_pattern_length, 4, R"( -Minimum length of the alphanumeric needle in a LIKE/ILIKE pattern required to use the text index LIKE evaluation by the dictionary scan. +Minimum length of the alphanumeric needle in a LIKE/ILIKE pattern, or of a `startsWith`/`endsWith` needle, +required to use the text index LIKE evaluation by the dictionary scan. Patterns shorter than this threshold match too many dictionary tokens and are skipped to avoid expensive scans. Requires `use_text_index_like_evaluation_by_dictionary_scan` to be enabled. diff --git a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp index 548148d12bd7..d8c5094ce71c 100644 --- a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp @@ -382,7 +382,6 @@ namespace /// A query in `All` mode folds postings by intersection, so a partially folded /// posting list is a superset of the result and can be used for pruning right away. bool queryMayBeTrueInRange( - const TextSearchQuery & query, const TextIndexAnalyzer::QueryBuilder & query_builder, const std::optional & current_range, TextSearchMode search_mode) @@ -391,8 +390,8 @@ bool queryMayBeTrueInRange( if (query_builder.is_failed) return false; - /// Pattern bypass means analysis is incomplete, so conservatively return true. - if (query_builder.is_bypassed && !query.getPatterns().empty()) + /// An incomplete scan may have missed matching tokens, so nothing can be pruned. + if (query_builder.is_analysis_incomplete) return true; if (!current_range.has_value()) @@ -424,7 +423,7 @@ bool hasAnyTokensInRange(const TextSearchQuery & query, const TextIndexAnalyzer: if (query.getTokens().empty()) return false; - return queryMayBeTrueInRange(query, query_builder, current_range, TextSearchMode::Any); + return queryMayBeTrueInRange(query_builder, current_range, TextSearchMode::Any); } bool hasAnyPatternsInRange(const TextSearchQuery & query, const TextIndexAnalyzer::QueryBuilder & query_builder, const std::optional & current_range) @@ -432,7 +431,7 @@ bool hasAnyPatternsInRange(const TextSearchQuery & query, const TextIndexAnalyze if (query.getPatterns().empty()) return false; - return queryMayBeTrueInRange(query, query_builder, current_range, TextSearchMode::Any); + return queryMayBeTrueInRange(query_builder, current_range, TextSearchMode::Any); } bool hasAllTokensOrEmptyInRange(const TextSearchQuery & query, const TextIndexAnalyzer::QueryBuilder & query_builder, const std::optional & current_range) @@ -440,7 +439,7 @@ bool hasAllTokensOrEmptyInRange(const TextSearchQuery & query, const TextIndexAn if (query.getTokens().empty()) return true; - return queryMayBeTrueInRange(query, query_builder, current_range, TextSearchMode::All); + return queryMayBeTrueInRange(query_builder, current_range, TextSearchMode::All); } bool hasAllTokensInRange(const TextSearchQuery & query, const TextIndexAnalyzer::QueryBuilder & query_builder, const std::optional & current_range) @@ -448,7 +447,7 @@ bool hasAllTokensInRange(const TextSearchQuery & query, const TextIndexAnalyzer: if (query.getTokens().empty()) return false; - return queryMayBeTrueInRange(query, query_builder, current_range, TextSearchMode::All); + return queryMayBeTrueInRange(query_builder, current_range, TextSearchMode::All); } } @@ -786,12 +785,21 @@ VectorWithMemoryTracking MergeTreeIndexConditionText::stringLikeToTokens return VectorWithMemoryTracking(unique_tokens.begin(), unique_tokens.end()); } -std::vector MergeTreeIndexConditionText::stringLikeToPatterns(const Field & field, bool case_insensitive) const +namespace +{ + +/// '%needle%' is anchored the same way in a token as in the whole value; an affix is not. +bool isInfixPattern(const String & pattern) { - /// Only handles the pure '%value%' form: one leading '%', a non-empty alphanumeric token immediately following, - /// then one trailing '%' immediately after the token, and nothing else. - /// Returns a single-element vector on success, empty on anything more complex. - /// Only this form is eligible for direct read mode. + return pattern.starts_with('%') && pattern.ends_with('%'); +} + +} + +std::vector +MergeTreeIndexConditionText::stringLikeToPatterns(const Field & field, bool case_insensitive) const +{ + /// Handles '%value%', 'value%' and '%value' with an alphanumeric needle; rejects anything more complex. const String value = preprocessor->processConstant(field.safeGet()); if (value.empty()) @@ -801,35 +809,22 @@ std::vector MergeTreeIndexConditionText::stringLikeT const size_t length = value.size(); size_t pos = 0; - const auto is_token_char = [](unsigned char c) { return isASCII(c) && isAlphaNumericASCII(static_cast(c)); }; - - const size_t min_pattern_length = getContext()->getSettingsRef()[Setting::text_index_like_min_pattern_length]; - - /// Must start with at least one '%'. - if (data[pos] != '%') - return {}; - while (pos < length && data[pos] == '%') ++pos; + const bool has_leading_wildcard = pos > 0; + /// Alphanumeric content must follow immediately. - if (pos >= length || !is_token_char(static_cast(data[pos]))) + if (pos >= length || !isAlphaNumericASCII(data[pos])) return {}; const size_t start = pos; - while (pos < length && is_token_char(static_cast(data[pos]))) + while (pos < length && isAlphaNumericASCII(data[pos])) ++pos; const size_t end = pos; - - /// Reject short needles: it might match too many dictionary tokens. - if (end - start < min_pattern_length) - return {}; - - /// Trailing '%' must follow immediately after the content. - if (pos >= length || data[pos] != '%') - return {}; + const bool has_trailing_wildcard = pos < length && data[pos] == '%'; while (pos < length && data[pos] == '%') ++pos; @@ -838,10 +833,20 @@ std::vector MergeTreeIndexConditionText::stringLikeT if (pos < length) return {}; + /// A pattern without wildcards is a plain equality comparison, which is served by the exact tokens path. + if (!has_leading_wildcard && !has_trailing_wildcard) + return {}; + + /// Reject short needles: they might match too many dictionary tokens. + if (end - start < getContext()->getSettingsRef()[Setting::text_index_like_min_pattern_length]) + return {}; + String pattern; - pattern += '%'; + if (has_leading_wildcard) + pattern += '%'; pattern.append(data + start, end - start); - pattern += '%'; + if (has_trailing_wildcard) + pattern += '%'; std::vector patterns; if (case_insensitive) @@ -958,6 +963,11 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( const bool is_array_tokenizer = (tokenizer->getType() == ITokenizer::Type::Array); + const auto * index_column_dag_node = index_column_node.getDAGNode(); + const DataTypePtr index_column_type = index_column_dag_node ? index_column_dag_node->result_type : nullptr; + /// A UInt8 virtual column cannot carry the NULL a predicate returns for a NULL value, which NOT flips to true. + const bool affix_patterns_allowed = index_column_type && !isNullableOrLowCardinalityNullable(index_column_type); + /// like/ilike optimization is only supported for splitByNonAlpha and array tokenizers. static const std::unordered_set like_optimization_supported_tokenizers = { ITokenizer::Type::SplitByNonAlpha, @@ -1256,20 +1266,39 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( out.text_search_queries.emplace_back(std::make_shared(function_name, TextSearchMode::All, direct_read_mode, std::move(tokens))); return true; } - if (function_name == "startsWith" && tokenizer->supportsStringLike()) + if (function_name == "startsWith" || function_name == "endsWith") { if (!value_data_type.isStringOrFixedString()) return false; - auto tokens = substringToTokens(value_field, true, false); - out.function = RPNElement::FUNCTION_EQUALS; - out.text_search_queries.emplace_back(std::make_shared(function_name, TextSearchMode::All, direct_read_mode, std::move(tokens))); - return true; - } - if (function_name == "endsWith" && tokenizer->supportsStringLike()) - { - if (!value_data_type.isStringOrFixedString()) + + const bool is_prefix = (function_name == "startsWith"); + + /// A needle inside a single token yields no complete token below, so evaluate it as `LIKE 'needle%'`. + /// Safe for a literal needle: one that would need LIKE escaping is not alphanumeric and is rejected. + if (like_optimization_supported_tokenizers.contains(tokenizer->getType()) && !has_preprocessor && !has_postprocessor + && settings[Setting::use_text_index_like_evaluation_by_dictionary_scan]) + { + const auto & needle = value_field.safeGet(); + const auto affix_pattern = is_prefix ? needle + "%" : "%" + needle; + auto patterns = stringLikeToPatterns(affix_pattern, /*case_insensitive=*/ false); + if (patterns.size() == 1 && affix_patterns_allowed) + { + const auto is_exact = is_array_tokenizer && candidate_for_exact_mode; + const auto pattern_read_mode = is_exact ? TextIndexDirectReadMode::Exact : direct_read_mode; + + out.function = RPNElement::FUNCTION_LIKE; + out.text_search_queries.emplace_back( + std::make_shared( + function_name, TextSearchMode::Any, pattern_read_mode, + VectorWithMemoryTracking(), std::move(patterns))); + return true; + } + } + + if (!tokenizer->supportsStringLike()) return false; - auto tokens = substringToTokens(value_field, false, true); + + auto tokens = substringToTokens(value_field, is_prefix, !is_prefix); out.function = RPNElement::FUNCTION_EQUALS; out.text_search_queries.emplace_back(std::make_shared(function_name, TextSearchMode::All, direct_read_mode, std::move(tokens))); return true; @@ -1282,7 +1311,7 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( if (like_optimization_supported_tokenizers.contains(tokenizer->getType()) && !has_preprocessor && !has_postprocessor && settings[Setting::use_text_index_like_evaluation_by_dictionary_scan]) { - /// TODO(ahmadov): Only '%foo%' pattern is eligible for direct read mode. An empty vector means the pattern is too complex. + /// TODO(ahmadov): Only the '%foo%', 'foo%' and '%foo' patterns are eligible for a dictionary scan. /// Add support for multiple patterns later with hint mode: /// 1. Handle multiple patterns e.g. %foo bar% -> postings_pattern(%foo) && postings_pattern(bar%) && regex(%foo bar%) /// 2. Handle exact tokens and patterns e.g. %foo bar baz% -> postings_exact(bar) && postings_pattern(%foo) && postings_pattern(bar%) @@ -1290,15 +1319,21 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( /// 4. Fall-back to the brute-force search for other cases for now. /// Follow-up: /// 1. Handle more complex patterns e.g. %foo%bar% -> (postings_pattern(%foo%) && postings_pattern(%bar%)) || postings_pattern(%foo%bar%) - auto patterns = stringLikeToPatterns(value_field, false); - if (patterns.size() == 1) + /// 2. Seek the sorted dictionary to the matching range of 'foo%' instead of scanning every block. + /// 3. Bypass a non-selective hint: it prunes nothing and still costs a dictionary scan. + const auto & like_pattern = value_field.safeGet(); + auto patterns = stringLikeToPatterns(value_field, /*case_insensitive=*/ false); + const bool is_infix = isInfixPattern(like_pattern); + if (patterns.size() == 1 && (is_infix || affix_patterns_allowed)) { - const auto pattern_read_mode = candidate_for_exact_mode ? TextIndexDirectReadMode::Exact : getHintOrNoneMode(); + const auto is_exact = (is_infix || is_array_tokenizer) && candidate_for_exact_mode; + const auto pattern_read_mode = is_exact ? TextIndexDirectReadMode::Exact : direct_read_mode; out.function = RPNElement::FUNCTION_LIKE; out.text_search_queries.emplace_back( std::make_shared( - function_name, TextSearchMode::Any, pattern_read_mode, VectorWithMemoryTracking(), std::move(patterns))); + function_name, TextSearchMode::Any, pattern_read_mode, + VectorWithMemoryTracking(), std::move(patterns))); return true; } } @@ -1321,15 +1356,20 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( if (has_postprocessor) return false; - auto patterns = stringLikeToPatterns(value_field, true); - if (patterns.size() == 1) + const auto & like_pattern = value_field.safeGet(); + auto patterns = stringLikeToPatterns(value_field, /*case_insensitive=*/ true); + const bool is_infix = isInfixPattern(like_pattern); + if (patterns.size() == 1 && (is_infix || affix_patterns_allowed)) { - const auto pattern_read_mode = candidate_for_exact_mode ? TextIndexDirectReadMode::Exact : getHintOrNoneMode(); + /// `getDirectReadMode` does not list `ilike`, which is served by the dictionary scan only. + const auto is_exact = (is_infix || is_array_tokenizer) && candidate_for_exact_mode; + const auto pattern_read_mode = is_exact ? TextIndexDirectReadMode::Exact : getHintOrNoneMode(); out.function = RPNElement::FUNCTION_LIKE; out.text_search_queries.emplace_back( std::make_shared( - function_name, TextSearchMode::Any, pattern_read_mode, VectorWithMemoryTracking(), std::move(patterns))); + function_name, TextSearchMode::Any, pattern_read_mode, + VectorWithMemoryTracking(), std::move(patterns))); return true; } return false; diff --git a/src/Storages/MergeTree/MergeTreeIndexConditionText.h b/src/Storages/MergeTree/MergeTreeIndexConditionText.h index 442e21d6a59d..5ac4da009fc1 100644 --- a/src/Storages/MergeTree/MergeTreeIndexConditionText.h +++ b/src/Storages/MergeTree/MergeTreeIndexConditionText.h @@ -185,6 +185,7 @@ class MergeTreeIndexConditionText final : public IMergeTreeIndexCondition, publi /// Builds the OR-list of token sets for a `match`-style regexp, folding the required substring /// into every alternative. Returns an empty list when the regexp imposes no token requirement. std::vector> regexpToTokensForQueries(const String & regexp_string) const; + /// Supports '%needle%', 'needle%' and '%needle'. See isInfixPattern for which of them is exact. std::vector stringLikeToPatterns(const Field & field, bool case_insensitive = false) const; bool tryPrepareSetForTextSearch(const RPNBuilderTreeNode & lhs, const RPNBuilderTreeNode & rhs, const String & function_name, RPNElement & out) const; diff --git a/src/Storages/MergeTree/MergeTreeReaderTextIndex.cpp b/src/Storages/MergeTree/MergeTreeReaderTextIndex.cpp index f58e8a1200c4..2bab2319f713 100644 --- a/src/Storages/MergeTree/MergeTreeReaderTextIndex.cpp +++ b/src/Storages/MergeTree/MergeTreeReaderTextIndex.cpp @@ -134,16 +134,16 @@ void MergeTreeReaderTextIndex::initializeFallbackReader(const IMergeTreeReader * /// - Pattern queries (LIKE): fallback when dictionary scan is abandoned. /// - Phrase queries (hasPhrase with Exact mode): fallback when estimated cardinality is too high /// and reading position data would be slower than evaluating directly. - bool has_fallback_candidates = condition_text->hasSearchPatterns() - || std::ranges::any_of( - search_queries, - [](const auto & search_query) - { - return search_query && search_query->getSearchMode() == TextSearchMode::Phrase - && search_query->getDirectReadMode() == TextIndexDirectReadMode::Exact; - }); + /// Only exact direct read needs it: a hint keeps the original predicate, so it can just be always true. + auto needs_fallback_for_query = [](const auto & search_query) + { + if (!search_query || search_query->getDirectReadMode() != TextIndexDirectReadMode::Exact) + return false; - if (!has_fallback_candidates) + return !search_query->getPatterns().empty() || search_query->getSearchMode() == TextSearchMode::Phrase; + }; + + if (std::ranges::none_of(search_queries, needs_fallback_for_query)) return; /// Build a fallback evaluation path. Compile each virtual column's default expression @@ -167,12 +167,7 @@ void MergeTreeReaderTextIndex::initializeFallbackReader(const IMergeTreeReader * { const auto & column = columns_to_read[i]; const auto & search_query = search_queries[i]; - if (!search_query) - continue; - - bool needs_fallback = !search_query->getPatterns().empty() - || (search_query->getSearchMode() == TextSearchMode::Phrase && search_query->getDirectReadMode() == TextIndexDirectReadMode::Exact); - if (!needs_fallback) + if (!needs_fallback_for_query(search_query)) continue; /// Compile the virtual column's default expression (the original search predicate). diff --git a/src/Storages/MergeTree/TextIndexAnalyzer.cpp b/src/Storages/MergeTree/TextIndexAnalyzer.cpp index c1bbb05fd9d9..c99d0db329e6 100644 --- a/src/Storages/MergeTree/TextIndexAnalyzer.cpp +++ b/src/Storages/MergeTree/TextIndexAnalyzer.cpp @@ -318,6 +318,7 @@ void TextIndexAnalyzer::bypassPatternQueries() { auto & query_builder = query_builders.at(query_hash); query_builder.markBypassed(); + query_builder.is_analysis_incomplete = true; for (const auto & [query_token, _] : query_builder.tokens) queries_by_token[query_token].erase(query_hash); @@ -327,7 +328,7 @@ void TextIndexAnalyzer::bypassPatternQueries() double TextIndexAnalyzer::estimateQueryCardinality(const QueryBuilder & query_builder, size_t total_rows) const { const auto & query = *query_builder.query; - chassert(!query.getTokens().empty()); + chassert(!query.getTokens().empty() || !query.getPatterns().empty()); const double n = static_cast(total_rows); switch (query.getSearchMode()) @@ -367,6 +368,20 @@ double TextIndexAnalyzer::estimateQueryCardinality(const QueryBuilder & query_bu ? 1.0 - static_cast(query_builder.postings->cardinality()) / n : 1.0; + /// A pattern query declares no tokens, it owns the ones the dictionary scan matched. + if (query.getTokens().empty()) + { + for (const auto & [token, token_info] : query_builder.tokens) + { + if (hasReadPostings(token)) + continue; + + not_in_any *= (1.0 - static_cast(token_info->cardinality) / n); + } + + return n * (1.0 - not_in_any); + } + for (const auto & token : query.getTokens()) { auto it = query_builder.tokens.find(token); @@ -404,10 +419,8 @@ void TextIndexAnalyzer::analyzeCardinalitiesAndBypassHints(double selectivity_th if (query.getDirectReadMode() != TextIndexDirectReadMode::Hint) continue; - /// Pure-pattern queries have no declared tokens at parse time; their tokens are - /// discovered dynamically during dictionary scan. Skip the cardinality check in - /// that case — it would have no inputs to work with. - if (query.getTokens().empty()) + /// A pure-pattern query is estimated from the tokens the dictionary scan discovered. + if (query.getTokens().empty() && query_builder.tokens.empty()) continue; double estimated_cardinality = estimateQueryCardinality(query_builder, total_rows); diff --git a/src/Storages/MergeTree/TextIndexAnalyzer.h b/src/Storages/MergeTree/TextIndexAnalyzer.h index 03eba3c3958c..fc199918aaeb 100644 --- a/src/Storages/MergeTree/TextIndexAnalyzer.h +++ b/src/Storages/MergeTree/TextIndexAnalyzer.h @@ -42,6 +42,8 @@ class TextIndexAnalyzer bool is_failed = false; /// Query was discarded (low-selectivity hint, pattern bypass). bool is_bypassed = false; + /// The dictionary scan stopped early, so the matched tokens are incomplete and nothing can be pruned. + bool is_analysis_incomplete = false; /// Number of tokens whose posting list has already been folded into `postings`. size_t num_read_postings = 0; /// Declared tokens (`query->getTokens`) that may still contribute to an `Any` query. diff --git a/tests/performance/text_index_like.xml b/tests/performance/text_index_like.xml index 5f38d8101570..935b025a5165 100644 --- a/tests/performance/text_index_like.xml +++ b/tests/performance/text_index_like.xml @@ -1,4 +1,9 @@ + + 0 + 4 + + CREATE TABLE tab ( @@ -14,14 +19,23 @@ INSERT INTO tab SELECT number, - if(number % 100 = 0, + trimRight(if(number < 100000, repeat('clickhouse is a fast column oriented database system ', 4), - repeat('the quick brown fox jumps over the lazy dog near the river ', 4)) + repeat('the quick brown fox jumps over the lazy dog near the river ', 4))) FROM numbers(10000000) SELECT count() FROM tab WHERE message LIKE '%clickhouse%' SELECT count() FROM tab WHERE message ILIKE '%CLICKHOUSE%' + SELECT count() FROM tab WHERE message LIKE 'clickhouse%' + SELECT count() FROM tab WHERE message ILIKE 'CLICKHOUSE%' + + SELECT count() FROM tab WHERE message LIKE '%system' + SELECT count() FROM tab WHERE message ILIKE '%SYSTEM' + + SELECT count() FROM tab WHERE startsWith(message, 'clickhouse') + SELECT count() FROM tab WHERE endsWith(message, 'system') + DROP TABLE IF EXISTS tab diff --git a/tests/queries/0_stateless/02346_text_index_function_like_affix.reference b/tests/queries/0_stateless/02346_text_index_function_like_affix.reference new file mode 100644 index 000000000000..89bb276d2e31 --- /dev/null +++ b/tests/queries/0_stateless/02346_text_index_function_like_affix.reference @@ -0,0 +1,123 @@ +Test results are same with/without the optimization +-- without optimization +[1,3] +[2,5] +[1,2,3,4,5,6] +[2,4,5,6,7] +[1,3,4,6,7] +[] +[] +[1] +[1,2,3,5] +[1] +[1,3] +[2,5] +[1] +[6] +[1,3] +[2,5] +[1,3] +[2,5] +-- with optimization +[1,3] +[2,5] +[1,2,3,4,5,6] +[2,4,5,6,7] +[1,3,4,6,7] +[] +[] +[1] +[1,2,3,5] +[1] +[1,3] +[2,5] +[1] +[6] +[1,3] +[2,5] +[1,3] +[2,5] +-- with optimization but without hints +[1,3] +[2,5] +[1,2,3,4,5,6] +[2,4,5,6,7] +[1,3,4,6,7] +[1,3] +[2,5] +Prefix and suffix patterns keep the original condition, infix patterns do not +prefix 1 1 +suffix 1 1 +prefix, no rewrite 1 1 +suffix, no rewrite 1 1 +infix 1 0 +ilike prefix 1 1 +Needles shorter than text_index_like_min_pattern_length are not evaluated by a dictionary scan +prefix 0 1 +[1,3] +[1] +Text index analysis +-- Prefix pattern should choose 1 part and 1024 granules out of 4 parts and 4096 granules +Description: text GRANULARITY 100000000 +Parts: 1/4 +Granules: 1024/4096 +-- Suffix pattern should choose 3 parts and 3072 granules out of 4 parts and 4096 granules +Description: text GRANULARITY 100000000 +Parts: 3/4 +Granules: 3072/4096 +-- Prefix pattern with a non-existent token should choose none +Description: text GRANULARITY 100000000 +Parts: 0/4 +Granules: 0/4096 +1024 +1024 +2048 +2048 +Test results are same with/without the optimization with array tokenizer +-- without optimization +[1,3] +[3] +[2,4] +[1,3] +[3] +[1,2,3,4] +[3] +-- with optimization +[1,3] +[3] +[2,4] +[1,3] +[3] +[1,2,3,4] +[3] +With the array tokenizer a token is the whole value, so an affix is exact +prefix 1 0 +suffix 1 0 +ilike prefix 1 0 +A nullable value is left to the original condition +[1,3] +[1,3] +[2] +[2] +[1,2] +[1,2] +[] +[] +[2] +[2] +[] +[] +An affix hint that prunes nothing is discarded, granule pruning is kept +1000 +0 +affix_hint_nonselective 0 1 1 +affix_hint_selective 1 0 1 +A nullable value reached through mapValues is left to the original condition +499 +499 +499 +499 +499 +499 +1 +1 diff --git a/tests/queries/0_stateless/02346_text_index_function_like_affix.sql b/tests/queries/0_stateless/02346_text_index_function_like_affix.sql new file mode 100644 index 000000000000..bf57731ff680 --- /dev/null +++ b/tests/queries/0_stateless/02346_text_index_function_like_affix.sql @@ -0,0 +1,338 @@ +-- Tags: no-parallel-replicas +-- Tests that affix LIKE/ILIKE patterns, i.e. prefix ('value%') and suffix ('%value'), use the text index as a hint. +-- By default the analyzer rewrites such patterns into startsWith/endsWith (optimize_rewrite_like_perfect_affix), +-- so both spellings are covered here. +SET explain_query_plan_default = 'legacy'; + +SET enable_analyzer = 1; +SET use_skip_indexes_on_data_read = 1; +SET query_plan_direct_read_from_text_index = 1; +SET query_plan_text_index_add_hint = 1; +-- Pinned because the queries below assert which of the two spellings the plan ends up with. +SET optimize_rewrite_like_perfect_affix = 1; + +DROP TABLE IF EXISTS tab; + +CREATE TABLE tab +( + id UInt32, + message String, + INDEX idx(message) TYPE text(tokenizer = splitByNonAlpha) +) +ENGINE = MergeTree +ORDER BY (id); + +-- The dictionary of this table contains tokens that match the prefix/suffix patterns below in rows +-- where the whole value does not match them, so a hint that is not verified would return extra rows. +INSERT INTO tab(id, message) VALUES + (1, 'foobar baz'), + (2, 'baz foobar'), + (3, 'foobarqux end'), + (4, 'end foobarqux'), + (5, 'quuxfoobar'), + (6, 'quuxfoobar tail'), + (7, 'nothing here'); + +SELECT 'Test results are same with/without the optimization'; + +SELECT '-- without optimization'; + +SET use_text_index_like_evaluation_by_dictionary_scan = 0; + +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar%'; +SELECT groupArray(id) FROM tab WHERE message NOT LIKE 'foobar%'; +SELECT groupArray(id) FROM tab WHERE message NOT LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'nonexistent%'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%nonexistent'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' AND message LIKE '%baz'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' OR message LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' AND hasToken(message, 'baz'); +SELECT groupArray(id) FROM tab WHERE startsWith(message, 'foobar'); +SELECT groupArray(id) FROM tab WHERE endsWith(message, 'foobar'); +SELECT groupArray(id) FROM tab WHERE startsWith(message, 'foobar baz'); +SELECT groupArray(id) FROM tab WHERE endsWith(message, 'quuxfoobar tail'); +SELECT groupArray(id) FROM tab WHERE message ILIKE 'FOOBAR%'; +SELECT groupArray(id) FROM tab WHERE message ILIKE '%FOOBAR'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' SETTINGS optimize_rewrite_like_perfect_affix = 0; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar' SETTINGS optimize_rewrite_like_perfect_affix = 0; + +SELECT '-- with optimization'; + +SET use_text_index_like_evaluation_by_dictionary_scan = 1; + +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar%'; +SELECT groupArray(id) FROM tab WHERE message NOT LIKE 'foobar%'; +SELECT groupArray(id) FROM tab WHERE message NOT LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'nonexistent%'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%nonexistent'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' AND message LIKE '%baz'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' OR message LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' AND hasToken(message, 'baz'); +SELECT groupArray(id) FROM tab WHERE startsWith(message, 'foobar'); +SELECT groupArray(id) FROM tab WHERE endsWith(message, 'foobar'); +SELECT groupArray(id) FROM tab WHERE startsWith(message, 'foobar baz'); +SELECT groupArray(id) FROM tab WHERE endsWith(message, 'quuxfoobar tail'); +SELECT groupArray(id) FROM tab WHERE message ILIKE 'FOOBAR%'; +SELECT groupArray(id) FROM tab WHERE message ILIKE '%FOOBAR'; +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%' SETTINGS optimize_rewrite_like_perfect_affix = 0; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar' SETTINGS optimize_rewrite_like_perfect_affix = 0; + +SELECT '-- with optimization but without hints'; + +SET query_plan_text_index_add_hint = 0; + +SELECT groupArray(id) FROM tab WHERE message LIKE 'foobar%'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%foobar%'; +SELECT groupArray(id) FROM tab WHERE message NOT LIKE 'foobar%'; +SELECT groupArray(id) FROM tab WHERE message NOT LIKE '%foobar'; +SELECT groupArray(id) FROM tab WHERE message ILIKE 'FOOBAR%'; +SELECT groupArray(id) FROM tab WHERE message ILIKE '%FOOBAR'; + +SET query_plan_text_index_add_hint = 1; + +SELECT 'Prefix and suffix patterns keep the original condition, infix patterns do not'; + +-- The columns are: whether the query plan contains a text index virtual column, and whether it still +-- evaluates the original search function. The latter is absent only for an exact direct read. +SELECT 'prefix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION startsWith(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message LIKE 'foobar%'); + +SELECT 'suffix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION endsWith(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message LIKE '%foobar'); + +SELECT 'prefix, no rewrite', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION like(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message LIKE 'foobar%' SETTINGS optimize_rewrite_like_perfect_affix = 0); + +SELECT 'suffix, no rewrite', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION like(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message LIKE '%foobar' SETTINGS optimize_rewrite_like_perfect_affix = 0); + +SELECT 'infix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION like(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message LIKE '%foobar%'); + +SELECT 'ilike prefix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION ilike(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message ILIKE 'FOOBAR%'); + +SELECT 'Needles shorter than text_index_like_min_pattern_length are not evaluated by a dictionary scan'; + +SELECT 'prefix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION startsWith(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE message LIKE 'foo%'); + +SELECT groupArray(id) FROM tab WHERE message LIKE 'foo%'; +SELECT groupArray(id) FROM tab WHERE message LIKE '%baz'; + +DROP TABLE tab; + +SELECT 'Text index analysis'; + +CREATE TABLE tab +( + id UInt32, + message String, + INDEX idx(message) TYPE text(tokenizer = splitByNonAlpha) GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY (id) +SETTINGS index_granularity = 1; + +INSERT INTO tab SELECT number, 'Hello ClickHouse' FROM numbers(1024); +INSERT INTO tab SELECT number, 'Hello World, ClickHouse is fast!' FROM numbers(1024); +INSERT INTO tab SELECT number, 'Hallo xClickHouse' FROM numbers(1024); +INSERT INTO tab SELECT number, 'ClickHousez rocks' FROM numbers(1024); + +SELECT '-- Prefix pattern should choose 1 part and 1024 granules out of 4 parts and 4096 granules'; +-- Only 'ClickHousez rocks' has a token starting with 'ClickHousez'. +SELECT trimLeft(explain) AS explain FROM ( + EXPLAIN indexes=1 + SELECT count() FROM tab WHERE message LIKE 'ClickHousez%' +) WHERE explain LIKE '%Description:%' OR explain LIKE '%Parts:%' OR explain LIKE '%Granules:%' +LIMIT 2, 3; + +SELECT '-- Suffix pattern should choose 3 parts and 3072 granules out of 4 parts and 4096 granules'; +-- 'ClickHouse' and 'xClickHouse' are tokens ending with 'ClickHouse', only 'ClickHousez' is not. +SELECT trimLeft(explain) AS explain FROM ( + EXPLAIN indexes=1 + SELECT count() FROM tab WHERE message LIKE '%ClickHouse' +) WHERE explain LIKE '%Description:%' OR explain LIKE '%Parts:%' OR explain LIKE '%Granules:%' +LIMIT 2, 3; + +SELECT '-- Prefix pattern with a non-existent token should choose none'; +SELECT trimLeft(explain) AS explain FROM ( + EXPLAIN indexes=1 + SELECT count() FROM tab WHERE message LIKE 'random%' +) WHERE explain LIKE '%Description:%' OR explain LIKE '%Parts:%' OR explain LIKE '%Granules:%' +LIMIT 2, 3; + +-- Three parts have a token starting with 'ClickHouse' but only one of them has rows starting with it, +-- so the hint must be verified by the original condition. +SELECT count() FROM tab WHERE message LIKE 'ClickHouse%'; +SELECT count() FROM tab WHERE message LIKE 'ClickHouse%' SETTINGS use_skip_indexes = 0; + +SELECT count() FROM tab WHERE message LIKE '%ClickHouse'; +SELECT count() FROM tab WHERE message LIKE '%ClickHouse' SETTINGS use_skip_indexes = 0; + +DROP TABLE tab; + +SELECT 'Test results are same with/without the optimization with array tokenizer'; + +CREATE TABLE tab +( + id UInt32, + tag String, + INDEX idx(tag) TYPE text(tokenizer = array) +) +ENGINE = MergeTree +ORDER BY (id); + +INSERT INTO tab(id, tag) VALUES + (1, 'ClickHouseServer'), + (2, 'clickhouseClient'), + (3, 'ClickHouseCloud'), + (4, 'CLICKHOUSE_SQL'); + +SELECT '-- without optimization'; + +SET use_text_index_like_evaluation_by_dictionary_scan = 0; + +SELECT groupArray(id) FROM tab WHERE tag LIKE 'ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE tag LIKE '%Cloud'; +SELECT groupArray(id) FROM tab WHERE tag NOT LIKE 'ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE startsWith(tag, 'ClickHouse'); +SELECT groupArray(id) FROM tab WHERE endsWith(tag, 'Cloud'); +SELECT groupArray(id) FROM tab WHERE tag ILIKE 'clickhouse%'; +SELECT groupArray(id) FROM tab WHERE tag ILIKE '%cloud'; + +SELECT '-- with optimization'; + +SET use_text_index_like_evaluation_by_dictionary_scan = 1; + +SELECT groupArray(id) FROM tab WHERE tag LIKE 'ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE tag LIKE '%Cloud'; +SELECT groupArray(id) FROM tab WHERE tag NOT LIKE 'ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE startsWith(tag, 'ClickHouse'); +SELECT groupArray(id) FROM tab WHERE endsWith(tag, 'Cloud'); +SELECT groupArray(id) FROM tab WHERE tag ILIKE 'clickhouse%'; +SELECT groupArray(id) FROM tab WHERE tag ILIKE '%cloud'; + +DROP TABLE tab; + +SELECT 'With the array tokenizer a token is the whole value, so an affix is exact'; + +DROP TABLE IF EXISTS tab; + +CREATE TABLE tab +( + id UInt32, + tag String, + INDEX idx(tag) TYPE text(tokenizer = array) +) +ENGINE = MergeTree +ORDER BY (id); + +INSERT INTO tab(id, tag) VALUES + (1, 'ClickHouseServer'), + (2, 'clickhouseClient'), + (3, 'ClickHouseCloud'), + (4, 'CLICKHOUSE_SQL'); + +SELECT 'prefix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION startsWith(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE tag LIKE 'ClickHouse%'); + +SELECT 'suffix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION endsWith(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE tag LIKE '%Cloud'); + +SELECT 'ilike prefix', countIf(explain LIKE '%\_\_text_index\_%') > 0, countIf(explain LIKE '%FUNCTION ilike(%') > 0 +FROM (EXPLAIN actions = 1 SELECT count() FROM tab WHERE tag ILIKE 'clickhouse%'); + +DROP TABLE tab; + +SELECT 'A nullable value is left to the original condition'; + +CREATE TABLE tab +( + id UInt32, + tag Nullable(String), + INDEX idx(tag) TYPE text(tokenizer = array) +) +ENGINE = MergeTree +ORDER BY (id); + +INSERT INTO tab(id, tag) VALUES + (1, 'ClickHouseServer'), + (2, 'clickhouseClient'), + (3, 'ClickHouseCloud'), + (4, NULL); + +SELECT groupArray(id) FROM tab WHERE tag LIKE 'ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE tag LIKE 'ClickHouse%' SETTINGS use_skip_indexes = 0; +SELECT groupArray(id) FROM tab WHERE tag NOT LIKE 'ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE tag NOT LIKE 'ClickHouse%' SETTINGS use_skip_indexes = 0; +SELECT groupArray(id) FROM tab WHERE NOT endsWith(tag, 'Cloud'); +SELECT groupArray(id) FROM tab WHERE NOT endsWith(tag, 'Cloud') SETTINGS use_skip_indexes = 0; +SELECT groupArray(id) FROM tab WHERE tag NOT ILIKE 'clickhouse%'; +SELECT groupArray(id) FROM tab WHERE tag NOT ILIKE 'clickhouse%' SETTINGS use_skip_indexes = 0; +SELECT groupArray(id) FROM tab WHERE tag NOT LIKE '%ClickHouse%'; +SELECT groupArray(id) FROM tab WHERE tag NOT LIKE '%ClickHouse%' SETTINGS use_skip_indexes = 0; +SELECT groupArray(id) FROM tab WHERE tag NOT ILIKE '%clickhouse%'; +SELECT groupArray(id) FROM tab WHERE tag NOT ILIKE '%clickhouse%' SETTINGS use_skip_indexes = 0; + +DROP TABLE tab; + +SELECT 'An affix hint that prunes nothing is discarded, granule pruning is kept'; + +CREATE TABLE tab +( + id UInt64, + message String, + INDEX idx(message) TYPE text(tokenizer = splitByNonAlpha) GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY id; + +INSERT INTO tab SELECT number, multiIf(number < 1000, 'clickhouse is fast', number < 30000, 'bank of the river', 'alpha beta gamma') FROM numbers(100000); + +SELECT count() FROM tab WHERE message LIKE 'clickhouse%' SETTINGS log_comment = 'affix_hint_selective'; +SELECT count() FROM tab WHERE message LIKE 'river%' SETTINGS log_comment = 'affix_hint_nonselective'; + +SYSTEM FLUSH LOGS query_log; + +SELECT log_comment, + max(ProfileEvents['TextIndexUseHint'] > 0) AS hint_used, + max(ProfileEvents['TextIndexDiscardHint'] > 0) AS hint_discarded, + max(read_rows < 100000) AS granules_pruned +FROM system.query_log +WHERE event_date >= yesterday() AND event_time >= now() - 600 AND current_database = currentDatabase() + AND type = 'QueryFinish' AND log_comment IN ('affix_hint_selective', 'affix_hint_nonselective') +GROUP BY log_comment +ORDER BY log_comment; + +DROP TABLE tab; + +SELECT 'A nullable value reached through mapValues is left to the original condition'; + +CREATE TABLE tab +( + id UInt64, + m Map(String, Nullable(String)), + INDEX idx mapValues(m) TYPE text(tokenizer = splitByNonAlpha) +) +ENGINE = MergeTree +ORDER BY id; + +INSERT INTO tab SELECT number, map('k', if(number = 0, 'foobar value', 'zulu yankee')) FROM numbers(500); +INSERT INTO tab VALUES (1000, map('k', NULL)); + +SELECT count() FROM tab WHERE NOT startsWith(m['k'], 'foobar'); +SELECT count() FROM tab WHERE NOT startsWith(m['k'], 'foobar') SETTINGS use_skip_indexes = 0; +SELECT count() FROM tab WHERE NOT m['k'] LIKE 'foobar%'; +SELECT count() FROM tab WHERE NOT m['k'] LIKE 'foobar%' SETTINGS use_skip_indexes = 0; +SELECT count() FROM tab WHERE NOT endsWith(m['k'], 'value'); +SELECT count() FROM tab WHERE NOT endsWith(m['k'], 'value') SETTINGS use_skip_indexes = 0; +SELECT count() FROM tab WHERE startsWith(m['k'], 'foobar'); +SELECT count() FROM tab WHERE startsWith(m['k'], 'foobar') SETTINGS use_skip_indexes = 0; + +DROP TABLE tab; diff --git a/tests/queries/0_stateless/04050_text_index_starts_ends_with.reference b/tests/queries/0_stateless/04050_text_index_starts_ends_with.reference index 51f83ff27583..6090a2c07c4d 100644 --- a/tests/queries/0_stateless/04050_text_index_starts_ends_with.reference +++ b/tests/queries/0_stateless/04050_text_index_starts_ends_with.reference @@ -6,7 +6,13 @@ Expression ((Project names + Projection)) Condition: true Parts: 3/3 Granules: 3/3 - Ranges: 3 + Skip + Name: idx_name_fts + Description: text GRANULARITY 100000000 + Condition: (mode: All; tokens: []) + Parts: 1/3 + Granules: 1/3 + Ranges: 1 2 Full-text search is now generally available Expression ((Project names + Projection)) ReadFromMergeTree (default.test_fts) @@ -15,7 +21,13 @@ Expression ((Project names + Projection)) Condition: true Parts: 3/3 Granules: 3/3 - Ranges: 3 + Skip + Name: idx_name_fts + Description: text GRANULARITY 100000000 + Condition: (mode: All; tokens: []) + Parts: 1/3 + Granules: 1/3 + Ranges: 1 2 Full-text search is now generally available Expression ((Project names + Projection)) ReadFromMergeTree (default.test_fts) @@ -112,9 +124,9 @@ Expression ((Project names + Projection)) Name: idx_name_fts Description: text GRANULARITY 100000000 Condition: (mode: All; tokens: []) - Parts: 3/3 - Granules: 3/3 - Ranges: 3 + Parts: 1/3 + Granules: 1/3 + Ranges: 1 2 Full-text search is now generally available Expression ((Project names + Projection)) ReadFromMergeTree (default.test_fts) @@ -127,9 +139,9 @@ Expression ((Project names + Projection)) Name: idx_name_fts Description: text GRANULARITY 100000000 Condition: (mode: All; tokens: []) - Parts: 3/3 - Granules: 3/3 - Ranges: 3 + Parts: 1/3 + Granules: 1/3 + Ranges: 1 2 Full-text search is now generally available Expression ((Project names + Projection)) ReadFromMergeTree (default.test_fts) diff --git a/tests/queries/0_stateless/04061_text_index_json_all_values.reference b/tests/queries/0_stateless/04061_text_index_json_all_values.reference index 494d66d4afc2..1bb69101acef 100644 --- a/tests/queries/0_stateless/04061_text_index_json_all_values.reference +++ b/tests/queries/0_stateless/04061_text_index_json_all_values.reference @@ -22,8 +22,8 @@ SELECT id FROM tab WHERE startsWith(data.key1, 'lazy') ORDER BY id 1 Description: text GRANULARITY 100000000 Condition: (mode: All; tokens: []) -Parts: 4/4 -Granules: 4/4 +Parts: 1/4 +Granules: 1/4 SELECT id FROM tab WHERE endsWith(data.key1, 'fox') ORDER BY id 0 Description: text GRANULARITY 100000000 @@ -61,8 +61,8 @@ SELECT id FROM tab WHERE startsWith(data.key1::String, 'lazy') ORDER BY id 1 Description: text GRANULARITY 100000000 Condition: (mode: All; tokens: []) -Parts: 4/4 -Granules: 4/4 +Parts: 1/4 +Granules: 1/4 SELECT id FROM tab WHERE hasToken(data.key1::String, 'quick') ORDER BY id 0 2 From 28fcece7a061b30a8c8ff7e68a46797daaf0afb0 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 15:14:56 +0000 Subject: [PATCH 024/185] Backport #121325 to 26.8: Bump `cctz` from 2026c to 2026d --- contrib/cctz | 2 +- ...imezone_name_file_prefix_rejected.reference | 1 + ...5233_timezone_name_file_prefix_rejected.sql | 18 ++++++++++++++++++ 3 files changed, 20 insertions(+), 1 deletion(-) create mode 100644 tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.reference create mode 100644 tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.sql diff --git a/contrib/cctz b/contrib/cctz index 8e694da054a9..bfd13df99e26 160000 --- a/contrib/cctz +++ b/contrib/cctz @@ -1 +1 @@ -Subproject commit 8e694da054a9a31d98392bf03ee188b04f810d0a +Subproject commit bfd13df99e26583c1a00edc5032964b01bab551b diff --git a/tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.reference b/tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.reference new file mode 100644 index 000000000000..22377f01772d --- /dev/null +++ b/tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.reference @@ -0,0 +1 @@ +1970-01-01 00:00:00 diff --git a/tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.sql b/tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.sql new file mode 100644 index 000000000000..7d2d21aa5244 --- /dev/null +++ b/tests/queries/0_stateless/05233_timezone_name_file_prefix_rejected.sql @@ -0,0 +1,18 @@ +-- Time zone names are untrusted input. `cctz` strips the `file:` prefix before it builds the path, +-- so `file:/abs/path` used to bypass the check that rejects names starting with `/` and opened an +-- arbitrary file. See https://github.com/ClickHouse/ClickHouse/issues/121367. + +SELECT toDateTime(0, 'file:UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, 'file:/usr/share/zoneinfo/UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, 'file:../zoneinfo/UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, '/usr/share/zoneinfo/UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, './UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, '~/UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, '../zoneinfo/UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, 'Etc/../Etc/UTC'); -- { serverError BAD_ARGUMENTS } +SELECT toTimeZone(toDateTime(0), 'file:UTC'); -- { serverError BAD_ARGUMENTS } +SELECT CAST(0 AS DateTime('file:UTC')); -- { serverError BAD_ARGUMENTS } +SELECT toDateTime(0, 'UTC') SETTINGS session_timezone = 'file:UTC'; -- { clientError BAD_ARGUMENTS } + +-- A regular name still works. +SELECT toDateTime(0, 'UTC'); From 4a03bb318cf3c84c914a7df4a48f33117ee99e9f Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 17:28:23 +0000 Subject: [PATCH 025/185] Backport #119585 to 26.8: Detach a patch part whose index lost its content to a zeroed block --- src/Common/FailPoint.cpp | 1 + src/Storages/MergeTree/IMergeTreeDataPart.cpp | 19 ++++ .../MergeTree/MergedBlockOutputStream.cpp | 18 ++- .../MergeTree/PatchParts/PatchPartIndex.cpp | 4 + .../tests/gtest_patch_part_index_read.cpp | 82 ++++++++++++++ ..._patch_part_empty_index_detached.reference | 7 ++ .../05229_patch_part_empty_index_detached.sh | 104 ++++++++++++++++++ 7 files changed, 234 insertions(+), 1 deletion(-) create mode 100644 src/Storages/MergeTree/PatchParts/tests/gtest_patch_part_index_read.cpp create mode 100644 tests/queries/0_stateless/05229_patch_part_empty_index_detached.reference create mode 100755 tests/queries/0_stateless/05229_patch_part_empty_index_detached.sh diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 1cc541d93098..15d565cee3fb 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -75,6 +75,7 @@ static struct InitFiu ONCE(rmt_lightweight_update_sleep_after_block_allocation) \ ONCE(rmt_merge_task_sleep_in_prepare) \ ONCE(merge_tree_refresh_parts_throw_once) \ + REGULAR(patch_part_index_write_empty) \ ONCE(s3_read_buffer_throw_expired_token) \ ONCE(s3_send_request_throw_expired_token) \ REGULAR(s3_read_inject_etag_mismatch) \ diff --git a/src/Storages/MergeTree/IMergeTreeDataPart.cpp b/src/Storages/MergeTree/IMergeTreeDataPart.cpp index 81ff0ca34e5b..d22df1d1f9ae 100644 --- a/src/Storages/MergeTree/IMergeTreeDataPart.cpp +++ b/src/Storages/MergeTree/IMergeTreeDataPart.cpp @@ -1516,6 +1516,17 @@ void IMergeTreeDataPart::loadColumnsChecksumsIndexes(bool require_columns_checks if (auto * constant_granularity = dynamic_cast(index_granularity.get())) constant_granularity->fixFromRowsCount(rows_count); + /// A patch part that holds rows names the parts it patches, so an index without source + /// parts is not a patch that applies to nothing - it is a file that lost its content. + /// Failing here is what keeps the acknowledged update recoverable: an empty index reports + /// data version 0, so `clearUnusedPatchParts` would find the patch materialized everywhere + /// and delete the only copy of it. An empty index belongs to an empty part alone (the + /// covering parts `cloneEmpty` creates). + if (info.isPatch() && rows_count > 0 && patch_part_index && patch_part_index->empty()) + throw Exception(ErrorCodes::CORRUPTED_DATA, + "Patch part {} has {} rows, but its index in {} references no source parts", + name, rows_count, PatchPartIndex::FILENAME); + loadExistingRowsCount(); /// Must be called after loadRowsCount() as it uses the value of `rows_count`. loadPartitionAndMinMaxIndex(); @@ -2001,7 +2012,15 @@ void IMergeTreeDataPart::loadPatchPartIndex() return; if (auto in = readFileIfExists(PatchPartIndex::FILENAME)) + { patch_part_index = PatchPartIndex::readBinary(*in); + + /// The file holds nothing but this index, so bytes left over mean its content is not what was + /// written. One corruption shape makes this check the difference between a loud and a silent + /// failure: a zeroed block parses as an index of format version `V1` with no source parts at + /// all, and everything after those nine bytes would otherwise be ignored. + assertEOF(*in); + } else throw Exception(ErrorCodes::CORRUPTED_DATA, "Missing file {} in patch part {}", PatchPartIndex::FILENAME, name); } diff --git a/src/Storages/MergeTree/MergedBlockOutputStream.cpp b/src/Storages/MergeTree/MergedBlockOutputStream.cpp index 87693b515ecc..167c6a53ef73 100644 --- a/src/Storages/MergeTree/MergedBlockOutputStream.cpp +++ b/src/Storages/MergeTree/MergedBlockOutputStream.cpp @@ -6,6 +6,7 @@ #include #include #include +#include #include #include #include @@ -29,6 +30,11 @@ namespace MergeTreeSetting extern const MergeTreeSettingsBool enable_index_granularity_compression; } +namespace FailPoints +{ + extern const char patch_part_index_write_empty[]; +} + MergedBlockOutputStream::MergedBlockOutputStream( const MergeTreeMutableDataPartPtr & data_part, MergeTreeSettingsPtr data_settings, @@ -380,9 +386,19 @@ MergedBlockOutputStream::WrittenFiles MergedBlockOutputStream::finalizePartOnDis /// throws `CORRUPTED_DATA` otherwise, including for empty covering parts. if (new_part->info.isPatch()) { + /// Writes an index without source parts, which is the corruption shape the load path + /// rejects: a patch part that holds rows but names no part it patches. Only for tests. + bool write_empty_index = false; + fiu_do_on(FailPoints::patch_part_index_write_empty, { write_empty_index = true; }); + write_hashed_file(PatchPartIndex::FILENAME, [&](auto & buffer) { - new_part->getPatchPartIndex().writeBinary(buffer); + const auto & patch_part_index = new_part->getPatchPartIndex(); + + if (write_empty_index) + patch_part_index.cloneEmpty().writeBinary(buffer); + else + patch_part_index.writeBinary(buffer); }); } } diff --git a/src/Storages/MergeTree/PatchParts/PatchPartIndex.cpp b/src/Storages/MergeTree/PatchParts/PatchPartIndex.cpp index 61bf1bb38a49..2ebcc2b6f0cb 100644 --- a/src/Storages/MergeTree/PatchParts/PatchPartIndex.cpp +++ b/src/Storages/MergeTree/PatchParts/PatchPartIndex.cpp @@ -271,6 +271,10 @@ PatchPartIndex PatchPartIndex::readBinary(ReadBuffer & in) } res.buildSourcePartsByVersion(); + + /// Consumes exactly the bytes of the index and nothing after them: the index is also embedded in + /// larger streams, so whether anything may follow it is for the caller to decide (see + /// `IMergeTreeDataPart::loadPatchPartIndex` for the file that holds nothing else). return res; } diff --git a/src/Storages/MergeTree/PatchParts/tests/gtest_patch_part_index_read.cpp b/src/Storages/MergeTree/PatchParts/tests/gtest_patch_part_index_read.cpp new file mode 100644 index 000000000000..92c3b967e9b8 --- /dev/null +++ b/src/Storages/MergeTree/PatchParts/tests/gtest_patch_part_index_read.cpp @@ -0,0 +1,82 @@ +#include + +#include +#include +#include +#include + +using namespace DB; + +/// A patch part carries the index of the parts it patches in `source_parts.dat`, and that file holds +/// nothing else. A zeroed block of the same size used to parse as a valid index: the first byte `0` is +/// the `V1` format version, the next eight zero bytes are `num_parts = 0`, and everything after them +/// was ignored. The load path now asserts that the file ends where the index ends, which relies on +/// `readBinary` consuming exactly the bytes of the index: the same parser reads the index out of larger +/// streams (the in-memory part data exchanged between replicas), where bytes do follow it. +TEST(PatchPartIndexRead, ConsumesExactlyTheIndex) +{ + PatchPartIndex index(MergeTreePatchPartsVersion::V1, ""); + index.addSourcePart("all_1_1_0", 2); + index.addSourcePart("all_2_2_0", 3); + + String written; + { + WriteBufferFromString out(written); + index.writeBinary(out); + } + + { + ReadBufferFromString in(written); + auto read_index = PatchPartIndex::readBinary(in); + EXPECT_FALSE(read_index.empty()); + EXPECT_EQ(read_index.getMinDataVersion("all_1_1_0"), 2); + EXPECT_EQ(read_index.getMaxDataVersion("all_2_2_0"), 3); + EXPECT_TRUE(in.eof()); + } + + /// The corruption shape from the issue: the file keeps its size, but its content is gone. The + /// parser stops after the nine bytes it understands, and what the loader does next is what turns + /// the rest of the block into a loud failure instead of an accepted empty index. + String zero_filled(written.size(), '\0'); + ASSERT_GT(zero_filled.size(), 9u); + { + ReadBufferFromString in(zero_filled); + auto read_index = PatchPartIndex::readBinary(in); + EXPECT_TRUE(read_index.empty()); + EXPECT_EQ(in.count(), 9u); + EXPECT_FALSE(in.eof()); + EXPECT_ANY_THROW(assertEOF(in)); + } + + /// Bytes after a well-formed index are left in the stream for the caller. + { + /// `ReadBufferFromString` only borrows the bytes, so the string has to outlive the buffer. + String with_trailing_bytes = written + String("tail"); + ReadBufferFromString in(with_trailing_bytes); + auto read_index = PatchPartIndex::readBinary(in); + EXPECT_FALSE(read_index.empty()); + EXPECT_EQ(in.count(), written.size()); + + String rest; + readStringUntilEOF(rest, in); + EXPECT_EQ(rest, "tail"); + } +} + +/// An index without source parts is still well-formed on its own - it is what an empty covering part +/// carries - so the parser accepts it and the load path is what rejects it for a part that holds rows. +TEST(PatchPartIndexRead, AcceptsAnEmptyIndex) +{ + PatchPartIndex index(MergeTreePatchPartsVersion::V1, ""); + + String written; + { + WriteBufferFromString out(written); + index.writeBinary(out); + } + + ReadBufferFromString in(written); + auto read_index = PatchPartIndex::readBinary(in); + EXPECT_TRUE(read_index.empty()); + EXPECT_TRUE(in.eof()); +} diff --git a/tests/queries/0_stateless/05229_patch_part_empty_index_detached.reference b/tests/queries/0_stateless/05229_patch_part_empty_index_detached.reference new file mode 100644 index 000000000000..973ca647e645 --- /dev/null +++ b/tests/queries/0_stateless/05229_patch_part_empty_index_detached.reference @@ -0,0 +1,7 @@ +updated sum: 999500 +active patch parts: 1 +active patch parts after reload: 0 +detached patch parts: 1 broken-on-start +sum after reload: 499500 +reloaded sum: 999500 +reloaded detached parts: 0 diff --git a/tests/queries/0_stateless/05229_patch_part_empty_index_detached.sh b/tests/queries/0_stateless/05229_patch_part_empty_index_detached.sh new file mode 100755 index 000000000000..dc15bec6c9af --- /dev/null +++ b/tests/queries/0_stateless/05229_patch_part_empty_index_detached.sh @@ -0,0 +1,104 @@ +#!/usr/bin/env bash +# Tags: no-replicated-database, no-shared-merge-tree, no-parallel +# no-replicated-database, no-shared-merge-tree: the test reloads the table and reads +# `system.detached_parts`, which a replicated table recovers from another replica. +# no-parallel: `patch_part_index_write_empty` is server-global, so a concurrent lightweight +# `UPDATE` would write a corrupted patch part too. + +# A patch part carries the index of the parts it patches in `source_parts.dat`. An index without +# source parts belongs to an empty covering part alone: for a patch part that holds rows it means the +# file lost its content, which used to load clean and silently unapply an acknowledged update - and +# then let `clearUnusedPatchParts` delete the only copy of it, because an empty index reports data +# version 0. The part is detached as broken on load now, so the rows stay recoverable. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +set -e + +${CLICKHOUSE_CLIENT} --query " + DROP TABLE IF EXISTS t_patch_empty_index SYNC; + CREATE TABLE t_patch_empty_index (id UInt64, v UInt64) ENGINE = MergeTree ORDER BY id + SETTINGS enable_block_number_column = 1, enable_block_offset_column = 1; + + INSERT INTO t_patch_empty_index SELECT number, number FROM numbers(1000); +" + +# The failpoint is server-global, so it is cleared even if a query below fails. +trap '${CLICKHOUSE_CLIENT} --query "SYSTEM DISABLE FAILPOINT patch_part_index_write_empty"' EXIT + +${CLICKHOUSE_CLIENT} --query "SYSTEM ENABLE FAILPOINT patch_part_index_write_empty" + +${CLICKHOUSE_CLIENT} --query " + SET enable_lightweight_update = 1; + UPDATE t_patch_empty_index SET v = v + 1000 WHERE id < 500; +" + +${CLICKHOUSE_CLIENT} --query "SYSTEM DISABLE FAILPOINT patch_part_index_write_empty" + +# The update is acknowledged and visible: the in-memory index of the patch part is intact, only its +# file is not. +echo -n 'updated sum: ' +${CLICKHOUSE_CLIENT} --query "SELECT sum(v) FROM t_patch_empty_index" + +echo -n 'active patch parts: ' +${CLICKHOUSE_CLIENT} --query " + SELECT count() FROM system.parts + WHERE database = currentDatabase() AND table = 't_patch_empty_index' + AND active AND startsWith(name, 'patch-') +" + +${CLICKHOUSE_CLIENT} --query "DETACH TABLE t_patch_empty_index SYNC" +# Loading the corrupted part logs the `CORRUPTED_DATA` exception it is detached for, and the test +# harness fails a test whose client writes to stderr. +${CLICKHOUSE_CLIENT} --send_logs_level=fatal --query "ATTACH TABLE t_patch_empty_index" + +# Before the fix the patch part loaded as an index with no source parts: still active, applying to +# nothing, and eligible for cleanup. +echo -n 'active patch parts after reload: ' +${CLICKHOUSE_CLIENT} --query " + SELECT count() FROM system.parts + WHERE database = currentDatabase() AND table = 't_patch_empty_index' + AND active AND startsWith(name, 'patch-') +" + +echo -n 'detached patch parts: ' +${CLICKHOUSE_CLIENT} --query " + SELECT count(), any(reason) FROM system.detached_parts + WHERE database = currentDatabase() AND table = 't_patch_empty_index' + AND startsWith(name, 'broken') +" + +# The update is not applied any more - but it is reported and its rows are in \`detached/\`, instead +# of being deleted by the cleanup of a patch that looks materialized everywhere. +echo -n 'sum after reload: ' +${CLICKHOUSE_CLIENT} --query "SELECT sum(v) FROM t_patch_empty_index" + +# An uncorrupted patch part still loads and still applies after a reload. +${CLICKHOUSE_CLIENT} --query " + DROP TABLE IF EXISTS t_patch_good_index SYNC; + CREATE TABLE t_patch_good_index (id UInt64, v UInt64) ENGINE = MergeTree ORDER BY id + SETTINGS enable_block_number_column = 1, enable_block_offset_column = 1; + + INSERT INTO t_patch_good_index SELECT number, number FROM numbers(1000); + SET enable_lightweight_update = 1; + UPDATE t_patch_good_index SET v = v + 1000 WHERE id < 500; +" + +${CLICKHOUSE_CLIENT} --query "DETACH TABLE t_patch_good_index SYNC" +${CLICKHOUSE_CLIENT} --query "ATTACH TABLE t_patch_good_index" + +echo -n 'reloaded sum: ' +${CLICKHOUSE_CLIENT} --query "SELECT sum(v) FROM t_patch_good_index" + +echo -n 'reloaded detached parts: ' +${CLICKHOUSE_CLIENT} --query " + SELECT count() FROM system.detached_parts + WHERE database = currentDatabase() AND table = 't_patch_good_index' +" + +${CLICKHOUSE_CLIENT} --query " + DROP TABLE t_patch_empty_index SYNC; + DROP TABLE t_patch_good_index SYNC; +" From 0f439b33d1ce51d8ec483c3ca1f82db29e8fe413 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 18:56:59 +0000 Subject: [PATCH 026/185] Backport #120488 to 26.8: Fix `INSERT ... SELECT FROM url()`/`s3()` with parallel replicas inserting the data once per replica --- src/Interpreters/InterpreterInsertQuery.cpp | 29 ++++-- src/Storages/IStorageCluster.h | 9 +- src/TableFunctions/TableFunctionURL.cpp | 14 ++- .../configs/config.d/parallel_replicas.xml | 23 +++++ tests/integration/test_s3_style_link/test.py | 53 +++++++++++ ...uster_function_parallel_replicas.reference | 8 ++ ...auto_cluster_function_parallel_replicas.sh | 91 +++++++++++++++++++ 7 files changed, 218 insertions(+), 9 deletions(-) create mode 100644 tests/integration/test_s3_style_link/configs/config.d/parallel_replicas.xml create mode 100644 tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.reference create mode 100755 tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.sh diff --git a/src/Interpreters/InterpreterInsertQuery.cpp b/src/Interpreters/InterpreterInsertQuery.cpp index afbd4a86e552..9e05cac8341c 100644 --- a/src/Interpreters/InterpreterInsertQuery.cpp +++ b/src/Interpreters/InterpreterInsertQuery.cpp @@ -1107,12 +1107,30 @@ std::optional InterpreterInsertQuery::distributedWriteIntoReplica /// query will be executed on all nodes of the cluster auto src_cluster = src_storage_cluster->getCluster(local_context); - /// Actually the query doesn't change, we just serialize it to string. Strip the initiator-only - /// settings from the forwarded query text (both `changes` and `default_settings`, across the INSERT - /// and its source SELECT) so those names — including the new HTTP table-as-file settings — do not reach - /// the shards and trip `UNKNOWN_SETTING` on a rolling upgrade; the per-shard context is stripped below. + src_storage_cluster->updateExternalDynamicMetadataIfExists(local_context); + + const auto src_metadata_snapshot = src_storage_cluster->getInMemoryMetadataPtr(local_context, false); + const auto src_snapshot = src_storage_cluster->getStorageSnapshot(src_metadata_snapshot, local_context); + + /// Strip the initiator-only settings from the forwarded query text (both `changes` and `default_settings`, + /// across the INSERT and its source SELECT) so those names — including the new HTTP table-as-file settings — + /// do not reach the shards and trip `UNKNOWN_SETTING` on a rolling upgrade; the per-shard context is + /// stripped below. auto query_to_send = query.clone(); ClusterProxy::stripInitiatorOnlySettingsFromQuery(query_to_send); + + /// The source storage may have been created by `parallel_replicas_for_cluster_engines` from a plain table + /// function (`url`, `s3`, ...), while the query text still names that plain function. A node that runs + /// the forwarded query as a secondary query does not convert it again: it creates a plain storage that + /// expands the globs and reads every file on its own instead of taking its share of the read tasks from + /// the initiator, so N nodes insert the data N times. Rewrite the source the same way `IStorageCluster::read` + /// does for a `SELECT`: the function becomes its `*Cluster` variant with the cluster name argument, and the + /// structure and format arguments are added so that the nodes do not infer the schema again. + { + auto & select_to_send = query_to_send->as().select->as(); + src_storage_cluster->updateQueryToSendIfNeeded(select_to_send.list_of_selects->children.at(0), src_snapshot, local_context); + } + String query_str; { WriteBufferFromOwnString buf; @@ -1134,8 +1152,6 @@ std::optional InterpreterInsertQuery::distributedWriteIntoReplica query_context->setSettings(stripped_settings); } - src_storage_cluster->updateExternalDynamicMetadataIfExists(local_context); - std::optional filter_dag; const ActionsDAG::Node * predicate = nullptr; if (select_query) @@ -1171,7 +1187,6 @@ std::optional InterpreterInsertQuery::distributedWriteIntoReplica } } } - const auto src_metadata_snapshot = src_storage_cluster->getInMemoryMetadataPtr(local_context, false); auto extension = src_storage_cluster->getTaskIteratorExtension( predicate, filter_dag ? &*filter_dag : nullptr, local_context, src_cluster, src_metadata_snapshot); diff --git a/src/Storages/IStorageCluster.h b/src/Storages/IStorageCluster.h index 43b2d690955d..6ecfdd22dc19 100644 --- a/src/Storages/IStorageCluster.h +++ b/src/Storages/IStorageCluster.h @@ -54,9 +54,16 @@ class IStorageCluster : public IStorage const String & getClusterName() const { return cluster_name; } + /// Prepare the `SELECT ... FROM f(...)` query (`f` is a table function) for the other nodes of the cluster: add the + /// structure and format arguments so that the nodes do not infer the schema again, and turn a plain table + /// function (`url`, `s3`, ...) that `parallel_replicas_for_cluster_engines` converted into this cluster + /// storage into its `*Cluster` variant with the cluster name argument, so that the nodes take their read + /// tasks from the initiator instead of reading every file on their own. Called by `read` and by the + /// distributed `INSERT ... SELECT` in `InterpreterInsertQuery`, which forwards the query the same way. + virtual void updateQueryToSendIfNeeded(ASTPtr & /*query*/, const StorageSnapshotPtr & /*storage_snapshot*/, const ContextPtr & /*context*/) {} + protected: virtual void updateBeforeRead(const ContextPtr &) {} - virtual void updateQueryToSendIfNeeded(ASTPtr & /*query*/, const StorageSnapshotPtr & /*storage_snapshot*/, const ContextPtr & /*context*/) {} virtual void updateConfigurationIfNeeded(ContextPtr /* context */) {} diff --git a/src/TableFunctions/TableFunctionURL.cpp b/src/TableFunctions/TableFunctionURL.cpp index b58a0b842505..1e95ac17dee1 100644 --- a/src/TableFunctions/TableFunctionURL.cpp +++ b/src/TableFunctions/TableFunctionURL.cpp @@ -267,15 +267,25 @@ StoragePtr TableFunctionURL::executeImpl( /// reports the delegate's engine name and access URI, so the outer check (or the caller that /// explicitly disabled it and took over) has already covered exactly the delegate's source. if (delegate) + { + /// The query text still names `url`, while the delegate is a different backend. If the delegate + /// created its `*Cluster` storage for `parallel_replicas_for_cluster_engines`, the forwarded query + /// would be rewritten from the surface AST name into `urlCluster(...)` - a function that rejects + /// every non-HTTP scheme - and with the argument grammar of the delegate rather than of `url`. + /// Scheme dispatch is therefore resolved on this node: the delegate builds its plain storage. + ContextMutablePtr delegate_context = Context::createCopy(context); + delegate_context->setSetting("parallel_replicas_for_cluster_engines", false); + return delegate->execute( ast_function, - context, + delegate_context, table_name, std::move(cached_columns), /*use_global_context=*/false, is_insert_query, /*check_create_temporary_table=*/false, /*check_source_access=*/false); + } /// Stored columns accompany a table definition rather than an ad-hoc query, so creation and /// replay must resolve to the same storage. @@ -554,6 +564,8 @@ SELECT * FROM url('s3://clickhouse-public-datasets/hits_compatible/hits.csv'); Scheme dispatch is not yet wired through [`urlCluster`](/reference/functions/table-functions/urlCluster): a non-`http(s)` scheme passed to `urlCluster` is rejected with an error. Use the corresponding cluster function (`s3Cluster`, `azureBlobStorageCluster`, `hdfsCluster`, …) for those backends instead. +For the same reason, a dispatched `url` call is read on the node that received the query: the [parallel_replicas_for_cluster_engines](/reference/settings/session-settings/parallel-replicas#parallel_replicas_for_cluster_engines) fan-out is not applied to it. Use the corresponding cluster function directly when you want the read distributed across replicas. + ## Globs in URL {#globs-in-url} Patterns in `{ }` are used to generate a set of shards or to specify failover addresses. Supported pattern types and examples see in the description of the [remote](/reference/functions/table-functions/remote#globs-in-addresses) function. diff --git a/tests/integration/test_s3_style_link/configs/config.d/parallel_replicas.xml b/tests/integration/test_s3_style_link/configs/config.d/parallel_replicas.xml new file mode 100644 index 000000000000..f8e015f0b761 --- /dev/null +++ b/tests/integration/test_s3_style_link/configs/config.d/parallel_replicas.xml @@ -0,0 +1,23 @@ + + + + + + + http://minio1:9001/root/ + minio + ClickHouse_Minio_P@ssw0rd + + + + + + + node9000 + node9000 + node9000 + + + + diff --git a/tests/integration/test_s3_style_link/test.py b/tests/integration/test_s3_style_link/test.py index da7e64676d17..353c3a0764bf 100644 --- a/tests/integration/test_s3_style_link/test.py +++ b/tests/integration/test_s3_style_link/test.py @@ -10,6 +10,7 @@ "node", main_configs=[ "configs/config.d/minio.xml", + "configs/config.d/parallel_replicas.xml", ], user_configs=[ "configs/users.d/users.xml", @@ -139,3 +140,55 @@ def test_s3_question_mark_wildcards(started_cluster): assert result_s3_scheme == result_http_scheme assert result_s3_scheme.startswith('20\t') assert "['a1','a2']" in result_s3_scheme or "['a2','a1']" in result_s3_scheme + + +def test_url_s3_scheme_with_parallel_replicas(started_cluster): + """ + `url('s3://...')` is delegated to the `s3` backend, but the query text still names `url`. + The cluster fan-out of `parallel_replicas_for_cluster_engines` rewrites the forwarded query + from that surface name, so it used to send `urlCluster('s3://...')` - a shape `urlCluster` + rejects - both for a plain `SELECT` and for the distributed `INSERT ... SELECT`. + """ + node.query( + f""" + INSERT INTO FUNCTION s3 + ( + 'minio://data/parallel_replicas_url.csv', 'minio', '{minio_secret_key}', + 'CSV', 'a UInt32' + ) SETTINGS s3_truncate_on_insert=1 + SELECT number FROM numbers(10); + """ + ) + + parallel_replicas_settings = """ + SETTINGS cluster_for_parallel_replicas = 'parallel_replicas', + enable_parallel_replicas = 1, + max_parallel_replicas = 3, + parallel_replicas_for_cluster_engines = 1 + """ + + assert ( + node.query( + f""" + SELECT count() FROM url('s3://data/parallel_replicas_url.csv', 'CSV', 'a UInt32') + {parallel_replicas_settings} + """ + ) + == "10\n" + ) + + node.query("DROP TABLE IF EXISTS url_s3_parallel_replicas SYNC") + node.query( + "CREATE TABLE url_s3_parallel_replicas (a UInt32) ENGINE = MergeTree ORDER BY a" + ) + node.query( + f""" + INSERT INTO url_s3_parallel_replicas + SELECT * FROM url('s3://data/parallel_replicas_url.csv', 'CSV', 'a UInt32') + {parallel_replicas_settings}, parallel_distributed_insert_select = 2 + """ + ) + + # The rows must be inserted exactly once, not once per replica. + assert node.query("SELECT count() FROM url_s3_parallel_replicas") == "10\n" + node.query("DROP TABLE url_s3_parallel_replicas SYNC") diff --git a/tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.reference b/tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.reference new file mode 100644 index 000000000000..4cd3790c6b88 --- /dev/null +++ b/tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.reference @@ -0,0 +1,8 @@ +--- url --- +60 30 +--- s3 --- +60 30 +--- forwarded queries of url --- +3 3 60 +--- forwarded queries of s3 --- +3 3 60 diff --git a/tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.sh b/tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.sh new file mode 100755 index 000000000000..c956ed1d537a --- /dev/null +++ b/tests/queries/0_stateless/05218_insert_select_auto_cluster_function_parallel_replicas.sh @@ -0,0 +1,91 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, zookeeper +# Tag no-fasttest: needs MinIO and a Replicated table + +# https://github.com/ClickHouse/ClickHouse/issues/120485 +# +# `INSERT INTO SELECT * FROM url(...)` (or `s3(...)`) with parallel replicas enabled: +# `parallel_replicas_for_cluster_engines` converts the plain table function into a cluster storage on the +# initiator, so the INSERT takes the distributed `parallel_distributed_insert_select` path and forwards the +# query to every replica of the cluster. The forwarded query text still named the plain function, and a replica +# running it as a secondary query created a plain storage that expanded the globs and read every file on its +# own instead of taking its share of the read tasks from the initiator, so N replicas inserted the data N times. +# The forwarded query must name the `*Cluster` variant, the same way the SELECT path does. +# +# The three "replicas" of `test_cluster_one_shard_three_replicas_localhost` are all this server, so the +# secondary queries are visible in the local query log. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +set -eu + +S3_DIR="http://localhost:11111/test/${CLICKHOUSE_DATABASE}/05218" + +# Three files of different sizes: 10 + 20 + 30 = 60 rows, so a duplicated file changes the count. +for i in 1 2 3 +do + $CLICKHOUSE_CLIENT -q "INSERT INTO FUNCTION s3('${S3_DIR}/part_${i}.tsv', 'TSV', 'x UInt32') SELECT number FROM numbers(${i} * 10)" +done + +$CLICKHOUSE_CLIENT -q " + DROP TABLE IF EXISTS dst_05218 SYNC; + CREATE TABLE dst_05218 (x UInt32) ENGINE = ReplicatedMergeTree('/clickhouse/tables/{database}/dst_05218', 'r1') ORDER BY x; +" + +SETTINGS="enable_parallel_replicas = 1, automatic_parallel_replicas_mode = 0, max_parallel_replicas = 3, + cluster_for_parallel_replicas = 'test_cluster_one_shard_three_replicas_localhost', + parallel_replicas_for_cluster_engines = 1, parallel_distributed_insert_select = 2, log_queries = 1" + +# The query ids and the query log lookup below must isolate this run: a re-run on the same server +# must not pick up the secondary queries of a previous run. +QUERY_ID_SUFFIX="${CLICKHOUSE_DATABASE}_$(date +%s%N)_${RANDOM}" +QUERY_ID_URL="05218_url_${QUERY_ID_SUFFIX}" +QUERY_ID_S3="05218_s3_${QUERY_ID_SUFFIX}" + +echo "--- url ---" +$CLICKHOUSE_CLIENT --query_id "${QUERY_ID_URL}" -q " + INSERT INTO dst_05218 SELECT * FROM url('${S3_DIR}/part_{1..3}.tsv', 'TSV', 'x UInt32') SETTINGS ${SETTINGS}" +$CLICKHOUSE_CLIENT -q "SELECT count(), uniqExact(x) FROM dst_05218" + +echo "--- s3 ---" +$CLICKHOUSE_CLIENT -q "TRUNCATE TABLE dst_05218" +$CLICKHOUSE_CLIENT --query_id "${QUERY_ID_S3}" -q " + INSERT INTO dst_05218 SELECT * FROM s3('${S3_DIR}/part_{1..3}.tsv', 'TSV', 'x UInt32') SETTINGS ${SETTINGS}" +$CLICKHOUSE_CLIENT -q "SELECT count(), uniqExact(x) FROM dst_05218" + +# The INSERT must really have been distributed: every replica ran the forwarded INSERT, the forwarded +# query names the `*Cluster` function, and the replicas together read every file exactly once. +$CLICKHOUSE_CLIENT -q "SYSTEM FLUSH LOGS query_log" +for pair in "${QUERY_ID_URL} urlCluster" "${QUERY_ID_S3} s3Cluster" +do + query_id="${pair%% *}" + cluster_function="${pair##* }" + echo "--- forwarded queries of ${cluster_function%Cluster} ---" + $CLICKHOUSE_CLIENT -q " + WITH initial AS + ( + SELECT query_id + FROM system.query_log + WHERE current_database = currentDatabase() + AND query_id = '${query_id}' + AND is_initial_query = 1 + AND type = 'QueryFinish' + AND event_date >= yesterday() + AND event_time >= now() - INTERVAL 10 MINUTE + ) + SELECT + count() AS replicas, + countIf(query ILIKE '%${cluster_function}(%') AS cluster_function_queries, + sum(read_rows) AS rows_read_by_replicas + FROM system.query_log + WHERE initial_query_id IN (SELECT query_id FROM initial) + AND is_initial_query = 0 + AND query_kind = 'Insert' + AND type = 'QueryFinish' + AND event_date >= yesterday() + AND event_time >= now() - INTERVAL 10 MINUTE" +done + +$CLICKHOUSE_CLIENT -q "DROP TABLE dst_05218 SYNC" From 25b0c2abd71160422501bbdc832073a98ad6ef2f Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 19:21:29 +0000 Subject: [PATCH 027/185] Backport #121037 to 26.8: Fix segmentation fault in `RefreshTask::startup()` racing a parallel shutdown --- src/Storages/MaterializedView/RefreshSet.cpp | 20 ++-- src/Storages/MaterializedView/RefreshSet.h | 2 + src/Storages/MaterializedView/RefreshTask.cpp | 28 ++++- ...st_refresh_task_startup_after_shutdown.cpp | 105 ++++++++++++++++++ 4 files changed, 137 insertions(+), 18 deletions(-) create mode 100644 src/Storages/MaterializedView/tests/gtest_refresh_task_startup_after_shutdown.cpp diff --git a/src/Storages/MaterializedView/RefreshSet.cpp b/src/Storages/MaterializedView/RefreshSet.cpp index c65838c16a73..9346f0c8bd35 100644 --- a/src/Storages/MaterializedView/RefreshSet.cpp +++ b/src/Storages/MaterializedView/RefreshSet.cpp @@ -85,18 +85,14 @@ RefreshSet::RefreshSet() = default; void RefreshSet::emplace(StorageID id, std::optional inner_table_id, const std::vector & dependencies, RefreshTaskPtr task) { - { - std::lock_guard guard(mutex); - const auto iter = addTaskLocked(id, task); - RefreshTaskList::iterator inner_table_iter; - if (inner_table_id) - inner_table_iter = addInnerTableLocked(*inner_table_id, task); - addDependenciesLocked(task, dependencies); - - task->setRefreshSetHandleUnlock(Handle(this, id, inner_table_id, iter, inner_table_iter, dependencies)); - } - - notifyDependents(id); + std::lock_guard guard(mutex); + const auto iter = addTaskLocked(id, task); + RefreshTaskList::iterator inner_table_iter; + if (inner_table_id) + inner_table_iter = addInnerTableLocked(*inner_table_id, task); + addDependenciesLocked(task, dependencies); + + task->setRefreshSetHandleUnlock(Handle(this, id, inner_table_id, iter, inner_table_iter, dependencies)); } RefreshTaskList::iterator RefreshSet::addTaskLocked(StorageID id, RefreshTaskPtr task) diff --git a/src/Storages/MaterializedView/RefreshSet.h b/src/Storages/MaterializedView/RefreshSet.h index a0ac15513882..7fa91d2ed244 100644 --- a/src/Storages/MaterializedView/RefreshSet.h +++ b/src/Storages/MaterializedView/RefreshSet.h @@ -52,6 +52,8 @@ class RefreshSet RefreshSet(); + /// Caller should then also call notifyDependents, because dependent views need to know when + /// their dependencies appear/disappear. void emplace(StorageID id, std::optional inner_table_id, const std::vector & dependencies, RefreshTaskPtr task); /// Finds active refreshable view(s) by database and table name. diff --git a/src/Storages/MaterializedView/RefreshTask.cpp b/src/Storages/MaterializedView/RefreshTask.cpp index faed43facb7c..2a3d98b5af9f 100644 --- a/src/Storages/MaterializedView/RefreshTask.cpp +++ b/src/Storages/MaterializedView/RefreshTask.cpp @@ -315,13 +315,29 @@ bool RefreshTask::canCreateOrDropOtherTables() const void RefreshTask::startup() { - if (start_paused || view->getContext()->getSettingsRef()[Setting::stop_refreshable_materialized_views_on_startup]) - scheduling.stop_requested = true; - auto inner_table_id = refresh_append ? std::nullopt : std::make_optional(view->getTargetTableId()); - view->getContext()->getRefreshSet().emplace(view->getStorageID(), inner_table_id, initial_dependencies, shared_from_this()); + ContextMutablePtr context; + StorageID view_id = StorageID::createEmpty(); + { + std::lock_guard guard(mutex); - std::lock_guard guard(mutex); - scheduleRefresh(guard); + /// shutdown() is allowed to run before or during startup() (see its declaration) and nulls `view`. + if (!view) + return; + + if (start_paused || view->getContext()->getSettingsRef()[Setting::stop_refreshable_materialized_views_on_startup]) + scheduling.stop_requested = true; + context = view->getContext(); + view_id = view->getStorageID(); + auto inner_table_id = refresh_append ? std::nullopt : std::make_optional(view->getTargetTableId()); + + /// `set_handle` is not thread safe and shutdown() resets it under `mutex`. + context->getRefreshSet().emplace(view_id, inner_table_id, initial_dependencies, shared_from_this()); + + scheduleRefresh(guard); + } + + /// Outside `mutex`: notifying a dependent view locks that view's own task mutex. + context->getRefreshSet().notifyDependents(view_id); } void RefreshTask::finalizeRestoreFromBackup() diff --git a/src/Storages/MaterializedView/tests/gtest_refresh_task_startup_after_shutdown.cpp b/src/Storages/MaterializedView/tests/gtest_refresh_task_startup_after_shutdown.cpp new file mode 100644 index 000000000000..01991dac8ab3 --- /dev/null +++ b/src/Storages/MaterializedView/tests/gtest_refresh_task_startup_after_shutdown.cpp @@ -0,0 +1,105 @@ +#include + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace DB; + +namespace +{ + +/// `DatabaseCatalog` is process-wide and shared by every suite in `unit_tests_dbms`, so the +/// database name has to be unique across the whole binary, not just readable here. +constexpr auto database_name = "refresh_task_startup_after_shutdown_db"; + +struct State +{ + State(const State &) = delete; + + ContextMutablePtr context; + + static const State & instance() + { + static State state; + return state; + } + +private: + State() : context(Context::createCopy(getContext().context)) + { + tryRegisterFunctions(); + tryRegisterAggregateFunctions(); + + DatabasePtr database = std::make_shared(database_name, context); + const ColumnsDescription columns{{"x", std::make_shared()}}; + for (const auto * table_name : {"src", "target"}) + database->attachTable( + context, + table_name, + std::make_shared( + StorageID(database_name, table_name), columns, ConstraintsDescription{}, String{}, MemorySettings{}), + {}); + + DatabaseCatalog::instance().attachDatabase(database_name, database); + context->setCurrentDatabase(database_name); + } +}; + +/// An explicit `TO` target and `LoadingStrictnessLevel::ATTACH` keep this to a plain constructor +/// call: neither creates an inner table, and a non-Replicated database leaves the refresh +/// uncoordinated, so no Keeper is needed. +std::shared_ptr attachRefreshableView(const ContextMutablePtr & context, const String & view_name) +{ + const String db = database_name; + const String query = "ATTACH MATERIALIZED VIEW " + db + "." + view_name + " REFRESH EVERY 1 YEAR TO " + db + + ".target AS SELECT x FROM " + db + ".src"; + + ParserCreateQuery parser; + ASTPtr ast = parseQuery(parser, query, 100000, 1000, 1000000); + + return std::make_shared( + StorageID(database_name, view_name), + context, + ast->as(), + ColumnsDescription{{"x", std::make_shared()}}, + LoadingStrictnessLevel::ATTACH, + /*comment=*/String{}, + /*is_restore_from_backup=*/false); +} + +} + +/// shutdown() is documented to be callable before or during startup(), and it nulls the `view` +/// back-pointer that startup() reads, so this order is sanctioned rather than a misuse. +/// +/// The two calls are made directly because no query sequence produces that order: ATTACH TABLE +/// builds a fresh RefreshTask, and a database sweep joins the outstanding startup jobs before it +/// begins (DatabaseOnDisk::shutdown() calls stopLoading()). +TEST(RefreshTaskStartupAfterShutdown, StartupAfterShutdownDoesNotDereferenceNullView) +{ + const auto & state = State::instance(); + + auto view = attachRefreshableView(state.context, "mv"); + /// Without a refresh task there is no `view` pointer to null and the test below is vacuous. + ASSERT_TRUE(view->isRefreshable()); + const StorageID view_id = view->getStorageID(); + + view->flushAndPrepareForShutdown(); + view->startup(); + + /// startup() must also decline to register the task: `RefreshSet` membership is what + /// `system.view_refreshes` reports and what schedules refreshes, and this view is shut down. + EXPECT_TRUE(state.context->getRefreshSet().findTasks(view_id).empty()); +} From 443fd99757c07200b73bc295dcd5499570d3a4f4 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 23 Sep 2026 20:51:23 +0000 Subject: [PATCH 028/185] Backport #121693 to 26.8: Guard the JSON AST deserialization recursions with checkStackSize --- src/Parsers/ASTFromJSON.cpp | 35 +++++++++++++++---- src/Parsers/ASTJSONReadHelpers.cpp | 3 ++ .../0_stateless/04056_ast_json_limits.sql | 14 ++++++++ ...n_depth_guard_max_ast_depth_zero.reference | 2 +- ...ast_json_depth_guard_max_ast_depth_zero.sh | 6 ++-- 5 files changed, 51 insertions(+), 9 deletions(-) diff --git a/src/Parsers/ASTFromJSON.cpp b/src/Parsers/ASTFromJSON.cpp index 40ceb1cf9583..bb4d0b6dd8f1 100644 --- a/src/Parsers/ASTFromJSON.cpp +++ b/src/Parsers/ASTFromJSON.cpp @@ -78,12 +78,14 @@ #include #include +#include #include #include #include #include +#include #include #include @@ -309,14 +311,27 @@ size_t computeJSONNestingDepth(const String & json) return max_depth; } +/// `Poco::JSON::Parser` descends recursively over the whole document before any AST node is built. +class StackCheckingParseHandler : public Poco::JSON::ParseHandler +{ +public: + void startObject() override + { + checkStackSize(); + Poco::JSON::ParseHandler::startObject(); + } + + void startArray() override + { + checkStackSize(); + Poco::JSON::ParseHandler::startArray(); + } +}; + } ASTPtr IAST::createFromJSON(const String & json) { - /// `Poco::JSON::Parser::setDepth` does not actually bound recursion in our Poco fork - /// (`ParserImpl` stores `_depth` but `handle`/`handleObject`/`handleArray` never read it), - /// so a hostile deeply-nested payload would recurse through the parser and overflow the - /// stack before any AST-level depth check runs. Enforce a raw-text bracket budget first. /// The budget is a safe multiple of the effective depth limit (the JSON encoding adds bracket /// levels per AST/`Field` level), not the limit itself, so a valid serialized AST is never /// rejected here — the constructed AST depth is still bounded by the counter check below. @@ -328,14 +343,19 @@ ASTPtr IAST::createFromJSON(const String & json) "JSON nesting depth exceeds the limit derived from max_ast_depth ({}) during JSON AST deserialization", json_nesting_budget); Poco::JSON::Parser parser; - /// Also request the parser-level bound (kept for forward compatibility if the fork starts - /// honouring it); the pre-scan above is the actual enforcement. + parser.setHandler(new StackCheckingParseHandler); + /// Poco's own default bound is 1000 levels (JSON_DEFAULT_DEPTH). parser.setDepth(json_nesting_budget); Poco::Dynamic::Var result; try { result = parser.parse(json); } + /// `DB::Exception` derives from `Poco::Exception`, so this clause must precede the one below. + catch (const Exception &) + { + throw; + } catch (const Poco::Exception & e) { throw Exception(ErrorCodes::BAD_ARGUMENTS, "Failed to parse JSON for AST deserialization: {}", e.displayText()); @@ -365,6 +385,9 @@ ASTPtr IAST::createFromJSON(const Poco::JSON::Object & json) throw Exception(ErrorCodes::TOO_DEEP_AST, "JSON AST deserialization exceeded maximum depth limit ({})", max_depth); + /// The limit above counts nodes, which is not a stack budget at any value. + checkStackSize(); + /// Check element count limit. if (json_deser_max_elements && json_deser_current_elements >= json_deser_max_elements) throw Exception(ErrorCodes::TOO_BIG_AST, diff --git a/src/Parsers/ASTJSONReadHelpers.cpp b/src/Parsers/ASTJSONReadHelpers.cpp index db816fc4d5f4..deed0dc9fddf 100644 --- a/src/Parsers/ASTJSONReadHelpers.cpp +++ b/src/Parsers/ASTJSONReadHelpers.cpp @@ -128,6 +128,9 @@ Field JSONObjectReader::readFieldFromObjectImpl(const Poco::JSON::Object & obj, "Structured Field value exceeds maximum AST depth limit ({}) during JSON AST deserialization", max_depth); + /// The limit above counts `Field` levels, which is not a stack budget at any value. + checkStackSize(); + /// Count every `Field` value (scalar or structured) against the element-count budget too, so a /// wide literal payload (e.g. one huge `Array`) cannot bypass `max_ast_elements` while adding no /// AST nodes. diff --git a/tests/queries/0_stateless/04056_ast_json_limits.sql b/tests/queries/0_stateless/04056_ast_json_limits.sql index e448e8ff8bb7..6aac3b420051 100644 --- a/tests/queries/0_stateless/04056_ast_json_limits.sql +++ b/tests/queries/0_stateless/04056_ast_json_limits.sql @@ -91,3 +91,17 @@ SELECT formatQueryFromJSON('{"type":"RefreshStrategy","schedule_kind":"EVERY"}') -- `clickhouse_json` client/server entry points (so a huge shallow document is rejected before Poco -- materializes it). Here `parseQueryToJSON('SELECT 1')` is well over 10 bytes. SELECT formatQueryFromJSON(parseQueryToJSON('SELECT 1')) SETTINGS max_query_size = 10; -- { serverError SYNTAX_ERROR } + +-- Reading an AST back from JSON descends over the document, so raising `max_ast_depth` must not turn +-- a deep payload into a stack overflow: each shape below must fail with a controlled error, never +-- crash. The depths sit far past every bound so that holds in any build; which of the descents +-- reports it first follows the build's stack budget, so no arm asserts a particular one. +-- A chain of nested AST nodes. +SELECT formatQueryFromJSON(concat(repeat('{"type":"ExpressionList","children":[', 20000), '{"type":"Literal","value":{"field_type":"UInt64","value":1}}', repeat(']}', 20000))) SETTINGS max_ast_depth = 100000, max_ast_elements = 0, max_query_size = 100000000; -- { serverError TOO_DEEP_RECURSION } + +-- A nested `field_type` value inside a single literal, which adds no AST nodes of its own. +SELECT formatQueryFromJSON(concat('{"type":"Literal","value":', repeat('{"field_type":"Array","value":[', 25000), '{"field_type":"UInt64","value":1}', repeat(']}', 25000), '}')) SETTINGS max_ast_depth = 200000, max_ast_elements = 0, max_query_size = 100000000; -- { serverError TOO_DEEP_RECURSION } + +-- The JSON text is parsed before any AST node exists, so a deep member that the reader never descends +-- into still has to be bounded while the text itself is being read. +SELECT formatQueryFromJSON(concat('{"type":"Literal","value":{"field_type":"UInt64","value":1},"junk":', repeat('[', 200000), repeat(']', 200000), '}')) SETTINGS max_ast_depth = 100000, max_ast_elements = 0, max_query_size = 100000000; -- { serverError TOO_DEEP_RECURSION } diff --git a/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.reference b/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.reference index 0fb27808c3ca..73b717c7d255 100644 --- a/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.reference +++ b/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.reference @@ -1,6 +1,6 @@ TOO_DEEP_AST TOO_DEEP_AST -Structured Field value exceeds maximum AST depth limit +Structured Field value rejected Field dump payload exceeds maximum AST depth limit 1 TOO_DEEP_AST diff --git a/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.sh b/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.sh index 455c50c4a2d5..67c08b16c6ba 100755 --- a/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.sh +++ b/tests/queries/0_stateless/04654_ast_json_depth_guard_max_ast_depth_zero.sh @@ -27,14 +27,16 @@ ${CLICKHOUSE_CLIENT} --max_ast_depth 0 --param_json "$BRACKET_BOMB" \ --query "SELECT formatQueryFromJSON({json:String})" 2>&1 | grep -om1 'TOO_DEEP_AST' # 3. A deeply nested structured `Field` value adds no AST nodes and stays under the bracket -# budget, so only the `Field` depth bound rejects it. +# budget, so it is rejected by its own depth bound or by the recursion's stack check. Which of +# the two reports it depends on the build's stack budget, so assert only that it is rejected. ${CLICKHOUSE_CLIENT} --max_ast_depth 0 --query " SELECT formatQueryFromJSON(concat( '{\"type\":\"Literal\",\"value\":', repeat('{\"field_type\":\"Array\",\"value\":[', 2000), '{\"field_type\":\"UInt64\",\"value\":1}', repeat(']}', 2000), '}'))" 2>&1 | - grep -om1 'Structured Field value exceeds maximum AST depth limit' + grep -qE 'Structured Field value exceeds maximum AST depth limit|TOO_DEEP_RECURSION' && + echo 'Structured Field value rejected' # 4. A deeply nested `Field` dump hides its nesting inside a JSON string, so the bracket # pre-scan does not see it either; `Field::restoreFromDump` must not recurse unbounded. From 4ccf093ab6b9a3e5cfc18a63d2c8e72c9379c29d Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 02:53:45 +0000 Subject: [PATCH 029/185] Backport #108626 to 26.8: Avoid vertical TTL delete for TTLDrop merges --- src/Storages/MergeTree/MergeTask.cpp | 4 + .../04411_ttl_drop_not_vertical.reference | 4 + .../04411_ttl_drop_not_vertical.sh | 154 ++++++++++++++++++ 3 files changed, 162 insertions(+) create mode 100644 tests/queries/0_stateless/04411_ttl_drop_not_vertical.reference create mode 100755 tests/queries/0_stateless/04411_ttl_drop_not_vertical.sh diff --git a/src/Storages/MergeTree/MergeTask.cpp b/src/Storages/MergeTree/MergeTask.cpp index 7138a2685bb3..fc33ad1cc3a8 100644 --- a/src/Storages/MergeTree/MergeTask.cpp +++ b/src/Storages/MergeTree/MergeTask.cpp @@ -3530,6 +3530,10 @@ MergeAlgorithm MergeTask::ExecuteAndFinalizeHorizontalPart::chooseMergeAlgorithm return MergeAlgorithm::Horizontal; if (ctx->need_remove_expired_values) { + /// `TTLTransform` stops reading after the first block when the rows TTL has expired for the whole part, + /// while a vertical merge reads every key row to write `rows_sources`. + if (global_ctx->future_part->merge_type == MergeType::TTLDrop && global_ctx->metadata_snapshot->hasRowsTTL()) + return MergeAlgorithm::Horizontal; if (!canVerticalTTLDelete(*global_ctx)) return MergeAlgorithm::Horizontal; } diff --git a/tests/queries/0_stateless/04411_ttl_drop_not_vertical.reference b/tests/queries/0_stateless/04411_ttl_drop_not_vertical.reference new file mode 100644 index 000000000000..876d94d1857a --- /dev/null +++ b/tests/queries/0_stateless/04411_ttl_drop_not_vertical.reference @@ -0,0 +1,4 @@ +TTLDropMerge Horizontal 0 +0 +TTLDropMerge Horizontal 0 +0 diff --git a/tests/queries/0_stateless/04411_ttl_drop_not_vertical.sh b/tests/queries/0_stateless/04411_ttl_drop_not_vertical.sh new file mode 100755 index 000000000000..d34ea2f74dc6 --- /dev/null +++ b/tests/queries/0_stateless/04411_ttl_drop_not_vertical.sh @@ -0,0 +1,154 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +set -e + +TABLE=t_ttl_drop_not_vertical + +function wait_for_ttl_drop_merge() +{ + for _ in $(seq 1 600); do + ${CLICKHOUSE_CLIENT} -q "SYSTEM FLUSH LOGS part_log" + local merge_count + merge_count=$(${CLICKHOUSE_CLIENT} -q " + SELECT count() + FROM system.part_log + WHERE database = currentDatabase() + AND table = '${TABLE}' + AND event_type = 'MergeParts' + AND merge_reason = 'TTLDropMerge'") + + if [ "${merge_count}" -gt "0" ]; then + return + fi + sleep 0.1 + done + + echo "Timed out waiting for a TTLDropMerge of ${TABLE}" >&2 + exit 1 +} + +${CLICKHOUSE_CLIENT} -q "DROP TABLE IF EXISTS ${TABLE}" + +${CLICKHOUSE_CLIENT} -q " + CREATE TABLE ${TABLE} + ( + id UInt64, + d DateTime DEFAULT '2000-01-01 00:00:00', + c1 UInt64, + c2 UInt64, + c3 UInt64, + c4 UInt64 + ) + ENGINE = MergeTree + ORDER BY id + TTL d + INTERVAL 1 DAY + SETTINGS + ttl_only_drop_parts = 1, + merge_with_ttl_timeout = 0, + min_bytes_for_wide_part = 0, + min_bytes_for_full_part_storage = 0, + enable_block_number_column = 0, + enable_block_offset_column = 0, + vertical_merge_algorithm_min_rows_to_activate = 1, + vertical_merge_algorithm_min_columns_to_activate = 1, + vertical_merge_optimize_ttl_delete = 1, + ratio_of_defaults_for_sparse_serialization = 1.0" + +${CLICKHOUSE_CLIENT} -q "SYSTEM STOP TTL MERGES ${TABLE}" +${CLICKHOUSE_CLIENT} -q "SYSTEM STOP MERGES ${TABLE}" + +${CLICKHOUSE_CLIENT} -q " + INSERT INTO ${TABLE} + SELECT number, '2000-01-01 00:00:00', number, number, number, number + FROM numbers(1000)" +${CLICKHOUSE_CLIENT} -q " + INSERT INTO ${TABLE} + SELECT number + 1000, '2000-01-01 00:00:00', number, number, number, number + FROM numbers(1000)" + +${CLICKHOUSE_CLIENT} -q "SYSTEM START TTL MERGES ${TABLE}" +${CLICKHOUSE_CLIENT} -q "SYSTEM START MERGES ${TABLE}" + +wait_for_ttl_drop_merge + +${CLICKHOUSE_CLIENT} -q " + SELECT + merge_reason, + merge_algorithm, + rows + FROM system.part_log + WHERE database = currentDatabase() + AND table = '${TABLE}' + AND event_type = 'MergeParts' + AND merge_reason = 'TTLDropMerge' + ORDER BY event_time_microseconds DESC + LIMIT 1" + +${CLICKHOUSE_CLIENT} -q "SELECT count() FROM ${TABLE}" +${CLICKHOUSE_CLIENT} -q "DROP TABLE ${TABLE}" + +TABLE=t_ttl_drop_not_vertical_mixed_ttl + +${CLICKHOUSE_CLIENT} -q "DROP TABLE IF EXISTS ${TABLE}" + +${CLICKHOUSE_CLIENT} -q " + CREATE TABLE ${TABLE} + ( + id UInt64, + d DateTime DEFAULT '2000-01-01 00:00:00', + c1 UInt64, + c2 UInt64, + c3 UInt64, + c4 UInt64 + ) + ENGINE = MergeTree + ORDER BY id + TTL d + INTERVAL 1 DAY, d + INTERVAL 2 DAY RECOMPRESS CODEC(ZSTD) + SETTINGS + ttl_only_drop_parts = 1, + merge_with_ttl_timeout = 0, + min_bytes_for_wide_part = 0, + min_bytes_for_full_part_storage = 0, + enable_block_number_column = 0, + enable_block_offset_column = 0, + vertical_merge_algorithm_min_rows_to_activate = 1, + vertical_merge_algorithm_min_columns_to_activate = 1, + vertical_merge_optimize_ttl_delete = 1, + ratio_of_defaults_for_sparse_serialization = 1.0" + +${CLICKHOUSE_CLIENT} -q "SYSTEM STOP TTL MERGES ${TABLE}" +${CLICKHOUSE_CLIENT} -q "SYSTEM STOP MERGES ${TABLE}" + +${CLICKHOUSE_CLIENT} -q " + INSERT INTO ${TABLE} + SELECT number, '2000-01-01 00:00:00', number, number, number, number + FROM numbers(1000)" +${CLICKHOUSE_CLIENT} -q " + INSERT INTO ${TABLE} + SELECT number + 1000, '2000-01-01 00:00:00', number, number, number, number + FROM numbers(1000)" + +${CLICKHOUSE_CLIENT} -q "SYSTEM START TTL MERGES ${TABLE}" +${CLICKHOUSE_CLIENT} -q "SYSTEM START MERGES ${TABLE}" + +wait_for_ttl_drop_merge + +${CLICKHOUSE_CLIENT} -q " + SELECT + merge_reason, + merge_algorithm, + rows + FROM system.part_log + WHERE database = currentDatabase() + AND table = '${TABLE}' + AND event_type = 'MergeParts' + AND merge_reason = 'TTLDropMerge' + ORDER BY event_time_microseconds DESC + LIMIT 1" + +${CLICKHOUSE_CLIENT} -q "SELECT count() FROM ${TABLE}" +${CLICKHOUSE_CLIENT} -q "DROP TABLE ${TABLE}" From cff4c0021e2136cca1c664f66572abb55182441f Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 08:35:40 +0000 Subject: [PATCH 030/185] Backport #118917 to 26.8: Bump `croaring` from v4.5.1 to v5.1.1 --- contrib/croaring | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/contrib/croaring b/contrib/croaring index 025ae3f7add1..342463b31b90 160000 --- a/contrib/croaring +++ b/contrib/croaring @@ -1 +1 @@ -Subproject commit 025ae3f7add169bc820dcfd46fa9304f382ec40a +Subproject commit 342463b31b909737bd6295c69b2f8e0ed9497424 From a7aa824c2e54f6c2733231a67402e7b0df63715e Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Ra=C3=BAl=20Mar=C3=ADn?= Date: Thu, 24 Sep 2026 09:04:58 +0000 Subject: [PATCH 031/185] Include checkStackSize for the checkStackSize() call added by the backport 26.8 lacked the earlier fix that added this include (and its own checkStackSize() call) to ASTJSONReadHelpers.cpp, so the cherry-picked checkStackSize() call in readFieldFromObjectImpl failed to compile. Co-Authored-By: Claude Sonnet 5 --- src/Parsers/ASTJSONReadHelpers.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/src/Parsers/ASTJSONReadHelpers.cpp b/src/Parsers/ASTJSONReadHelpers.cpp index deed0dc9fddf..ae4111a79448 100644 --- a/src/Parsers/ASTJSONReadHelpers.cpp +++ b/src/Parsers/ASTJSONReadHelpers.cpp @@ -4,6 +4,7 @@ #include #include #include +#include #include #include From 0653b9ea269820006c1aee957f5b31847b89fdfd Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 09:58:14 +0000 Subject: [PATCH 032/185] Backport #121864 to 26.8: Small performance optimizations --- .../Iceberg/ManifestFileIterator.cpp | 15 ++++++----- .../DataLakes/Iceberg/ManifestFileIterator.h | 1 + .../DataLakes/Iceberg/SchemaProcessor.cpp | 27 +++++++++++++------ 3 files changed, 28 insertions(+), 15 deletions(-) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp index 9c0d2db914de..9fed28a86496 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp @@ -464,13 +464,14 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) /// those manifests still carry the original snapshot_id. The manifest file's own Avro header /// records the correct schema_id for the data files it describes, so falling back to /// manifest_schema_id is safe and correct in this case. - LOG_DEBUG( - getLogger("ManifestFileIterator"), - "Manifest file '{}' has entry with snapshot_id '{}' whose snapshot metadata is not present " - "(snapshot may have been expired by the catalog). Falling back to manifest schema_id {}.", - path_to_manifest_file, - resolved_snapshot_id, - manifest_schema_id); + if (!logged_missing_snapshot_metadata.exchange(true, std::memory_order_relaxed)) + LOG_TEST( + getLogger("ManifestFileIterator"), + "Manifest file '{}' has entry with snapshot_id '{}' whose snapshot metadata is not present " + "(snapshot may have been expired by the catalog). Falling back to manifest schema_id {}.", + path_to_manifest_file, + resolved_snapshot_id, + manifest_schema_id); } const auto resolved_schema_id = schema_id_opt.has_value() ? *schema_id_opt : manifest_schema_id; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.h index 40bb9b4c45c1..d6e90f78590e 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.h @@ -160,6 +160,7 @@ class ManifestFileIterator : public boost::noncopyable std::atomic current_row_index{0}; std::atomic fully_initialized{false}; std::atomic active_fetchers{0}; + std::atomic logged_missing_snapshot_metadata{false}; /// Cached results accumulated during iteration mutable SharedMutex files_mutex; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp index c154a025ab3a..30c5177ce13c 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp @@ -395,8 +395,6 @@ std::string IcebergSchemaProcessor::default_link{}; void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr) { - std::lock_guard lock(mutex); - Int32 schema_id = schema_ptr->getValue(f_schema_id); /// Databricks UniForm writes a degenerate placeholder schema (e.g. {"schema-id":0,"fields":[]}) @@ -404,16 +402,29 @@ void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schem if (!schema_ptr->isArray(f_fields) || schema_ptr->getArray(f_fields)->size() == 0) return; + std::unordered_map type_mapping; + if (allow_geo_parser) + { + type_mapping[f_geography] = f_binary; + type_mapping[f_geometry] = f_binary; + } + + Poco::JSON::Object::Ptr registered_schema; + { + SharedLockGuard lock(mutex); + auto it = iceberg_table_schemas_by_ids.find(schema_id); + if (it != iceberg_table_schemas_by_ids.end()) + registered_schema = it->second; + } + if (registered_schema && schemasAreIdentical(*registered_schema, *schema_ptr, type_mapping)) + return; + + std::lock_guard lock(mutex); + current_schema_id = schema_id; if (iceberg_table_schemas_by_ids.contains(schema_id)) { chassert(clickhouse_table_schemas_by_ids.contains(schema_id)); - std::unordered_map type_mapping; - if (allow_geo_parser) - { - type_mapping[f_geography] = f_binary; - type_mapping[f_geometry] = f_binary; - } /// A schema-id is immutable per the Iceberg spec: re-binding it to different fields is malformed metadata. if (!schemasAreIdentical(*iceberg_table_schemas_by_ids.at(schema_id), *schema_ptr, type_mapping)) throw Exception( From 87208465d5fd9cbb0a0c9e1306f0ed5bd0d2cb58 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 10:29:53 +0000 Subject: [PATCH 033/185] Backport #121856 to 26.8: Do not exhaust the stack when serializing a deeply nested query to JSON --- src/Parsers/ASTJSONHelpers.cpp | 2 ++ src/Parsers/ASTJSONHelpers.h | 3 +++ ...uery_to_json_deep_query_stack_guard.reference | 2 ++ ...arse_query_to_json_deep_query_stack_guard.sql | 16 ++++++++++++++++ 4 files changed, 23 insertions(+) create mode 100644 tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.reference create mode 100644 tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.sql diff --git a/src/Parsers/ASTJSONHelpers.cpp b/src/Parsers/ASTJSONHelpers.cpp index 1d8e3e28cdad..63cad18b82c8 100644 --- a/src/Parsers/ASTJSONHelpers.cpp +++ b/src/Parsers/ASTJSONHelpers.cpp @@ -3,6 +3,7 @@ #include #include #include +#include #include @@ -22,6 +23,7 @@ void JSONObjectWriter::writeAlias(const ASTWithAlias & node) static void writeFieldJSON(WriteBuffer & out, const FormatSettings & fs, const Field & field) { + checkStackSize(); out << "{\"field_type\":"; writeJSONString(field.getTypeName(), out, fs); diff --git a/src/Parsers/ASTJSONHelpers.h b/src/Parsers/ASTJSONHelpers.h index 68afb51b2aae..7cc6c5d7d467 100644 --- a/src/Parsers/ASTJSONHelpers.h +++ b/src/Parsers/ASTJSONHelpers.h @@ -5,6 +5,7 @@ #include #include #include +#include namespace DB { @@ -30,6 +31,8 @@ class JSONObjectWriter JSONObjectWriter(WriteBuffer & out_, const char * type_name) : out(out_) { + /// One of these is constructed per node by the AST-to-JSON walk, which recurses as deep as the query nests. + checkStackSize(); out << "{\"type\":"; writeJSONString(std::string_view(type_name), out, fs); } diff --git a/tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.reference b/tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.reference new file mode 100644 index 000000000000..6ed281c757a9 --- /dev/null +++ b/tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.reference @@ -0,0 +1,2 @@ +1 +1 diff --git a/tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.sql b/tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.sql new file mode 100644 index 000000000000..6501d6bb073a --- /dev/null +++ b/tests/queries/0_stateless/05254_parse_query_to_json_deep_query_stack_guard.sql @@ -0,0 +1,16 @@ +-- `parseQueryToJSON` walks the parsed query recursively, so a query nested deeper than that walk can +-- follow must be reported rather than crash the server. `max_parser_depth` is raised far above the +-- nesting so that the parser's own limit does not answer first, and the refused depths are far past +-- what any build's stack can hold, so they do not depend on the build. The accepted depths are kept +-- small instead: under TSan only 5% of a thread's stack may be used, which this walk exhausts at +-- under a hundred levels of nesting. + +SELECT length(parseQueryToJSON(concat('SELECT ', repeat('abs(', 10), '1', repeat(')', 10)))) > 0; +SELECT length(parseQueryToJSON(concat('SELECT ', repeat('abs(', 40000), '1', repeat(')', 40000)))) +SETTINGS max_parser_depth = 200000, max_ast_depth = 200000, max_ast_elements = 0, max_query_size = 100000000; -- { serverError TOO_DEEP_RECURSION } + +-- A single nested value nests the same way while adding one node to the query, so the depth of the +-- query does not bound it. +SELECT length(parseQueryToJSON(concat('SELECT ', repeat('[', 10), '1', repeat(']', 10)))) > 0; +SELECT length(parseQueryToJSON(concat('SELECT ', repeat('[', 500000), '1', repeat(']', 500000)))) +SETTINGS max_parser_depth = 2000000, max_ast_elements = 0, max_query_size = 100000000; -- { serverError TOO_DEEP_RECURSION } From cac86dcf32e00c796da943572917f1c3c284e7f3 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 14:22:37 +0000 Subject: [PATCH 034/185] Backport #117421 to 26.8: Fix SQL injection through identifiers when inserting into a SQLite table --- src/Common/quoteString.cpp | 11 +++ src/Common/quoteString.h | 2 + src/IO/WriteHelpers.h | 13 +++ src/Storages/StorageSQLite.cpp | 4 +- ...qlite_insert_identifier_escaping.reference | 20 +++++ ...05055_sqlite_insert_identifier_escaping.sh | 81 +++++++++++++++++++ 6 files changed, 129 insertions(+), 2 deletions(-) create mode 100644 tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.reference create mode 100755 tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.sh diff --git a/src/Common/quoteString.cpp b/src/Common/quoteString.cpp index 0559e85d8a50..cc3dbaac0e62 100644 --- a/src/Common/quoteString.cpp +++ b/src/Common/quoteString.cpp @@ -39,6 +39,17 @@ String doubleQuoteString(std::string_view x) } +String doubleQuoteStringSQLite(std::string_view x) +{ + String res(2 + x.size(), '\0'); + { + WriteBufferFromString wb(res); + writeDoubleQuotedStringSQLite(x, wb); + } + return res; +} + + String backQuote(std::string_view x) { String res(2 + x.size(), '\0'); diff --git a/src/Common/quoteString.h b/src/Common/quoteString.h index b6ee14c0b878..15733c7c2d6a 100644 --- a/src/Common/quoteString.h +++ b/src/Common/quoteString.h @@ -30,6 +30,8 @@ namespace DB /// Double quote the string. String doubleQuoteString(std::string_view x); +String doubleQuoteStringSQLite(std::string_view x); + /// Quote the identifier with backquotes. String backQuote(std::string_view x); diff --git a/src/IO/WriteHelpers.h b/src/IO/WriteHelpers.h index 185c22b38caf..1afee1a496e0 100644 --- a/src/IO/WriteHelpers.h +++ b/src/IO/WriteHelpers.h @@ -642,6 +642,19 @@ inline void writeQuotedStringSQLite(std::string_view ref, WriteBuffer & buf) writeChar('\'', buf); } +/// SQLite identifiers: a " is escaped by doubling it; every other byte, backslash included, is literal. +inline void writeDoubleQuotedStringSQLite(std::string_view ref, WriteBuffer & buf) +{ + writeChar('"', buf); + for (char c : ref) + { + if (c == '"') + writeChar('"', buf); + writeChar(c, buf); + } + writeChar('"', buf); +} + inline void writeDoubleQuotedString(const String & s, WriteBuffer & buf) { writeAnyQuotedString<'"'>(s, buf); diff --git a/src/Storages/StorageSQLite.cpp b/src/Storages/StorageSQLite.cpp index aef8d35af603..a5964106745e 100644 --- a/src/Storages/StorageSQLite.cpp +++ b/src/Storages/StorageSQLite.cpp @@ -184,14 +184,14 @@ class SQLiteSink final : public SinkToStorage WriteBufferFromOwnString sqlbuf; sqlbuf << "INSERT INTO "; - sqlbuf << doubleQuoteString(remote_table_name); + sqlbuf << doubleQuoteStringSQLite(remote_table_name); sqlbuf << " ("; for (auto it = block.begin(); it != block.end(); ++it) { if (it != block.begin()) sqlbuf << ", "; - sqlbuf << quoteString(it->name); + sqlbuf << doubleQuoteStringSQLite(it->name); } sqlbuf << ") VALUES "; diff --git a/tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.reference b/tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.reference new file mode 100644 index 000000000000..d50220788b0b --- /dev/null +++ b/tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.reference @@ -0,0 +1,20 @@ +--- A1 control: the boundary the injection bypasses is enforced when the path is named directly +PATH_ACCESS_DENIED +--- A2 injection through the remote table name +no such table +objects created in the anchor: none +canary read from outside user_files: none +objects written outside user_files: none +--- A3 injection through a column name +no column named +objects created in the anchor: none +canary read from outside user_files: none +objects written outside user_files: none +--- A4 control: a table and column name that legitimately contain a double quote round-trip +NO_ERROR +11 +--- A5 control: a table name that legitimately contains a backslash round-trips +NO_ERROR +22 +--- A6 a NUL in the remote table name fails loudly, it is never silently truncated +unrecognized token diff --git a/tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.sh b/tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.sh new file mode 100755 index 000000000000..9ec9abb0b9ba --- /dev/null +++ b/tests/queries/0_stateless/05055_sqlite_insert_identifier_escaping.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: Fast tests don't build external libraries (SQLite) + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +ANCHOR="${USER_FILES_PATH}/05055_anchor_${CLICKHOUSE_DATABASE}.db" +# Outside the user_files confinement. +OUTSIDE="${CLICKHOUSE_TMP}/05055_outside_${CLICKHOUSE_DATABASE}.db" + +cleanup() +{ + ${CLICKHOUSE_CLIENT} --query="DROP TABLE IF EXISTS t_name_inj" + ${CLICKHOUSE_CLIENT} --query="DROP TABLE IF EXISTS t_col_inj" + ${CLICKHOUSE_CLIENT} --query="DROP TABLE IF EXISTS t_quote_id" + ${CLICKHOUSE_CLIENT} --query="DROP TABLE IF EXISTS t_backslash_id" + ${CLICKHOUSE_CLIENT} --query="DROP TABLE IF EXISTS t_nul_name" + rm -f "${ANCHOR}" "${OUTSIDE}" +} +trap cleanup EXIT +cleanup + +# Odd names: the old backslash escaping produced them, so the injected script's first statement succeeds and the rest runs. +sqlite3 "${ANCHOR}" 'CREATE TABLE "target\"(id INTEGER)' +sqlite3 "${ANCHOR}" 'CREATE TABLE col_anchor("c\" INTEGER)' +sqlite3 "${ANCHOR}" 'CREATE TABLE "ta""ble" ("c""1" INTEGER)' +sqlite3 "${ANCHOR}" 'CREATE TABLE "a\b" (x INTEGER)' +chmod ugo+rw "${ANCHOR}" + +# A database outside user_files, holding a canary the user must never be able to reach. +sqlite3 "${OUTSIDE}" "CREATE TABLE secrets(s TEXT); INSERT INTO secrets VALUES ('CANARY')" +chmod ugo+rw "${OUTSIDE}" + +# One stable token per failure, so the reference embeds no path and no server version. +classify() +{ + local out + out=$(${CLICKHOUSE_CLIENT} --query="$1" 2>&1 \ + | grep -oF -e 'no such table' -e 'no column named' -e 'unrecognized token' \ + -e 'syntax error' -e 'PATH_ACCESS_DENIED' \ + | sed -n 1p) + echo "${out:-NO_ERROR}" +} + +# Read out of band: the ClickHouse read path cannot address these names, so an in-band readback would be a false negative. +injected_state() +{ + sqlite3 "${ANCHOR}" "SELECT 'objects created in the anchor: ' || coalesce((SELECT group_concat(name) FROM (SELECT name FROM sqlite_master WHERE name LIKE '%\_marker' ESCAPE '\' ORDER BY name)), 'none')" + sqlite3 "${ANCHOR}" "SELECT 'canary read from outside user_files: ' || coalesce((SELECT group_concat(s) FROM stolen_marker), 'none')" 2>/dev/null \ + || echo 'canary read from outside user_files: none' + sqlite3 "${OUTSIDE}" "SELECT 'objects written outside user_files: ' || coalesce((SELECT group_concat(name) FROM (SELECT name FROM sqlite_master WHERE name LIKE '%\_marker' ESCAPE '\' ORDER BY name)), 'none')" +} + +echo '--- A1 control: the boundary the injection bypasses is enforced when the path is named directly' +classify "CREATE TABLE t_ctl (s String) ENGINE = SQLite('${OUTSIDE}', 'secrets')" + +echo '--- A2 injection through the remote table name' +${CLICKHOUSE_CLIENT} --query="CREATE TABLE t_name_inj (id UInt32) ENGINE = SQLite('${ANCHOR}', \$\$target\" (id) VALUES (999); ATTACH DATABASE '${OUTSIDE}' AS v; CREATE TABLE stolen_marker AS SELECT s FROM v.secrets; CREATE TABLE v.written_marker(z); DETACH v; --\$\$)" +classify "INSERT INTO t_name_inj VALUES (0)" +injected_state + +echo '--- A3 injection through a column name' +${CLICKHOUSE_CLIENT} --query="CREATE TABLE t_col_inj (\`c') VALUES (1); CREATE TABLE col_marker(z); --\` UInt32) ENGINE = SQLite('${ANCHOR}', 'col_anchor')" +classify "INSERT INTO t_col_inj VALUES (0)" +injected_state + +echo '--- A4 control: a table and column name that legitimately contain a double quote round-trip' +${CLICKHOUSE_CLIENT} --query="CREATE TABLE t_quote_id (\`c\"1\` UInt32) ENGINE = SQLite('${ANCHOR}', 'ta\"ble')" +classify "INSERT INTO t_quote_id VALUES (11)" +sqlite3 "${ANCHOR}" 'SELECT "c""1" FROM "ta""ble"' + +echo '--- A5 control: a table name that legitimately contains a backslash round-trips' +${CLICKHOUSE_CLIENT} --query="CREATE TABLE t_backslash_id (x UInt32) ENGINE = SQLite('${ANCHOR}', 'a\\\\b')" +classify "INSERT INTO t_backslash_id VALUES (22)" +sqlite3 "${ANCHOR}" 'SELECT x FROM "a\b"' + +echo '--- A6 a NUL in the remote table name fails loudly, it is never silently truncated' +${CLICKHOUSE_CLIENT} --query="CREATE TABLE t_nul_name (id UInt32) ENGINE = SQLite('${ANCHOR}', 'a\0b')" +classify "INSERT INTO t_nul_name VALUES (0)" From 6b68f29f4e46aaf493f3f3ee9a148d956972c449 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 16:28:59 +0000 Subject: [PATCH 035/185] Backport #118360 to 26.8: Require the local privilege for KILL and SYSTEM statements over ON CLUSTER --- .../InterpreterKillQueryQuery.cpp | 36 ++++-- src/Interpreters/InterpreterSystemQuery.cpp | 13 +- ...d_system_on_cluster_access_types.reference | 12 ++ ...kill_and_system_on_cluster_access_types.sh | 111 ++++++++++++++++++ 4 files changed, 160 insertions(+), 12 deletions(-) create mode 100644 tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.reference create mode 100755 tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.sh diff --git a/src/Interpreters/InterpreterKillQueryQuery.cpp b/src/Interpreters/InterpreterKillQueryQuery.cpp index 177ca1a051c0..32b725ce3474 100644 --- a/src/Interpreters/InterpreterKillQueryQuery.cpp +++ b/src/Interpreters/InterpreterKillQueryQuery.cpp @@ -474,17 +474,31 @@ AccessRightsElements InterpreterKillQueryQuery::getRequiredAccessForDDLOnCluster { const auto & query = query_ptr->as(); AccessRightsElements required_access; - if (query.type == ASTKillQueryQuery::Type::Query) - required_access.emplace_back(AccessType::KILL_QUERY); - else if (query.type == ASTKillQueryQuery::Type::Mutation) - required_access.emplace_back( - AccessType::ALTER_UPDATE - | AccessType::ALTER_DELETE - | AccessType::ALTER_MATERIALIZE_INDEX - | AccessType::ALTER_MATERIALIZE_COLUMN - | AccessType::ALTER_MATERIALIZE_TTL - | AccessType::ALTER_REWRITE_PARTS - ); + /// This switch has no `default:`, so a new Type has to be mapped here to compile. + switch (query.type) + { + case ASTKillQueryQuery::Type::Query: + required_access.emplace_back(AccessType::KILL_QUERY); + break; + case ASTKillQueryQuery::Type::Mutation: + required_access.emplace_back( + AccessType::ALTER_UPDATE + | AccessType::ALTER_DELETE + | AccessType::ALTER_MATERIALIZE_INDEX + | AccessType::ALTER_MATERIALIZE_COLUMN + | AccessType::ALTER_MATERIALIZE_TTL + | AccessType::ALTER_REWRITE_PARTS + ); + break; + case ASTKillQueryQuery::Type::PartMoveToShard: + required_access.emplace_back(AccessType::SELECT, DatabaseCatalog::SYSTEM_DATABASE, "part_moves_between_shards"); + required_access.emplace_back(AccessType::ALTER_MOVE_PARTITION | AccessType::MOVE_PARTITION_BETWEEN_SHARDS); + break; + case ASTKillQueryQuery::Type::Transaction: + required_access.emplace_back(AccessType::KILL_TRANSACTION); + required_access.emplace_back(AccessType::SELECT, DatabaseCatalog::SYSTEM_DATABASE, "transactions"); + break; + } return required_access; } diff --git a/src/Interpreters/InterpreterSystemQuery.cpp b/src/Interpreters/InterpreterSystemQuery.cpp index 9efabe1c7877..9a9ff9e46785 100644 --- a/src/Interpreters/InterpreterSystemQuery.cpp +++ b/src/Interpreters/InterpreterSystemQuery.cpp @@ -3229,11 +3229,22 @@ AccessRightsElements InterpreterSystemQuery::getRequiredAccessForDDLOnCluster() } case Type::STOP_THREAD_FUZZER: case Type::START_THREAD_FUZZER: + { + required_access.emplace_back(AccessType::SYSTEM_THREAD_FUZZER); + break; + } + case Type::RESET_COVERAGE: + { + required_access.emplace_back(AccessType::SYSTEM); + break; + } + /// The parser cases of the failpoint statements and of SYSTEM SET COVERAGE TEST never read an + /// ON CLUSTER clause, so those cluster spellings do not parse and reach no host. UNKNOWN and + /// END are not statements. case Type::ENABLE_FAILPOINT: case Type::WAIT_FAILPOINT: case Type::NOTIFY_FAILPOINT: case Type::DISABLE_FAILPOINT: - case Type::RESET_COVERAGE: case Type::SET_COVERAGE_TEST: case Type::UNKNOWN: case Type::END: break; diff --git a/tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.reference b/tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.reference new file mode 100644 index 000000000000..e9efacb33c69 --- /dev/null +++ b/tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.reference @@ -0,0 +1,12 @@ +KILL TRANSACTION -> KILL TRANSACTION ON *.* +KILL PART_MOVE_TO_SHARD -> SELECT ON system.part_moves_between_shards +SYSTEM STOP THREAD FUZZER -> SYSTEM THREAD FUZZER ON *.* +SYSTEM START THREAD FUZZER -> SYSTEM THREAD FUZZER ON *.* +SYSTEM RESET COVERAGE -> SYSTEM ON *.* +KILL TRANSACTION without system.transactions -> SELECT ON system.transactions +KILL PART_MOVE_TO_SHARD without move privileges -> ALTER MOVE PARTITION, MOVE PARTITION BETWEEN SHARDS ON *.* +no CLUSTER grant -> CLUSTER ON *.* +KILL TRANSACTION local -> allowed +KILL TRANSACTION on cluster -> allowed +KILL PART_MOVE_TO_SHARD local -> allowed +KILL PART_MOVE_TO_SHARD on cluster -> allowed diff --git a/tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.sh b/tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.sh new file mode 100755 index 000000000000..9f246c25da6d --- /dev/null +++ b/tests/queries/0_stateless/05082_kill_and_system_on_cluster_access_types.sh @@ -0,0 +1,111 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Holds CLUSTER and nothing else. +cluster_user="cluster_$CLICKHOUSE_TEST_UNIQUE_NAME" +# Holds every statement privilege but not CLUSTER. +no_cluster_user="no_cluster_$CLICKHOUSE_TEST_UNIQUE_NAME" +# Hold the complete privilege set of one statement each. +kill_txn_user="kill_txn_$CLICKHOUSE_TEST_UNIQUE_NAME" +move_user="move_$CLICKHOUSE_TEST_UNIQUE_NAME" +# Hold all but one privilege of one statement each. +partial_txn_user="partial_txn_$CLICKHOUSE_TEST_UNIQUE_NAME" +partial_move_user="partial_move_$CLICKHOUSE_TEST_UNIQUE_NAME" + +function cleanup() +{ + $CLICKHOUSE_CLIENT -mq " + DROP USER IF EXISTS $cluster_user; + DROP USER IF EXISTS $no_cluster_user; + DROP USER IF EXISTS $kill_txn_user; + DROP USER IF EXISTS $move_user; + DROP USER IF EXISTS $partial_txn_user; + DROP USER IF EXISTS $partial_move_user; + " +} +cleanup +trap cleanup EXIT + +# CLUSTER is a global privilege, so it cannot share a GRANT with a table-scoped one. +$CLICKHOUSE_CLIENT -mq " + CREATE USER $cluster_user, $no_cluster_user, $kill_txn_user, $move_user, + $partial_txn_user, $partial_move_user IDENTIFIED WITH no_password; + + GRANT CLUSTER ON *.* TO $cluster_user; + + GRANT KILL TRANSACTION, SYSTEM THREAD FUZZER, SYSTEM ON *.* TO $no_cluster_user; + + GRANT CLUSTER, KILL TRANSACTION ON *.* TO $kill_txn_user; + GRANT SELECT ON system.transactions TO $kill_txn_user; + + GRANT CLUSTER, ALTER MOVE PARTITION, MOVE PARTITION BETWEEN SHARDS ON *.* TO $move_user; + GRANT SELECT ON system.part_moves_between_shards TO $move_user; + + GRANT CLUSTER, KILL TRANSACTION ON *.* TO $partial_txn_user; + + GRANT CLUSTER ON *.* TO $partial_move_user; + GRANT SELECT ON system.part_moves_between_shards TO $partial_move_user; +" + +cluster="test_shard_localhost" +# Match no live transaction and no live part move, so every allowed arm below is a no-op. The +# transaction predicate deliberately does not compare `tid`: its tuple shape is not the same in +# every build, so a literal tuple fails type analysis before the access check this test is about. +txn_predicate="tid_hash = 0" +task_uuid="'00000000-0000-0000-0000-000000000000'" + +# Report the privilege the server asked for and let the reference hold the expected mapping. The +# privilege name is what discriminates: an ACCESS_DENIED-only assertion would also pass when the +# CLUSTER check, which runs before the required-access check, is what fired. +required_privilege() { + $CLICKHOUSE_CLIENT --distributed_ddl_output_mode none --user "$1" --query "$2" 2>&1 | + sed -n "/necessary to have the grant/{s/.*grant \(.*\)\. (ACCESS_DENIED).*/\1/p;q;}" +} + +# Each statement must name over ON CLUSTER the same privilege its local spelling names. +while IFS= read -r statement; do + echo "${statement%% ON CLUSTER*} -> $(required_privilege "$cluster_user" "$statement")" +done < $(required_privilege "$partial_txn_user" "KILL TRANSACTION ON CLUSTER $cluster WHERE $txn_predicate")" +echo "KILL PART_MOVE_TO_SHARD without move privileges -> $(required_privilege "$partial_move_user" "KILL PART_MOVE_TO_SHARD ON CLUSTER $cluster WHERE task_uuid = $task_uuid")" + +# In-range control: holding the statement privileges without CLUSTER is refused by the earlier check, +# so a mapping that refuses everything would not produce the five lines above. +echo "no CLUSTER grant -> $(required_privilege "$no_cluster_user" "KILL TRANSACTION ON CLUSTER $cluster WHERE $txn_predicate")" + +# The two statements whose privileges can be granted in full are allowed in both spellings, which +# proves the new elements are a gate rather than an unconditional refusal. The local half is asserted +# too: a user who is cluster-allowed while locally refused is the bypass this test exists to catch. +# The thread fuzzer and coverage statements get no allowed arm, because executing them would change +# server-global state that concurrent tests read and neither takes an argument that lets the host +# reject them harmlessly. +allowed() { + local out + out=$($CLICKHOUSE_CLIENT --distributed_ddl_output_mode none --user "$1" --query "$2" 2>&1) + if [ -z "$out" ]; then + echo "$3 -> allowed" + else + echo "$3 -> FAIL: $out" + fi +} + +allowed "$kill_txn_user" "KILL TRANSACTION WHERE $txn_predicate" "KILL TRANSACTION local" +allowed "$kill_txn_user" "KILL TRANSACTION ON CLUSTER $cluster WHERE $txn_predicate" "KILL TRANSACTION on cluster" +allowed "$move_user" "KILL PART_MOVE_TO_SHARD WHERE task_uuid = $task_uuid" "KILL PART_MOVE_TO_SHARD local" +allowed "$move_user" "KILL PART_MOVE_TO_SHARD ON CLUSTER $cluster WHERE task_uuid = $task_uuid" "KILL PART_MOVE_TO_SHARD on cluster" From f6351a498d1084bdba618af41021d798f689bf22 Mon Sep 17 00:00:00 2001 From: Robert Schulze Date: Wed, 23 Sep 2026 20:27:00 +0000 Subject: [PATCH 036/185] Fix a bug that MySQL/Postgres queries may run in an expired session --- src/Interpreters/Session.cpp | 9 +- src/Interpreters/Session.h | 2 +- .../__init__.py | 0 .../configs/protocols.xml | 4 + .../test.py | 112 ++++++++++++++++++ 5 files changed, 125 insertions(+), 2 deletions(-) create mode 100644 tests/integration/test_auth_method_valid_until_stateful_protocols/__init__.py create mode 100644 tests/integration/test_auth_method_valid_until_stateful_protocols/configs/protocols.xml create mode 100644 tests/integration/test_auth_method_valid_until_stateful_protocols/test.py diff --git a/src/Interpreters/Session.cpp b/src/Interpreters/Session.cpp index 6b6cc000b61f..17521ee3d1bc 100644 --- a/src/Interpreters/Session.cpp +++ b/src/Interpreters/Session.cpp @@ -413,7 +413,7 @@ void Session::authenticate(const Credentials & credentials_, const Poco::Net::So prepared_client_info->connection_address = Poco::Net::SocketAddress(connection_address ? *connection_address : address); } -void Session::checkIfUserIsStillValid() +void Session::checkIfUserIsStillValid() const { if (const auto valid_until = user_authenticated_with.getValidUntil()) { @@ -696,6 +696,13 @@ ContextMutablePtr Session::makeQueryContextImpl(const ClientInfo * client_info_t if (!user_id && getClientInfo().interface != ClientInfo::Interface::TCP_INTERSERVER) throw Exception(ErrorCodes::LOGICAL_ERROR, "Query context must be created after authentication"); + /// The authentication method's `VALID UNTIL` must be enforced per query, not only at login: + /// stateful protocols (MySQL, PostgreSQL, native TCP, gRPC, Arrow Flight) authenticate once and + /// then create a query context per command, so an expired credential must stop working here. + /// Interserver connections replay an already-checked initiator identity, so they are exempt. + if (getClientInfo().interface != ClientInfo::Interface::TCP_INTERSERVER) + checkIfUserIsStillValid(); + /// We can create a query context either from a session context or from a global context. const bool from_session_context = static_cast(session_context) && !detached; diff --git a/src/Interpreters/Session.h b/src/Interpreters/Session.h index 055c8b8bcbd9..4a8bf55154f3 100644 --- a/src/Interpreters/Session.h +++ b/src/Interpreters/Session.h @@ -62,7 +62,7 @@ class Session // Verifies whether the user's validity extends beyond the current time. // Throws an exception if the user's validity has expired. - void checkIfUserIsStillValid(); + void checkIfUserIsStillValid() const; /// Writes a row about login failure into session log (if enabled) void onAuthenticationFailure(const std::optional & user_name, const Poco::Net::SocketAddress & address_, const Exception & e); diff --git a/tests/integration/test_auth_method_valid_until_stateful_protocols/__init__.py b/tests/integration/test_auth_method_valid_until_stateful_protocols/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_auth_method_valid_until_stateful_protocols/configs/protocols.xml b/tests/integration/test_auth_method_valid_until_stateful_protocols/configs/protocols.xml new file mode 100644 index 000000000000..65eafc59c3f8 --- /dev/null +++ b/tests/integration/test_auth_method_valid_until_stateful_protocols/configs/protocols.xml @@ -0,0 +1,4 @@ + + 9001 + 5433 + diff --git a/tests/integration/test_auth_method_valid_until_stateful_protocols/test.py b/tests/integration/test_auth_method_valid_until_stateful_protocols/test.py new file mode 100644 index 000000000000..1faf74afbbeb --- /dev/null +++ b/tests/integration/test_auth_method_valid_until_stateful_protocols/test.py @@ -0,0 +1,112 @@ +import time + +import psycopg2 +import pymysql.connections +import pytest + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +# The authentication method's `VALID UNTIL` must be enforced per query, not only at login. +# Stateful protocols (MySQL, PostgreSQL) authenticate once at connection startup and then run +# every later command through `Session::makeQueryContext`, so without a per-query re-check a +# credential that expires after login would keep working for the lifetime of the connection. +# The check lives in `Session::makeQueryContextImpl`, so every protocol shares it. +node = cluster.add_instance("node", main_configs=["configs/protocols.xml"]) + +MYSQL_PORT = 9001 +POSTGRES_PORT = 5433 + +# Lifetime of the expiring credential, measured from user creation. The first query on each +# connection runs within a fraction of a second of creation (well inside the lifetime); the second +# query runs after sleeping past the expiry. +EXPIRING_LIFETIME_S = 6 + + +@pytest.fixture(scope="module") +def started_cluster(): + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def create_users(): + node.query("DROP USER IF EXISTS u_expiring, u_lasting") + expiry = node.query(f"SELECT toString(now() + INTERVAL {EXPIRING_LIFETIME_S} SECOND)").strip() + # Plaintext-stored passwords so both the MySQL and PostgreSQL frontends can verify them. + node.query(f"CREATE USER u_expiring IDENTIFIED WITH plaintext_password BY 'pw' VALID UNTIL '{expiry}'") + node.query("CREATE USER u_lasting IDENTIFIED WITH plaintext_password BY 'pw' VALID UNTIL '2999-01-01 00:00:00'") + node.query("GRANT SELECT ON system.one TO u_expiring, u_lasting") + return expiry + + +def sleep_past(expiry): + while int(node.query(f"SELECT now() > toDateTime('{expiry}')").strip()) != 1: + time.sleep(0.5) + # The check is `now > valid_until` with second precision, so cross the boundary decisively. + time.sleep(1.5) + + +def test_mysql_connection_stops_working_after_expiry(started_cluster): + expiry = create_users() + host = started_cluster.get_instance_ip("node") + + def connect(user): + return pymysql.connections.Connection(host=host, user=user, password="pw", database="default", port=MYSQL_PORT) + + expiring = connect("u_expiring") + lasting = connect("u_lasting") + + # Both credentials are valid at connection time and for the first query. + for client in (expiring, lasting): + cursor = client.cursor() + cursor.execute("SELECT 1") + assert cursor.fetchall() == ((1,),) + + sleep_past(expiry) + + # The same, still-open connection must stop working once the method has expired ... + with pytest.raises(pymysql.err.MySQLError, match="expired"): + expiring.cursor().execute("SELECT 1") + + # ... while a connection under a non-expired method keeps working. + cursor = lasting.cursor() + cursor.execute("SELECT 1") + assert cursor.fetchall() == ((1,),) + + lasting.close() + + +def test_postgresql_connection_stops_working_after_expiry(started_cluster): + expiry = create_users() + host = started_cluster.get_instance_ip("node") + + def connect(user): + conn = psycopg2.connect(host=host, user=user, password="pw", dbname="default", port=POSTGRES_PORT) + conn.autocommit = True + return conn + + expiring = connect("u_expiring") + lasting = connect("u_lasting") + + for client in (expiring, lasting): + cursor = client.cursor() + cursor.execute("SELECT 1") + assert cursor.fetchall() == [(1,)] + + sleep_past(expiry) + + # The PostgreSQL handler sends an ErrorResponse and then terminates the connection on an expired + # credential, and psycopg2 may surface either the message or the connection loss - both are + # fail-close; the point is that the query must not succeed. + with pytest.raises(psycopg2.Error, match="expired|closed the connection"): + expiring.cursor().execute("SELECT 1") + + cursor = lasting.cursor() + cursor.execute("SELECT 1") + assert cursor.fetchall() == [(1,)] + + lasting.close() From 8fa70cde6514bd319633ad16e3329a1f1233269e Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 20:12:00 +0000 Subject: [PATCH 037/185] Backport #114344 to 26.8: Validate a verbatim codec's declared size in the passthrough read shortcut --- src/Compression/CompressedReadBufferBase.cpp | 6 ++ ...mpressed_decompressed_size_bound.reference | 13 +++ ...4848_compressed_decompressed_size_bound.sh | 85 +++++++++++++++++++ 3 files changed, 104 insertions(+) create mode 100644 tests/queries/0_stateless/04848_compressed_decompressed_size_bound.reference create mode 100755 tests/queries/0_stateless/04848_compressed_decompressed_size_bound.sh diff --git a/src/Compression/CompressedReadBufferBase.cpp b/src/Compression/CompressedReadBufferBase.cpp index b32d51ef27e4..d116c7baa1d0 100644 --- a/src/Compression/CompressedReadBufferBase.cpp +++ b/src/Compression/CompressedReadBufferBase.cpp @@ -310,6 +310,12 @@ void CompressedReadBufferBase::decompress(BufferBase::Buffer & to, size_t size_d "Can't decompress data: the compressed data size ({}, this should include header size) is less than the header size ({})", size_compressed_without_checksum, static_cast(header_size)); + if (size_compressed_without_checksum - header_size != size_decompressed) + throw Exception(external_data ? ErrorCodes::CANNOT_DECOMPRESS : ErrorCodes::CORRUPTED_DATA, + "Can't decompress data: the compressed data size without header ({}) does not match size_decompressed ({}) " + "for a codec that stores data uncompressed", + size_compressed_without_checksum - header_size, size_decompressed); + to = BufferBase::Buffer(compressed_buffer + header_size, compressed_buffer + size_compressed_without_checksum); } else diff --git a/tests/queries/0_stateless/04848_compressed_decompressed_size_bound.reference b/tests/queries/0_stateless/04848_compressed_decompressed_size_bound.reference new file mode 100644 index 000000000000..74b069ccd930 --- /dev/null +++ b/tests/queries/0_stateless/04848_compressed_decompressed_size_bound.reference @@ -0,0 +1,13 @@ +-- a valid frame still executes (proves the arms below fail for the intended reason) +1 +-- a codec that stores data uncompressed must not lie about the uncompressed size +1 +-- and neither may the other verbatim codec +1 +-- nor may it understate the uncompressed size +1 +-- engines keep working on ordinary data +1000 100000 +1000 100000 +1 +1 diff --git a/tests/queries/0_stateless/04848_compressed_decompressed_size_bound.sh b/tests/queries/0_stateless/04848_compressed_decompressed_size_bound.sh new file mode 100755 index 000000000000..a0248bf8eb06 --- /dev/null +++ b/tests/queries/0_stateless/04848_compressed_decompressed_size_bound.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +set -o pipefail + +# Wire layout of a compressed frame: +# [16B checksum][1B method][4B LE size_compressed][4B LE size_decompressed][payload] +# size_compressed counts the 9-byte header. The checksum is an unkeyed CityHash128 of everything +# after it, so a peer can compute it: these frames are accepted at default settings, and +# http_native_compression_disable_checksumming_on_decompress is deliberately NOT used. +# +# checksum-for-compressed-block prints CityHash128 of every single-bit mutation of its input, so +# feeding it the body with bit 0 flipped yields CityHash128(body) on the line labelled "0, 0". The +# wire order is low64 then high64, each little-endian, i.e. the reverse of the printed hex. +emit() { # $1 = frame body as hex, checksum excluded + local checksum + checksum=$(python3 -c " +import sys +b = bytearray.fromhex(sys.argv[1]); b[0] ^= 1 +sys.stdout.buffer.write(bytes(b)) +" "$1" | $CLICKHOUSE_BINARY checksum-for-compressed-block | awk -F'\t' '$2 == "0, 0" { print $1; exit }') + python3 -c " +import sys +sys.stdout.buffer.write(bytearray.fromhex(sys.argv[1])[::-1] + bytearray.fromhex(sys.argv[2])) +" "$checksum" "$1" +} + +# Method bytes are from CompressionInfo.h: NONE is 0x02 = 2 and Quantized is 0x9e = 158. +frame() { # $1 = method byte (decimal), $2 = size_decompressed, $3 = payload + emit "$(python3 -c " +import struct, sys +payload = sys.argv[3].encode() +sys.stdout.buffer.write(bytes([int(sys.argv[1])]) + struct.pack('&1 | grep -c '(8) does not match size_decompressed (999)' +# Quantized is the other codec reporting isNone(), so it takes the same shortcut. The read path +# builds it from the method byte alone, without the enable_quantized_codec setting its DDL requires. +echo '-- and neither may the other verbatim codec' +frame 158 999 'SELECT 1' | post 2>&1 | grep -c '(8) does not match size_decompressed (999)' + +# The comparison is for inequality, so arms that only ever declare more than the body pin one side of +# it: narrowed to "body shorter than declared", both frames above would still be refused. This one +# declares less instead, and without the check it is not refused at all, it executes its real body. +echo '-- nor may it understate the uncompressed size' +frame 2 3 'SELECT 1' | post 2>&1 | grep -c '(8) does not match size_decompressed (3)' + +# No-regression control: only a NONE-coded frame takes the shortcut, so the column, the marks and +# the primary key are all stored uncompressed here. StripeLog is absent because it compresses its +# whole stream with the default codec and ignores the column codec, so its reads never take it. +echo '-- engines keep working on ordinary data' +${CLICKHOUSE_CLIENT} --query " + DROP TABLE IF EXISTS t_log_bound; + DROP TABLE IF EXISTS t_mt_bound; + + CREATE TABLE t_log_bound (s String CODEC(NONE)) ENGINE = Log; + CREATE TABLE t_mt_bound (k UInt64, s String CODEC(NONE), INDEX idx_s s TYPE minmax GRANULARITY 1) + ENGINE = MergeTree ORDER BY k + SETTINGS index_granularity = 8, compress_marks = 1, compress_primary_key = 1, + min_bytes_for_wide_part = 0, packed_skip_index_max_bytes = 0, + marks_compression_codec = 'NONE', primary_key_compression_codec = 'NONE'; + + INSERT INTO t_log_bound SELECT repeat('a', 100) FROM numbers(1000); + INSERT INTO t_mt_bound SELECT number, repeat('a', 100) FROM numbers(1000); + + SELECT count(), sum(length(s)) FROM t_log_bound; + SELECT count(), sum(length(s)) FROM t_mt_bound WHERE s LIKE '%a%'; + SELECT count() FROM t_mt_bound WHERE k = 42; + -- packed_skip_index_max_bytes = 0 keeps the index in its own file, which is the skip-index class + -- checkDataPart iterates over; a packed one carries no per-file checksum entry to visit. + CHECK TABLE t_mt_bound SETTINGS check_query_single_value_result = 1; + + DROP TABLE t_log_bound; + DROP TABLE t_mt_bound; +" From 0963256af9c256b82cd7e6083911c0b60b33ee82 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 24 Sep 2026 20:44:46 +0000 Subject: [PATCH 038/185] Backport #120282 to 26.8: Bound column count in inserted RowBinary header --- .../Formats/Impl/BinaryRowInputFormat.cpp | 11 +++++ ...owbinary_header_too_many_columns.reference | 3 ++ ...04603_rowbinary_header_too_many_columns.sh | 48 +++++++++++++++++++ 3 files changed, 62 insertions(+) create mode 100644 tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.reference create mode 100755 tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.sh diff --git a/src/Processors/Formats/Impl/BinaryRowInputFormat.cpp b/src/Processors/Formats/Impl/BinaryRowInputFormat.cpp index 3b10b98a2783..bc380233d0cc 100644 --- a/src/Processors/Formats/Impl/BinaryRowInputFormat.cpp +++ b/src/Processors/Formats/Impl/BinaryRowInputFormat.cpp @@ -6,6 +6,7 @@ #include #include #include +#include #include namespace DB @@ -14,8 +15,12 @@ namespace DB namespace ErrorCodes { extern const int CANNOT_SKIP_UNKNOWN_FIELD; + extern const int TOO_LARGE_ARRAY_SIZE; } +/// Bound number of columns in header so user cannot reserve() arbitrarily large amount of memory +static constexpr auto TOO_MANY_COLUMNS_MESSAGE = "Suspiciously many columns in RowBinary header: {}"; + template BinaryRowInputFormat::BinaryRowInputFormat(ReadBuffer & in_, SharedHeader header, IRowInputFormat::Params params_, bool with_names_, bool with_types_, const FormatSettings & format_settings_) : RowInputFormatWithNamesAndTypes>( @@ -88,6 +93,8 @@ template std::vector BinaryFormatReader::readNames() { readVarUInt(read_columns, *in); + if (read_columns > DEFAULT_NATIVE_BINARY_MAX_NUM_COLUMNS) + throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, TOO_MANY_COLUMNS_MESSAGE, read_columns); return readHeaderRow(); } @@ -150,6 +157,8 @@ template void BinaryFormatReader::skipNames() { readVarUInt(read_columns, *in); + if (read_columns > DEFAULT_NATIVE_BINARY_MAX_NUM_COLUMNS) + throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, TOO_MANY_COLUMNS_MESSAGE, read_columns); skipHeaderRow(); } @@ -160,6 +169,8 @@ void BinaryFormatReader::skipTypes() { /// It's possible only when with_names = false and with_types = true readVarUInt(read_columns, *in); + if (read_columns > DEFAULT_NATIVE_BINARY_MAX_NUM_COLUMNS) + throw Exception(ErrorCodes::TOO_LARGE_ARRAY_SIZE, TOO_MANY_COLUMNS_MESSAGE, read_columns); } skipHeaderRow(); diff --git a/tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.reference b/tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.reference new file mode 100644 index 000000000000..6cc8f3258bf9 --- /dev/null +++ b/tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.reference @@ -0,0 +1,3 @@ +1 +TOO_LARGE_ARRAY_SIZE +TOO_LARGE_ARRAY_SIZE diff --git a/tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.sh b/tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.sh new file mode 100755 index 000000000000..627efc72b50a --- /dev/null +++ b/tests/queries/0_stateless/04603_rowbinary_header_too_many_columns.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +# Tests that a RowBinaryWithNames[AndTypes] header with a suspiciously large column +# count is rejected with TOO_LARGE_ARRAY_SIZE instead of amplifying into a huge +# allocation. See https://github.com/ClickHouse/clickhouse-private/issues/69219 + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +${CLICKHOUSE_CLIENT} -q "DROP TABLE IF EXISTS t_rowbinary_header" +${CLICKHOUSE_CLIENT} -q "CREATE TABLE t_rowbinary_header (x UInt64) ENGINE = Memory" + +# LEB128 encoder. +varint() { + python3 -c ' +import sys +n = int(sys.argv[1]) +out = bytearray() +while True: + b = n & 0x7F + n >>= 7 + if n: + b |= 0x80 + out.append(b) + if not n: + break +sys.stdout.buffer.write(bytes(out)) +' "$1" +} + +# Negative control: a well-formed one-column header (name "x", type "UInt64") plus one +# 8-byte UInt64 row inserts fine. +{ varint 1; printf '\x01x\x06UInt64'; printf '\x01\x00\x00\x00\x00\x00\x00\x00'; } \ + | ${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&query=INSERT+INTO+t_rowbinary_header+FORMAT+RowBinaryWithNamesAndTypes" --data-binary @- +${CLICKHOUSE_CLIENT} -q "SELECT count() FROM t_rowbinary_header" + +# Attack: a header claiming far more columns than the ceiling (1'000'000). +# The tiny body must be rejected up-front, not amplified. +varint 2000000 \ + | ${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&query=INSERT+INTO+t_rowbinary_header+FORMAT+RowBinaryWithNamesAndTypes" --data-binary @- 2>&1 \ + | grep -o "TOO_LARGE_ARRAY_SIZE" | head -n1 + +# Same for the WithNames variant (the names header is a separate read path). +varint 2000000 \ + | ${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&query=INSERT+INTO+t_rowbinary_header+FORMAT+RowBinaryWithNames" --data-binary @- 2>&1 \ + | grep -o "TOO_LARGE_ARRAY_SIZE" | head -n1 + +${CLICKHOUSE_CLIENT} -q "DROP TABLE t_rowbinary_header" From a8a37bb99b61b2db79fb711bf3ebcefc7891ecc3 Mon Sep 17 00:00:00 2001 From: George Larionov Date: Thu, 24 Sep 2026 23:36:09 +0000 Subject: [PATCH 039/185] Fix settings history in backport of #121424 to 26.8 Drop the leaked `26.10` and `26.9` blocks; record the changes in the `26.8` block. `ai_function_max_api_calls_per_query` no longer changes in 26.8, so its entry is removed. Co-Authored-By: Claude Opus 5.5 (1M context) --- src/Core/SettingsChangesHistory.cpp | 105 +----------------- ...functions_compatibility_defaults.reference | 2 - ...28_ai_functions_compatibility_defaults.sql | 15 +-- 3 files changed, 6 insertions(+), 116 deletions(-) diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 97a310296201..867808a27809 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -41,108 +41,6 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() /// controls new feature and it's 'true' by default, use 'false' as previous_value). /// It's used to implement `compatibility` setting (see https://github.com/ClickHouse/ClickHouse/issues/35972) /// Note: please check if the key already exists to prevent duplicate entries. - addSettingsChanges(settings_changes_history, "26.10", - { - {"allow_executable_tables", true, true, "New setting to disable reading through the `executable` table function and from `Executable` and `ExecutablePool` tables."}, - {"ai_function_max_input_tokens_per_query", 1000000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, - {"ai_function_max_output_tokens_per_query", 500000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, - {"ai_function_max_api_calls_per_query", 1000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, - {"iceberg_tolerate_conflicting_manifest_schemas", false, true, "New setting: when an Iceberg manifest file header carries a schema that conflicts with the schema registered for the same schema-id from metadata.json, prefer the metadata.json schema and log a warning instead of failing the query, matching the behavior of other query engines. `compatibility` below 26.10 restores the previous strict behavior."}, - {"legacy_join_size_limits_trigger_spilling", true, false, "`max_rows_in_join` / `max_bytes_in_join` are now hard caps for every hash join, including the ones that spill to disk, where they used to act as the spill trigger. Spilling is driven by `max_bytes_before_external_join` / `max_bytes_ratio_before_external_join` alone."}, - {"prefer_optimize_projection", false, false, "New setting: choose a usable projection regardless of its estimated cost, like `force_optimize_projection`, but without failing the query when no projection is used."}, - {"query_plan_optimize_join_order_conflict_detector", "", "", "New setting selecting the conflict detector that decides join reordering validity in the DPsub join order algorithm: `a` for the (correct but incomplete) CD-A, `c` for the (correct and complete) CD-C, empty for none."}, - {"use_text_index_postings_cache", false, true, "Enabled the text index posting lists cache globally. Previously each query used a small private cache, which caused posting lists and phrase search results to be recomputed within a single query on large tables."}, - {"output_format_arrow_unsupported_types", "binary", "binary", "New setting superseding `output_format_arrow_unsupported_types_as_binary`, adding a `text` mode. Its default matches the previous behavior, so `compatibility` must not change it."}, - }); - addSettingsChanges(settings_changes_history, "26.9", - { - {"max_bytes_before_external_distinct", 0, 0, "New setting to enable spilling of `DISTINCT` to disk when memory usage exceeds the given threshold in bytes. If 0, only `max_bytes_ratio_before_external_distinct` applies."}, - {"max_bytes_ratio_before_external_distinct", 0., 0.5, "New setting to enable spilling of `DISTINCT` to disk when memory usage exceeds the given ratio of available memory. If 0, only `max_bytes_before_external_distinct` applies."}, - {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7 or 26.8: the previous value is deliberately equal to the new one, because those versions already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, - {"allow_delta_lake_create_table", false, false, "New setting: allow creating a new DeltaLake table using delta-kernel-rs or registering an existing one into a catalog."}, - {"delta_lake_accurate_write_cast", false, true, "New setting: cast written values to the Delta write-schema type with an accurate cast that throws on a value that does not fit the target type instead of silently truncating; `compatibility` below 26.9 uses the plain, non-throwing cast."}, - {"allow_experimental_nullable_tuple_type", false, true, "`Nullable(Tuple)` is now GA"}, - {"enable_nullable_tuple_type", false, true, "`Nullable(Tuple)` is now GA"}, - {"allow_nullable_tuple_in_extracted_subcolumns", false, true, "`Nullable(Tuple)` is now GA: a `Tuple` subcolumn extracted from a `Tuple`, `Variant`, `Dynamic` or `JSON` column is `Nullable(Tuple)` and is NULL in the rows where the subcolumn is missing. The setting is read once at server startup, so `compatibility` restores the previous behavior only from the startup profile (for example, users.xml), not from a session-level `SET`."}, - {"workload_admission_timeout_ms", 0, 0, "New setting bounding how long a query waits to be admitted by workload scheduling (acquiring its query slot and memory reservation) before failing; 0 (default) preserves the previous unbounded wait."}, - {"s3_disable_checksum", false, false, "Obsolete setting: checksum calculation no longer re-reads the source"}, - {"session_query_ids_history_size", 0, 1000, "New setting limiting the size of the session-local query id history exposed through the new `system.session_query_ids` system table. The previous value `0` (recording disabled) reproduces the pre-26.9 behavior."}, - {"query_plan_optimize_join_order_conflict_detector", "", "", "New setting selecting the conflict detector that decides join reordering validity in the DPsub join order algorithm: `a` for the (correct but incomplete) CD-A, `c` for the (correct and complete) CD-C, empty for none."}, - {"reader_executor_window_size", 4194304, 8388608, "Raised the default read window of the experimental `ReaderExecutor` from 4 MiB to 8 MiB. Under memory pressure the window is reduced from this base, floored at 128 KiB."}, - {"webassembly_udf_input_split_memory_ratio", 0.0, 0.5, "New setting controlling the fraction of a WebAssembly UDF instance's linear memory that one call's serialized input may occupy, which also enables the dynamic splitting of that input by its serialized size; `compatibility` below 26.9 sets it to 0 and restores the previous behavior, where `webassembly_udf_max_input_block_size = 0` meant one call per pipeline block."}, - {"cascades_aggregation_pushdown", false, true, "New setting to consider pushing partial aggregation below a join (eager aggregation) in the Cascades optimizer."}, - {"optimize_read_in_reverse_order_final", false, true, "New setting to enable the read-in-order optimization when reading in reverse order of the sorting key with the `FINAL` modifier from `ReplacingMergeTree` tables."}, - {"load_marks_asynchronously", false, true, "Load marks of all streams in parallel by default. On remote disks, synchronous loading of marks of columns with many substreams (such as `JSON`) took one network round trip per stream."}, - {"ast_fuzzer_oracle", false, false, "New setting to enable correctness oracle checks in the server-side AST fuzzer."}, - {"create_token_default_ttl_seconds", 1800, 1800, "New setting giving a lifetime to a token created by `CREATE TOKEN` without an explicit `VALID UNTIL` or `VALID FOR` clause. The statement is new, so there is no earlier behavior to restore and the previous value is the default itself: a `compatibility` with an older version must not turn tokens into never-expiring ones."}, - {"enable_hash_join_row_store", false, true, "New setting to enable transforming the payload of a hash join into a row-major layout."}, - {"min_rows_ratio_for_hash_join_row_store", 5.0, 5.0, "New setting to control the minimum estimated ratio of join output rows to build-side rows to enable transforming hash join payload to row-major. 0 means the transformation is always allowed."}, - {"enable_sharding_aggregator", false, false, "Obsolete setting, the sharded aggregator has been removed in favor of the adaptive aggregator (`enable_adaptive_aggregator`)."}, - {"allow_preliminary_distinct_abandoning", false, true, "New setting that lets the preliminary `DISTINCT` give up deduplicating mostly-unique input, because the final `DISTINCT` deduplicates its output again."}, - {"query_plan_fuse_filter_into_array_join", false, true, "New optimization to fuse a filter on ARRAY JOINed columns into the ARRAY JOIN step, enabled by default."}, - {"iceberg_file_entries_queue_size", 100, 100, "New setting for the previously hardcoded capacity of the queue between the Iceberg data manifest decode tasks and the query."}, - {"iceberg_manifest_decode_concurrency", 2, 4, "New setting bounding how many Iceberg manifest files are decoded concurrently, for delete and data manifests alike. It replaces `iceberg_delete_manifest_decode_concurrency` (kept as an alias). `2` approximates the pre-26.9 data path, which decoded one manifest at a time with the next one's fetch already in flight; under `compatibility` at or below 26.8 the delete decode therefore also runs at 2 rather than its released default of 4, preserving the older data-path memory envelope at the cost of some delete-decode overlap."}, - {"query_plan_lower_array_join_function", false, false, "New optimization to lower an arrayJoin function into a real ARRAY JOIN step; disabled by default."}, - {"adaptive_aggregator_freeze_threshold_bytes", 4194304, 4194304, "New setting bounding the adaptive aggregator's frozen local tables in bytes, whichever of it and the key-count threshold is reached first; 0 disables the byte bound."}, - {"allow_experimental_ai_functions", false, false, "The setting is obsolete, AI functions are beta now and enabled by default."}, - {"allow_experimental_analyzer", true, true, "The setting is obsolete: the analyzer is mandatory and the old query analysis is no longer supported. Disabling it is refused instead of being ignored, and `compatibility` with a version below 24.3 no longer reverts it."}, - {"allow_url_wildcard_from_index_pages", false, false, "Added an alias for setting `allow_experimental_url_wildcard_from_index_pages`."}, - {"allow_kafka_offsets_storage_in_keeper", false, false, "Added an alias for setting `allow_experimental_kafka_offsets_storage_in_keeper`."}, - {"allow_correlated_subqueries", true, true, "Added an alias for setting `allow_experimental_correlated_subqueries`."}, - {"allow_geo_types_in_iceberg", false, false, "Added an alias for setting `allow_experimental_geo_types_in_iceberg`."}, - {"enable_materialized_postgresql_table", false, false, "Added an alias for setting `allow_experimental_materialized_postgresql_table`."}, - {"enable_funnel_functions", false, false, "Added an alias for setting `allow_experimental_funnel_functions`."}, - {"enable_unique_key", false, false, "Added an alias for setting `allow_experimental_unique_key`."}, - {"allow_join_right_table_sorting", false, false, "Added an alias for setting `allow_experimental_join_right_table_sorting`."}, - {"enable_json_lazy_type_hints", false, false, "Added an alias for setting `allow_experimental_json_lazy_type_hints`."}, - {"enable_join_runtime_filters", true, true, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_filter_exact_values_limit", 10000, 10000, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_bloom_filter_bytes", 512_KiB, 512_KiB, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_bloom_filter_hash_functions", 3, 3, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_filter_pass_ratio_threshold_for_disabling", 0.7, 0.7, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_filter_blocks_to_skip_before_reenabling", 30, 30, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_bloom_filter_max_ratio_of_set_bits", 0.7, 0.7, "The JOIN runtime filters became a Production tier feature."}, - {"join_runtime_filter_min_probe_rows", 1000, 1000, "The JOIN runtime filters became a Production tier feature."}, - {"enable_join_runtime_filters_index_analysis", false, false, "The JOIN runtime filters became a Production tier feature."}, - {"ai_function_max_retries", 0, 1, "Retry a transient API error once by default, so a single 429 or 5xx from the provider does not fail the query."}, - {"query_plan_aggregation_bucket_top_k", false, true, "New setting to toggle the plan optimization that materializes only each two-level bucket's best n groups when a final aggregation feeds ORDER BY over its outputs with LIMIT n and the per-bucket selection is provably exact."}, - {"enable_trino_dialect", false, false, "New setting to enable the `trino` value of the `dialect` setting, which translates Trino SQL syntax and maps Trino function names to ClickHouse equivalents."}, - {"enable_join_key_only_hash_tables", false, true, "New setting to store the join keys alone, without a reference to a right row, in the hash tables of joins whose result can never contain a value taken from a right row (`LEFT ANTI`, and `LEFT SEMI` when no right column is selected)."}, - {"distributed_plan_read_in_order", false, false, "New setting to allow the read-in-order optimization for `ORDER BY` in a distributed query plan, so a sorted read of the table's sorting key can skip the sort and stop early. Off by default: only shapes where no exchange survives between the read and the sort are safe today."}, - {"distributed_cache_client_id", "", "", "New setting (CI tests only) to override the distributed cache client id per query."}, - {"query_plan_propagate_predicate_across_join", false, true, "New setting that lifts filter conjuncts across equi-join keys so primary-key pruning fires on both sides."}, - {"read_through_distributed_cache", false, false, "The setting moved to the server configuration and is ignored as a profile setting: reading from the distributed cache is now switched by the server setting `enable_read_through_distributed_cache`, which is applied without a restart (so that merges, mutations and `Buffer` flushes follow it too, instead of being pinned to the value the server started with). Use `force_read_through_distributed_cache` to deviate from the server setting per query."}, - {"write_through_distributed_cache", false, false, "The setting moved to the server configuration and is ignored as a profile setting: writing to the distributed cache is now switched by the server setting `enable_write_through_distributed_cache`, which is applied without a restart (so that merges, mutations and `Buffer` flushes follow it too, instead of being pinned to the value the server started with). Use `force_write_through_distributed_cache` to deviate from the server setting per query."}, - {"force_read_through_distributed_cache", "auto", "auto", "New setting overriding the server setting `enable_read_through_distributed_cache` for a single query."}, - {"force_write_through_distributed_cache", "auto", "auto", "New setting overriding the server setting `enable_write_through_distributed_cache` for a single query."}, - {"distributed_cache_min_inflight_bytes_to_discard_connection_on_seek", 0, 4 * 1024 * 1024, "New setting to drop and reopen a distributed cache connection on a seek when too many in-flight bytes would otherwise be discarded. Defaults to 4 MiB; 0 restores the previous behavior (always reuse the connection via the read range id)."}, - {"distributed_plan_workers_provisioning_timeout_ms", 10000, 10000, "New setting bounding how long a query waits for leased stateless workers to become reachable before execution."}, - {"distributed_plan_fallback_to_local_execution", false, true, "New setting to fall back to local execution when a plan cannot be distributed (only takes effect under `make_distributed_plan`)."}, - {"query_plan_optimize_lazy_materialization_for_object_storage", false, true, "New setting to use lazy materialization for `ORDER BY ... LIMIT n` queries reading Parquet files from object storage (including Iceberg tables)."}, - {"iceberg_compaction_commit_batch_size", 100, 100, "New setting"}, - {"iceberg_compaction_max_rows_in_data_file", std::numeric_limits::max(), std::numeric_limits::max(), "New setting for the max rows of an iceberg data file produced by compaction, separate from the insert-time limit."}, - {"iceberg_compaction_max_bytes_in_data_file", std::numeric_limits::max(), std::numeric_limits::max(), "New setting for the max bytes of an iceberg data file produced by compaction, separate from the insert-time limit."}, - {"enable_json_lazy_type_hints", false, false, "Lazy JSON type hints are now Beta. An alias for setting 'allow_experimental_json_lazy_type_hints'."}, - {"s3_upload_checksum_algorithm", "", "", "New setting to choose the checksum algorithm for S3 uploads."}, - {"network_compression_method", "LZ4", "ZSTD", "Switched the default compression method for client/server and server/server communication from `LZ4` to `ZSTD` to reduce network traffic."}, - {"network_zstd_compression_level", 1, 3, "Aligned the default network `ZSTD` compression level with the new default on-disk `ZSTD(3)` compression."}, - {"use_statistics_for_min_max_aggregation", false, true, "New setting to answer `min`, `max` and `count` aggregations without `GROUP BY` and filters from per-part column statistics for parts that have them materialized, reading only the remaining parts. previous_value=false so `compatibility` with versions before 26.9 keeps the optimization disabled and restores the pre-existing plan."}, - {"parallel_replicas_allow_merge_tables", false, false, "New setting to allow reading from a `Merge` table with plan-based parallel replicas, by expanding the `Merge` read into a union of the reads from the underlying `MergeTree` tables. It only has an effect together with `parallel_replicas_plan_based`."}, - {"iceberg_compaction_max_bytes_in_data_file", std::numeric_limits::max(), 512 * 1024 * 1024, "New setting for the max bytes of an iceberg data file produced by compaction, separate from the insert-time limit. The default is aligned with the documented default of the Iceberg table property `write.target-file-size-bytes` (512 MiB), see https://iceberg.apache.org/docs/1.5.2/configuration/. Previously compaction merged all eligible files of a partition into a single output file."}, - {"iceberg_insert_max_bytes_in_data_file", 1024 * 1024 * 1024, 512 * 1024 * 1024, "Aligned with the documented default of the Iceberg table property `write.target-file-size-bytes` (512 MiB), see https://iceberg.apache.org/docs/1.5.2/configuration/."}, - {"iceberg_data_file_size_lower_threshold_compaction", 10 * 1024 * 1024, 384 * 1024 * 1024, "Aligned with how the Iceberg `rewrite_data_files` procedure derives `min-file-size-bytes`: 0.75 of the target file size (512 MiB). Compaction now selects files below 384 MiB instead of below 10 MiB."}, - {"iceberg_data_file_size_upper_threshold_compaction", 10ULL * 1024 * 1024 * 1024, 512ULL * 1024 * 1024 * 9 / 5, "Aligned with how the Iceberg `rewrite_data_files` procedure derives `max-file-size-bytes`: 1.8 of the target file size (512 MiB)."}, - {"iceberg_manifest_min_count_to_compact", 30, 100, "Aligned with the documented default of the Iceberg table property `commit.manifest.min-count-to-merge` (100), see https://iceberg.apache.org/docs/1.5.2/configuration/."}, - {"parallel_replicas_allow_merge_tables", false, false, "New setting to allow reading from a `Merge` table with plan-based parallel replicas, by expanding the `Merge` read into a union of the reads from the underlying `MergeTree` tables. It only has an effect together with `parallel_replicas_plan_based`."}, - {"optimize_mutations_with_partition_pruning", false, true, "New setting to automatically prune partitions for mutations based on WHERE clause"}, - {"statistics_max_set_size_for_exact_selectivity_estimation", 10000, 10000, "The bound on the cost of estimating the selectivity of `IN` with a large set is kept under `compatibility` with an earlier version: the previous value is deliberately equal to the new one, so that the uncapped estimation, which could add hundreds of milliseconds to the planning of a single query, is not restored."}, - {"type_json_skip_null_typed_paths", false, false, "New setting to treat NULL values in typed JSON paths as absent"}, - {"use_iceberg_manifest_list_partition_pruning", false, true, "New setting to skip Iceberg manifest files whose manifest-list partition summaries cannot match the query filter, without reading them."}, - {"enable_time_series_table", false, false, "The `TimeSeries` table engine and the `promql` dialect were moved to the private preview tier. Added an alias for setting `allow_experimental_time_series_table`."}, - {"enable_time_series_aggregate_functions", false, false, "The `timeSeries*` aggregate functions were moved to the private preview tier. Added an alias for setting `allow_experimental_time_series_aggregate_functions`."}, - {"output_format_arrow_record_batch_size", 0, 0, "New setting to combine small blocks in `Arrow` and `ArrowStream` output using a target row count. The default `0` preserves one record batch per block."}, - {"output_format_arrow_record_batch_size_bytes", 0, 0, "New setting to combine small blocks in `Arrow` and `ArrowStream` output using a target size in bytes of accumulated data. The default `0` preserves one record batch per block."}, - }); addSettingsChanges(settings_changes_history, "26.8", { {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7: the previous value is deliberately equal to the new one, because 26.7 already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, @@ -204,7 +102,8 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"shrink_over_allocated_columns_min_waste_ratio", 1.0, 1.0, "New setting to shrink over-allocated columns to fit on INSERT to reduce peak memory usage. Disabled by default (1.0)."}, {"shrink_over_allocated_columns_min_waste_bytes", 16 * 1024 * 1024, 16 * 1024 * 1024, "New setting: minimum absolute wasted memory in a column for it to be shrunk to fit on INSERT."}, {"ai_function_allow_insecure_endpoint", true, false, "AI functions now reject insecure (http) endpoints to remote hosts by default."}, - {"ai_function_max_api_calls_per_query", 0, 1000, "Bound outbound AI function HTTP calls per query by default (previously 0 - unlimited)."}, + {"ai_function_max_input_tokens_per_query", 1000000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, + {"ai_function_max_output_tokens_per_query", 500000, 0, "The AI function per-query quotas are disabled by default: 0 means no limit."}, {"join_runtime_filter_min_probe_rows", 0, 1000, "New setting to control minimum probe side size for installing JOIN runtime filters. It wasn't limited before, so previous value is 0 meaning always install."}, {"read_in_order_use_virtual_row", false, true, "Enable the virtual row optimization by default. When reading in order of the primary key over many parts, it lets `MergingSortedTransform` reprioritize sources using primary key values from the sparse index, so parts that are not relevant for the query are not read, plus a bounded read-ahead window of at most `max_threads` parts that keeps reads parallel. This significantly reduces peak memory consumption (see https://github.com/ClickHouse/ClickHouse/issues/52624)."}, {"page", 0., 0., "New setting for paginated HTTP responses, equivalent to offset = limit * (page - 1). Float so it can hold negative or fractional values (passed through to SQL `LIMIT`/`OFFSET`)."}, diff --git a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference index b4345fecb835..078656c7dc9d 100644 --- a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference +++ b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.reference @@ -1,7 +1,5 @@ -- Current defaults false 0 1 --- compatibility = 26.9 restores the API call quota -false 1000 1 -- compatibility = 26.6 restores the legacy defaults true 0 0 0 diff --git a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql index 1881d8b85a6f..1fe196405fc4 100644 --- a/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql +++ b/tests/queries/0_stateless/04628_ai_functions_compatibility_defaults.sql @@ -3,12 +3,10 @@ -- no-replicated-database: named collections are server-global, not database-scoped -- ============================================================================= --- Four AI function default flips: `ai_function_allow_insecure_endpoint` from 1 to 0 and --- `ai_function_max_api_calls_per_query` from 0 (unlimited) to 1000 in 26.8, then --- `ai_function_max_retries` from 0 to 1 in 26.9, and `ai_function_max_api_calls_per_query` --- back to 0 (unlimited) in 26.10. `compatibility = 26.6` predates all of them and --- `compatibility = 26.9` reverts only the last one, which pins the previous_value/new_value --- pairs in `SettingsChangesHistory`. +-- Two AI function defaults were flipped in 26.8: `ai_function_allow_insecure_endpoint` from 1 +-- to 0 and `ai_function_max_retries` from 0 to 1. `ai_function_max_api_calls_per_query` stays 0 +-- (unlimited). `compatibility = 26.6` predates the flips and restores them, which pins the +-- previous_value/new_value pairs in `SettingsChangesHistory`. -- -- The endpoint check runs in `resolveAIParams`, before the zero-row early return -- in `executeImpl`, so an empty source table exercises it without any real HTTP @@ -26,11 +24,6 @@ SELECT '-- Current defaults'; SELECT getSetting('ai_function_allow_insecure_endpoint'), getSetting('ai_function_max_api_calls_per_query'), getSetting('ai_function_max_retries'); SELECT aiGenerate(x, map('credentials', 'ai_compat_remote_http')) FROM tab; -- { serverError BAD_ARGUMENTS } -SELECT '-- compatibility = 26.9 restores the API call quota'; -SET compatibility = '26.9'; -SELECT getSetting('ai_function_allow_insecure_endpoint'), getSetting('ai_function_max_api_calls_per_query'), getSetting('ai_function_max_retries'); -SELECT aiGenerate(x, map('credentials', 'ai_compat_remote_http')) FROM tab; -- { serverError BAD_ARGUMENTS } - SELECT '-- compatibility = 26.6 restores the legacy defaults'; SET compatibility = '26.6'; SELECT getSetting('ai_function_allow_insecure_endpoint'), getSetting('ai_function_max_api_calls_per_query'), getSetting('ai_function_max_retries'); From 9801cd0c7891285a89d69816b6d8a9ca8c4a60d4 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 08:26:43 +0000 Subject: [PATCH 040/185] Backport #118674 to 26.8: Keep a `Backup` database loadable after `ALTER DATABASE ... MODIFY COMMENT` --- src/Databases/DatabaseBackup.cpp | 71 +++++++++-- src/Parsers/FunctionSecretArgumentsFinder.cpp | 14 +++ .../configs/backups.xml | 9 ++ .../test_backup_source_grants/test.py | 113 ++++++++++++++---- .../integration/test_database_backup/test.py | 102 ++++++++++++++++ ...base_backup_database_disk_engine.reference | 2 +- ...3_explicit_url_named_secret_mask.reference | 1 + ...4510_s3_explicit_url_named_secret_mask.sql | 7 ++ ...8_database_backup_modify_comment.reference | 4 + .../05148_database_backup_modify_comment.sh | 48 ++++++++ ...kup_quoted_locator_parallel_with.reference | 5 + ...ase_backup_quoted_locator_parallel_with.sh | 80 +++++++++++++ 12 files changed, 419 insertions(+), 37 deletions(-) create mode 100644 tests/queries/0_stateless/05148_database_backup_modify_comment.reference create mode 100755 tests/queries/0_stateless/05148_database_backup_modify_comment.sh create mode 100644 tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.reference create mode 100755 tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.sh diff --git a/src/Databases/DatabaseBackup.cpp b/src/Databases/DatabaseBackup.cpp index 0fa7ae4fe9c7..c46a7a663b1f 100644 --- a/src/Databases/DatabaseBackup.cpp +++ b/src/Databases/DatabaseBackup.cpp @@ -43,6 +43,7 @@ #include #include #include +#include namespace CurrentMetrics @@ -71,6 +72,7 @@ namespace ErrorCodes extern const int LOGICAL_ERROR; extern const int INCORRECT_FILE_NAME; extern const int NUMBER_OF_ARGUMENTS_DOESNT_MATCH; + extern const int BAD_ARGUMENTS; extern const int CANNOT_GET_CREATE_TABLE_QUERY; } @@ -464,8 +466,11 @@ ASTPtr DatabaseBackup::getCreateDatabaseQueryImpl() const { const auto & settings = getContext()->getSettingsRef(); + /// The locator is emitted as the function it is, not as a string literal holding its text: this + /// definition is what `ALTER DATABASE ... MODIFY COMMENT` writes back into the metadata file, and + /// the load path parses the second argument with `BackupInfo::fromAST`, which takes a function. const String query = fmt::format("CREATE DATABASE {} ENGINE = Backup({}, {})", - backQuoteIfNeed(database_name), quoteString(config.database_name), quoteString(config.backup_info.toString())); + backQuoteIfNeed(database_name), quoteString(config.database_name), config.backup_info.toString()); ParserCreateQuery parser; ASTPtr ast = parseQuery(parser, @@ -493,7 +498,7 @@ std::vector> DatabaseBackup::getTablesForBackup(co namespace { -DatabaseBackup::Configuration parseArguments(ASTs engine_args, ContextPtr) +DatabaseBackup::Configuration parseArguments(ASTs engine_args, ContextPtr, bool allow_locator_in_string_literal) { if (engine_args.size() != 2) throw Exception::createRuntime(ErrorCodes::NUMBER_OF_ARGUMENTS_DOESNT_MATCH, @@ -502,6 +507,33 @@ DatabaseBackup::Configuration parseArguments(ASTs engine_args, ContextPtr) DatabaseBackup::Configuration result; result.database_name = checkAndGetLiteralArgument(engine_args[0], "database_name"); + + /** A locator held in a string literal (`Backup('db', 'File(\'backup.zip\')')`) is the form that + * metadata rewritten by an older server carries, so it has to keep loading - a server that cannot + * parse its own metadata does not start at all. + * + * Only there: in a statement a user writes the locator must be the function it is, because that is + * the form `FunctionSecretArgumentsFinder` knows how to redact, and a quoted one would carry its + * credentials verbatim into `query_log`, `SHOW PROCESSLIST` and the distributed DDL payload. + */ + if (allow_locator_in_string_literal) + { + if (const auto * locator = engine_args[1]->as(); locator && locator->value.getType() == Field::Types::String) + { + result.backup_info = BackupInfo::fromString(locator->value.safeGet()); + return result; + } + } + + /// `BackupInfo::fromAST` puts the offending argument into its message, and that message reaches the + /// error log and the `exception` column of `query_log`. A locator held in a string literal is exactly + /// the text that can carry credentials, so refuse it here without echoing it. + if (!engine_args[1]->as()) + throw Exception(ErrorCodes::BAD_ARGUMENTS, + "Expected function as the backup destination of a `Backup` database. It must be spelled as the " + "function it is, such as `File('backup')` or `S3(...)`. The text given is not shown, because a " + "destination held in a string literal cannot be redacted and may carry credentials"); + result.backup_info = BackupInfo::fromAST(*engine_args[1]); return result; @@ -511,11 +543,20 @@ DatabaseBackup::Configuration parseArguments(ASTs engine_args, ContextPtr) void DatabaseBackup::parseAndAuthorizeLocator(const ASTs & engine_args, ContextPtr query_context) { - /// A locator we cannot parse opens nothing: creation rejects it, so there is nothing to authorize. - if (engine_args.size() == 2 && !engine_args[1]->as()) - return; - - auto config = parseArguments(engine_args, query_context); + /** A locator that is not a function is refused right here rather than left to creation time. + * + * This preflight is the last point that still runs as the real user, and the creation it would + * otherwise rely on does not always run: `RESTORE DATABASE` issues `CREATE DATABASE IF NOT EXISTS`, + * which returns from `InterpreterCreateQuery::createDatabase` before `DatabaseFactory::get` when the + * target database already exists. With `allow_different_database_def = 1` the definition mismatch is + * waived as well, so a manifest an older server left holding a quoted locator would pass through the + * whole restore with its embedded source never authorized, while the function form of the same + * manifest is checked. + * + * Only `parseArguments` may formulate the refusal: the offending text can carry credentials and must + * not be echoed. + */ + auto config = parseArguments(engine_args, query_context, /*allow_locator_in_string_literal=*/ false); BackupFactory::instance().checkSourceAccess(config.backup_info, query_context, IBackup::OpenMode::READ); } @@ -531,13 +572,23 @@ void registerDatabaseBackup(DatabaseFactory & factory) if (engine->arguments) engine_args = engine->arguments->children; - auto config = parseArguments(engine_args, args.context); - /// Authorize only a newly introduced definition: one read back from this server's metadata was /// already validated, and a context with no user cannot be checked per user. + /// + /// Metadata is read back on three paths: the short `ATTACH DATABASE db`, a load under `force_restore_data`, + /// and the replay of the stored full `ATTACH DATABASE db ENGINE = Backup(...)` statement at server start. + /// The last one runs in plain `ATTACH` mode, so neither of the first two conditions covers it - and it is + /// exactly the path that has to load metadata an older server rewrote. + /// + /// The loader flag, not `internal`, is the discriminator: wrappers such as `PARALLEL WITH` run user + /// statements as internal ones, and a user's `ATTACH DATABASE ... ENGINE = Backup(...)` must neither + /// skip the source authorization nor get its locator accepted in the form the secret masker cannot redact. const bool has_real_user = args.context->getAccess()->getUserID().has_value(); + const bool is_internal_metadata_replay = args.is_metadata_replay && args.mode >= LoadingStrictnessLevel::ATTACH; const bool from_existing_metadata - = isLoadingFromExistingMetadata(args.mode) || args.create_query.attach_short_syntax; + = isLoadingFromExistingMetadata(args.mode) || args.create_query.attach_short_syntax || is_internal_metadata_replay; + + auto config = parseArguments(engine_args, args.context, /*allow_locator_in_string_literal=*/ from_existing_metadata); if (has_real_user && !from_existing_metadata) BackupFactory::instance().checkSourceAccess(config.backup_info, args.context, IBackup::OpenMode::READ); diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 0275cd816508..8a6138d23552 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -1181,6 +1181,20 @@ void FunctionSecretArgumentsFinder::findBackupDatabaseSecretArguments() auto storage_arg = function->arguments->at(1); auto storage_function = storage_arg->getFunction(); + /// A locator that is not a function - a string literal holding its text, or an expression - carries + /// the destination as text this finder cannot parse, and that text can hold an access key, a secret + /// access key or a presigned URL. The engine accepts such a locator only while replaying its own + /// metadata, but a statement carrying it is formatted before the engine rejects it: by `PARALLEL WITH`, + /// by the distributed DDL queue, and by `query_log`. Hide it whole rather than let it through verbatim. + if (!storage_function) + { + result.start = 1; + result.count = 1; + result.replacement = "'[HIDDEN]'"; + result.quote_replacement = false; + return; + } + /// The nested S3 destination is not recognized as an S3 engine when the formatter recurses into it, /// so its secrets must be masked here. Handle both forms: /// Backup('', S3('url', 'access_key_id', 'secret_access_key' [, ...])) diff --git a/tests/integration/test_backup_source_grants/configs/backups.xml b/tests/integration/test_backup_source_grants/configs/backups.xml index 2d99cb1d1ef9..acf77f033a2f 100644 --- a/tests/integration/test_backup_source_grants/configs/backups.xml +++ b/tests/integration/test_backup_source_grants/configs/backups.xml @@ -1,5 +1,14 @@ + + + + local + /var/lib/clickhouse/disks/backup_disk/ + + + /var/lib/clickhouse/backups + backup_disk diff --git a/tests/integration/test_backup_source_grants/test.py b/tests/integration/test_backup_source_grants/test.py index 42b44da713dc..ff8efd1037bd 100644 --- a/tests/integration/test_backup_source_grants/test.py +++ b/tests/integration/test_backup_source_grants/test.py @@ -256,9 +256,8 @@ def test_restore_on_cluster_authorizes_an_embedded_definition_over_an_existing_d # the check is skipped here and the restricted user reaches the embedded locator unchecked. # # The outer locator is File and IS granted, so a denial naming READ ON S3 can only come from the - # embedded S3 locator. `BACKUP DATABASE` serializes the locator as a string that - # `BackupInfo::fromAST` rejects before any check, so the function form is written into the - # manifest directly - the attacker-controlled-manifest shape this authorization exists for. + # embedded S3 locator, which is why the manifest is rewritten to hold one - the + # attacker-controlled-manifest shape this authorization exists for. # Every locator here is credential-free (File locally, a 1-argument S3 URL in the manifest): the # later definition-mismatch error logs both definitions, and a masked credential in that line # trips the tests-only `throw_on_match` masking rule and aborts the server. @@ -289,7 +288,7 @@ def test_restore_on_cluster_authorizes_an_embedded_definition_over_an_existing_d assert "READ ON S3" in error, error # With the grant, authorization passes: the restore proceeds past CHECKING_ACCESS_RIGHTS and - # fails later on the pre-existing string-vs-function definition mismatch instead. + # fails later on the mismatch between the pre-existing definition and the rewritten one instead. node.query(f"GRANT READ ON S3 TO {USER}") error = node.query_and_get_error( "RESTORE DATABASE dbembedded ON CLUSTER one_shard FROM File('outer10')", user=USER @@ -300,41 +299,103 @@ def test_restore_on_cluster_authorizes_an_embedded_definition_over_an_existing_d node.query("DROP DATABASE dbembedded SYNC") -def test_restore_on_cluster_of_a_real_backup_engine_manifest_is_unchanged(started_cluster): - # The manifest is NOT crafted here, which is the point: `BACKUP DATABASE` serializes the inner - # locator as a string, and authorizing it would parse it and reject it with BAD_ARGUMENTS before - # any access decision. Only this shape can catch that, so it is a separate case from the crafted - # one - which asserts the security property but cannot see this class. +def test_restore_on_cluster_refuses_a_quoted_embedded_locator_over_an_existing_database( + started_cluster, +): + # The pre-fix carrier: metadata an older server rewrote holds the locator as a string literal, and + # such a manifest can end up inside a backup. On this path the creation-time refusal never runs - + # the target database exists, so `CREATE DATABASE IF NOT EXISTS` returns before `DatabaseFactory`, + # and `allow_different_database_def = 1` waives the definition mismatch that would stop it next. + # The preflight is therefore the only place that can see the quoted locator, and it must refuse it + # rather than wave it through unauthorized: the function form of the very same manifest is denied + # for a missing `READ ON S3` in the case above. + # + # The refusal must not print the locator, which is why the assertion below looks for the host that + # a quoted locator could carry credentials next to. + node.query("DROP DATABASE IF EXISTS dbquoted SYNC") + node.query("BACKUP DATABASE d67785 TO File('inner13') FORMAT Null") + node.query("CREATE DATABASE dbquoted ENGINE = Backup('d67785', File('inner13'))") + node.query("BACKUP DATABASE dbquoted TO File('outer13') FORMAT Null") + manifest = ( + "CREATE DATABASE dbquoted ENGINE = Backup('d67785', " + "'S3(\\'http://minio1:9001/root/data/denied/b13\\')')" + ) + node.exec_in_container( + [ + "bash", + "-c", + "cat > /var/lib/clickhouse/backups/outer13/metadata/dbquoted.sql <<'MANIFEST'\n" + f"{manifest}\nMANIFEST", + ], + user="root", + ) + + assert node.query("SELECT count() FROM system.databases WHERE name = 'dbquoted'") == "1\n" + node.query(f"GRANT READ ON FILE TO {USER}") + + error = node.query_and_get_error( + "RESTORE DATABASE dbquoted ON CLUSTER one_shard FROM File('outer13') " + "SETTINGS allow_different_database_def = 1", + user=USER, + ) + assert "BAD_ARGUMENTS" in error, error + assert "Expected function as the backup destination" in error, error + assert "minio1" not in error, error + + node.query("DROP DATABASE dbquoted SYNC") + + +def test_restore_on_cluster_of_a_real_backup_engine_manifest_authorizes_the_inner_locator( + started_cluster, +): + # The manifest is NOT crafted here, which is the point: `BACKUP DATABASE` writes the inner locator + # as the function it is, so a real manifest carries a locator that authorization can decode, and + # the embedded one is checked exactly as a rewritten one is. Only this shape can see that, so it is + # a separate case from the rewritten one - which asserts the security property against a locator + # the definition never held. The outer locator is a Disk backup, which needs READ ON DISK, so a + # denial naming READ ON FILE can only come from the inner one. node.query("DROP DATABASE IF EXISTS dbreal SYNC") node.query("BACKUP DATABASE d67785 TO File('inner11') FORMAT Null") node.query("CREATE DATABASE dbreal ENGINE = Backup('d67785', File('inner11'))") - node.query("BACKUP DATABASE dbreal TO File('outer11') FORMAT Null") + node.query("BACKUP DATABASE dbreal TO Disk('backup_disk', 'outer11') FORMAT Null") manifest = node.exec_in_container( - ["bash", "-c", "cat /var/lib/clickhouse/backups/outer11/metadata/dbreal.sql"], + [ + "bash", + "-c", + "cat /var/lib/clickhouse/disks/backup_disk/outer11/metadata/dbreal.sql", + ], user="root", ) - # The locator really is the string form: an ASTLiteral, not an ASTFunction. - assert "Backup('d67785', 'File(\\'inner11\\')')" in manifest, manifest + # The locator really is the function form: an ASTFunction, not an ASTLiteral holding its text. + assert "Backup('d67785', File('inner11'))" in manifest, manifest - # Only the outer locator's own grant; the string-form inner one must not be authorized at all. - node.query(f"GRANT READ ON FILE TO {USER}") + # Only the outer locator's own grant, so the embedded File locator is the one that is missing. + node.query(f"GRANT READ ON DISK TO {USER}") + + error = node.query_and_get_error( + "RESTORE DATABASE dbreal ON CLUSTER one_shard FROM Disk('backup_disk', 'outer11')", + user=USER, + ) + assert "ACCESS_DENIED" in error, error + assert "READ ON FILE" in error, error - # Target exists, so nothing is created and the restore fully succeeds. Authorizing the string - # form turns this into `Code: 36` out of CHECKING_ACCESS_RIGHTS. + # Target exists and its definition matches the manifest, so nothing is created and the restore + # fully succeeds. + node.query(f"GRANT READ ON FILE TO {USER}") node.query( - "RESTORE DATABASE dbreal ON CLUSTER one_shard FROM File('outer11') FORMAT Null", user=USER + "RESTORE DATABASE dbreal ON CLUSTER one_shard FROM Disk('backup_disk', 'outer11') FORMAT Null", + user=USER, ) - # Target absent, so creation runs and rejects the string form - as it does without this feature. - # `While creating database` is what pins the failure to the creation stage rather than the - # access-check one, which is where the same code would report it. + # Target absent, so creation runs: the locator parses, and the created database reads the tables of + # the inner backup. node.query("DROP DATABASE dbreal SYNC") - error = node.query_and_get_error( - "RESTORE DATABASE dbreal ON CLUSTER one_shard FROM File('outer11')", user=USER + node.query( + "RESTORE DATABASE dbreal ON CLUSTER one_shard FROM Disk('backup_disk', 'outer11') FORMAT Null", + user=USER, ) - assert "ACCESS_DENIED" not in error, error - assert "BAD_ARGUMENTS" in error, error - assert "While creating database" in error, error + assert node.query("SELECT x FROM dbreal.secrets") == "42\n" + node.query("DROP DATABASE dbreal SYNC") def test_explicit_base_backup_locator_is_authorized_on_the_initiator(started_cluster): diff --git a/tests/integration/test_database_backup/test.py b/tests/integration/test_database_backup/test.py index 8316d798e14f..40aa406a050e 100644 --- a/tests/integration/test_database_backup/test.py +++ b/tests/integration/test_database_backup/test.py @@ -12,6 +12,17 @@ with_minio=True, ) +# `test_database_backup_metadata_with_quoted_locator_loads_on_restart` rewrites a database metadata file +# in place, and such a file only exists when the metadata lives on the local disk, so that test runs on an +# instance which keeps the local database disk. +instance_local_metadata = cluster.add_instance( + "instance_local_metadata", + main_configs=["configs/backups.xml"], + stay_alive=True, + with_minio=True, + with_remote_database_disk=False, +) + @pytest.fixture(scope="module", autouse=True) def start_cluster(): @@ -271,3 +282,94 @@ def test_database_backup_unavailable_but_server_starts(backup_destination): instance.query("DROP DATABASE IF EXISTS test_database_backup SYNC") instance.query("DROP DATABASE IF EXISTS test_database SYNC") cleanup_backup_files(instance) + + +def test_database_backup_metadata_with_quoted_locator_loads_on_restart(): + # Regression test for https://github.com/ClickHouse/ClickHouse/issues/118349 + # An older server regenerated the definition of a `Backup` database with the locator quoted into a + # string literal, and `ALTER DATABASE ... MODIFY COMMENT` wrote that back into `metadata/.sql`. + # The next start replays the stored full `ATTACH DATABASE ... ENGINE = Backup(...)` statement, which + # is neither the short `ATTACH` nor a force-restore load, so it has to accept that form on its own. + # + # The metadata file is rewritten in place here, so this runs on the instance whose metadata is a file + # on the local disk rather than an object on a remote database disk. + instance = instance_local_metadata + + cleanup_backup_files(instance) + + instance.query( + """ + DROP DATABASE IF EXISTS test_database SYNC; + DROP DATABASE IF EXISTS test_database_backup SYNC; + + CREATE DATABASE test_database; + + CREATE TABLE test_database.test_table (id UInt64, value String) ENGINE=MergeTree ORDER BY id; + INSERT INTO test_database.test_table VALUES (0, 'test_database.test_table'); + + BACKUP DATABASE test_database TO File('test_database_backup_file'); + CREATE DATABASE test_database_backup ENGINE = Backup('test_database', File('test_database_backup_file')); + """ + ) + assert ( + instance.query("SELECT id, value FROM test_database_backup.test_table") + == "0\ttest_database.test_table\n" + ) + + # The metadata file exactly as a pre-fix server left it after a comment change. + instance.stop_clickhouse() + metadata = ( + "ATTACH DATABASE test_database_backup\n" + "ENGINE = Backup('test_database', 'File(\\'test_database_backup_file\\')')\n" + "COMMENT 'written by an older server'\n" + ) + instance.exec_in_container( + [ + "bash", + "-c", + "cat > /var/lib/clickhouse/metadata/test_database_backup.sql <<'SQL'\n" + + metadata + + "SQL\n", + ], + user="root", + ) + assert "'File(\\'test_database_backup_file\\')'" in instance.exec_in_container( + ["cat", "/var/lib/clickhouse/metadata/test_database_backup.sql"] + ) + instance.start_clickhouse() + + # The server started and the database loaded with its tables and comment. + assert ( + instance.query("SELECT id, value FROM test_database_backup.test_table") + == "0\ttest_database.test_table\n" + ) + assert ( + instance.query( + "SELECT comment FROM system.databases WHERE name = 'test_database_backup'" + ) + == "written by an older server\n" + ) + # The definition is regenerated with the locator as the function it is. + assert ( + "ENGINE = Backup('test_database', File('test_database_backup_file'))" + in instance.query("SHOW CREATE DATABASE test_database_backup FORMAT TSVRaw") + ) + + # A comment change on this server writes the function form, and that survives a restart too. + instance.query( + "ALTER DATABASE test_database_backup MODIFY COMMENT 'written by this server'" + ) + instance.restart_clickhouse() + assert ( + instance.query("SELECT id, value FROM test_database_backup.test_table") + == "0\ttest_database.test_table\n" + ) + assert ( + instance.query( + "SELECT comment FROM system.databases WHERE name = 'test_database_backup'" + ) + == "written by this server\n" + ) + + instance.query("DROP DATABASE test_database_backup SYNC") + instance.query("DROP DATABASE test_database SYNC") diff --git a/tests/queries/0_stateless/03279_database_backup_database_disk_engine.reference b/tests/queries/0_stateless/03279_database_backup_database_disk_engine.reference index 4f8c3ae7dc13..80083db7617a 100644 --- a/tests/queries/0_stateless/03279_database_backup_database_disk_engine.reference +++ b/tests/queries/0_stateless/03279_database_backup_database_disk_engine.reference @@ -20,7 +20,7 @@ 8 1500 9 1500 -- -CREATE DATABASE default_inner_backup_database\nENGINE = Backup(\'default_inner\', \'Disk(\\\'backups\\\', \\\'default_inner\\\')\') +CREATE DATABASE default_inner_backup_database\nENGINE = Backup(\'default_inner\', Disk(\'backups\', \'default_inner\')) test_table_1 15000 test_table_2 15000 -- diff --git a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference index eb36b31ef3f9..030f5c5e6031 100644 --- a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference +++ b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference @@ -207,6 +207,7 @@ CREATE DATABASE db_04510_mixed ENGINE = Backup(\'\', S3(\'url_dbmixed\', access_ CREATE DATABASE db_04510_ncurl ENGINE = Backup(\'\', S3(nc_dburl_missing, url = \'[HIDDEN]\')) CREATE DATABASE db_04510_hdr ENGINE = Backup(\'\', S3(\'url_dbhdr\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\')) CREATE DATABASE db_04510_expr ENGINE = Backup(\'\', S3(\'url_dbexpr\', \'ak\', \'[HIDDEN]\', \'[HIDDEN]\')) +CREATE DATABASE db_04510_quoted ENGINE = Backup(\'\', \'[HIDDEN]\') CREATE DATABASE db_04510_s3pos ENGINE = S3(\'url_dbs3pos\', \'ak\', \'[HIDDEN]\', \'[HIDDEN]\') DROP DATABASE IF EXISTS default_1 CREATE DATABASE default_1 ENGINE = S3(\'url_dbenv\', \'ak\', \'[HIDDEN]\', use_environment_credentials = 1) diff --git a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql index 9cded2e5e118..569086cf03b2 100644 --- a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql +++ b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql @@ -280,6 +280,13 @@ CREATE DATABASE db_04510_hdr ENGINE = Backup('', S3('url_dbhdr', 'ak', 'SEKRIT_S CREATE DATABASE db_04510_expr ENGINE = Backup('', S3('url_dbexpr', 'ak', 'SEKRIT_SAK', extra_credentials(concat('extern', 'al_id') = 'SEKRIT_EXPR'))); -- { serverError BAD_ARGUMENTS } +-- A locator held in a string literal is text this finder cannot parse, and that text can carry an +-- access key or a presigned URL. The `Backup` engine accepts that form only while replaying its own +-- metadata, but a statement carrying it is formatted - by `PARALLEL WITH`, by the distributed DDL +-- queue, by `query_log` - before the engine rejects it, so it must be masked whole. +CREATE DATABASE db_04510_quoted ENGINE = Backup('', + 'S3(\'https://user:SEKRIT_PW@localhost:11111/x?X-Amz-Signature=SEKRIT_SIG\', \'ak\', \'SEKRIT_QUOTED\')'); -- { serverError BAD_ARGUMENTS } + -- The S3 database engine accepts no positional beyond secret_access_key; an extra positional must -- be masked in the logged query text. CREATE DATABASE db_04510_s3pos ENGINE = S3('url_dbs3pos', 'ak', 'SEKRIT_SAK', diff --git a/tests/queries/0_stateless/05148_database_backup_modify_comment.reference b/tests/queries/0_stateless/05148_database_backup_modify_comment.reference new file mode 100644 index 000000000000..b3863579a954 --- /dev/null +++ b/tests/queries/0_stateless/05148_database_backup_modify_comment.reference @@ -0,0 +1,4 @@ +10 +10 +a comment +refused diff --git a/tests/queries/0_stateless/05148_database_backup_modify_comment.sh b/tests/queries/0_stateless/05148_database_backup_modify_comment.sh new file mode 100755 index 000000000000..4af56ac37b84 --- /dev/null +++ b/tests/queries/0_stateless/05148_database_backup_modify_comment.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash +# Tags: no-encrypted-storage + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +BACKUP_DATABASE_NAME=${CLICKHOUSE_TEST_UNIQUE_NAME}_backup +ATTACHED_DATABASE_NAME=${CLICKHOUSE_TEST_UNIQUE_NAME}_attached + +$CLICKHOUSE_CLIENT -q """ +DROP DATABASE IF EXISTS $BACKUP_DATABASE_NAME; +CREATE DATABASE $BACKUP_DATABASE_NAME; + +CREATE TABLE $BACKUP_DATABASE_NAME.test_table (id UInt64) ENGINE = MergeTree ORDER BY id; +INSERT INTO $BACKUP_DATABASE_NAME.test_table SELECT number FROM numbers(10); + +BACKUP DATABASE $BACKUP_DATABASE_NAME TO Disk('backups', '$BACKUP_DATABASE_NAME') FORMAT Null; + +DROP DATABASE IF EXISTS $ATTACHED_DATABASE_NAME; +CREATE DATABASE $ATTACHED_DATABASE_NAME ENGINE = Backup('$BACKUP_DATABASE_NAME', Disk('backups', '$BACKUP_DATABASE_NAME')); + +SELECT count() FROM $ATTACHED_DATABASE_NAME.test_table; +""" + +# A comment change rewrites the metadata file of the database, so the locator has to survive the +# round trip: loading it back is what the server does at every start. +$CLICKHOUSE_CLIENT -q """ +ALTER DATABASE $ATTACHED_DATABASE_NAME MODIFY COMMENT 'a comment'; +DETACH DATABASE $ATTACHED_DATABASE_NAME; +ATTACH DATABASE $ATTACHED_DATABASE_NAME; + +SELECT count() FROM $ATTACHED_DATABASE_NAME.test_table; +SELECT comment FROM system.databases WHERE name = '$ATTACHED_DATABASE_NAME'; +""" + +# A locator serialized as a string literal is accepted only while loading metadata an older server +# rewrote: in a statement a user writes it must be the function it is, because that is the form the +# secret masker redacts. +$CLICKHOUSE_CLIENT -q """ +DROP DATABASE $ATTACHED_DATABASE_NAME; +CREATE DATABASE $ATTACHED_DATABASE_NAME ENGINE = Backup('$BACKUP_DATABASE_NAME', 'Disk(\\'backups\\', \\'$BACKUP_DATABASE_NAME\\')'); +""" 2>&1 | grep -q -F 'Expected function' && echo 'refused' + +$CLICKHOUSE_CLIENT -q """ +DROP DATABASE IF EXISTS $ATTACHED_DATABASE_NAME; +DROP DATABASE $BACKUP_DATABASE_NAME; +""" diff --git a/tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.reference b/tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.reference new file mode 100644 index 000000000000..8b916e9e9712 --- /dev/null +++ b/tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.reference @@ -0,0 +1,5 @@ +refused +refused +refused +0 1 +10 diff --git a/tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.sh b/tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.sh new file mode 100755 index 000000000000..454cce40ebff --- /dev/null +++ b/tests/queries/0_stateless/05218_database_backup_quoted_locator_parallel_with.sh @@ -0,0 +1,80 @@ +#!/usr/bin/env bash +# Tags: no-encrypted-storage + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +BACKUP_DATABASE_NAME=${CLICKHOUSE_TEST_UNIQUE_NAME}_backup +ATTACHED_DATABASE_NAME=${CLICKHOUSE_TEST_UNIQUE_NAME}_attached +OTHER_DATABASE_NAME=${CLICKHOUSE_TEST_UNIQUE_NAME}_other + +$CLICKHOUSE_CLIENT -q """ +DROP DATABASE IF EXISTS $BACKUP_DATABASE_NAME; +DROP DATABASE IF EXISTS $ATTACHED_DATABASE_NAME; +DROP DATABASE IF EXISTS $OTHER_DATABASE_NAME; +CREATE DATABASE $BACKUP_DATABASE_NAME; + +CREATE TABLE $BACKUP_DATABASE_NAME.test_table (id UInt64) ENGINE = MergeTree ORDER BY id; +INSERT INTO $BACKUP_DATABASE_NAME.test_table SELECT number FROM numbers(10); + +BACKUP DATABASE $BACKUP_DATABASE_NAME TO Disk('backups', '$BACKUP_DATABASE_NAME') FORMAT Null; +""" + +# `PARALLEL WITH` runs its statements as internal queries, and the full `ATTACH DATABASE ... ENGINE = ...` +# form runs in `ATTACH` mode - the same pair the server's own replay of stored metadata shows. The quoted +# locator is accepted only on that replay, never in a statement a user writes, however it is wrapped. +$CLICKHOUSE_CLIENT -q """ +CREATE DATABASE $OTHER_DATABASE_NAME +PARALLEL WITH +ATTACH DATABASE $ATTACHED_DATABASE_NAME ENGINE = Backup('$BACKUP_DATABASE_NAME', 'Disk(\\'backups\\', \\'$BACKUP_DATABASE_NAME\\')'); +""" 2>&1 | grep -q -F 'Expected function' && echo 'refused' + +# Each block below must start from the same state: whether the sibling `CREATE DATABASE` of a +# `PARALLEL WITH` whose other statement threw is committed or rolled back depends on how the +# statements were scheduled, and with `max_threads = 1` it is committed. Drop it in between, so the +# refusal under test is what the block observes, not a leftover "database already exists". +$CLICKHOUSE_CLIENT -q "DROP DATABASE IF EXISTS $OTHER_DATABASE_NAME" + +$CLICKHOUSE_CLIENT -q """ +CREATE DATABASE $OTHER_DATABASE_NAME +PARALLEL WITH +CREATE DATABASE $ATTACHED_DATABASE_NAME ENGINE = Backup('$BACKUP_DATABASE_NAME', 'Disk(\\'backups\\', \\'$BACKUP_DATABASE_NAME\\')'); +""" 2>&1 | grep -q -F 'Expected function' && echo 'refused' + +$CLICKHOUSE_CLIENT -q "DROP DATABASE IF EXISTS $OTHER_DATABASE_NAME" + +# A quoted locator can carry credentials, and `PARALLEL WITH` formats its statements before the engine +# refuses them, so the formatted text must hide it: neither the logged query text (the same text an +# `ON CLUSTER` statement puts into the distributed DDL payload) nor the refusal message may carry it. +$CLICKHOUSE_CLIENT -q """ +CREATE DATABASE $OTHER_DATABASE_NAME +PARALLEL WITH +ATTACH DATABASE $ATTACHED_DATABASE_NAME ENGINE = Backup('$BACKUP_DATABASE_NAME', 'S3(\\'http://localhost:11111/05218\\', \\'ak\\', \\'SEKRIT_05218\\')'); +""" 2>&1 | grep -q -F 'Expected function' && echo 'refused' + +$CLICKHOUSE_CLIENT -q "DROP DATABASE IF EXISTS $OTHER_DATABASE_NAME" + +$CLICKHOUSE_CLIENT -q "SYSTEM FLUSH LOGS query_log" +$CLICKHOUSE_CLIENT -q """ +SELECT countIf(query LIKE '%SEKRIT_05218%' OR exception LIKE '%SEKRIT_05218%'), countIf(query LIKE '%Backup(%[HIDDEN]%') > 0 +FROM system.query_log +WHERE current_database = currentDatabase() + AND type != 'QueryStart' + AND event_date >= yesterday() AND event_time > now() - INTERVAL 5 MINUTE; +""" + +# The function form goes through the same wrapper. +$CLICKHOUSE_CLIENT -q """ +CREATE DATABASE IF NOT EXISTS $OTHER_DATABASE_NAME +PARALLEL WITH +ATTACH DATABASE $ATTACHED_DATABASE_NAME ENGINE = Backup('$BACKUP_DATABASE_NAME', Disk('backups', '$BACKUP_DATABASE_NAME')); + +SELECT count() FROM $ATTACHED_DATABASE_NAME.test_table; +""" + +$CLICKHOUSE_CLIENT -q """ +DROP DATABASE IF EXISTS $ATTACHED_DATABASE_NAME; +DROP DATABASE IF EXISTS $OTHER_DATABASE_NAME; +DROP DATABASE $BACKUP_DATABASE_NAME; +""" From 52129a0751d4daf45fe114a48aa4b2fd1d8b986e Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 08:37:19 +0000 Subject: [PATCH 041/185] Backport #115693 to 26.8: Do not assert scan-progress values decided by an async race in test_early_return_limit --- tests/integration/test_storage_delta/test.py | 16 ++++------------ 1 file changed, 4 insertions(+), 12 deletions(-) diff --git a/tests/integration/test_storage_delta/test.py b/tests/integration/test_storage_delta/test.py index cc8f7034012f..efc207fa0d18 100644 --- a/tests/integration/test_storage_delta/test.py +++ b/tests/integration/test_storage_delta/test.py @@ -5358,24 +5358,16 @@ def test_early_return_limit(started_cluster, use_delta_kernel): assert first_check_hits > 0 or queue_check_hits > 0 - assert 1 == int(instance.query( - f"SELECT count() FROM system.text_log WHERE query_id = '{query_id}' AND message LIKE '%List batch size is 1/1, shutdown: true%'" - )) - # Early return should scan significantly fewer files # With s3_list_object_keys_size=1, queue pauses frequently forcing shutdown checks # Should stop very early after consuming just a few files assert scanned_files < full_scan_files, \ f"Early return should scan fewer files: {scanned_files} >= {full_scan_files}" - # 3 because: - # we have async reader creation with 2 existing readers at a moment of time, - # each calls next() and consumes 2 files from the scan. - # It takes 1 file for the query to stop because of LIMIT 1. - # But because scan is also asynchronous and continues once batch limit is not reached, - # we get +1 scanned file. - assert scanned_files == 3, \ - f"Early return should scan 3 files with LIMIT 1, but scanned {scanned_files}" + # At most 3: two async readers each consume one file, plus one the scan produces before + # it observes shutdown. Fewer is legal when shutdown lands earlier. + assert scanned_files <= 3, \ + f"Early return should scan at most 3 files with LIMIT 1, but scanned {scanned_files}" def test_struct_dotted_field_names(started_cluster): From cdbaf82420964d5c329cc964b86562c93f4e8214 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 08:56:30 +0000 Subject: [PATCH 042/185] Backport #119686 to 26.8: Fix flaky `01780_column_sparse`: order the `arrayFilter` result before limiting it --- tests/queries/0_stateless/01780_column_sparse.reference | 2 +- tests/queries/0_stateless/01780_column_sparse.sql | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/tests/queries/0_stateless/01780_column_sparse.reference b/tests/queries/0_stateless/01780_column_sparse.reference index 3393361a19fb..7452be01648e 100644 --- a/tests/queries/0_stateless/01780_column_sparse.reference +++ b/tests/queries/0_stateless/01780_column_sparse.reference @@ -132,7 +132,7 @@ SELECT id % 7, sum(u) FROM t_sparse GROUP BY id % 7 ORDER BY id % 7; 4 190 5 330 6 270 -SELECT arrayFilter(x -> x % 2 = 1, arr2) FROM t_sparse WHERE arr2 != [] LIMIT 5; +SELECT arrayFilter(x -> x % 2 = 1, arr2) FROM t_sparse WHERE arr2 != [] ORDER BY id LIMIT 5; [1] [1,3] [1,3,5] diff --git a/tests/queries/0_stateless/01780_column_sparse.sql b/tests/queries/0_stateless/01780_column_sparse.sql index 8e3c4372d05a..c5b98b1fd422 100644 --- a/tests/queries/0_stateless/01780_column_sparse.sql +++ b/tests/queries/0_stateless/01780_column_sparse.sql @@ -27,7 +27,7 @@ SELECT * FROM t_sparse WHERE arr2 != [] ORDER BY id; SELECT sum(u) FROM t_sparse; SELECT id % 7, sum(u) FROM t_sparse GROUP BY id % 7 ORDER BY id % 7; -SELECT arrayFilter(x -> x % 2 = 1, arr2) FROM t_sparse WHERE arr2 != [] LIMIT 5; +SELECT arrayFilter(x -> x % 2 = 1, arr2) FROM t_sparse WHERE arr2 != [] ORDER BY id LIMIT 5; CREATE TABLE t_sparse_1 (id UInt64, v Int64) ENGINE = MergeTree ORDER BY tuple() From fdec066dc3af8d90dfa2f28e934935ff36300818 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 09:04:27 +0000 Subject: [PATCH 043/185] Backport #119743 to 26.8: Tolerate Iceberg manifest header schemas that conflict with metadata.json --- src/Core/Settings.cpp | 5 + src/Core/SettingsChangesHistory.cpp | 4 + .../Iceberg/ManifestFileIterator.cpp | 12 +- .../DataLakes/Iceberg/SchemaProcessor.cpp | 132 ++++++++++++---- .../DataLakes/Iceberg/SchemaProcessor.h | 26 +++- .../tests/gtest_iceberg_schema_processor.cpp | 144 ++++++++++++++++++ ...tion_conflicting_manifest_schema.reference | 7 + ..._compaction_conflicting_manifest_schema.sh | 81 ++++++++++ 8 files changed, 376 insertions(+), 35 deletions(-) create mode 100644 tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.reference create mode 100755 tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.sh diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 304d1177a179..1a73f7833bd1 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -8101,6 +8101,11 @@ Default partition strategy for file like engines. Applied only to `CREATE` queri )", 0) \ DECLARE(Bool, use_iceberg_partition_pruning, true, R"( Use Iceberg partition pruning for Iceberg tables +)", 0) \ + DECLARE(Bool, iceberg_tolerate_conflicting_manifest_schemas, true, R"( +If enabled and the `schema` key of an Iceberg manifest file header carries a schema that differs from the schema already registered for the same schema-id from metadata.json, the metadata.json schema is used and the manifest header copy is ignored with a warning. If disabled, such a conflict fails the query with an ICEBERG_SPECIFICATION_VIOLATION error. + +The manifest header schema is only a copy of the table schema at the time the manifest was written, and some writers (e.g. AWS S3 Tables maintenance jobs) have been observed storing degraded copies there. Other query engines resolve schemas from metadata.json and ignore divergent header copies, so the default follows them. A conflict between two metadata.json schema definitions still always fails the query. )", 0) \ DECLARE(Bool, optimize_distinct_in_order, true, R"( Enable DISTINCT optimization if some columns in DISTINCT form a prefix of sorting. For example, prefix of sorting key in merge tree or ORDER BY statement diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index a13257bba353..62d536511a0c 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -41,6 +41,10 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() /// controls new feature and it's 'true' by default, use 'false' as previous_value). /// It's used to implement `compatibility` setting (see https://github.com/ClickHouse/ClickHouse/issues/35972) /// Note: please check if the key already exists to prevent duplicate entries. + addSettingsChanges(settings_changes_history, "26.10", + { + {"iceberg_tolerate_conflicting_manifest_schemas", false, true, "New setting: when an Iceberg manifest file header carries a schema that conflicts with the schema registered for the same schema-id from metadata.json, prefer the metadata.json schema and log a warning instead of failing the query, matching the behavior of other query engines. `compatibility` below 26.10 restores the previous strict behavior."}, + }); addSettingsChanges(settings_changes_history, "26.8", { {"validate_group_by_all_key_types", true, true, "The validation of the key types that `GROUP BY ALL` expands the `SELECT` expressions into is kept under `compatibility` with 26.7: the previous value is deliberately equal to the new one, because 26.7 already rejected such a key and only a version before 26.7 restores the earlier acceptance."}, diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp index 9c0d2db914de..189f1e700d1b 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp @@ -9,6 +9,7 @@ #include +#include #include #include @@ -17,6 +18,7 @@ #include #include +#include #include #include #include @@ -39,6 +41,11 @@ namespace DB::ErrorCodes extern const int BAD_ARGUMENTS; } +namespace DB::Setting +{ + extern const SettingsBool iceberg_tolerate_conflicting_manifest_schemas; +} + namespace ProfileEvents { extern const Event IcebergPartitionPrunedFiles; @@ -319,7 +326,10 @@ std::shared_ptr ManifestFileIterator::create( const Poco::JSON::Object::Ptr & schema_object = json.extract(); Int32 manifest_schema_id = schema_object->getValue(f_schema_id); - schema_processor.addIcebergTableSchema(schema_object); + schema_processor.addIcebergTableSchema( + schema_object, + IcebergSchemaProcessor::SchemaSource::ManifestFile, + context_->getSettingsRef()[Setting::iceberg_tolerate_conflicting_manifest_schemas]); PartitionSpecification partition_spec_vec; for (size_t i = 0; i != partition_specification->size(); ++i) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp index c154a025ab3a..978f863f9dfd 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp @@ -393,7 +393,17 @@ namespace Iceberg std::string IcebergSchemaProcessor::default_link{}; -void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr) +void IcebergSchemaProcessor::dropCachedSchema(Int32 schema_id) +{ + iceberg_table_schemas_by_ids.erase(schema_id); + clickhouse_table_schemas_by_ids.erase(schema_id); + std::erase_if(transform_dags_by_ids, [schema_id](const auto & item) { return item.first.first == schema_id || item.first.second == schema_id; }); + std::erase_if(clickhouse_types_by_source_ids, [schema_id](const auto & item) { return item.first.first == schema_id; }); + std::erase_if(clickhouse_ids_by_source_names, [schema_id](const auto & item) { return item.first.first == schema_id; }); +} + +void IcebergSchemaProcessor::addIcebergTableSchema( + Poco::JSON::Object::Ptr schema_ptr, SchemaSource source, bool tolerate_conflicting_manifest_schemas) { std::lock_guard lock(mutex); @@ -404,7 +414,6 @@ void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schem if (!schema_ptr->isArray(f_fields) || schema_ptr->getArray(f_fields)->size() == 0) return; - current_schema_id = schema_id; if (iceberg_table_schemas_by_ids.contains(schema_id)) { chassert(clickhouse_table_schemas_by_ids.contains(schema_id)); @@ -414,44 +423,101 @@ void IcebergSchemaProcessor::addIcebergTableSchema(Poco::JSON::Object::Ptr schem type_mapping[f_geography] = f_binary; type_mapping[f_geometry] = f_binary; } - /// A schema-id is immutable per the Iceberg spec: re-binding it to different fields is malformed metadata. - if (!schemasAreIdentical(*iceberg_table_schemas_by_ids.at(schema_id), *schema_ptr, type_mapping)) - throw Exception( - ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, - "Iceberg schema with schema-id {} is bound to two different schemas across metadata versions", - schema_id); - } - else - { - auto fields = schema_ptr->get(f_fields).extract(); - /// A field name is required per the Iceberg spec, and an empty column name is not representable in ClickHouse. - for (size_t i = 0; i != fields->size(); ++i) + if (schemasAreIdentical(*iceberg_table_schemas_by_ids.at(schema_id), *schema_ptr, type_mapping)) + { + /// An identical metadata.json copy confirms a copy that was registered from a manifest header. + if (source == SchemaSource::Metadata) + manifest_sourced_schema_ids.erase(schema_id); + return; + } + + /// The 'schema' key in a manifest file header is only a copy of the table schema at the + /// time the manifest was written; metadata.json is the authoritative source. Broken writers + /// have been observed storing degraded copies in manifest headers under an already-used + /// schema-id (e.g. AWS S3 Tables maintenance jobs writing `timestamp` instead of + /// `timestamptz`, or a schema containing only the partition source columns). Other engines + /// (Spark, Trino, PyIceberg, DuckDB) resolve schemas from metadata.json and ignore such + /// divergent copies that came from a manifest. + const bool registered_from_manifest = manifest_sourced_schema_ids.contains(schema_id); + if (registered_from_manifest) { - auto field = fields->getObject(static_cast(i)); - if (field->getValue(f_name).empty()) + if (source == SchemaSource::Metadata) + { + /// A read registers the metadata.json schemas before it walks any manifest, but the + /// maintenance entrypoints (`remove_orphan_files`, `expire_snapshots`, manifest + /// compaction, mutation validation) can reach a manifest header first on an empty + /// processor. A schema that came from a manifest is never authoritative, so the + /// metadata.json copy replaces it, along with everything that was derived from it. + LOG_WARNING( + getLogger("IcebergSchemaProcessor"), + "Schema-id {} was registered from a manifest file header and differs from the schema " + "metadata.json binds to that id; replacing the schema that came from the manifest", + schema_id); + dropCachedSchema(schema_id); + manifest_sourced_schema_ids.erase(schema_id); + } + else + { + /// Two manifest headers disagree on a schema-id that metadata.json has not defined: + /// there is no authoritative copy to decide which one the data was written with. throw Exception( ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, - "Iceberg schema with schema-id {} has a field with id {} whose name is empty", - schema_id, - field->getValue(f_id)); + "Iceberg schema with schema-id {} is bound to two different schemas by manifest file headers, " + "and metadata.json does not define it", + schema_id); + } } - - auto clickhouse_schema = std::make_shared(); - String current_full_name{}; - for (size_t i = 0; i != fields->size(); ++i) + else { - auto field = fields->getObject(static_cast(i)); - auto name = field->getValue(f_name); - bool required = field->getValue(f_required); - current_full_name = name; - auto type = getFieldType(field, f_type, required, current_full_name, true); - clickhouse_schema->push_back(NameAndTypePair{name, type}); - clickhouse_types_by_source_ids[{schema_id, field->getValue(f_id)}] = NameAndTypePair{current_full_name, type}; - clickhouse_ids_by_source_names[{schema_id, current_full_name}] = field->getValue(f_id); + if (source == SchemaSource::ManifestFile && tolerate_conflicting_manifest_schemas) + { + LOG_WARNING( + getLogger("IcebergSchemaProcessor"), + "Manifest file header carries schema-id {} which differs from the schema already " + "registered for that id from metadata.json; ignoring the manifest header copy " + "(disable setting `iceberg_tolerate_conflicting_manifest_schemas` to make this an error)", + schema_id); + return; + } + /// A schema-id is immutable per the Iceberg spec: re-binding it to different fields is malformed metadata. + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Iceberg schema with schema-id {} is bound to two different schemas across metadata versions", + schema_id); } - clickhouse_table_schemas_by_ids[schema_id] = clickhouse_schema; - iceberg_table_schemas_by_ids[schema_id] = schema_ptr; } + + auto fields = schema_ptr->get(f_fields).extract(); + /// A field name is required per the Iceberg spec, and an empty column name is not representable in ClickHouse. + for (size_t i = 0; i != fields->size(); ++i) + { + auto field = fields->getObject(static_cast(i)); + if (field->getValue(f_name).empty()) + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Iceberg schema with schema-id {} has a field with id {} whose name is empty", + schema_id, + field->getValue(f_id)); + } + + current_schema_id = schema_id; + auto clickhouse_schema = std::make_shared(); + String current_full_name{}; + for (size_t i = 0; i != fields->size(); ++i) + { + auto field = fields->getObject(static_cast(i)); + auto name = field->getValue(f_name); + bool required = field->getValue(f_required); + current_full_name = name; + auto type = getFieldType(field, f_type, required, current_full_name, true); + clickhouse_schema->push_back(NameAndTypePair{name, type}); + clickhouse_types_by_source_ids[{schema_id, field->getValue(f_id)}] = NameAndTypePair{current_full_name, type}; + clickhouse_ids_by_source_names[{schema_id, current_full_name}] = field->getValue(f_id); + } + clickhouse_table_schemas_by_ids[schema_id] = clickhouse_schema; + iceberg_table_schemas_by_ids[schema_id] = schema_ptr; + if (source == SchemaSource::ManifestFile) + manifest_sourced_schema_ids.insert(schema_id); current_schema_id = std::nullopt; } diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h index a072f5ebc127..396f244ec07c 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.h @@ -88,9 +88,23 @@ class IcebergSchemaProcessor using Node = ActionsDAG::Node; public: + /// Where a schema copy being registered comes from. metadata.json is the authoritative source; + /// the 'schema' key of a manifest file header is only a snapshot of the table schema at the time + /// the manifest was written. A schema that came from a manifest is never authoritative: it is + /// replaced by the metadata.json copy of the same schema-id whenever that one is registered, and + /// it may be ignored if it conflicts with an already registered metadata.json copy. + enum class SchemaSource + { + Metadata, + ManifestFile, + }; + explicit IcebergSchemaProcessor(bool allow_geo_parser_ = false) : allow_geo_parser(allow_geo_parser_) {} - void addIcebergTableSchema(Poco::JSON::Object::Ptr schema_ptr); + void addIcebergTableSchema( + Poco::JSON::Object::Ptr schema_ptr, + SchemaSource source = SchemaSource::Metadata, + bool tolerate_conflicting_manifest_schemas = false); std::shared_ptr getClickHouseTableSchemaById(Int32 id); std::shared_ptr getSchemaTransformationDagByIds(Int32 old_id, Int32 new_id); NameAndTypePair getFieldCharacteristics(Int32 schema_version, Int32 source_id) const; @@ -125,6 +139,16 @@ class IcebergSchemaProcessor mutable std::map, Int32> clickhouse_ids_by_source_names TSA_GUARDED_BY(mutex); std::optional current_schema_id TSA_GUARDED_BY(mutex) = 0; std::unordered_map schema_id_by_snapshot TSA_GUARDED_BY(mutex); + /// Schema-ids whose registered copy came from a manifest file header and has not been confirmed + /// by an identical metadata.json copy yet. Such a copy is replaced when metadata.json binds the + /// schema-id to a different schema, and two manifest headers disagreeing on such a schema-id is + /// an error, because no authoritative copy is left to decide between them. + std::unordered_set manifest_sourced_schema_ids TSA_GUARDED_BY(mutex); + + /// Forget the schema registered for `schema_id` together with everything derived from it: the + /// per-field lookups and the cached schema transformation DAGs in either direction. They are + /// keyed by schema-id and never rebuilt once populated, so a stale entry would keep answering. + void dropCachedSchema(Int32 schema_id) TSA_REQUIRES(mutex); NamesAndTypesList getSchemaType(const Poco::JSON::Object::Ptr & schema); DataTypePtr getComplexTypeFromObject(const Poco::JSON::Object::Ptr & type, String & current_full_name, bool is_subfield_of_root); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp index 504f0f8fc7e9..b79d3ba94033 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/tests/gtest_iceberg_schema_processor.cpp @@ -7,6 +7,11 @@ #include #include +namespace DB::ErrorCodes +{ + extern const int ICEBERG_SPECIFICATION_VIOLATION; +} + using namespace DB::Iceberg; namespace @@ -185,6 +190,145 @@ TEST(IcebergSchemaProcessor, RebindingSchemaIdToDifferentTypeStillRejected) EXPECT_THROW(processor.addIcebergTableSchema(second), DB::Exception); } +/// The manifest header 'schema' key is only a copy of the table schema at write time; metadata.json +/// is authoritative. Broken writers (observed with AWS S3 Tables maintenance jobs) store degraded +/// copies in manifest headers under an already-used schema-id, e.g. `timestamp` instead of +/// `timestamptz`. With toleration enabled (the default of `iceberg_tolerate_conflicting_manifest_schemas`), +/// the divergent manifest copy must be ignored and the metadata.json schema kept. +TEST(IcebergSchemaProcessor, ConflictingManifestSchemaToleratedWhenEnabled) +{ + auto from_metadata = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + auto from_manifest = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + IcebergSchemaProcessor processor; + processor.addIcebergTableSchema(from_metadata); + EXPECT_NO_THROW(processor.addIcebergTableSchema( + from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true)); + + /// The metadata.json copy must win: the field keeps the timestamptz type. + auto schema = processor.getClickHouseTableSchemaById(0); + ASSERT_EQ(schema->size(), 1u); + EXPECT_EQ(schema->front().type->getName(), "Nullable(DateTime64(6, 'UTC'))"); +} + +/// With toleration disabled (`compatibility` below 26.9), the same conflict must still fail. +TEST(IcebergSchemaProcessor, ConflictingManifestSchemaRejectedWhenDisabled) +{ + auto from_metadata = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + auto from_manifest = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + IcebergSchemaProcessor processor; + processor.addIcebergTableSchema(from_metadata); + EXPECT_THROW( + processor.addIcebergTableSchema( + from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/false), + DB::Exception); +} + +/// Toleration only applies to manifest header copies: two conflicting metadata.json definitions of +/// the same schema-id are genuine catalog corruption and must always be rejected. +TEST(IcebergSchemaProcessor, ConflictingMetadataSchemaAlwaysRejected) +{ + auto first = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + auto second = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + IcebergSchemaProcessor processor; + processor.addIcebergTableSchema(first); + EXPECT_THROW( + processor.addIcebergTableSchema( + second, IcebergSchemaProcessor::SchemaSource::Metadata, /*tolerate_conflicting_manifest_schemas=*/true), + DB::Exception); +} + +/// A manifest header carrying a schema-id NOT registered from metadata.json (e.g. an expired schema +/// still referenced by an old manifest) must register normally regardless of the toleration flag. +TEST(IcebergSchemaProcessor, ManifestSchemaWithNewIdRegistersNormally) +{ + auto from_manifest = parseSchema(R"json({"schema-id":5,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + IcebergSchemaProcessor processor; + EXPECT_NO_THROW(processor.addIcebergTableSchema( + from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true)); + EXPECT_TRUE(processor.hasClickHouseTableSchemaById(5)); +} + +/// The maintenance entrypoints (`remove_orphan_files`, `expire_snapshots`, manifest compaction) can +/// register a schema that came from a manifest before metadata.json is read. Such a schema is never +/// authoritative: the metadata.json copy of the same schema-id must replace it, together with the +/// per-field lookups and the cached transformation DAGs derived from it. +TEST(IcebergSchemaProcessor, ManifestHeaderRegisteredFirstIsReplacedByMetadata) +{ + auto from_manifest = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + auto other = parseSchema(R"json({"schema-id":1,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"},{"id":2,"name":"v","required":false,"type":"int"}]})json"); + auto from_metadata = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + IcebergSchemaProcessor processor; + processor.addIcebergTableSchema(from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true); + processor.addIcebergTableSchema(other); + + /// Populate everything derived from the schema that came from the manifest. + EXPECT_EQ(processor.getFieldCharacteristics(0, 1).type->getName(), "Nullable(DateTime64(6))"); + auto stale_dag = processor.getSchemaTransformationDagByIds(0, 1); + ASSERT_NE(stale_dag, nullptr); + + EXPECT_NO_THROW(processor.addIcebergTableSchema(from_metadata)); + + auto schema = processor.getClickHouseTableSchemaById(0); + ASSERT_EQ(schema->size(), 1u); + EXPECT_EQ(schema->front().type->getName(), "Nullable(DateTime64(6, 'UTC'))"); + EXPECT_EQ(processor.getFieldCharacteristics(0, 1).type->getName(), "Nullable(DateTime64(6, 'UTC'))"); + EXPECT_NE(processor.getSchemaTransformationDagByIds(0, 1).get(), stale_dag.get()); + + /// Once metadata.json has settled the schema-id, a later divergent schema from a manifest is handled like + /// any header conflicting with a known metadata.json copy: ignored when tolerated, rejected otherwise. + EXPECT_NO_THROW(processor.addIcebergTableSchema( + from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true)); + EXPECT_THROW( + processor.addIcebergTableSchema( + from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/false), + DB::Exception); + EXPECT_EQ(processor.getClickHouseTableSchemaById(0)->front().type->getName(), "Nullable(DateTime64(6, 'UTC'))"); +} + +/// An identical metadata.json copy confirms a schema that came from a manifest and was registered before it; the schema-id then +/// behaves as if it had been registered from metadata.json in the first place. +TEST(IcebergSchemaProcessor, ManifestHeaderRegisteredFirstIsConfirmedByIdenticalMetadata) +{ + auto from_manifest = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + auto from_metadata = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + auto degraded = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + IcebergSchemaProcessor processor; + processor.addIcebergTableSchema(from_manifest, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true); + EXPECT_NO_THROW(processor.addIcebergTableSchema(from_metadata)); + + EXPECT_NO_THROW(processor.addIcebergTableSchema( + degraded, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true)); + EXPECT_THROW( + processor.addIcebergTableSchema( + degraded, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/false), + DB::Exception); + EXPECT_EQ(processor.getClickHouseTableSchemaById(0)->front().type->getName(), "Nullable(DateTime64(6, 'UTC'))"); +} + +/// Two manifest headers that disagree on a schema-id metadata.json has not defined leave no +/// authoritative copy to decide between them, so this is a specification violation even when +/// conflicts with metadata.json are tolerated. Other schema-ids are unaffected. +TEST(IcebergSchemaProcessor, ConflictingManifestHeadersWithoutMetadataSchemaRejected) +{ + auto first = parseSchema(R"json({"schema-id":5,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamp"}]})json"); + auto second = parseSchema(R"json({"schema-id":5,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + auto other = parseSchema(R"json({"schema-id":0,"fields":[{"id":1,"name":"ts","required":false,"type":"timestamptz"}]})json"); + IcebergSchemaProcessor processor; + processor.addIcebergTableSchema(other); + processor.addIcebergTableSchema(first, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true); + try + { + processor.addIcebergTableSchema(second, IcebergSchemaProcessor::SchemaSource::ManifestFile, /*tolerate_conflicting_manifest_schemas=*/true); + FAIL() << "expected ICEBERG_SPECIFICATION_VIOLATION"; + } + catch (const DB::Exception & e) + { + EXPECT_EQ(e.code(), DB::ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION); + } + EXPECT_TRUE(processor.hasClickHouseTableSchemaById(0)); + EXPECT_EQ(processor.getClickHouseTableSchemaById(0)->front().type->getName(), "Nullable(DateTime64(6, 'UTC'))"); +} + /// A renamed field bound to the same schema-id must still be rejected (issue #107316). TEST(IcebergSchemaProcessor, RebindingSchemaIdToRenamedFieldStillRejected) { diff --git a/tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.reference b/tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.reference new file mode 100644 index 000000000000..a3a948fe5178 --- /dev/null +++ b/tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.reference @@ -0,0 +1,7 @@ +strict read +ICEBERG_SPECIFICATION_VIOLATION +tolerant compaction +3 +DateTime64(6, \'UTC\') 2024-01-01 00:00:00.000000 1 +DateTime64(6, \'UTC\') 2024-01-02 00:00:00.000000 2 +DateTime64(6, \'UTC\') 2024-01-03 00:00:00.000000 3 diff --git a/tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.sh b/tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.sh new file mode 100755 index 000000000000..9a72f718081a --- /dev/null +++ b/tests/queries/0_stateless/05212_iceberg_manifest_compaction_conflicting_manifest_schema.sh @@ -0,0 +1,81 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-parallel +# - no-fasttest: requires `IcebergLocal` (USE_AVRO build option) +# - no-parallel: uses DETACH/ATTACH which serializes per database + +# A manifest file header keeps a copy of the schema the data was written with, while metadata.json is +# the authoritative source. The conflict below is produced the way broken writers produce it: the +# manifest headers keep the `timestamp` type, while metadata.json binds the same schema-id to +# `timestamptz`. +# +# A read registers the metadata.json schemas before it walks the manifests, so a strict read meets a +# header conflicting with a known metadata.json copy and fails with `ICEBERG_SPECIFICATION_VIOLATION`. +# `OPTIMIZE TABLE ... MANIFEST` walks the current manifests first and registers the metadata.json +# schemas only afterwards, so on a freshly attached table the schemas that came from the manifests are registered before +# metadata.json: the metadata.json copy has to replace them and the compaction has to succeed. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# The data is written as `timestamp` (parsed in the session time zone) and read back as `timestamptz` +# (rendered in UTC), so the session time zone has to be UTC for the values to round-trip unchanged. +CLICKHOUSE_CLIENT="${CLICKHOUSE_CLIENT} --session_timezone UTC" + +TABLE="t_${CLICKHOUSE_DATABASE}_${RANDOM}" +TABLE_PATH="${USER_FILES_PATH}/${TABLE}/" + +trap 'rm -rf "${TABLE_PATH}" 2>/dev/null' EXIT + +${CLICKHOUSE_CLIENT} --query " + CREATE TABLE ${TABLE} (ts DateTime64(6), v Int32) + ENGINE = IcebergLocal('${TABLE_PATH}', 'Parquet') +" + +# One INSERT per manifest, so the table has enough manifests to compact. +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --use_iceberg_metadata_files_cache=0 -m --query " + INSERT INTO ${TABLE} VALUES ('2024-01-01 00:00:00', 1); + INSERT INTO ${TABLE} VALUES ('2024-01-02 00:00:00', 2); + INSERT INTO ${TABLE} VALUES ('2024-01-03 00:00:00', 3); +" + +# Rebind the schema-id in metadata.json to a schema that differs from the schemas that came from the manifests. +LATEST_METADATA=$(ls "${TABLE_PATH}"metadata/v*.metadata.json | sed 's#.*/v##;s#\.metadata.json##' | sort -n | tail -1) +python3 - "${TABLE_PATH}metadata/v${LATEST_METADATA}.metadata.json" <<'PY' +import json, sys +path = sys.argv[1] +meta = json.load(open(path)) +for schema in meta["schemas"]: + for field in schema["fields"]: + if field["type"] == "timestamp": + field["type"] = "timestamptz" +json.dump(meta, open(path, "w")) +PY + +# Drop the in-memory metadata and the shared schema processor, so the read starts from an empty +# processor and registers the metadata.json schemas before it meets the schemas that came from the manifests. +${CLICKHOUSE_CLIENT} --use_iceberg_metadata_files_cache=0 --query "DETACH TABLE ${TABLE}" +${CLICKHOUSE_CLIENT} --use_iceberg_metadata_files_cache=0 --send_logs_level=fatal --query "ATTACH TABLE ${TABLE}" + +echo "strict read" +${CLICKHOUSE_CLIENT} --use_iceberg_metadata_files_cache=0 --iceberg_tolerate_conflicting_manifest_schemas=0 \ + --query "SELECT count() FROM ${TABLE}" 2>&1 \ + | grep -oF 'ICEBERG_SPECIFICATION_VIOLATION' | head -n1 + +# Start from an empty processor again, so the compaction registers the schemas that came from the manifests first +# and the metadata.json schemas replace them. +${CLICKHOUSE_CLIENT} --use_iceberg_metadata_files_cache=0 --query "DETACH TABLE ${TABLE}" +${CLICKHOUSE_CLIENT} --use_iceberg_metadata_files_cache=0 --send_logs_level=fatal --query "ATTACH TABLE ${TABLE}" + +echo "tolerant compaction" +${CLICKHOUSE_CLIENT} --allow_experimental_iceberg_compaction=1 --use_iceberg_metadata_files_cache=0 \ + --iceberg_tolerate_conflicting_manifest_schemas=1 --send_logs_level=error \ + --query "OPTIMIZE TABLE ${TABLE} MANIFEST SETTINGS iceberg_manifest_min_count_to_compact=2" + +# The compaction rewrote three data manifests into one, and the data is intact. +${CLICKHOUSE_CLIENT} --use_iceberg_metadata_files_cache=0 --iceberg_tolerate_conflicting_manifest_schemas=1 --query " + SELECT count() FROM ${TABLE}; + SELECT toTypeName(ts), ts, v FROM ${TABLE} ORDER BY v; +" + +${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS ${TABLE} SYNC" From 4963403fe6e249b587212db9624ec2518d68bd0d Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 09:35:45 +0000 Subject: [PATCH 044/185] Backport #120036 to 26.8: Fix silent truncation at `LimitReadBuffer` limits and the multipart form budget --- src/Core/ExternalTable.cpp | 13 +- src/Core/ExternalTable.h | 3 + src/Core/MySQL/IMySQLReadPacket.cpp | 11 +- src/Core/MySQL/IMySQLReadPacket.h | 3 + .../tests/gtest_mysql_limited_read_packet.cpp | 121 +++++++++++++ src/IO/LimitReadBuffer.cpp | 11 +- src/IO/LimitReadBuffer.h | 3 + src/IO/tests/gtest_limit_read_buffer.cpp | 162 ++++++++++++++++++ src/Server/HTTP/HTTPServerRequest.cpp | 4 +- src/Server/TCPHandler.cpp | 5 +- ...p_multipart_form_data_total_size.reference | 3 + ...213_http_multipart_form_data_total_size.sh | 32 ++++ 12 files changed, 359 insertions(+), 12 deletions(-) create mode 100644 src/Core/tests/gtest_mysql_limited_read_packet.cpp create mode 100644 src/IO/tests/gtest_limit_read_buffer.cpp create mode 100644 tests/queries/0_stateless/05213_http_multipart_form_data_total_size.reference create mode 100755 tests/queries/0_stateless/05213_http_multipart_form_data_total_size.sh diff --git a/src/Core/ExternalTable.cpp b/src/Core/ExternalTable.cpp index 584b2b50f1fb..a03f0109a3ae 100644 --- a/src/Core/ExternalTable.cpp +++ b/src/Core/ExternalTable.cpp @@ -216,11 +216,12 @@ void ExternalTablesHandler::handlePart(const Poco::Net::MessageHeader & header, const Settings & settings = getContext()->getSettingsRef(); - if (settings[Setting::http_max_multipart_form_data_size]) + const size_t form_data_size_limit = settings[Setting::http_max_multipart_form_data_size]; + if (form_data_size_limit) read_buffer = std::make_unique( stream, LimitReadBuffer::Settings{ - .read_no_more = settings[Setting::http_max_multipart_form_data_size], + .read_no_more = form_data_size_limit > form_data_bytes_read ? form_data_size_limit - form_data_bytes_read : 0, .expect_eof = true, .excetion_hint = "the maximum size of multipart/form-data. This limit can be tuned by 'http_max_multipart_form_data_size' setting", }); @@ -293,6 +294,14 @@ void ExternalTablesHandler::handlePart(const Poco::Net::MessageHeader & header, CompletedPipelineExecutor executor(pipeline); executor.execute(); + + /// Whatever the format left unread still belongs to the part, and `HTMLForm` would skip it + /// outside the limiter. Read it out through the limiter so it counts against the budget, and so + /// that a part running past the budget trips `expect_eof` rather than slipping by. + if (form_data_size_limit) + read_buffer->ignoreAll(); + + form_data_bytes_read += read_buffer->count(); } } diff --git a/src/Core/ExternalTable.h b/src/Core/ExternalTable.h index e7c5918a209d..c120fb22e2cd 100644 --- a/src/Core/ExternalTable.h +++ b/src/Core/ExternalTable.h @@ -91,6 +91,9 @@ class ExternalTablesHandler : public HTMLForm::PartHandler, BaseExternalTable, W private: const Poco::Net::NameValueCollection & params; + /// `http_max_multipart_form_data_size` is one budget across the form's external-table parts, not a + /// fresh limit per part. Other fields have their own bound, `http_max_field_value_size`. + size_t form_data_bytes_read = 0; }; diff --git a/src/Core/MySQL/IMySQLReadPacket.cpp b/src/Core/MySQL/IMySQLReadPacket.cpp index e29fb0385a56..d40af18658ba 100644 --- a/src/Core/MySQL/IMySQLReadPacket.cpp +++ b/src/Core/MySQL/IMySQLReadPacket.cpp @@ -18,6 +18,11 @@ namespace MySQLProtocol void IMySQLReadPacket::readPayload(ReadBuffer & in, uint8_t & sequence_id) { MySQLPacketPayloadReadBuffer payload(in, sequence_id); + readPayloadFrom(payload); +} + +void IMySQLReadPacket::readPayloadFrom(ReadBuffer & payload) +{ payload.next(); readPayloadImpl(payload); if (!payload.eof()) @@ -35,8 +40,10 @@ void IMySQLReadPacket::readPayloadWithUnpacked(ReadBuffer & in) void LimitedReadPacket::readPayload(ReadBuffer &in, uint8_t &sequence_id) { - LimitReadBuffer limited(in, {.read_no_more = 10000, .expect_eof = true, .excetion_hint = "too long MySQL packet."}); - IMySQLReadPacket::readPayload(limited, sequence_id); + /// On the payload, not on `in`: `in` is the connection, which continues with the next packet. + MySQLPacketPayloadReadBuffer payload(in, sequence_id); + LimitReadBuffer limited(payload, {.read_no_more = 10000, .expect_eof = true, .excetion_hint = "too long MySQL packet."}); + readPayloadFrom(limited); } void LimitedReadPacket::readPayloadWithUnpacked(ReadBuffer & in) diff --git a/src/Core/MySQL/IMySQLReadPacket.h b/src/Core/MySQL/IMySQLReadPacket.h index b6c3d59f5eef..1e7fb1b501b4 100644 --- a/src/Core/MySQL/IMySQLReadPacket.h +++ b/src/Core/MySQL/IMySQLReadPacket.h @@ -23,6 +23,9 @@ class IMySQLReadPacket protected: virtual void readPayloadImpl(ReadBuffer & buf) = 0; + + /// `payload` has to end where the packet ends. + void readPayloadFrom(ReadBuffer & payload); }; class LimitedReadPacket : public IMySQLReadPacket diff --git a/src/Core/tests/gtest_mysql_limited_read_packet.cpp b/src/Core/tests/gtest_mysql_limited_read_packet.cpp new file mode 100644 index 000000000000..62fbdb893a44 --- /dev/null +++ b/src/Core/tests/gtest_mysql_limited_read_packet.cpp @@ -0,0 +1,121 @@ +#include + +#include +#include +#include +#include + +#include + +using namespace DB; +using namespace DB::MySQLProtocol; + +namespace DB::ErrorCodes +{ +extern const int LIMIT_EXCEEDED; +} + +namespace +{ + +/// The limit `LimitedReadPacket` applies, from IMySQLReadPacket.cpp. +constexpr size_t PACKET_PAYLOAD_LIMIT = 10000; + +/// Serves its data a few bytes at a time. The limiter only reaches its `expect_eof` check when it +/// has to refill, which a fully buffered source never makes it do. +class ChunkedSource : public ReadBuffer +{ +public: + ChunkedSource(String data_, size_t chunk_) : ReadBuffer(nullptr, 0), data(std::move(data_)), chunk(chunk_) {} + +private: + bool nextImpl() override + { + if (consumed >= data.size()) + return false; + + const size_t count = std::min(chunk, data.size() - consumed); + BufferBase::set(data.data() + consumed, count, 0); + consumed += count; + return true; + } + + String data; + size_t chunk; + size_t consumed = 0; +}; + +struct PayloadCollector : public LimitedReadPacket +{ + String payload; + + void readPayloadImpl(ReadBuffer & buf) override { readStringUntilEOF(payload, buf); } +}; + +String packet(const String & payload, uint8_t sequence_id) +{ + String result; + const size_t length = payload.size(); + result += static_cast(length & 0xFF); + result += static_cast((length >> 8) & 0xFF); + result += static_cast((length >> 16) & 0xFF); + result += static_cast(sequence_id); + result += payload; + return result; +} + +} + +/// The limit counts one packet's payload, not the connection. A payload exactly at the limit is +/// allowed even though the 4-byte header pushes the connection past it, and the next packet is not +/// mistaken for overflow. +TEST(MySQLLimitedReadPacket, PayloadAtTheLimitFollowedByAnotherPacket) +{ + const String first(PACKET_PAYLOAD_LIMIT, 'a'); + ChunkedSource in(packet(first, 0) + packet("second", 1), 64); + + uint8_t sequence_id = 0; + PayloadCollector collector; + collector.readPayload(in, sequence_id); + + EXPECT_EQ(collector.payload, first); + EXPECT_EQ(sequence_id, 1); + + PayloadCollector next; + next.readPayload(in, sequence_id); + EXPECT_EQ(next.payload, "second"); +} + +TEST(MySQLLimitedReadPacket, PayloadOverTheLimitIsRejected) +{ + ChunkedSource in(packet(String(PACKET_PAYLOAD_LIMIT + 1, 'a'), 0), 64); + + uint8_t sequence_id = 0; + PayloadCollector collector; + try + { + collector.readPayload(in, sequence_id); + FAIL() << "An oversized payload was accepted"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::LIMIT_EXCEEDED); + } +} + +/// The unpacked variant is handed a payload that already ends where the packet ends. +TEST(MySQLLimitedReadPacket, UnpackedPayloadOverTheLimitIsRejected) +{ + ChunkedSource in(String(PACKET_PAYLOAD_LIMIT + 1, 'a'), 64); + + PayloadCollector collector; + try + { + collector.readPayloadWithUnpacked(in); + FAIL() << "An oversized unpacked payload was accepted"; + } + catch (const Exception & e) + { + EXPECT_EQ(e.code(), ErrorCodes::LIMIT_EXCEEDED); + } +} diff --git a/src/IO/LimitReadBuffer.cpp b/src/IO/LimitReadBuffer.cpp index 6f580b64f9aa..aa727c5b57cd 100644 --- a/src/IO/LimitReadBuffer.cpp +++ b/src/IO/LimitReadBuffer.cpp @@ -20,15 +20,14 @@ bool LimitReadBuffer::nextImpl() /// Let underlying buffer calculate read bytes in `next()` call. in->position() = position(); - if (bytes >= settings.read_no_less) + if (bytes >= settings.read_no_more) { - if (settings.expect_eof && bytes > settings.read_no_more) + /// A stream ending exactly at the limit is not an error, so the check waits until the limit is + /// reached. `eof` refills the nested buffer, it consumes nothing. + if (settings.expect_eof && !in->eof()) throw Exception(ErrorCodes::LIMIT_EXCEEDED, "Limit for LimitReadBuffer exceeded: {}", settings.excetion_hint); - if (bytes >= settings.read_no_more) - return false; - - //throw Exception(ErrorCodes::CANNOT_READ_ALL_DATA, "Unexpected data, got {} bytes, expected {}", bytes, settings.read_atmost); + return false; } if (!in->next()) diff --git a/src/IO/LimitReadBuffer.h b/src/IO/LimitReadBuffer.h index fbcb4570e83d..af4626a11cc2 100644 --- a/src/IO/LimitReadBuffer.h +++ b/src/IO/LimitReadBuffer.h @@ -19,6 +19,9 @@ class LimitReadBuffer : public ReadBuffer { size_t read_no_less = 0; size_t read_no_more = std::numeric_limits::max(); + /// Throw instead of reporting EOF when the nested buffer still has data at the limit. Do not set + /// it when the nested buffer carries an unrelated message after this one, such as the next + /// keep-alive request: those bytes are not part of what the limit counts. bool expect_eof = false; std::string excetion_hint = {}; }; diff --git a/src/IO/tests/gtest_limit_read_buffer.cpp b/src/IO/tests/gtest_limit_read_buffer.cpp new file mode 100644 index 000000000000..04dde06885f1 --- /dev/null +++ b/src/IO/tests/gtest_limit_read_buffer.cpp @@ -0,0 +1,162 @@ +#include + +#include +#include +#include +#include +#include + +#include + +using namespace DB; + +namespace DB::ErrorCodes +{ +extern const int LIMIT_EXCEEDED; +extern const int CANNOT_READ_ALL_DATA; +} + +namespace +{ + +String readAll(ReadBuffer & in) +{ + String result; + WriteBufferFromString out(result); + copyData(in, out); + out.finalize(); + return result; +} + +int codeOfThrown(ReadBuffer & in) +{ + try + { + readAll(in); + } + catch (const Exception & e) + { + return e.code(); + } + return 0; +} + +} + +TEST(LimitReadBuffer, StreamEndingAtTheLimitIsNotAnError) +{ + ReadBufferFromString nested(std::string_view("0123456789")); + LimitReadBuffer limited(nested, {.read_no_more = 10, .expect_eof = true, .excetion_hint = "hint"}); + EXPECT_EQ(readAll(limited), "0123456789"); +} + +TEST(LimitReadBuffer, ExpectEofRejectsDataPastTheLimit) +{ + ReadBufferFromString nested(std::string_view("0123456789abc")); + LimitReadBuffer limited(nested, {.read_no_more = 10, .expect_eof = true, .excetion_hint = "hint"}); + EXPECT_EQ(codeOfThrown(limited), ErrorCodes::LIMIT_EXCEEDED); +} + +TEST(LimitReadBuffer, ZeroLimitRejectsANonEmptyStream) +{ + ReadBufferFromString nested(std::string_view("a")); + LimitReadBuffer limited(nested, {.read_no_more = 0, .expect_eof = true, .excetion_hint = "hint"}); + EXPECT_EQ(codeOfThrown(limited), ErrorCodes::LIMIT_EXCEEDED); +} + +TEST(LimitReadBuffer, WithoutExpectEofDataPastTheLimitIsCutOff) +{ + ReadBufferFromString nested(std::string_view("0123456789abc")); + LimitReadBuffer limited(nested, {.read_no_more = 10}); + EXPECT_EQ(readAll(limited), "0123456789"); +} + +/// The check runs in `nextImpl`, so a consumer that reads exactly the limit and stops never asks for +/// the byte that would reveal the overflow. `expect_eof` is therefore best-effort, not a guarantee. +TEST(LimitReadBuffer, ExpectEofIsNotCheckedWhenTheConsumerStopsAtTheLimit) +{ + ReadBufferFromString nested(std::string_view("0123456789abc")); + LimitReadBuffer limited(nested, {.read_no_more = 10, .expect_eof = true, .excetion_hint = "hint"}); + char buf[10] = {}; + limited.readStrict(buf, sizeof(buf)); + EXPECT_EQ(String(buf, sizeof(buf)), "0123456789"); +} + +TEST(LimitReadBuffer, ExpectEofShortStreamIsNotAnError) +{ + ReadBufferFromString nested(std::string_view("012")); + LimitReadBuffer limited(nested, {.read_no_more = 10, .expect_eof = true, .excetion_hint = "hint"}); + EXPECT_EQ(readAll(limited), "012"); +} + +TEST(LimitReadBuffer, StreamShorterThanReadNoLessIsAnError) +{ + ReadBufferFromString nested(std::string_view("012")); + LimitReadBuffer limited(nested, {.read_no_less = 10, .read_no_more = 10}); + EXPECT_EQ(codeOfThrown(limited), ErrorCodes::CANNOT_READ_ALL_DATA); +} + +TEST(LimitReadBuffer, WithoutExpectEofStreamEndingAtTheLimitIsFine) +{ + ReadBufferFromString nested(std::string_view("0123456789")); + LimitReadBuffer limited(nested, {.read_no_more = 10}); + EXPECT_EQ(readAll(limited), "0123456789"); +} + +TEST(LimitReadBuffer, WithoutExpectEofShortStreamIsFine) +{ + ReadBufferFromString nested(std::string_view("012")); + LimitReadBuffer limited(nested, {.read_no_more = 10}); + EXPECT_EQ(readAll(limited), "012"); +} + +TEST(LimitReadBuffer, WithoutExpectEofEmptyStreamIsFine) +{ + ReadBufferFromString nested(std::string_view("")); + LimitReadBuffer limited(nested, {.read_no_more = 10}); + EXPECT_EQ(readAll(limited), ""); +} + +TEST(LimitReadBuffer, WithoutExpectEofZeroLimitReadsNothing) +{ + ReadBufferFromString nested(std::string_view("abc")); + { + LimitReadBuffer limited(nested, {.read_no_more = 0}); + EXPECT_EQ(readAll(limited), ""); + } + EXPECT_EQ(readAll(nested), "abc"); +} + +TEST(LimitReadBuffer, WithoutExpectEofNestedBufferContinuesPastTheLimit) +{ + ReadBufferFromString nested(std::string_view("0123456789abc")); + { + LimitReadBuffer limited(nested, {.read_no_more = 10}); + EXPECT_EQ(readAll(limited), "0123456789"); + } + EXPECT_EQ(readAll(nested), "abc"); +} + +/// The `Content-Length` shape of `HTTPServerRequest`. With keep-alive the nested buffer holds the next +/// request, so the bytes past the limit have to stay readable. +TEST(LimitReadBuffer, ExactLengthLeavesTheRestForTheNextReader) +{ + ReadBufferFromString nested(std::string_view("0123456789abc")); + { + LimitReadBuffer limited(nested, {.read_no_less = 10, .read_no_more = 10}); + EXPECT_EQ(readAll(limited), "0123456789"); + } + EXPECT_EQ(readAll(nested), "abc"); +} + +TEST(LimitReadBuffer, PartialReadLeavesNestedBufferAtTheConsumedOffset) +{ + ReadBufferFromString nested(std::string_view("0123456789abc")); + { + LimitReadBuffer limited(nested, {.read_no_more = 10}); + char buf[4] = {}; + limited.readStrict(buf, sizeof(buf)); + EXPECT_EQ(String(buf, sizeof(buf)), "0123"); + } + EXPECT_EQ(readAll(nested), "456789abc"); +} diff --git a/src/Server/HTTP/HTTPServerRequest.cpp b/src/Server/HTTP/HTTPServerRequest.cpp index c580ee955f4b..a0cbe4e8a2c0 100644 --- a/src/Server/HTTP/HTTPServerRequest.cpp +++ b/src/Server/HTTP/HTTPServerRequest.cpp @@ -93,7 +93,9 @@ HTTPServerRequest::HTTPServerRequest(HTTPContextPtr context, HTTPServerResponse else if (hasContentLength()) { size_t content_length = getContentLength(); - stream = std::make_shared(std::move(in), LimitReadBuffer::Settings{.read_no_less = content_length, .read_no_more = content_length, .expect_eof = true}); + /// No `expect_eof`: `Content-Length` already defines where the body ends, and with keep-alive + /// the socket legitimately holds the bytes of the next request. + stream = std::make_shared(std::move(in), LimitReadBuffer::Settings{.read_no_less = content_length, .read_no_more = content_length}); stream_is_bounded = true; } else if (getMethod() != HTTPRequest::HTTP_GET && getMethod() != HTTPRequest::HTTP_HEAD && getMethod() != HTTPRequest::HTTP_DELETE) diff --git a/src/Server/TCPHandler.cpp b/src/Server/TCPHandler.cpp index 18d9711bf0c5..c681f104ec35 100644 --- a/src/Server/TCPHandler.cpp +++ b/src/Server/TCPHandler.cpp @@ -1947,7 +1947,10 @@ bool TCPHandler::receiveProxyHeader() /// Only PROXYv1 is supported. /// Validation of protocol is not fully performed. - LimitReadBuffer limit_in(*in, {.read_no_more=107, .expect_eof=true}); /// Maximum length from the specs. + /// No `expect_eof`: except for the `UNKNOWN` health check below, the client sends its handshake + /// right after the header, so the connection does not end at the limit. An over-long header is + /// rejected anyway, by carrying no `\r\n` within these 107 bytes. + LimitReadBuffer limit_in(*in, {.read_no_more=107}); /// Maximum length from the specs. assertString("PROXY ", limit_in); diff --git a/tests/queries/0_stateless/05213_http_multipart_form_data_total_size.reference b/tests/queries/0_stateless/05213_http_multipart_form_data_total_size.reference new file mode 100644 index 000000000000..a9796c468d97 --- /dev/null +++ b/tests/queries/0_stateless/05213_http_multipart_form_data_total_size.reference @@ -0,0 +1,3 @@ +the maximum size of multipart/form-data +30 +50 diff --git a/tests/queries/0_stateless/05213_http_multipart_form_data_total_size.sh b/tests/queries/0_stateless/05213_http_multipart_form_data_total_size.sh new file mode 100755 index 000000000000..4d497382264e --- /dev/null +++ b/tests/queries/0_stateless/05213_http_multipart_form_data_total_size.sh @@ -0,0 +1,32 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# `http_max_multipart_form_data_size` cannot be set through URL parameters, so it comes from a user. +user="u_${CLICKHOUSE_DATABASE}" +${CLICKHOUSE_CLIENT} -q "CREATE USER $user IDENTIFIED WITH plaintext_password BY 'pw' SETTINGS http_max_multipart_form_data_size = 100" +${CLICKHOUSE_CLIENT} -q "GRANT CREATE TEMPORARY TABLE, SELECT ON *.* TO $user" + +url="${CLICKHOUSE_URL}&user=${user}&password=pw" +part="${CLICKHOUSE_TMP}/${CLICKHOUSE_DATABASE}_part.tsv" +python3 -c 'import sys; sys.stdout.write("x\n" * 30)' > "$part" + +# Two parts of 60 bytes: each one fits the 100 byte limit on its own, the form as a whole does not. +${CLICKHOUSE_CURL} -sS -F "a=@$part" -F "b=@$part" \ + "${url}&query=SELECT+count()+FROM+a&a_structure=s+String&b_structure=s+String&a_format=TSV&b_format=TSV" \ + | grep -o -m1 'the maximum size of multipart/form-data' + +# One part of the same size stays under the limit. +${CLICKHOUSE_CURL} -sS -F "a=@$part" \ + "${url}&query=SELECT+count()+FROM+a&a_structure=s+String&a_format=TSV" + +# A part that ends exactly at the limit is not over it. +exact="${CLICKHOUSE_TMP}/${CLICKHOUSE_DATABASE}_exact.tsv" +python3 -c 'import sys; sys.stdout.write("x\n" * 50)' > "$exact" +${CLICKHOUSE_CURL} -sS -F "a=@$exact" \ + "${url}&query=SELECT+count()+FROM+a&a_structure=s+String&a_format=TSV" + +${CLICKHOUSE_CLIENT} -q "DROP USER $user" +rm -f "$part" "$exact" From 04596a5465e15351f23350e6245c416604099f35 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 09:37:47 +0000 Subject: [PATCH 045/185] Backport #116692 to 26.8: Prevent stale OpenSSL errors from breaking asynchronous TLS connections --- .../include/Poco/Net/SecureSocketImpl.h | 9 +- .../include/Poco/Net/SecureStreamSocketImpl.h | 9 + .../NetSSL_OpenSSL/src/SecureSocketImpl.cpp | 172 ++++++-- src/Common/tests/gtest_ssl_error_queue.cpp | 392 ++++++++++++++++++ .../tests/gtest_ssl_send_pending_data.cpp | 177 ++++++++ src/IO/SilkFiberStreamSocketImpl.h | 5 - src/IO/SilkSecureFiberStreamSocketImpl.cpp | 150 ++----- src/IO/SilkSecureFiberStreamSocketImpl.h | 5 - src/IO/SocketPeerClosed.cpp | 99 +++-- src/IO/SocketPeerClosed.h | 4 +- .../tests/gtest_silk_fiber_stream_socket.cpp | 34 +- 11 files changed, 819 insertions(+), 237 deletions(-) create mode 100644 src/Common/tests/gtest_ssl_error_queue.cpp create mode 100644 src/Common/tests/gtest_ssl_send_pending_data.cpp diff --git a/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h b/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h index 475226652f5a..82b38890b791 100644 --- a/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h +++ b/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h @@ -149,6 +149,10 @@ namespace Net /// The object is normally guarded by the socket's mutex; a caller that uses it /// directly must ensure the socket is not accessed concurrently. + void markFatalError(); + /// Records that an external operation on the underlying `SSL` object failed fatally. + /// An orderly SSL shutdown must not be attempted afterwards. + X509 * peerCertificate() const; /// Returns the peer's certificate. @@ -233,7 +237,7 @@ namespace Net /// Returns true iff the given host name is the local host /// (either "localhost" or "127.0.0.1"). - bool mustRetry(int rc, Poco::Timespan & remaining_time); + bool mustRetry(int rc, int sslError, int socketError, Poco::Timespan & remaining_time); /// Returns true if the last operation should be retried, /// otherwise false. /// @@ -247,7 +251,7 @@ namespace Net /// not become readable or writable within the sockets /// receive or send timeout. - int handleError(int rc); + int handleError(int rc, int sslError, int socketError, unsigned long errorCode); /// Handles an SSL error by throwing an appropriate exception. void reset(); @@ -281,6 +285,7 @@ namespace Net Poco::AutoPtr _pSocket; Context::Ptr _pContext; bool _needHandshake; + bool _fatalError; std::string _peerHostName; Session::Ptr _pSession; const BIO_METHOD * _bioMethod = nullptr; diff --git a/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureStreamSocketImpl.h b/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureStreamSocketImpl.h index a222ee01914c..c5420492ec66 100644 --- a/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureStreamSocketImpl.h +++ b/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureStreamSocketImpl.h @@ -183,6 +183,9 @@ namespace Net /// Returns the underlying OpenSSL SSL object, or null if the SSL handshake /// has not been performed yet. + void markFatalError(); + /// Records that an external operation on the underlying `SSL` object failed fatally. + void setLazyHandshake(bool flag = true); /// Enable lazy SSL handshake. If enabled, the SSL handshake /// will be performed the first time date is sent or @@ -301,6 +304,12 @@ namespace Net } + inline void SecureStreamSocketImpl::markFatalError() + { + _impl.markFatalError(); + } + + inline Session::Ptr SecureStreamSocketImpl::currentSession() { return _impl.currentSession(); diff --git a/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp b/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp index 60a55fec88bb..3636d4b5fdd3 100644 --- a/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp +++ b/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp @@ -25,6 +25,7 @@ #include "Poco/NumberFormatter.h" #include "Poco/NumberParser.h" #include "Poco/Format.h" +#include #include #include @@ -60,11 +61,52 @@ struct RemainingTimeCounter Poco::Timestamp start; }; +struct SSLOperationResult +{ + int rc = 0; + int sslError = SSL_ERROR_NONE; + int socketError = 0; + unsigned long errorCode = 0; +}; + + +template +SSLOperationResult performSSLOperation(SSL * ssl, Operation && operation, bool zeroIsError = true) +{ + /// The error queue contract for TLS I/O and `SSL_get_error`: + /// https://docs.openssl.org/3.5/man3/SSL_get_error/ + /// The queue is cleared with `ERR_clear_error`: + /// https://docs.openssl.org/3.5/man3/ERR_clear_error/ + ERR_clear_error(); + + SSLOperationResult result; + /// `errno` is only meaningful if it was set by this operation. In particular, a custom `BIO` + /// can report `SSL_ERROR_SYSCALL` without changing it, so do not inherit another syscall result. + errno = 0; + result.rc = operation(); + + if (result.rc < 0 || (zeroIsError && result.rc == 0)) + { + /// Save `errno` before calling another function. `SSL_get_error` must be the first + /// OpenSSL call after the operation and must observe the queue created by that operation. + result.socketError = errno; + result.sslError = SSL_get_error(ssl, result.rc); + result.errorCode = ERR_get_error(); + } + + /// `errorCode` above preserves the diagnostic used by `handleError`. Do not leave any + /// additional entries in the thread-local queue for another connection on this thread. + ERR_clear_error(); + return result; +} + + SecureSocketImpl::SecureSocketImpl(Poco::AutoPtr pSocketImpl, Context::Ptr pContext): _pSSL(nullptr), _pSocket(pSocketImpl), _pContext(pContext), - _needHandshake(false) + _needHandshake(false), + _fatalError(false) { poco_check_ptr (_pSocket); poco_check_ptr (_pContext); @@ -111,6 +153,13 @@ void SecureSocketImpl::setMutex(std::unique_ptr mutex) } +void SecureSocketImpl::markFatalError() +{ + ScopedLock lock(*_mutex); + _fatalError = true; +} + + const BIO_METHOD * SecureSocketImpl::getBioMethod() const { return _bioMethod ? _bioMethod : BIO_s_socket(); @@ -121,6 +170,7 @@ void SecureSocketImpl::acceptSSL() { ScopedLock lock(*_mutex); poco_assert (!_pSSL); + _fatalError = false; BIO* pBIO = BIO_new(getBioMethod()); if (!pBIO) throw SSLException("Cannot create BIO object"); @@ -190,6 +240,7 @@ void SecureSocketImpl::connectSSL(bool performHandshake) ScopedLock lock(*_mutex); poco_assert (!_pSSL); poco_assert (_pSocket->initialized()); + _fatalError = false; BIO* pBIO = BIO_new(getBioMethod()); if (!pBIO) throw SSLException("Cannot create SSL BIO object"); @@ -218,31 +269,39 @@ void SecureSocketImpl::connectSSL(bool performHandshake) SSL_set_session(_pSSL, _pSession->sslSession()); } + SSL_set_connect_state(_pSSL); + _needHandshake = true; + try { if (performHandshake && _pSocket->getBlocking()) { - int ret; + SSLOperationResult result; Poco::Timespan remaining_time = getMaxTimeoutOrLimit(); do { RemainingTimeCounter counter(remaining_time); - ret = SSL_connect(_pSSL); + result = performSSLOperation(_pSSL, [this] + { + return SSL_connect(_pSSL); + }); + } + while (mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); + if (result.rc <= 0) + { + if (handleError(result.rc, result.sslError, result.socketError, result.errorCode) < 0) + throw Poco::TimeoutException("SSL handshake timed out"); + throw SSLConnectionUnexpectedlyClosedException(); } - while (mustRetry(ret, remaining_time)); - handleError(ret); + _needHandshake = false; verifyPeerCertificate(); } - else - { - SSL_set_connect_state(_pSSL); - _needHandshake = true; - } } catch (...) { SSL_free(_pSSL); _pSSL = 0; + _fatalError = false; throw; } } @@ -271,6 +330,16 @@ void SecureSocketImpl::shutdown() ScopedLock lock(*_mutex); if (_pSSL) { + if (_fatalError) + { + /// OpenSSL forbids `SSL_shutdown` after a fatal TLS or syscall error. + /// https://docs.openssl.org/3.5/man3/SSL_shutdown/ + /// Close the underlying transport without attempting an orderly TLS shutdown. + if (_pSocket->getBlocking()) + _pSocket->shutdown(); + return; + } + // Don't shut down the socket more than once. int shutdownState = SSL_get_shutdown(_pSSL); bool shutdownSent = (shutdownState & SSL_SENT_SHUTDOWN) == SSL_SENT_SHUTDOWN; @@ -283,8 +352,21 @@ void SecureSocketImpl::shutdown() // most web browsers, so we just set the shutdown // flag by calling SSL_shutdown() once and be // done with it. - int rc = SSL_shutdown(_pSSL); - if (rc < 0) handleError(rc); + /// A zero result is not an error for `SSL_shutdown` and must not be passed to + /// `SSL_get_error`; it means that `close_notify` was sent but not received yet. + SSLOperationResult result; + Poco::Timespan remaining_time = getMaxTimeoutOrLimit(); + do + { + RemainingTimeCounter counter(remaining_time); + result = performSSLOperation(_pSSL, [this] + { + return SSL_shutdown(_pSSL); + }, false); + } + while (result.rc < 0 && mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); + if (result.rc < 0) + handleError(result.rc, result.sslError, result.socketError, result.errorCode); if (_pSocket->getBlocking()) { _pSocket->shutdown(); @@ -328,24 +410,21 @@ int SecureSocketImpl::sendBytes(const void* buffer, int length, int flags) return rc; } + SSLOperationResult result; Poco::Timespan remaining_time = getMaxTimeoutOrLimit(); do { RemainingTimeCounter counter(remaining_time); - rc = SSL_write(_pSSL, buffer, length); + result = performSSLOperation(_pSSL, [this, buffer, length] + { + return SSL_write(_pSSL, buffer, length); + }); } - while (mustRetry(rc, remaining_time)); + while (mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); + rc = result.rc; if (rc <= 0) { - // At this stage we still can have last not yet received SSL message containing SSL error - // so make a read to force SSL to process possible SSL error - if (SSL_get_error(_pSSL, rc) == SSL_ERROR_SYSCALL && SocketImpl::lastError() == POCO_ECONNRESET) - { - char c = 0; - SSL_read(_pSSL, &c, 1); - } - - rc = handleError(rc); + rc = handleError(rc, result.sslError, result.socketError, result.errorCode); if (rc == 0) throw SSLConnectionUnexpectedlyClosedException(); if (rc < 0 && _pSocket->getBlocking()) throw Poco::TimeoutException("SSL_write timed out"); @@ -379,6 +458,7 @@ int SecureSocketImpl::receiveBytes(void* buffer, int length, int flags) return rc; } + SSLOperationResult result; Poco::Timespan remaining_time = getMaxTimeoutOrLimit(); do { @@ -386,12 +466,16 @@ int SecureSocketImpl::receiveBytes(void* buffer, int length, int flags) /// so thread can be blocked on recv/send and epoll_wait several times /// until SSL_read will return rc > 0. Let's use our own time counter. RemainingTimeCounter counter(remaining_time); - rc = SSL_read(_pSSL, buffer, length); + result = performSSLOperation(_pSSL, [this, buffer, length] + { + return SSL_read(_pSSL, buffer, length); + }); } - while (mustRetry(rc, remaining_time)); + while (mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); + rc = result.rc; if (rc <= 0) { - rc = handleError(rc); + rc = handleError(rc, result.sslError, result.socketError, result.errorCode); if (rc < 0 && _pSocket->getBlocking()) throw Poco::TimeoutException("SSL_read timed out"); return rc; @@ -419,16 +503,21 @@ int SecureSocketImpl::completeHandshake() poco_check_ptr (_pSSL); int rc; + SSLOperationResult result; Poco::Timespan remaining_time = getMaxTimeoutOrLimit(); do { RemainingTimeCounter counter(remaining_time); - rc = SSL_do_handshake(_pSSL); + result = performSSLOperation(_pSSL, [this] + { + return SSL_do_handshake(_pSSL); + }); } - while (mustRetry(rc, remaining_time)); + while (mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); + rc = result.rc; if (rc <= 0) { - rc = handleError(rc); + rc = handleError(rc, result.sslError, result.socketError, result.errorCode); if (rc < 0 && _pSocket->getBlocking()) throw Poco::TimeoutException("SSL handshake timed out"); return rc; @@ -536,15 +625,13 @@ Poco::Timespan SecureSocketImpl::getMaxTimeoutOrLimit() return remaining_time; } -bool SecureSocketImpl::mustRetry(int rc, Poco::Timespan& remaining_time) +bool SecureSocketImpl::mustRetry(int rc, int sslError, int socketError, Poco::Timespan& remaining_time) { if (remaining_time == 0) return false; ScopedLock lock(*_mutex); if (rc <= 0) { - int sslError = SSL_get_error(_pSSL, rc); - int socketError = _pSocket->lastError(); switch (sslError) { case SSL_ERROR_WANT_READ: @@ -571,21 +658,20 @@ bool SecureSocketImpl::mustRetry(int rc, Poco::Timespan& remaining_time) case SSL_ERROR_SYSCALL: return socketError == POCO_EAGAIN || socketError == POCO_EINTR; default: - return socketError == POCO_EINTR; + /// `errno` is only meaningful for `SSL_ERROR_SYSCALL`; other errors must not + /// be retried because of a leftover `EINTR`, especially `SSL_ERROR_SSL`. + return false; } } return false; } -int SecureSocketImpl::handleError(int rc) +int SecureSocketImpl::handleError(int rc, int sslError, int error, unsigned long errorCode) { ScopedLock lock(*_mutex); if (rc > 0) return rc; - int sslError = SSL_get_error(_pSSL, rc); - int error = SocketImpl::lastError(); - switch (sslError) { case SSL_ERROR_ZERO_RETURN: @@ -598,18 +684,21 @@ int SecureSocketImpl::handleError(int rc) case SSL_ERROR_WANT_ACCEPT: case SSL_ERROR_WANT_X509_LOOKUP: // these should not occur + _fatalError = true; poco_bugcheck(); return rc; case SSL_ERROR_SYSCALL: + _fatalError = true; if (error != 0) { SocketImpl::error(error); } - // fallthrough + [[fallthrough]]; + case SSL_ERROR_SSL: default: { - long lastError = ERR_get_error(); - if (lastError == 0) + _fatalError = true; + if (errorCode == 0) { if (rc == 0) { @@ -631,7 +720,7 @@ int SecureSocketImpl::handleError(int rc) else { char buffer[256]; - ERR_error_string_n(lastError, buffer, sizeof(buffer)); + ERR_error_string_n(errorCode, buffer, sizeof(buffer)); std::string msg(buffer); throw SSLException(msg); } @@ -658,6 +747,7 @@ void SecureSocketImpl::reset() SSL_free(_pSSL); _pSSL = nullptr; } + _fatalError = false; } diff --git a/src/Common/tests/gtest_ssl_error_queue.cpp b/src/Common/tests/gtest_ssl_error_queue.cpp new file mode 100644 index 000000000000..53e2a560ecf5 --- /dev/null +++ b/src/Common/tests/gtest_ssl_error_queue.cpp @@ -0,0 +1,392 @@ +#include "config.h" + +#if USE_SSL + +#include + +#include +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +#include +#include +#include +#include +#include +#include +#include + +#include + + +namespace +{ + +void leaveShutdownWhileInInit(Poco::Net::Context::Ptr context) +{ + ERR_clear_error(); + SSL * ssl = SSL_new(context->sslContext()); + ASSERT_NE(ssl, nullptr); + SSL_set_connect_state(ssl); + ASSERT_LT(SSL_shutdown(ssl), 0); + SSL_free(ssl); + + const auto error = ERR_peek_last_error(); + /// Unwrap the error manually because OpenSSL declares `ERR_GET_LIB` and + /// `ERR_GET_REASON` with `ossl_unused`, which causes compiler warnings when used. + ASSERT_EQ((error >> ERR_LIB_OFFSET) & ERR_LIB_MASK, ERR_LIB_SSL); + ASSERT_EQ(error & ERR_REASON_MASK, SSL_R_SHUTDOWN_WHILE_IN_INIT); +} + +int failReadWithoutErrno(BIO *, char *, int) +{ + /// Return a fatal `BIO` error without setting either `errno` or the OpenSSL error queue. + return -1; +} + +long failReadCtrl(BIO *, int command, long, void *) // NOLINT(google-runtime-int) +{ + if (command == BIO_CTRL_FLUSH) + return 1; + return 0; +} + +int failReadCreate(BIO * bio) +{ + BIO_set_init(bio, 1); + BIO_set_data(bio, nullptr); + return 1; +} + +int failReadDestroy(BIO *) +{ + return 1; +} + +const BIO_METHOD * failReadBioMethod() +{ + static const BIO_METHOD * method = [] + { + BIO_METHOD * result = BIO_meth_new(BIO_get_new_index() | BIO_TYPE_SOURCE_SINK, "fail-read-without-errno"); + BIO_meth_set_read(result, failReadWithoutErrno); + BIO_meth_set_ctrl(result, failReadCtrl); + BIO_meth_set_create(result, failReadCreate); + BIO_meth_set_destroy(result, failReadDestroy); + return result; + }(); + return method; +} + +int retryRead(BIO * bio, char *, int) +{ + BIO_clear_retry_flags(bio); + BIO_set_retry_read(bio); + return -1; +} + +int retryWrite(BIO * bio, const char *, int) +{ + BIO_clear_retry_flags(bio); + BIO_set_retry_write(bio); + return -1; +} + +long retryCtrl(BIO *, int command, long, void *) // NOLINT(google-runtime-int) +{ + if (command == BIO_CTRL_FLUSH) + return 1; + return 0; +} + +int retryCreate(BIO * bio) +{ + BIO_set_init(bio, 1); + BIO_set_data(bio, nullptr); + return 1; +} + +int retryDestroy(BIO *) +{ + return 1; +} + +const BIO_METHOD * retryBioMethod() +{ + static const BIO_METHOD * method = [] + { + BIO_METHOD * result = BIO_meth_new(BIO_get_new_index() | BIO_TYPE_SOURCE_SINK, "always-retry"); + BIO_meth_set_read(result, retryRead); + BIO_meth_set_write(result, retryWrite); + BIO_meth_set_ctrl(result, retryCtrl); + BIO_meth_set_create(result, retryCreate); + BIO_meth_set_destroy(result, retryDestroy); + return result; + }(); + return method; +} + +class ExhaustTimeoutStreamSocketImpl final : public Poco::Net::StreamSocketImpl +{ +public: + bool pollImpl(Poco::Timespan & timeout, int) override + { + timeout = 0; + return true; + } +}; + +class LiveTLSPair +{ +public: + LiveTLSPair() + : server_context(cert.makeContext(Poco::Net::Context::SERVER_USE)) + , client_context(cert.makeContext(Poco::Net::Context::CLIENT_USE)) + , listener(Poco::Net::SocketAddress("127.0.0.1", 0), 1, server_context) + , server_thread([this] + { + runServer(); + }) + { + try + { + client = std::make_unique( + Poco::Net::SocketAddress("127.0.0.1", listener.address().port()), client_context); + if (client->sendBytes("x", 1) != 1) + throw std::runtime_error("TLS client could not send the handshake byte"); + server_ready.wait(); + if (server_exception) + std::rethrow_exception(server_exception); + } + catch (...) + { + stop(); + throw; + } + } + + ~LiveTLSPair() + { + stop(); + } + + void sendMalformedRecord() + { + send_malformed_record.store(true, std::memory_order_release); + action_requested.count_down(); + action_was_requested = true; + action_done.wait(); + if (server_exception) + std::rethrow_exception(server_exception); + } + + Poco::Net::SecureStreamSocket & getClient() + { + return *client; + } + + Poco::Net::Context::Ptr getClientContext() const + { + return client_context; + } + +private: + void runServer() + { + try + { + Poco::Net::SecureStreamSocket peer(listener.acceptConnection()); + char byte = 0; + if (peer.receiveBytes(&byte, 1) != 1) + throw std::runtime_error("TLS server did not receive the handshake byte"); + + server_ready.count_down(); + action_requested.wait(); + + if (send_malformed_record.load(std::memory_order_acquire)) + { + /// A syntactically TLS-looking application-data record with an invalid payload. + /// It bypasses the server's `SSL` object and causes a fatal record-layer error on + /// the client while leaving the underlying TCP connection open. + constexpr unsigned char malformed_record[] = {0x17, 0x03, 0x03, 0x00, 0x01, 0xff}; + const ssize_t sent = ::send( + peer.impl()->sockfd(), malformed_record, sizeof(malformed_record), MSG_NOSIGNAL); + if (sent != static_cast(sizeof(malformed_record))) + throw std::runtime_error("TLS server could not send the malformed record"); + } + + action_done.count_down(); + finish.wait(); + peer.abort(); + } + catch (...) + { + server_exception = std::current_exception(); + server_ready.count_down(); + action_done.count_down(); + } + } + + void stop() noexcept + { + if (stopped) + return; + stopped = true; + + if (client) + { + try + { + client->abort(); + } + catch (...) + { + /// Ok: report the failure but continue cleanup so that the server thread is always joined. + ADD_FAILURE() << "Failed to abort the TLS client during cleanup"; + } + } + + if (!action_was_requested) + action_requested.count_down(); + finish.count_down(); + try + { + listener.close(); + } + catch (...) + { + /// Ok: report the failure but continue cleanup so that the server thread is always joined. + ADD_FAILURE() << "Failed to close the TLS listener during cleanup"; + } + if (server_thread.joinable()) + server_thread.join(); + } + + EphemeralCert cert; + Poco::Net::Context::Ptr server_context; + Poco::Net::Context::Ptr client_context; + Poco::Net::SecureServerSocket listener; + std::latch server_ready{1}; + std::latch action_requested{1}; + std::latch action_done{1}; + std::latch finish{1}; + std::atomic send_malformed_record{false}; + std::exception_ptr server_exception; + std::thread server_thread; + std::unique_ptr client; + bool action_was_requested = false; + bool stopped = false; +}; + +} + + +TEST(SSLErrorQueue, StaleErrorDoesNotChangeNonBlockingReadRetry) +{ + LiveTLSPair pair; + auto & client = pair.getClient(); + client.setBlocking(false); + + char byte = 0; + leaveShutdownWhileInInit(pair.getClientContext()); + EXPECT_EQ(client.receiveBytes(&byte, 1), Poco::Net::SecureStreamSocket::ERR_SSL_WANT_READ); + EXPECT_EQ(ERR_peek_error(), 0UL); + + /// The external async socket path retries `receiveBytes` after polling. The queue must be + /// cleared for every attempt, not only for the first operation on the connection. + leaveShutdownWhileInInit(pair.getClientContext()); + EXPECT_EQ(client.receiveBytes(&byte, 1), Poco::Net::SecureStreamSocket::ERR_SSL_WANT_READ); + EXPECT_EQ(ERR_peek_error(), 0UL); +} + + +TEST(SSLErrorQueue, StaleErrnoDoesNotRetryFatalBioError) +{ + LiveTLSPair pair; + auto & client = pair.getClient(); + client.setSendTimeout(Poco::Timespan(0, 10'000)); + client.setReceiveTimeout(Poco::Timespan(0, 10'000)); + + auto * client_impl = static_cast(client.impl()); + SSL * ssl = client_impl->ssl(); + ASSERT_NE(ssl, nullptr); + + /// Replace only the read `BIO`; `SSL_set0_rbio` transfers its ownership to `SSL`. + BIO * failing_read_bio = BIO_new(failReadBioMethod()); + ASSERT_NE(failing_read_bio, nullptr); + SSL_set0_rbio(ssl, failing_read_bio); + + /// The failing `BIO` deliberately leaves `errno` unchanged. A stale retriable value must not + /// make `SecureSocketImpl::mustRetry` repeat the operation or report an unrelated socket error. + errno = EINTR; + char byte = 0; + EXPECT_THROW(client.receiveBytes(&byte, 1), Poco::Net::SSLConnectionUnexpectedlyClosedException); + EXPECT_EQ(ERR_peek_error(), 0UL); +} + + +TEST(SSLErrorQueue, BlockingConnectDoesNotAcceptTimedOutHandshake) +{ + Poco::Net::ServerSocket listener(Poco::Net::SocketAddress("127.0.0.1", 0)); + EphemeralCert cert; + auto client_context = cert.makeContext(Poco::Net::Context::CLIENT_USE); + + Poco::AutoPtr client_impl + = new Poco::Net::SecureStreamSocketImpl(new ExhaustTimeoutStreamSocketImpl, client_context); + client_impl->setBioMethod(retryBioMethod()); + + /// The custom `BIO` keeps returning `WANT_READ` or `WANT_WRITE`. The first poll reports + /// readiness but consumes the entire timeout, so a blocking handshake must not be accepted. + EXPECT_THROW(client_impl->connect(listener.address()), Poco::TimeoutException); +} + + +TEST(SSLErrorQueue, FatalReadSkipsTLSShutdown) +{ + LiveTLSPair pair; + pair.sendMalformedRecord(); + + char byte = 0; + EXPECT_THROW(pair.getClient().receiveBytes(&byte, 1), Poco::Net::SSLException); + + /// `shutdown` must close the transport directly. Calling `SSL_shutdown` after the fatal read + /// would raise another SSL exception and could contaminate the thread's error queue. + EXPECT_NO_THROW(pair.getClient().shutdown()); + EXPECT_EQ(ERR_peek_error(), 0UL); +} + + +TEST(SSLErrorQueue, FatalPeekSkipsTLSShutdown) +{ + LiveTLSPair pair; + pair.sendMalformedRecord(); + + auto state = DB::SocketState::Idle; + /// `poll` can first wake for a TLS 1.3 post-handshake message. Repeat while `SSL_peek` + /// consumes only such control records, with a total timeout of five seconds. + for (int attempt = 0; attempt < 50 && state == DB::SocketState::Idle; ++attempt) + { + if (pair.getClient().poll(Poco::Timespan(0, 100'000), Poco::Net::Socket::SELECT_READ)) + state = DB::getSocketState(pair.getClient()); + } + EXPECT_EQ(state, DB::SocketState::Closed); + EXPECT_EQ(ERR_peek_error(), 0UL); + + /// The diagnostic `SSL_peek` saw a fatal TLS error, so `shutdown` must close the transport + /// directly instead of attempting the forbidden `SSL_shutdown` operation. + EXPECT_NO_THROW(pair.getClient().shutdown()); + EXPECT_EQ(ERR_peek_error(), 0UL); +} + +#endif diff --git a/src/Common/tests/gtest_ssl_send_pending_data.cpp b/src/Common/tests/gtest_ssl_send_pending_data.cpp new file mode 100644 index 000000000000..20a26adc9613 --- /dev/null +++ b/src/Common/tests/gtest_ssl_send_pending_data.cpp @@ -0,0 +1,177 @@ +#include "config.h" + +#if USE_SSL + +#include + +#include + +#include + +#include +#include +#include +#include +#include +#include + +#include +#include +#include + +#include +#include +#include +#include +#include + + +namespace +{ + +/// This custom `BIO` makes every TLS write fail as if the underlying socket was reset. +/// It is installed as the write `BIO` only, so the real read `BIO` and buffered application data remain intact. +int resetWrite(BIO * bio, const char *, int) +{ + BIO_clear_retry_flags(bio); + errno = ECONNRESET; + return -1; +} + +long resetCtrl(BIO *, int command, long, void *) // NOLINT(google-runtime-int) +{ + if (command == BIO_CTRL_FLUSH) + return 1; + return 0; +} + +int resetCreate(BIO * bio) +{ + BIO_set_init(bio, 1); + BIO_set_data(bio, nullptr); + return 1; +} + +int resetDestroy(BIO *) +{ + return 1; +} + +const BIO_METHOD * resetWriteBioMethod() +{ + /// `SSL_set0_wbio` transfers ownership of the `BIO` to `SSL`, so the method must support its full lifetime. + /// A successful flush also lets OpenSSL perform routine `BIO` housekeeping without obscuring the write failure. + static const BIO_METHOD * method = [] + { + BIO_METHOD * result = BIO_meth_new(BIO_get_new_index() | BIO_TYPE_SOURCE_SINK, "reset-write"); + BIO_meth_set_write(result, resetWrite); + BIO_meth_set_ctrl(result, resetCtrl); + BIO_meth_set_create(result, resetCreate); + BIO_meth_set_destroy(result, resetDestroy); + return result; + }(); + return method; +} + +} + + +TEST(SSLSocketError, SendFailureDoesNotConsumePendingApplicationData) +{ + EphemeralCert cert; + auto server_context = cert.makeContext(Poco::Net::Context::SERVER_USE); + auto client_context = cert.makeContext(Poco::Net::Context::CLIENT_USE); + + Poco::Net::SecureServerSocket server_socket(Poco::Net::SocketAddress("127.0.0.1", 0), 1, server_context); + const Poco::Net::SocketAddress server_address("127.0.0.1", server_socket.address().port()); + const Poco::Timespan timeout(5, 0); + + /// Establish TCP before starting the server thread. The lazy TLS handshake is completed below, + /// concurrently with the first server-side write. + Poco::Net::SecureStreamSocket client(client_context); + client.setLazyHandshake(true); + client.connect(server_address); + client.setSendTimeout(timeout); + client.setReceiveTimeout(timeout); + + constexpr std::array payload{'p', 'e', 'n', 'd', 'i', 'n', 'g', '!'}; + std::exception_ptr server_exception; + std::jthread server_thread([&] + { + try + { + auto accepted = server_socket.acceptConnection(); + accepted.setSendTimeout(timeout); + accepted.setReceiveTimeout(timeout); + if (accepted.sendBytes(payload.data(), static_cast(payload.size())) != static_cast(payload.size())) + throw Poco::Net::NetException("Could not send the complete TLS test payload"); + } + catch (...) + { + server_exception = std::current_exception(); + } + }); + + auto * client_impl = static_cast(client.impl()); + ASSERT_EQ(client_impl->completeHandshake(), 1); + + SSL * ssl = client_impl->ssl(); + ASSERT_NE(ssl, nullptr); + + /// The failing write `BIO` installed below would also reject a TLS shutdown write. Mark shutdown as complete + /// during scope cleanup, while still letting `SSL` own and destroy the failing `BIO`. + SCOPE_EXIT(SSL_set_shutdown(ssl, SSL_SENT_SHUTDOWN | SSL_RECEIVED_SHUTDOWN)); + + /// `SSL_peek` processes the incoming TLS record without consuming its first application-data byte. + /// Because the server sends the payload in one call, the complete payload must then be visible in `SSL_pending`. + char first_byte = 0; + const int peek_result = SSL_peek(ssl, &first_byte, 1); + + /// Wait until the server has completed its single write and propagate any server-side exception. + server_thread.join(); + if (server_exception) + std::rethrow_exception(server_exception); + + ASSERT_EQ(peek_result, 1); + EXPECT_EQ(first_byte, payload.front()); + const int pending_before_send = SSL_pending(ssl); + ASSERT_EQ(pending_before_send, static_cast(payload.size())); + + /// Replace only the write `BIO`. The read `BIO` and application data already buffered by `SSL` are preserved, + /// and `SSL_set0_wbio` transfers ownership of the failing `BIO` to `SSL`. + BIO * reset_write_bio = BIO_new(resetWriteBioMethod()); + ASSERT_NE(reset_write_bio, nullptr); + SSL_set0_wbio(ssl, reset_write_bio); + + /// OpenSSL requires the current thread's error queue to be empty before TLS I/O so that `SSL_get_error` + /// classifies this `SSL_write` result rather than an earlier error. Check the exact exception type because + /// the removed workaround also changed a connection reset into an `SSLException`. + ERR_clear_error(); + errno = 0; + const char byte_to_send = 'x'; + try + { + client.sendBytes(&byte_to_send, 1); + ADD_FAILURE() << "Expected Poco::Net::ConnectionResetException"; + } + catch (const Poco::Net::ConnectionResetException & exception) + { + EXPECT_EQ(typeid(exception), typeid(Poco::Net::ConnectionResetException)); + } + catch (const Poco::Exception & exception) + { + ADD_FAILURE() << "Expected Poco::Net::ConnectionResetException, got " << exception.className(); + } + catch (...) + { + /// Ok: this catch turns a non-Poco exception into a test failure. + ADD_FAILURE() << "Expected Poco::Net::ConnectionResetException, got a non-Poco exception"; + } + + /// Previously `SecureSocketImpl::sendBytes` called `SSL_read` after this write failure, silently consuming + /// one pending application-data byte. Handling the write error must leave all incoming data untouched. + EXPECT_EQ(SSL_pending(ssl), pending_before_send); +} + + +#endif diff --git a/src/IO/SilkFiberStreamSocketImpl.h b/src/IO/SilkFiberStreamSocketImpl.h index efbeaf042de2..b6662336e4ef 100644 --- a/src/IO/SilkFiberStreamSocketImpl.h +++ b/src/IO/SilkFiberStreamSocketImpl.h @@ -24,11 +24,6 @@ class FiberStreamSocketImpl final : public Poco::Net::StreamSocketImpl void setBlocking(bool flag) override; bool supportsExternalPolling() const override { return false; } - bool getDontWait() const { return dont_wait; } - void setDontWait(bool flag) { dont_wait = flag; } - -private: - bool dont_wait = false; }; } diff --git a/src/IO/SilkSecureFiberStreamSocketImpl.cpp b/src/IO/SilkSecureFiberStreamSocketImpl.cpp index 871baaefc896..78416e37310a 100644 --- a/src/IO/SilkSecureFiberStreamSocketImpl.cpp +++ b/src/IO/SilkSecureFiberStreamSocketImpl.cpp @@ -8,7 +8,6 @@ #include #include -#include #include #include @@ -16,12 +15,10 @@ #include #include #include -#include #include #include #include -#include namespace Silk @@ -30,11 +27,6 @@ namespace Silk namespace { -uint64_t timeoutNs(const Poco::Timespan & timeout) -{ - return static_cast(timeout.totalMicroseconds()) * 1000ULL; -} - FiberStreamSocketImpl * getUnderlyingSocket(BIO * bio) { return static_cast(static_cast(BIO_get_data(bio))); @@ -45,68 +37,30 @@ int silkBioRead(BIO * bio, char * buf, int len) auto * socket_impl = getUnderlyingSocket(bio); const int fd = socket_impl->sockfd(); - if (socket_impl->getDontWait()) - { - /// Honor non-blocking mode instead of parking the caller on an io_uring read: a - /// non-blocking caller (e.g. `SSL_peek` from the pool's staleness probe) expects an - /// immediate `EAGAIN`/`SSL_ERROR_WANT_READ`, not a wait for `getReceiveTimeout()`. - BIO_clear_retry_flags(bio); - const ssize_t n = ::recv(fd, buf, len, MSG_DONTWAIT); - if (n < 0) - { - /// `recv` left the error in `errno`; the flag call below cannot touch it - /// (`BIO_set_retry_read` is a pure in-struct flag set), so it stays valid for the - /// caller through `return`. Read it into a local for the fatality test per the - /// errno rule in silk's docs/tls.md. - const int err = errno; - if (BIO_sock_non_fatal_error(err)) - BIO_set_retry_read(bio); - return -1; - } - - /// TODO(mstetsyuk): should be done at Silk level. - __msan_unpoison(buf, static_cast(n)); - - if (n == 0) - BIO_set_flags(bio, BIO_FLAGS_IN_EOF); - return static_cast(n); - } - - const uint64_t timeout_ns = timeoutNs(socket_impl->getReceiveTimeout()); - - uint64_t bytes_read = 0; - silk::FiberScheduler::IoFuture future; - iovec iov{buf, static_cast(len)}; - silk::FiberScheduler::read(fd, &iov, 1, 0, &bytes_read, &future); - - int r = timeout_ns > 0 - ? silk::FiberFuture::waitWithTimeout(&future, timeout_ns) - : future.wait(); - - if (r == ETIMEDOUT) - { - future.cancel(); - r = future.wait(); - if (r == ECANCELED) - r = ETIMEDOUT; - } - + /// Do not schedule a fiber-aware read or wait for its future here. This callback runs + /// inside an OpenSSL operation, and suspending the fiber could resume it on another + /// OS thread before `SSL_get_error` reads that thread's error queue. Use a non-blocking + /// syscall and report retry through the `BIO` flags instead. `SecureSocketImpl` saves + /// the OpenSSL result first, then `SecureSocketImpl::mustRetry` performs the timed, + /// fiber-aware wait through `pollImpl`. + /// https://docs.openssl.org/3.5/man3/SSL_get_error/ BIO_clear_retry_flags(bio); - - if (r == 0) + const ssize_t n = ::recv(fd, buf, len, MSG_DONTWAIT); + if (n < 0) { - /// TODO(mstetsyuk): should be done at Silk level. - __msan_unpoison(buf, bytes_read); - - if (bytes_read == 0) - BIO_set_flags(bio, BIO_FLAGS_IN_EOF); - return static_cast(bytes_read); + /// Capture `errno` immediately; fiber migration makes a later read invalid. + const int err = errno; + if (BIO_sock_non_fatal_error(err)) + BIO_set_retry_read(bio); + return -1; } - errno = r; - if (BIO_sock_non_fatal_error(r) || r == ETIMEDOUT) - BIO_set_retry_read(bio); - return -1; + /// TODO(mstetsyuk): should be done at Silk level. + __msan_unpoison(buf, static_cast(n)); + + if (n == 0) + BIO_set_flags(bio, BIO_FLAGS_IN_EOF); + return static_cast(n); } int silkBioWrite(BIO * bio, const char * buf, int len) @@ -114,52 +68,17 @@ int silkBioWrite(BIO * bio, const char * buf, int len) auto * socket_impl = getUnderlyingSocket(bio); const int fd = socket_impl->sockfd(); - if (socket_impl->getDontWait()) - { - /// See the matching branch in silkBioRead: honor non-blocking mode rather than parking - /// the caller in a fiber wait for getSendTimeout(). - BIO_clear_retry_flags(bio); - const ssize_t n = ::send(fd, buf, len, MSG_DONTWAIT | MSG_NOSIGNAL); - if (n < 0) - { - /// See the matching note in silkBioRead: `send` left the error in `errno`, the flag - /// call cannot clobber it, so it survives to the caller. - const int err = errno; - if (BIO_sock_non_fatal_error(err)) - BIO_set_retry_write(bio); - return -1; - } - return static_cast(n); - } - - const uint64_t timeout_ns = timeoutNs(socket_impl->getSendTimeout()); - - uint64_t bytes_written = 0; - silk::FiberScheduler::IoFuture future; - iovec iov{const_cast(buf), static_cast(len)}; - silk::FiberScheduler::write(fd, &iov, 1, 0, &bytes_written, &future); - - int r = timeout_ns > 0 - ? silk::FiberFuture::waitWithTimeout(&future, timeout_ns) - : future.wait(); - - if (r == ETIMEDOUT) + /// See `silkBioRead`: never suspend from inside an OpenSSL operation. + BIO_clear_retry_flags(bio); + const ssize_t n = ::send(fd, buf, len, MSG_DONTWAIT | MSG_NOSIGNAL); + if (n < 0) { - future.cancel(); - r = future.wait(); - if (r == ECANCELED) - r = ETIMEDOUT; + const int err = errno; + if (BIO_sock_non_fatal_error(err)) + BIO_set_retry_write(bio); + return -1; } - - BIO_clear_retry_flags(bio); - - if (r == 0) - return static_cast(bytes_written); - - errno = r; - if (BIO_sock_non_fatal_error(r) || r == ETIMEDOUT) - BIO_set_retry_write(bio); - return -1; + return static_cast(n); } long silkBioCtrl(BIO * bio, int cmd, [[maybe_unused]] long larg, void * parg) // NOLINT(google-runtime-int) @@ -259,22 +178,11 @@ SecureFiberStreamSocketImpl::SecureFiberStreamSocketImpl(Poco::Net::Context::Ptr SecureFiberStreamSocketImpl::SecureFiberStreamSocketImpl(FiberStreamSocketImpl * underlying_, Poco::Net::Context::Ptr context) : Poco::Net::SecureStreamSocketImpl(underlying_, context) - , underlying(underlying_) { setBioMethod(silkBioMethod()); setMutex(std::make_unique()); } -bool SecureFiberStreamSocketImpl::getDontWait() const -{ - return underlying->getDontWait(); -} - -void SecureFiberStreamSocketImpl::setDontWait(bool flag) -{ - underlying->setDontWait(flag); -} - bool SecureFiberStreamSocketImpl::pollImpl(Poco::Timespan & timeout, int mode) { uint32_t events = 0; diff --git a/src/IO/SilkSecureFiberStreamSocketImpl.h b/src/IO/SilkSecureFiberStreamSocketImpl.h index ccf8383413a0..1afd28d4a574 100644 --- a/src/IO/SilkSecureFiberStreamSocketImpl.h +++ b/src/IO/SilkSecureFiberStreamSocketImpl.h @@ -17,16 +17,11 @@ class SecureFiberStreamSocketImpl final : public Poco::Net::SecureStreamSocketIm public: explicit SecureFiberStreamSocketImpl(Poco::Net::Context::Ptr context); - bool getDontWait() const; - void setDontWait(bool flag); - bool pollImpl(Poco::Timespan & timeout, int mode) override; bool supportsExternalPolling() const override { return false; } private: SecureFiberStreamSocketImpl(FiberStreamSocketImpl * underlying_, Poco::Net::Context::Ptr context); - - FiberStreamSocketImpl * underlying; }; } diff --git a/src/IO/SocketPeerClosed.cpp b/src/IO/SocketPeerClosed.cpp index c82b138b7ede..af0017fa7494 100644 --- a/src/IO/SocketPeerClosed.cpp +++ b/src/IO/SocketPeerClosed.cpp @@ -44,7 +44,16 @@ SocketState getSocketState(int fd) #if USE_SSL -SocketState getSSLSocketState(ssl_st * ssl) +namespace +{ + +struct SSLSocketStateResult +{ + SocketState state; + bool fatal_error; +}; + +SSLSocketStateResult getSSLSocketStateImpl(ssl_st * ssl) { /// `SSL_peek` decrypts just enough of the pending records to tell real application data and /// harmless post-handshake messages (session tickets, `KeyUpdate`) apart from a `close_notify`. @@ -54,32 +63,47 @@ SocketState getSSLSocketState(ssl_st * ssl) ERR_clear_error(); char c = 0; int res = SSL_peek(ssl, &c, 1); + SSLSocketStateResult result{SocketState::Closed, false}; if (res > 0) - return SocketState::DataPending; /// Application data is waiting to be read; the peer is alive. - - switch (SSL_get_error(ssl, res)) { - case SSL_ERROR_WANT_READ: [[fallthrough]]; - case SSL_ERROR_WANT_WRITE: - /// `SSL_peek` found no complete application-data record, but that alone does not prove - /// the connection is idle: the bytes of a record that has only partially arrived (e.g. - /// the first fragment of a queued response) are buffered inside the SSL object too, and - /// look identical from here - both end in `SSL_ERROR_WANT_READ`. `SSL_has_pending` - /// reports on that internal buffer regardless of whether the record is complete, so a - /// session ticket / `KeyUpdate` that was fully consumed reads as idle (nothing left - /// buffered), while a partial record correctly reads as pending. - return SSL_has_pending(ssl) ? SocketState::DataPending : SocketState::Idle; - case SSL_ERROR_ZERO_RETURN: - return SocketState::Closed; /// The peer sent `close_notify`: an orderly TLS shutdown. - default: - /// A FIN without `close_notify` (`SSL_ERROR_SYSCALL`), a protocol error (`SSL_ERROR_SSL`), - /// or anything else: treat as closed/broken. - return SocketState::Closed; + result = {SocketState::DataPending, false}; /// Application data is waiting to be read; the peer is alive. + } + else + { + switch (SSL_get_error(ssl, res)) + { + case SSL_ERROR_WANT_READ: [[fallthrough]]; + case SSL_ERROR_WANT_WRITE: + /// `SSL_peek` found no complete application-data record, but that alone does not prove + /// the connection is idle: the bytes of a record that has only partially arrived (e.g. + /// the first fragment of a queued response) are buffered inside the SSL object too, and + /// look identical from here - both end in `SSL_ERROR_WANT_READ`. `SSL_has_pending` + /// reports on that internal buffer regardless of whether the record is complete, so a + /// session ticket / `KeyUpdate` that was fully consumed reads as idle (nothing left + /// buffered), while a partial record correctly reads as pending. + result = {SSL_has_pending(ssl) ? SocketState::DataPending : SocketState::Idle, false}; + break; + case SSL_ERROR_ZERO_RETURN: + result = {SocketState::Closed, false}; /// The peer sent `close_notify`: an orderly TLS shutdown. + break; + case SSL_ERROR_SYSCALL: [[fallthrough]]; + case SSL_ERROR_SSL: + /// A FIN without `close_notify` or a protocol error is fatal. OpenSSL forbids + /// `SSL_shutdown` afterwards. + result = {SocketState::Closed, true}; + break; + default: + /// Any other unexpected result is treated as closed/broken, but only + /// `SSL_ERROR_SYSCALL` and `SSL_ERROR_SSL` make the connection fatal. + result = {SocketState::Closed, false}; + break; + } } -} -namespace -{ + /// Do not leak errors from this diagnostic probe into subsequent operations on this thread. + ERR_clear_error(); + return result; +} /// Force the socket into non-blocking mode for the duration of a call, restoring the original /// mode afterwards, so that `SSL_peek` on an idle pooled connection can never block. @@ -90,13 +114,10 @@ class ScopedNonBlocking : socket_impl(socket_impl_) { #if USE_SILK - if (auto * fiber_socket_impl = dynamic_cast(&socket_impl)) - { - was_blocking = !fiber_socket_impl->getDontWait(); - if (was_blocking) - fiber_socket_impl->setDontWait(true); + /// The Silk TLS BIO is always non-blocking so that an OpenSSL operation cannot + /// suspend and migrate between the operation and `SSL_get_error`. + if (dynamic_cast(&socket_impl)) return; - } #endif was_blocking = socket_impl.getBlocking(); if (was_blocking) @@ -110,13 +131,6 @@ class ScopedNonBlocking try { -#if USE_SILK - if (auto * fiber_socket_impl = dynamic_cast(&socket_impl)) - { - fiber_socket_impl->setDontWait(false); - return; - } -#endif socket_impl.setBlocking(true); } catch (...) @@ -130,13 +144,17 @@ class ScopedNonBlocking private: Poco::Net::SocketImpl & socket_impl; - /// For regular (non-silk) socket: whether it was blocking before. - /// For silk socket: whether it was dont-wait before. + /// Whether a regular (non-Silk) socket was blocking before the probe. bool was_blocking = false; }; } +SocketState getSSLSocketState(ssl_st * ssl) +{ + return getSSLSocketStateImpl(ssl).state; +} + #endif SocketState getSocketState(const Poco::Net::StreamSocket & socket) @@ -149,7 +167,10 @@ SocketState getSocketState(const Poco::Net::StreamSocket & socket) if (auto * ssl = secure->ssl()) { ScopedNonBlocking non_blocking(*secure); - return getSSLSocketState(ssl); + auto result = getSSLSocketStateImpl(ssl); + if (result.fatal_error) + secure->markFatalError(); + return result.state; } } #endif diff --git a/src/IO/SocketPeerClosed.h b/src/IO/SocketPeerClosed.h index 0d50cc64e307..fb7ad792164f 100644 --- a/src/IO/SocketPeerClosed.h +++ b/src/IO/SocketPeerClosed.h @@ -52,8 +52,8 @@ SocketState getSocketState(const Poco::Net::StreamSocket & socket); #if USE_SSL /// TLS-aware core of the check, exposed for testing. The socket underlying `ssl` MUST be in -/// non-blocking mode - don't-wait mode for a fiber socket - (the caller guarantees this) so that -/// `SSL_peek` cannot block. +/// non-blocking mode (the caller guarantees this) so that `SSL_peek` cannot block. The Silk TLS +/// BIO is always non-blocking. SocketState getSSLSocketState(ssl_st * ssl); #endif diff --git a/src/IO/tests/gtest_silk_fiber_stream_socket.cpp b/src/IO/tests/gtest_silk_fiber_stream_socket.cpp index 3afc0f233ca4..79fc9dafca47 100644 --- a/src/IO/tests/gtest_silk_fiber_stream_socket.cpp +++ b/src/IO/tests/gtest_silk_fiber_stream_socket.cpp @@ -310,8 +310,7 @@ TYPED_TEST(SilkFiberSocketTest, ThrottlerLimitEnforced) } -/// Secure-only: the bug is TLS-specific (a blocking-only `silkBioRead`/`silkBioWrite` surfaces -/// through `SSL_peek`, not through a raw, BIO-less socket read). Reuses the +/// Secure-only tests for the TLS BIO and direct OpenSSL operations. Reuses the /// `SecurePolicy policy` member from the typed fixture rather than redeclaring it. using SilkFiberSecureSocketTest = SilkFiberSocketTest; @@ -345,15 +344,12 @@ TEST_F(SilkFiberSecureSocketTest, NonBlockingPeekDoesNotBlockOnIdleConnection) char pong[1] = {}; EXPECT_EQ(socket.receiveBytes(pong, sizeof(pong)), 1); - /// A long receive timeout. Pre-fix, `silkBioRead` has no non-blocking mode and always - /// parks the caller in a fiber wait up to this timeout, so a slow probe below proves - /// the bug; a fast one proves the fix. + /// A long receive timeout. The probe must remain non-blocking regardless of it. socket.setReceiveTimeout(Poco::Timespan(5, 0)); /// The actual production sequence (`DB::getSocketState(StreamSocket)`, the core of the - /// connection pool's staleness check in `HTTPConnectionPool.cpp`): it puts the silk - /// socket into don't-wait mode with `setDontWait` - `Socket::setBlocking(false)` is rejected - /// by silk sockets - and calls `SSL_peek`, which reaches OpenSSL's socket BIO, i.e. `silkBioRead`. + /// connection pool's staleness check in `HTTPConnectionPool.cpp`) calls `SSL_peek`, + /// which reaches the always non-blocking Silk TLS BIO. Stopwatch watch; *p->state = DB::getSocketState(socket); *p->elapsed_us = watch.elapsedMicroseconds(); @@ -378,17 +374,13 @@ TEST_F(SilkFiberSecureSocketTest, NonBlockingPeekDoesNotBlockOnIdleConnection) EXPECT_EQ(state, DB::SocketState::Idle); EXPECT_LT(elapsed_us, 500'000U) << "getSocketState() took " << elapsed_us - << "us: silkBioRead ignored non-blocking mode and blocked on the receive timeout instead " - "of returning EAGAIN immediately"; + << "us: the Silk TLS BIO blocked on the receive timeout instead of returning EAGAIN immediately"; } -/// The same bug at the raw level, without any ClickHouse helper: a plain `SSL_peek` on a -/// non-blocking TLS connection with no data pending must return `SSL_ERROR_WANT_READ` -/// immediately. This is exactly how a non-blocking consumer uses the socket - and the only way, -/// since silk sockets reject `Socket::setBlocking(false)`: the socket is put into don't-wait -/// mode with `setDontWait` and only `silkBioRead` ever observes it. Pre-fix, the BIO has no -/// non-blocking mode and parks the caller for the full receive timeout. +/// At the raw level, without a ClickHouse helper, a plain `SSL_peek` on an idle TLS connection +/// must return `SSL_ERROR_WANT_READ` immediately. Silk sockets reject `Socket::setBlocking(false)`, +/// so the BIO itself has to be non-blocking. TEST_F(SilkFiberSecureSocketTest, NonBlockingSslPeekReturnsWantReadImmediately) { auto listener = policy.makeListener(); @@ -419,11 +411,10 @@ TEST_F(SilkFiberSecureSocketTest, NonBlockingSslPeekReturnsWantReadImmediately) socket.setReceiveTimeout(Poco::Timespan(5, 0)); - /// Put the socket into don't-wait mode - what any non-blocking user must do here, - /// since `Socket::setBlocking(false)` throws on a silk socket. + /// The Silk TLS BIO is always non-blocking. OpenSSL operations return WANT_READ or + /// WANT_WRITE and `SecureSocketImpl` performs the fiber-aware wait outside OpenSSL. auto * secure = dynamic_cast(socket.impl()); SSL * ssl = secure->ssl(); - secure->setDontWait(true); char c = 0; ERR_clear_error(); @@ -432,7 +423,6 @@ TEST_F(SilkFiberSecureSocketTest, NonBlockingSslPeekReturnsWantReadImmediately) *p->ssl_error = SSL_get_error(ssl, res); *p->elapsed_us = watch.elapsedMicroseconds(); - secure->setDontWait(false); socket.close(); return 0; }, @@ -452,8 +442,8 @@ TEST_F(SilkFiberSecureSocketTest, NonBlockingSslPeekReturnsWantReadImmediately) EXPECT_EQ(ssl_error, SSL_ERROR_WANT_READ); EXPECT_LT(elapsed_us, 500'000U) << "SSL_peek() took " << elapsed_us - << "us on an idle non-blocking connection: silkBioRead ignored non-blocking mode and " - "blocked on the receive timeout instead of returning EAGAIN immediately"; + << "us on an idle connection: the Silk TLS BIO blocked on the receive timeout instead " + "of returning EAGAIN immediately"; } From 931083f201549c9a79eaa19446e1e0e7a86387ec Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 10:57:59 +0000 Subject: [PATCH 046/185] Backport #117429 to 26.8: Mask secrets in SHOW CREATE DATABASE of a Backup database --- src/Backups/BackupInfo.cpp | 6 +- src/Backups/BackupMetadataFinder.cpp | 8 +- src/Backups/registerBackupEngineS3.cpp | 15 +- src/Databases/DatabaseBackup.cpp | 17 +- src/Parsers/FunctionSecretArgumentsFinder.cpp | 264 +++++++++++++++++- src/Parsers/FunctionSecretArgumentsFinder.h | 21 ++ .../test_mask_sensitive_info/test.py | 56 ++++ ...3_explicit_url_named_secret_mask.reference | 56 +++- ...4510_s3_explicit_url_named_secret_mask.sql | 207 +++++++++++++- ...abase_backup_locator_secret_mask.reference | 14 + ...233_database_backup_locator_secret_mask.sh | 58 ++++ ...duplicate_definition_secret_mask.reference | 34 +++ ...estore_duplicate_definition_secret_mask.sh | 164 +++++++++++ 13 files changed, 889 insertions(+), 31 deletions(-) create mode 100644 tests/queries/0_stateless/05233_database_backup_locator_secret_mask.reference create mode 100755 tests/queries/0_stateless/05233_database_backup_locator_secret_mask.sh create mode 100644 tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.reference create mode 100755 tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.sh diff --git a/src/Backups/BackupInfo.cpp b/src/Backups/BackupInfo.cpp index 0357ab785981..755027d19075 100644 --- a/src/Backups/BackupInfo.cpp +++ b/src/Backups/BackupInfo.cpp @@ -286,7 +286,7 @@ BackupInfo BackupInfo::fromAST(const IAST & ast) { const auto * func = ast.as(); if (!func) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected function, got {}", ast.formatForErrorMessage()); + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected a function as the backup locator"); BackupInfo res; res.backup_engine_name = func->name; @@ -295,7 +295,7 @@ BackupInfo BackupInfo::fromAST(const IAST & ast) { const auto * list = func->arguments->as(); if (!list) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected list, got {}", func->arguments->formatForErrorMessage()); + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected an argument list in the backup locator"); size_t index = 0; if (!list->children.empty()) @@ -329,7 +329,7 @@ BackupInfo BackupInfo::fromAST(const IAST & ast) res.function_arg = elem; break; } - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected literal, got {}", elem->formatForErrorMessage()); + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected a literal argument in the backup locator"); } res.args.push_back(lit->value); } diff --git a/src/Backups/BackupMetadataFinder.cpp b/src/Backups/BackupMetadataFinder.cpp index c9b6fee3f362..cfc3c257784d 100644 --- a/src/Backups/BackupMetadataFinder.cpp +++ b/src/Backups/BackupMetadataFinder.cpp @@ -304,8 +304,8 @@ void BackupMetadataFinder::findTableInBackupImpl( ErrorCodes::CANNOT_RESTORE_TABLE, "Extracted two different create queries for the same {}: {} and {}", tableNameWithTypeToString(table_name.database, table_name.table, false), - table_info.create_table_query_str, - create_table_query_str); + table_info.create_table_query->formatForErrorMessage(), + create_table_query->formatForErrorMessage()); } } @@ -395,8 +395,8 @@ void BackupMetadataFinder::findDatabaseInBackupImpl( ErrorCodes::CANNOT_RESTORE_DATABASE, "Extracted two different create queries for the same database {}: {} and {}", backQuoteIfNeed(database_name), - database_info.create_database_query_str, - create_database_query_str); + database_info.create_database_query->formatForErrorMessage(), + create_database_query->formatForErrorMessage()); } database_info.create_database_query = create_database_query; diff --git a/src/Backups/registerBackupEngineS3.cpp b/src/Backups/registerBackupEngineS3.cpp index 4f3f9b075dee..feddeddd7887 100644 --- a/src/Backups/registerBackupEngineS3.cpp +++ b/src/Backups/registerBackupEngineS3.cpp @@ -304,8 +304,19 @@ void registerBackupEngineS3(BackupFactory & factory) /// copy: `BackupImpl::writeBackupMetadata` decides whether the base backup may take this /// backup's credentials by comparing the two locators as text, and a locator resolved on /// one side alone no longer matches the other written the same way. - if (!StorageS3Configuration::collectCredentials(params.backup_info.function_arg->clone(), auth_settings, params.context)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid argument: {}", params.backup_info.function_arg->formatForErrorMessage()); + /// The rejected argument is never echoed, so both failure modes report the same way: + /// a nested map like `headers(..)` is masked only by its enclosing formatter, so + /// formatting one alone prints its values, and evaluating one can throw quoting it. + try + { + if (!StorageS3Configuration::collectCredentials( + params.backup_info.function_arg->clone(), auth_settings, params.context)) + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid S3 extra credentials"); + } + catch (const Exception &) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid S3 extra credentials"); + } role_arn = std::move(auth_settings[S3AuthSetting::role_arn]); role_session_name = std::move(auth_settings[S3AuthSetting::role_session_name]); diff --git a/src/Databases/DatabaseBackup.cpp b/src/Databases/DatabaseBackup.cpp index c46a7a663b1f..97d063e5fa30 100644 --- a/src/Databases/DatabaseBackup.cpp +++ b/src/Databases/DatabaseBackup.cpp @@ -506,7 +506,16 @@ DatabaseBackup::Configuration parseArguments(ASTs engine_args, ContextPtr, bool DatabaseBackup::Configuration result; - result.database_name = checkAndGetLiteralArgument(engine_args[0], "database_name"); + /// `checkAndGetLiteralArgument` formats the argument it rejects, and a locator written in this + /// position would format its credentials in plaintext. + try + { + result.database_name = checkAndGetLiteralArgument(engine_args[0], "database_name"); + } + catch (const Exception &) + { + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Argument 'database_name' must be a string literal"); + } /** A locator held in a string literal (`Backup('db', 'File(\'backup.zip\')')`) is the form that * metadata rewritten by an older server carries, so it has to keep loading - a server that cannot @@ -525,9 +534,9 @@ DatabaseBackup::Configuration parseArguments(ASTs engine_args, ContextPtr, bool } } - /// `BackupInfo::fromAST` puts the offending argument into its message, and that message reaches the - /// error log and the `exception` column of `query_log`. A locator held in a string literal is exactly - /// the text that can carry credentials, so refuse it here without echoing it. + /// A locator held in a string literal is exactly the text that can carry credentials, and a rejection + /// message reaches the error log and the `exception` column of `query_log`, so name the accepted form + /// here rather than the text given. if (!engine_args[1]->as()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Expected function as the backup destination of a `Backup` database. It must be spelled as the " diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 8a6138d23552..95bfbc06b57e 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -26,6 +26,89 @@ namespace changed |= maskPresignedURLParameters(url); return changed; } + + /// How an Azure destination reads a connection value, and whether the masking here can show it. + enum class AzureConnectionValue + { + /// An http(s) scheme and a host, with no userinfo, query or fragment: each of those carries a + /// credential of its own (`http://user:key@host`, a SAS `?sig=`). + PlainStorageAccountURL, + /// A connection string, whose secret keys `maskAzureConnectionString` masks in place. + ConnectionString, + /// A value that can carry a credential no rule here masks. + Unmaskable, + }; + + AzureConnectionValue classifyAzureConnectionValue(const String & value) + { + static constexpr std::string_view SEPARATOR = "://"; + const size_t separator = value.find(SEPARATOR); + const std::string_view scheme = std::string_view(value).substr(0, std::min(separator, value.length())); + /// The scheme grammar `maskURIUserinfo` reads. A connection string does not match it, even when + /// one of its values embeds an endpoint URL. + const bool is_url = separator != String::npos && !scheme.empty() && isAlphaASCII(scheme.front()) + && std::all_of( + scheme.begin(), scheme.end(), [](char c) { return isAlphaNumericASCII(c) || c == '+' || c == '.' || c == '-'; }); + /// `maskAzureConnectionString` masks nothing in a value starting with `http`, so one that is no + /// URL either would be left as written. + if (!is_url) + return value.starts_with("http") ? AzureConnectionValue::Unmaskable : AzureConnectionValue::ConnectionString; + + if ((!equalsCaseInsensitive(scheme, "http") && !equalsCaseInsensitive(scheme, "https")) + || value.find_first_of("?#") != String::npos) + return AzureConnectionValue::Unmaskable; + + const size_t authority_begin = separator + SEPARATOR.length(); + const size_t authority_end = std::min(value.find('/', authority_begin), value.length()); + if (authority_end == authority_begin || value.find('@', authority_begin) < authority_end) + return AzureConnectionValue::Unmaskable; + return AzureConnectionValue::PlainStorageAccountURL; + } + + /// The backup engines whose locator names a destination with no credential in it, each with the + /// argument count it accepts. `BackupFactory` registers exactly these plus `S3` and `AzureBlobStorage`. + std::optional credentialFreeBackupEngineArity(const String & engine_name) + { + if (engine_name == "File" || engine_name == "Memory") + return 1; + if (engine_name == "Disk") + return 2; + if (engine_name == "Null") + return 0; + return {}; + } +} + +void FunctionSecretArgumentsFinder::maskEveryArgument() +{ + for (size_t i = 0, size = function->arguments->size(); i < size; ++i) + markSecretArgument(i); +} + +bool FunctionSecretArgumentsFinder::hasOnlyLiteralArguments(const AbstractFunction & function) +{ + if (!function.hasArguments()) + return true; + for (size_t i = 0, size = function.arguments->size(); i < size; ++i) + if (!function.arguments->at(i)->tryGetLiteralText(nullptr)) + return false; + return true; +} + +bool FunctionSecretArgumentsFinder::isCredentialFreeBackupLocator(const AbstractFunction & function) +{ + auto arity = credentialFreeBackupEngineArity(function.name()); + if (!arity) + return false; + const size_t count = function.hasArguments() ? function.arguments->size() : 0; + if (count != *arity) + return false; + /// Each of these reads every argument of its own as a string, so another shape - an array among + /// them - is read by none of them and can carry a string of its own. + for (size_t i = 0; i < count; ++i) + if (!tryGetStringFromArgument(*function.arguments->at(i), nullptr, /* allow_identifier= */ false)) + return false; + return true; } void FunctionSecretArgumentsFinder::markSecretArgument(size_t index, bool argument_is_named) @@ -599,6 +682,54 @@ bool FunctionSecretArgumentsFinder::maskAzureConnectionString(ssize_t url_arg_id return false; } +bool FunctionSecretArgumentsFinder::azureCollectionArgumentsAreShowable(size_t start, size_t positional_limit) +{ + size_t positionals = 0; + for (size_t i = start, size = function->arguments->size(); i < size; ++i) + { + const auto argument_function = function->arguments->at(i)->getFunction(); + if (argument_function && argument_function->name() == "equals") + { + /// A key this rule cannot read hides which credential the override carries; a value that is + /// no plain literal or identifier can nest one (`headers('Authorization' = '...')`). + if (argument_function->arguments && argument_function->arguments->size() == 2 + && tryGetStringFromArgument(*argument_function->arguments->at(0), nullptr) + && (tryGetStringFromArgument(*argument_function->arguments->at(1), nullptr) + || argument_function->arguments->at(1)->tryGetLiteralText(nullptr))) + continue; + return false; + } + if (++positionals > positional_limit || !function->arguments->at(i)->tryGetLiteralText(nullptr)) + return false; + } + + /// A destination reads at most one of the two mutually exclusive connection keys, and rejects a + /// second one only after the statement has been formatted, so a surplus one stays as written. + size_t connection_overrides = 0; + for (const auto & key : {"connection_string", "storage_account_url"}) + for (ssize_t i = findNamedArgument(nullptr, key, start); i >= 0; + i = findNamedArgument(nullptr, key, static_cast(i) + 1)) + ++connection_overrides; + + if (connection_overrides > 1) + return false; + + for (const auto & key : {"connection_string", "storage_account_url"}) + { + String value; + if (findNamedArgument(&value, key, start) < 0) + continue; + /// Hiding a connection string replaces its whole argument, which cannot be combined with + /// hiding `account_key`. + const auto shape = classifyAzureConnectionValue(value); + if (value.empty() || shape == AzureConnectionValue::Unmaskable + || (shape == AzureConnectionValue::ConnectionString + && findNamedArgument(nullptr, "account_key", start) >= 0)) + return false; + } + return true; +} + void FunctionSecretArgumentsFinder::findURLSecretArguments(size_t url_offset) { /// `headers(...)` can appear at any position in every url form (function, cluster function, engine, @@ -966,31 +1097,52 @@ void FunctionSecretArgumentsFinder::findAzureBlobStorageTableEngineSecretArgumen if (isNamedCollectionName(url_arg_idx)) { /// AzureBlobStorage(named_collection, ..., account_key = 'account_key', ...) + if (!azureCollectionArgumentsAreShowable(url_arg_idx + 1, /* positional_limit= */ 0)) + { + maskEveryArgument(); + return; + } if (maskAzureConnectionString(-1, true, 1)) return; findSecretNamedArgument("account_key", 1); return; } - if (maskAzureConnectionString(url_arg_idx)) - return; - /// We should check other arguments first because we don't need to do any replacement in case of /// AzureBlobStorage(connection_string|storage_account_url, container_name, blobpath, format) -- in this case there is no account_key argument size_t count = function->arguments->size(); + bool fourth_argument_is_format = false; if ((url_arg_idx + 4 <= count) && (count <= url_arg_idx + 7)) { String fourth_arg; if (tryGetStringFromArgument(url_arg_idx + 3, &fourth_arg)) - { - if (fourth_arg == "auto" || KnownFormatNames::instance().exists(fourth_arg)) - return; - } + fourth_argument_is_format = fourth_arg == "auto" || KnownFormatNames::instance().exists(fourth_arg); + } + /// Which argument holds a credential: the two-argument shape takes a shared access signature beside + /// the url (`endpoint.sas_auth`), the longer ones an `account_key` - unless the fourth names a format. + std::optional credential_arg_idx; + if (count == url_arg_idx + 2) + credential_arg_idx = url_arg_idx + 1; + else if (!fourth_argument_is_format && (url_arg_idx + 4 < count)) + credential_arg_idx = url_arg_idx + 4; + + /// The engine reads this argument as a connection string or as a plain account url; a value of + /// another shape is read by neither rule below, and a hidden connection string replaces it whole. + String connection_value; + const auto shape = tryGetStringFromArgument(url_arg_idx, &connection_value) + ? classifyAzureConnectionValue(connection_value) + : AzureConnectionValue::Unmaskable; + if (shape == AzureConnectionValue::Unmaskable || (shape == AzureConnectionValue::ConnectionString && credential_arg_idx)) + { + maskEveryArgument(); + return; } - /// We're going to replace 'account_key' with '[HIDDEN]' if account_key is used in the signature - if (url_arg_idx + 4 < count) - markSecretArgument(url_arg_idx + 4); + if (maskAzureConnectionString(url_arg_idx)) + return; + + if (credential_arg_idx) + markSecretArgument(*credential_arg_idx); } void FunctionSecretArgumentsFinder::findRedisFunctionSecretArguments() @@ -1175,8 +1327,13 @@ void FunctionSecretArgumentsFinder::findDataLakeCatalogSecretArguments() void FunctionSecretArgumentsFinder::findBackupDatabaseSecretArguments() { - if (function->arguments->size() < 2) + /// `Backup(database_name, locator)` is the only valid shape, a locator carrying credentials can be + /// written in either position, and the query is formatted for logging before validation rejects it. + if (function->arguments->size() != 2 || !function->arguments->at(0)->tryGetLiteralText(nullptr)) + { + maskEveryArgument(); return; + } auto storage_arg = function->arguments->at(1); auto storage_function = storage_arg->getFunction(); @@ -1200,7 +1357,26 @@ void FunctionSecretArgumentsFinder::findBackupDatabaseSecretArguments() /// Backup('', S3('url', 'access_key_id', 'secret_access_key' [, ...])) /// Backup('', S3(named_collection, ..., secret_access_key = '...', session_token = '...', ...)) /// by reconstructing the nested `S3(...)` with the secret arguments replaced by `[HIDDEN]`. - if (!storage_function || storage_function->name() != "S3" || !storage_function->hasArguments()) + if (storage_function->name() != "S3") + { + if (isCredentialFreeBackupLocator(*storage_function)) + return; + + /// Any other locator holds a credential no rule below reconstructs (`AzureBlobStorage` holds + /// `account_key` and connection-string material); its engine name and arity are not secrets. + std::string replacement = storage_function->name() + "("; + for (size_t i = 0, size = storage_function->hasArguments() ? storage_function->arguments->size() : 0; i < size; ++i) + replacement += i > 0 ? ", '[HIDDEN]'" : "'[HIDDEN]'"; + replacement += ")"; + + result.start = 1; + result.count = 1; + result.replacement = std::move(replacement); + result.quote_replacement = false; + return; + } + + if (!storage_function->hasArguments()) return; const auto & nested_args = *storage_function->arguments; @@ -1383,10 +1559,70 @@ void FunctionSecretArgumentsFinder::findBackupNameSecretArguments() maskS3UrlArgument(positional, 0); maskS3PositionalsFrom(positional, positional.size() == 3 ? 2 : 1); } - else if (engine_name == "AzureBlobStorage" || engine_name == "AzureQueue") + else if (engine_name == "AzureBlobStorage") { - findAzureBlobStorageTableEngineSecretArguments(); + findAzureBlobStorageBackupSecretArguments(); + } + else if (!isCredentialFreeBackupLocator(*function)) + { + /// Everything else either is an engine no rule here reconstructs, or has arguments the named + /// engine does not read (an override, a nested map, a surplus slot), which can carry a credential. + /// `AzureQueue` reaches this branch: it is a table engine, not a registered backup engine. + maskEveryArgument(); + } +} + +void FunctionSecretArgumentsFinder::findAzureBlobStorageBackupSecretArguments() +{ + /// The destination reads AzureBlobStorage(named_collection [, 'filename'] [, key = value, ...]), + /// ('connection_string|storage_account_url', 'container', 'path'), or those three followed by + /// ('account_name', 'account_key'). An argument no shape reads holds whatever was written in it. + const size_t count = function->arguments->size(); + + if (isNamedCollectionName(0)) + { + if (!azureCollectionArgumentsAreShowable(1, /* positional_limit= */ 1)) + { + maskEveryArgument(); + return; + } + if (maskAzureConnectionString(-1, /* argument_is_named= */ true, 1)) + return; + findSecretNamedArgument("account_key", 1); + return; + } + + if ((count != 3 && count != 5) || !hasOnlyLiteralArguments(*function)) + { + maskEveryArgument(); + return; } + + if (count == 3) + { + /// Only this shape accepts a connection string, which can embed `AccountKey`. A value that is no + /// string is read by neither the classification below nor the destination. + String connection_value; + if (!tryGetStringFromArgument(0, &connection_value) + || classifyAzureConnectionValue(connection_value) == AzureConnectionValue::Unmaskable) + { + maskEveryArgument(); + return; + } + maskAzureConnectionString(0); + return; + } + + String storage_account_url; + if (!tryGetStringFromArgument(0, &storage_account_url) + || classifyAzureConnectionValue(storage_account_url) != AzureConnectionValue::PlainStorageAccountURL) + { + /// This shape requires a plain account URL. A connection string here can only be hidden whole, + /// which cannot be combined with hiding `account_key`. + maskEveryArgument(); + return; + } + markSecretArgument(4); } bool FunctionSecretArgumentsFinder::isNamedCollectionName(size_t arg_idx) const diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index 10342f18cc6b..ee6b4922923b 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -171,6 +171,11 @@ class FunctionSecretArgumentsFinder void findS3FunctionSecretArguments(bool is_cluster_function); void findAzureBlobStorageFunctionSecretArguments(bool is_cluster_function); bool maskAzureConnectionString(ssize_t url_arg_idx, bool argument_is_named = false, size_t start = 0); + /// Whether the arguments an `AzureBlobStorage(named_collection, ...)` destination or table takes + /// from `start` can be shown: only an argument written here can carry a credential, and each has + /// to be readable enough to tell that it does not. `positional_limit` bounds the plain literals + /// read beside the overrides: one filename for a backup locator, none for a table engine. + bool azureCollectionArgumentsAreShowable(size_t start, size_t positional_limit); /// Masks the secrets of every URL form (`url`/`urlCluster` table functions, the `URL` table /// engine, and their named-collection variants): the userinfo password of the url positional or a /// named `url = ...` override, and the `headers(...)` values at any position. `url` is at @@ -180,6 +185,18 @@ class FunctionSecretArgumentsFinder bool tryGetStringFromArgument(size_t arg_idx, String * res, bool allow_identifier = true) const; static bool tryGetStringFromArgument(const AbstractFunction::Argument & argument, String * res, bool allow_identifier = true); + /// `BackupInfo` keeps named overrides and a trailing map for every backup engine, including the + /// ones that read neither, so an argument that is not a plain literal can carry a credential. + static bool hasOnlyLiteralArguments(const AbstractFunction & function); + + /// Whether a backup locator names its destination with exactly the literal arguments its engine + /// accepts, and therefore holds no credential. An engine that takes fewer rejects the rest only + /// after the statement has been formatted for logging, so the count has to be checked here too. + static bool isCredentialFreeBackupLocator(const AbstractFunction & function); + + /// Hides every argument, for a shape whose valid slots cannot be established. + void maskEveryArgument(); + void findRemoteFunctionSecretArguments(); /// Tries to get either a database name or a qualified table name from an argument. @@ -207,6 +224,10 @@ class FunctionSecretArgumentsFinder void findBackupDatabaseSecretArguments(); void findBackupNameSecretArguments(); + /// A backup destination reads a different signature than the table engine of the same name, so the + /// table-engine rule leaves an argument it does not model visible. + void findAzureBlobStorageBackupSecretArguments(); + /// Whether a specified argument can be the name of a named collection? bool isNamedCollectionName(size_t arg_idx) const; diff --git a/tests/integration/test_mask_sensitive_info/test.py b/tests/integration/test_mask_sensitive_info/test.py index d16d8f5623f7..4ea3878add59 100644 --- a/tests/integration/test_mask_sensitive_info/test.py +++ b/tests/integration/test_mask_sensitive_info/test.py @@ -1058,6 +1058,62 @@ def test_database_backup_engine_s3(): node.query("DROP DATABASE IF EXISTS backup_db_s3_2") +def test_database_backup_engine_azure_display_surfaces(): + """A live `Backup` database over an `AzureBlobStorage` locator: `account_key` must not reach + `SHOW CREATE DATABASE` or `system.databases.engine_full`. Only `S3` locators are reconstructed + argument by argument, so an Azure one keeps its engine name and arity and hides every argument.""" + azure_storage_account_url = cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"] + azure_account_name = "devstoreaccount1" + azure_account_key = "Eby8vdM02xNOcqFlqUwJPLlmEtlCDXJ1OUzFT50uSRZ6IFsuFq2UVErCz4I6tq/K1SZFPTOtr/KBHBeksoGMGw==" + + # A backup destination is write-once and the container is emptied only when the cluster starts, + # so give every run its own blob path: a repeat inside one session hits BACKUP_ALREADY_EXISTS. + blob = "db_backup_azure_surfaces_" + "".join( + random.choice(string.ascii_lowercase) for _ in range(8) + ) + locator = ( + f"AzureBlobStorage('{azure_storage_account_url}', '{cluster.azurite_container}', " + f"'{blob}', '{azure_account_name}', '{azure_account_key}')" + ) + masked_locator = ( + "AzureBlobStorage('[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]')" + ) + surfaces = [ + "SHOW CREATE DATABASE backup_db_azure_view", + "SELECT engine_full FROM system.databases WHERE name = 'backup_db_azure_view'", + ] + + node.query("DROP DATABASE IF EXISTS backup_db_azure_src SYNC") + node.query("DROP DATABASE IF EXISTS backup_db_azure_view SYNC") + node.query("CREATE DATABASE backup_db_azure_src") + node.query( + "CREATE TABLE backup_db_azure_src.t (x int) ENGINE = MergeTree ORDER BY x" + ) + node.query("INSERT INTO backup_db_azure_src.t SELECT * FROM numbers(10)") + node.query(f"BACKUP DATABASE backup_db_azure_src TO {locator} FORMAT Null") + node.query( + f"CREATE DATABASE backup_db_azure_view ENGINE = Backup('backup_db_azure_src', {locator})" + ) + + # TSVRaw so the locator is compared as written rather than through TSV escaping. + for surface in surfaces: + shown = node.query(f"{surface} FORMAT TSVRaw") + assert masked_locator in shown, shown + assert azure_account_key not in shown, shown + + # The key is in the locator, so its absence above is masking rather than an argument that the + # regenerated definition never carried. + for surface in surfaces: + shown = node.query(f"{surface} {show_secrets}=1 FORMAT TSVRaw") + assert azure_account_key in shown, shown + + # The masked surfaces belong to a working database, not to one that failed to open. + assert node.query("SELECT count() FROM backup_db_azure_view.t") == "10\n" + + node.query("DROP DATABASE IF EXISTS backup_db_azure_view SYNC") + node.query("DROP DATABASE IF EXISTS backup_db_azure_src SYNC") + + def test_backup_table_azure_named_collection(): """Test that secrets in Azure named collection backups are masked in system.backups and logs.""" azure_storage_account_url = cluster.env_variables["AZURITE_STORAGE_ACCOUNT_URL"] diff --git a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference index 030f5c5e6031..0b581d50ec4a 100644 --- a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference +++ b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference @@ -197,6 +197,46 @@ BACKUP TABLE nonexistent_04510 TO S3(\'url_bkp_pos\', \'[HIDDEN]\', \'[HIDDEN]\' BACKUP TABLE nonexistent_04510 TO S3(nc_bkp_missing, \'visible_bkp_dir\', \'[HIDDEN]\') BACKUP TABLE nonexistent_04510 TO S3(nc_bkporder_missing, secret_access_key = \'[HIDDEN]\', \'visible_bkp_dir2\') BACKUP TABLE nonexistent_04510 TO S3(\'url_bkp_mixed\', equals(access_key_id, \'ak\'), \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO File(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO Disk(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO Memory(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO File(\'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO `Null`(\'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO Foo(\'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO Null(); +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureQueue(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=[HIDDEN];\', \'cont\', \'visible_04510_dir/b.zip\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'http://localhost:11111/acct\', \'visible_04510_cont\', \'visible_04510_dir/b.zip\'); +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, connection_string = \'DefaultEndpointsProtocol=https;AccountName=visible_04510_acct;AccountKey=[HIDDEN];\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, account_key = \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, equals(storage_account_url, \'http://localhost:11111/visible_04510_url\'), account_key = \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, container = \'visible_04510_c1\', container = \'visible_04510_c2\'); +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(\'http://localhost:11111/visible_04510_acct5\', \'visible_04510_cont5\', \'visible_04510_blob5\', \'visible_04510_acctname5\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte1 (`x` UInt8) ENGINE = AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte2 (`x` UInt8) ENGINE = AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte3 (`x` UInt8) ENGINE = AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte4 (`x` UInt8) ENGINE = AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte5 (`x` UInt8) ENGINE = AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte6 (`x` UInt8) ENGINE = AzureBlobStorage(\'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=[HIDDEN];\', \'visible_04510_teco\', \'visible_04510_teblob\') +CREATE TABLE t_04510_azte7 (`x` UInt8) ENGINE = AzureBlobStorage(\'http://localhost:11111/visible_04510_teacct\', \'visible_04510_tec5\', \'visible_04510_teb5\', \'visible_04510_teacctname\', \'[HIDDEN]\') +CREATE TABLE t_04510_azte8 (`x` UInt8) ENGINE = AzureQueue(\'http://localhost:11111/visible_04510_teq/cont/*\', \'[HIDDEN]\') SETTINGS mode = \'unordered\' CREATE DATABASE db_04510_ec ENGINE = Backup(\'\', S3(\'url_dbec\', \'ak\', \'[HIDDEN]\', extra_credentials(external_id = \'[HIDDEN]\'))) CREATE DATABASE db_04510_postok ENGINE = Backup(\'\', S3(\'url_dbpostok\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\')) CREATE DATABASE db_04510_ncpos ENGINE = Backup(\'\', S3(nc_dbnc_missing, \'visible_dbnc_dir\', \'[HIDDEN]\')) @@ -208,6 +248,20 @@ CREATE DATABASE db_04510_ncurl ENGINE = Backup(\'\', S3(nc_dburl_missing, url = CREATE DATABASE db_04510_hdr ENGINE = Backup(\'\', S3(\'url_dbhdr\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\')) CREATE DATABASE db_04510_expr ENGINE = Backup(\'\', S3(\'url_dbexpr\', \'ak\', \'[HIDDEN]\', \'[HIDDEN]\')) CREATE DATABASE db_04510_quoted ENGINE = Backup(\'\', \'[HIDDEN]\') +CREATE DATABASE db_04510_tail ENGINE = Backup(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +CREATE DATABASE db_04510_tmap ENGINE = Backup(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\') +CREATE DATABASE db_04510_lone ENGINE = Backup(\'[HIDDEN]\') +CREATE DATABASE db_04510_swap ENGINE = Backup(\'[HIDDEN]\', \'[HIDDEN]\') +CREATE DATABASE db_04510_ident ENGINE = Backup(\'[HIDDEN]\', \'[HIDDEN]\') +CREATE DATABASE db_04510_ftail ENGINE = Backup(\'src_04510\', File(\'[HIDDEN]\', \'[HIDDEN]\')) +CREATE DATABASE db_04510_fover ENGINE = Backup(\'src_04510\', File(\'[HIDDEN]\', \'[HIDDEN]\')) +CREATE DATABASE db_04510_dover ENGINE = Backup(\'src_04510\', Disk(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\')) +CREATE DATABASE db_04510_mover ENGINE = Backup(\'src_04510\', Memory(\'[HIDDEN]\', \'[HIDDEN]\')) +CREATE DATABASE db_04510_fvalid ENGINE = Backup(\'src_04510\', File(\'nonexistent_04510\')); +CREATE DATABASE db_04510_dvalid ENGINE = Backup(\'src_04510\', Disk(\'backups\', \'nonexistent_04510\')); +CREATE DATABASE db_04510_eval ENGINE = Backup(\'\', S3(\'http://localhost:11111/test/04510eval\', \'ak\', \'[HIDDEN]\', extra_credentials(external_id = \'[HIDDEN]\'))) +CREATE DATABASE db_04510_nonlit ENGINE = Backup(\'\', S3(\'[HIDDEN]\', \'[HIDDEN]\')) +CREATE DATABASE db_04510_azure ENGINE = Backup(\'\', AzureBlobStorage(\'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\', \'[HIDDEN]\')) CREATE DATABASE db_04510_s3pos ENGINE = S3(\'url_dbs3pos\', \'ak\', \'[HIDDEN]\', \'[HIDDEN]\') DROP DATABASE IF EXISTS default_1 CREATE DATABASE default_1 ENGINE = S3(\'url_dbenv\', \'ak\', \'[HIDDEN]\', use_environment_credentials = 1) @@ -219,4 +273,4 @@ EXPLAIN QUERY TREE run_passes = 0 SELECT * FROM s3(\'http://localhost:11111/test EXPLAIN QUERY TREE run_passes = 0 SELECT * FROM s3(\'http://localhost:11111/test/04510qt\', NOSIGN, \'TSV\', \'x UInt8\', headers(\'Authorization\' = \'[HIDDEN]\')) EXPLAIN QUERY TREE run_passes = 0 SELECT * FROM s3(nc_04510_missing, url = \'https://[HIDDEN]@localhost:11111/test/04510qt?X-Amz-Signature=[HIDDEN]\', structure = \'x UInt8\') EXPLAIN QUERY TREE run_passes = 0 SELECT * FROM s3(\'http://localhost:11111/test/04510qt\', \'ak\', \'[HIDDEN]\', \'TSV\', \'x UInt8\') UNION ALL SELECT 1 -1 0 +1 0 0 diff --git a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql index 569086cf03b2..f3258d0805df 100644 --- a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql +++ b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql @@ -234,6 +234,138 @@ BACKUP TABLE nonexistent_04510 TO S3(nc_bkporder_missing, BACKUP TABLE nonexistent_04510 TO S3('url_bkp_mixed', access_key_id = 'ak', 'SEKRIT_BKPMIX'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +-- The credential-free engines (Disk, File, Memory, Null) are credential-free only in the shape they +-- read: File and Memory take one argument, Disk two, Null none. A surplus argument is rejected after +-- the statement is logged, so it reaches the log holding whatever was written in it, an argument that is +-- no string is read by none of these engines and can carry a string of its own, and an engine name that +-- is not registered at all has no known shape to trust; all three must be hidden. The last statement is +-- the control: the shape Null does read carries nothing to hide, so it stays visible verbatim and gets +-- as far as resolving the table. +BACKUP TABLE nonexistent_04510 TO File('nonexistent_04510', + 'SEKRIT_TOFILEOVER'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +BACKUP TABLE nonexistent_04510 TO Disk('backups', 'nonexistent_04510', + 'SEKRIT_TODISKOVER'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +BACKUP TABLE nonexistent_04510 TO Memory('nonexistent_04510', + 'SEKRIT_TOMEMOVER'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +BACKUP TABLE nonexistent_04510 TO File(['SEKRIT_TOFILEARR']); -- { serverError BAD_GET } +BACKUP TABLE nonexistent_04510 TO Null('SEKRIT_TONULLOVER'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +BACKUP TABLE nonexistent_04510 TO Foo('SEKRIT_TOUNKNOWN'); -- { serverError BACKUP_ENGINE_NOT_FOUND } +BACKUP TABLE nonexistent_04510 TO Null(); -- { serverError UNKNOWN_TABLE } + +-- The AzureBlobStorage backup destination reads a different signature than the table engine of the same +-- name: a named collection with an optional filename, three arguments (connection string or account url, +-- container, path), or five (adding account_name and account_key). An argument outside those shapes is +-- rejected only after the statement is logged, and AzureQueue has no backup engine at all. The last +-- two statements are the controls: a connection string hides its AccountKey, and the three-argument +-- shape has nothing to hide, so it stays visible verbatim. +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('http://localhost:11111/acct', 'cont', 'blob', + 'SEKRIT_AZTO4'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, 'dir', + 'SEKRIT_AZTONCPOS'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureQueue('http://localhost:11111/acct', 'cont', 'blob', + 'SEKRIT_AZQTO'); -- { serverError BACKUP_ENGINE_NOT_FOUND } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_AZTOCSKEY==;', + 'cont', 'blob', 'acct', 'SEKRIT_AZTOCS5'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('DefaultEndpointsProtocol=https;AccountName=a;AccountKey=c2VrcmV0Cg==;', + 'cont', 'visible_04510_dir/b.zip'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('http://localhost:11111/acct', 'visible_04510_cont', 'visible_04510_dir/b.zip'); -- { serverError BAD_ARGUMENTS } + +-- A named collection can be overridden per statement, and the destination evaluates those overrides as +-- constant expressions. An override this rule cannot read may hold either credential, and hiding a +-- connection string replaces the whole argument, which cannot be combined with hiding account_key, so +-- both shapes hide the locator whole. The last three statements are the controls: a connection string +-- alone still hides only its AccountKey, account_key alone is hidden by itself, and an account url +-- override does not take the replacement path, so it stays visible next to a hidden account_key. +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + connection_string = concat('DefaultEndpointsProtocol=https;AccountName=a;AccountKey=', + 'SEKRIT_AZNCEXPR')); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + connection_string = 'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=c2VrcmV0Cg==;', + account_key = 'SEKRIT_AZNCBOTH'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + storage_account_url = concat('https://a.blob.core.windows.net/c?sig=', + 'SEKRIT_AZNCURLEXPR')); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + connection_string = 'DefaultEndpointsProtocol=https;AccountName=visible_04510_acct;AccountKey=SEKRIT_AZNCCS==;'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + account_key = 'SEKRIT_AZNCKEY'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + storage_account_url = 'http://localhost:11111/visible_04510_url', + account_key = 'SEKRIT_AZNCURLKEY'); -- { serverError BAD_ARGUMENTS } + +-- An override key can be an expression too, and one this rule cannot read hides which credential the +-- value is: both the computed key and a malformed override hide the locator whole, as the S3 form above +-- already does for a key it cannot read. +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + concat('account_', 'key') = 'SEKRIT_AZNCKEYEXPR'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + equals('account_key', 'SEKRIT_AZNCEQ3', 'surplus')); -- { serverError BAD_ARGUMENTS } + +-- A readable key does not make the override readable: the destination evaluates its value as a constant +-- expression, so a value that is no plain literal or identifier can nest a credential of its own and is +-- formatted verbatim before that evaluation rejects it, as the S3 form above already is for such a value. +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + container = headers('Authorization' = 'SEKRIT_AZNCHDR')); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + blob_path = concat('SEKRIT_AZNCPATHEXPR', '/b.zip')); -- { serverError BAD_ARGUMENTS } + +-- connection_string and storage_account_url are mutually exclusive, and the destination reads at most +-- one of them, so an override it never reads holds whatever was written there. The last statement is +-- the control: a key that carries no credential stays visible however often it is repeated. +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + connection_string = 'http://localhost:11111/acct', + storage_account_url = 'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_AZNCPAIR==;'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + connection_string = 'http://localhost:11111/acct', + connection_string = 'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_AZNCDUP==;'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, container = 'visible_04510_c1', container = 'visible_04510_c2'); -- { serverError BAD_ARGUMENTS } + +-- An account url is shown only when it is a plain storage account URL, which is what the destination +-- requires beside explicit credentials: userinfo, a query string (a SAS is a credential) and a fragment +-- each carry a credential of their own, in any shape and under any scheme spelling. A connection value +-- that is no string is read by neither this rule nor the destination, so it is hidden as well. The last +-- statement is the control: a plain url keeps the account url, container, path and account name visible +-- while account_key is hidden, and reaches the account_key decoding that rejects it. +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + storage_account_url = 'http://user:SEKRIT_AZNCUSERINFO@localhost:11111/acct'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(nc_04510_missing, + storage_account_url = 'HTTPS://localhost:11111/acct?sig=SEKRIT_AZNCSAS'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('http://user:SEKRIT_AZ3USERINFO@localhost:11111/acct', + 'cont', 'blob'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('http://localhost:11111/acct#f', 'cont', 'blob', + 'acct', 'SEKRIT_AZTO5KEY'); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage(['SEKRIT_AZ3ARR'], 'cont', 'blob'); -- { serverError BAD_GET } +BACKUP TABLE nonexistent_04510 TO AzureBlobStorage('http://localhost:11111/visible_04510_acct5', + 'visible_04510_cont5', 'visible_04510_blob5', 'visible_04510_acctname5', + 'SEKRIT_AZ5PLAINKEY'); -- { serverError STD_EXCEPTION } + +-- The AzureBlobStorage table engine reads its arguments the same way: an override whose value or key +-- this rule cannot read may hold a credential the engine does read (the parser evaluates both as +-- constant expressions), an account url is shown only when it is a plain storage account url, and a +-- connection value that is no string is read by neither. A connection string is hidden by replacing +-- its whole argument, which cannot be combined with hiding a following account_key. The last two +-- statements are the controls: the connection string alone still hides only its AccountKey, and a +-- plain account url keeps url, container, path and account name visible beside a hidden account_key. +CREATE TABLE t_04510_azte1 (x UInt8) ENGINE = AzureBlobStorage(nc_04510_missing, + connection_string = headers('Authorization' = 'SEKRIT_AZTEHDR')); -- { serverError NAMED_COLLECTION_DOESNT_EXIST } +CREATE TABLE t_04510_azte2 (x UInt8) ENGINE = AzureBlobStorage(nc_04510_missing, + upper('account_key') = 'SEKRIT_AZTEKEY'); -- { serverError NAMED_COLLECTION_DOESNT_EXIST } +CREATE TABLE t_04510_azte3 (x UInt8) ENGINE = AzureBlobStorage(['SEKRIT_AZTEARR'], 'cont', 'blob'); -- { serverError BAD_ARGUMENTS } +CREATE TABLE t_04510_azte4 (x UInt8) ENGINE = AzureBlobStorage('http://localhost:11111/acct?sig=SEKRIT_AZTESAS', + 'cont', 'blob', 'acct', 'SEKRIT_AZTE5KEY'); -- { serverError STD_EXCEPTION } +CREATE TABLE t_04510_azte5 (x UInt8) ENGINE = AzureBlobStorage('DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_AZTECS==;', + 'cont', 'blob', 'acct', 'SEKRIT_AZTEMIXKEY'); -- { serverError STD_EXCEPTION } +CREATE TABLE t_04510_azte6 (x UInt8) ENGINE = AzureBlobStorage('DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_AZTECTLCS==;', + 'visible_04510_teco', 'visible_04510_teblob'); -- { serverError STD_EXCEPTION } +CREATE TABLE t_04510_azte7 (x UInt8) ENGINE = AzureBlobStorage('http://localhost:11111/visible_04510_teacct', + 'visible_04510_tec5', 'visible_04510_teb5', 'visible_04510_teacctname', + 'SEKRIT_AZTECTLKEY'); -- { serverError STD_EXCEPTION } + +-- The two-argument signature takes container and path from the url and a shared access signature +-- beside it, so its second argument is a credential wherever the same rule serves the engine. +CREATE TABLE t_04510_azte8 (x UInt8) ENGINE = AzureQueue('http://localhost:11111/visible_04510_teq/cont/*', + 'SEKRIT_AZTESAS2') SETTINGS mode = 'unordered'; -- { serverError UNKNOWN_FORMAT } + -- Backup database engine reconstructs the nested S3 destination; extra_credentials must be masked. CREATE DATABASE db_04510_ec ENGINE = Backup('', S3('url_dbec', 'ak', 'SEKRIT_SAK', extra_credentials(external_id = 'SEKRIT_EID'))); -- { serverError BAD_ARGUMENTS } @@ -272,9 +404,11 @@ CREATE DATABASE db_04510_mixed ENGINE = Backup('', S3('url_dbmixed', CREATE DATABASE db_04510_ncurl ENGINE = Backup('', S3(nc_dburl_missing, url = concat('https://user:SEKRIT_PW@', 'localhost/x?X-Amz-Signature=SEKRIT_SIG'))); -- { serverError BAD_ARGUMENTS } --- The reconstructor must fail closed on an unsupported tail (headers), not emit it verbatim. +-- The reconstructor must fail closed on an unsupported tail (headers), not emit it verbatim, and the +-- rejection message must not echo it either: a nested map's values are hidden only by its parent's +-- formatter, so such a node formatted on its own carries them in plaintext. CREATE DATABASE db_04510_hdr ENGINE = Backup('', S3('url_dbhdr', 'ak', 'SEKRIT_SAK', - headers('X-Auth' = 'SEKRIT_HDR'))); -- { serverError BAD_ARGUMENTS } + headers('X-Auth' = 'SEKRIT_DBHDR'))); -- { serverError BAD_ARGUMENTS } -- The reconstructor must also fail closed on a constant-expression extra_credentials key. CREATE DATABASE db_04510_expr ENGINE = Backup('', S3('url_dbexpr', 'ak', 'SEKRIT_SAK', @@ -287,6 +421,65 @@ CREATE DATABASE db_04510_expr ENGINE = Backup('', S3('url_dbexpr', 'ak', 'SEKRIT CREATE DATABASE db_04510_quoted ENGINE = Backup('', 'S3(\'https://user:SEKRIT_PW@localhost:11111/x?X-Amz-Signature=SEKRIT_SIG\', \'ak\', \'SEKRIT_QUOTED\')'); -- { serverError BAD_ARGUMENTS } +-- `Backup(database_name, locator)` is the only valid shape. The statement is logged before the arity +-- is validated, so a credential parked in a surplus argument must be hidden as well. +CREATE DATABASE db_04510_tail ENGINE = Backup('', S3('url_dbtail', 'ak', 'SEKRIT_SAK'), + 'SEKRIT_DBTAIL'); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +CREATE DATABASE db_04510_tmap ENGINE = Backup('', S3('url_dbtmap', 'ak', 'SEKRIT_SAK'), + extra_credentials(external_id = 'SEKRIT_DBTMAP')); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +CREATE DATABASE db_04510_lone ENGINE = Backup(S3('url_dblone', 'ak', + 'SEKRIT_DBLONE')); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } + +-- A locator is also a credential carrier in the database-name position, where the argument is neither +-- masked by the S3 rule below nor safe to echo: the message that rejects it formats it standalone. +CREATE DATABASE db_04510_swap ENGINE = Backup(S3('url_dbswap', 'ak', 'SEKRIT_DBSWAP'), + ''); -- { serverError BAD_ARGUMENTS } + +-- An identifier is not a literal either, and the parsers evaluate one as a literal, so the +-- database-name position must hold a literal before the rest of the shape is trusted. +CREATE DATABASE db_04510_ident ENGINE = Backup(SEKRIT_DBIDENT, + S3('url_dbident', 'ak', 'SEKRIT_SAK')); -- { serverError BAD_ARGUMENTS } + +-- A credential-free locator (Disk, File, Memory, Null) names its destination with literals and holds +-- no credential, but it keeps a named override or nested map that it never reads, and that can carry +-- one. Both the logged text and the not-found message identify the locator, so both must hide it. +CREATE DATABASE db_04510_ftail ENGINE = Backup('src_04510', + File('nonexistent_04510', extra_credentials(external_id = 'SEKRIT_DBFTAIL'))); -- { serverError BACKUP_NOT_FOUND } + +-- A surplus literal argument reaches the same locator through the shape check instead of the tail, and +-- it is rejected only after the statement is logged, so it has to be hidden as well. The two controls +-- that follow hold the argument counts these engines do read (File one, Disk two), where the locator +-- names a destination only: nothing is masked, so both the logged text and the not-found message keep +-- it visible verbatim, and the transcript records the statement as sent rather than a re-formatted AST. +CREATE DATABASE db_04510_fover ENGINE = Backup('src_04510', + File('nonexistent_04510', 'SEKRIT_DBFILEOVER')); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +CREATE DATABASE db_04510_dover ENGINE = Backup('src_04510', + Disk('backups', 'nonexistent_04510', 'SEKRIT_DBDISKOVER')); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +CREATE DATABASE db_04510_mover ENGINE = Backup('src_04510', + Memory('nonexistent_04510', 'SEKRIT_DBMEMOVER')); -- { serverError NUMBER_OF_ARGUMENTS_DOESNT_MATCH } +CREATE DATABASE db_04510_fvalid ENGINE = Backup('src_04510', File('nonexistent_04510')); -- { serverError BACKUP_NOT_FOUND } +CREATE DATABASE db_04510_dvalid ENGINE = Backup('src_04510', Disk('backups', 'nonexistent_04510')); -- { serverError BACKUP_NOT_FOUND } + +-- A credential value given as an expression is hidden in the logged text, but evaluating it can fail +-- with a message that quotes its input, so the rejection must not carry that message either. The url +-- has to parse for the locator to reach credential evaluation at all. +CREATE DATABASE db_04510_eval ENGINE = Backup('', S3('http://localhost:11111/test/04510eval', 'ak', 'SEKRIT_SAK', + extra_credentials(external_id = toUInt64('SEKRIT_DBEVAL')))); -- { serverError BAD_ARGUMENTS } + +-- A non-tail argument that is neither a literal nor `key = value` cannot be reconstructed either. It +-- is rejected by a different message than the quoted locator above, and that message used to echo the +-- offending argument verbatim, so it needs its own tag. +CREATE DATABASE db_04510_nonlit ENGINE = Backup('', S3(concat('SEKRIT_NONLIT', 'x'), + 'url_dbnonlit')); -- { serverError BAD_ARGUMENTS } + +-- S3 is not the only backup engine whose locator carries credentials: AzureBlobStorage takes an +-- account_key and accepts connection strings and named-collection overrides that carry one too. +-- Only S3 is reconstructed above, so an Azure locator keeps its engine name and argument count +-- (neither is a secret) and every argument is hidden. The url carries a query string, which the +-- engine rejects before it reaches the network. +CREATE DATABASE db_04510_azure ENGINE = Backup('', AzureBlobStorage('http://localhost:11111/acct?sig=x', + 'cont', 'blob', 'account', 'SEKRIT_AZUREKEY')); -- { serverError BAD_ARGUMENTS } + -- The S3 database engine accepts no positional beyond secret_access_key; an extra positional must -- be masked in the logged query text. CREATE DATABASE db_04510_s3pos ENGINE = S3('url_dbs3pos', 'ak', 'SEKRIT_SAK', @@ -345,7 +538,15 @@ ORDER BY event_time_microseconds; -- logs each DDL from the replay worker, which re-masks a rewritten AST independently, so assert -- the masking property over every row this test produced, replay rows included. count() > 0 keeps -- an empty row set from passing vacuously. -SELECT count() > 0, countIf(query LIKE '%SEKRIT%') +-- The third column covers the recorded exception messages of the seven unreconstructible-locator +-- statements above, whose rejections used to echo the locator verbatim - one tag per message, since +-- they are thrown at different sites. It is scoped to those tags because widening it to every +-- deliberately-failing statement here would report unrelated pre-existing echoes. +SELECT count() > 0, countIf(query LIKE '%SEKRIT%'), + countIf(exception LIKE '%SEKRIT_QUOTED%' OR exception LIKE '%SEKRIT_NONLIT%' + OR exception LIKE '%SEKRIT_DBHDR%' OR exception LIKE '%SEKRIT_DBSWAP%' + OR exception LIKE '%SEKRIT_DBEVAL%' OR exception LIKE '%SEKRIT_DBFTAIL%' + OR exception LIKE '%SEKRIT_DBIDENT%') FROM system.query_log WHERE current_database = currentDatabase() AND type != 'QueryStart' diff --git a/tests/queries/0_stateless/05233_database_backup_locator_secret_mask.reference b/tests/queries/0_stateless/05233_database_backup_locator_secret_mask.reference new file mode 100644 index 000000000000..38dae7585877 --- /dev/null +++ b/tests/queries/0_stateless/05233_database_backup_locator_secret_mask.reference @@ -0,0 +1,14 @@ +-- SHOW CREATE DATABASE +CREATE DATABASE default_view\nENGINE = Backup(\'default_src\', S3(\'http://localhost:11111/test/backups/default/locator_mask\', \'test\', \'[HIDDEN]\')) +-- system.databases.engine_full +Backup(\'default_src\', S3(\'http://localhost:11111/test/backups/default/locator_mask\', \'test\', \'[HIDDEN]\')) +-- secret occurrences in SHOW CREATE (must be 0) +0 +-- [HIDDEN] present in SHOW CREATE (must be 1) +1 +-- secret occurrences in system.databases.engine_full (must be 0) +0 +-- [HIDDEN] present in system.databases.engine_full (must be 1) +1 +-- rows visible through the Backup database +10 diff --git a/tests/queries/0_stateless/05233_database_backup_locator_secret_mask.sh b/tests/queries/0_stateless/05233_database_backup_locator_secret_mask.sh new file mode 100755 index 000000000000..cb236e68b03c --- /dev/null +++ b/tests/queries/0_stateless/05233_database_backup_locator_secret_mask.sh @@ -0,0 +1,58 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-encrypted-storage +# Tag no-fasttest: requires S3 +# Tag no-encrypted-storage: a backup from an encrypted disk restores only to an encrypted disk, so the Backup database gets no parts. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +client_opts=( + --allow_repeated_settings + --send_logs_level 'error' +) + +db=${CLICKHOUSE_DATABASE}_src +view=${CLICKHOUSE_DATABASE}_view +# The credentials the stateless suite uses for S3 backups: access key id 'test', secret 'testtest'. +# They are two distinct strings, so the assertions below tell the id apart from the secret. +dest="S3('http://localhost:11111/test/backups/${CLICKHOUSE_DATABASE}/locator_mask', 'test', 'testtest')" + +# BACKUP TO S3 emits a bandwidth warning on stderr under some configurations, which would break +# the reference in a bare run, so keep the log level at error throughout (as 02843 and 04327 do). +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP DATABASE IF EXISTS ${db}; +DROP DATABASE IF EXISTS ${view}; +CREATE DATABASE ${db}; +CREATE TABLE ${db}.t (id UInt64) ENGINE = MergeTree ORDER BY id; +INSERT INTO ${db}.t SELECT number FROM numbers(10); +BACKUP DATABASE ${db} TO ${dest} FORMAT Null; +CREATE DATABASE ${view} ENGINE = Backup('${db}', ${dest}); +" + +# Both display surfaces of a real, successfully created Backup database. The locator must stay a +# nested S3 function so that only its secret_access_key is replaced by [HIDDEN]: the url and the +# access key id must remain visible, and the two surfaces must agree. +echo '-- SHOW CREATE DATABASE' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE DATABASE ${view}" +echo '-- system.databases.engine_full' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SELECT engine_full FROM system.databases WHERE name = '${view}'" + +# Explicit counters, so a regression is visible on its own line rather than buried in a long one. +echo '-- secret occurrences in SHOW CREATE (must be 0)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE DATABASE ${view}" | grep -c testtest +echo '-- [HIDDEN] present in SHOW CREATE (must be 1)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE DATABASE ${view}" | grep -c -m1 '\[HIDDEN\]' +echo '-- secret occurrences in system.databases.engine_full (must be 0)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SELECT engine_full FROM system.databases WHERE name = '${view}'" | grep -c testtest +echo '-- [HIDDEN] present in system.databases.engine_full (must be 1)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SELECT engine_full FROM system.databases WHERE name = '${view}'" | grep -c -m1 '\[HIDDEN\]' + +# The Backup database is readable, so the masked surfaces above are not the output of a broken database. +echo '-- rows visible through the Backup database' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SELECT count() FROM ${view}.t" + +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP DATABASE ${view}; +DROP DATABASE ${db}; +" diff --git a/tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.reference b/tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.reference new file mode 100644 index 000000000000..2d33a25de9b3 --- /dev/null +++ b/tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.reference @@ -0,0 +1,34 @@ +-- database-level duplicate definition reached (must be 1) +1 +-- CANNOT_RESTORE_DATABASE (must be 1) +1 +-- secret occurrences in the error (must be 0) +0 +-- [HIDDEN] present in the error (must be 1) +1 +-- archived locator still identifiable in the error (must be 1) +1 +-- restored Backup database reads its source table (must be 3) +3 +-- the restored locator is hidden on display (must be 1) +1 +-- table-level duplicate definition reached (must be 1) +1 +-- CANNOT_RESTORE_TABLE (must be 1) +1 +-- secret occurrences in the error (must be 0) +0 +-- [HIDDEN] present in the error (must be 1) +1 +-- archived table definition still identifiable in the error (must be 1) +1 +-- Azure table-level duplicate definition reached (must be 1) +1 +-- CANNOT_RESTORE_TABLE for the Azure definitions (must be 1) +1 +-- signature occurrences in the error (must be 0) +0 +-- [HIDDEN] present in the error (must be 1) +1 +-- restore target still named in the error (must be 1) +1 diff --git a/tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.sh b/tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.sh new file mode 100755 index 000000000000..8ca86313e9b8 --- /dev/null +++ b/tests/queries/0_stateless/05234_restore_duplicate_definition_secret_mask.sh @@ -0,0 +1,164 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-encrypted-storage +# Tag no-fasttest: requires the S3 endpoint +# Tag no-encrypted-storage: a backup from an encrypted disk restores only to an encrypted disk, so the restored Backup database gets no parts. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +client_opts=( + --allow_repeated_settings + --send_logs_level 'error' +) + +s1=${CLICKHOUSE_DATABASE}_s1 +s2=${CLICKHOUSE_DATABASE}_s2 +v1=${CLICKHOUSE_DATABASE}_v1 +v2=${CLICKHOUSE_DATABASE}_v2 +merged=${CLICKHOUSE_DATABASE}_merged +restored=${CLICKHOUSE_DATABASE}_restored + +# Access key id 'test', secret 'testtest': the credentials the stateless suite uses for S3. They are +# two distinct strings, so the assertions below tell the id apart from the secret. +inner1="S3('http://localhost:11111/test/backups/${CLICKHOUSE_DATABASE}/dupdef1', 'test', 'testtest')" +inner2="S3('http://localhost:11111/test/backups/${CLICKHOUSE_DATABASE}/dupdef2', 'test', 'testtest')" +# The outer locator carries no credential, so a secret found in the restore error can only have come +# from one of the two archived definitions and not from the statement being executed. +outer="Disk('backups', '${CLICKHOUSE_DATABASE}_dupdef_outer')" + +# The two tables have different names on purpose: two tables renamed into one target would reach the +# table-level duplicate-definition check first, and the database-level one is what is asserted here. +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP DATABASE IF EXISTS ${s1}; +DROP DATABASE IF EXISTS ${s2}; +DROP DATABASE IF EXISTS ${v1}; +DROP DATABASE IF EXISTS ${v2}; +DROP DATABASE IF EXISTS ${merged}; +DROP DATABASE IF EXISTS ${restored}; +CREATE DATABASE ${s1}; +CREATE TABLE ${s1}.t1 (id UInt64) ENGINE = MergeTree ORDER BY id; +INSERT INTO ${s1}.t1 SELECT number FROM numbers(3); +CREATE DATABASE ${s2}; +CREATE TABLE ${s2}.t2 (id UInt64) ENGINE = MergeTree ORDER BY id; +INSERT INTO ${s2}.t2 SELECT number FROM numbers(3); +BACKUP DATABASE ${s1} TO ${inner1} FORMAT Null; +BACKUP DATABASE ${s2} TO ${inner2} FORMAT Null; +CREATE DATABASE ${v1} ENGINE = Backup('${s1}', ${inner1}); +CREATE DATABASE ${v2} ENGINE = Backup('${s2}', ${inner2}); +BACKUP DATABASE ${v1}, DATABASE ${v2} TO ${outer} FORMAT Null; +" + +# Both elements rename to one target, so the definition read second reaches the duplicate-definition +# check, which reports both archived definitions. +err=$(${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q \ + "RESTORE DATABASE ${v1} AS ${merged}, DATABASE ${v2} AS ${merged} FROM ${outer}" 2>&1) + +# The first two counters pin the failure to that check rather than to an earlier one, so the +# remaining three cannot be satisfied by an unrelated error or by an empty result. +echo '-- database-level duplicate definition reached (must be 1)' +echo "$err" | grep -c -m1 'Extracted two different create queries for the same database' +echo '-- CANNOT_RESTORE_DATABASE (must be 1)' +echo "$err" | grep -c -m1 CANNOT_RESTORE_DATABASE +echo '-- secret occurrences in the error (must be 0)' +echo "$err" | grep -c testtest +echo '-- [HIDDEN] present in the error (must be 1)' +echo "$err" | grep -c -m1 '\[HIDDEN\]' +echo '-- archived locator still identifiable in the error (must be 1)' +echo "$err" | grep -c -m1 dupdef1 + +# The archived definition has to keep the credential as written: attaching a Backup database checks the +# backup with the credentials the locator carries, so a definition archived with '[HIDDEN]' would fail +# the restore below with ACCESS_DENIED instead of reading the source table through it. +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "RESTORE DATABASE ${v1} AS ${restored} FROM ${outer}" > /dev/null +echo '-- restored Backup database reads its source table (must be 3)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SELECT count() FROM ${restored}.t1" +echo '-- the restored locator is hidden on display (must be 1)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q \ + "SHOW CREATE DATABASE ${restored} SETTINGS format_display_secrets_in_show_and_select = 0" | grep -c -m1 '\[HIDDEN\]' + +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP DATABASE IF EXISTS ${restored}; +DROP DATABASE IF EXISTS ${merged}; +DROP DATABASE ${v1}; +DROP DATABASE ${v2}; +DROP DATABASE ${s1}; +DROP DATABASE ${s2}; +" + +# The same function checks table definitions the same way, and a table engine keeps its credential in +# the same argument position a database locator does. +ta=${CLICKHOUSE_DATABASE}.dup_a +tb=${CLICKHOUSE_DATABASE}.dup_b +tmerged=${CLICKHOUSE_DATABASE}.dup_merged +table_outer="Disk('backups', '${CLICKHOUSE_DATABASE}_tabledef_outer')" + +# Two tables renamed into one target: the definition read second is compared against one that is +# already recorded for that target and differs from it, which is what the check requires. +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP TABLE IF EXISTS ${ta}; +DROP TABLE IF EXISTS ${tb}; +DROP TABLE IF EXISTS ${tmerged}; +CREATE TABLE ${ta} (id UInt64) ENGINE = S3('http://localhost:11111/test/${CLICKHOUSE_DATABASE}/dup_a.csv', 'test', 'testtest', 'CSV'); +CREATE TABLE ${tb} (id UInt64) ENGINE = S3('http://localhost:11111/test/${CLICKHOUSE_DATABASE}/dup_b.csv', 'test', 'testtest', 'CSV'); +BACKUP TABLE ${ta}, TABLE ${tb} TO ${table_outer} FORMAT Null; +" + +err=$(${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q \ + "RESTORE TABLE ${ta} AS ${tmerged}, TABLE ${tb} AS ${tmerged} FROM ${table_outer}" 2>&1) + +echo '-- table-level duplicate definition reached (must be 1)' +echo "$err" | grep -c -m1 'Extracted two different create queries for the same table' +echo '-- CANNOT_RESTORE_TABLE (must be 1)' +echo "$err" | grep -c -m1 CANNOT_RESTORE_TABLE +echo '-- secret occurrences in the error (must be 0)' +echo "$err" | grep -c testtest +echo '-- [HIDDEN] present in the error (must be 1)' +echo "$err" | grep -c -m1 '\[HIDDEN\]' +echo '-- archived table definition still identifiable in the error (must be 1)' +echo "$err" | grep -c -m1 dup_a + +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP TABLE ${ta}; +DROP TABLE ${tb}; +" + +# An Azure table definition reaches the same check, and an Azure locator can hold its credential inside +# the account url rather than in an argument of its own. +aza=${CLICKHOUSE_DATABASE}.az_a +azb=${CLICKHOUSE_DATABASE}.az_b +azmerged=${CLICKHOUSE_DATABASE}.az_merged +azure_outer="Disk('backups', '${CLICKHOUSE_DATABASE}_azuredef_outer')" +# A shared access signature is a credential, and the account url is plain apart from it, so a secret +# found in the restore error can only have come from one of the two archived definitions. Neither +# creating nor backing up the tables reads the endpoint, so no Azure service is needed here. +azure_a="AzureBlobStorage('http://localhost:11111/devstoreaccount1?sig=SEKRITAZURESAS', 'cont', 'az_a.csv', 'CSV')" +azure_b="AzureBlobStorage('http://localhost:11111/devstoreaccount1?sig=SEKRITAZURESAS', 'cont', 'az_b.csv', 'CSV')" + +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP TABLE IF EXISTS ${aza}; +DROP TABLE IF EXISTS ${azb}; +DROP TABLE IF EXISTS ${azmerged}; +CREATE TABLE ${aza} (id UInt64) ENGINE = ${azure_a}; +CREATE TABLE ${azb} (id UInt64) ENGINE = ${azure_b}; +BACKUP TABLE ${aza}, TABLE ${azb} TO ${azure_outer} FORMAT Null; +" + +err=$(${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q \ + "RESTORE TABLE ${aza} AS ${azmerged}, TABLE ${azb} AS ${azmerged} FROM ${azure_outer}" 2>&1) + +echo '-- Azure table-level duplicate definition reached (must be 1)' +echo "$err" | grep -c -m1 'Extracted two different create queries for the same table' +echo '-- CANNOT_RESTORE_TABLE for the Azure definitions (must be 1)' +echo "$err" | grep -c -m1 CANNOT_RESTORE_TABLE +echo '-- signature occurrences in the error (must be 0)' +echo "$err" | grep -c SEKRITAZURESAS +echo '-- [HIDDEN] present in the error (must be 1)' +echo "$err" | grep -c -m1 '\[HIDDEN\]' +echo '-- restore target still named in the error (must be 1)' +echo "$err" | grep -c -m1 az_merged + +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP TABLE ${aza}; +DROP TABLE ${azb}; +" From 317cd284cb24361af84aa3a2cc1f3ad6e3865d2e Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 10:58:36 +0000 Subject: [PATCH 047/185] Backport #116378 to 26.8: Do not put a row into a dictionary source's pipeline header --- src/Dictionaries/CassandraSource.cpp | 2 +- src/Dictionaries/DictionarySourceFactory.cpp | 2 ++ src/Dictionaries/DirectDictionary.cpp | 4 ++-- src/Dictionaries/NullDictionarySource.cpp | 2 +- src/Dictionaries/RedisSource.cpp | 2 +- src/Processors/Sources/MongoDBSource.cpp | 2 +- src/Processors/Sources/ShellCommandSource.cpp | 2 +- src/Processors/Sources/YTsaurusSource.cpp | 4 ++-- .../05043_direct_dictionary_in_merge.reference | 1 + .../05043_direct_dictionary_in_merge.sh | 15 +++++++++++++++ 10 files changed, 27 insertions(+), 9 deletions(-) create mode 100644 tests/queries/0_stateless/05043_direct_dictionary_in_merge.reference create mode 100755 tests/queries/0_stateless/05043_direct_dictionary_in_merge.sh diff --git a/src/Dictionaries/CassandraSource.cpp b/src/Dictionaries/CassandraSource.cpp index 83e7869ddeb6..8030b120b82c 100644 --- a/src/Dictionaries/CassandraSource.cpp +++ b/src/Dictionaries/CassandraSource.cpp @@ -25,7 +25,7 @@ CassandraSource::CassandraSource( const String & query_str, SharedHeader & sample_block, size_t max_block_size_) - : ISource(sample_block) + : ISource(std::make_shared(sample_block->cloneEmpty())) , session(session_) , statement(query_str.c_str(), /*parameters count*/ 0) , max_block_size(max_block_size_) diff --git a/src/Dictionaries/DictionarySourceFactory.cpp b/src/Dictionaries/DictionarySourceFactory.cpp index 248efbe0e5ac..c96a071d945d 100644 --- a/src/Dictionaries/DictionarySourceFactory.cpp +++ b/src/Dictionaries/DictionarySourceFactory.cpp @@ -20,6 +20,8 @@ namespace ErrorCodes namespace { + /// Holds one row per column, an attribute's row being its `null_value` default. A port header + /// must have no rows, so a source publishing this block as its header has to strip them first. Block createSampleBlock(const DictionaryStructure & dict_struct) { Block block; diff --git a/src/Dictionaries/DirectDictionary.cpp b/src/Dictionaries/DirectDictionary.cpp index 684bd77f4c83..05d98fc6c70a 100644 --- a/src/Dictionaries/DirectDictionary.cpp +++ b/src/Dictionaries/DirectDictionary.cpp @@ -346,14 +346,14 @@ class SourceFromQueryPipeline : public ISource { public: explicit SourceFromQueryPipeline(QueryPipeline & pipeline_) - : ISource(pipeline_.getSharedHeader()) + : ISource(std::make_shared(pipeline_.getSharedHeader()->cloneEmpty())) , executor(pipeline_) { pipeline_.setConcurrencyControl(false); } explicit SourceFromQueryPipeline(BlockIO io) - : ISource(io.pipeline.getSharedHeader()) + : ISource(std::make_shared(io.pipeline.getSharedHeader()->cloneEmpty())) , io_holder(std::move(io)) , executor(io_holder->pipeline) { diff --git a/src/Dictionaries/NullDictionarySource.cpp b/src/Dictionaries/NullDictionarySource.cpp index c940c28e5ef7..beb30e8b8f8d 100644 --- a/src/Dictionaries/NullDictionarySource.cpp +++ b/src/Dictionaries/NullDictionarySource.cpp @@ -22,7 +22,7 @@ BlockIO NullDictionarySource::loadAll() { LOG_TRACE(getLogger("NullDictionarySource"), "loadAll {}", toString()); BlockIO io; - io.pipeline = QueryPipeline(std::make_shared(sample_block)); + io.pipeline = QueryPipeline(std::make_shared(std::make_shared(sample_block->cloneEmpty()))); return io; } diff --git a/src/Dictionaries/RedisSource.cpp b/src/Dictionaries/RedisSource.cpp index a606d9cc69cb..c2c38623dce3 100644 --- a/src/Dictionaries/RedisSource.cpp +++ b/src/Dictionaries/RedisSource.cpp @@ -26,7 +26,7 @@ namespace DB const RedisStorageType & storage_type_, SharedHeader sample_block, size_t max_block_size_) - : ISource(sample_block) + : ISource(std::make_shared(sample_block->cloneEmpty())) , connection(std::move(connection_)) , keys(keys_) , storage_type(storage_type_) diff --git a/src/Processors/Sources/MongoDBSource.cpp b/src/Processors/Sources/MongoDBSource.cpp index 4dca8b937265..e5993dde6597 100644 --- a/src/Processors/Sources/MongoDBSource.cpp +++ b/src/Processors/Sources/MongoDBSource.cpp @@ -160,7 +160,7 @@ MongoDBSource::MongoDBSource( const mongocxx::options::find & options, SharedHeader sample_block_, const UInt64 & max_block_size_) - : ISource{sample_block_} + : ISource{std::make_shared(sample_block_->cloneEmpty())} , client{uri} , database{client.database(uri.database())} , collection{database.collection(collection_name)} diff --git a/src/Processors/Sources/ShellCommandSource.cpp b/src/Processors/Sources/ShellCommandSource.cpp index b39668905a7f..01a0eca5d00e 100644 --- a/src/Processors/Sources/ShellCommandSource.cpp +++ b/src/Processors/Sources/ShellCommandSource.cpp @@ -503,7 +503,7 @@ namespace const ShellCommandSourceConfiguration & configuration_ = {}, std::unique_ptr && command_holder_ = nullptr, std::shared_ptr process_pool_ = nullptr) - : ISource(sample_block_) + : ISource(std::make_shared(sample_block_->cloneEmpty())) , context(context_) , format(format_) , sample_block(sample_block_) diff --git a/src/Processors/Sources/YTsaurusSource.cpp b/src/Processors/Sources/YTsaurusSource.cpp index 64e1b1250626..48e8faa9a12d 100644 --- a/src/Processors/Sources/YTsaurusSource.cpp +++ b/src/Processors/Sources/YTsaurusSource.cpp @@ -31,7 +31,7 @@ namespace YTsaurusSetting YTsaurusTableSourceStaticTable::YTsaurusTableSourceStaticTable( YTsaurusClientPtr client_, const String & cypress_path_, std::pair rows_range_, const YTsaurusTableSourceOptions & source_options_, const SharedHeader & sample_block_, const UInt64 & max_block_size_) - : ISource(sample_block_) + : ISource(std::make_shared(sample_block_->cloneEmpty())) , client(std::move(client_)) , cypress_path(cypress_path_) , rows_range(std::move(rows_range_)) @@ -62,7 +62,7 @@ Chunk YTsaurusTableSourceStaticTable::generate() YTsaurusTableSourceDynamicTable::YTsaurusTableSourceDynamicTable( YTsaurusClientPtr client_, const String & cypress_path, const YTsaurusTableSourceOptions & source_options_, const SharedHeader & sample_block_, const UInt64 & max_block_size_) - : ISource(sample_block_) + : ISource(std::make_shared(sample_block_->cloneEmpty())) , client(std::move(client_)) , sample_block(sample_block_) , max_block_size(max_block_size_) diff --git a/tests/queries/0_stateless/05043_direct_dictionary_in_merge.reference b/tests/queries/0_stateless/05043_direct_dictionary_in_merge.reference new file mode 100644 index 000000000000..72c3262a309d --- /dev/null +++ b/tests/queries/0_stateless/05043_direct_dictionary_in_merge.reference @@ -0,0 +1 @@ +Hello 1 diff --git a/tests/queries/0_stateless/05043_direct_dictionary_in_merge.sh b/tests/queries/0_stateless/05043_direct_dictionary_in_merge.sh new file mode 100755 index 000000000000..8385d6f97658 --- /dev/null +++ b/tests/queries/0_stateless/05043_direct_dictionary_in_merge.sh @@ -0,0 +1,15 @@ +#!/usr/bin/env bash + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +# A `direct` dictionary reads through the source's own pipeline, and `merge` wraps every child in +# `materialize`, which rejects a header that carries rows. +$CLICKHOUSE_CLIENT -q " + CREATE DICTIONARY dict (word String, counter UInt32) + PRIMARY KEY word + SOURCE(HTTP(url '${CLICKHOUSE_URL}&query=SELECT+%27Hello%27,1+FORMAT+CSV' format 'CSV')) + LAYOUT(DIRECT())" + +$CLICKHOUSE_CLIENT -q "SELECT * FROM merge(currentDatabase(), '^dict\$')" From 8fd1e7f2d12a44613508a1d0b1b1b3cc88c0aa84 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 11:32:33 +0000 Subject: [PATCH 048/185] Backport #120453 to 26.8: Speed up Map key lookup for LowCardinality keys --- src/Columns/LowCardinalityValueIndex.h | 70 ++++++ src/DataTypes/DataTypeMapHelpers.cpp | 164 ++++++++++++-- src/Functions/array/arrayElement.cpp | 57 +++-- .../map_lowcardinality_key_lookup.xml | 43 ++++ ...18_map_lowcardinality_key_lookup.reference | 102 +++++++++ .../05218_map_lowcardinality_key_lookup.sql | 203 ++++++++++++++++++ .../05234_map_key_lookup_nan.reference | 22 ++ .../0_stateless/05234_map_key_lookup_nan.sql | 61 ++++++ 8 files changed, 671 insertions(+), 51 deletions(-) create mode 100644 src/Columns/LowCardinalityValueIndex.h create mode 100644 tests/performance/map_lowcardinality_key_lookup.xml create mode 100644 tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.reference create mode 100644 tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.sql create mode 100644 tests/queries/0_stateless/05234_map_key_lookup_nan.reference create mode 100644 tests/queries/0_stateless/05234_map_key_lookup_nan.sql diff --git a/src/Columns/LowCardinalityValueIndex.h b/src/Columns/LowCardinalityValueIndex.h new file mode 100644 index 000000000000..fb6fbecdad7d --- /dev/null +++ b/src/Columns/LowCardinalityValueIndex.h @@ -0,0 +1,70 @@ +#pragma once + +#include +#include +#include +#include +#include + +#include +#include + + +namespace DB +{ + +enum class LowCardinalityValueLookupResult +{ + /// The dictionary does not hold strings, so the caller has to compare the values itself. + Unsupported, + /// No row of the column can hold the value. + NotFound, + Found, +}; + +/// Resolves `value` to its position in the dictionary of `column` and calls +/// `callback(const IndexType * indexes, IndexType value_index)`, so that comparing a row of `column` +/// with `value` becomes an integer comparison of dictionary positions. +template +LowCardinalityValueLookupResult callWithLowCardinalityValueIndex( + const ColumnLowCardinality & column, std::string_view value, Callback && callback) +{ + const auto & dictionary = column.getDictionary(); + const auto & dictionary_values = *dictionary.getNestedNotNullableColumn(); + if (!typeid_cast(&dictionary_values) && !typeid_cast(&dictionary_values)) + return LowCardinalityValueLookupResult::Unsupported; + + auto value_index = dictionary.getOrFindValueIndex(value); + if (!value_index) + return LowCardinalityValueLookupResult::NotFound; + + const IColumn & indexes = column.getIndexes(); + + auto call_for_index_type = [&](IndexType) -> LowCardinalityValueLookupResult + { + /// A shared dictionary also holds values of other columns, so a dictionary position is not + /// necessarily representable in the index type of this column. No row here references it then. + if constexpr (!std::is_same_v) + { + if (*value_index > std::numeric_limits::max()) + return LowCardinalityValueLookupResult::NotFound; + } + + callback( + assert_cast &>(indexes).getData().data(), + static_cast(*value_index)); + + return LowCardinalityValueLookupResult::Found; + }; + + switch (column.getSizeOfIndexType()) + { + case sizeof(UInt8): return call_for_index_type(UInt8{}); + case sizeof(UInt16): return call_for_index_type(UInt16{}); + case sizeof(UInt32): return call_for_index_type(UInt32{}); + case sizeof(UInt64): return call_for_index_type(UInt64{}); + default: throwUnexpectedLowCardinalityIndexType(column.getSizeOfIndexType()); + } +} + +} diff --git a/src/DataTypes/DataTypeMapHelpers.cpp b/src/DataTypes/DataTypeMapHelpers.cpp index 63aa451627f6..5443542a3be6 100644 --- a/src/DataTypes/DataTypeMapHelpers.cpp +++ b/src/DataTypes/DataTypeMapHelpers.cpp @@ -1,11 +1,15 @@ #include #include +#include #include +#include #include #include #include #include +#include +#include #include #include @@ -35,21 +39,30 @@ struct KeyMatcherGeneric bool match(size_t keys_row) const { - return keys_column.compareAt(keys_row, 0, key, 0) == 0; + /// The direction hint must not be zero. When exactly one side is a `NaN`, or a NULL of a + /// `Nullable` nested in the key, `compareAt` answers with the hint itself, so a zero hint + /// reports them as equal to every other value: a row holding such a key would match any + /// requested key, and requesting such a key would match any key. + return keys_column.compareAt(keys_row, 0, key, 1) == 0; } }; -/// Specialized key matcher for ColumnVector. Compares values directly -/// without virtual dispatch. -template +/// Specialized key matcher for columns holding a flat array of values (ColumnVector, +/// ColumnDecimal). Compares values directly without virtual dispatch. +template struct KeyMatcherVector { - const typename ColumnVector::Container & data; + using T = typename ColumnType::ValueType; + + const typename ColumnType::Container & data; T key_value; bool match(size_t keys_row) const { - return data[keys_row] == key_value; + /// `CompareHelper` is a plain `==` for every type but the floating point ones, where it + /// keeps the matcher in agreement with `KeyMatcherGeneric`: a `NaN` key is found by a + /// requested `NaN` and by nothing else. + return CompareHelper::equals(data[keys_row], key_value, 1); } }; @@ -86,6 +99,20 @@ struct KeyMatcherFixedString } }; +/// Specialized key matcher for ColumnLowCardinality. The requested key is resolved to its dictionary +/// position once, so a key is matched without comparing the key values. +template +struct KeyMatcherLowCardinality +{ + const IndexType * indexes; + IndexType key_index; + + bool match(size_t keys_row) const + { + return indexes[keys_row] == key_index; + } +}; + /// The core position-finding loop, parametrized by Matcher type. /// For each row in [start, end), finds the flat index of the matching key. /// @@ -126,6 +153,59 @@ void findKeyPositions( } } +/// The value of the single-row key column, if the key is a String or a FixedString, +/// possibly wrapped in LowCardinality. +std::optional tryGetStringKey(const IColumn & key) +{ + const IColumn * key_values = &key; + if (const auto * key_low_cardinality = typeid_cast(&key)) + key_values = key_low_cardinality->getDictionary().getNestedNotNullableColumn().get(); + + if (!typeid_cast(key_values) && !typeid_cast(key_values)) + return {}; + + return key.getDataAt(0); +} + +/// Finds the key positions by comparing dictionary positions when the keys are LowCardinality. +/// Returns false if the key column types are not supported, leaving it to the generic matcher. +bool tryFindKeyPositionsLowCardinality( + const IColumn & keys_column, + const ColumnArray::Offsets & offsets, + const IColumn & key, + size_t start, + size_t end, + PaddedPODArray & matched_positions) +{ + const auto * keys_low_cardinality = typeid_cast(&keys_column); + if (!keys_low_cardinality) + return false; + + auto key_value = tryGetStringKey(key); + if (!key_value) + return false; + + auto lookup_result = callWithLowCardinalityValueIndex( + *keys_low_cardinality, + *key_value, + [&](const auto * indexes, auto key_index) + { + KeyMatcherLowCardinality matcher{indexes, key_index}; + findKeyPositions(offsets, matcher, start, end, matched_positions); + }); + + if (lookup_result == LowCardinalityValueLookupResult::Unsupported) + return false; + + if (lookup_result == LowCardinalityValueLookupResult::NotFound) + { + matched_positions.clear(); + matched_positions.resize_fill(end - start, KEY_NOT_FOUND); + } + + return true; +} + /// Dispatches to the appropriate specialized matcher based on the key column type, /// then calls findKeyPositions with that matcher. void findKeyPositionsDispatch( @@ -138,31 +218,47 @@ void findKeyPositionsDispatch( { TypeIndex type_id = keys_column.getDataType(); - /// Try ColumnVector specializations. switch (type_id) { -#define DISPATCH_VECTOR(T) \ +#define DISPATCH_COLUMN(T, ColType) \ case TypeIndex::T: \ { \ - using ColType = ColumnVector; \ const auto & typed_col = assert_cast(keys_column); \ const auto & key_col = assert_cast(key); \ - KeyMatcherVector matcher{typed_col.getData(), key_col.getData()[0]}; \ + KeyMatcherVector matcher{typed_col.getData(), key_col.getData()[0]}; \ findKeyPositions(offsets, matcher, start, end, matched_positions); \ return; \ } +#define DISPATCH_VECTOR(T) DISPATCH_COLUMN(T, ColumnVector) +#define DISPATCH_DECIMAL(T) DISPATCH_COLUMN(T, ColumnDecimal) DISPATCH_VECTOR(UInt8) DISPATCH_VECTOR(UInt16) DISPATCH_VECTOR(UInt32) DISPATCH_VECTOR(UInt64) + DISPATCH_VECTOR(UInt128) + DISPATCH_VECTOR(UInt256) DISPATCH_VECTOR(Int8) DISPATCH_VECTOR(Int16) DISPATCH_VECTOR(Int32) DISPATCH_VECTOR(Int64) + DISPATCH_VECTOR(Int128) + DISPATCH_VECTOR(Int256) + DISPATCH_VECTOR(BFloat16) DISPATCH_VECTOR(Float32) DISPATCH_VECTOR(Float64) + DISPATCH_VECTOR(UUID) + DISPATCH_VECTOR(IPv4) + DISPATCH_VECTOR(IPv6) + DISPATCH_DECIMAL(Decimal32) + DISPATCH_DECIMAL(Decimal64) + DISPATCH_DECIMAL(Decimal128) + DISPATCH_DECIMAL(Decimal256) + DISPATCH_DECIMAL(DateTime64) + DISPATCH_DECIMAL(Time64) +#undef DISPATCH_DECIMAL #undef DISPATCH_VECTOR +#undef DISPATCH_COLUMN case TypeIndex::String: { @@ -181,13 +277,19 @@ void findKeyPositionsDispatch( findKeyPositions(offsets, matcher, start, end, matched_positions); return; } - default: + case TypeIndex::LowCardinality: { - /// Fallback: generic matcher using virtual compareAt. - KeyMatcherGeneric matcher{keys_column, key}; - findKeyPositions(offsets, matcher, start, end, matched_positions); + if (tryFindKeyPositionsLowCardinality(keys_column, offsets, key, start, end, matched_positions)) + return; + break; } + default: + break; } + + /// Fallback: generic matcher using virtual compareAt. + KeyMatcherGeneric matcher{keys_column, key}; + findKeyPositions(offsets, matcher, start, end, matched_positions); } /// --------------------------------------------------------------------------- @@ -213,17 +315,20 @@ void extractValuesGeneric( } } -/// Specialized value extractor for ColumnVector, with optional Nullable support. +/// Specialized value extractor for columns holding a flat array of values (ColumnVector, +/// ColumnDecimal), with optional Nullable support. /// If src_null_map / dst_null_map are non-null, propagates null flags. /// For missing keys, inserts a default value and sets the null flag to 1. -template +template void extractValuesVector( - const ColumnVector & values_column, - ColumnVector & result, + const ColumnType & values_column, + ColumnType & result, const PaddedPODArray & matched_positions, const NullMap * src_null_map = nullptr, NullMap * dst_null_map = nullptr) { + using T = typename ColumnType::ValueType; + const auto & src_data = values_column.getData(); auto & dst_data = result.getData(); size_t old_size = dst_data.size(); @@ -374,28 +479,45 @@ void extractValuesDispatch( switch (type_id) { -#define DISPATCH_VECTOR(T) \ +#define DISPATCH_COLUMN(T, ColType) \ case TypeIndex::T: \ { \ - using ColType = ColumnVector; \ - extractValuesVector( \ + extractValuesVector( \ assert_cast(*data_column), \ assert_cast(*result_data_column), \ matched_positions, src_null_map, dst_null_map); \ return; \ } +#define DISPATCH_VECTOR(T) DISPATCH_COLUMN(T, ColumnVector) +#define DISPATCH_DECIMAL(T) DISPATCH_COLUMN(T, ColumnDecimal) DISPATCH_VECTOR(UInt8) DISPATCH_VECTOR(UInt16) DISPATCH_VECTOR(UInt32) DISPATCH_VECTOR(UInt64) + DISPATCH_VECTOR(UInt128) + DISPATCH_VECTOR(UInt256) DISPATCH_VECTOR(Int8) DISPATCH_VECTOR(Int16) DISPATCH_VECTOR(Int32) DISPATCH_VECTOR(Int64) + DISPATCH_VECTOR(Int128) + DISPATCH_VECTOR(Int256) + DISPATCH_VECTOR(BFloat16) DISPATCH_VECTOR(Float32) DISPATCH_VECTOR(Float64) + DISPATCH_VECTOR(UUID) + DISPATCH_VECTOR(IPv4) + DISPATCH_VECTOR(IPv6) + DISPATCH_DECIMAL(Decimal32) + DISPATCH_DECIMAL(Decimal64) + DISPATCH_DECIMAL(Decimal128) + DISPATCH_DECIMAL(Decimal256) + DISPATCH_DECIMAL(DateTime64) + DISPATCH_DECIMAL(Time64) +#undef DISPATCH_DECIMAL #undef DISPATCH_VECTOR +#undef DISPATCH_COLUMN case TypeIndex::String: { diff --git a/src/Functions/array/arrayElement.cpp b/src/Functions/array/arrayElement.cpp index 663b58c53340..a041de61cb25 100644 --- a/src/Functions/array/arrayElement.cpp +++ b/src/Functions/array/arrayElement.cpp @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -1855,6 +1856,17 @@ struct MatcherNumberConst bool match(size_t row_data, size_t /* row_index */) const { return data[row_data] == index; } }; +/// Matcher for keys of a ColumnLowCardinality. The requested key is resolved to its dictionary +/// position once, so a key is matched without comparing the key values. +template +struct MatcherLowCardinalityConst +{ + const IndexType * indexes; + IndexType key_index; + + bool match(size_t row_data, size_t /* row_index */) const { return indexes[row_data] == key_index; } +}; + } template @@ -1921,11 +1933,6 @@ bool castColumnString(const IColumn * column, F && f) return castTypeToEither(column, std::forward(f)); } -bool isStringOrFixedStringColumn(const IColumn & column) -{ - return typeid_cast(&column) || typeid_cast(&column); -} - template bool FunctionArrayElement::matchKeyToIndexStringConst( const IColumn & data, const Offsets & offsets, const Field & index, PaddedPODArray & matched_idxs) @@ -1933,37 +1940,27 @@ bool FunctionArrayElement::matchKeyToIndexStringConst( if (index.getType() != Field::Types::String) return false; - /// The dictionary lookup below is defined only for String and FixedString keys. For other - /// LowCardinality key types, fall through so that the regular dispatch reports the type error - /// instead of silently finding no match. - const auto * low_cardinality_data = typeid_cast(&data); - if (low_cardinality_data - && isStringOrFixedStringColumn(*low_cardinality_data->getDictionary().getNestedNotNullableColumn())) + if (const auto * low_cardinality_data = typeid_cast(&data)) { - const auto & requested_key = index.safeGet(); - auto dictionary_index = low_cardinality_data->getDictionary().getOrFindValueIndex(requested_key); matched_idxs.reserve(offsets.size()); - if (!dictionary_index) + auto lookup_result = callWithLowCardinalityValueIndex( + *low_cardinality_data, + index.safeGet(), + [&](const auto * indexes, auto key_index) + { + MatcherLowCardinalityConst matcher{indexes, key_index}; + executeMatchKeyToIndex(offsets, matched_idxs, matcher); + }); + + /// For LowCardinality key types without a dictionary lookup, fall through so that the regular + /// dispatch reports the type error instead of silently finding no match. + if (lookup_result != LowCardinalityValueLookupResult::Unsupported) { - matched_idxs.resize_fill(offsets.size()); + if (lookup_result == LowCardinalityValueLookupResult::NotFound) + matched_idxs.resize_fill(offsets.size()); return true; } - - struct MatcherLowCardinalityStringConst - { - const ColumnLowCardinality & data; - UInt64 dictionary_index; - - bool match(size_t row_data, size_t /* row_index */) const - { - return data.getIndexAt(row_data) == dictionary_index; - } - }; - - MatcherLowCardinalityStringConst matcher{*low_cardinality_data, *dictionary_index}; - executeMatchKeyToIndex(offsets, matched_idxs, matcher); - return true; } return castColumnString( diff --git a/tests/performance/map_lowcardinality_key_lookup.xml b/tests/performance/map_lowcardinality_key_lookup.xml new file mode 100644 index 000000000000..ee61624b8bf3 --- /dev/null +++ b/tests/performance/map_lowcardinality_key_lookup.xml @@ -0,0 +1,43 @@ + + + + + CREATE TABLE map_lc_keys (id UInt64, m Map(LowCardinality(String), String)) + ENGINE = MergeTree ORDER BY id + + + + INSERT INTO map_lc_keys + SELECT number, mapFromArrays( + arrayMap(x -> concat('k', leftPad(toString(x), 2, '0')), range(20)), + arrayMap(x -> concat('v', toString(cityHash64(number, x) % 1000)), range(20))) + FROM numbers(1500000) + + + SELECT sum(ignore(m['k00'])) FROM map_lc_keys SETTINGS max_threads = 1 + SELECT sum(ignore(m['k09'])) FROM map_lc_keys SETTINGS max_threads = 1 + SELECT sum(ignore(m['k19'])) FROM map_lc_keys SETTINGS max_threads = 1 + SELECT sum(ignore(m['absent'])) FROM map_lc_keys SETTINGS max_threads = 1 + + DROP TABLE map_lc_keys + + + + + CREATE TABLE map_uuid_keys (id UInt64, m Map(UUID, DateTime64(3))) + ENGINE = MergeTree ORDER BY id + + + + INSERT INTO map_uuid_keys + SELECT number, mapFromArrays( + arrayMap(x -> reinterpretAsUUID(toFixedString(leftPad(toString(x), 16, '0'), 16)), range(20)), + arrayMap(x -> toDateTime64(number + x, 3), range(20))) + FROM numbers(1000000) + + + SELECT sum(ignore(m[reinterpretAsUUID(toFixedString('0000000000000019', 16))])) FROM map_uuid_keys SETTINGS max_threads = 1 + + DROP TABLE map_uuid_keys + diff --git a/tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.reference b/tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.reference new file mode 100644 index 000000000000..06f2d273f643 --- /dev/null +++ b/tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.reference @@ -0,0 +1,102 @@ +present key, subcolumns=0 1 b1 +present key, subcolumns=0 2 +present key, subcolumns=0 3 +present key, subcolumns=0 4 FIRST +present key, subcolumns=1 1 b1 +present key, subcolumns=1 2 +present key, subcolumns=1 3 +present key, subcolumns=1 4 FIRST +absent key, subcolumns=0 1 +absent key, subcolumns=0 2 +absent key, subcolumns=0 3 +absent key, subcolumns=0 4 +absent key, subcolumns=1 1 +absent key, subcolumns=1 2 +absent key, subcolumns=1 3 +absent key, subcolumns=1 4 +empty key, subcolumns=0 1 +empty key, subcolumns=0 2 +empty key, subcolumns=0 3 +empty key, subcolumns=0 4 empty +empty key, subcolumns=1 1 +empty key, subcolumns=1 2 +empty key, subcolumns=1 3 +empty key, subcolumns=1 4 empty +direct subcolumn 1 b1 +direct subcolumn 2 +direct subcolumn 3 +direct subcolumn 4 FIRST +compact, subcolumns=0 1 b1 +compact, subcolumns=0 2 +compact, subcolumns=0 3 +compact, subcolumns=0 4 FIRST +compact, subcolumns=1 1 b1 +compact, subcolumns=1 2 +compact, subcolumns=1 3 +compact, subcolumns=1 4 FIRST +fixed string key, subcolumns=0 1 b1 +fixed string key, subcolumns=0 2 +fixed string key, subcolumns=1 1 b1 +fixed string key, subcolumns=1 2 +wide map, subcolumns=0 17633783096591347703 10736917531793588130 +wide map, subcolumns=1 17633783096591347703 10736917531793588130 +high dictionary position, subcolumns=0 100000 100 +high dictionary position, subcolumns=1 100000 100 +random maps, subcolumns=0 9043450906938593318 +random maps, subcolumns=1 9043450906938593318 +UUID key, subcolumns=0 1 hit +UUID key, subcolumns=0 2 +UUID key, subcolumns=1 1 hit +UUID key, subcolumns=1 2 +IPv4 key, subcolumns=0 1 hit +IPv4 key, subcolumns=0 2 +IPv4 key, subcolumns=1 1 hit +IPv4 key, subcolumns=1 2 +IPv6 key, subcolumns=0 1 hit +IPv6 key, subcolumns=0 2 +IPv6 key, subcolumns=1 1 hit +IPv6 key, subcolumns=1 2 +Int128 key, subcolumns=0 1 hit +Int128 key, subcolumns=0 2 +Int128 key, subcolumns=1 1 hit +Int128 key, subcolumns=1 2 +UInt256 key, subcolumns=0 1 hit +UInt256 key, subcolumns=0 2 +UInt256 key, subcolumns=1 1 hit +UInt256 key, subcolumns=1 2 +Decimal64 key subcolumn 1 hit +Decimal64 key subcolumn 2 +DateTime64 key subcolumn 1 hit +DateTime64 key subcolumn 2 +UUID value, subcolumns=0 1 61f0c404-5cb3-11e7-907b-a6006ad3dba0 +UUID value, subcolumns=0 2 00000000-0000-0000-0000-000000000000 +UUID value, subcolumns=1 1 61f0c404-5cb3-11e7-907b-a6006ad3dba0 +UUID value, subcolumns=1 2 00000000-0000-0000-0000-000000000000 +IPv6 value, subcolumns=0 1 ::1 +IPv6 value, subcolumns=0 2 :: +IPv6 value, subcolumns=1 1 ::1 +IPv6 value, subcolumns=1 2 :: +Int128 value, subcolumns=0 1 -123456789012345678901234567890 +Int128 value, subcolumns=0 2 0 +Int128 value, subcolumns=1 1 -123456789012345678901234567890 +Int128 value, subcolumns=1 2 0 +Decimal64 value, subcolumns=0 1 1.25 +Decimal64 value, subcolumns=0 2 0 +Decimal64 value, subcolumns=1 1 1.25 +Decimal64 value, subcolumns=1 2 0 +DateTime64 value, subcolumns=0 1 2020-01-01 00:00:00.500 +DateTime64 value, subcolumns=0 2 1970-01-01 00:00:00.000 +DateTime64 value, subcolumns=1 1 2020-01-01 00:00:00.500 +DateTime64 value, subcolumns=1 2 1970-01-01 00:00:00.000 +Nullable(UUID) value, subcolumns=0 1 61f0c404-5cb3-11e7-907b-a6006ad3dba0 +Nullable(UUID) value, subcolumns=0 2 \N +Nullable(UUID) value, subcolumns=0 3 \N +Nullable(UUID) value, subcolumns=1 1 61f0c404-5cb3-11e7-907b-a6006ad3dba0 +Nullable(UUID) value, subcolumns=1 2 \N +Nullable(UUID) value, subcolumns=1 3 \N +Nullable(Decimal64) value, subcolumns=0 1 1.25 +Nullable(Decimal64) value, subcolumns=0 2 \N +Nullable(Decimal64) value, subcolumns=0 3 \N +Nullable(Decimal64) value, subcolumns=1 1 1.25 +Nullable(Decimal64) value, subcolumns=1 2 \N +Nullable(Decimal64) value, subcolumns=1 3 \N diff --git a/tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.sql b/tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.sql new file mode 100644 index 000000000000..925533b788e1 --- /dev/null +++ b/tests/queries/0_stateless/05218_map_lowcardinality_key_lookup.sql @@ -0,0 +1,203 @@ +-- m[key] on a Map with LowCardinality keys is resolved through dictionary positions instead of +-- comparing the key values. The result must be the same as with the generic comparison, and the +-- same for both the arrayElement path (optimize_functions_to_subcolumns = 0) and the subcolumn +-- path (optimize_functions_to_subcolumns = 1). + +DROP TABLE IF EXISTS t_map_lc; + +CREATE TABLE t_map_lc (id UInt64, m Map(LowCardinality(String), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; + +-- Row 2 has no 'b' at all, row 3 is an empty map, row 4 has 'b' twice and '' as a key. +INSERT INTO t_map_lc VALUES (1, map('a', 'a1', 'b', 'b1')), (2, map('a', 'a2')), (3, map()), (4, map('', 'empty', 'b', 'FIRST', 'b', 'SECOND')); + +SELECT 'present key, subcolumns=0', id, m['b'] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'present key, subcolumns=1', id, m['b'] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; + +-- A key that is in no row is absent from the dictionary as well, and is resolved without +-- looking at the rows. +SELECT 'absent key, subcolumns=0', id, m['zz'] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'absent key, subcolumns=1', id, m['zz'] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; + +-- The empty string is kept in the reserved default position of the dictionary, not in its index. +SELECT 'empty key, subcolumns=0', id, m[''] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'empty key, subcolumns=1', id, m[''] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; + +-- Reading the subcolumn directly goes through the same key lookup. +SELECT 'direct subcolumn', id, m.key_b FROM t_map_lc ORDER BY id; + +DROP TABLE t_map_lc; + +-- Compact parts exercise the other reader path. +CREATE TABLE t_map_lc (id UInt64, m Map(LowCardinality(String), String)) ENGINE = MergeTree ORDER BY id; +INSERT INTO t_map_lc VALUES (1, map('a', 'a1', 'b', 'b1')), (2, map('a', 'a2')), (3, map()), (4, map('', 'empty', 'b', 'FIRST', 'b', 'SECOND')); + +SELECT 'compact, subcolumns=0', id, m['b'] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'compact, subcolumns=1', id, m['b'] FROM t_map_lc ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; + +DROP TABLE t_map_lc; + +-- LowCardinality(FixedString) keys use the same dictionary lookup. +CREATE TABLE t_map_lc_fixed (id UInt64, m Map(LowCardinality(FixedString(3)), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_lc_fixed VALUES (1, map('aaa', 'a1', 'bbb', 'b1')), (2, map('aaa', 'a2')); + +SELECT 'fixed string key, subcolumns=0', id, m['bbb'], m['zzz'] FROM t_map_lc_fixed ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'fixed string key, subcolumns=1', id, m['bbb'], m['zzz'] FROM t_map_lc_fixed ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; + +DROP TABLE t_map_lc_fixed; + +-- A wide map, where the looked up key is the last one in every row: the shape the dictionary +-- lookup is meant to make cheap. +DROP TABLE IF EXISTS t_map_lc_wide; + +CREATE TABLE t_map_lc_wide (id UInt64, m Map(LowCardinality(String), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; + +INSERT INTO t_map_lc_wide +SELECT number, mapFromArrays(arrayMap(x -> concat('k', leftPad(toString(x), 2, '0')), range(20)), arrayMap(x -> concat('v', toString(number), '_', toString(x)), range(20))) +FROM numbers(1000); + +SELECT 'wide map, subcolumns=0', sum(cityHash64(m['k19'])), sum(cityHash64(m['k00'])) FROM t_map_lc_wide SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'wide map, subcolumns=1', sum(cityHash64(m['k19'])), sum(cityHash64(m['k00'])) FROM t_map_lc_wide SETTINGS optimize_functions_to_subcolumns = 1; + +DROP TABLE t_map_lc_wide; + +-- A key at a high dictionary position looked up over rows that do not contain it. The dictionary of +-- a part is shared by all of its blocks, so a dictionary position can be out of the range of the +-- index type of an individual block, and must not be truncated into a false match. +DROP TABLE IF EXISTS t_map_lc_many_keys; + +CREATE TABLE t_map_lc_many_keys (id UInt64, m Map(LowCardinality(String), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; + +INSERT INTO t_map_lc_many_keys +SELECT number, map(concat('k', leftPad(toString(number % 1000), 4, '0')), concat('v', toString(number))) +FROM numbers(100000); + +SELECT 'high dictionary position, subcolumns=0', count(), countIf(m['k0999'] != '') FROM t_map_lc_many_keys SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'high dictionary position, subcolumns=1', count(), countIf(m['k0999'] != '') FROM t_map_lc_many_keys SETTINGS optimize_functions_to_subcolumns = 1; + +DROP TABLE t_map_lc_many_keys; + +-- Pseudo-random wide maps with duplicate keys: both paths must agree, which is the property that +-- the first-match fix of issue #111203 established. +DROP TABLE IF EXISTS t_map_lc_random; + +CREATE TABLE t_map_lc_random (id UInt64, m Map(LowCardinality(String), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; + +INSERT INTO t_map_lc_random +SELECT + number, + mapFromArrays( + arrayMap(x -> concat('k', toString(cityHash64(number, x) % 12)), range(1 + (number % 15))), + arrayMap(x -> concat('v', toString(cityHash64(number, x, 'value'))), range(1 + (number % 15)))) +FROM numbers(20000); + +SELECT 'random maps, subcolumns=0', sum(cityHash64(m['k0'], m['k5'], m['k11'], m['k99'])) FROM t_map_lc_random SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'random maps, subcolumns=1', sum(cityHash64(m['k0'], m['k5'], m['k11'], m['k99'])) FROM t_map_lc_random SETTINGS optimize_functions_to_subcolumns = 1; + +DROP TABLE t_map_lc_random; + +-- Key types that used to be compared through virtual compareAt. + +DROP TABLE IF EXISTS t_map_key_types; + +CREATE TABLE t_map_key_types (id UInt64, m Map(UUID, String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types VALUES (1, map('61f0c404-5cb3-11e7-907b-a6006ad3dba0', 'hit')), (2, map('00000000-0000-0000-0000-000000000001', 'other')); +SELECT 'UUID key, subcolumns=0', id, m[toUUID('61f0c404-5cb3-11e7-907b-a6006ad3dba0')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'UUID key, subcolumns=1', id, m[toUUID('61f0c404-5cb3-11e7-907b-a6006ad3dba0')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_key_types; + +CREATE TABLE t_map_key_types (id UInt64, m Map(IPv4, String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types VALUES (1, map('1.2.3.4', 'hit')), (2, map('5.6.7.8', 'other')); +SELECT 'IPv4 key, subcolumns=0', id, m[toIPv4('1.2.3.4')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'IPv4 key, subcolumns=1', id, m[toIPv4('1.2.3.4')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_key_types; + +CREATE TABLE t_map_key_types (id UInt64, m Map(IPv6, String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types VALUES (1, map('::1', 'hit')), (2, map('::2', 'other')); +SELECT 'IPv6 key, subcolumns=0', id, m[toIPv6('::1')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'IPv6 key, subcolumns=1', id, m[toIPv6('::1')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_key_types; + +-- The wide integers are built from strings, because a literal that wide goes through Float64 and +-- loses precision. +CREATE TABLE t_map_key_types (id UInt64, m Map(Int128, String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types SELECT 1, map(toInt128('-123456789012345678901234567890'), 'hit'); +INSERT INTO t_map_key_types SELECT 2, map(toInt128(1), 'other'); +SELECT 'Int128 key, subcolumns=0', id, m[toInt128('-123456789012345678901234567890')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'Int128 key, subcolumns=1', id, m[toInt128('-123456789012345678901234567890')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_key_types; + +CREATE TABLE t_map_key_types (id UInt64, m Map(UInt256, String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types SELECT 1, map(toUInt256('12345678901234567890123456789012345678901234567890'), 'hit'); +INSERT INTO t_map_key_types SELECT 2, map(toUInt256(1), 'other'); +SELECT 'UInt256 key, subcolumns=0', id, m[toUInt256('12345678901234567890123456789012345678901234567890')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'UInt256 key, subcolumns=1', id, m[toUInt256('12345678901234567890123456789012345678901234567890')] FROM t_map_key_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_key_types; + +-- Decimal and DateTime64 keys are rejected by m[key], but reachable through the subcolumn name. + +CREATE TABLE t_map_key_types (id UInt64, m Map(Decimal64(2), String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types SELECT 1, map(toDecimal64(1.5, 2), 'hit', toDecimal64(2.5, 2), 'other'); +INSERT INTO t_map_key_types SELECT 2, map(toDecimal64(2.5, 2), 'other'); +SELECT 'Decimal64 key subcolumn', id, `m.key_1.50` FROM t_map_key_types ORDER BY id; +DROP TABLE t_map_key_types; + +-- The timezone is pinned, because the key is rendered into the subcolumn name. +CREATE TABLE t_map_key_types (id UInt64, m Map(DateTime64(3, 'UTC'), String)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_key_types SELECT 1, map(toDateTime64('2020-01-01 00:00:00.500', 3, 'UTC'), 'hit', toDateTime64('2021-01-01 00:00:00.500', 3, 'UTC'), 'other'); +INSERT INTO t_map_key_types SELECT 2, map(toDateTime64('2021-01-01 00:00:00.500', 3, 'UTC'), 'other'); +SELECT 'DateTime64 key subcolumn', id, `m.key_2020-01-01 00:00:00.500` FROM t_map_key_types ORDER BY id; +DROP TABLE t_map_key_types; + +-- Value types that used to be copied through virtual insertFrom. The rows not holding the key +-- cover the default-insert branch, the Nullable variants the null map. + +DROP TABLE IF EXISTS t_map_value_types; + +CREATE TABLE t_map_value_types (id UInt64, m Map(String, UUID)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types VALUES (1, map('a', '61f0c404-5cb3-11e7-907b-a6006ad3dba0')), (2, map('b', '00000000-0000-0000-0000-000000000001')); +SELECT 'UUID value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'UUID value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; + +CREATE TABLE t_map_value_types (id UInt64, m Map(String, IPv6)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types VALUES (1, map('a', '::1')), (2, map('b', '::2')); +SELECT 'IPv6 value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'IPv6 value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; + +CREATE TABLE t_map_value_types (id UInt64, m Map(String, Int128)) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types SELECT 1, map('a', toInt128('-123456789012345678901234567890')); +INSERT INTO t_map_value_types SELECT 2, map('b', toInt128(1)); +SELECT 'Int128 value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'Int128 value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; + +CREATE TABLE t_map_value_types (id UInt64, m Map(String, Decimal64(3))) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types VALUES (1, map('a', 1.25)), (2, map('b', 2.5)); +SELECT 'Decimal64 value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'Decimal64 value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; + +-- The timezone is pinned, because the missing key renders the epoch as the default value. +CREATE TABLE t_map_value_types (id UInt64, m Map(String, DateTime64(3, 'UTC'))) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types VALUES (1, map('a', '2020-01-01 00:00:00.500')), (2, map('b', '2021-01-01 00:00:00.500')); +SELECT 'DateTime64 value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'DateTime64 value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; + +CREATE TABLE t_map_value_types (id UInt64, m Map(String, Nullable(UUID))) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types VALUES (1, map('a', '61f0c404-5cb3-11e7-907b-a6006ad3dba0')), (2, map('a', NULL)), (3, map('b', NULL)); +SELECT 'Nullable(UUID) value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'Nullable(UUID) value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; + +CREATE TABLE t_map_value_types (id UInt64, m Map(String, Nullable(Decimal64(3)))) ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_value_types VALUES (1, map('a', 1.25)), (2, map('a', NULL)), (3, map('b', NULL)); +SELECT 'Nullable(Decimal64) value, subcolumns=0', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'Nullable(Decimal64) value, subcolumns=1', id, m['a'] FROM t_map_value_types ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_value_types; diff --git a/tests/queries/0_stateless/05234_map_key_lookup_nan.reference b/tests/queries/0_stateless/05234_map_key_lookup_nan.reference new file mode 100644 index 000000000000..62d2b83565c4 --- /dev/null +++ b/tests/queries/0_stateless/05234_map_key_lookup_nan.reference @@ -0,0 +1,22 @@ +Float64 1 nan-hit one-five +Float64 2 other +Float64 3 plain +Float32 1 nan-hit one-five +Float32 2 other +Float32 3 plain +BFloat16 1 nan-hit one-five +BFloat16 2 other +BFloat16 3 plain +LowCardinality(Float64) 1 nan-hit one-five +LowCardinality(Float64) 2 other +LowCardinality(Float64) 3 plain +Tuple(Float64, UInt8) 1 nan-hit one-five +Tuple(Float64, UInt8) 2 other +Tuple(Float64, UInt8) 3 plain +Tuple(Nullable(UInt8), UInt8) 1 null-hit five +Tuple(Nullable(UInt8), UInt8) 2 other +Tuple(Nullable(UInt8), UInt8) 3 plain +NaN value 1 1 1.5 +NaN value 2 0 2.5 +NaN value 1 1 1.5 +NaN value 2 0 2.5 diff --git a/tests/queries/0_stateless/05234_map_key_lookup_nan.sql b/tests/queries/0_stateless/05234_map_key_lookup_nan.sql new file mode 100644 index 000000000000..bfcb9deffeee --- /dev/null +++ b/tests/queries/0_stateless/05234_map_key_lookup_nan.sql @@ -0,0 +1,61 @@ +-- A `NaN` key of a Map is found by a requested `NaN` and by nothing else, and a requested `NaN` +-- finds nothing but a `NaN` key. The subcolumn extractor used to ask `compareAt` with a direction +-- hint of zero, which answers with the hint itself when one side is a `NaN`, so a row holding a +-- `NaN` key matched every requested key and a requested `NaN` matched every key. + +SET allow_suspicious_low_cardinality_types = 1; + +DROP TABLE IF EXISTS t_map_nan_key; + +-- The specialized matchers: BFloat16, Float32 and Float64 keys are compared value by value. + +CREATE TABLE t_map_nan_key (id UInt64, m Map(Float64, String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map(nan, 'nan-hit', 1.5, 'one-five')), (2, map(2.5, 'other')), (3, map(1.5, 'plain')); +SELECT 'Float64', id, `m.key_nan`, `m.key_1.5`, `m.key_2.5` FROM t_map_nan_key ORDER BY id; +DROP TABLE t_map_nan_key; + +CREATE TABLE t_map_nan_key (id UInt64, m Map(Float32, String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map(nan, 'nan-hit', 1.5, 'one-five')), (2, map(2.5, 'other')), (3, map(1.5, 'plain')); +SELECT 'Float32', id, `m.key_nan`, `m.key_1.5`, `m.key_2.5` FROM t_map_nan_key ORDER BY id; +DROP TABLE t_map_nan_key; + +CREATE TABLE t_map_nan_key (id UInt64, m Map(BFloat16, String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map(nan, 'nan-hit', 1.5, 'one-five')), (2, map(2.5, 'other')), (3, map(1.5, 'plain')); +SELECT 'BFloat16', id, `m.key_nan`, `m.key_1.5`, `m.key_2.5` FROM t_map_nan_key ORDER BY id; +DROP TABLE t_map_nan_key; + +-- The generic matcher: a LowCardinality dictionary that does not hold strings, and a Tuple, are +-- compared through the virtual `compareAt`. + +CREATE TABLE t_map_nan_key (id UInt64, m Map(LowCardinality(Float64), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map(nan, 'nan-hit', 1.5, 'one-five')), (2, map(2.5, 'other')), (3, map(1.5, 'plain')); +SELECT 'LowCardinality(Float64)', id, `m.key_nan`, `m.key_1.5`, `m.key_2.5` FROM t_map_nan_key ORDER BY id; +DROP TABLE t_map_nan_key; + +CREATE TABLE t_map_nan_key (id UInt64, m Map(Tuple(Float64, UInt8), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map((nan, 1), 'nan-hit', (1.5, 1), 'one-five')), (2, map((2.5, 1), 'other')), (3, map((1.5, 1), 'plain')); +SELECT 'Tuple(Float64, UInt8)', id, `m.key_(nan,1)`, `m.key_(1.5,1)`, `m.key_(2.5,1)` FROM t_map_nan_key ORDER BY id; +DROP TABLE t_map_nan_key; + +-- The same hint decides where a NULL goes, so a NULL inside a composite key used to be equal to +-- every other value as well. + +CREATE TABLE t_map_nan_key (id UInt64, m Map(Tuple(Nullable(UInt8), UInt8), String)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map((NULL, 1), 'null-hit', (5, 1), 'five')), (2, map((7, 1), 'other')), (3, map((5, 1), 'plain')); +SELECT 'Tuple(Nullable(UInt8), UInt8)', id, `m.key_(NULL,1)`, `m.key_(5,1)`, `m.key_(7,1)` FROM t_map_nan_key ORDER BY id; +DROP TABLE t_map_nan_key; + +-- A `NaN` in the value is copied as it is, whichever key it is reached by. + +CREATE TABLE t_map_nan_key (id UInt64, m Map(String, Float64)) +ENGINE = MergeTree ORDER BY id SETTINGS min_bytes_for_wide_part = 0; +INSERT INTO t_map_nan_key VALUES (1, map('a', nan, 'b', 1.5)), (2, map('b', 2.5)); +SELECT 'NaN value', id, isNaN(m['a']), m['b'] FROM t_map_nan_key ORDER BY id SETTINGS optimize_functions_to_subcolumns = 0; +SELECT 'NaN value', id, isNaN(m['a']), m['b'] FROM t_map_nan_key ORDER BY id SETTINGS optimize_functions_to_subcolumns = 1; +DROP TABLE t_map_nan_key; From fec0ce2bc4e25b3a7255d95552c088435a757c03 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 12:20:51 +0000 Subject: [PATCH 049/185] Backport #121375 to 26.8: Use text and token skip indexes for IN with transform_null_in = 1 (Human alternative for #112785) --- .../MergeTreeIndexBloomFilterText.cpp | 22 +-- .../MergeTree/MergeTreeIndexConditionText.cpp | 47 +++++- ...xt_token_index_transform_null_in.reference | 35 +++++ ...346_text_token_index_transform_null_in.sql | 136 ++++++++++++++++++ 4 files changed, 223 insertions(+), 17 deletions(-) create mode 100644 tests/queries/0_stateless/02346_text_token_index_transform_null_in.reference create mode 100644 tests/queries/0_stateless/02346_text_token_index_transform_null_in.sql diff --git a/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp b/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp index c36427c44164..d9d478b11dfb 100644 --- a/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp @@ -387,16 +387,11 @@ bool MergeTreeConditionBloomFilterText::extractAtomFromTree(const RPNBuilderTree { if (tryPrepareSetBloomFilter(left_argument, right_argument, out)) { - if (function_name == "notIn") - { - out.function = RPNElement::FUNCTION_NOT_IN; - return true; - } - if (function_name == "in") - { - out.function = RPNElement::FUNCTION_IN; - return true; - } + /// `transform_null_in = 1` renames the family; a NULL element is refused above. + const bool negated = function_name == "notIn" || function_name == "globalNotIn" + || function_name == "notNullIn" || function_name == "globalNotNullIn"; + out.function = negated ? RPNElement::FUNCTION_NOT_IN : RPNElement::FUNCTION_IN; + return true; } } else if (function_name == "equals" || @@ -807,7 +802,8 @@ bool MergeTreeConditionBloomFilterText::tryPrepareSetBloomFilter( for (const auto & prepared_set_data_type : prepared_set->getDataTypes()) { - auto prepared_set_data_type_id = prepared_set_data_type->getTypeId(); + /// A `Nullable` key keeps the wrapper on its elements at `transform_null_in = 1`. + auto prepared_set_data_type_id = removeNullable(prepared_set_data_type)->getTypeId(); if (prepared_set_data_type_id != TypeIndex::String && prepared_set_data_type_id != TypeIndex::FixedString) return false; } @@ -828,6 +824,10 @@ bool MergeTreeConditionBloomFilterText::tryPrepareSetBloomFilter( for (size_t row = 0; row < prepared_set_total_row_count; ++row) { + /// A NULL element also matches the column's NULL rows, which the filter cannot express. + if (column->isNullAt(row)) + return false; + bloom_filters.back().emplace_back(params); auto ref = column->getDataAt(row); forEachTokenToBloomFilter(*tokenizer, ref.data(), ref.size(), bloom_filters.back().back()); diff --git a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp index d8c5094ce71c..d64931e0ed45 100644 --- a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp @@ -30,6 +30,7 @@ #include #include #include +#include #include #include #include @@ -650,7 +651,8 @@ bool MergeTreeIndexConditionText::traverseAtomNode(const RPNBuilderTreeNode & no auto lhs_argument = function.getArgumentAt(0); auto rhs_argument = function.getArgumentAt(1); - if ((function_name == "in" || function_name == "globalIn") + if ((function_name == "in" || function_name == "globalIn" + || function_name == "nullIn" || function_name == "globalNullIn") && tryPrepareSetForTextSearch(lhs_argument, rhs_argument, function_name, out)) { out.function = RPNElement::FUNCTION_HAS_ANY_ELEMENTS; @@ -894,6 +896,19 @@ static void validateRegexpPatterns(const Array & patterns, const Settings & sett #endif } +/// The value an absent map key reads: `''`, or all NUL when the value type is `FixedString`. +/// `mapValues` stores neither. +static bool isMapValueDefault(std::string_view value, const Block & header) +{ + /// A text index is always defined on a single expression. + chassert(header.columns() == 1); + auto value_type = removeNullable(removeLowCardinality(header.getByPosition(0).type)); + if (const auto * array_type = typeid_cast(value_type.get())) + value_type = removeNullable(removeLowCardinality(array_type->getNestedType())); + + return value.empty() || (isFixedString(value_type) && value.find_first_not_of('\0') == std::string_view::npos); +} + bool MergeTreeIndexConditionText::traverseFunctionNode( const RPNBuilderFunctionTreeNode & function_node, const RPNBuilderTreeNode & index_column_node, @@ -943,7 +958,7 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( auto & [map_column_name, _] = *parsed; if (header.has(fmt::format("mapValues({})", map_column_name)) && value_field.getType() == Field::Types::String - && !value_field.safeGet().empty()) + && !isMapValueDefault(value_field.safeGet(), header)) { has_index_column = true; direct_read_mode = getHintOrNoneMode(); @@ -1641,7 +1656,7 @@ bool MergeTreeIndexConditionText::traverseMapElementValueNode(const RPNBuilderTr /// for functions like `func(arrayElement(m, 'const_key'), ...)`. /// If index can be used, than we can analyze the index as for scalar string column /// because `arrayElement(m, 'const_key')` projects Array(String) to String. - if (const_value.getType() != Field::Types::String || const_value.safeGet().empty()) + if (const_value.getType() != Field::Types::String || isMapValueDefault(const_value.safeGet(), header)) return false; return hasIndexForMapElementValue(index_column_node); @@ -1721,10 +1736,17 @@ bool MergeTreeIndexConditionText::tryPrepareSetForTextSearch( { std::optional set_key_position; + /// `m['key']` answered by a `mapValues(m)` index: an absent key reads the value type's default. + bool has_index_for_map_element_value = false; + auto has_index = [&](const RPNBuilderTreeNode & node) { + if (hasIndexForMapElementValue(node)) + { + has_index_for_map_element_value = true; + return true; + } return hasIndexForColumn(node.getColumnName()) - || hasIndexForMapElementValue(node) || tryMatchNodeToJSONIndex(node, header, "JSONAllValues"); }; @@ -1773,19 +1795,32 @@ bool MergeTreeIndexConditionText::tryPrepareSetForTextSearch( return false; const auto & set_column = *columns[*set_key_position]; - if (!WhichDataType(set_column.getDataType()).isStringOrFixedString()) + + /// With setting `transform_null_in = 1`, the IN set can be nullable. + const auto * set_column_nullable = typeid_cast(&set_column); + const auto & set_column_values = set_column_nullable ? set_column_nullable->getNestedColumn() : set_column; + + if (!WhichDataType(set_column_values.getDataType()).isStringOrFixedString()) return false; size_t total_row_count = prepared_set->getTotalRowCount(); for (size_t row = 0; row < total_row_count; ++row) { + /// The atom is an OR over the elements, and the index skips NULL rows when building a + /// granule, so a NULL element is a disjunct it cannot bind. Decline the atom. + if (set_column.isNullAt(row)) + { + out.text_search_queries.clear(); + return false; + } + auto ref = set_column.getDataAt(row); /// Reject the index usage when there is an empty string in the set. /// The condition with such a predicate will be always true on granule. /// See MergeTreeIndexGranuleText::hasAllQueryTokensOrEmpty. - if (ref.empty()) + if (ref.empty() || (has_index_for_map_element_value && isMapValueDefault(ref, header))) { out.text_search_queries.clear(); return false; diff --git a/tests/queries/0_stateless/02346_text_token_index_transform_null_in.reference b/tests/queries/0_stateless/02346_text_token_index_transform_null_in.reference new file mode 100644 index 000000000000..7f82e6a6845d --- /dev/null +++ b/tests/queries/0_stateless/02346_text_token_index_transform_null_in.reference @@ -0,0 +1,35 @@ +Granules: 1/2 +1 +1 +1 +Granules: 1/2 +1 +7 +1 +7 +1 +Granules: 1/2 +1 +7 +1 +7 +1 +1 +1 +Granules: 1/2 +1 +7 +1 +7 +1 +Granules: 1/2 +2 +3 +1 +1 +2 +2 +1 +1 +1 +1 diff --git a/tests/queries/0_stateless/02346_text_token_index_transform_null_in.sql b/tests/queries/0_stateless/02346_text_token_index_transform_null_in.sql new file mode 100644 index 000000000000..6127b05004cd --- /dev/null +++ b/tests/queries/0_stateless/02346_text_token_index_transform_null_in.sql @@ -0,0 +1,136 @@ +-- Tests that the `text`, `tokenbf_v1`, `ngrambf_v1` and `sparse_grams` skip indexes prune an `IN` +-- when `transform_null_in = 1`, where the predicate arrives as `nullIn`/`globalNullIn`, and that a +-- set holding a NULL element is refused because `nullIn` also matches the column's NULL rows. +-- +-- `index_granularity = 4` over 8 rows, so every granule is mixed: granule 0 holds word0..word3. + +SET enable_full_text_index = 1; +SET transform_null_in = 1; + +DROP TABLE IF EXISTS tab; + +-- The same block runs for every index type. `tokenbf_v1`, `ngrambf_v1` and `sparse_grams` reject a +-- `Nullable` column at DDL, so the common block uses `String` and NULL enters through the set only. + +-- text + +CREATE TABLE tab (s String, INDEX idx s TYPE text(tokenizer = splitByNonAlpha)) ENGINE = MergeTree ORDER BY tuple() SETTINGS index_granularity = 4; +INSERT INTO tab SELECT 'word' || toString(number) FROM numbers(8); + +SELECT extract(explain, 'Granules: \\d+/\\d+') FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE s IN ('word1')) WHERE explain LIKE '%Granules: %/%'; +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- The text index does not model NOT IN at all, unlike the token bloom filter family below. +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +-- Without transform_null_in the same predicates keep the globalIn/globalNotIn spellings, which must prune too. +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +-- A Nullable element type is what transform_null_in = 1 adds, so a null-free set of it must prune. +SELECT count() FROM tab WHERE s IN (SELECT CAST('word1', 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s IN (SELECT CAST(NULL, 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } + +-- tokenbf_v1 + +DROP TABLE tab; +CREATE TABLE tab (s String, INDEX idx s TYPE tokenbf_v1(256, 2, 0)) ENGINE = MergeTree ORDER BY tuple() SETTINGS index_granularity = 4; +INSERT INTO tab SELECT 'word' || toString(number) FROM numbers(8); + +SELECT extract(explain, 'Granules: \\d+/\\d+') FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE s IN ('word1')) WHERE explain LIKE '%Granules: %/%'; +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- NOT IN never prunes on these indexes, but the index must still be used. +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- Without transform_null_in the same predicates keep the globalIn/globalNotIn spellings, which must prune too. +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +-- A Nullable element type is what transform_null_in = 1 adds, so a null-free set of it must prune. +SELECT count() FROM tab WHERE s IN (SELECT CAST('word1', 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s IN (SELECT CAST(NULL, 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } + +-- ngrambf_v1 + +DROP TABLE tab; +CREATE TABLE tab (s String, INDEX idx s TYPE ngrambf_v1(3, 256, 2, 0)) ENGINE = MergeTree ORDER BY tuple() SETTINGS index_granularity = 4; +INSERT INTO tab SELECT 'word' || toString(number) FROM numbers(8); + +SELECT extract(explain, 'Granules: \\d+/\\d+') FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE s IN ('word1')) WHERE explain LIKE '%Granules: %/%'; +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- NOT IN never prunes on these indexes, but the index must still be used. +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- Without transform_null_in the same predicates keep the globalIn/globalNotIn spellings, which must prune too. +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +-- A Nullable element type is what transform_null_in = 1 adds, so a null-free set of it must prune. +SELECT count() FROM tab WHERE s IN (SELECT CAST('word1', 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s IN (SELECT CAST(NULL, 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } + +-- A FixedString element has its NUL padding stripped before tokenization, so it keeps pruning. +SELECT count() FROM tab WHERE s IN (SELECT toFixedString('word1', 12)) SETTINGS force_data_skipping_indices = 'idx'; +SELECT (SELECT count() FROM tab WHERE s IN (SELECT toFixedString('word1', 12))) = (SELECT count() FROM tab WHERE s IN (SELECT toFixedString('word1', 12)) SETTINGS use_skip_indexes = 0); + +-- sparse_grams + +DROP TABLE tab; +CREATE TABLE tab (s String, INDEX idx s TYPE sparse_grams(3, 100, 512, 2, 0)) ENGINE = MergeTree ORDER BY tuple() SETTINGS index_granularity = 4; +INSERT INTO tab SELECT 'word' || toString(number) FROM numbers(8); + +SELECT extract(explain, 'Granules: \\d+/\\d+') FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE s IN ('word1')) WHERE explain LIKE '%Granules: %/%'; +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- NOT IN never prunes on these indexes, but the index must still be used. +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS force_data_skipping_indices = 'idx'; +-- Without transform_null_in the same predicates keep the globalIn/globalNotIn spellings, which must prune too. +SELECT count() FROM tab WHERE s GLOBAL IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s GLOBAL NOT IN ('word1') SETTINGS transform_null_in = 0, force_data_skipping_indices = 'idx'; +-- A Nullable element type is what transform_null_in = 1 adds, so a null-free set of it must prune. +SELECT count() FROM tab WHERE s IN (SELECT CAST('word1', 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE s IN (SELECT CAST(NULL, 'Nullable(String)')) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } + +-- A Nullable column: only the text index accepts one, and only there can a NULL element of the set +-- meet a NULL row of the column. The set is a disjunction, so one NULL element refuses the index +-- whatever else the set holds. + +DROP TABLE tab; +CREATE TABLE tab (s Nullable(String), INDEX idx s TYPE text(tokenizer = splitByNonAlpha)) ENGINE = MergeTree ORDER BY tuple() SETTINGS index_granularity = 4; +INSERT INTO tab SELECT if(number = 7, NULL, 'word' || toString(number)) FROM numbers(8); + +SELECT extract(explain, 'Granules: \\d+/\\d+') FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE s IN ('word1')) WHERE explain LIKE '%Granules: %/%'; +SELECT count() FROM tab WHERE s IN ('word1', NULL) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +SELECT count() FROM tab WHERE s IN ('word1', NULL); +SELECT count() FROM tab WHERE s IN ('word1', NULL, 'word2') SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +SELECT count() FROM tab WHERE s IN ('word1', NULL, 'word2'); + +-- A tuple set is one Tuple column unpacked by position, so the indexed component is addressed by it. + +DROP TABLE tab; +CREATE TABLE tab (id UInt64, s String, INDEX idx s TYPE text(tokenizer = splitByNonAlpha)) ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 4; +INSERT INTO tab SELECT number, 'word' || toString(number) FROM numbers(8); + +SELECT count() FROM tab WHERE (id, s) IN ((1, 'word1')) SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE (id, s) IN (SELECT tuple(number, 'word1') FROM numbers(8)) SETTINGS force_data_skipping_indices = 'idx'; + +-- An absent map key reads the value type's default, which mapValues never stores. The FixedString default +-- is all NUL and the array tokenizer keeps the padding, so it must be recognised as the default. + +DROP TABLE tab; +CREATE TABLE tab (id UInt64, m Map(String, FixedString(4)), INDEX idx mapValues(m) TYPE text(tokenizer = array)) ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; +INSERT INTO tab VALUES (0, {'k':'val0'}), (1, {'other':'xxxx'}), (2, {'k':'val2'}), (3, {}); + +SELECT count() FROM tab WHERE m['k'] IN (toFixedString('', 4)) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +SELECT count() FROM tab WHERE m['k'] IN (toFixedString('', 4)); +SELECT count() FROM tab WHERE m['k'] = toFixedString('', 4) SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +SELECT count() FROM tab WHERE m['k'] = toFixedString('', 4); +SELECT count() FROM tab WHERE m['k'] IN ('val0') SETTINGS force_data_skipping_indices = 'idx'; + +-- A String map value defaults to '' only, so an all-NUL value is a real value and must prune. + +DROP TABLE tab; +CREATE TABLE tab (id UInt64, m Map(String, String), INDEX idx mapValues(m) TYPE text(tokenizer = array)) ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; +INSERT INTO tab SELECT 0, map('k', unhex('00')); +INSERT INTO tab SELECT number, map('k', 'val' || toString(number)) FROM numbers(1, 2); +INSERT INTO tab SELECT 3, map('other', 'x'); + +SELECT count() FROM tab WHERE m['k'] = unhex('00') SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE m['k'] IN (unhex('00')) SETTINGS force_data_skipping_indices = 'idx'; +SELECT count() FROM tab WHERE m['k'] = '' SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +SELECT count() FROM tab WHERE m['k'] IN ('') SETTINGS force_data_skipping_indices = 'idx'; -- { serverError INDEX_NOT_USED } +SELECT count() FROM tab WHERE m['k'] = ''; + +DROP TABLE tab; From 7b3ca31f99fd6ec8d2eaf0a983006dea8df72956 Mon Sep 17 00:00:00 2001 From: Jimmy Aguilar Mena Date: Fri, 25 Sep 2026 13:29:37 +0000 Subject: [PATCH 050/185] Add missing includes --- src/Storages/MergeTree/MergeTreeIndexConditionText.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp index d64931e0ed45..1640e2b731c1 100644 --- a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp @@ -7,6 +7,8 @@ #include #include #include +#include +#include #include #include #include From c8e1f583abbb0a0581f550eee23add17d1a1a658 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 13:49:46 +0000 Subject: [PATCH 051/185] Backport #111287 to 26.8: Fix double free when finalizing -State aggregates under looping combinators --- .../Combinators/AggregateFunctionArray.h | 5 + ...gregateFunctionCombinatorsArgMinArgMax.cpp | 5 + .../Combinators/AggregateFunctionDistinct.h | 5 + .../Combinators/AggregateFunctionForEach.h | 57 +++++++- .../Combinators/AggregateFunctionIf.h | 5 + .../Combinators/AggregateFunctionMap.cpp | 132 ++++++++++++++---- .../Combinators/AggregateFunctionMerge.h | 5 + .../Combinators/AggregateFunctionNull.h | 39 +++++- .../Combinators/AggregateFunctionOrFill.h | 63 +++++++-- .../Combinators/AggregateFunctionResample.h | 71 +++++++++- .../AggregateFunctionSimpleState.h | 5 + .../Combinators/AggregateFunctionState.h | 31 +++- .../Combinators/AggregateFunctionTuple.cpp | 7 + .../Combinators/AggregateFunctionTuple.h | 65 ++++++++- src/AggregateFunctions/IAggregateFunction.h | 9 ++ src/Columns/ColumnAggregateFunction.cpp | 8 ++ src/Columns/ColumnAggregateFunction.h | 16 +++ src/Common/FailPoint.cpp | 4 +- ...state_resample_finalize_no_crash.reference | 14 ++ ...array_state_resample_finalize_no_crash.sql | 102 ++++++++++++++ 20 files changed, 590 insertions(+), 58 deletions(-) create mode 100644 tests/queries/0_stateless/04614_group_array_state_resample_finalize_no_crash.reference create mode 100644 tests/queries/0_stateless/04614_group_array_state_resample_finalize_no_crash.sql diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionArray.h b/src/AggregateFunctions/Combinators/AggregateFunctionArray.h index 614bf5aa1470..a98b076c532f 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionArray.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionArray.h @@ -231,6 +231,11 @@ class AggregateFunctionArray final : public IAggregateFunctionHelperinsertMergeResultInto(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + nested_func->rollbackInsertResult(place, to); + } + bool allocatesMemoryInArena() const override { return nested_func->allocatesMemoryInArena(); diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionCombinatorsArgMinArgMax.cpp b/src/AggregateFunctions/Combinators/AggregateFunctionCombinatorsArgMinArgMax.cpp index 2dd9f6bfe11f..f73eed07a9ab 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionCombinatorsArgMinArgMax.cpp +++ b/src/AggregateFunctions/Combinators/AggregateFunctionCombinatorsArgMinArgMax.cpp @@ -222,6 +222,11 @@ class AggregateFunctionCombinatorArgMinArgMax final : public IAggregateFunctionH nested_function->insertMergeResultInto(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + nested_function->rollbackInsertResult(place, to); + } + AggregateFunctionPtr getNestedFunction() const override { return nested_function; } }; diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionDistinct.h b/src/AggregateFunctions/Combinators/AggregateFunctionDistinct.h index 69c2af03fef1..036884376681 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionDistinct.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionDistinct.h @@ -291,6 +291,11 @@ class AggregateFunctionDistinct final : public IAggregateFunctionDataHelperinsertMergeResultInto(getNestedPlace(place), to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + nested_func->rollbackInsertResult(getNestedPlace(place), to); + } + size_t sizeOfData() const override { return prefix_size + nested_func->sizeOfData(); diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionForEach.h b/src/AggregateFunctions/Combinators/AggregateFunctionForEach.h index 219c94ec9cd3..17b499acd3bf 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionForEach.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionForEach.h @@ -342,17 +342,15 @@ class AggregateFunctionForEach final : public IAggregateFunctionDataHelper - void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const + void transferElements(AggregateDataPtr __restrict place, ColumnArray & arr_to, size_t & transferred, Arena * arena) const { AggregateFunctionForEachData & state = data(place); - - ColumnArray & arr_to = assert_cast(to); - ColumnArray::Offsets & offsets_to = arr_to.getOffsets(); IColumn & elems_to = arr_to.getData(); - char * nested_state = state.array_of_aggregate_datas; - for (size_t i = 0; i < state.dynamic_array_size; ++i) + char * nested_state = state.array_of_aggregate_datas + transferred * nested_size_of_data; + for (; transferred < state.dynamic_array_size; ++transferred) { if constexpr (merge) nested_func->insertMergeResultInto(nested_state, elems_to, arena); @@ -361,9 +359,43 @@ class AggregateFunctionForEach final : public IAggregateFunctionDataHelper + void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const + { + ColumnArray & arr_to = assert_cast(to); + size_t transferred = 0; + + if constexpr (!merge) + { + /// A nested function that is not a state aliases nothing and need not be atomic. + if (nested_func->isState()) + { + const size_t offsets_before = arr_to.getOffsets().size(); + + try + { + transferElements(place, arr_to, transferred, arena); + } + catch (...) + { + arr_to.getOffsets().resize_assume_reserved(offsets_before); + const char * nested_state = data(place).array_of_aggregate_datas; + for (size_t i = transferred; i-- > 0;) + nested_func->rollbackInsertResult(nested_state + i * nested_size_of_data, arr_to.getData()); + throw; + } + + return; + } + } + + transferElements(place, arr_to, transferred, arena); + } + void insertResultInto(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const override { insertResultIntoImpl(place, to, arena); @@ -374,6 +406,19 @@ class AggregateFunctionForEach final : public IAggregateFunctionDataHelper(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + const AggregateFunctionForEachData & state = data(place); + + ColumnArray & arr_to = assert_cast(to); + ColumnArray::Offsets & offsets_to = arr_to.getOffsets(); + + offsets_to.resize_assume_reserved(offsets_to.size() - 1); + const char * nested_state = state.array_of_aggregate_datas; + for (size_t i = state.dynamic_array_size; i-- > 0;) + nested_func->rollbackInsertResult(nested_state + i * nested_size_of_data, arr_to.getData()); + } + bool allocatesMemoryInArena() const override { return true; diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionIf.h b/src/AggregateFunctions/Combinators/AggregateFunctionIf.h index ed6c8e84f8a6..f5cdb6b30e84 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionIf.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionIf.h @@ -240,6 +240,11 @@ class AggregateFunctionIf final : public IAggregateFunctionHelperinsertMergeResultInto(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + nested_func->rollbackInsertResult(place, to); + } + bool allocatesMemoryInArena() const override { return nested_func->allocatesMemoryInArena(); diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionMap.cpp b/src/AggregateFunctions/Combinators/AggregateFunctionMap.cpp index b711b1288624..1cdc6b1881f2 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionMap.cpp +++ b/src/AggregateFunctions/Combinators/AggregateFunctionMap.cpp @@ -15,6 +15,7 @@ #include #include #include +#include #include #include @@ -154,6 +155,23 @@ class AggregateFunctionMap final return map_type->getKeyType(); } + /// Reads one key out of a key column in the representation `merged_maps` is keyed by. + typename Data::SearchType keyAt(const IColumn & key_column, size_t row) const + { + if constexpr (std::is_same_v) + { + if (key_type->getTypeId() == TypeIndex::FixedString) + return assert_cast(key_column).getDataAt(row); + if (key_type->getTypeId() == TypeIndex::IPv6) + return assert_cast(key_column).getDataAt(row); + return assert_cast(key_column).getDataAt(row); + } + else + { + return assert_cast &>(key_column).getData()[row]; + } + } + void add(AggregateDataPtr __restrict place, const IColumn ** columns, size_t row_num, Arena * arena) const override { const auto & map_column = assert_cast(*columns[0]); @@ -170,24 +188,7 @@ class AggregateFunctionMap final for (size_t i = 0; i < size; ++i) { - typename Data::SearchType key; - - if constexpr (std::is_same_v) - { - std::string_view key_ref; - if (key_type->getTypeId() == TypeIndex::FixedString) - key_ref = assert_cast(key_column).getDataAt(offset + i); - else if (key_type->getTypeId() == TypeIndex::IPv6) - key_ref = assert_cast(key_column).getDataAt(offset + i); - else - key_ref = assert_cast(key_column).getDataAt(offset + i); - - key = key_ref; - } - else - { - key = assert_cast &>(key_column).getData()[offset + i]; - } + const typename Data::SearchType key = keyAt(key_column, offset + i); auto it = merged_maps.find(key); @@ -339,13 +340,40 @@ class AggregateFunctionMap final } } + /// `transferred` counts the values whose transfer returned, so a caller that catches can undo those. + template + void transferValues( + AggregateDataPtr __restrict place, + ColumnMap & map_column, + const VectorWithMemoryTracking & keys, + size_t & transferred, + Arena * arena) const + { + auto & nested_data_column = map_column.getNestedData(); + auto & key_column = nested_data_column.getColumn(0); + auto & val_column = nested_data_column.getColumn(1); + + auto & merged_maps = this->data(place).merged_maps; + + // insert using sorted keys to result column + for (; transferred < keys.size(); ++transferred) + { + key_column.insert(keys[transferred]); + if constexpr (merge) + nested_func->insertMergeResultInto(merged_maps[keys[transferred]], val_column, arena); + else + nested_func->insertResultInto(merged_maps[keys[transferred]], val_column, arena); + } + + IColumn::Offsets & res_offsets = map_column.getNestedColumn().getOffsets(); + res_offsets.push_back(val_column.size()); + } + template void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const { auto & map_column = assert_cast(to); - auto & nested_column = map_column.getNestedColumn(); auto & nested_data_column = map_column.getNestedData(); - auto & key_column = nested_data_column.getColumn(0); auto & val_column = nested_data_column.getColumn(1); @@ -360,18 +388,37 @@ class AggregateFunctionMap final } ::sort(keys.begin(), keys.end()); - // insert using sorted keys to result column - for (auto & key : keys) + size_t transferred = 0; + + if constexpr (!merge) { - key_column.insert(key); - if constexpr (merge) - nested_func->insertMergeResultInto(merged_maps[key], val_column, arena); - else - nested_func->insertResultInto(merged_maps[key], val_column, arena); + /// A nested function that is not a state aliases nothing and need not be atomic. + if (nested_func->isState()) + { + /// `ColumnString::insert` grows `chars` before it appends the offset, so a row count + /// cannot undo an interrupted key insert but a checkpoint, which restores both, can. + const auto keys_checkpoint = key_column.getCheckpoint(); + IColumn::Offsets & res_offsets = map_column.getNestedColumn().getOffsets(); + const size_t offsets_before = res_offsets.size(); + + try + { + transferValues(place, map_column, keys, transferred, arena); + } + catch (...) + { + res_offsets.resize_assume_reserved(offsets_before); + for (size_t i = transferred; i-- > 0;) + nested_func->rollbackInsertResult(merged_maps[keys[i]], val_column); + key_column.rollback(*keys_checkpoint); + throw; + } + + return; + } } - IColumn::Offsets & res_offsets = nested_column.getOffsets(); - res_offsets.push_back(val_column.size()); + transferValues(place, map_column, keys, transferred, arena); } void insertResultInto(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const override @@ -384,6 +431,33 @@ class AggregateFunctionMap final insertResultIntoImpl(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + auto & map_column = assert_cast(to); + auto & nested_data_column = map_column.getNestedData(); + auto & key_column = nested_data_column.getColumn(0); + auto & val_column = nested_data_column.getColumn(1); + + auto & merged_maps = this->data(place).merged_maps; + const size_t appended = merged_maps.size(); + + IColumn::Offsets & res_offsets = map_column.getNestedColumn().getOffsets(); + res_offsets.resize_assume_reserved(res_offsets.size() - 1); + + /// The values were appended in sorted key order while `merged_maps` is unordered, and a row's + /// undo depends on which place produced it, so recover that order from the keys themselves. + const size_t appended_end = key_column.size(); + for (size_t pos = appended_end; pos-- > appended_end - appended;) + { + auto it = merged_maps.find(keyAt(key_column, pos)); + if (it == merged_maps.end()) + abortOnFailedAssertion("AggregateFunctionMap::rollbackInsertResult: appended key is missing from merged_maps"); + nested_func->rollbackInsertResult(it->second, val_column); + } + + key_column.popBack(appended); + } + bool allocatesMemoryInArena() const override { return true; } AggregateFunctionPtr getNestedFunction() const override { return nested_func; } diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionMerge.h b/src/AggregateFunctions/Combinators/AggregateFunctionMerge.h index b80eb99657f7..113cb24963d0 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionMerge.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionMerge.h @@ -139,6 +139,11 @@ class AggregateFunctionMerge final : public IAggregateFunctionHelperinsertMergeResultInto(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + nested_func->rollbackInsertResult(place, to); + } + bool allocatesMemoryInArena() const override { return nested_func->allocatesMemoryInArena(); diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionNull.h b/src/AggregateFunctions/Combinators/AggregateFunctionNull.h index 232fe5d8fac6..18cd17272fa4 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionNull.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionNull.h @@ -248,10 +248,25 @@ class AggregateFunctionNullBase : public IAggregateFunctionHelper if (getFlag(place)) { if constexpr (merge) + { nested_function->insertMergeResultInto(nestedPlace(place), to_concrete.getNestedColumn(), arena); + to_concrete.getNullMapData().push_back(false); + } else + { nested_function->insertResultInto(nestedPlace(place), to_concrete.getNestedColumn(), arena); - to_concrete.getNullMapData().push_back(false); + + /// A nested call that threw has already restored the nested column itself. + try + { + to_concrete.getNullMapData().push_back(false); + } + catch (...) + { + nested_function->rollbackInsertResult(nestedPlace(place), to_concrete.getNestedColumn()); + throw; + } + } } else { @@ -277,6 +292,28 @@ class AggregateFunctionNullBase : public IAggregateFunctionHelper insertResultIntoImpl(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + if constexpr (result_is_nullable) + { + ColumnNullable & to_concrete = assert_cast(to); + if (getFlag(place)) + { + to_concrete.getNullMapData().pop_back(); + nested_function->rollbackInsertResult(nestedPlace(place), to_concrete.getNestedColumn()); + } + else + { + /// insertResultInto appended a state the column itself owns, so this pop must destroy it. + to_concrete.popBack(1); + } + } + else + { + nested_function->rollbackInsertResult(nestedPlace(place), to); + } + } + bool allocatesMemoryInArena() const override { return nested_function->allocatesMemoryInArena(); diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionOrFill.h b/src/AggregateFunctions/Combinators/AggregateFunctionOrFill.h index 3f51d7b5723e..1bf5cbc176dd 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionOrFill.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionOrFill.h @@ -319,10 +319,20 @@ class AggregateFunctionOrFill final : public IAggregateFunctionHelper(to); col.getNullMapColumn().insertDefault(); - if constexpr (merge) - nested_function->insertMergeResultInto(place, col.getNestedColumn(), arena); - else - nested_function->insertResultInto(place, col.getNestedColumn(), arena); + /// The null map entry is this call's own, and the nested transfer restores the nested + /// column itself when it throws, so only the entry has to go. + try + { + if constexpr (merge) + nested_function->insertMergeResultInto(place, col.getNestedColumn(), arena); + else + nested_function->insertResultInto(place, col.getNestedColumn(), arena); + } + catch (...) + { + col.getNullMapColumn().getData().pop_back(); + throw; + } } } else @@ -356,10 +366,18 @@ class AggregateFunctionOrFill final : public IAggregateFunctionHelper(to); col.getNullMapColumn().getData().push_back(static_cast(1)); - if constexpr (merge) - nested_function->insertMergeResultInto(place, col.getNestedColumn(), arena); - else - nested_function->insertResultInto(place, col.getNestedColumn(), arena); + try + { + if constexpr (merge) + nested_function->insertMergeResultInto(place, col.getNestedColumn(), arena); + else + nested_function->insertResultInto(place, col.getNestedColumn(), arena); + } + catch (...) + { + col.getNullMapColumn().getData().pop_back(); + throw; + } } } else @@ -384,6 +402,35 @@ class AggregateFunctionOrFill final : public IAggregateFunctionHelper(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + /// The flag-unset branches are indistinguishable from the appended row alone: with a state-nested + /// function they alias like the flag-set branch, otherwise they append a default `to` owns. + if (!place[size_of_data] && !nested_function->isState()) + { + to.popBack(1); + return; + } + + if constexpr (UseNull) + { + if (!result_is_nullable || inner_nullable) + { + nested_function->rollbackInsertResult(place, to); + } + else + { + ColumnNullable & col = typeid_cast(to); + col.getNullMapColumn().getData().pop_back(); + nested_function->rollbackInsertResult(place, col.getNestedColumn()); + } + } + else + { + nested_function->rollbackInsertResult(place, to); + } + } + AggregateFunctionPtr getNestedFunction() const override { return nested_function; } /// After `Nullable(Tuple)` was introduced, Tuple's `canBeInsideNullable` now returns true, diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionResample.h b/src/AggregateFunctions/Combinators/AggregateFunctionResample.h index b418906d739a..21defecebde7 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionResample.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionResample.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -15,6 +16,12 @@ struct Settings; namespace ErrorCodes { extern const int ARGUMENT_OUT_OF_BOUND; + extern const int MEMORY_LIMIT_EXCEEDED; +} + +namespace FailPoints +{ +extern const char aggregate_function_state_transfer_throw[]; } template @@ -229,23 +236,63 @@ class AggregateFunctionResample final : public IAggregateFunctionHelper(nested_function_->getResultType()); } + /// `transferred` counts the buckets whose transfer returned, so a caller that catches can undo those. template - void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const + void transferBuckets(AggregateDataPtr __restrict place, ColumnArray & col, size_t & transferred, Arena * arena) const { - auto & col = assert_cast(to); - auto & col_offsets = assert_cast(col.getOffsetsColumn()); - - for (size_t i = 0; i < total; ++i) + for (; transferred < total; ++transferred) { if constexpr (merge) - nested_function->insertMergeResultInto(place + i * size_of_data, col.getData(), arena); + nested_function->insertMergeResultInto(place + transferred * size_of_data, col.getData(), arena); else - nested_function->insertResultInto(place + i * size_of_data, col.getData(), arena); + nested_function->insertResultInto(place + transferred * size_of_data, col.getData(), arena); } + if constexpr (!merge) + { + fiu_do_on(FailPoints::aggregate_function_state_transfer_throw, + { + throw Exception(ErrorCodes::MEMORY_LIMIT_EXCEEDED, "Injected failure in AggregateFunctionResample::insertResultInto"); + }); + } + + auto & col_offsets = assert_cast(col.getOffsetsColumn()); col_offsets.getData().push_back(col.getData().size()); } + template + void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const + { + auto & col = assert_cast(to); + size_t transferred = 0; + + if constexpr (!merge) + { + /// A nested function that is not a state aliases nothing and need not be atomic. + if (nested_function->isState()) + { + const size_t offsets_before = assert_cast(col.getOffsetsColumn()).size(); + + try + { + transferBuckets(place, col, transferred, arena); + } + catch (...) + { + auto & col_offsets = assert_cast(col.getOffsetsColumn()); + col_offsets.getData().resize_assume_reserved(offsets_before); + for (size_t i = transferred; i-- > 0;) + nested_function->rollbackInsertResult(place + i * size_of_data, col.getData()); + throw; + } + + return; + } + } + + transferBuckets(place, col, transferred, arena); + } + void insertResultInto(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const override { insertResultIntoImpl(place, to, arena); @@ -256,6 +303,16 @@ class AggregateFunctionResample final : public IAggregateFunctionHelper(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + auto & col = assert_cast(to); + auto & col_offsets = assert_cast(col.getOffsetsColumn()); + + col_offsets.getData().resize_assume_reserved(col_offsets.size() - 1); + for (size_t i = total; i-- > 0;) + nested_function->rollbackInsertResult(place + i * size_of_data, col.getData()); + } + AggregateFunctionPtr getNestedFunction() const override { return nested_function; } }; diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionSimpleState.h b/src/AggregateFunctions/Combinators/AggregateFunctionSimpleState.h index 19a786db1f20..7385a3e944db 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionSimpleState.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionSimpleState.h @@ -103,6 +103,11 @@ class AggregateFunctionSimpleState final : public IAggregateFunctionHelperinsertResultInto(place, to, arena); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept override + { + nested_func->rollbackInsertResult(place, to); + } + bool allocatesMemoryInArena() const override { return nested_func->allocatesMemoryInArena(); } AggregateFunctionPtr getNestedFunction() const override { return nested_func; } diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionState.h b/src/AggregateFunctions/Combinators/AggregateFunctionState.h index 5cb41d1d8716..7318c51022c0 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionState.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionState.h @@ -3,6 +3,7 @@ #include #include #include +#include #include @@ -10,6 +11,16 @@ namespace DB { struct Settings; +namespace ErrorCodes +{ +extern const int MEMORY_LIMIT_EXCEEDED; +} + +namespace FailPoints +{ +extern const char aggregate_function_state_transfer_throw[]; +} + /** Not an aggregate function, but an adapter of aggregate functions, * Aggregate functions with the `State` suffix differ from the corresponding ones in that their states are not finalized. @@ -144,7 +155,18 @@ class AggregateFunctionState final : public IAggregateFunctionHelper(to).getData().push_back(place); + auto & column = assert_cast(to); + + /// Only once the column holds an aliased state, which is the partial transfer to undo. + if (unlikely(!column.empty())) + { + fiu_do_on(FailPoints::aggregate_function_state_transfer_throw, + { + throw Exception(ErrorCodes::MEMORY_LIMIT_EXCEEDED, "Injected failure in AggregateFunctionState::insertResultInto"); + }); + } + + column.getData().push_back(place); } void insertMergeResultInto(AggregateDataPtr __restrict place, IColumn & to, Arena *) const override @@ -152,6 +174,13 @@ class AggregateFunctionState final : public IAggregateFunctionHelper(to).insertFrom(place); } + void rollbackInsertResult(ConstAggregateDataPtr __restrict, IColumn & to) const noexcept override + { + /// insertResultInto aliased a state that `place` still owns, so the row must go without the + /// destroy that ColumnAggregateFunction::popBack performs. + assert_cast(to).popBackWithoutDestroy(1); + } + /// Aggregate function or aggregate function state. bool isState() const override { return true; } diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionTuple.cpp b/src/AggregateFunctions/Combinators/AggregateFunctionTuple.cpp index 4001af145fa1..13b9d034887e 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionTuple.cpp +++ b/src/AggregateFunctions/Combinators/AggregateFunctionTuple.cpp @@ -499,6 +499,13 @@ void AggregateFunctionTuple::insertMergeResultInto(AggregateDataPtr __restrict p insertResultIntoImpl(place, to, arena); } +void AggregateFunctionTuple::rollbackInsertResult(ConstAggregateDataPtr __restrict place, IColumn & to) const noexcept +{ + auto & tuple_to = assert_cast(to); + for (size_t i = nested_functions.size(); i-- > 0;) + nested_functions[i]->rollbackInsertResult(place + state_offsets[i], tuple_to.getColumn(i)); +} + bool AggregateFunctionTuple::allocatesMemoryInArena() const { for (const auto & func : nested_functions) diff --git a/src/AggregateFunctions/Combinators/AggregateFunctionTuple.h b/src/AggregateFunctions/Combinators/AggregateFunctionTuple.h index 4cefe838a21b..3c03a82531c3 100644 --- a/src/AggregateFunctions/Combinators/AggregateFunctionTuple.h +++ b/src/AggregateFunctions/Combinators/AggregateFunctionTuple.h @@ -1,6 +1,7 @@ #pragma once #include +#include #include #include #include @@ -11,6 +12,16 @@ namespace DB { struct Settings; +namespace ErrorCodes +{ +extern const int MEMORY_LIMIT_EXCEEDED; +} + +namespace FailPoints +{ +extern const char aggregate_function_state_transfer_throw_after_child[]; +} + /** Adaptor for aggregate functions. * Adding -Tuple suffix to aggregate function * will convert that aggregate function to a function, accepting Tuples, @@ -130,6 +141,7 @@ class AggregateFunctionTuple final : public IAggregateFunctionHelper - void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const + void transferElements(AggregateDataPtr __restrict place, ColumnTuple & tuple_to, size_t & transferred, Arena * arena) const { - auto & tuple_to = assert_cast(to); - for (size_t i = 0; i < nested_functions.size(); ++i) + for (; transferred < nested_functions.size(); ++transferred) { + if constexpr (!for_merge) + { + /// Only past the first element, whose transfer returned, is there a completed child to undo. + if (unlikely(transferred > 0)) + { + fiu_do_on(FailPoints::aggregate_function_state_transfer_throw_after_child, + { + throw Exception(ErrorCodes::MEMORY_LIMIT_EXCEEDED, "Injected failure in AggregateFunctionTuple::insertResultInto"); + }); + } + } + if constexpr (for_merge) - nested_functions[i]->insertMergeResultInto(place + state_offsets[i], tuple_to.getColumn(i), arena); + nested_functions[transferred]->insertMergeResultInto( + place + state_offsets[transferred], tuple_to.getColumn(transferred), arena); else - nested_functions[i]->insertResultInto(place + state_offsets[i], tuple_to.getColumn(i), arena); + nested_functions[transferred]->insertResultInto( + place + state_offsets[transferred], tuple_to.getColumn(transferred), arena); } } + template + void insertResultIntoImpl(AggregateDataPtr __restrict place, IColumn & to, Arena * arena) const + { + auto & tuple_to = assert_cast(to); + size_t transferred = 0; + + if constexpr (!for_merge) + { + /// An element that is not a state aliases nothing and need not be atomic. + if (isState()) + { + try + { + transferElements(place, tuple_to, transferred, arena); + } + catch (...) + { + for (size_t i = transferred; i-- > 0;) + nested_functions[i]->rollbackInsertResult(place + state_offsets[i], tuple_to.getColumn(i)); + throw; + } + + return; + } + } + + transferElements(place, tuple_to, transferred, arena); + } + /// Shared implementation of the batch add overrides. Hoists the per-element column pointers, so /// no per-row unwrapping work remains in the row loop. /// `get_place` returns the aggregation state for a row, or nullptr when the row has none. diff --git a/src/AggregateFunctions/IAggregateFunction.h b/src/AggregateFunctions/IAggregateFunction.h index 7f7e321d3ad0..ffa802107efe 100644 --- a/src/AggregateFunctions/IAggregateFunction.h +++ b/src/AggregateFunctions/IAggregateFunction.h @@ -294,6 +294,15 @@ class IAggregateFunction : public std::enable_shared_from_this size()) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Cannot pop {} rows from {}: there are only {} rows", n, getName(), size()); + + data.resize_assume_reserved(data.size() - n); +} + ColumnPtr ColumnAggregateFunction::replicate(const IColumn::Offsets & offsets) const { size_t size = data.size(); diff --git a/src/Columns/ColumnAggregateFunction.h b/src/Columns/ColumnAggregateFunction.h index 2689cb00816a..c9da4d7f5647 100644 --- a/src/Columns/ColumnAggregateFunction.h +++ b/src/Columns/ColumnAggregateFunction.h @@ -131,6 +131,18 @@ class ColumnAggregateFunction final : public COWHelper arrayMap(y -> length(finalizeAggregation(y)), x), groupArrayStateForEachResample(0, 2, 1)([number], number % 2)) FROM numbers(2000) SETTINGS max_threads = 1; + +-- -ForEach alone: the throw is in its own loop, with the first element already aliased. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateForEach([number, number + 1]) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), groupArrayStateForEach([number, number + 1])) FROM numbers(2000) SETTINGS max_threads = 1; + +-- -Tuple over -Map with String keys: the first element completes, the second throws on its second +-- key, so -Tuple has to undo a completed -Map child in a different subcolumn. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateMapTuple((map('a', number), map('b', number, 'c', number))) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT mapKeys(t.1), arrayMap(x -> length(finalizeAggregation(x)), mapValues(t.1)), mapKeys(t.2), arrayMap(x -> length(finalizeAggregation(x)), mapValues(t.2)) FROM (SELECT groupArrayStateMapTuple((map('a', number), map('b', number, 'c', number))) AS t FROM numbers(2000)) SETTINGS max_threads = 1; + +-- -Tuple over -ForEach, first element shorter than the second. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateForEachTuple(([number], [number, number])) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), t.1), arrayMap(x -> length(finalizeAggregation(x)), t.2) FROM (SELECT groupArrayStateForEachTuple(([number], [number, number])) AS t FROM numbers(2000)) SETTINGS max_threads = 1; + +-- Exactly one bucket: the -State transfer cannot throw (the column is still empty), so the throw +-- comes from -Resample appending its own offset with the single sub-state already aliased. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateResample(0, 1, 1)(number, 0) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), groupArrayStateResample(0, 1, 1)(number, 0)) FROM numbers(2000) SETTINGS max_threads = 1; + +-- A -Map whose value rows are not uniform: the key with non-NULL values aliases a sub-state, the +-- all-NULL keys make the null adapter insert a default the column itself owns. Undoing such a map in +-- any order other than the reverse of the sorted-key append order applies the wrong row's semantics. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateTupleMapTuple((CAST(map('b', NULL, 'a', (number, number + 1), 'c', NULL), 'Map(String, Nullable(Tuple(UInt64, UInt64)))'), CAST(map('x', (number, number + 1), 'y', (number + 2, number + 3)), 'Map(String, Nullable(Tuple(UInt64, UInt64)))'))) FROM numbers(2000) SETTINGS max_threads = 1, enable_nullable_tuple_type = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT mapKeys(t.1), arrayMap(x -> isNull(x), mapValues(t.1)), arrayMap(x -> length(finalizeAggregation(assumeNotNull(x).1)), mapValues(t.1)), mapKeys(t.2), arrayMap(x -> isNull(x), mapValues(t.2)) FROM (SELECT groupArrayStateTupleMapTuple((CAST(map('b', NULL, 'a', (number, number + 1), 'c', NULL), 'Map(String, Nullable(Tuple(UInt64, UInt64)))'), CAST(map('x', (number, number + 1), 'y', (number + 2, number + 3)), 'Map(String, Nullable(Tuple(UInt64, UInt64)))'))) AS t FROM numbers(2000)) SETTINGS max_threads = 1, enable_nullable_tuple_type = 1; + +-- -Tuple over -Resample. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateResampleTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2)) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), t.1), arrayMap(x -> length(finalizeAggregation(x)), t.2) FROM (SELECT groupArrayStateResampleTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2)) AS t FROM numbers(2000)) SETTINGS max_threads = 1; + +-- The same, through a transparent -If wrapper. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateResampleIfTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2), (number > 0, number > 0)) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), t.1), arrayMap(x -> length(finalizeAggregation(x)), t.2) FROM (SELECT groupArrayStateResampleIfTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2), (number > 0, number > 0)) AS t FROM numbers(2000)) SETTINGS max_threads = 1; + +-- -OrNull around a Tuple of -State results: Tuple can be inside Nullable, so the transfer goes into +-- the nested column of a ColumnNullable. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateResampleTupleOrNullTuple(0, 2, 1)(((number, number + 1), (number + 2, number + 3)), ((number % 2, number % 2), (number % 2, number % 2))) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), assumeNotNull(t.1).1), arrayMap(x -> length(finalizeAggregation(x)), assumeNotNull(t.2).2) FROM (SELECT groupArrayStateResampleTupleOrNullTuple(0, 2, 1)(((number, number + 1), (number + 2, number + 3)), ((number % 2, number % 2), (number % 2, number % 2))) AS t FROM numbers(2000)) SETTINGS max_threads = 1; + +-- The same window through the implicit null adapter instead of -OrNull, plus a -Distinct wrapper. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT groupArrayStateResampleTupleDistinctTuple(0, 2, 1)(CAST(((number, number + 1), (number + 2, number + 3)), 'Tuple(Nullable(Tuple(UInt64, UInt64)), Nullable(Tuple(UInt64, UInt64)))'), ((number % 2, number % 2), (number % 2, number % 2))) FROM numbers(2000) SETTINGS max_threads = 1, enable_nullable_tuple_type = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw; +SELECT arrayMap(x -> length(finalizeAggregation(x)), assumeNotNull(t.1).1), arrayMap(x -> length(finalizeAggregation(x)), assumeNotNull(t.2).2) FROM (SELECT groupArrayStateResampleTupleDistinctTuple(0, 2, 1)(CAST(((number, number + 1), (number + 2, number + 3)), 'Tuple(Nullable(Tuple(UInt64, UInt64)), Nullable(Tuple(UInt64, UInt64)))'), ((number % 2, number % 2), (number % 2, number % 2))) AS t FROM numbers(2000)) SETTINGS max_threads = 1, enable_nullable_tuple_type = 1; + +-- A completed -Resample child undone by its parent: with the second failpoint the first tuple element +-- transfers all of its buckets, and the throw lands before the second element. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw_after_child; +SELECT groupArrayStateResampleTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2)) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw_after_child; +SELECT arrayMap(x -> length(finalizeAggregation(x)), t.1), arrayMap(x -> length(finalizeAggregation(x)), t.2) FROM (SELECT groupArrayStateResampleTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2)) AS t FROM numbers(2000)) SETTINGS max_threads = 1; + +-- The same completed child behind an -OrNull that forwards: Array cannot be inside Nullable, so -OrFill +-- passes the array column straight through and has to forward the undo as well. +SYSTEM ENABLE FAILPOINT aggregate_function_state_transfer_throw_after_child; +SELECT groupArrayStateResampleOrNullTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2)) FROM numbers(2000) SETTINGS max_threads = 1 FORMAT Null; -- { serverError MEMORY_LIMIT_EXCEEDED } +SYSTEM DISABLE FAILPOINT aggregate_function_state_transfer_throw_after_child; +SELECT arrayMap(x -> length(finalizeAggregation(x)), t.1), arrayMap(x -> length(finalizeAggregation(x)), t.2) FROM (SELECT groupArrayStateResampleOrNullTuple(0, 2, 1)((number, number + 1), (number % 2, number % 2)) AS t FROM numbers(2000)) SETTINGS max_threads = 1; From b6dd300fa9b56a0478e85e08f71932bcd1ecc375 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 14:32:50 +0000 Subject: [PATCH 052/185] Backport #121894 to 26.8: Fix a server crash when the flameGraph aggregate function is constructed without a query context --- .../AggregateFunctionFactory.h | 2 ++ .../AggregateFunctionFlameGraph.cpp | 4 +++- ...graph_attach_without_query_context.reference | 1 + ...3_flamegraph_attach_without_query_context.sh | 17 +++++++++++++++++ 4 files changed, 23 insertions(+), 1 deletion(-) create mode 100644 tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.reference create mode 100755 tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.sh diff --git a/src/AggregateFunctions/AggregateFunctionFactory.h b/src/AggregateFunctions/AggregateFunctionFactory.h index 53248dae079f..c3323b1a2177 100644 --- a/src/AggregateFunctions/AggregateFunctionFactory.h +++ b/src/AggregateFunctions/AggregateFunctionFactory.h @@ -31,6 +31,8 @@ class ASTFunction; * The invoker has arguments: name of aggregate function, types of arguments, values of parameters. * Parameters are for "parametric" aggregate functions. * For example, in quantileWeighted(0.9)(x, weight), 0.9 is "parameter" and x, weight are "arguments". + * `Settings` is null when the function is constructed outside a query, e.g. while a background + * thread parses an `AggregateFunction(...)` type name. */ using AggregateFunctionCreator = std::function; diff --git a/src/AggregateFunctions/AggregateFunctionFlameGraph.cpp b/src/AggregateFunctions/AggregateFunctionFlameGraph.cpp index eb5f188de545..b84aaa5daf12 100644 --- a/src/AggregateFunctions/AggregateFunctionFlameGraph.cpp +++ b/src/AggregateFunctions/AggregateFunctionFlameGraph.cpp @@ -691,7 +691,9 @@ static void check(const std::string & name, const DataTypes & argument_types, co static AggregateFunctionPtr createAggregateFunctionFlameGraph(const std::string & name, const DataTypes & argument_types, const Array & params, const Settings * settings) { - if (!(*settings)[Setting::allow_introspection_functions]) + /// The factory passes null when there is no query context to take settings from, and with no + /// principal there is nothing to authorize. + if (settings && !(*settings)[Setting::allow_introspection_functions]) throw Exception(ErrorCodes::FUNCTION_NOT_ALLOWED, "Introspection functions are disabled, because setting 'allow_introspection_functions' is set to 0"); diff --git a/tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.reference b/tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.reference new file mode 100644 index 000000000000..573541ac9702 --- /dev/null +++ b/tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.reference @@ -0,0 +1 @@ +0 diff --git a/tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.sh b/tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.sh new file mode 100755 index 000000000000..3e7efa42e6b8 --- /dev/null +++ b/tests/queries/0_stateless/05243_flamegraph_attach_without_query_context.sh @@ -0,0 +1,17 @@ +#!/usr/bin/env bash +# A stored table whose column type names AggregateFunction(flameGraph, ...) used to crash the +# process when a second run opened it. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +cd "${CLICKHOUSE_TMP}" || exit +rm -rf "05243_flamegraph_attach" + +$CLICKHOUSE_LOCAL --path "05243_flamegraph_attach" --query "SET allow_introspection_functions = 1; CREATE TABLE t (c AggregateFunction(flameGraph, Array(UInt64))) ENGINE = Memory" + +# Read the table rather than a constant, so that opening it is what the output depends on. +$CLICKHOUSE_LOCAL --path "05243_flamegraph_attach" --query "SELECT count() FROM t" + +rm -rf "05243_flamegraph_attach" From 974a1c0b3cae4f8a951c9f8dcd6c90a30adfe2b6 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 15:23:43 +0000 Subject: [PATCH 053/185] Backport #120376 to 26.8: Enforce settings constraints installed by a profile in the same statement --- src/Interpreters/Context.cpp | 77 +++++++++++++++++++ src/Interpreters/Context.h | 2 + src/Interpreters/InterpreterSetQuery.cpp | 4 +- src/Server/ArrowFlight/ArrowFlightServer.cpp | 17 +++- src/Server/GRPCServer.cpp | 25 +++--- .../test_session_options_settings_profile.py | 76 ++++++++++++++++++ tests/integration/test_grpc_protocol/test.py | 47 +++++++++++ ...raints_set_profile_in_same_query.reference | 1 + ...s_constraints_set_profile_in_same_query.sh | 39 ++++++++++ ...aints_profile_in_nested_settings.reference | 3 + ..._constraints_profile_in_nested_settings.sh | 18 +++++ ...nstraints_profile_in_http_params.reference | 6 ++ ...ings_constraints_profile_in_http_params.sh | 28 +++++++ 13 files changed, 328 insertions(+), 15 deletions(-) create mode 100644 tests/integration/test_arrowflight_interface/test_session_options_settings_profile.py create mode 100644 tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.reference create mode 100755 tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.sh create mode 100644 tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.reference create mode 100755 tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.sh create mode 100644 tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.reference create mode 100755 tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.sh diff --git a/src/Interpreters/Context.cpp b/src/Interpreters/Context.cpp index 5649d6f637b9..4bcc3b6bc767 100644 --- a/src/Interpreters/Context.cpp +++ b/src/Interpreters/Context.cpp @@ -1,3 +1,4 @@ +#include #include #include #include @@ -3398,6 +3399,41 @@ Settings Context::getSettingsCopy() const return *settings; } +namespace +{ +bool isProfileChange(const SettingChange & change) +{ + return change.name == "profile"; +} + +/// Enforces the constraints on `changes` the way `applySettingsChanges` applies them: a `profile` change +/// installs a new constraint set for the changes after it. Each run of changes up to the next `profile` +/// change is enforced against the constraints in force before it, then applied together with that `profile` +/// change to a scratch copy of `context`, so a rejected list leaves `context` untouched. Returns the enforced list. +template +SettingsChanges enforceConstraintsAlongProfileChanges(const ContextPtr & context, const SettingsChanges & changes, Enforce && enforce) +{ + auto scratch_context = Context::createCopy(context); + SettingsChanges enforced; + for (auto begin = changes.begin(); begin != changes.end();) + { + auto profile = std::find_if(begin, changes.end(), isProfileChange); + SettingsChanges segment(begin, profile); + enforce(*scratch_context, segment); + begin = profile; + if (profile != changes.end()) + { + segment.push_back(*profile); + ++begin; + } + /// `setCurrentProfile` checks the profile's own settings against the constraints in force before it. + scratch_context->applySettingsChanges(segment); + enforced.insert(enforced.end(), segment.begin(), segment.end()); + } + return enforced; +} +} + void Context::setSettings(const Settings & settings_) { std::lock_guard lock(mutex); @@ -3558,6 +3594,15 @@ void Context::checkSettingsConstraints(const SettingChange & change, SettingSour void Context::checkSettingsConstraints(const SettingsChanges & changes, SettingSource source) { + if (std::ranges::any_of(changes, isProfileChange)) + { + enforceConstraintsAlongProfileChanges(shared_from_this(), changes, [source](Context & context, SettingsChanges & segment) + { + context.checkSettingsConstraints(std::as_const(segment), source); + }); + return; + } + SharedLockGuard lock(mutex); settings->checkShorthandChanges(changes); getSettingsConstraintsAndCurrentProfilesWithLock()->constraints.check(*settings, changes, source); @@ -3570,14 +3615,46 @@ void Context::checkSettingsConstraintsForSettingsReset(const std::vector getSettingsConstraintsAndCurrentProfilesWithLock()->constraints.checkResetToDefault(*settings, names, source); } +void Context::checkSettingsConstraintsForSettingsReset( + const std::vector & names, const SettingsChanges & changes_applied_first, SettingSource source) +{ + if (std::ranges::none_of(changes_applied_first, isProfileChange)) + { + checkSettingsConstraintsForSettingsReset(names, source); + return; + } + /// The resets take effect after the rest of the statement, so a `profile` change in it decides the constraints. + auto scratch_context = Context::createCopy(shared_from_this()); + scratch_context->applySettingsChanges(changes_applied_first); + scratch_context->checkSettingsConstraintsForSettingsReset(names, source); +} + void Context::checkSettingsConstraints(SettingsChanges & changes, SettingSource source) { + if (std::ranges::any_of(changes, isProfileChange)) + { + changes = enforceConstraintsAlongProfileChanges(shared_from_this(), changes, [source](Context & context, SettingsChanges & segment) + { + context.checkSettingsConstraints(segment, source); + }); + return; + } + SharedLockGuard lock(mutex); checkSettingsConstraintsWithLock(changes, source); } void Context::clampToSettingsConstraints(SettingsChanges & changes, SettingSource source) { + if (std::ranges::any_of(changes, isProfileChange)) + { + changes = enforceConstraintsAlongProfileChanges(shared_from_this(), changes, [source](Context & context, SettingsChanges & segment) + { + context.clampToSettingsConstraints(segment, source); + }); + return; + } + SharedLockGuard lock(mutex); clampToSettingsConstraintsWithLock(changes, source); } diff --git a/src/Interpreters/Context.h b/src/Interpreters/Context.h index f40b65b73f91..6c2dcdcff8f5 100644 --- a/src/Interpreters/Context.h +++ b/src/Interpreters/Context.h @@ -1228,6 +1228,8 @@ class Context: public ContextData, public std::enable_shared_from_this void checkSettingsConstraints(const SettingsChanges & changes, SettingSource source); void checkSettingsConstraints(SettingsChanges & changes, SettingSource source); void checkSettingsConstraintsForSettingsReset(const std::vector & names, SettingSource source); + /// For the resets of a statement that also changes `profile`: `changes_applied_first` decides their constraints. + void checkSettingsConstraintsForSettingsReset(const std::vector & names, const SettingsChanges & changes_applied_first, SettingSource source); void clampToSettingsConstraints(SettingsChanges & changes, SettingSource source); void checkMergeTreeSettingsConstraints(const MergeTreeSettings & merge_tree_settings, const SettingsChanges & changes) const; diff --git a/src/Interpreters/InterpreterSetQuery.cpp b/src/Interpreters/InterpreterSetQuery.cpp index fd7ff81b7a77..cef2ade6d97d 100644 --- a/src/Interpreters/InterpreterSetQuery.cpp +++ b/src/Interpreters/InterpreterSetQuery.cpp @@ -89,7 +89,7 @@ BlockIO InterpreterSetQuery::execute() /// explicitly set to its current value. The original code applies const `ast.changes`. getContext()->checkSettingsConstraints(std::as_const(changes), SettingSource::QUERY); /// Checked before anything is applied, so that a violation leaves the whole statement without effect. - getContext()->checkSettingsConstraintsForSettingsReset(ast.default_settings, SettingSource::QUERY); + getContext()->checkSettingsConstraintsForSettingsReset(ast.default_settings, changes, SettingSource::QUERY); auto session_context = getContext()->getSessionContext(); session_context->applySettingsChanges(changes); session_context->addQueryParameters(NameToNameMap{ast.query_parameters.begin(), ast.query_parameters.end()}); @@ -109,7 +109,7 @@ void InterpreterSetQuery::executeForCurrentContext(bool ignore_setting_constrain if (!ignore_setting_constraints) { getContext()->checkSettingsConstraints(std::as_const(changes), SettingSource::QUERY); - getContext()->checkSettingsConstraintsForSettingsReset(ast.default_settings, SettingSource::QUERY); + getContext()->checkSettingsConstraintsForSettingsReset(ast.default_settings, changes, SettingSource::QUERY); rejectHTTPOnlyConstructionSettings(ast); } getContext()->applySettingsChanges(changes); diff --git a/src/Server/ArrowFlight/ArrowFlightServer.cpp b/src/Server/ArrowFlight/ArrowFlightServer.cpp index 87627bbce6af..c560ba7838ed 100644 --- a/src/Server/ArrowFlight/ArrowFlightServer.cpp +++ b/src/Server/ArrowFlight/ArrowFlightServer.cpp @@ -1376,14 +1376,14 @@ arrow::Status ArrowFlightServer::DoAction( } }; - for (const auto & [setting, value] : request.session_options) + auto apply_option = [&](const std::string & setting, const auto & value) { if (!isValidIdentifier(setting)) { result.errors[setting] = arrow::flight::SetSessionOptionsResult::Error{ arrow::flight::SetSessionOptionErrorValue::kInvalidName }; - continue; + return; } try @@ -1391,14 +1391,14 @@ arrow::Status ArrowFlightServer::DoAction( if (std::holds_alternative(value)) { /// std::monostate means "reset to default" (SET setting = DEFAULT). - query_context->checkSettingsConstraintsForSettingsReset({setting}, SettingSource::QUERY); + session_context->checkSettingsConstraintsForSettingsReset({setting}, SettingSource::QUERY); session_context->resetSettingsToDefaultValue({setting}); } else { auto string_value = std::visit(to_string_value, value); SettingChange change{setting, Field{string_value}}; - query_context->checkSettingsConstraints(change, SettingSource::QUERY); + session_context->checkSettingsConstraints(change, SettingSource::QUERY); session_context->setSetting(setting, string_value); } } @@ -1416,6 +1416,15 @@ arrow::Status ArrowFlightServer::DoAction( result.errors[setting] = arrow::flight::SetSessionOptionsResult::Error{error_value}; } + }; + + /// The options arrive in a map with no order, so `profile` goes first for its constraints to bind the rest. + if (auto profile = request.session_options.find("profile"); profile != request.session_options.end()) + apply_option(profile->first, profile->second); + for (const auto & [setting, value] : request.session_options) + { + if (setting != "profile") + apply_option(setting, value); } ARROW_ASSIGN_OR_RAISE(auto serialized, result.SerializeToString()) diff --git a/src/Server/GRPCServer.cpp b/src/Server/GRPCServer.cpp index f8dd4d717c2a..ae5a796bfe12 100644 --- a/src/Server/GRPCServer.cpp +++ b/src/Server/GRPCServer.cpp @@ -310,6 +310,20 @@ namespace } }; + /// A protobuf map has no order, so `profile` goes first for its constraints to bind the other settings. + SettingsChanges settingsChangesFromMap(const google::protobuf::Map & map) + { + SettingsChanges changes; + for (const auto & [key, value] : map) + { + if (key == "profile") + changes.insert(changes.begin(), {key, value}); + else + changes.push_back({key, value}); + } + return changes; + } + /// Gets session's timeout from query info or from the server config. std::chrono::steady_clock::duration getSessionTimeout(const GRPCQueryInfo & query_info, const Poco::Util::AbstractConfiguration & config) { @@ -945,12 +959,7 @@ namespace query_context = session->makeQueryContext(std::move(client_info)); - /// Prepare settings. - SettingsChanges settings_changes; - for (const auto & [key, value] : query_info.settings()) - { - settings_changes.push_back({key, value}); - } + auto settings_changes = settingsChangesFromMap(query_info.settings()); query_context->checkSettingsConstraints(settings_changes, SettingSource::QUERY); query_context->applySettingsChanges(settings_changes); @@ -1277,9 +1286,7 @@ namespace { temp_context = Context::createCopy(query_context); external_table_context = temp_context; - SettingsChanges settings_changes; - for (const auto & [key, value] : external_table.settings()) - settings_changes.push_back({key, value}); + auto settings_changes = settingsChangesFromMap(external_table.settings()); external_table_context->checkSettingsConstraints(settings_changes, SettingSource::QUERY); external_table_context->applySettingsChanges(settings_changes); } diff --git a/tests/integration/test_arrowflight_interface/test_session_options_settings_profile.py b/tests/integration/test_arrowflight_interface/test_session_options_settings_profile.py new file mode 100644 index 000000000000..d998b1690edf --- /dev/null +++ b/tests/integration/test_arrowflight_interface/test_session_options_settings_profile.py @@ -0,0 +1,76 @@ +# coding: utf-8 + +import pytest +import random +import string + +from .flight_sql_client import FlightSQLClient, SetSessionOptionsResult + +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) +node = cluster.add_instance( + "node", + main_configs=[ + "configs/flight_port.xml", + ], +) + +PROFILE_NAME = "profile_arrowflight_session_options_constraints" + + +def get_client(): + session_id = ''.join(random.choices(string.ascii_letters + string.digits, k=16)) + return FlightSQLClient( + host=node.ip_address, + port=8888, + insecure=True, + disable_server_verification=True, + metadata={'x-clickhouse-session-id': session_id}, + features={'metadata-reflection': 'true'}, + ) + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + try: + cluster.start() + node.wait_until_port_is_ready(8888, timeout=10) + yield cluster + finally: + cluster.shutdown() + + +@pytest.fixture(autouse=True) +def settings_profile(): + node.query(f"DROP SETTINGS PROFILE IF EXISTS {PROFILE_NAME}") + node.query( + f"CREATE SETTINGS PROFILE {PROFILE_NAME} SETTINGS max_execution_time = 10 CONST" + ) + try: + yield PROFILE_NAME + finally: + node.query(f"DROP SETTINGS PROFILE IF EXISTS {PROFILE_NAME}") + + +def _query_scalar(client, query): + flight_info = client.execute(query) + table = client.do_get(flight_info.endpoints[0].ticket).read_all() + return table.column(0)[0].as_py() + + +def test_profile_constraints_apply_within_same_request(): + """A profile set in a SetSessionOptions request constrains the other options of the same request.""" + client = get_client() + + result = client.set_session_options( + {"profile": PROFILE_NAME, "max_execution_time": "999"} + ) + + assert "profile" not in result.errors + # A constraint violation is neither a parse nor an unknown-setting error, so it maps to UNSPECIFIED. + assert ( + result.errors["max_execution_time"].value + == SetSessionOptionsResult.UNSPECIFIED + ) + assert float(_query_scalar(client, "SELECT getSetting('max_execution_time')")) == 10 diff --git a/tests/integration/test_grpc_protocol/test.py b/tests/integration/test_grpc_protocol/test.py index 915eaef3ad15..9aa9080bf737 100644 --- a/tests/integration/test_grpc_protocol/test.py +++ b/tests/integration/test_grpc_protocol/test.py @@ -611,6 +611,53 @@ def test_external_table(): ) +def test_settings_profile_constraints(): + profile = "grpc_profile_constraints" + query(f"DROP SETTINGS PROFILE IF EXISTS {profile}") + query( + f"CREATE SETTINGS PROFILE {profile} SETTINGS max_execution_time = 10 CONST, max_result_rows = 12345, format_csv_delimiter = '|'" + ) + try: + # `settings` is a protobuf map with no order: the profile goes first, so its constraints bind the + # other settings and they override the values it sets. + e = query_and_get_error( + "SELECT 1", settings={"profile": profile, "max_execution_time": "999"} + ) + assert "Setting max_execution_time should not be changed" in e.display_text + assert ( + query( + "SELECT getSetting('max_result_rows')", + settings={"profile": profile, "max_result_rows": "7"}, + ) + == "7\n" + ) + + columns = [ + clickhouse_grpc_pb2.NameAndType(name="UserID", type="UInt64"), + clickhouse_grpc_pb2.NameAndType(name="UserName", type="String"), + ] + + def ext(settings): + return clickhouse_grpc_pb2.ExternalTable( + name="ext1", columns=columns, data=b"1;Alex\n", format="CSV", settings=settings + ) + + e = query_and_get_error( + "SELECT * FROM ext1", + external_tables=[ext({"profile": profile, "max_execution_time": "999"})], + ) + assert "Setting max_execution_time should not be changed" in e.display_text + assert ( + query( + "SELECT * FROM ext1", + external_tables=[ext({"profile": profile, "format_csv_delimiter": ";"})], + ) + == "1\tAlex\n" + ) + finally: + query(f"DROP SETTINGS PROFILE {profile}") + + def test_external_table_streaming(): columns = [ clickhouse_grpc_pb2.NameAndType(name="UserID", type="UInt64"), diff --git a/tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.reference b/tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.reference new file mode 100644 index 000000000000..17c209230693 --- /dev/null +++ b/tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.reference @@ -0,0 +1 @@ +7 8 diff --git a/tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.sh b/tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.sh new file mode 100755 index 000000000000..e4d11a0e11bf --- /dev/null +++ b/tests/queries/0_stateless/05218_settings_constraints_set_profile_in_same_query.sh @@ -0,0 +1,39 @@ +#!/usr/bin/env bash +# A `SET` statement that changes `profile` installs a new constraint set halfway through itself, and the +# assignments and resets after that point must pass it, as they would in a separate statement. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +CONST_PROFILE="profile_const_$CLICKHOUSE_DATABASE" +READONLY_PROFILE="profile_readonly_$CLICKHOUSE_DATABASE" +MAX_PROFILE="profile_max_$CLICKHOUSE_DATABASE" +RELAXED_PROFILE="profile_relaxed_$CLICKHOUSE_DATABASE" + +$CLICKHOUSE_CLIENT -m -q " +DROP SETTINGS PROFILE IF EXISTS $CONST_PROFILE, $READONLY_PROFILE, $MAX_PROFILE, $RELAXED_PROFILE; +CREATE SETTINGS PROFILE $CONST_PROFILE SETTINGS max_execution_time = 10 CONST, SQL_tenant_id = 1 CONST; +CREATE SETTINGS PROFILE $READONLY_PROFILE SETTINGS readonly = 1; +CREATE SETTINGS PROFILE $MAX_PROFILE SETTINGS max_memory_usage MAX 1000000; +CREATE SETTINGS PROFILE $RELAXED_PROFILE SETTINGS max_memory_usage MAX 1099511627776; +" + +# One session on purpose: a rejected statement must leave it untouched, which the statements after it prove. +$CLICKHOUSE_CLIENT -m -q " +SET profile = '$CONST_PROFILE', max_execution_time = 999; -- { serverError SETTING_CONSTRAINT_VIOLATION } +SET profile = '$CONST_PROFILE', max_execution_time = DEFAULT; -- { serverError SETTING_CONSTRAINT_VIOLATION } +SET profile = '$READONLY_PROFILE', max_memory_usage = 1099511627776; -- { serverError READONLY } +SET profile = '$MAX_PROFILE', max_memory_usage = 1099511627776; -- { serverError SETTING_CONSTRAINT_VIOLATION } +-- a later, looser profile does not lift the constraint in force at the assignment +SET profile = '$MAX_PROFILE', max_memory_usage = 5000000, profile = '$RELAXED_PROFILE'; -- { serverError SETTING_CONSTRAINT_VIOLATION } +-- none of the rejected statements applied its profile, which would have installed this setting +SELECT getSetting('SQL_tenant_id'); -- { serverError UNKNOWN_SETTING } +-- assigning the value the profile installs is a no-op and stays allowed +SET profile = '$CONST_PROFILE', max_execution_time = 10; +-- what is not constrained still applies, before and after the profile change +SET SQL_before = 7, profile = '$CONST_PROFILE', SQL_after = 8; +SELECT getSetting('SQL_before'), getSetting('SQL_after'); +" + +$CLICKHOUSE_CLIENT -q "DROP SETTINGS PROFILE $CONST_PROFILE, $READONLY_PROFILE, $MAX_PROFILE, $RELAXED_PROFILE" diff --git a/tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.reference b/tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.reference new file mode 100644 index 000000000000..4f2ffd17cb8d --- /dev/null +++ b/tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.reference @@ -0,0 +1,3 @@ +10 +1000000 +10 diff --git a/tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.sh b/tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.sh new file mode 100755 index 000000000000..613eece11dcd --- /dev/null +++ b/tests/queries/0_stateless/05219_settings_constraints_profile_in_nested_settings.sh @@ -0,0 +1,18 @@ +#!/usr/bin/env bash +# A `profile` change in a nested `SETTINGS` clause constrains the settings after it. Settings crossing into +# another execution context are clamped rather than rejected, so the nested value loses to the constraint. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +PROFILE="profile_nested_$CLICKHOUSE_DATABASE" + +$CLICKHOUSE_CLIENT -m -q " +DROP SETTINGS PROFILE IF EXISTS $PROFILE; +CREATE SETTINGS PROFILE $PROFILE SETTINGS max_execution_time = 10 CONST, max_memory_usage MAX 1000000; +SELECT * FROM (SELECT getSetting('max_execution_time') SETTINGS profile = '$PROFILE', max_execution_time = 999); +SELECT * FROM (SELECT getSetting('max_memory_usage') SETTINGS profile = '$PROFILE', max_memory_usage = 1099511627776); +WITH w AS (SELECT getSetting('max_execution_time') SETTINGS profile = '$PROFILE', max_execution_time = 999) SELECT * FROM w; +DROP SETTINGS PROFILE $PROFILE; +" diff --git a/tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.reference b/tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.reference new file mode 100644 index 000000000000..5521a7189933 --- /dev/null +++ b/tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.reference @@ -0,0 +1,6 @@ +-- a parameter after the profile is checked against the constraints it installs +SETTING_CONSTRAINT_VIOLATION +-- a parameter before the profile is overridden by it +12345 +-- a parameter equal to the value before the profile still overrides the profile +0 diff --git a/tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.sh b/tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.sh new file mode 100755 index 000000000000..d060d5ee55c4 --- /dev/null +++ b/tests/queries/0_stateless/05220_settings_constraints_profile_in_http_params.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# HTTP URL parameters are a settings change list in URL order, so a `profile` parameter constrains the +# parameters after it. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +PROFILE="profile_http_$CLICKHOUSE_DATABASE" + +$CLICKHOUSE_CLIENT -m -q " +DROP SETTINGS PROFILE IF EXISTS $PROFILE; +CREATE SETTINGS PROFILE $PROFILE SETTINGS max_execution_time = 10 CONST, max_result_rows = 12345; +" + +function run_with_params() +{ + ${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&$1" -d "$2" +} + +echo '-- a parameter after the profile is checked against the constraints it installs' +run_with_params "profile=$PROFILE&max_execution_time=999" "SELECT 1" | grep -o -m1 SETTING_CONSTRAINT_VIOLATION +echo '-- a parameter before the profile is overridden by it' +run_with_params "max_result_rows=8&profile=$PROFILE" "SELECT getSetting('max_result_rows')" +echo '-- a parameter equal to the value before the profile still overrides the profile' +run_with_params "profile=$PROFILE&max_result_rows=0" "SELECT getSetting('max_result_rows')" + +$CLICKHOUSE_CLIENT -q "DROP SETTINGS PROFILE $PROFILE" From 8137e730c439f2c183619be1876be4356b6156b4 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 25 Sep 2026 17:55:16 +0000 Subject: [PATCH 054/185] Update autogenerated version to 26.8.11.7 and contributors --- cmake/autogenerated_versions.txt | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cmake/autogenerated_versions.txt b/cmake/autogenerated_versions.txt index d14f61b49bdc..16b893d9fd0d 100644 --- a/cmake/autogenerated_versions.txt +++ b/cmake/autogenerated_versions.txt @@ -2,11 +2,11 @@ # NOTE: VERSION_REVISION has nothing common with DBMS_TCP_PROTOCOL_VERSION, # only DBMS_TCP_PROTOCOL_VERSION should be incremented on protocol changes. -SET(VERSION_REVISION 54523) +SET(VERSION_REVISION 54524) SET(VERSION_MAJOR 26) SET(VERSION_MINOR 8) -SET(VERSION_PATCH 11) -SET(VERSION_GITHASH bedf2ab54b8a0c34afbf2d907eaf933324f43cd6) -SET(VERSION_DESCRIBE v26.8.11.1-lts) -SET(VERSION_STRING 26.8.11.1) +SET(VERSION_PATCH 12) +SET(VERSION_GITHASH 97da26e305d9764402a30702c9b779745bf557c4) +SET(VERSION_DESCRIBE v26.8.12.1-lts) +SET(VERSION_STRING 26.8.12.1) # end of autochange From 17861e3f82cefdcff6e0ff2ebc5d66afc143e1be Mon Sep 17 00:00:00 2001 From: Jimmy Aguilar Mena Date: Fri, 25 Sep 2026 20:47:55 +0000 Subject: [PATCH 055/185] Strip the padding of an set element before tokenizing it --- .../MergeTree/MergeTreeIndexBloomFilterText.cpp | 10 +++++++++- 1 file changed, 9 insertions(+), 1 deletion(-) diff --git a/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp b/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp index d9d478b11dfb..e8e299d1a699 100644 --- a/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp @@ -822,6 +822,9 @@ bool MergeTreeConditionBloomFilterText::tryPrepareSetBloomFilter( size_t tuple_idx = elem.tuple_index; const auto & column = columns[tuple_idx]; + const DataTypePtr & element_type = prepared_set->getElementsTypes()[tuple_idx]; + const bool is_fixed_string_element = WhichDataType(removeNullable(element_type)).isFixedString(); + for (size_t row = 0; row < prepared_set_total_row_count; ++row) { /// A NULL element also matches the column's NULL rows, which the filter cannot express. @@ -829,7 +832,12 @@ bool MergeTreeConditionBloomFilterText::tryPrepareSetBloomFilter( return false; bloom_filters.back().emplace_back(params); - auto ref = column->getDataAt(row); + + /// `FixedString` element carries its padding, which the comparison ignores but the tokenizer would not. + std::string_view ref = column->getDataAt(row); + if (is_fixed_string_element) + ref = ref.substr(0, ref.find_last_not_of('\0') + 1); + forEachTokenToBloomFilter(*tokenizer, ref.data(), ref.size(), bloom_filters.back().back()); } } From 7504aa878eebe7f65c6735e36268ef9b060371d7 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sat, 26 Sep 2026 04:07:42 +0000 Subject: [PATCH 056/185] Backport #119302 to 26.8: Revert "Revert "Proper partition pruning with dates"" --- src/Functions/FunctionsConversion.h | 16 +- .../Iceberg/ManifestFilesPruning.cpp | 206 +++++++++++++- .../test_partition_pruning_with_functions.py | 268 ++++++++++++++++++ ...nullable_key_todate_monotonicity.reference | 8 + ...05054_nullable_key_todate_monotonicity.sql | 33 +++ 5 files changed, 523 insertions(+), 8 deletions(-) create mode 100644 tests/integration/test_storage_iceberg_with_spark/test_partition_pruning_with_functions.py create mode 100644 tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.reference create mode 100644 tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.sql diff --git a/src/Functions/FunctionsConversion.h b/src/Functions/FunctionsConversion.h index e3158387728c..477c0e010bbf 100644 --- a/src/Functions/FunctionsConversion.h +++ b/src/Functions/FunctionsConversion.h @@ -4365,8 +4365,14 @@ struct ToDateMonotonicity { static bool has() { return true; } - static IFunction::Monotonicity get(const IDataType & type, const Field & left, const Field & right) + static IFunction::Monotonicity get(const IDataType & type_with_wrappers, const Field & left, const Field & right) { + const IDataType * type_without_wrappers = &type_with_wrappers; + if (const auto * low_cardinality_type = typeid_cast(type_without_wrappers)) + type_without_wrappers = low_cardinality_type->getDictionaryType().get(); + if (const auto * nullable_type = typeid_cast(type_without_wrappers)) + type_without_wrappers = nullable_type->getNestedType().get(); + const IDataType & type = *type_without_wrappers; auto which = WhichDataType(type); if (which.isDateOrDate32() || which.isTime() || which.isTime64() || which.isDateTime() || which.isDateTime64() || which.isInt8() || which.isInt16() || which.isUInt8() || which.isUInt16()) @@ -4404,8 +4410,14 @@ struct ToDateTimeMonotonicity { static bool has() { return true; } - static IFunction::Monotonicity get(const IDataType & type, const Field &, const Field &) + static IFunction::Monotonicity get(const IDataType & type_with_wrappers, const Field &, const Field &) { + const IDataType * type_without_wrappers = &type_with_wrappers; + if (const auto * low_cardinality_type = typeid_cast(type_without_wrappers)) + type_without_wrappers = low_cardinality_type->getDictionaryType().get(); + if (const auto * nullable_type = typeid_cast(type_without_wrappers)) + type_without_wrappers = nullable_type->getNestedType().get(); + const IDataType & type = *type_without_wrappers; if (type.isValueRepresentedByNumber()) { auto which = WhichDataType(type); diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp index 011cc96f16a0..c7c76cbd1483 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp @@ -6,6 +6,9 @@ #include #include #include +#include +#include +#include #include #include #include @@ -187,12 +190,185 @@ Field decodePartitionDecimalByType(const String & bytes, const IDataType & type) } +namespace +{ + +enum class PartitionTransformKind : uint8_t +{ + Day, + Month, + Year, + Hour, + NotInvertible, +}; + +PartitionTransformKind parsePartitionTransformKind(const String & transform_name_src) +{ + const String transform_name = Poco::toLower(transform_name_src); + + if (transform_name == "day" || transform_name == "days" || transform_name == "date" || transform_name == "dates") + return PartitionTransformKind::Day; + if (transform_name == "month" || transform_name == "months") + return PartitionTransformKind::Month; + if (transform_name == "year" || transform_name == "years") + return PartitionTransformKind::Year; + if (transform_name == "hour" || transform_name == "hours") + return PartitionTransformKind::Hour; + return PartitionTransformKind::NotInvertible; +} + +struct Interval +{ + Int64 first; + Int64 past_last; +}; + +std::optional unitInterval(Int64 value) +{ + Int64 past_last = 0; + if (common::addOverflow(value, Int64{1}, past_last)) + return {}; + return Interval{value, past_last}; +} + +std::optional refineInterval(std::optional interval, Int64 factor) +{ + Int64 first = 0; + Int64 past_last = 0; + if (!interval || common::mulOverflow(interval->first, factor, first) || common::mulOverflow(interval->past_last, factor, past_last)) + return {}; + return Interval{first, past_last}; +} + +std::optional closedRange(std::optional interval, std::optional decimal_scale) +{ + Int64 last = 0; + if (!interval || common::subOverflow(interval->past_last, Int64{1}, last)) + return {}; + + if (decimal_scale) + return Range( + DecimalField(interval->first, *decimal_scale), true, DecimalField(last, *decimal_scale), true); + return Range(interval->first, true, last, true); +} + +std::optional dayIntervalOfMonthNum(Int64 month) +{ + auto months = unitInterval(month); + if (!months) + return {}; + + const auto & utc = DateLUT::instance("UTC"); + const auto epoch = ExtendedDayNum(0); + const auto first = utc.addMonths(epoch, months->first); + const auto past_last = utc.addMonths(epoch, months->past_last); + if (utc.toMonthNumSinceEpoch(first) != months->first || utc.toMonthNumSinceEpoch(past_last) != months->past_last) + return {}; + + return Interval{Int64{first}, Int64{past_last}}; +} + +std::optional dayIntervalOfYearNum(Int64 year) +{ + auto years = unitInterval(year); + if (!years) + return {}; + + const auto & utc = DateLUT::instance("UTC"); + const auto epoch = ExtendedDayNum(0); + const auto first = utc.addYears(epoch, years->first); + const auto past_last = utc.addYears(epoch, years->past_last); + if (utc.toYearSinceEpoch(first) != years->first || utc.toYearSinceEpoch(past_last) != years->past_last) + return {}; + + return Interval{Int64{first}, Int64{past_last}}; +} + +std::optional dayIntervalOfPartitionValue(PartitionTransformKind kind, Int64 value) +{ + switch (kind) + { + case PartitionTransformKind::Day: + return unitInterval(value); + case PartitionTransformKind::Month: + return dayIntervalOfMonthNum(value); + case PartitionTransformKind::Year: + return dayIntervalOfYearNum(value); + case PartitionTransformKind::Hour: + case PartitionTransformKind::NotInvertible: + return {}; + } + UNREACHABLE(); +} + +std::optional secondIntervalOfPartitionValue(PartitionTransformKind kind, Int64 value) +{ + static constexpr Int64 seconds_per_hour = 3600; + static constexpr Int64 seconds_per_day = 86400; + + switch (kind) + { + case PartitionTransformKind::Hour: + return refineInterval(unitInterval(value), seconds_per_hour); + case PartitionTransformKind::Day: + case PartitionTransformKind::Month: + case PartitionTransformKind::Year: + return refineInterval(dayIntervalOfPartitionValue(kind, value), seconds_per_day); + case PartitionTransformKind::NotInvertible: + return {}; + } + UNREACHABLE(); +} + +std::optional partitionValueAsInt64(const Field & partition_value) +{ + if (partition_value.getType() == Field::Types::Int64) + return partition_value.safeGet(); + + if (partition_value.getType() == Field::Types::UInt64) + { + const UInt64 value = partition_value.safeGet(); + if (value <= static_cast(std::numeric_limits::max())) + return static_cast(value); + } + + return {}; +} + +std::optional rangeOfPartitionValue(const String & transform_name, const Field & partition_value, const IDataType & source_type) +{ + const auto value = partitionValueAsInt64(partition_value); + if (!value) + return {}; + + const PartitionTransformKind kind = parsePartitionTransformKind(transform_name); + const WhichDataType which(source_type); + + if (which.isDateOrDate32()) + return closedRange(dayIntervalOfPartitionValue(kind, *value), std::nullopt); + + if (which.isDateTime()) + return closedRange(secondIntervalOfPartitionValue(kind, *value), std::nullopt); + + if (which.isDateTime64()) + { + const UInt32 scale = getDecimalScale(source_type); + return closedRange( + refineInterval(secondIntervalOfPartitionValue(kind, *value), DecimalUtils::scaleMultiplier(scale)), scale); + } + + return {}; +} + +} + PruningReturnStatus ManifestFilesPruner::canBePruned( const ProcessedManifestFileEntryPtr & entry, const std::unordered_map & entry_hyperrectangles) const { + const auto & partition_value = entry->parsed_entry->partition_key_value; + if (partition_key_condition.has_value()) { - const auto & partition_value = entry->parsed_entry->partition_key_value; std::vector index_value(partition_value.begin(), partition_value.end()); for (size_t i = 0; i < index_value.size(); ++i) { @@ -226,15 +402,33 @@ PruningReturnStatus ManifestFilesPruner::canBePruned( continue; } - auto rect_it = entry_hyperrectangles.find(column_id); - if (rect_it == entry_hyperrectangles.end()) - continue; - auto info_it = entry->parsed_entry->columns_infos.find(column_id); bool has_no_nulls = info_it != entry->parsed_entry->columns_infos.end() && info_it->second.nulls_count.has_value() && *info_it->second.nulls_count == 0; - if (has_no_nulls && !key_condition.mayBeTrueInRange(1, &rect_it->second.left, &rect_it->second.right, {name_and_type->type})) + const DataTypes data_types{name_and_type->type}; + + if (entry->common_partition_specification) + { + for (const auto & partition_field : *entry->common_partition_specification) + { + if (partition_field.source_id != column_id || partition_field.tuple_index < 0 + || static_cast(partition_field.tuple_index) >= partition_value.size()) + continue; + + auto range = rangeOfPartitionValue( + partition_field.transform_name, + partition_value[partition_field.tuple_index], + *removeNullable(name_and_type->type)); + + if (range && !key_condition.mayBeTrueInRange(1, &range->left, &range->right, data_types)) + return PruningReturnStatus::PARTITION_PRUNED; + } + } + + auto rect_it = entry_hyperrectangles.find(column_id); + if (has_no_nulls && rect_it != entry_hyperrectangles.end() + && !key_condition.mayBeTrueInRange(1, &rect_it->second.left, &rect_it->second.right, data_types)) { return PruningReturnStatus::MIN_MAX_INDEX_PRUNED; } diff --git a/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning_with_functions.py b/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning_with_functions.py new file mode 100644 index 000000000000..51bb446b5f31 --- /dev/null +++ b/tests/integration/test_storage_iceberg_with_spark/test_partition_pruning_with_functions.py @@ -0,0 +1,268 @@ +import pytest + +from helpers.iceberg_utils import ( + check_validity_and_get_prunned_files_general, + execute_spark_query_general, + get_creation_expression, + get_uuid_str, +) + + +@pytest.mark.parametrize("storage_type", ["s3", "local"]) +def test_partition_pruning_with_functions(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_partition_pruning_with_functions_" + storage_type + "_" + get_uuid_str() + + def execute_spark_query(query: str): + return execute_spark_query_general( + spark, started_cluster_iceberg_with_spark, storage_type, TABLE_NAME, query + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} ( + ts_day TIMESTAMP, + ts_hour TIMESTAMP, + ts_month TIMESTAMP, + d_year DATE, + tag INT + ) + USING iceberg + PARTITIONED BY (days(ts_day), hours(ts_hour), months(ts_month), years(d_year)) + OPTIONS('format-version'='2') + """ + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (TIMESTAMP '2024-01-20 10:00:00', TIMESTAMP '2024-01-20 10:00:00', TIMESTAMP '2024-01-20 10:00:00', DATE '2024-01-20', 1), + (TIMESTAMP '2024-01-21 11:00:00', TIMESTAMP '2024-01-20 11:00:00', TIMESTAMP '2024-02-20 10:00:00', DATE '2025-01-20', 2), + (TIMESTAMP '2024-02-20 10:00:00', TIMESTAMP '2024-01-21 10:00:00', TIMESTAMP '2024-03-20 10:00:00', DATE '2026-01-20', 3), + (TIMESTAMP '2025-02-20 10:00:00', TIMESTAMP '2024-01-21 11:00:00', TIMESTAMP '2025-01-20 10:00:00', DATE '2027-01-20', 4); + """ + ) + + creation_expression = get_creation_expression( + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True + ) + + def check_validity_and_get_prunned_files(select_expression): + settings1 = {"use_iceberg_partition_pruning": 0, "session_timezone": "UTC"} + settings2 = {"use_iceberg_partition_pruning": 1, "session_timezone": "UTC"} + return check_validity_and_get_prunned_files_general( + instance, TABLE_NAME, settings1, settings2, "IcebergPartitionPrunedFiles", select_expression + ) + + def select(where): + return f"SELECT * FROM {creation_expression} WHERE {where} ORDER BY ALL" + + # A filter that wraps the partition source column in a monotonic function must still be able to + # use the partition value, which is what https://github.com/ClickHouse/ClickHouse/issues/103433 + # reported for `toDate`. + assert check_validity_and_get_prunned_files(select("toDate(ts_day) = toDate('2024-01-20')")) == 3 + assert check_validity_and_get_prunned_files(select("toStartOfDay(ts_day) = toDateTime64('2024-01-20 00:00:00', 6)")) == 3 + assert check_validity_and_get_prunned_files(select("toYYYYMMDD(ts_day) = 20240120")) == 3 + assert check_validity_and_get_prunned_files(select("toDate(ts_day) IN (toDate('2024-01-20'), toDate('2024-02-20'))")) == 2 + assert check_validity_and_get_prunned_files(select("toDate(ts_day) > toDate('2024-02-01')")) == 2 + assert check_validity_and_get_prunned_files(select("toYear(ts_day) = 2025")) == 3 + assert check_validity_and_get_prunned_files(select("toStartOfHour(ts_hour) = toDateTime64('2024-01-20 10:00:00', 6)")) == 3 + assert check_validity_and_get_prunned_files(select("toDate(ts_hour) = toDate('2024-01-21')")) == 2 + assert check_validity_and_get_prunned_files(select("toStartOfMonth(ts_month) = toDate('2024-02-01')")) == 3 + assert check_validity_and_get_prunned_files(select("toYYYYMM(ts_month) = 202503")) == 4 + assert check_validity_and_get_prunned_files(select("toYear(d_year) = 2026")) == 3 + assert check_validity_and_get_prunned_files(select("toStartOfYear(d_year) = toDate('2028-01-01')")) == 4 + + # A constant that is not aligned to the transform's granularity: 10:30 is inside the 10:00 hour, + # so the file of that hour must survive, and the same for a day and a month boundary. + assert check_validity_and_get_prunned_files(select("toStartOfHour(ts_hour) < toDateTime64('2024-01-20 10:30:00', 6)")) == 3 + assert check_validity_and_get_prunned_files(select("toStartOfHour(ts_hour) >= toDateTime64('2024-01-20 10:30:00', 6)")) == 1 + assert check_validity_and_get_prunned_files(select("toDate(ts_day) < toDate('2024-01-20') + INTERVAL 12 HOUR")) == 3 + assert check_validity_and_get_prunned_files(select("toStartOfMonth(ts_month) > toDate('2024-02-10')")) == 2 + + # A single day is one weekday, so the partition value answers this too: only 2024-01-20 is a Saturday. + assert check_validity_and_get_prunned_files(select("toDayOfWeek(ts_day) = 6")) == 3 + + # The partition value of a `day` transform says nothing about the hour of the day, so a filter on + # it must not prune anything. + assert check_validity_and_get_prunned_files(select("toHour(ts_day) = 10")) == 0 + + +@pytest.mark.parametrize("storage_type", ["s3", "local"]) +def test_partition_pruning_with_functions_before_epoch(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_partition_pruning_before_epoch_" + storage_type + "_" + get_uuid_str() + + def execute_spark_query(query: str): + return execute_spark_query_general( + spark, started_cluster_iceberg_with_spark, storage_type, TABLE_NAME, query + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} (ts TIMESTAMP, d DATE, tag INT) + USING iceberg + PARTITIONED BY (days(ts), years(d)) + OPTIONS('format-version'='2') + """ + ) + + # The transforms count from 1970, so these partition values are negative. + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (TIMESTAMP '1969-11-15 07:00:00', DATE '1969-11-15', 1), + (TIMESTAMP '1969-12-31 23:00:00', DATE '1970-01-01', 2), + (TIMESTAMP '1970-01-01 00:30:00', DATE '1971-06-01', 3), + (TIMESTAMP '2024-01-20 10:00:00', DATE '2024-01-20', 4); + """ + ) + + creation_expression = get_creation_expression( + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True + ) + + def check_validity_and_get_prunned_files(select_expression): + settings1 = {"use_iceberg_partition_pruning": 0, "session_timezone": "UTC"} + settings2 = {"use_iceberg_partition_pruning": 1, "session_timezone": "UTC"} + return check_validity_and_get_prunned_files_general( + instance, TABLE_NAME, settings1, settings2, "IcebergPartitionPrunedFiles", select_expression + ) + + def select(where): + return f"SELECT * FROM {creation_expression} WHERE {where} ORDER BY ALL" + + assert check_validity_and_get_prunned_files(select("toDate32(ts) = toDate32('1969-11-15')")) == 3 + assert check_validity_and_get_prunned_files(select("toDate32(ts) >= toDate32('1970-01-01')")) == 2 + assert check_validity_and_get_prunned_files(select("toDate32(ts) < toDate32('1970-01-01')")) == 2 + assert check_validity_and_get_prunned_files(select("toYear(d) = 1969")) == 3 + assert check_validity_and_get_prunned_files(select("toYear(d) >= 1971")) == 2 + + +@pytest.mark.parametrize("storage_type", ["s3", "local"]) +def test_partition_pruning_with_functions_before_epoch_separate_commits(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_partition_pruning_before_epoch_separate_commits_" + storage_type + "_" + get_uuid_str() + + def execute_spark_query(query: str): + return execute_spark_query_general( + spark, started_cluster_iceberg_with_spark, storage_type, TABLE_NAME, query + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} (ts TIMESTAMP, d DATE, tag INT) + USING iceberg + PARTITIONED BY (days(ts), years(d)) + OPTIONS('format-version'='2') + """ + ) + + # One manifest file per commit, so the partition summaries of the first manifest are negative on + # both of their sides, while the ones of the second manifest are positive on both. + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (TIMESTAMP '1969-11-15 07:00:00', DATE '1969-11-15', 1), + (TIMESTAMP '1969-12-31 23:00:00', DATE '1969-06-01', 2); + """ + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (TIMESTAMP '2024-01-20 10:00:00', DATE '2024-01-20', 3); + """ + ) + + creation_expression = get_creation_expression( + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True + ) + + def check_validity_and_get_prunned_files(select_expression): + settings1 = {"use_iceberg_partition_pruning": 0, "session_timezone": "UTC"} + settings2 = {"use_iceberg_partition_pruning": 1, "session_timezone": "UTC"} + return check_validity_and_get_prunned_files_general( + instance, TABLE_NAME, settings1, settings2, "IcebergPartitionPrunedFiles", select_expression + ) + + def select(where): + return f"SELECT * FROM {creation_expression} WHERE {where} ORDER BY ALL" + + assert check_validity_and_get_prunned_files(select("toDate32(ts) = toDate32('1969-11-15')")) == 2 + assert check_validity_and_get_prunned_files(select("toDate32(ts) >= toDate32('1970-01-01')")) == 2 + assert check_validity_and_get_prunned_files(select("toYear(d) = 1969")) == 1 + + +@pytest.mark.parametrize("storage_type", ["s3", "local"]) +def test_partition_pruning_without_functions(started_cluster_iceberg_with_spark, storage_type): + instance = started_cluster_iceberg_with_spark.instances["node1"] + spark = started_cluster_iceberg_with_spark.spark_session + TABLE_NAME = "test_partition_pruning_without_functions_" + storage_type + "_" + get_uuid_str() + + def execute_spark_query(query: str): + return execute_spark_query_general( + spark, started_cluster_iceberg_with_spark, storage_type, TABLE_NAME, query + ) + + execute_spark_query( + f""" + CREATE TABLE {TABLE_NAME} ( + ts_day TIMESTAMP, + ts_hour TIMESTAMP, + ts_month TIMESTAMP, + d_year DATE, + tag INT + ) + USING iceberg + PARTITIONED BY (days(ts_day), hours(ts_hour), months(ts_month), years(d_year)) + OPTIONS('format-version'='2') + """ + ) + + execute_spark_query( + f""" + INSERT INTO {TABLE_NAME} VALUES + (TIMESTAMP '2024-01-20 10:00:00', TIMESTAMP '2024-01-20 10:00:00', TIMESTAMP '2024-01-20 10:00:00', DATE '2024-01-20', 1), + (TIMESTAMP '2024-01-21 11:00:00', TIMESTAMP '2024-01-20 11:00:00', TIMESTAMP '2024-02-20 10:00:00', DATE '2025-01-20', 2), + (TIMESTAMP '2024-02-20 10:00:00', TIMESTAMP '2024-01-21 10:00:00', TIMESTAMP '2024-03-20 10:00:00', DATE '2026-01-20', 3), + (TIMESTAMP '2025-02-20 10:00:00', TIMESTAMP '2024-01-21 11:00:00', TIMESTAMP '2025-01-20 10:00:00', DATE '2027-01-20', 4); + """ + ) + + creation_expression = get_creation_expression( + storage_type, TABLE_NAME, started_cluster_iceberg_with_spark, table_function=True + ) + + def check_validity_and_get_prunned_files(select_expression): + settings1 = {"use_iceberg_partition_pruning": 0, "session_timezone": "UTC"} + settings2 = {"use_iceberg_partition_pruning": 1, "session_timezone": "UTC"} + return check_validity_and_get_prunned_files_general( + instance, TABLE_NAME, settings1, settings2, "IcebergPartitionPrunedFiles", select_expression + ) + + def select(where): + return f"SELECT * FROM {creation_expression} WHERE {where} ORDER BY ALL" + + assert check_validity_and_get_prunned_files(select("ts_day = '2024-01-20 10:00:00'")) == 3 + assert check_validity_and_get_prunned_files(select("ts_day < '2024-01-21 00:00:00'")) == 3 + assert check_validity_and_get_prunned_files(select("ts_day >= '2024-02-01 00:00:00'")) == 2 + assert check_validity_and_get_prunned_files(select("ts_day BETWEEN '2024-01-21 00:00:00' AND '2024-02-21 00:00:00'")) == 2 + assert check_validity_and_get_prunned_files(select("ts_day IN ('2024-01-20 10:00:00', '2024-02-20 10:00:00')")) == 2 + + assert check_validity_and_get_prunned_files(select("ts_hour = '2024-01-20 11:00:00'")) == 3 + assert check_validity_and_get_prunned_files(select("ts_hour > '2024-01-21 09:00:00'")) == 2 + + assert check_validity_and_get_prunned_files(select("ts_month = '2024-02-20 10:00:00'")) == 3 + assert check_validity_and_get_prunned_files(select("ts_month < '2024-03-01 00:00:00'")) == 2 + + assert check_validity_and_get_prunned_files(select("d_year = '2026-01-20'")) == 3 + assert check_validity_and_get_prunned_files(select("d_year < '2025-01-01'")) == 3 + assert check_validity_and_get_prunned_files(select("d_year >= '2026-01-01'")) == 2 + assert check_validity_and_get_prunned_files(select("d_year IN ('2024-01-20', '2026-01-20')")) == 2 + + assert check_validity_and_get_prunned_files(select("ts_day = '2024-01-20 12:00:00'")) == 3 + assert check_validity_and_get_prunned_files(select("d_year = '2026-06-01'")) == 3 diff --git a/tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.reference b/tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.reference new file mode 100644 index 000000000000..19219d2f6180 --- /dev/null +++ b/tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.reference @@ -0,0 +1,8 @@ +Granules: 2/10 +Granules: 2/10 +Granules: 2/10 +3 15 +3 6 +6 105 +6 105 +9 120 diff --git a/tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.sql b/tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.sql new file mode 100644 index 000000000000..aae7082ce17b --- /dev/null +++ b/tests/queries/0_stateless/05054_nullable_key_todate_monotonicity.sql @@ -0,0 +1,33 @@ +-- The monotonicity of a conversion does not depend on the `Nullable` wrapper of its argument, so a +-- `toDate`/`toDateTime` predicate over a `Nullable` key must still be usable for index analysis. + +SET session_timezone = 'UTC'; + +DROP TABLE IF EXISTS t_nullable_key; +CREATE TABLE t_nullable_key (x Nullable(DateTime64(6)), y Int64) +ENGINE = MergeTree ORDER BY x SETTINGS index_granularity = 4, allow_nullable_key = 1; + +INSERT INTO t_nullable_key +SELECT if(number % 7 = 0, NULL, toDateTime64('2026-03-01 00:00:00', 6) + INTERVAL number * 6 HOUR), number +FROM numbers(40); + +SELECT extract(explain, 'Granules: \\d+/\\d+') AS granules FROM ( + EXPLAIN indexes = 1 SELECT count() FROM t_nullable_key WHERE toDate(x) = toDate('2026-03-02') +) WHERE granules != ''; + +SELECT extract(explain, 'Granules: \\d+/\\d+') AS granules FROM ( + EXPLAIN indexes = 1 SELECT count() FROM t_nullable_key WHERE toDate32(x) = toDate32('2026-03-02') +) WHERE granules != ''; + +SELECT extract(explain, 'Granules: \\d+/\\d+') AS granules FROM ( + EXPLAIN indexes = 1 SELECT count() FROM t_nullable_key WHERE toDateTime(x) = toDateTime('2026-03-02 06:00:00') +) WHERE granules != ''; + +-- The rows the index analysis keeps must be the rows the predicate selects, `NULL`s included. +SELECT count(), sum(y) FROM t_nullable_key WHERE toDate(x) = toDate('2026-03-02'); +SELECT count(), sum(y) FROM t_nullable_key WHERE toDate(x) < toDate('2026-03-02'); +SELECT count(), sum(y) FROM t_nullable_key WHERE toDate(x) IS NULL; +SELECT count(), sum(y) FROM t_nullable_key WHERE toDate(x) IN (toDate('2026-03-02'), toDate('2026-03-08')); +SELECT count(), sum(y) FROM t_nullable_key WHERE isNull(x) OR toDate(x) = toDate('2026-03-02'); + +DROP TABLE t_nullable_key; From 2e64dd6dcd788d25c8fa269a2dc20e232bae8cbd Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sat, 26 Sep 2026 09:16:29 +0000 Subject: [PATCH 057/185] Backport #122222 to 26.8: Do not wait for the peer in TLS shutdown when a write is still pending --- .../include/Poco/Net/SecureSocketImpl.h | 3 + .../NetSSL_OpenSSL/src/SecureSocketImpl.cpp | 8 +- src/Common/tests/gtest_ssl_send_timeout.cpp | 126 ++++++++++++++++++ 3 files changed, 136 insertions(+), 1 deletion(-) diff --git a/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h b/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h index 82b38890b791..9e295ce5d299 100644 --- a/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h +++ b/base/poco/NetSSL_OpenSSL/include/Poco/Net/SecureSocketImpl.h @@ -286,6 +286,9 @@ namespace Net Context::Ptr _pContext; bool _needHandshake; bool _fatalError; + bool _pendingWrite = false; + /// Whether the last `SSL_write` returned `SSL_ERROR_WANT_WRITE`. OpenSSL keeps that record pending + /// until `SSL_write` is retried, even after other operations change what `SSL_get_error` reports. std::string _peerHostName; Session::Ptr _pSession; const BIO_METHOD * _bioMethod = nullptr; diff --git a/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp b/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp index 3636d4b5fdd3..61e0a720ba61 100644 --- a/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp +++ b/base/poco/NetSSL_OpenSSL/src/SecureSocketImpl.cpp @@ -171,6 +171,7 @@ void SecureSocketImpl::acceptSSL() ScopedLock lock(*_mutex); poco_assert (!_pSSL); _fatalError = false; + _pendingWrite = false; BIO* pBIO = BIO_new(getBioMethod()); if (!pBIO) throw SSLException("Cannot create BIO object"); @@ -241,6 +242,7 @@ void SecureSocketImpl::connectSSL(bool performHandshake) poco_assert (!_pSSL); poco_assert (_pSocket->initialized()); _fatalError = false; + _pendingWrite = false; BIO* pBIO = BIO_new(getBioMethod()); if (!pBIO) throw SSLException("Cannot create SSL BIO object"); @@ -364,7 +366,10 @@ void SecureSocketImpl::shutdown() return SSL_shutdown(_pSSL); }, false); } - while (result.rc < 0 && mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); + /// OpenSSL does not dispatch the `close_notify` alert while a record write from an + /// earlier `SSL_write` is still pending, so retrying cannot make progress then. + while (!_pendingWrite && result.rc < 0 + && mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); if (result.rc < 0) handleError(result.rc, result.sslError, result.socketError, result.errorCode); if (_pSocket->getBlocking()) @@ -419,6 +424,7 @@ int SecureSocketImpl::sendBytes(const void* buffer, int length, int flags) { return SSL_write(_pSSL, buffer, length); }); + _pendingWrite = result.sslError == SSL_ERROR_WANT_WRITE; } while (mustRetry(result.rc, result.sslError, result.socketError, remaining_time)); rc = result.rc; diff --git a/src/Common/tests/gtest_ssl_send_timeout.cpp b/src/Common/tests/gtest_ssl_send_timeout.cpp index 7f56fd605945..e17f43b4add7 100644 --- a/src/Common/tests/gtest_ssl_send_timeout.cpp +++ b/src/Common/tests/gtest_ssl_send_timeout.cpp @@ -6,6 +6,7 @@ #include #include +#include #include #include #include @@ -15,6 +16,12 @@ #include +#include + +#include + +#include +#include #include #include #include @@ -100,6 +107,125 @@ TEST(SSLSocketTimeout, SendBytesThrowsTimeoutOnBlockingSocket) } +namespace +{ + +/// Checks that shutting a blocking SSL socket down after its write timed out returns +/// promptly, instead of waiting for the peer for another full I/O timeout and throwing. +void checkShutdownAfterSendTimeout(bool receive_timeout_before_shutdown) +{ + EphemeralCert cert; + auto server_ctx = cert.makeContext(Poco::Net::Context::SERVER_USE); + auto client_ctx = cert.makeContext(Poco::Net::Context::CLIENT_USE); + + Poco::Net::SecureServerSocket server_socket( + Poco::Net::SocketAddress("127.0.0.1", 0), 1, server_ctx); + auto port = server_socket.address().port(); + + std::atomic server_done{false}; + + /// Server thread: accept and handshake, then sit idle (never read). + std::jthread server_thread([&] + { + try + { + auto accepted = server_socket.acceptConnection(); + /// Handshake happens on first I/O. Do a small read to trigger it. + char buf[1]; + try { accepted.receiveBytes(buf, 1); } catch (...) {} /// Ok: handshake may fail. NOLINT(bugprone-empty-catch) + /// Keep the connection open, and unread, until the test completes. + while (!server_done.load()) + std::this_thread::sleep_for(std::chrono::milliseconds(50)); + } + catch (...) {} /// Ok: server thread cleanup, test checks client-side behavior. NOLINT(bugprone-empty-catch) + }); + + /// Declared after the thread so that it runs before the thread is joined, on every exit path. + /// Closing the listening socket also unblocks acceptConnection if the client never connected. + SCOPE_EXIT(server_done.store(true); server_socket.close()); + + std::optional client; + try + { + client.emplace(Poco::Net::SocketAddress("127.0.0.1", port), client_ctx); + } + catch (const Poco::Exception & e) + { + /// Connection setup can fail on some systems; skip gracefully. + GTEST_SKIP() << "SSL setup failed: " << e.displayText(); + } + + /// Very short send timeout so the test doesn't wait long. + client->setSendTimeout(Poco::Timespan(0, 200'000)); /// 200ms + + /// Write enough data to fill the TCP send buffer and SSL buffer. + /// Typical TCP buffer is 128KB-256KB. Write 4MB to be sure. + std::vector data(4 * 1024 * 1024, 'X'); + + bool got_timeout = false; + try + { + size_t offset = 0; + while (offset < data.size()) + { + int sent = client->sendBytes(data.data() + offset, static_cast(data.size() - offset)); + if (sent > 0) + offset += sent; + else + break; + } + } + catch (const Poco::TimeoutException &) + { + got_timeout = true; + } + + /// Without a timed out write there is no unsent data left behind and nothing to test. + ASSERT_TRUE(got_timeout) << "Expected Poco::TimeoutException when writing to a non-reading SSL peer"; + + auto * client_impl = static_cast(client->impl()); + SSL * ssl = client_impl->ssl(); + ASSERT_NE(ssl, nullptr); + + if (receive_timeout_before_shutdown) + { + /// The peer never writes either, so the read times out too. + client->setReceiveTimeout(Poco::Timespan(0, 200'000)); /// 200ms + char buf[1]; + EXPECT_THROW(client->receiveBytes(buf, 1), Poco::TimeoutException); + /// The read replaces the state that `SSL_want_write` reports, while the write stays pending. + ASSERT_FALSE(SSL_want_write(ssl)); + } + + /// The shutdown budget is max(send timeout, receive timeout), so raising the receive + /// timeout now separates a shutdown that waits for the peer from one that does not. + client->setReceiveTimeout(Poco::Timespan(10, 0)); /// 10s + + const auto started = std::chrono::steady_clock::now(); + EXPECT_NO_THROW(client->shutdown()); + const auto elapsed_ms = std::chrono::duration_cast( + std::chrono::steady_clock::now() - started).count(); + + EXPECT_LT(elapsed_ms, 2000) << "shutdown() blocked for " << elapsed_ms + << "ms waiting for a peer that is not reading"; + + /// The orderly TLS shutdown must still be attempted once, so skipping it entirely does not pass. + EXPECT_TRUE(SSL_get_shutdown(ssl) & SSL_SENT_SHUTDOWN); +} + +} + +TEST(SSLSocketTimeout, ShutdownAfterSendTimeoutDoesNotWaitForPeer) +{ + checkShutdownAfterSendTimeout(/* receive_timeout_before_shutdown= */ false); +} + +TEST(SSLSocketTimeout, ShutdownAfterSendAndReceiveTimeoutsDoesNotWaitForPeer) +{ + checkShutdownAfterSendTimeout(/* receive_timeout_before_shutdown= */ true); +} + + /// Test that SSL handshake throws TimeoutException when the peer /// is a plain TCP listener that never speaks SSL. /// No server thread needed -- the kernel's listen backlog completes the From d7f2b255cfa15d6e1e7f2b421743ad66f4b58758 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sat, 26 Sep 2026 15:54:22 +0000 Subject: [PATCH 058/185] Update autogenerated version to 26.8.12.53 and contributors --- cmake/autogenerated_versions.txt | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cmake/autogenerated_versions.txt b/cmake/autogenerated_versions.txt index 16b893d9fd0d..88fe83ae7ce7 100644 --- a/cmake/autogenerated_versions.txt +++ b/cmake/autogenerated_versions.txt @@ -2,11 +2,11 @@ # NOTE: VERSION_REVISION has nothing common with DBMS_TCP_PROTOCOL_VERSION, # only DBMS_TCP_PROTOCOL_VERSION should be incremented on protocol changes. -SET(VERSION_REVISION 54524) +SET(VERSION_REVISION 54525) SET(VERSION_MAJOR 26) SET(VERSION_MINOR 8) -SET(VERSION_PATCH 12) -SET(VERSION_GITHASH 97da26e305d9764402a30702c9b779745bf557c4) -SET(VERSION_DESCRIBE v26.8.12.1-lts) -SET(VERSION_STRING 26.8.12.1) +SET(VERSION_PATCH 13) +SET(VERSION_GITHASH 1968835eab6e2d2030697e4ff1ec5692d628fb13) +SET(VERSION_DESCRIBE v26.8.13.1-lts) +SET(VERSION_STRING 26.8.13.1) # end of autochange From 18eed45f781a95d4b9540f1cb7a2a8e83c72d172 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sat, 26 Sep 2026 17:33:37 +0000 Subject: [PATCH 059/185] Backport #121711 to 26.8: Split a small single-level `GROUP BY` result into several chunks when the aggregation output fans out --- src/Interpreters/Aggregator.cpp | 62 ++++++---- src/Interpreters/Aggregator.h | 23 ++-- src/Processors/QueryPlan/AggregatingStep.cpp | 14 ++- .../QueryPlan/MergingAggregatedStep.cpp | 3 +- .../Transforms/AggregatingTransform.cpp | 29 ++++- .../Transforms/AggregatingTransform.h | 8 +- .../Transforms/MergingAggregatedTransform.cpp | 6 +- .../Transforms/MergingAggregatedTransform.h | 5 +- ...aggregation_single_level_result_fanout.xml | 111 ++++++++++++++++++ ...ation_single_level_result_fanout.reference | 6 + ...aggregation_single_level_result_fanout.sql | 30 +++++ 11 files changed, 255 insertions(+), 42 deletions(-) create mode 100644 tests/performance/aggregation_single_level_result_fanout.xml create mode 100644 tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.reference create mode 100644 tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.sql diff --git a/src/Interpreters/Aggregator.cpp b/src/Interpreters/Aggregator.cpp index 0d5901bff0a5..b526a8ac129d 100644 --- a/src/Interpreters/Aggregator.cpp +++ b/src/Interpreters/Aggregator.cpp @@ -117,6 +117,21 @@ bool worthConvertToTwoLevel( || (group_by_two_level_threshold_bytes && result_size_bytes >= static_cast(group_by_two_level_threshold_bytes)); } +/// The row capacity of each chunk that `convertToBlockImpl` emits. +1 for `nullKeyData`: if the table +/// doesn't have it, that's not a problem, just memory for one excessive row is preallocated. +/// A non-zero `max_rows_per_block` lowers the `max_block_size` bound so that a table smaller than +/// one block can still be emitted as several chunks. +size_t convertedBlockSize(size_t table_size, size_t max_block_size, size_t max_rows_per_block, bool return_single_block) +{ + if (return_single_block) + return table_size + 1; + + if (max_rows_per_block) + max_block_size = std::min(max_block_size, max_rows_per_block); + + return std::min(max_block_size, table_size) + 1; +} + void initDataVariantsWithSizeHint( DB::AggregatedDataVariants & result, DB::AggregatedDataVariants::Type method_chosen, const DB::Aggregator::Params & params) { @@ -3334,7 +3349,7 @@ void Aggregator::disableMinMaxOptimizationForFixedHashMaps(ManyAggregatedDataVar template requires SetAggregationMethod Chunks -Aggregator::convertToBlockImpl(Method & method, Table & data, Arena *, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block) const +Aggregator::convertToBlockImpl(Method & method, Table & data, Arena *, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block, size_t max_rows_per_block) const { if (data.empty()) { @@ -3344,7 +3359,7 @@ Aggregator::convertToBlockImpl(Method & method, Table & data, Arena *, Arenas & return result; } - Chunks res = convertToBlockImplKeysOnly(method, data, aggregates_pools, final, return_single_block); + Chunks res = convertToBlockImplKeysOnly(method, data, aggregates_pools, final, return_single_block, max_rows_per_block); /// In order to release memory early. data.clearAndShrink(); @@ -3355,7 +3370,7 @@ Aggregator::convertToBlockImpl(Method & method, Table & data, Arena *, Arenas & template requires MapAggregationMethod Chunks -Aggregator::convertToBlockImpl(Method & method, Table & data, Arena * arena, Arenas & aggregates_pools, bool final,size_t rows, bool return_single_block) const +Aggregator::convertToBlockImpl(Method & method, Table & data, Arena * arena, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block, size_t max_rows_per_block) const { if (data.empty()) { @@ -3368,8 +3383,7 @@ Aggregator::convertToBlockImpl(Method & method, Table & data, Arena * arena, Are if (is_simple_count) { - /// +1 for nullKeyData, if `data` doesn't have it - not a problem, just some memory for one excessive row will be preallocated - const size_t max_block_size = (return_single_block ? data.size() : std::min(params.max_block_size, data.size())) + 1; + const size_t max_block_size = convertedBlockSize(data.size(), params.max_block_size, max_rows_per_block, return_single_block); std::optional out_cols; std::optional shuffled_key_sizes; @@ -3483,11 +3497,11 @@ Aggregator::convertToBlockImpl(Method & method, Table & data, Arena * arena, Are #if USE_EMBEDDED_COMPILER use_compiled_functions = compiled_aggregate_functions_holder != nullptr && !Method::low_cardinality_optimization; #endif - res = convertToBlockImplFinal(method, data, arena, aggregates_pools, use_compiled_functions, return_single_block); + res = convertToBlockImplFinal(method, data, arena, aggregates_pools, use_compiled_functions, return_single_block, max_rows_per_block); } else { - res = convertToBlockImplNotFinal(method, data, aggregates_pools, rows, return_single_block); + res = convertToBlockImplNotFinal(method, data, aggregates_pools, rows, return_single_block, max_rows_per_block); } /// In order to release memory early. @@ -3661,10 +3675,9 @@ Chunk Aggregator::insertResultsIntoColumns( template requires SetAggregationMethod Chunks Aggregator::convertToBlockImplKeysOnly( - Method & method, Table & data, Arenas & aggregates_pools, bool final, bool return_single_block) const + Method & method, Table & data, Arenas & aggregates_pools, bool final, bool return_single_block, size_t max_rows_per_block) const { - /// +1 for nullKeyData, if `data` doesn't have it - not a problem, just some memory for one excessive row will be preallocated - const size_t max_block_size = (return_single_block ? data.size() : std::min(params.max_block_size, data.size())) + 1; + const size_t max_block_size = convertedBlockSize(data.size(), params.max_block_size, max_rows_per_block, return_single_block); std::optional out_cols; std::optional shuffled_key_sizes; @@ -3730,10 +3743,10 @@ Chunks Aggregator::convertToBlockImplFinal( Arena * arena, Arenas & aggregates_pools, bool use_compiled_functions [[maybe_unused]], - bool return_single_block) const + bool return_single_block, + size_t max_rows_per_block) const { - /// +1 for nullKeyData, if `data` doesn't have it - not a problem, just some memory for one excessive row will be preallocated - const size_t max_block_size = (return_single_block ? data.size() : std::min(params.max_block_size, data.size())) + 1; + const size_t max_block_size = convertedBlockSize(data.size(), params.max_block_size, max_rows_per_block, return_single_block); const bool final = true; std::optional out_cols; @@ -3811,10 +3824,9 @@ Chunks Aggregator::convertToBlockImplFinal( template Chunks NO_INLINE -Aggregator::convertToBlockImplNotFinal(Method & method, Table & data, Arenas & aggregates_pools, size_t, bool return_single_block) const +Aggregator::convertToBlockImplNotFinal(Method & method, Table & data, Arenas & aggregates_pools, size_t, bool return_single_block, size_t max_rows_per_block) const { - /// +1 for nullKeyData, if `data` doesn't have it - not a problem, just some memory for one excessive row will be preallocated - const size_t max_block_size = (return_single_block ? data.size() : std::min(params.max_block_size, data.size())) + 1; + const size_t max_block_size = convertedBlockSize(data.size(), params.max_block_size, max_rows_per_block, return_single_block); const bool final = false; Chunks res_chunks; @@ -3989,7 +4001,7 @@ Aggregator::AggregatedChunk Aggregator::prepareChunkAndFillWithoutKey(Aggregated template std::conditional_t -Aggregator::prepareChunkAndFillSingleLevel(AggregatedDataVariants & data_variants, bool final) const +Aggregator::prepareChunkAndFillSingleLevel(AggregatedDataVariants & data_variants, bool final, size_t max_rows_per_block) const { Chunks res_variant; const size_t rows = data_variants.sizeWithoutOverflowRow(); @@ -3997,7 +4009,7 @@ Aggregator::prepareChunkAndFillSingleLevel(AggregatedDataVariants & data_variant else if (data_variants.type == AggregatedDataVariants::Type::NAME) \ { \ res_variant = convertToBlockImpl( \ - *data_variants.NAME, data_variants.NAME->data, data_variants.aggregates_pool, data_variants.aggregates_pools, final, rows, return_single_block); \ + *data_variants.NAME, data_variants.NAME->data, data_variants.aggregates_pool, data_variants.aggregates_pools, final, rows, return_single_block, max_rows_per_block); \ } if (false) {} // NOLINT @@ -4092,7 +4104,7 @@ Aggregator::AggregatedChunks Aggregator::prepareChunksAndFillTwoLevelImpl(Aggreg } -Aggregator::AggregatedChunks Aggregator::convertToChunks(AggregatedDataVariants & data_variants, bool final) const +Aggregator::AggregatedChunks Aggregator::convertToChunks(AggregatedDataVariants & data_variants, bool final, size_t max_rows_per_block) const { LOG_TRACE(log, "Converting aggregated data to chunks"); @@ -4111,7 +4123,7 @@ Aggregator::AggregatedChunks Aggregator::convertToChunks(AggregatedDataVariants if (data_variants.type != AggregatedDataVariants::Type::without_key) { if (!data_variants.isTwoLevel()) - chunks.splice(chunks.end(), prepareChunkAndFillSingleLevel(data_variants, final)); + chunks.splice(chunks.end(), prepareChunkAndFillSingleLevel(data_variants, final, max_rows_per_block)); else chunks.splice(chunks.end(), prepareChunksAndFillTwoLevel(data_variants, final)); } @@ -4142,6 +4154,16 @@ Aggregator::AggregatedChunks Aggregator::convertToChunks(AggregatedDataVariants return chunks; } +size_t Aggregator::singleLevelChunkRowsForFanOut(size_t rows, size_t output_streams) +{ + static constexpr size_t MIN_ROWS_PER_CHUNK{512}; + const size_t num_chunks = std::clamp(rows / MIN_ROWS_PER_CHUNK, 1, std::max(output_streams, 1)); + if (num_chunks <= 1) + return 0; + + return (rows + num_chunks - 1) / num_chunks; +} + template void NO_INLINE Aggregator::mergeDataNullKey( diff --git a/src/Interpreters/Aggregator.h b/src/Interpreters/Aggregator.h index 2ae75015b007..48c2e5c8a9f2 100644 --- a/src/Interpreters/Aggregator.h +++ b/src/Interpreters/Aggregator.h @@ -460,8 +460,15 @@ class Aggregator final * If final = false, then ColumnAggregateFunction is created as the aggregation columns with the state of the calculations, * which can then be combined with other states (for distributed query processing). * If final = true, then columns with ready values are created as aggregate columns. + * A non-zero `max_rows_per_block` caps the size of the emitted single-level chunks below `max_block_size`. */ - AggregatedChunks convertToChunks(AggregatedDataVariants & data_variants, bool final) const; + AggregatedChunks convertToChunks(AggregatedDataVariants & data_variants, bool final, size_t max_rows_per_block = 0) const; + + /// A single-level result smaller than `max_block_size` is converted to one chunk, and a `Resize` + /// hands out whole chunks, so everything downstream of it runs in one thread. Returns a chunk size + /// that splits `rows` into about one chunk per output stream, never below 512 rows per chunk, or 0 + /// to leave the result as is. + static size_t singleLevelChunkRowsForFanOut(size_t rows, size_t output_streams); /// `adaptive_session` (or nullptr when the adaptive aggregation is off) feeds the /// thaw verdict into the hash-table statistics next to the observed sizes. @@ -1034,13 +1041,13 @@ class Aggregator final template requires MapAggregationMethod Chunks - convertToBlockImpl(Method & method, Table & data, Arena * arena, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block) const; + convertToBlockImpl(Method & method, Table & data, Arena * arena, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block, size_t max_rows_per_block = 0) const; /// A set method skips the inline-count and compiled-function paths; it only emits keys. template requires SetAggregationMethod Chunks - convertToBlockImpl(Method & method, Table & data, Arena * arena, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block) const; + convertToBlockImpl(Method & method, Table & data, Arena * arena, Arenas & aggregates_pools, bool final, size_t rows, bool return_single_block, size_t max_rows_per_block = 0) const; template void insertAggregatesIntoColumns( @@ -1060,7 +1067,7 @@ class Aggregator final template requires SetAggregationMethod Chunks convertToBlockImplKeysOnly( - Method & method, Table & data, Arenas & aggregates_pools, bool final, bool return_single_block) const; + Method & method, Table & data, Arenas & aggregates_pools, bool final, bool return_single_block, size_t max_rows_per_block) const; template Chunks convertToBlockImplFinal( @@ -1069,11 +1076,12 @@ class Aggregator final Arena * arena, Arenas & aggregates_pools, bool use_compiled_functions, - bool return_single_block) const; + bool return_single_block, + size_t max_rows_per_block) const; template Chunks - convertToBlockImplNotFinal(Method & method, Table & data, Arenas & aggregates_pools, size_t rows, bool return_single_block) const; + convertToBlockImplNotFinal(Method & method, Table & data, Arenas & aggregates_pools, size_t rows, bool return_single_block, size_t max_rows_per_block) const; /// `topk_full_key_bytes`, when non-null and the bucket goes through the Top-K conversion, /// receives the byte size all of the bucket's keys would occupy materialized: the runtime @@ -1124,9 +1132,10 @@ class Aggregator final AggregatedChunk prepareChunkAndFillWithoutKey(AggregatedDataVariants & data_variants, bool final, bool is_overflows) const; AggregatedChunks prepareChunksAndFillTwoLevel(AggregatedDataVariants & data_variants, bool final) const; + /// A non-zero `max_rows_per_block` caps the size of the emitted chunks below `max_block_size`. template std::conditional_t - prepareChunkAndFillSingleLevel(AggregatedDataVariants & data_variants, bool final) const; + prepareChunkAndFillSingleLevel(AggregatedDataVariants & data_variants, bool final, size_t max_rows_per_block = 0) const; template AggregatedChunks prepareChunksAndFillTwoLevelImpl(AggregatedDataVariants & data_variants, Method & method, bool final) const; diff --git a/src/Processors/QueryPlan/AggregatingStep.cpp b/src/Processors/QueryPlan/AggregatingStep.cpp index 272d0f4f0e35..08fe3c68f92f 100644 --- a/src/Processors/QueryPlan/AggregatingStep.cpp +++ b/src/Processors/QueryPlan/AggregatingStep.cpp @@ -563,6 +563,10 @@ void AggregatingStep::transformPipeline(QueryPipelineBuilder & pipeline, const B }); } + /// The per-set results are spread over `max_threads` streams below, so split a small single-level + /// result of each set for that width as for an ordinary aggregation. + const size_t grouping_sets_output_streams = should_produce_results_in_order_of_bucket_number ? 1 : params.max_threads; + pipeline.transform([&](OutputPortRawPtrs ports) { chassert(streams * grouping_sets_size == ports.size()); @@ -586,7 +590,8 @@ void AggregatingStep::transformPipeline(QueryPipelineBuilder & pipeline, const B new_temporary_data_merge_threads, should_produce_results_in_order_of_bucket_number, skip_merging, - nullptr); + nullptr, + grouping_sets_output_streams); // For each input stream we have `grouping_sets_size` copies, so port index // for transform #j should skip ports of first (j-1) streams. connect(*ports[i + grouping_sets_size * j], aggregation_for_set->getInputs().front()); @@ -597,7 +602,7 @@ void AggregatingStep::transformPipeline(QueryPipelineBuilder & pipeline, const B else { auto aggregation_for_set - = std::make_shared(input_header, transform_params_for_set, dataflow_cache_updater); + = std::make_shared(input_header, transform_params_for_set, dataflow_cache_updater, grouping_sets_output_streams); connect(*ports[i], aggregation_for_set->getInputs().front()); ports[i] = &aggregation_for_set->getOutputs().front(); processors.push_back(aggregation_for_set); @@ -859,7 +864,8 @@ void AggregatingStep::transformPipeline(QueryPipelineBuilder & pipeline, const B new_temporary_data_merge_threads, should_produce_results_in_order_of_bucket_number, skip_merging, - dataflow_cache_updater); + dataflow_cache_updater, + streams_after_aggregation); }); pipeline.resize(streams_after_aggregation, false, settings.min_outstreams_per_resize_after_split); @@ -869,7 +875,7 @@ void AggregatingStep::transformPipeline(QueryPipelineBuilder & pipeline, const B else { pipeline.addSimpleTransform([&](const SharedHeader & header) - { return std::make_shared(header, transform_params, dataflow_cache_updater); }); + { return std::make_shared(header, transform_params, dataflow_cache_updater, streams_after_aggregation); }); pipeline.resize(streams_after_aggregation); diff --git a/src/Processors/QueryPlan/MergingAggregatedStep.cpp b/src/Processors/QueryPlan/MergingAggregatedStep.cpp index 95528696962e..3f64bfea764a 100644 --- a/src/Processors/QueryPlan/MergingAggregatedStep.cpp +++ b/src/Processors/QueryPlan/MergingAggregatedStep.cpp @@ -151,7 +151,8 @@ void MergingAggregatedStep::transformPipeline(QueryPipelineBuilder & pipeline, c pipeline.resize(1); /// Now merge the aggregated blocks - auto transform = std::make_shared(pipeline.getSharedHeader(), params, final, grouping_sets_params); + auto transform = std::make_shared(pipeline.getSharedHeader(), params, final, grouping_sets_params, + should_produce_results_in_order_of_bucket_number ? 1 : max_threads); pipeline.addTransform(std::move(transform)); } else diff --git a/src/Processors/Transforms/AggregatingTransform.cpp b/src/Processors/Transforms/AggregatingTransform.cpp index e03ad193b77e..6a7c49764e0f 100644 --- a/src/Processors/Transforms/AggregatingTransform.cpp +++ b/src/Processors/Transforms/AggregatingTransform.cpp @@ -546,6 +546,7 @@ class ConvertingAggregatedToChunksTransform final : public IProcessor AggregatingTransformParamsPtr params_, ManyAggregatedDataVariantsPtr data_, size_t num_threads_, + size_t output_streams_, RuntimeDataflowStatisticsCacheUpdaterPtr updater_, AdaptiveAggregationSessionPtr adaptive_session_) : IProcessor({}, {params_->getHeader()}) @@ -553,6 +554,7 @@ class ConvertingAggregatedToChunksTransform final : public IProcessor , data(std::move(data_)) , shared_data(std::make_shared()) , num_threads(num_threads_) + , output_streams(output_streams_) , updater(std::move(updater_)) , adaptive_session(std::move(adaptive_session_)) { @@ -907,6 +909,11 @@ class ConvertingAggregatedToChunksTransform final : public IProcessor size_t num_threads; + /// How many streams the output is spread over downstream. It is not `num_threads`. That is capped by the + /// number of aggregating streams (1 for a single input stream), while the `Resize` after the aggregation + /// fans out to `max_threads`. 1 when the results must go out in bucket order. + size_t output_streams; + RuntimeDataflowStatisticsCacheUpdaterPtr updater; AdaptiveAggregationSessionPtr adaptive_session; @@ -1003,7 +1010,11 @@ class ConvertingAggregatedToChunksTransform final : public IProcessor throw Exception(ErrorCodes::UNKNOWN_AGGREGATED_DATA_VARIANT, "Unknown aggregated data variant."); } - auto agg_chunks = params->aggregator.prepareChunkAndFillSingleLevel(*first, params->final); + const size_t max_rows_per_block = Aggregator::singleLevelChunkRowsForFanOut(first->sizeWithoutOverflowRow(), output_streams); + if (max_rows_per_block) + LOG_TRACE(getLogger("AggregatingTransform"), "Split single level result into chunks of at most {} rows.", max_rows_per_block); + + auto agg_chunks = params->aggregator.prepareChunkAndFillSingleLevel(*first, params->final, max_rows_per_block); for (auto & agg_chunk : agg_chunks) { if (agg_chunk.chunk.getNumRows() > 0) @@ -1086,7 +1097,7 @@ class ConvertingAggregatedToChunksTransform final : public IProcessor }; AggregatingTransform::AggregatingTransform( - SharedHeader header, AggregatingTransformParamsPtr params_, RuntimeDataflowStatisticsCacheUpdaterPtr updater_) + SharedHeader header, AggregatingTransformParamsPtr params_, RuntimeDataflowStatisticsCacheUpdaterPtr updater_, size_t output_streams_) : AggregatingTransform( std::move(header), std::move(params_), @@ -1096,7 +1107,8 @@ AggregatingTransform::AggregatingTransform( 1, true /* should_produce_results_in_order_of_bucket_number */, false /* skip_merging */, - updater_) + updater_, + output_streams_) { } @@ -1109,7 +1121,8 @@ AggregatingTransform::AggregatingTransform( size_t temporary_data_merge_threads_, bool should_produce_results_in_order_of_bucket_number_, bool skip_merging_, - RuntimeDataflowStatisticsCacheUpdaterPtr updater_) + RuntimeDataflowStatisticsCacheUpdaterPtr updater_, + size_t output_streams_) : IProcessor({std::move(header)}, {params_->getHeader()}) , params(std::move(params_)) , key_columns(params->params.keys_size) @@ -1121,6 +1134,7 @@ AggregatingTransform::AggregatingTransform( , should_produce_results_in_order_of_bucket_number(should_produce_results_in_order_of_bucket_number_) , skip_merging(skip_merging_) , updater(std::move(updater_)) + , output_streams(output_streams_) { /// `AggregatingStep` leaves its engagement verdict in the flag. Without a producer nothing is ever /// staged, so the merge-time drains find empty backlogs and do nothing. @@ -1412,7 +1426,12 @@ void AggregatingTransform::initGenerate() std::move(many_data->variants), adaptive_context ? adaptive_context->session.get() : nullptr); auto prepared_data_ptr = std::make_shared(std::move(prepared_data)); processors.emplace_back(std::make_shared( - params, std::move(prepared_data_ptr), max_threads, updater, adaptive_engaged ? adaptive_context->session : nullptr)); + params, + std::move(prepared_data_ptr), + max_threads, + output_streams, + updater, + adaptive_engaged ? adaptive_context->session : nullptr)); } else { diff --git a/src/Processors/Transforms/AggregatingTransform.h b/src/Processors/Transforms/AggregatingTransform.h index 6be36c2f7677..9842346a5e2f 100644 --- a/src/Processors/Transforms/AggregatingTransform.h +++ b/src/Processors/Transforms/AggregatingTransform.h @@ -115,7 +115,7 @@ using ManyAggregatedDataPtr = std::shared_ptr; class AggregatingTransform final : public IProcessor { public: - AggregatingTransform(SharedHeader header, AggregatingTransformParamsPtr params_, RuntimeDataflowStatisticsCacheUpdaterPtr updater_); + AggregatingTransform(SharedHeader header, AggregatingTransformParamsPtr params_, RuntimeDataflowStatisticsCacheUpdaterPtr updater_, size_t output_streams_ = 1); /// For Parallel aggregating. AggregatingTransform( @@ -127,7 +127,8 @@ class AggregatingTransform final : public IProcessor size_t temporary_data_merge_threads, bool should_produce_results_in_order_of_bucket_number_ = true, bool skip_merging_ = false, - RuntimeDataflowStatisticsCacheUpdaterPtr updater_ = nullptr); + RuntimeDataflowStatisticsCacheUpdaterPtr updater_ = nullptr, + size_t output_streams_ = 1); ~AggregatingTransform() override; @@ -195,6 +196,9 @@ class AggregatingTransform final : public IProcessor RuntimeDataflowStatisticsCacheUpdaterPtr updater; + /// How many streams `AggregatingStep` spreads this transform's output over; 1 when it doesn't. + size_t output_streams = 1; + void initGenerate(); }; diff --git a/src/Processors/Transforms/MergingAggregatedTransform.cpp b/src/Processors/Transforms/MergingAggregatedTransform.cpp index 6c3a11a3944f..4e080b4931ae 100644 --- a/src/Processors/Transforms/MergingAggregatedTransform.cpp +++ b/src/Processors/Transforms/MergingAggregatedTransform.cpp @@ -64,8 +64,9 @@ static ActionsDAG makeReorderingActions(const Block & in_header, const GroupingS MergingAggregatedTransform::~MergingAggregatedTransform() = default; MergingAggregatedTransform::MergingAggregatedTransform( - SharedHeader header_, Aggregator::Params params, bool final, GroupingSetsParamsList grouping_sets_params) + SharedHeader header_, Aggregator::Params params, bool final, GroupingSetsParamsList grouping_sets_params, size_t output_streams_) : IAccumulatingTransform(header_, std::make_shared(appendGroupingIfNeeded(*header_, params.getHeader(*header_, final)))) + , output_streams(output_streams_) { if (!grouping_sets_params.empty()) { @@ -258,7 +259,8 @@ Chunk MergingAggregatedTransform::generate() /// TODO: this operation can be made async. Add async for IAccumulatingTransform. params->aggregator.mergeBlocks(std::move(bucket_to_chunks), data_variants, is_cancelled); - auto merged_chunks = params->aggregator.convertToChunks(data_variants, params->final); + const size_t max_rows_per_block = Aggregator::singleLevelChunkRowsForFanOut(data_variants.sizeWithoutOverflowRow(), output_streams); + auto merged_chunks = params->aggregator.convertToChunks(data_variants, params->final, max_rows_per_block); if (grouping_set.creating_missing_keys_actions) { diff --git a/src/Processors/Transforms/MergingAggregatedTransform.h b/src/Processors/Transforms/MergingAggregatedTransform.h index 69388c021dd3..4ace709bafe9 100644 --- a/src/Processors/Transforms/MergingAggregatedTransform.h +++ b/src/Processors/Transforms/MergingAggregatedTransform.h @@ -15,7 +15,7 @@ using ExpressionActionsPtr = std::shared_ptr; class MergingAggregatedTransform final : public IAccumulatingTransform { public: - MergingAggregatedTransform(SharedHeader header_, Aggregator::Params params_, bool final_, GroupingSetsParamsList grouping_sets_params); + MergingAggregatedTransform(SharedHeader header_, Aggregator::Params params_, bool final_, GroupingSetsParamsList grouping_sets_params, size_t output_streams_ = 1); ~MergingAggregatedTransform() override; @@ -50,6 +50,9 @@ class MergingAggregatedTransform final : public IAccumulatingTransform bool consume_started = false; bool generate_started = false; + /// How many streams `MergingAggregatedStep` spreads the output over; 1 when it doesn't. + size_t output_streams = 1; + void addChunk(Columns columns, size_t num_rows, Int32 bucket_num, bool is_overflows); }; diff --git a/tests/performance/aggregation_single_level_result_fanout.xml b/tests/performance/aggregation_single_level_result_fanout.xml new file mode 100644 index 000000000000..e7a9f16cc9f3 --- /dev/null +++ b/tests/performance/aggregation_single_level_result_fanout.xml @@ -0,0 +1,111 @@ + + + + 8 + 0 + 0 + + + CREATE TABLE single_level_fanout_t1 (oid UInt64, cid Int32, cr DateTime, h String, s1 Int32) ENGINE = MergeTree ORDER BY oid + CREATE TABLE single_level_fanout_t2 (oid UInt64) ENGINE = MergeTree ORDER BY oid + CREATE TABLE single_level_fanout_t3 (cid Int32, h String, ref DateTime, s2 Float64) ENGINE = MergeTree ORDER BY (cid, h, ref) + + + INSERT INTO single_level_fanout_t1 + SELECT + intDiv(number, 5), + toInt32(intDiv(number, 5) % 4), + toDateTime('2026-09-08 00:00:00') + toIntervalMinute(intDiv(number, 5) % 1440) + toIntervalSecond(number % 5), + toString(intDiv(number, 5) % 200), + toInt32(number % 300) + FROM numbers(300000) + + INSERT INTO single_level_fanout_t2 SELECT number FROM numbers(60000) + + INSERT INTO single_level_fanout_t3 + SELECT + toInt32(number % 4), + toString(intDiv(number, 4) % 200), + toDateTime('2026-09-08 00:00:00') + toIntervalMinute(intDiv(number, 800)), + toFloat64(number % 5) / 2 + FROM numbers(1152000) + + + + + join_algorithm + + hash + parallel_hash + + + + + = e.t0 - toIntervalMinute(3) AND p.ref <= e.t0 + toIntervalMinute(1) + GROUP BY e.oid, e.cid, e.h, e.s + ) + GROUP BY cid + FORMAT Null + SETTINGS join_algorithm = '{join_algorithm}' + ]]> + + + = e.t0 - toIntervalMinute(3) AND p.ref <= e.t0 + toIntervalMinute(1) + GROUP BY e.oid, e.cid, e.h, e.s + ) + GROUP BY cid + FORMAT Null + SETTINGS enable_parallel_single_level_merge = 1, enable_adaptive_aggregator = 1 + ]]> + + DROP TABLE IF EXISTS single_level_fanout_t1 + DROP TABLE IF EXISTS single_level_fanout_t2 + DROP TABLE IF EXISTS single_level_fanout_t3 + diff --git a/tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.reference b/tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.reference new file mode 100644 index 000000000000..a9e2f17562ae --- /dev/null +++ b/tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.reference @@ -0,0 +1,6 @@ +1 +1 +1 +1 +1 +1 diff --git a/tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.sql b/tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.sql new file mode 100644 index 000000000000..73843a147383 --- /dev/null +++ b/tests/queries/0_stateless/05257_aggregation_single_level_result_fanout.sql @@ -0,0 +1,30 @@ +-- A single-level result smaller than max_block_size is split into several chunks when the aggregation +-- output fans out, so the step after it does not run in one thread. 20000 groups stay single-level. +DROP TABLE IF EXISTS t_fanout; +CREATE TABLE t_fanout (n UInt64) ENGINE = MergeTree ORDER BY n AS SELECT number FROM numbers(200000); + +-- Several aggregating streams, serial single-level merge. +SELECT max(bs) < 20000 FROM (SELECT blockSize() AS bs FROM (SELECT number % 20000 AS k, count() FROM numbers_mt(200000) GROUP BY k)) +SETTINGS max_threads = 4, enable_parallel_single_level_merge = 0; + +-- One aggregating stream. +SELECT max(bs) < 20000 FROM (SELECT blockSize() AS bs FROM (SELECT number % 20000 AS k, count() FROM numbers(200000) GROUP BY k)) +SETTINGS max_threads = 4; + +-- Merge of results from remote shards. +SELECT max(bs) < 20000 FROM (SELECT blockSize() AS bs FROM (SELECT n % 20000 AS k, count() FROM remote('127.0.0.{1,2}', currentDatabase(), t_fanout) GROUP BY k)) +SETTINGS max_threads = 4, distributed_aggregation_memory_efficient = 0; + +-- Results are unchanged by the split. +SELECT sum(c) = 200000 AND count() = 20000 FROM (SELECT number % 20000 AS k, count() AS c FROM numbers(200000) GROUP BY k) SETTINGS max_threads = 4; + +-- A result above max_block_size is also split into about one chunk per output stream, not into +-- max_block_size chunks: 3000 groups used to come out as 2001 + 999 rows, now as four chunks of about 750. +SELECT max(bs) < 1000 FROM (SELECT blockSize() AS bs FROM (SELECT number % 3000 AS k, count() FROM numbers_mt(30000) GROUP BY k)) +SETTINGS max_threads = 4, max_block_size = 2000, enable_parallel_single_level_merge = 0; + +-- The same split applies to each set of GROUPING SETS. +SELECT max(bs) < 20000 FROM (SELECT blockSize() AS bs FROM (SELECT number % 20000 AS k, number % 3 AS g, count() FROM numbers_mt(200000) GROUP BY GROUPING SETS ((k), (g)))) +SETTINGS max_threads = 4; + +DROP TABLE t_fanout; From 783d557efa25bfee0b5d35c86787b819f5f5ddbf Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sun, 27 Sep 2026 13:33:29 +0000 Subject: [PATCH 060/185] Update autogenerated version to 26.8.13.2 and contributors --- cmake/autogenerated_versions.txt | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cmake/autogenerated_versions.txt b/cmake/autogenerated_versions.txt index 88fe83ae7ce7..af630df25c06 100644 --- a/cmake/autogenerated_versions.txt +++ b/cmake/autogenerated_versions.txt @@ -2,11 +2,11 @@ # NOTE: VERSION_REVISION has nothing common with DBMS_TCP_PROTOCOL_VERSION, # only DBMS_TCP_PROTOCOL_VERSION should be incremented on protocol changes. -SET(VERSION_REVISION 54525) +SET(VERSION_REVISION 54526) SET(VERSION_MAJOR 26) SET(VERSION_MINOR 8) -SET(VERSION_PATCH 13) -SET(VERSION_GITHASH 1968835eab6e2d2030697e4ff1ec5692d628fb13) -SET(VERSION_DESCRIBE v26.8.13.1-lts) -SET(VERSION_STRING 26.8.13.1) +SET(VERSION_PATCH 14) +SET(VERSION_GITHASH 3eac80eef9b1f81e88b581a6c107c59f21d9e15c) +SET(VERSION_DESCRIBE v26.8.14.1-lts) +SET(VERSION_STRING 26.8.14.1) # end of autochange From b44da491bb423ce04ede58be4a607ff8b83b4baf Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sun, 27 Sep 2026 15:36:45 +0000 Subject: [PATCH 061/185] Backport #117387 to 26.8: Reload TLS CA certificates (caConfig) without restart --- .../features/security/tls/configuring-tls.mdx | 13 + .../_server_settings_outside_source.mdx | 4 +- programs/keeper/Keeper.cpp | 14 +- programs/server/Server.cpp | 33 ++- src/Common/Config/ConfigReloader.cpp | 13 +- src/Coordination/KeeperServer.cpp | 6 +- src/Core/ServerSettings.cpp | 4 +- src/Server/CertificateReloader.cpp | 215 +++++++++++++- src/Server/CertificateReloader.h | 54 +++- .../System/StorageSystemCertificates.cpp | 17 +- .../test_reload_ca_certificate/__init__.py | 0 .../test_reload_ca_certificate/certs/ca.crt | 20 ++ .../test_reload_ca_certificate/certs/ca1.crt | 20 ++ .../test_reload_ca_certificate/certs/ca1.key | 28 ++ .../test_reload_ca_certificate/certs/ca2.crt | 20 ++ .../test_reload_ca_certificate/certs/ca2.key | 28 ++ .../certs/cert1.crt | 21 ++ .../certs/cert1.key | 28 ++ .../certs/cert2.crt | 21 ++ .../certs/cert2.key | 28 ++ .../certs/generate_certs.sh | 41 +++ .../test_reload_ca_certificate/certs/node.crt | 21 ++ .../test_reload_ca_certificate/certs/node.key | 28 ++ .../configs/keeper1.xml | 36 +++ .../configs/keeper2.xml | 36 +++ .../configs/keeper3.xml | 36 +++ .../configs/ssl.xml | 31 ++ .../configs/ssl_with_default_cas.xml | 29 ++ .../test_reload_ca_certificate/test.py | 275 ++++++++++++++++++ 29 files changed, 1078 insertions(+), 42 deletions(-) create mode 100644 tests/integration/test_reload_ca_certificate/__init__.py create mode 100644 tests/integration/test_reload_ca_certificate/certs/ca.crt create mode 100644 tests/integration/test_reload_ca_certificate/certs/ca1.crt create mode 100644 tests/integration/test_reload_ca_certificate/certs/ca1.key create mode 100644 tests/integration/test_reload_ca_certificate/certs/ca2.crt create mode 100644 tests/integration/test_reload_ca_certificate/certs/ca2.key create mode 100644 tests/integration/test_reload_ca_certificate/certs/cert1.crt create mode 100644 tests/integration/test_reload_ca_certificate/certs/cert1.key create mode 100644 tests/integration/test_reload_ca_certificate/certs/cert2.crt create mode 100644 tests/integration/test_reload_ca_certificate/certs/cert2.key create mode 100755 tests/integration/test_reload_ca_certificate/certs/generate_certs.sh create mode 100644 tests/integration/test_reload_ca_certificate/certs/node.crt create mode 100644 tests/integration/test_reload_ca_certificate/certs/node.key create mode 100644 tests/integration/test_reload_ca_certificate/configs/keeper1.xml create mode 100644 tests/integration/test_reload_ca_certificate/configs/keeper2.xml create mode 100644 tests/integration/test_reload_ca_certificate/configs/keeper3.xml create mode 100644 tests/integration/test_reload_ca_certificate/configs/ssl.xml create mode 100644 tests/integration/test_reload_ca_certificate/configs/ssl_with_default_cas.xml create mode 100644 tests/integration/test_reload_ca_certificate/test.py diff --git a/docs/concepts/features/security/tls/configuring-tls.mdx b/docs/concepts/features/security/tls/configuring-tls.mdx index be54610c4cdc..ca80687a4e25 100644 --- a/docs/concepts/features/security/tls/configuring-tls.mdx +++ b/docs/concepts/features/security/tls/configuring-tls.mdx @@ -645,6 +645,19 @@ For `clickhouse-client`, you can also use the `--accept-invalid-certificate` CLI ``` +## Rotating certificates and CA certificates without a restart {#rotating-certificates-without-restart} + +ClickHouse server and ClickHouse Keeper watch the files referenced by `certificateFile`, `privateKeyFile` and `caConfig` +in the `openSSL.server` and `openSSL.client` sections (and in `protocols.*` for composable protocols), or the files in the +directory if `caConfig` is a directory. When one of these files changes, or when `SYSTEM RELOAD CONFIG` is executed, the +certificates are reloaded and used for all new TLS connections, including HTTPS, the secure native protocol, interserver +connections, connections to Keeper and the Raft connections between Keeper nodes. Established connections keep using the +certificates they were opened with. + +To rotate a CA without downtime, first replace the `caConfig` file with a bundle that contains both the old and the new +CA certificate, then switch the node and client certificates to ones issued by the new CA, and finally replace the bundle +with the new CA certificate only. + ## Summary {#summary} This article focused on getting a ClickHouse environment configured with TLS. The settings will differ for different requirements in production environments; for example, certificate verification levels, protocols, ciphers, etc. But you should now have a good understanding of the steps involved in configuring and implementing secure connections. diff --git a/docs/reference/settings/server-settings/_server_settings_outside_source.mdx b/docs/reference/settings/server-settings/_server_settings_outside_source.mdx index 03357b3470b0..23c3fedd9bc7 100644 --- a/docs/reference/settings/server-settings/_server_settings_outside_source.mdx +++ b/docs/reference/settings/server-settings/_server_settings_outside_source.mdx @@ -1159,12 +1159,14 @@ SSL client/server configuration. Support for SSL is provided by the `libpoco` library. The available configuration options are explained in [SSLManager.h](https://github.com/ClickHouse-Extras/poco/blob/master/NetSSL_OpenSSL/include/Poco/Net/SSLManager.h). Default values can be found in [SSLManager.cpp](https://github.com/ClickHouse-Extras/poco/blob/master/NetSSL_OpenSSL/src/SSLManager.cpp). +The files referenced by `certificateFile`, `privateKeyFile` and `caConfig` are reloaded without a restart when they change or on `SYSTEM RELOAD CONFIG`. New connections use the reloaded certificates, established connections are not affected. + Keys for server/client settings: | Option | Description | Default Value | |-------------------------------|------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------|--------------------------------------------| | `cacheSessions` | Enables or disables caching sessions. Must be used in combination with `sessionIdContext`. Acceptable values: `true`, `false`. | `false` | -| `caConfig` | Path to the file or directory that contains trusted CA certificates. If this points to a file, it must be in PEM format and can contain several CA certificates. If this points to a directory, it must contain one .pem file per CA certificate. The filenames are looked up by the CA subject name hash value. Details can be found in the man page of [SSL_CTX_load_verify_locations](https://www.openssl.org/docs/man3.0/man3/SSL_CTX_load_verify_locations.html). | | +| `caConfig` | Path to the file or directory that contains trusted CA certificates. If this points to a file, it must be in PEM format and can contain several CA certificates. If this points to a directory, it must contain one .pem file per CA certificate. The filenames are looked up by the CA subject name hash value. Details can be found in the man page of [SSL_CTX_load_verify_locations](https://www.openssl.org/docs/man3.0/man3/SSL_CTX_load_verify_locations.html). Like `certificateFile` and `privateKeyFile`, the CA certificates are reloaded without a restart when the file changes or on `SYSTEM RELOAD CONFIG`. | | | `certificateFile` | Path to the client/server certificate file in PEM format. You can omit it if `privateKeyFile` contains the certificate. | | | `cipherList` | Supported OpenSSL encryptions. | `ALL:!ADH:!LOW:!EXP:!MD5:!3DES:@STRENGTH` | | `disableProtocols` | Protocols that are not allowed to be used. | | diff --git a/programs/keeper/Keeper.cpp b/programs/keeper/Keeper.cpp index fe530bd9d2f6..f86cb3a01521 100644 --- a/programs/keeper/Keeper.cpp +++ b/programs/keeper/Keeper.cpp @@ -626,14 +626,14 @@ try Coordination::EventPtr unused_event = std::make_shared(); - const std::string cert_path = config().getString("openSSL.server.certificateFile", ""); - const std::string key_path = config().getString("openSSL.server.privateKeyFile", ""); - + /// TLS certificates, keys and CA certificates are reloaded by CertificateReloader when these files change. std::vector extra_paths = {include_from_path}; - if (!cert_path.empty()) - extra_paths.emplace_back(cert_path); - if (!key_path.empty()) - extra_paths.emplace_back(key_path); + for (const auto * key : {"openSSL.server.certificateFile", "openSSL.server.privateKeyFile", "openSSL.server.caConfig", + "openSSL.client.certificateFile", "openSSL.client.privateKeyFile", "openSSL.client.caConfig"}) + { + if (auto file_path = config().getString(key, ""); !file_path.empty()) + extra_paths.emplace_back(std::move(file_path)); + } /// ConfigReloader have to strict parameters which are redundant in our case auto main_config_reloader = std::make_unique( diff --git a/programs/server/Server.cpp b/programs/server/Server.cpp index ad78a0d204a2..8877da89122b 100644 --- a/programs/server/Server.cpp +++ b/programs/server/Server.cpp @@ -487,6 +487,10 @@ namespace ServerSetting extern const ServerSettingsString logger_shutdown_level; extern const ServerSettingsString openssl_server_certificate_file; extern const ServerSettingsString openssl_server_private_key_file; + extern const ServerSettingsString openssl_server_ca_config; + extern const ServerSettingsString openssl_client_certificate_file; + extern const ServerSettingsString openssl_client_private_key_file; + extern const ServerSettingsString openssl_client_ca_config; extern const ServerSettingsString distributed_ddl_path; extern const ServerSettingsString distributed_ddl_replicas_path; extern const ServerSettingsInt32 distributed_ddl_pool_size; @@ -2668,26 +2672,25 @@ try tryLogCurrentException(log, "Disabling cgroup memory observer because of an error during initialization"); } - std::string cert_path = server_settings[ServerSetting::openssl_server_certificate_file]; - std::string key_path = server_settings[ServerSetting::openssl_server_private_key_file]; - + /// TLS certificates, keys and CA certificates are reloaded by CertificateReloader when these files change. std::vector extra_paths = {include_from_path}; - if (!cert_path.empty()) - extra_paths.emplace_back(cert_path); - if (!key_path.empty()) - extra_paths.emplace_back(key_path); + auto watch_path = [&](const std::string & file_path) + { + if (!file_path.empty()) + extra_paths.emplace_back(file_path); + }; + watch_path(server_settings[ServerSetting::openssl_server_certificate_file]); + watch_path(server_settings[ServerSetting::openssl_server_private_key_file]); + watch_path(server_settings[ServerSetting::openssl_server_ca_config]); + watch_path(server_settings[ServerSetting::openssl_client_certificate_file]); + watch_path(server_settings[ServerSetting::openssl_client_private_key_file]); + watch_path(server_settings[ServerSetting::openssl_client_ca_config]); Poco::Util::AbstractConfiguration::Keys protocols; config().keys("protocols", protocols); for (const auto & protocol : protocols) - { - cert_path = config().getString("protocols." + protocol + ".certificateFile", ""); - key_path = config().getString("protocols." + protocol + ".privateKeyFile", ""); - if (!cert_path.empty()) - extra_paths.emplace_back(cert_path); - if (!key_path.empty()) - extra_paths.emplace_back(key_path); - } + for (const auto * key : {"certificateFile", "privateKeyFile", "caConfig"}) + watch_path(config().getString("protocols." + protocol + "." + key, "")); DNSResolver::instance().setFilterSettings(server_settings[ServerSetting::dns_allow_resolve_names_to_ipv4], server_settings[ServerSetting::dns_allow_resolve_names_to_ipv6]); /// DNSCacheUpdater uses BackgroundSchedulePool which lives in shared context diff --git a/src/Common/Config/ConfigReloader.cpp b/src/Common/Config/ConfigReloader.cpp index dd9531c6493d..1dc2fd0ab747 100644 --- a/src/Common/Config/ConfigReloader.cpp +++ b/src/Common/Config/ConfigReloader.cpp @@ -222,8 +222,17 @@ struct ConfigReloader::FileWithTimestamp void ConfigReloader::FilesChangesTracker::addIfExists(const std::string & path_to_add) { - if (!path_to_add.empty() && fs::exists(path_to_add)) - files.emplace(path_to_add); + if (path_to_add.empty() || !fs::exists(path_to_add)) + return; + + files.emplace(path_to_add); + + /// E.g. a directory with CA certificates, a change of a file in it should be noticed too. + std::error_code ec; + if (fs::is_directory(path_to_add, ec)) + for (const auto & entry : fs::directory_iterator(path_to_add, ec)) + if (entry.is_regular_file(ec)) + files.emplace(entry.path().string()); } bool ConfigReloader::FilesChangesTracker::isDifferOrNewerThan(const FilesChangesTracker & rhs) diff --git a/src/Coordination/KeeperServer.cpp b/src/Coordination/KeeperServer.cpp index c1d19bbc92bd..b453b94dea7c 100644 --- a/src/Coordination/KeeperServer.cpp +++ b/src/Coordination/KeeperServer.cpp @@ -148,7 +148,9 @@ auto getSslContextProvider(const Poco::Util::AbstractConfiguration & config, std if (config.has(root_ca_file_property)) params.caLocation = config.getString(root_ca_file_property); - params.loadDefaultCAs = config.getBool(load_default_ca_file_property, false); + /// Unlike `Poco::Net::SSLManager`, the default CA certificates are not trusted unless `loadDefaultCAFile` is set. + constexpr bool load_default_cas_default = false; + params.loadDefaultCAs = config.getBool(load_default_ca_file_property, load_default_cas_default); params.verificationMode = Poco::Net::Utility::convertVerificationMode(config.getString(verification_mode_property, "none")); const String cipher_list_property = config_prefix + "cipherList"; @@ -194,7 +196,7 @@ auto getSslContextProvider(const Poco::Util::AbstractConfiguration & config, std /// Try to register with CertificateReloader for hot-reload support. /// If registration fails, fall back to static certificate loading. - if (!CertificateReloader::instance().registerAdditionalContext(ssl_ctx, config_prefix)) + if (!CertificateReloader::instance().registerAdditionalContext(ssl_ctx, config_prefix, load_default_cas_default)) { /// For passphrase-protected keys, load certificates manually if (certificate_data) diff --git a/src/Core/ServerSettings.cpp b/src/Core/ServerSettings.cpp index 0f94a06ba3bc..209c0521f77b 100644 --- a/src/Core/ServerSettings.cpp +++ b/src/Core/ServerSettings.cpp @@ -1886,7 +1886,7 @@ Configured as `named_collections_storage.type` (`` contains the certificate.)", 0, "openSSL.server.certificateFile") \ - DECLARE(String, openssl_server_ca_config, "", R"(Path to the file or directory that contains trusted CA certificates. If this points to a file, it must be in PEM format and can contain several CA certificates. If this points to a directory, it must contain one .pem file per CA certificate. The filenames are looked up by the CA subject name hash value. Details can be found in the man page of [SSL_CTX_load_verify_locations](https://docs.openssl.org/3.0/man3/SSL_CTX_load_verify_locations/).)", 0, "openSSL.server.caConfig") \ + DECLARE(String, openssl_server_ca_config, "", R"(Path to the file or directory that contains trusted CA certificates. If this points to a file, it must be in PEM format and can contain several CA certificates. If this points to a directory, it must contain one .pem file per CA certificate. The filenames are looked up by the CA subject name hash value. Details can be found in the man page of [SSL_CTX_load_verify_locations](https://docs.openssl.org/3.0/man3/SSL_CTX_load_verify_locations/). The CA certificates are reloaded without a restart when the file changes or on `SYSTEM RELOAD CONFIG`; new connections are verified against the reloaded certificates.)", 0, "openSSL.server.caConfig") \ DECLARE(String, openssl_server_verification_mode, "relaxed", R"(The method for checking the node's certificates. Details are in the description of the [Context](https://github.com/ClickHouse/poco/blob/master/NetSSL_OpenSSL/include/Poco/Net/Context.h) class. Possible values: ``, ``, ``, ``.)", 0, "openSSL.server.verificationMode") \ DECLARE(UInt64, openssl_server_verification_depth, 9, R"(The maximum length of the verification chain. Verification will fail if the certificate chain length exceeds the set value.)", 0, "openSSL.server.verificationDepth") \ DECLARE(Bool, openssl_server_load_default_ca_file, true, R"(Determines whether built-in CA certificates for OpenSSL will be used. ClickHouse assumes that builtin CA certificates are in the file `` (resp. the directory ``) or in file (resp. directory) specified by the environment variable `` (resp. ``).)", 0, "openSSL.server.loadDefaultCAFile") \ @@ -1906,7 +1906,7 @@ Configured as `named_collections_storage.type` (`` contains the certificate.)", 0, "openSSL.client.certificateFile") \ - DECLARE(String, openssl_client_ca_config, "", R"(Path to the file or directory that contains trusted CA certificates. If this points to a file, it must be in PEM format and can contain several CA certificates. If this points to a directory, it must contain one .pem file per CA certificate. The filenames are looked up by the CA subject name hash value. Details can be found in the man page of [SSL_CTX_load_verify_locations](https://docs.openssl.org/3.0/man3/SSL_CTX_load_verify_locations/).)", 0, "openSSL.client.caConfig") \ + DECLARE(String, openssl_client_ca_config, "", R"(Path to the file or directory that contains trusted CA certificates. If this points to a file, it must be in PEM format and can contain several CA certificates. If this points to a directory, it must contain one .pem file per CA certificate. The filenames are looked up by the CA subject name hash value. Details can be found in the man page of [SSL_CTX_load_verify_locations](https://docs.openssl.org/3.0/man3/SSL_CTX_load_verify_locations/). The CA certificates are reloaded without a restart when the file changes or on `SYSTEM RELOAD CONFIG`; new connections are verified against the reloaded certificates.)", 0, "openSSL.client.caConfig") \ DECLARE(String, openssl_client_verification_mode, "relaxed", R"(The method for checking the node's certificates. Details are in the description of the [Context](https://github.com/ClickHouse/poco/blob/master/NetSSL_OpenSSL/include/Poco/Net/Context.h) class. Possible values: ``, ``, ``, ``.)", 0, "openSSL.client.verificationMode") \ DECLARE(UInt64, openssl_client_verification_depth, 9, R"(The maximum length of the verification chain. Verification will fail if the certificate chain length exceeds the set value.)", 0, "openSSL.client.verificationDepth") \ DECLARE(Bool, openssl_client_load_default_ca_file, true, R"(Determines whether built-in CA certificates for OpenSSL will be used. ClickHouse assumes that builtin CA certificates are in the file `` (resp. the directory ``) or in file (resp. directory) specified by the environment variable `` (resp. ``).)", 0, "openSSL.client.loadDefaultCAFile") \ diff --git a/src/Server/CertificateReloader.cpp b/src/Server/CertificateReloader.cpp index 787c4e1fba8a..543508d2c9e4 100644 --- a/src/Server/CertificateReloader.cpp +++ b/src/Server/CertificateReloader.cpp @@ -3,6 +3,7 @@ #if USE_SSL #include +#include #include #include #include @@ -23,6 +24,7 @@ CertificateReloader & CertificateReloader::instance() namespace ErrorCodes { extern const int INVALID_CONFIG_PARAMETER; + extern const int OPENSSL_ERROR; } namespace @@ -38,6 +40,91 @@ int callSetCertificate(SSL * ssl, void * arg) return CertificateReloader::instance().setCertificate(ssl, pdata); } +/// Called by OpenSSL instead of `X509_verify_cert` to verify the peer's certificate. +int callVerifyCertificate(X509_STORE_CTX * store_ctx, void * arg) +{ + const MultiVersion * ca_store = reinterpret_cast *>(arg); + return CertificateReloader::instance().verifyCertificate(store_ctx, ca_store); +} + +int defaultVerifyCertificate(X509_STORE_CTX * store_ctx) +{ + /// Same as libssl does when no callback is installed: an error is treated as a verification failure. + int ok = X509_verify_cert(store_ctx); + return ok < 0 ? 0 : ok; +} + +/// Verify the certificate from `original_ctx` like `X509_verify_cert(original_ctx)` would, but against the trusted certificates in `store`. +/// libssl prepares `original_ctx` in `ssl_verify_internal` (ssl/ssl_cert.c) before calling the application callback. +/// The same preparation is carried over to a new `X509_STORE_CTX` that uses `store`, and the outcome is reported back +/// through `original_ctx`, because that is what libssl looks at after the callback returns. +int verifyCertificateWithStore(X509_STORE_CTX * original_ctx, X509 * certificate, X509_STORE * store) +{ + std::unique_ptr verify_ctx(X509_STORE_CTX_new(), X509_STORE_CTX_free); + if (!verify_ctx) + { + X509_STORE_CTX_set_error(original_ctx, X509_V_ERR_OUT_OF_MEM); + return 0; + } + + if (X509_STORE_CTX_init(verify_ctx.get(), store, certificate, X509_STORE_CTX_get0_untrusted(original_ctx)) != 1) + { + X509_STORE_CTX_set_error(original_ctx, X509_V_ERR_UNSPECIFIED); + return 0; + } + + /// The connection, for the per-connection verification callbacks (e.g. the ones of Poco and boost::asio) that look it up. + int ssl_idx = SSL_get_ex_data_X509_STORE_CTX_idx(); + SSL * ssl = static_cast(X509_STORE_CTX_get_ex_data(original_ctx, ssl_idx)); + if (ssl) + { + if (X509_STORE_CTX_set_ex_data(verify_ctx.get(), ssl_idx, ssl) != 1) + { + X509_STORE_CTX_set_error(original_ctx, X509_V_ERR_UNSPECIFIED); + return 0; + } + /// Has an effect only if DANE is enabled for the connection. + X509_STORE_CTX_set0_dane(verify_ctx.get(), SSL_get0_dane(ssl)); + } + + /// The default purpose depends on which side is verified. Everything libssl derived from the connection and its `SSL_CTX` + /// (verification depth, expected host name, security level, flags, ...) is already in the parameters of `original_ctx`. + X509_STORE_CTX_set_default(verify_ctx.get(), (ssl && SSL_is_server(ssl)) ? "ssl_client" : "ssl_server"); + if (X509_VERIFY_PARAM_set1(X509_STORE_CTX_get0_param(verify_ctx.get()), X509_STORE_CTX_get0_param(original_ctx)) != 1) + { + X509_STORE_CTX_set_error(original_ctx, X509_V_ERR_UNSPECIFIED); + return 0; + } + X509_STORE_CTX_set_verify_cb(verify_ctx.get(), X509_STORE_CTX_get_verify_cb(original_ctx)); + + int ok = defaultVerifyCertificate(verify_ctx.get()); + + X509_STORE_CTX_set_error(original_ctx, X509_STORE_CTX_get_error(verify_ctx.get())); + X509_STORE_CTX_set_error_depth(original_ctx, X509_STORE_CTX_get_error_depth(verify_ctx.get())); + if (STACK_OF(X509) * chain = X509_STORE_CTX_get1_chain(verify_ctx.get())) + X509_STORE_CTX_set0_verified_chain(original_ctx, chain); + X509_VERIFY_PARAM_move_peername(X509_STORE_CTX_get0_param(original_ctx), X509_STORE_CTX_get0_param(verify_ctx.get())); + + return ok; +} + +/// Load the trusted CA certificates the same way `Poco::Net::Context` does it for the contexts created at startup, +/// so that a reload results in the same set of trusted certificates as a restart would. +std::unique_ptr loadCAStore(const std::string & ca_path, bool load_default_cas) +{ + Poco::Net::Context::Params params; + params.caLocation = ca_path; + params.loadDefaultCAs = load_default_cas; + /// Only the certificate store is taken from this context, so the usage does not matter. + Poco::Net::Context context(Poco::Net::Context::CLIENT_USE, params); + + X509_STORE * store = SSL_CTX_get_cert_store(context.sslContext()); + if (!store || X509_STORE_up_ref(store) != 1) + throw Exception(ErrorCodes::OPENSSL_ERROR, "Cannot get CA certificates from SSL context: {}", Poco::Net::Utility::getLastError()); + + return std::make_unique(store, context.getCAPaths()); +} + } /// This is callback for OpenSSL. It will be called on every connection to obtain a certificate and private key. @@ -50,6 +137,28 @@ int CertificateReloader::setCertificate(SSL * ssl, const CertificateReloader::Mu return setCertificateCallback(ssl, current.get(), log); } +/// This is callback for OpenSSL. It will be called on every connection that verifies the certificate of the peer. +int CertificateReloader::verifyCertificate(X509_STORE_CTX * store_ctx, const MultiVersion * ca_store) const +{ + try + { + auto current = ca_store->get(); + X509 * certificate = X509_STORE_CTX_get0_cert(store_ctx); + + /// Raw public keys (RFC 7250) are verified without CA certificates. + if (!current || !certificate) + return defaultVerifyCertificate(store_ctx); + + return verifyCertificateWithStore(store_ctx, certificate, current->store); + } + catch (...) + { + LOG_ERROR(log, getCurrentExceptionMessageAndPattern(/* with_stacktrace */ false)); + X509_STORE_CTX_set_error(store_ctx, X509_V_ERR_UNSPECIFIED); + return 0; + } +} + int setCertificateCallback(SSL * ssl, const CertificateReloader::Data * current_data, LoggerPtr log) { if (current_data->certs_chain.empty()) @@ -142,6 +251,10 @@ std::list::iterator CertificateReloader::findOrI data.push_back(MultiData(ctx)); --it; data_index[prefix] = it; + + /// Verify peer certificates against the reloadable CA certificates of this prefix. + /// Until (and unless) they are loaded, the callback does exactly what OpenSSL does without it. + SSL_CTX_set_cert_verify_callback(ctx, callVerifyCertificate, reinterpret_cast(&it->ca.store)); } return it; } @@ -179,8 +292,51 @@ void CertificateReloader::tryLoadACMECertificate(SSL_CTX * ctx, const std::strin } } +void CertificateReloader::tryLoadCAImpl(const Poco::Util::AbstractConfiguration & config, SSL_CTX * ctx, const std::string & prefix) +{ + std::string new_ca_path = config.getString(prefix + Poco::Net::SSLManager::CFG_CA_LOCATION, ""); + + /// Without `caConfig` the trusted certificates come only from the system locations (if at all), there is nothing to reload. + /// But if `caConfig` was there before, keep following the configuration, like a restart would. + if (new_ca_path.empty()) + { + auto index_it = data_index.find(prefix); + if (index_it == data_index.end() || index_it->second->ca_file.path.empty()) + return; + } + + try + { + auto it = findOrInsert(ctx, prefix); + bool ca_file_changed = it->ca_file.changeIfModified(std::move(new_ca_path), log); + + auto load = [&](CAData & ca, bool load_default_cas_default) + { + bool new_load_default_cas = config.getBool(prefix + Poco::Net::SSLManager::CFG_ENABLE_DEFAULT_CA, load_default_cas_default); + if (!ca_file_changed && ca.store.get() && new_load_default_cas == ca.load_default_cas) + return; + + LOG_DEBUG(log, "Reloading CA certificates ({}), load default CAs: {}.", it->ca_file.path, new_load_default_cas); + ca.store.set(loadCAStore(it->ca_file.path, new_load_default_cas)); + ca.load_default_cas = new_load_default_cas; + LOG_INFO(log, "Reloaded CA certificates ({}), load default CAs: {}.", it->ca_file.path, new_load_default_cas); + }; + + load(it->ca, Poco::Net::SSLManager::VAL_ENABLE_DEFAULT_CA); + if (it->ca_with_other_default) + load(*it->ca_with_other_default, it->other_load_default_cas_default); + } + catch (...) + { + LOG_ERROR(log, getCurrentExceptionMessageAndPattern(/* with_stacktrace */ false)); + } +} + void CertificateReloader::tryLoadImpl(const Poco::Util::AbstractConfiguration & config, SSL_CTX * ctx, const std::string & prefix) { + /// Trusted CA certificates do not depend on how the own certificate is configured. + tryLoadCAImpl(config, ctx, prefix); + /// If at least one of the files is modified - recreate std::string new_cert_path = config.getString(prefix + "certificateFile", ""); std::string new_key_path = config.getString(prefix + "privateKeyFile", ""); @@ -239,7 +395,7 @@ void CertificateReloader::tryReloadAll(const Poco::Util::AbstractConfiguration & } -bool CertificateReloader::registerAdditionalContext(SSL_CTX * ctx, const std::string & prefix) +bool CertificateReloader::registerAdditionalContext(SSL_CTX * ctx, const std::string & prefix, bool load_default_cas_default) { if (!ctx) return false; @@ -256,9 +412,24 @@ bool CertificateReloader::registerAdditionalContext(SSL_CTX * ctx, const std::st MultiData * pdata = &*(it->second); + /// Share the reloadable CA certificates of this prefix (see `findOrInsert`). A context that assumes another default + /// for `loadDefaultCAFile` may trust other CA certificates than the primary one, so it gets its own ones. + /// They are loaded by the next `tryLoad`, until then the context keeps using the store it was created with. + CAData * ca = &pdata->ca; + if (load_default_cas_default != Poco::Net::SSLManager::VAL_ENABLE_DEFAULT_CA) + { + if (!pdata->ca_with_other_default) + { + pdata->ca_with_other_default.emplace(); + pdata->other_load_default_cas_default = load_default_cas_default; + } + ca = &*pdata->ca_with_other_default; + } + SSL_CTX_set_cert_verify_callback(ctx, callVerifyCertificate, reinterpret_cast(&ca->store)); + /// Verify that certificate data was actually loaded, not just the entry created. /// If data is null, return false so caller can use fallback (static cert loading). - /// This can happen if initial cert parsing failed in tryLoadImpl. + /// This can happen if initial cert parsing failed in tryLoadImpl or if only `caConfig` is set for the prefix. if (!pdata->data.get()) { LOG_WARNING(log, "Cannot register additional context for prefix '{}': certificate data not loaded. " @@ -292,6 +463,22 @@ std::optional CertificateReloader::getCertificate(const std::st } +std::optional CertificateReloader::getCAPaths(const std::string & prefix) const +{ + std::lock_guard lock{data_mutex}; + + auto it = data_index.find(prefix); + if (it == data_index.end()) + return {}; + + auto current = it->second->ca.store.get(); + if (!current) + return {}; + + return current->paths; +} + + CertificateReloader::Data::Data(std::string cert_path, std::string key_path, std::string pass_phrase) : certs_chain(X509Certificate::fromFile(cert_path)), key(KeyPair::fromFile(key_path, pass_phrase)) { @@ -305,6 +492,14 @@ CertificateReloader::Data::Data(KeyPair _pkey, X509Certificate::List _certs_chai bool CertificateReloader::File::changeIfModified(std::string new_path, LoggerPtr logger) { + if (new_path.empty()) + { + bool changed = !path.empty(); + path.clear(); + modification_time = {}; + return changed; + } + std::error_code ec; std::filesystem::file_time_type new_modification_time = std::filesystem::last_write_time(new_path, ec); if (ec) @@ -318,10 +513,24 @@ bool CertificateReloader::File::changeIfModified(std::string new_path, LoggerPtr return false; } - if (new_path != path || new_modification_time != modification_time) + /// `caConfig` can be a directory with certificates, replacing one of them is a change too. + UInt64 new_directory_contents_hash = 0; + if (std::filesystem::is_directory(new_path, ec)) + { + SipHash hash; + for (const auto & entry : std::filesystem::directory_iterator(new_path, ec)) + { + hash.update(entry.path().filename().string()); + hash.update(std::filesystem::last_write_time(entry.path(), ec).time_since_epoch().count()); + } + new_directory_contents_hash = hash.get64(); + } + + if (new_path != path || new_modification_time != modification_time || new_directory_contents_hash != directory_contents_hash) { path = new_path; modification_time = new_modification_time; + directory_contents_hash = new_directory_contents_hash; return true; } diff --git a/src/Server/CertificateReloader.h b/src/Server/CertificateReloader.h index 44330a1f7007..735b40392d43 100644 --- a/src/Server/CertificateReloader.h +++ b/src/Server/CertificateReloader.h @@ -11,6 +11,7 @@ #include #include +#include #include #include #include @@ -27,17 +28,34 @@ namespace DB { -/// The CertificateReloader singleton performs 2 functions: -/// 1. Dynamic reloading of TLS key-pair when requested by server: +/// The CertificateReloader singleton performs 3 functions: +/// 1. Dynamic reloading of TLS key-pair and of the trusted CA certificates (`caConfig`) when requested by server: /// Server config reloader notifies CertificateReloader when the config changes. /// On changed config, CertificateReloader reloads certs from disk. /// 2. Implement `SSL_CTX_set_cert_cb` to set certificate for a new connection: /// OpenSSL invokes a callback to setup a connection. +/// 3. Implement `SSL_CTX_set_cert_verify_callback` to verify the peer's certificate of a new connection +/// against the most recently loaded CA certificates. +/// +/// An `SSL_CTX` that is shared between threads must not be modified, so instead of touching the contexts on reload, +/// both callbacks apply the current immutable snapshot (`MultiVersion`) to each new connection. class CertificateReloader { public: using stat_t = struct stat; + /// Owns a reference to a set of trusted CA certificates and remembers where they were loaded from. + struct CAStore + { + CAStore(X509_STORE * store_, Poco::Net::Context::CAPaths paths_) : store(store_), paths(std::move(paths_)) {} + CAStore(const CAStore &) = delete; + CAStore & operator=(const CAStore &) = delete; + ~CAStore() { X509_STORE_free(store); } + + X509_STORE * const store; + const Poco::Net::Context::CAPaths paths; + }; + struct Data { X509Certificate::List certs_chain; @@ -56,10 +74,19 @@ class CertificateReloader std::string path; std::filesystem::file_time_type modification_time; + /// For a directory: the names and modification times of the files in it. + UInt64 directory_contents_hash = 0; bool changeIfModified(std::string new_path, LoggerPtr logger); }; + /// Trusted CA certificates from `caConfig` and, if `load_default_cas`, the default ones. + struct CAData + { + MultiVersion store; + bool load_default_cas = false; + }; + struct MultiData { SSL_CTX * ctx = nullptr; @@ -69,6 +96,14 @@ class CertificateReloader File cert_file{"certificate"}; File key_file{"key"}; + /// Empty if `caConfig` is not set for the prefix, then verification keeps using the store the context was created with. + CAData ca; + /// For additional contexts that assume another `loadDefaultCAFile` than Poco when it is not configured (the ones of Keeper). + std::optional ca_with_other_default; + bool other_load_default_cas_default = false; + + File ca_file{"CA"}; + explicit MultiData(SSL_CTX * ctx_) : ctx(ctx_) {} }; @@ -88,8 +123,11 @@ class CertificateReloader /// Handle configuration reload void tryLoad(const Poco::Util::AbstractConfiguration & config, SSL_CTX * ctx, const std::string & prefix); - /// Register an additional SSL_CTX to share certificates with the primary context - bool registerAdditionalContext(SSL_CTX * ctx, const std::string & prefix); + /// Register an additional SSL_CTX to share certificates and trusted CAs with the primary context of `prefix`. + /// `load_default_cas_default` is what the caller assumes for `loadDefaultCAFile` when it is not configured. + /// Returns true if the context will get its certificate and key from CertificateReloader, + /// false if the caller has to configure them on the context itself. + bool registerAdditionalContext(SSL_CTX * ctx, const std::string & prefix, bool load_default_cas_default); /// Handle configuration reload for all contexts void tryReloadAll(const Poco::Util::AbstractConfiguration & config); @@ -97,11 +135,18 @@ class CertificateReloader /// A callback for OpenSSL int setCertificate(SSL * ssl, const MultiData * pdata); + /// A callback for OpenSSL: verify the peer certificate in `store_ctx` against the current CA certificates in `ca_store`. + int verifyCertificate(X509_STORE_CTX * store_ctx, const MultiVersion * ca_store) const; + /// The leaf certificate that is currently served for `prefix` connections, if there is one. /// It is not necessarily the certificate of the corresponding `SSL_CTX`: certificates are installed /// per connection, and with `` the context itself never gets a certificate at all. std::optional getCertificate(const std::string & prefix) const; + /// Where the CA certificates that are currently used to verify peers of `prefix` connections were loaded from, + /// if they are managed by CertificateReloader (i.e. `caConfig` is set for the prefix). + std::optional getCAPaths(const std::string & prefix) const; + private: CertificateReloader() = default; @@ -111,6 +156,7 @@ class CertificateReloader /// Unsafe implementation void tryLoadImpl(const Poco::Util::AbstractConfiguration & config, SSL_CTX * ctx, const std::string & prefix) TSA_REQUIRES(data_mutex); void tryLoadACMECertificate(SSL_CTX * ctx, const std::string & prefix) TSA_REQUIRES(data_mutex); + void tryLoadCAImpl(const Poco::Util::AbstractConfiguration & config, SSL_CTX * ctx, const std::string & prefix) TSA_REQUIRES(data_mutex); std::list::iterator findOrInsert(SSL_CTX * ctx, const std::string & prefix) TSA_REQUIRES(data_mutex); diff --git a/src/Storages/System/StorageSystemCertificates.cpp b/src/Storages/System/StorageSystemCertificates.cpp index 3b8acceea62a..2ae44bb8a5f3 100644 --- a/src/Storages/System/StorageSystemCertificates.cpp +++ b/src/Storages/System/StorageSystemCertificates.cpp @@ -15,6 +15,7 @@ #include #include #include + #include #endif #include @@ -140,12 +141,19 @@ void StorageSystemCertificates::fillData([[maybe_unused]] MutableColumns & res_c } }; + /// The CA certificates may have been reloaded since the context was created, `CertificateReloader` knows the current ones. + auto current_ca_paths = [](const std::string & prefix, Poco::Net::Context::Ptr ssl_context) + { + if (auto reloaded_ca_paths = CertificateReloader::instance().getCAPaths(prefix)) + return *reloaded_ca_paths; + return ssl_context->getCAPaths(); + }; + const auto & config = Context::getGlobalContextInstance()->getConfigRef(); try { - const auto & ca_paths = Poco::Net::SSLManager::instance().defaultServerContext()->getCAPaths(); - process_ca_paths(ca_paths, ""); + process_ca_paths(current_ca_paths(Poco::Net::SSLManager::CFG_SERVER_PREFIX, Poco::Net::SSLManager::instance().defaultServerContext()), ""); } catch (const Poco::Net::SSLException &) { @@ -162,10 +170,7 @@ void StorageSystemCertificates::fillData([[maybe_unused]] MutableColumns & res_c continue; if (auto ctx = Poco::Net::SSLManager::instance().getCustomServerContext(prefix)) - { - const auto & ca_paths = ctx->getCAPaths(); - process_ca_paths(ca_paths, protocol_name); - } + process_ca_paths(current_ca_paths(prefix, ctx), protocol_name); } #endif } diff --git a/tests/integration/test_reload_ca_certificate/__init__.py b/tests/integration/test_reload_ca_certificate/__init__.py new file mode 100644 index 000000000000..e69de29bb2d1 diff --git a/tests/integration/test_reload_ca_certificate/certs/ca.crt b/tests/integration/test_reload_ca_certificate/certs/ca.crt new file mode 100644 index 000000000000..b836c066c43e --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/ca.crt @@ -0,0 +1,20 @@ +-----BEGIN CERTIFICATE----- +MIIDRzCCAi+gAwIBAgIURxZvEwx6gcq/IbNHsPts6W5c7P0wDQYJKoZIhvcNAQEL +BQAwMzEYMBYGA1UECgwPQ2xpY2tIb3VzZSBUZXN0MRcwFQYDVQQDDA5UZXN0IFJv +b3QgQ0EgMTAeFw0yNjA4MzExNzI4MTRaFw0zNjA4MjgxNzI4MTRaMDMxGDAWBgNV +BAoMD0NsaWNrSG91c2UgVGVzdDEXMBUGA1UEAwwOVGVzdCBSb290IENBIDEwggEi +MA0GCSqGSIb3DQEBAQUAA4IBDwAwggEKAoIBAQCQMESmjuEmo6A25FY5Bt11iOrs +ySvgaWC7DkFaRRy/EZOvQg/HiGXG/oRtt1BsKLYdxRrvtxgJ9H5Uei/SsY0kLTiz +PZFsEomesNwxV1Y8dyeKf9EDSKCjCnrlJwaTmpJ2FpjJRQHf5YJtwVf9E13eKGw7 +qGoCxGGUx14kU7ftdfpEtCC8HRjktWlkXt16nCAHInXx8WSi6SAgLoLUeVC/HpNQ +dVWyPHUhYyFppkYnYONIAJa2UaXB4OlaE+d9/edJeKWGc46oIFK4bJV53nOJ+1Jw +IEf13ax0IcKc1I0q1hJrsnB68Wowi6NR33THHbPNCvU4h/+1qcTYdzicaA9JAgMB +AAGjUzBRMB0GA1UdDgQWBBSJsd7zDPOJkBExjUxJ8bKVSg2pHTAfBgNVHSMEGDAW +gBSJsd7zDPOJkBExjUxJ8bKVSg2pHTAPBgNVHRMBAf8EBTADAQH/MA0GCSqGSIb3 +DQEBCwUAA4IBAQBiopVu+5LqgH4FwKh6YLvWEWdtw5gWs1Eda27cN2b1nTnJ+Xfn +UlwAdaS1lzGBkgUDYfbL0OJo9wx/yg5MuakE2kNO1QALb4r5mpGqTBMYT1XVzcGx +6LCOweHBI4a/ff4s1rj1xRR+PIPIHnCd5mKEgscKMvCmR5kUzkkmjRPFnBJFLHH9 +Z88L4J5NBwF8wJsFCv9td8aeeeTc9rg19zKOn6vkR5AczeXVAeBDBxAtcmX2YRMt +ytIWGmKfW3OE8kxsZdJJn4b+9jE5pLbTEL/So0YuNELRgcpK+9MGApOKjcc3DJg0 +LhsYev318QV4SjluwR1T/jgRvAlFbwk8+10L +-----END CERTIFICATE----- diff --git a/tests/integration/test_reload_ca_certificate/certs/ca1.crt b/tests/integration/test_reload_ca_certificate/certs/ca1.crt new file mode 100644 index 000000000000..b836c066c43e --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/ca1.crt @@ -0,0 +1,20 @@ +-----BEGIN CERTIFICATE----- +MIIDRzCCAi+gAwIBAgIURxZvEwx6gcq/IbNHsPts6W5c7P0wDQYJKoZIhvcNAQEL +BQAwMzEYMBYGA1UECgwPQ2xpY2tIb3VzZSBUZXN0MRcwFQYDVQQDDA5UZXN0IFJv +b3QgQ0EgMTAeFw0yNjA4MzExNzI4MTRaFw0zNjA4MjgxNzI4MTRaMDMxGDAWBgNV +BAoMD0NsaWNrSG91c2UgVGVzdDEXMBUGA1UEAwwOVGVzdCBSb290IENBIDEwggEi +MA0GCSqGSIb3DQEBAQUAA4IBDwAwggEKAoIBAQCQMESmjuEmo6A25FY5Bt11iOrs +ySvgaWC7DkFaRRy/EZOvQg/HiGXG/oRtt1BsKLYdxRrvtxgJ9H5Uei/SsY0kLTiz +PZFsEomesNwxV1Y8dyeKf9EDSKCjCnrlJwaTmpJ2FpjJRQHf5YJtwVf9E13eKGw7 +qGoCxGGUx14kU7ftdfpEtCC8HRjktWlkXt16nCAHInXx8WSi6SAgLoLUeVC/HpNQ +dVWyPHUhYyFppkYnYONIAJa2UaXB4OlaE+d9/edJeKWGc46oIFK4bJV53nOJ+1Jw +IEf13ax0IcKc1I0q1hJrsnB68Wowi6NR33THHbPNCvU4h/+1qcTYdzicaA9JAgMB +AAGjUzBRMB0GA1UdDgQWBBSJsd7zDPOJkBExjUxJ8bKVSg2pHTAfBgNVHSMEGDAW +gBSJsd7zDPOJkBExjUxJ8bKVSg2pHTAPBgNVHRMBAf8EBTADAQH/MA0GCSqGSIb3 +DQEBCwUAA4IBAQBiopVu+5LqgH4FwKh6YLvWEWdtw5gWs1Eda27cN2b1nTnJ+Xfn +UlwAdaS1lzGBkgUDYfbL0OJo9wx/yg5MuakE2kNO1QALb4r5mpGqTBMYT1XVzcGx +6LCOweHBI4a/ff4s1rj1xRR+PIPIHnCd5mKEgscKMvCmR5kUzkkmjRPFnBJFLHH9 +Z88L4J5NBwF8wJsFCv9td8aeeeTc9rg19zKOn6vkR5AczeXVAeBDBxAtcmX2YRMt +ytIWGmKfW3OE8kxsZdJJn4b+9jE5pLbTEL/So0YuNELRgcpK+9MGApOKjcc3DJg0 +LhsYev318QV4SjluwR1T/jgRvAlFbwk8+10L +-----END CERTIFICATE----- diff --git a/tests/integration/test_reload_ca_certificate/certs/ca1.key b/tests/integration/test_reload_ca_certificate/certs/ca1.key new file mode 100644 index 000000000000..562b4debf1ac --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/ca1.key @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCQMESmjuEmo6A2 +5FY5Bt11iOrsySvgaWC7DkFaRRy/EZOvQg/HiGXG/oRtt1BsKLYdxRrvtxgJ9H5U +ei/SsY0kLTizPZFsEomesNwxV1Y8dyeKf9EDSKCjCnrlJwaTmpJ2FpjJRQHf5YJt +wVf9E13eKGw7qGoCxGGUx14kU7ftdfpEtCC8HRjktWlkXt16nCAHInXx8WSi6SAg +LoLUeVC/HpNQdVWyPHUhYyFppkYnYONIAJa2UaXB4OlaE+d9/edJeKWGc46oIFK4 +bJV53nOJ+1JwIEf13ax0IcKc1I0q1hJrsnB68Wowi6NR33THHbPNCvU4h/+1qcTY +dzicaA9JAgMBAAECggEABbduL1gkknXkBcJw2IRec8OaBHQs2cdfwwbGM0b1xEL/ +2Dp9+j7ngcJTO9or1XVElCKoMNxmgaAzK9RSJfqtV3j2AaWKyOO6wk3YKwTKXy2+ +WvS6k7RgTwUnjm3TykwTa1xgZ2wQQ8EmC3CCCtScMPu+CnGgw1Y9GVXmqaQf5jQ1 +xAoxP8/POBLjco8PAFWtai+wK9X1OVxQ6gr5Tw3TwAIHrYGsTY19g5c5IM4Y2k34 +QoaSGHkmSFOrBZs2DaWCaJ08Iz3zWUBP0VV2/H0szdIIrkv/Tqw2K6peRbJB8mYN +TID/BpbAadbEllzfn31O3Si0urHNZMrJI5IS2IChEQKBgQDHXrc6AUQQArbEKCxB +dq6B9+tLukaiMctxLHNXhY2NYlJxtASxkmYJewmU/d16lMyo3S+IPkEueo0Q1u3Z +k/MotSiis5kkL2EJWJjHq4H/ktw6k2a0vRd4DtcoysJ2qkCcgubSGYqSE+YcI1KC +MyOoKfb2IffCucyYAzntKmh+2QKBgQC5JQH9+CdyPlcWTJsXMztgDwyhzBixSqIl +Ovfrzzi7QSIX9csccZoXWkOfDEUJp2CZImErASO06+RM2ETQ1n2art4eWxX3UubD +pLXfr1m8NL/LJb56k9sWT+YwLXm4JeHeaErK6xTQCDNxH6szXUXQdXUpnbrrjg0S +6xVG0tCt8QKBgBI+U4vmQ8EnTmwitPIElzFja0+Rqxb6cYBYrfFLUkmmvp6S9378 +Q4QIkzbkCBlIdnXZT5krATHsmu34jOlFBZIrCZ3hy1ipUTrWtZxH0Gx/ltFxXYua +ZgRhb0TXUPYk3Ca2P8Ln/WsikQLwJIOvhErGFEgvkYlrERKz8OAH6mn5AoGAXQxd +YO9bm8365KkhdNp5p8BIf/RcIJY6wW1OdkPR5kJIyTPtnWD2qW/i9kcrVzu4j524 +qe1Lrby0I265vx9dRuVFmon6ky8l7QOVqFKvTahRD97rSR2QCTmknWfteYAIcUeG +906ISjkk6WCaIRlqYeb2ODEeZQ4iQfTF369J03ECgYA3NPk8j5ma59w/TYUPgZWN +aFlEKWLO8xOal9kWhx5dpXi1YCv3YoQztrfYus5b2nj7iDshK5cK6TzZaxPahqxA +rW9ExDjN2q89AvqP9vKPDJZ8Rk7+vTlLSNkFSe9LvvtfnU6IHdNcj+t/CI2823GC +Myj/NocqikSkeRCcq4SGCA== +-----END PRIVATE KEY----- diff --git a/tests/integration/test_reload_ca_certificate/certs/ca2.crt b/tests/integration/test_reload_ca_certificate/certs/ca2.crt new file mode 100644 index 000000000000..eff5f7fd97bc --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/ca2.crt @@ -0,0 +1,20 @@ +-----BEGIN CERTIFICATE----- +MIIDRzCCAi+gAwIBAgIUdDLEtjezpK/A2pIn/0YjbO3JhyAwDQYJKoZIhvcNAQEL +BQAwMzEYMBYGA1UECgwPQ2xpY2tIb3VzZSBUZXN0MRcwFQYDVQQDDA5UZXN0IFJv +b3QgQ0EgMjAeFw0yNjA4MzExNzI4MTVaFw0zNjA4MjgxNzI4MTVaMDMxGDAWBgNV +BAoMD0NsaWNrSG91c2UgVGVzdDEXMBUGA1UEAwwOVGVzdCBSb290IENBIDIwggEi +MA0GCSqGSIb3DQEBAQUAA4IBDwAwggEKAoIBAQC1bXrQUqqds9kXelFc7J+IrXsE +UdFZtxktFglh9eFrffqr0/+vthrcnqZ/OZrLmZV5PSW+gSNuaAVArbgGwYWW1lqL +0eRfQOktX/NcjoYlWUYlK7lWReFQj2CTCUrP3ojrdo/Oe0/0LsV3kctfUagcNxMX +oV5NAwHd9dDf9aL1ulljc+RO4ym8BlRqSApShkyH3xklkDLBneHOUJncwrWcC12p +1kV3fvt9iBURGK/MYAhhBJGnBeHM+OHLcyVZaUhqXcZ4+fFA2dEA5yGvPs2Gaf/1 +aAE/FsU/zLZkHmI8gslCoBrLtc10dgiUne07Hphu14wplVrSCWJImFMNN31fAgMB +AAGjUzBRMB0GA1UdDgQWBBRQyTcy9OASBuFcqS92wJ9LmtdwejAfBgNVHSMEGDAW +gBRQyTcy9OASBuFcqS92wJ9LmtdwejAPBgNVHRMBAf8EBTADAQH/MA0GCSqGSIb3 +DQEBCwUAA4IBAQCnd9ONRHb484+BSpRjfWpuDwGgjdqW9aglEFHOVsUestWe8rSl +9lZBveng/4zIqpIKZpOK04shPwP28lKwPim7EFp9iMSd7lnvSy0vHUhYoVzDtK/V ++QoYcrS/LgijOd+ohaszn2hzWWenNdWkgzSfjzp0WgiGg6j2R5pM9fmX/Wh0KtaC +XF23cM8eLaA5CyVCGRmH637qircr9wLbulld14GNBa9hTqcsp12TE+UX5ULwrGdO +PfiFeC8dmtiNW1p1dhdas52kB92e8EWU1wjFMZACyWpY6x7fxmdLexys5TqKkSn+ +JODwownsXCO4RGHTgQUuZNKFcyfbQGUbF3Ig +-----END CERTIFICATE----- diff --git a/tests/integration/test_reload_ca_certificate/certs/ca2.key b/tests/integration/test_reload_ca_certificate/certs/ca2.key new file mode 100644 index 000000000000..14928a74fcad --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/ca2.key @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvQIBADANBgkqhkiG9w0BAQEFAASCBKcwggSjAgEAAoIBAQC1bXrQUqqds9kX +elFc7J+IrXsEUdFZtxktFglh9eFrffqr0/+vthrcnqZ/OZrLmZV5PSW+gSNuaAVA +rbgGwYWW1lqL0eRfQOktX/NcjoYlWUYlK7lWReFQj2CTCUrP3ojrdo/Oe0/0LsV3 +kctfUagcNxMXoV5NAwHd9dDf9aL1ulljc+RO4ym8BlRqSApShkyH3xklkDLBneHO +UJncwrWcC12p1kV3fvt9iBURGK/MYAhhBJGnBeHM+OHLcyVZaUhqXcZ4+fFA2dEA +5yGvPs2Gaf/1aAE/FsU/zLZkHmI8gslCoBrLtc10dgiUne07Hphu14wplVrSCWJI +mFMNN31fAgMBAAECggEADGEHnovqwk88LSLrxvMKQPF9UaJ3g1nqk4NLvphf+iKi +u53/yzBvYsWccf/Z37HUsMIK/5WgzO63X7NlZBNI7ILRs3VDJ3tZWFQ8rwBhMiPb +lREHrgBJ9beNYlr29M9So6ZM/Pdi/DPuG1DNAe2jZ9Fw3GLmo/XkPGMtDlWcpIwm +DpJDpFDCPfpi0g0z5gIyqVQQW3m9dE8WFvBPUYoDALmKfj8xvPuw1ezEsSkUj06N +YD+/PTOvpQyDhJGVVS9fZyQvdfCszTPMYsNx7e3FVsDdrXRwCFr4gbzxkjQhcwoq +VeZc+1k9R6V17uY3lPL5hfUpLBz8CxfitBzHJjH7QQKBgQDhiAXlNGn0+ggKg3r0 +H2wJ9E6ecyhUVOXoS9f65b+V6sqPSEHWk0jKPJqMXZV6WY4lQHq90WU3hxx2sNiX +jIP6AZdTUVfqJ7gAvBIn14kkLrirmPYx3MURFYoPm+B4LIHcvMlnMtY7BzO3eiv9 +1GDZYDl0mNcx/m5ipYxYLIH34QKBgQDN8CKzju8R1Ogp7m7bbjxSv3PCEzV0jHrs +jQBTSjb60d+Lz/STI6N0H7BMkwvNewnMwQU1xZ8DgGsDXQa4/6x2UiDGHvsCDFTF +HHsXzQCHP8n10vaU4xeCFqyj9/KYElqS8WOn/faH+c1N246+ATTMRDiU3lhI64Bo +iXKmuQcdPwKBgHKYPwabf0su0G8nJ45reOYF8Pyp3tAa40cJYpDltFdkmc/8ExgI +dm/sI0s3MgCdCJD9FmDkyN1SFbBpY2R9zYF21YFMT7N2wxP8e+0qo1BzPPpUGqRz +XN61ZxVPSttFIica9est9ZTAsBKGTVwIUb2iGw+XqaCJe2U8YPdchh2BAoGBAJ+L +fAbqJHMHJDpgG4hqlddxtafUo+RAdXdQIcFlTMTy1aKGoK9hu99aMYaRoWI3ATed +DoFDMldPJRj8+BlZEu6z3+o91C8ZCI+Q6hhdXRxrIfcN0rU0XmENWgDKNir0hTE0 +TAW5LkbYE+NOxv6TBql97OwAehs8QEY8vhNGY6mXAoGASwKe9byhQKXk0X6Q31g+ +5qaDc+fAqirKnPkqexw+1pUv/j6pK0QFBmPAT46Iwt/znAXqXCfSCNwIpR2UW78n +2J9UQR076SUat71MvcbPhesoUhLw4ZyFQYZ1Lo3piokhzjI0zwlmMusDpUdOjdB6 +JJcZmm30hrRMYNncfobFvXc= +-----END PRIVATE KEY----- diff --git a/tests/integration/test_reload_ca_certificate/certs/cert1.crt b/tests/integration/test_reload_ca_certificate/certs/cert1.crt new file mode 100644 index 000000000000..3c2faa29525a --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/cert1.crt @@ -0,0 +1,21 @@ +-----BEGIN CERTIFICATE----- +MIIDbjCCAlagAwIBAgIUB2rETXTp2XN0iWipA9fQo9alsGQwDQYJKoZIhvcNAQEL +BQAwMzEYMBYGA1UECgwPQ2xpY2tIb3VzZSBUZXN0MRcwFQYDVQQDDA5UZXN0IFJv +b3QgQ0EgMTAeFw0yNjA4MzExNzI4MTRaFw0zNjA4MjgxNzI4MTRaMBoxGDAWBgNV +BAMMD2NsaWNraG91c2UtdGVzdDCCASIwDQYJKoZIhvcNAQEBBQADggEPADCCAQoC +ggEBAN7xjZT2K9ocTmXgCKTgRU70rTvDrW9lpupOAASUm+pwWjtwtYs6AyI1eOlM +5ecZlMHC5zhgngxHlQ1MKqMmvFeMJmPZlpG4xPiwXQXxoFvbQ9T6nRvcnxhB2R/p +wSxBvVwwX2Tf7h+f03XwVndC3CkPwdmwR3VsnVld7WPsLaTemUDKefZ6yTR0OVYZ +fN86bw5Ao9ZzjyH079fAvL2BAQ27I64WJz2480dJX9xgp9wzo91oYyy/ox6cA+ls +lPxUIvV/Se9mV0NhtkkdEgC0DidXyKZM79PsDTxnQUYiqF8p6YGv+bDbvBgyaoTJ +uJQ5MyEDKEl6PcTWq06mHWGF7iMCAwEAAaOBkjCBjzAJBgNVHRMEAjAAMAsGA1Ud +DwQEAwIFoDA1BgNVHREELjAsgglsb2NhbGhvc3SCBG5vZGWCBW5vZGUxggVub2Rl +MoIFbm9kZTOHBH8AAAEwHQYDVR0OBBYEFCLJtr15jan+ipoJ4MHQ8hHuj5SmMB8G +A1UdIwQYMBaAFImx3vMM84mQETGNTEnxspVKDakdMA0GCSqGSIb3DQEBCwUAA4IB +AQAhoPNnfpwuqjZClvtMNGZz4He2g09XO7RgE6uFBPG4DKZyJoRwNkrUrdqpkG13 +2D1lV8IUUzgRBqzP4JmVjmm6Vb7106V6epaZXQPar9PaM9pS/whf6G+5lXTVbn0G +skjMAuDeqZLIPh+547hvdI3WguJ+Lt+iqtlpVpqjPqySgG21H7CFMEZ7pKz7bKgc +Q7qzlvVNJc/XDE8nkuxfdBDZODQD3HEQXOVVGx/1x82HXy/EsTS6nBZnO9gOyj5H +gd/+uN2S5oIuxs1a1b1b54MuWZaFz9OGaqGr3CAjSZO1imIfgT4Ftfoa0FAFrSdO +pXpfpZLT5iOYwG+WxcYu1fW5 +-----END CERTIFICATE----- diff --git a/tests/integration/test_reload_ca_certificate/certs/cert1.key b/tests/integration/test_reload_ca_certificate/certs/cert1.key new file mode 100644 index 000000000000..61143a040b98 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/cert1.key @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQDe8Y2U9ivaHE5l +4Aik4EVO9K07w61vZabqTgAElJvqcFo7cLWLOgMiNXjpTOXnGZTBwuc4YJ4MR5UN +TCqjJrxXjCZj2ZaRuMT4sF0F8aBb20PU+p0b3J8YQdkf6cEsQb1cMF9k3+4fn9N1 +8FZ3QtwpD8HZsEd1bJ1ZXe1j7C2k3plAynn2esk0dDlWGXzfOm8OQKPWc48h9O/X +wLy9gQENuyOuFic9uPNHSV/cYKfcM6PdaGMsv6MenAPpbJT8VCL1f0nvZldDYbZJ +HRIAtA4nV8imTO/T7A08Z0FGIqhfKemBr/mw27wYMmqEybiUOTMhAyhJej3E1qtO +ph1hhe4jAgMBAAECggEAOTrp3OTqseVVTLqbjXOS5ydRNwfOxEtkcz5Nq99YPPDV +gO+4csKUHlp6rO0UEWSUNr8pKuRGfiF8BjtYsKQXciPkkPpAuCylx69CWe3CfAIH +4irpXMcgQhJZQeN4Nruzd/Bk9Ji1YIHfPyXQlHHh4VqNqSui1GZq6A+ACogM2Ybe +Qu2tEKGYC7LIYATfcnWMF8QpdF+FDCQmjL8yZTz6WhCDTk+h58y/lDfFwzhCKEwx +Y3g69S16lgIRlzxqoXU9UR31ieGsG+op6hmgrdO57chTUF/+g3qS7fGoFN6g6gsm +GeanWfM2nHpM4A73+TF7aI8xf4IHIgkXYa/KEXXh4QKBgQDz+Kj7OEQvv2NIs+lY +RkYTl0ljof8IzFPp8qZEo9z0PpD/Nl9sphfoV/ZYj79Ijnb8bkPcK2aqn2f9Ortx +I28XtzZog/p4XVSLw0o6hRRENsZEpsxYebaGp6kighZM9JHPm67T4neQ5PcR0r4g +Kz6DpoN/bM+uiWiULIMBKeHlwwKBgQDp73x77kvFXGb/Xc67qjfT7H8gPdBcAajd +SGeQHP+AbvsxZaEp0hMSXHYoJRq4eSsA5h8BOHDjM39LZ4uvlcphQDpuCHC/cbUI +bt5PmuddY/YTDkvEi/WIZi9+HSuAZ+RnVgbZowuTV0/t776Fj8R/MefDuWSuiDvF +Bt2z1XZwIQKBgBGZxtcY4BJxxD/ietsbdsLDD1BYx4Vi+ErQbp5VFAOq39sJmSjF +csQYVHVfKXWakYr0iYDAwM9eYKosKomm/MTBOvOfUdqNISRUGm7OWv/w06zwO53G +ahycy97pc6JpontPx/URSX7yhcCLa5v2grQMtz/iIbl9wEWwUGMtGlbxAoGANQD3 +IplWf6w1Bg06JxklNxYxo5t91yrlGOYr2OJJHc+HiKSvRGt9uL5MY0Is8Lk7fiOl +yMACC+iCIhKe+rSkuy4zTvUInsfjrbp5Em5Vl7pradvmXO0dP79vaVKwpZJklOlP ++gXQPJ0e1hlpAJgXfH5RNe6OmmDxse2hU/q8sCECgYBmK1MbPoX0SbuzPmAT6iu8 +4spHZp/AsiWOAnXU42VgEJCRfH+086CvZdf8N7Q6cJDfxofO95syDIMkC/7bkTGT +WT+R+1s0ouZFBtcX+6uPf9eR0/5W3SCzV1YkQtw6iXhoFociRdXDGO8JON80+x8e +BWtbo3oEMlwc6/rNAELLbQ== +-----END PRIVATE KEY----- diff --git a/tests/integration/test_reload_ca_certificate/certs/cert2.crt b/tests/integration/test_reload_ca_certificate/certs/cert2.crt new file mode 100644 index 000000000000..4fb702bf143b --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/cert2.crt @@ -0,0 +1,21 @@ +-----BEGIN CERTIFICATE----- +MIIDbjCCAlagAwIBAgIUIcogGF2yo/W8Z+f1vXaSVWnpDO4wDQYJKoZIhvcNAQEL +BQAwMzEYMBYGA1UECgwPQ2xpY2tIb3VzZSBUZXN0MRcwFQYDVQQDDA5UZXN0IFJv +b3QgQ0EgMjAeFw0yNjA4MzExNzI4MTVaFw0zNjA4MjgxNzI4MTVaMBoxGDAWBgNV +BAMMD2NsaWNraG91c2UtdGVzdDCCASIwDQYJKoZIhvcNAQEBBQADggEPADCCAQoC +ggEBAJKHwEUY5CUuRs7G9Zv+slTng5op18vSYWpj9m1/VagGhkRLlp6FGqf0+pnb +m1A0T0sXKVeuKFgJWqkTTJ5BuP2IOnS+HsfJ2oc0g2vRpO8v0mtZBdT8KA/W6fur +bxQeUOAoEIGca1r0ha/nRAdwvptFRIiIbDgIe9NXR7DQVBWe7Q+DxpzDhjR3/EvZ +W1/jjvTyg/WRhnM3WhrxfeBrHm6o1d8SsDR1t2FTe+7tiBh4ItI1ZWQUSpPhs9qf +LGe5jDx0+9fy4bj8isyMfGvnT+Ll5QFQn34DZrhQ+CAnU6AukZw2Bc3tIfLqtx28 +dh1VUDl0j/BRdbXerEzla1AVCb8CAwEAAaOBkjCBjzAJBgNVHRMEAjAAMAsGA1Ud +DwQEAwIFoDA1BgNVHREELjAsgglsb2NhbGhvc3SCBG5vZGWCBW5vZGUxggVub2Rl +MoIFbm9kZTOHBH8AAAEwHQYDVR0OBBYEFCjHe0KJ1rhLUhZfsMAtv5KSBYC0MB8G +A1UdIwQYMBaAFFDJNzL04BIG4VypL3bAn0ua13B6MA0GCSqGSIb3DQEBCwUAA4IB +AQAbU63n0EM1JWgCDaHYTo9FvnCmnmc1Ge2Nnz5lCwwom5kzB7GPjHljsx3L93OG +ETPbFD1On4OZI9ynURh+9mGlTAE3RiCHT+UjnHnQlKpn0mEjNihXGGM9iMIe4a/q +iVM2To3ayfjuMviLqgpRk48941rEbSSlnaZu+9bXG87QfRdRww2brRF1YOh3IvdT +lawdGc4Rlw/Hh9XjEBJ4u9dtRneoScKI//dzkTKuY0Jj3LeWRq8quX4+DnJZaBlx +vhU9YHstiNIooT1WGlHfNErOB7rsjWbAexWAdIo4gJVTM8caQtpaYPbOBxvCMaMX +/GD8THgD62RZxT80PWXdr4RW +-----END CERTIFICATE----- diff --git a/tests/integration/test_reload_ca_certificate/certs/cert2.key b/tests/integration/test_reload_ca_certificate/certs/cert2.key new file mode 100644 index 000000000000..beeebe86de1f --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/cert2.key @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQCSh8BFGOQlLkbO +xvWb/rJU54OaKdfL0mFqY/Ztf1WoBoZES5aehRqn9PqZ25tQNE9LFylXrihYCVqp +E0yeQbj9iDp0vh7HydqHNINr0aTvL9JrWQXU/CgP1un7q28UHlDgKBCBnGta9IWv +50QHcL6bRUSIiGw4CHvTV0ew0FQVnu0Pg8acw4Y0d/xL2Vtf44708oP1kYZzN1oa +8X3gax5uqNXfErA0dbdhU3vu7YgYeCLSNWVkFEqT4bPanyxnuYw8dPvX8uG4/IrM +jHxr50/i5eUBUJ9+A2a4UPggJ1OgLpGcNgXN7SHy6rcdvHYdVVA5dI/wUXW13qxM +5WtQFQm/AgMBAAECggEAF9DYR21vbhBH8kV0gl69Zb8H7O9zw2KAcoDTJY072FFQ +fbAwHQ1dku071UmpbuSZpgbv3Brn3sCNIOGMo0TjstLIoX83KdodL7rcHhLsdPMi +XE/opUCN/bO8wwAO+pz0yyy9MLvCLSiExAguzpYcvMg02+DEf2pNlJY2hj6jXqyq +PKQXqTdtclrTfsjLBDJfqysZA5TrEV8tglU4uYr7fD2q3iFJ90VsBOKZaNs5jEsC +TJfXYWOf5OngwUD5nVCWHfsltJyysbhHUPQJcsk4G2zzQdX9ASS6wMc67rBuz25B +EMzR0ZP2eGAYqhbmYa6oVr0kE1GnukxLkQqtHOeAuQKBgQDEKEu4CRri/fw+RXOB +S33jdEPFP5hzPhx/41bKQzvZ/VkVRH7SLxMGIQjgkd1j7Klx7o42iUJPUYtlN8Gc +0+HxmnUjAh81F2fAq+C3rqRo2D0KQT3izHPGKxfmyPhPXVKrL8FVXGWMv1wvSHpJ +HQ3M4KaD9GMKmnswwyPArM0OBwKBgQC/O5/bqJf+ltCVeE4q77VxsHNcrJfxwgn6 +1I6ZcoAUPPyywOAG4WLaiH41I3CPydFYLfDk6AnbO9JQybvnu69OPa+At++Dggm6 +MSAYvYigQ4RsZuc+kzaV9//zhWIVa5kcfljyo1L49cLf2uBz3JWziEqYsl4TmCtp +bs8oWME4iQKBgAJU5EmEujAWisgGtU/FIPLyL9gJYHuGMnqGrkJrOCvoKgXpsYQ4 +EQbSn7NjqHkGmCEFj+UwDny44GpMll2R2y6vAlNvNAXCiHYu1NX6GnQwldEoY17t +xTaGzprsqp7u4gus3qRwG7jnkWXye5mg4cgcp34MCp1Wpr42o5cntqxDAoGAKP89 +XDgercPjX8f06huNyJvNf5a41GmG/jFHiPoVH0Gb4y6aWJ9FNBiDBh1c6laX/NGM +jWZ5hniitBMrp5iDEsECuRO103mzYClb+jHX8pPG9f5xoOaqkyghxTFZP8JbhtJH +e20sQpddeeRQrkYiCeU0KNxEcuryk53f54RvmBECgYAO8Yo/NA6Cz7FIidmb4tL5 +n3frL1fdA7bY74ltjMJp4dCmvlezbKG2xdSKqDlEvfoWgRrZEBtjH/i5esZhzH/H +6z0rK/TmoLpW5HLmBJ5Fqmj5lWOvmtNVh63QcW/hTFOGRSzuESzTD1qafMjtAAvZ +HBvkHzbEbZ6tVlKpNnNZzg== +-----END PRIVATE KEY----- diff --git a/tests/integration/test_reload_ca_certificate/certs/generate_certs.sh b/tests/integration/test_reload_ca_certificate/certs/generate_certs.sh new file mode 100755 index 000000000000..c487b3941426 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/generate_certs.sh @@ -0,0 +1,41 @@ +#!/bin/bash +# Generates the certificates for test_reload_ca_certificate: +# ca1.crt / ca1.key - first root CA +# ca2.crt / ca2.key - second, unrelated root CA +# cert1.crt / cert1.key - leaf certificate issued by ca1 +# cert2.crt / cert2.key - leaf certificate issued by ca2 +# Both leaf certificates are valid for all host names used in the test and carry no extended key usage, +# so they can be used as server, client and Keeper (Raft) certificates. +# ca.crt, node.crt, node.key - copies of ca1.crt, cert1.crt, cert1.key: the initial content of the files +# that the configs point to and that the test overwrites to rotate certificates. +set -e +cd "$(dirname "${BASH_SOURCE[0]}")" + +DAYS=3650 + +cat > leaf.cnf << 'EOC' +[req] +distinguished_name = dn +prompt = no +[dn] +CN = clickhouse-test +[v3_leaf] +basicConstraints = CA:FALSE +keyUsage = digitalSignature, keyEncipherment +subjectAltName = DNS:localhost, DNS:node, DNS:node1, DNS:node2, DNS:node3, IP:127.0.0.1 +EOC + +for i in 1 2; do + openssl req -x509 -newkey rsa:2048 -nodes -batch -sha256 -days $DAYS \ + -subj "/O=ClickHouse Test/CN=Test Root CA $i" -keyout ca$i.key -out ca$i.crt + + openssl req -newkey rsa:2048 -nodes -batch -config leaf.cnf -keyout cert$i.key -out cert$i.csr + openssl x509 -req -in cert$i.csr -CA ca$i.crt -CAkey ca$i.key -CAcreateserial -sha256 -days $DAYS \ + -extfile leaf.cnf -extensions v3_leaf -out cert$i.crt + rm -f cert$i.csr ca$i.srl +done +rm -f leaf.cnf + +cp ca1.crt ca.crt +cp cert1.crt node.crt +cp cert1.key node.key diff --git a/tests/integration/test_reload_ca_certificate/certs/node.crt b/tests/integration/test_reload_ca_certificate/certs/node.crt new file mode 100644 index 000000000000..3c2faa29525a --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/node.crt @@ -0,0 +1,21 @@ +-----BEGIN CERTIFICATE----- +MIIDbjCCAlagAwIBAgIUB2rETXTp2XN0iWipA9fQo9alsGQwDQYJKoZIhvcNAQEL +BQAwMzEYMBYGA1UECgwPQ2xpY2tIb3VzZSBUZXN0MRcwFQYDVQQDDA5UZXN0IFJv +b3QgQ0EgMTAeFw0yNjA4MzExNzI4MTRaFw0zNjA4MjgxNzI4MTRaMBoxGDAWBgNV +BAMMD2NsaWNraG91c2UtdGVzdDCCASIwDQYJKoZIhvcNAQEBBQADggEPADCCAQoC +ggEBAN7xjZT2K9ocTmXgCKTgRU70rTvDrW9lpupOAASUm+pwWjtwtYs6AyI1eOlM +5ecZlMHC5zhgngxHlQ1MKqMmvFeMJmPZlpG4xPiwXQXxoFvbQ9T6nRvcnxhB2R/p +wSxBvVwwX2Tf7h+f03XwVndC3CkPwdmwR3VsnVld7WPsLaTemUDKefZ6yTR0OVYZ +fN86bw5Ao9ZzjyH079fAvL2BAQ27I64WJz2480dJX9xgp9wzo91oYyy/ox6cA+ls +lPxUIvV/Se9mV0NhtkkdEgC0DidXyKZM79PsDTxnQUYiqF8p6YGv+bDbvBgyaoTJ +uJQ5MyEDKEl6PcTWq06mHWGF7iMCAwEAAaOBkjCBjzAJBgNVHRMEAjAAMAsGA1Ud +DwQEAwIFoDA1BgNVHREELjAsgglsb2NhbGhvc3SCBG5vZGWCBW5vZGUxggVub2Rl +MoIFbm9kZTOHBH8AAAEwHQYDVR0OBBYEFCLJtr15jan+ipoJ4MHQ8hHuj5SmMB8G +A1UdIwQYMBaAFImx3vMM84mQETGNTEnxspVKDakdMA0GCSqGSIb3DQEBCwUAA4IB +AQAhoPNnfpwuqjZClvtMNGZz4He2g09XO7RgE6uFBPG4DKZyJoRwNkrUrdqpkG13 +2D1lV8IUUzgRBqzP4JmVjmm6Vb7106V6epaZXQPar9PaM9pS/whf6G+5lXTVbn0G +skjMAuDeqZLIPh+547hvdI3WguJ+Lt+iqtlpVpqjPqySgG21H7CFMEZ7pKz7bKgc +Q7qzlvVNJc/XDE8nkuxfdBDZODQD3HEQXOVVGx/1x82HXy/EsTS6nBZnO9gOyj5H +gd/+uN2S5oIuxs1a1b1b54MuWZaFz9OGaqGr3CAjSZO1imIfgT4Ftfoa0FAFrSdO +pXpfpZLT5iOYwG+WxcYu1fW5 +-----END CERTIFICATE----- diff --git a/tests/integration/test_reload_ca_certificate/certs/node.key b/tests/integration/test_reload_ca_certificate/certs/node.key new file mode 100644 index 000000000000..61143a040b98 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/certs/node.key @@ -0,0 +1,28 @@ +-----BEGIN PRIVATE KEY----- +MIIEvAIBADANBgkqhkiG9w0BAQEFAASCBKYwggSiAgEAAoIBAQDe8Y2U9ivaHE5l +4Aik4EVO9K07w61vZabqTgAElJvqcFo7cLWLOgMiNXjpTOXnGZTBwuc4YJ4MR5UN +TCqjJrxXjCZj2ZaRuMT4sF0F8aBb20PU+p0b3J8YQdkf6cEsQb1cMF9k3+4fn9N1 +8FZ3QtwpD8HZsEd1bJ1ZXe1j7C2k3plAynn2esk0dDlWGXzfOm8OQKPWc48h9O/X +wLy9gQENuyOuFic9uPNHSV/cYKfcM6PdaGMsv6MenAPpbJT8VCL1f0nvZldDYbZJ +HRIAtA4nV8imTO/T7A08Z0FGIqhfKemBr/mw27wYMmqEybiUOTMhAyhJej3E1qtO +ph1hhe4jAgMBAAECggEAOTrp3OTqseVVTLqbjXOS5ydRNwfOxEtkcz5Nq99YPPDV +gO+4csKUHlp6rO0UEWSUNr8pKuRGfiF8BjtYsKQXciPkkPpAuCylx69CWe3CfAIH +4irpXMcgQhJZQeN4Nruzd/Bk9Ji1YIHfPyXQlHHh4VqNqSui1GZq6A+ACogM2Ybe +Qu2tEKGYC7LIYATfcnWMF8QpdF+FDCQmjL8yZTz6WhCDTk+h58y/lDfFwzhCKEwx +Y3g69S16lgIRlzxqoXU9UR31ieGsG+op6hmgrdO57chTUF/+g3qS7fGoFN6g6gsm +GeanWfM2nHpM4A73+TF7aI8xf4IHIgkXYa/KEXXh4QKBgQDz+Kj7OEQvv2NIs+lY +RkYTl0ljof8IzFPp8qZEo9z0PpD/Nl9sphfoV/ZYj79Ijnb8bkPcK2aqn2f9Ortx +I28XtzZog/p4XVSLw0o6hRRENsZEpsxYebaGp6kighZM9JHPm67T4neQ5PcR0r4g +Kz6DpoN/bM+uiWiULIMBKeHlwwKBgQDp73x77kvFXGb/Xc67qjfT7H8gPdBcAajd +SGeQHP+AbvsxZaEp0hMSXHYoJRq4eSsA5h8BOHDjM39LZ4uvlcphQDpuCHC/cbUI +bt5PmuddY/YTDkvEi/WIZi9+HSuAZ+RnVgbZowuTV0/t776Fj8R/MefDuWSuiDvF +Bt2z1XZwIQKBgBGZxtcY4BJxxD/ietsbdsLDD1BYx4Vi+ErQbp5VFAOq39sJmSjF +csQYVHVfKXWakYr0iYDAwM9eYKosKomm/MTBOvOfUdqNISRUGm7OWv/w06zwO53G +ahycy97pc6JpontPx/URSX7yhcCLa5v2grQMtz/iIbl9wEWwUGMtGlbxAoGANQD3 +IplWf6w1Bg06JxklNxYxo5t91yrlGOYr2OJJHc+HiKSvRGt9uL5MY0Is8Lk7fiOl +yMACC+iCIhKe+rSkuy4zTvUInsfjrbp5Em5Vl7pradvmXO0dP79vaVKwpZJklOlP ++gXQPJ0e1hlpAJgXfH5RNe6OmmDxse2hU/q8sCECgYBmK1MbPoX0SbuzPmAT6iu8 +4spHZp/AsiWOAnXU42VgEJCRfH+086CvZdf8N7Q6cJDfxofO95syDIMkC/7bkTGT +WT+R+1s0ouZFBtcX+6uPf9eR0/5W3SCzV1YkQtw6iXhoFociRdXDGO8JON80+x8e +BWtbo3oEMlwc6/rNAELLbQ== +-----END PRIVATE KEY----- diff --git a/tests/integration/test_reload_ca_certificate/configs/keeper1.xml b/tests/integration/test_reload_ca_certificate/configs/keeper1.xml new file mode 100644 index 000000000000..427a4b8f6872 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/configs/keeper1.xml @@ -0,0 +1,36 @@ + + 0.0.0.0 + 0.0.0.0 + + 0 + 9181 + 1 + /var/lib/clickhouse/coordination/log + /var/lib/clickhouse/coordination/snapshots + + + 5000 + 10000 + trace + + + + true + + 1 + node1 + 9234 + + + 2 + node2 + 9234 + + + 3 + node3 + 9234 + + + + diff --git a/tests/integration/test_reload_ca_certificate/configs/keeper2.xml b/tests/integration/test_reload_ca_certificate/configs/keeper2.xml new file mode 100644 index 000000000000..5dac68a6af56 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/configs/keeper2.xml @@ -0,0 +1,36 @@ + + 0.0.0.0 + 0.0.0.0 + + 0 + 9181 + 2 + /var/lib/clickhouse/coordination/log + /var/lib/clickhouse/coordination/snapshots + + + 5000 + 10000 + trace + + + + true + + 1 + node1 + 9234 + + + 2 + node2 + 9234 + + + 3 + node3 + 9234 + + + + diff --git a/tests/integration/test_reload_ca_certificate/configs/keeper3.xml b/tests/integration/test_reload_ca_certificate/configs/keeper3.xml new file mode 100644 index 000000000000..28a155e92631 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/configs/keeper3.xml @@ -0,0 +1,36 @@ + + 0.0.0.0 + 0.0.0.0 + + 0 + 9181 + 3 + /var/lib/clickhouse/coordination/log + /var/lib/clickhouse/coordination/snapshots + + + 5000 + 10000 + trace + + + + true + + 1 + node1 + 9234 + + + 2 + node2 + 9234 + + + 3 + node3 + 9234 + + + + diff --git a/tests/integration/test_reload_ca_certificate/configs/ssl.xml b/tests/integration/test_reload_ca_certificate/configs/ssl.xml new file mode 100644 index 000000000000..3b93b0a7552a --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/configs/ssl.xml @@ -0,0 +1,31 @@ + + 8443 + + + + + /etc/clickhouse-server/config.d/node.crt + /etc/clickhouse-server/config.d/node.key + /etc/clickhouse-server/config.d/ca.crt + false + + relaxed + false + sslv2,sslv3 + true + + + /etc/clickhouse-server/config.d/node.crt + /etc/clickhouse-server/config.d/node.key + /etc/clickhouse-server/config.d/ca.crt + false + relaxed + false + sslv2,sslv3 + true + + RejectCertificateHandler + + + + diff --git a/tests/integration/test_reload_ca_certificate/configs/ssl_with_default_cas.xml b/tests/integration/test_reload_ca_certificate/configs/ssl_with_default_cas.xml new file mode 100644 index 000000000000..6a1077ce40b9 --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/configs/ssl_with_default_cas.xml @@ -0,0 +1,29 @@ + + 8443 + + + + + /etc/clickhouse-server/config.d/node.crt + /etc/clickhouse-server/config.d/node.key + /etc/clickhouse-server/config.d/ca.crt + + relaxed + false + sslv2,sslv3 + true + + + /etc/clickhouse-server/config.d/node.crt + /etc/clickhouse-server/config.d/node.key + /etc/clickhouse-server/config.d/ca.crt + relaxed + false + sslv2,sslv3 + true + + RejectCertificateHandler + + + + diff --git a/tests/integration/test_reload_ca_certificate/test.py b/tests/integration/test_reload_ca_certificate/test.py new file mode 100644 index 000000000000..473a2e850e2c --- /dev/null +++ b/tests/integration/test_reload_ca_certificate/test.py @@ -0,0 +1,275 @@ +""" +Hot reload of the trusted CA certificates (`openSSL.*.caConfig`). + +The certificates in certs/ are produced by certs/generate_certs.sh: +two unrelated root CAs (ca1, ca2) and one leaf certificate issued by each of them (cert1, cert2). +Every instance starts with ca.crt = ca1 as the trusted CA and node.crt/node.key = cert1 as its own certificate, +and the tests overwrite these files in the container to rotate them without restarting anything. +""" + +import time +import uuid + +import pytest + +import helpers.keeper_utils as ku +from helpers.cluster import ClickHouseCluster + +cluster = ClickHouseCluster(__file__) + +CONFIG_DIR = "/etc/clickhouse-server/config.d" +CERT_FILES = [ + "certs/ca.crt", + "certs/ca1.crt", + "certs/ca2.crt", + "certs/node.crt", + "certs/node.key", + "certs/cert1.crt", + "certs/cert1.key", + "certs/cert2.crt", + "certs/cert2.key", +] + +# Serves HTTPS and acts as a TLS client towards itself. +node = cluster.add_instance("node", main_configs=["configs/ssl.xml"] + CERT_FILES) + +# Does not set `loadDefaultCAFile`, and OpenSSL's default CA file is ca2 for this instance. +node_with_default_cas = cluster.add_instance( + "node_with_default_cas", + main_configs=["configs/ssl_with_default_cas.xml"] + CERT_FILES, + env_variables={"SSL_CERT_FILE": f"{CONFIG_DIR}/ca2.crt"}, +) + +# Three nodes with embedded Keeper talking Raft over TLS. `loadDefaultCAFile` is not set for them: Keeper assumes `false` for +# the Raft connections then, unlike everything else, and their CA certificates have to be reloaded all the same. +keeper_nodes = [ + cluster.add_instance(f"node{i}", main_configs=[f"configs/keeper{i}.xml", "configs/ssl_with_default_cas.xml"] + CERT_FILES) for i in (1, 2, 3) +] + + +@pytest.fixture(scope="module") +def started_cluster(): + try: + cluster.start() + yield cluster + finally: + cluster.shutdown() + + +def set_trusted_cas(instance, *cas): + """Overwrite ca.crt (the configured `caConfig`) with the given CA certificates.""" + sources = " ".join(f"{CONFIG_DIR}/{ca}.crt" for ca in cas) + instance.exec_in_container(["bash", "-c", f"cat {sources} > {CONFIG_DIR}/ca.crt.tmp && mv {CONFIG_DIR}/ca.crt.tmp {CONFIG_DIR}/ca.crt"]) + + +def set_own_certificate(instance, cert): + """Overwrite node.crt/node.key (the configured `certificateFile`/`privateKeyFile`) with the given leaf certificate.""" + instance.exec_in_container( + [ + "bash", + "-c", + f"cp {CONFIG_DIR}/{cert}.crt {CONFIG_DIR}/node.crt.tmp && mv {CONFIG_DIR}/node.crt.tmp {CONFIG_DIR}/node.crt && " + f"cp {CONFIG_DIR}/{cert}.key {CONFIG_DIR}/node.key.tmp && mv {CONFIG_DIR}/node.key.tmp {CONFIG_DIR}/node.key", + ] + ) + + +@pytest.fixture(autouse=True) +def restore_certificates(started_cluster): + yield + for instance in [node, node_with_default_cas] + keeper_nodes: + set_trusted_cas(instance, "ca1") + set_own_certificate(instance, "cert1") + instance.query("SYSTEM RELOAD CONFIG") + for instance in keeper_nodes: + kill_raft_connections(instance) + ku.wait_nodes(cluster, keeper_nodes) + + +def https_request_with_client_certificate(cert, instance=node): + """Query the HTTPS port of `instance` presenting the given client certificate. Returns the response, or None if the TLS handshake failed.""" + result = instance.exec_in_container( + [ + "bash", + "-c", + f"curl --silent --show-error --cacert {CONFIG_DIR}/ca1.crt --cert {CONFIG_DIR}/{cert}.crt --key {CONFIG_DIR}/{cert}.key " + f"'https://localhost:8443/?query=SELECT%201' 2>&1 || echo CURL_FAILED", + ] + ) + return None if "CURL_FAILED" in result else result + + +def assert_eventually(predicate, description, timeout=60): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if predicate(): + return + time.sleep(0.5) + assert predicate(), description + + +def test_server_reloads_ca(started_cluster): + """The CAs used to verify client certificates follow the content of `caConfig` without a restart or an explicit reload.""" + assert https_request_with_client_certificate("cert1") == "1\n" + assert https_request_with_client_certificate("cert2") is None + + # Trust both CAs: the file change alone triggers the reload. + set_trusted_cas(node, "ca1", "ca2") + assert_eventually(lambda: https_request_with_client_certificate("cert2") == "1\n", "cert2 is accepted after ca2 was added") + assert https_request_with_client_certificate("cert1") == "1\n" + + # Drop the old CA: certificates issued by it are not accepted anymore. + set_trusted_cas(node, "ca2") + assert_eventually(lambda: https_request_with_client_certificate("cert1") is None, "cert1 is rejected after ca1 was removed") + assert https_request_with_client_certificate("cert2") == "1\n" + + assert node.contains_in_log("Reloaded CA certificates") + + +def test_client_reloads_ca(started_cluster): + """The CAs used to verify server certificates of outgoing connections follow the content of `caConfig`.""" + # `node` connects to its own HTTPS port, which serves cert1. `Connection: close` rules out reusing a pooled connection. + query = "SELECT * FROM url('https://localhost:8443/?query=SELECT%201', 'TSV', 'x UInt8', headers('Connection'='close'))" + + set_trusted_cas(node, "ca2") + node.query("SYSTEM RELOAD CONFIG") + error = node.query_and_get_error(query) + assert "certificate verify failed" in error, error + + set_trusted_cas(node, "ca1") + node.query("SYSTEM RELOAD CONFIG") + assert node.query(query) == "1\n" + + +def test_default_cas_are_kept(started_cluster): + """With `loadDefaultCAFile` left at its default, both the CAs from `caConfig` and the default ones are trusted, before and after a reload.""" + # cert1 is trusted through `caConfig` (ca.crt = ca1), cert2 through the default CA file (SSL_CERT_FILE = ca2). + assert https_request_with_client_certificate("cert1", node_with_default_cas) == "1\n" + assert https_request_with_client_certificate("cert2", node_with_default_cas) == "1\n" + + reloads = int(node_with_default_cas.count_in_log("Reloaded CA certificates").strip()) + set_trusted_cas(node_with_default_cas, "ca1") # same content, new modification time + node_with_default_cas.query("SYSTEM RELOAD CONFIG") + assert int(node_with_default_cas.count_in_log("Reloaded CA certificates").strip()) > reloads + + assert https_request_with_client_certificate("cert1", node_with_default_cas) == "1\n" + assert https_request_with_client_certificate("cert2", node_with_default_cas) == "1\n" + + +def test_system_certificates_follows_reload(started_cluster): + """`system.certificates` shows the CA certificates that are currently used, also after `caConfig` is changed to another file.""" + query = "SELECT path, subject LIKE '%Test Root CA {}%' FROM system.certificates WHERE NOT default" + assert node.query(query.format(1)) == f"{CONFIG_DIR}/ca.crt\t1\n" + + node.replace_in_config(f"{CONFIG_DIR}/ssl.xml", f"{CONFIG_DIR}/ca.crt", f"{CONFIG_DIR}/ca2.crt") + try: + node.query("SYSTEM RELOAD CONFIG") + assert node.query(query.format(2)) == f"{CONFIG_DIR}/ca2.crt\t1\n" + finally: + node.replace_in_config(f"{CONFIG_DIR}/ssl.xml", f"{CONFIG_DIR}/ca2.crt", f"{CONFIG_DIR}/ca.crt") + node.query("SYSTEM RELOAD CONFIG") + + +def test_ca_directory(started_cluster): + """`caConfig` can be a directory with certificates named by their subject hash. Replacing a certificate in it is noticed too.""" + ca_dir = f"{CONFIG_DIR}/ca_dir" + node.exec_in_container(["bash", "-c", f"mkdir -p {ca_dir} && cp {CONFIG_DIR}/ca1.crt {ca_dir}/$(openssl x509 -noout -subject_hash -in {CONFIG_DIR}/ca1.crt).0"]) + node.replace_in_config(f"{CONFIG_DIR}/ssl.xml", f"{CONFIG_DIR}/ca.crt", ca_dir) + try: + node.query("SYSTEM RELOAD CONFIG") + assert https_request_with_client_certificate("cert1") == "1\n" + assert https_request_with_client_certificate("cert2") is None + + # Overwrite the only file in place: the set of file names in the directory does not change. + node.exec_in_container(["bash", "-c", f"cat {CONFIG_DIR}/ca2.crt > {ca_dir}/$(openssl x509 -noout -subject_hash -in {CONFIG_DIR}/ca1.crt).0"]) + node.query("SYSTEM RELOAD CONFIG") + assert https_request_with_client_certificate("cert1") is None + finally: + node.replace_in_config(f"{CONFIG_DIR}/ssl.xml", ca_dir, f"{CONFIG_DIR}/ca.crt") + node.exec_in_container(["rm", "-rf", ca_dir]) + node.query("SYSTEM RELOAD CONFIG") + + +def kill_raft_connections(instance): + instance.exec_in_container( + ["bash", "-c", "ss --kill -tn state established '( dport = :9234 or sport = :9234 )' > /dev/null"], nothrow=True + ) + + +def raft_port_accepts_client_certificate(instance, target, cert): + """Whether the Raft port of `target` completes a TLS handshake with a client that presents `cert`.""" + # With TLS 1.2 the server verifies the client certificate before the handshake completes on the client side, + # so a rejected certificate reliably shows up as a failed handshake in s_client. + result = instance.exec_in_container( + [ + "bash", + "-c", + f"openssl s_client -brief -tls1_2 -connect {target.name}:9234 -cert {CONFIG_DIR}/{cert}.crt -key {CONFIG_DIR}/{cert}.key " + f"&1 || true", + ] + ) + return "CONNECTION ESTABLISHED" in result + + +def raft_port_certificate_issuer(instance, target): + return instance.exec_in_container( + [ + "bash", + "-c", + f"openssl s_client -connect {target.name}:9234 /dev/null | openssl x509 -noout -issuer 2>/dev/null || true", + ] + ).strip() + + +def check_keeper_cluster_works(path): + connections = [] + try: + for instance in keeper_nodes: + connections.append(ku.get_fake_zk(cluster, instance.name)) + connections[0].create(path, b"data") + for connection in connections: + connection.sync(path) + assert connection.get(path)[0] == b"data" + finally: + for connection in connections: + connection.stop() + connection.close() + + +def reload_and_reconnect_raft(): + for instance in keeper_nodes: + instance.query("SYSTEM RELOAD CONFIG") + for instance in keeper_nodes: + kill_raft_connections(instance) + ku.wait_nodes(cluster, keeper_nodes) + + +def test_keeper_raft_reloads_ca(started_cluster): + """Rotate the CA of the Raft connections between Keeper nodes without restarting them.""" + run = uuid.uuid4().hex + ku.wait_nodes(cluster, keeper_nodes) + check_keeper_cluster_works(f"/before_{run}") + assert "Test Root CA 1" in raft_port_certificate_issuer(keeper_nodes[0], keeper_nodes[1]) + + # 1. Trust the new CA in addition to the old one. + for instance in keeper_nodes: + set_trusted_cas(instance, "ca1", "ca2") + reload_and_reconnect_raft() + check_keeper_cluster_works(f"/both_cas_{run}") + + # 2. Switch the nodes to certificates issued by the new CA. + for instance in keeper_nodes: + set_own_certificate(instance, "cert2") + reload_and_reconnect_raft() + check_keeper_cluster_works(f"/new_certs_{run}") + for target in keeper_nodes[1:]: + assert "Test Root CA 2" in raft_port_certificate_issuer(keeper_nodes[0], target) + + # 3. Stop trusting the old CA. + for instance in keeper_nodes: + set_trusted_cas(instance, "ca2") + reload_and_reconnect_raft() + check_keeper_cluster_works(f"/new_ca_{run}") + + assert raft_port_accepts_client_certificate(keeper_nodes[0], keeper_nodes[1], "cert2") + assert not raft_port_accepts_client_certificate(keeper_nodes[0], keeper_nodes[1], "cert1") From 36971ad11119f280dfec6bcfe25c1391ca784659 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Sun, 27 Sep 2026 22:52:09 +0000 Subject: [PATCH 062/185] Backport #122337 to 26.8: Fix Keeper crash after a failed changelog preallocation --- src/Common/FailPoint.cpp | 1 + src/Coordination/Changelog.cpp | 13 +++++--- .../tests/gtest_coordination_changelog.cpp | 31 +++++++++++++++++++ 3 files changed, 41 insertions(+), 4 deletions(-) diff --git a/src/Common/FailPoint.cpp b/src/Common/FailPoint.cpp index 4ea56fc9fb76..60ae93c6ff4c 100644 --- a/src/Common/FailPoint.cpp +++ b/src/Common/FailPoint.cpp @@ -349,6 +349,7 @@ static struct InitFiu PAUSEABLE(keeper_changelog_readahead_pre_drain) \ PAUSEABLE(object_storage_source_pause_before_virtual_columns) \ REGULAR(keeper_changelog_readahead_fill_exception) \ + ONCE(keeper_changelog_preallocate_no_space) \ REGULAR(distributed_plan_record_failure_while_starting_tasks) \ ONCE(zk_send_thread_request_window_throw) \ ONCE(zk_send_thread_operations_insert_throw) \ diff --git a/src/Coordination/Changelog.cpp b/src/Coordination/Changelog.cpp index 4b358776b2c1..a5d204ea5bce 100644 --- a/src/Coordination/Changelog.cpp +++ b/src/Coordination/Changelog.cpp @@ -92,6 +92,7 @@ namespace FailPoints extern const char keeper_changelog_readahead_park_armed[]; extern const char keeper_changelog_readahead_pre_drain[]; extern const char keeper_changelog_readahead_fill_exception[]; + extern const char keeper_changelog_preallocate_no_space[]; } namespace @@ -347,7 +348,8 @@ class ChangelogWriter bool appendRecord(ChangelogRecord && record) { const auto * file_buffer = tryGetFileBaseBuffer(); - chassert(file_buffer && current_file_description); + if (!file_buffer || !current_file_description) + throw Exception(ErrorCodes::LOGICAL_ERROR, "Log writer wasn't initialized for any file"); chassert(record.header.index - getStartIndex() <= current_file_description->expectedEntriesCountInLog()); // check if log file reached the limit for amount of records it can contain @@ -590,6 +592,12 @@ class ChangelogWriter file_buffer->getFD(), FALLOC_FL_KEEP_SIZE, 0, log_file_settings.max_size + log_file_settings.overallocate_size); } while (res < 0 && errno == EINTR); + fiu_do_on(FailPoints::keeper_changelog_preallocate_no_space, + { + res = -1; + errno = ENOSPC; + }); + if (res != 0) { if (errno == ENOSPC) @@ -4154,9 +4162,6 @@ void Changelog::appendCompletionThread() bool append_ok = false; while (append_completion_queue.pop(append_ok)) { - if (!append_ok) - current_writer->finalize(); - // we shouldn't start the raft_server before sending it here if (auto raft_server_locked = raft_server.lock()) raft_server_locked->notify_log_append_completion(append_ok); diff --git a/src/Coordination/tests/gtest_coordination_changelog.cpp b/src/Coordination/tests/gtest_coordination_changelog.cpp index 74e6c2b1a2ff..ef2a02920b17 100644 --- a/src/Coordination/tests/gtest_coordination_changelog.cpp +++ b/src/Coordination/tests/gtest_coordination_changelog.cpp @@ -30,6 +30,7 @@ namespace FailPoints { extern const char keeper_changelog_read_plan_resolved[]; extern const char keeper_changelog_removed_from_disk_set[]; + extern const char keeper_changelog_preallocate_no_space[]; } namespace ErrorCodes @@ -346,6 +347,36 @@ TEST_P(CoordinationTestWithCompression, ChangelogTestFlushThrottling) EXPECT_GE(watch.elapsedMilliseconds(), 100); } +/// A failed preallocation (e.g. `ENOSPC`) fails the batch, but must leave the writer usable: +/// the next append retries the preallocation. Previously the append completion thread +/// finalized the writer without holding the writer lock, and the next append dereferenced +/// the destroyed file buffer. +TEST_P(CoordinationTestWithCompression, ChangelogTestAppendAfterPreallocationFailure) +{ + ChangelogDirTest test("./logs"); + this->setLogDirectory("./logs"); + + DB::KeeperLogStore changelog( + DB::LogFileSettings{ + .force_sync = true, .compress_logs = this->enable_compression, .rotate_interval = 1000, .max_size = 1024 * 1024}, + DB::FlushSettings(), + DB::ReadAheadSettings{}, + this->keeper_context); + changelog.init(0, 0); + + DB::FailPointInjection::enableFailPoint(DB::FailPoints::keeper_changelog_preallocate_no_space); + + auto entry = getLogEntry("hello world", 77); + changelog.append(entry); + EXPECT_FALSE(changelog.flush()); + + for (size_t i = 0; i < 10; ++i) + { + changelog.append(entry); + EXPECT_TRUE(changelog.flush()); + } +} + TEST_P(CoordinationTestWithCompression, ChangelogTestFile) { ChangelogDirTest test("./logs"); From 9b80c8730866a0f6d6c7b11e5d298a58bda8eda3 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 28 Sep 2026 10:10:38 +0000 Subject: [PATCH 063/185] Backport #121397 to 26.8: Do not materialize a lazily replicated array column in ARRAY JOIN --- src/Interpreters/ArrayJoinAction.cpp | 84 ++++++++++++------- src/Interpreters/ArrayJoinAction.h | 9 +- tests/performance/array_join.xml | 8 ++ ...join_replicated_array_stays_lazy.reference | 25 ++++++ ...array_join_replicated_array_stays_lazy.sql | 30 +++++++ 5 files changed, 127 insertions(+), 29 deletions(-) create mode 100644 tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference create mode 100644 tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql diff --git a/src/Interpreters/ArrayJoinAction.cpp b/src/Interpreters/ArrayJoinAction.cpp index ef87afc35c30..1b5126a77070 100644 --- a/src/Interpreters/ArrayJoinAction.cpp +++ b/src/Interpreters/ArrayJoinAction.cpp @@ -42,6 +42,15 @@ static ColumnPtr getArrayJoinColumn(const ColumnPtr & column) return column; if (const auto * map = typeid_cast(column.get())) return map->getNestedColumnPtr(); + /// Keep replicated arrays lazy, only unwrap maps. + if (const auto * replicated = typeid_cast(column.get())) + { + const auto & nested = replicated->getNestedColumn(); + if (typeid_cast(nested.get())) + return column; + if (const auto * map = typeid_cast(nested.get())) + return ColumnReplicated::create(map->getNestedColumnPtr(), replicated->getIndexesColumn()); + } return nullptr; } @@ -151,11 +160,8 @@ ArrayJoinResultIterator::ArrayJoinResultIterator(const ArrayJoinAction * array_j const auto & function_array_resize = array_join->function_array_resize; const auto & function_builder = array_join->function_builder; - /// TODO: avoid convertToFullColumnIfReplicated - any_array_map_ptr = block.getByName(*columns.begin()).column->convertToFullColumnIfConst()->convertToFullColumnIfReplicated(); - any_array = getArrayJoinColumnRawPtr(any_array_map_ptr); - if (!any_array) - throw Exception(ErrorCodes::TYPE_MISMATCH, "ARRAY JOIN requires array or map argument"); + any_array_map_ptr = block.getByName(*columns.begin()).column->convertToFullColumnIfConst(); + initAnyArray(); if (is_unaligned) { @@ -187,27 +193,60 @@ ArrayJoinResultIterator::ArrayJoinResultIterator(const ArrayJoinAction * array_j any_array_map_ptr = src_col.column->convertToFullColumnIfConst(); } - any_array = getArrayJoinColumnRawPtr(any_array_map_ptr); - if (!any_array) - throw Exception(ErrorCodes::TYPE_MISMATCH, "ARRAY JOIN requires array or map argument"); + initAnyArray(); } else if (is_left) { for (const auto & name : columns) { const auto & src_col = block.getByName(name); - ColumnWithTypeAndName array_col = convertArrayJoinColumn(src_col); + /// emptyArrayToSingle is fine with a replicated input, no need to materialize it. + ColumnWithTypeAndName array_col{getArrayJoinColumn(src_col.column->convertToFullColumnIfConst()), getArrayJoinDataType(src_col.type), src_col.name}; ColumnsWithTypeAndName tmp_block{array_col}; non_empty_array_columns[name] = function_builder->build(tmp_block)->execute(tmp_block, array_col.type, array_col.column->size(), /* dry_run = */ false); } any_array_map_ptr = non_empty_array_columns.begin()->second->convertToFullColumnIfConst(); - any_array = getArrayJoinColumnRawPtr(any_array_map_ptr); - if (!any_array) - throw Exception(ErrorCodes::TYPE_MISMATCH, "ARRAY JOIN requires array or map argument"); + initAnyArray(); } } +void ArrayJoinResultIterator::initAnyArray() +{ + any_array = getArrayJoinColumnRawPtr(any_array_map_ptr); + if (any_array) + return; + + /// Replicated arrays are materialized per window, here we only need the row sizes. + const auto * replicated = typeid_cast(any_array_map_ptr.get()); + const auto * nested_array = replicated ? getArrayJoinColumnRawPtr(replicated->getNestedColumn()) : nullptr; + if (!nested_array) + throw Exception(ErrorCodes::TYPE_MISMATCH, "ARRAY JOIN requires array or map argument"); + + const auto & nested_offsets = nested_array->getOffsets(); + const auto & indexes = replicated->getIndexes(); + replicated_offsets.resize(replicated->size()); + size_t accumulated = 0; + for (size_t row = 0; row < replicated_offsets.size(); ++row) + { + size_t index = indexes.getIndexAt(row); + accumulated += nested_offsets[index] - nested_offsets[index - 1]; + replicated_offsets[row] = accumulated; + } +} + +const IColumn::Offsets & ArrayJoinResultIterator::anyOffsets() const +{ + return any_array ? any_array->getOffsets() : replicated_offsets; +} + +ColumnPtr ArrayJoinResultIterator::cutAnyArray(size_t start, size_t length) const +{ + if (any_array) + return any_array->cut(start, length); + return getArrayJoinColumn(any_array_map_ptr->cut(start, length)->convertToFullColumnIfReplicated()); +} + bool ArrayJoinResultIterator::hasNext() const { return total_rows != 0 && current_row < total_rows; @@ -220,7 +259,7 @@ Block ArrayJoinResultIterator::next() throw Exception(ErrorCodes::LOGICAL_ERROR, "No more elements in ArrayJoinResultIterator."); size_t max_block_size = array_join->max_block_size; - const auto & offsets = any_array->getOffsets(); + const auto & offsets = anyOffsets(); /// Make sure output block rows do not exceed max_block_size. size_t next_row = current_row; @@ -237,7 +276,7 @@ Block ArrayJoinResultIterator::next() const auto & columns = array_join->columns; bool is_unaligned = array_join->is_unaligned; bool is_left = array_join->is_left; - auto cut_any_col = any_array->cut(current_row, next_row - current_row); + auto cut_any_col = cutAnyArray(current_row, next_row - current_row); const auto * cut_any_array = typeid_cast(cut_any_col.get()); ColumnPtr indexes_for_lazy_replication; @@ -258,20 +297,9 @@ Block ArrayJoinResultIterator::next() { if (const auto & type = getArrayJoinDataType(current.type)) { - ColumnPtr array_ptr; - if (typeid_cast(current.type.get())) - { - array_ptr = (is_left && !is_unaligned) ? non_empty_array_columns[current.name]->cut(current_row, next_row - current_row) - : current.column; - array_ptr = array_ptr->convertToFullColumnIfConst()->convertToFullColumnIfReplicated(); - } - else - { - ColumnPtr map_ptr = current.column->convertToFullColumnIfConst()->convertToFullColumnIfReplicated(); - const ColumnMap & map = typeid_cast(*map_ptr); - array_ptr = (is_left && !is_unaligned) ? non_empty_array_columns[current.name]->cut(current_row, next_row - current_row) - : map.getNestedColumnPtr(); - } + ColumnPtr array_ptr = (is_left && !is_unaligned) ? non_empty_array_columns[current.name]->cut(current_row, next_row - current_row) + : getArrayJoinColumn(current.column->convertToFullColumnIfConst()->convertToFullColumnIfReplicated()); + array_ptr = array_ptr->convertToFullColumnIfConst()->convertToFullColumnIfReplicated(); const ColumnArray & array = typeid_cast(*array_ptr); if (!is_unaligned && !array.hasEqualOffsets(*cut_any_array)) diff --git a/src/Interpreters/ArrayJoinAction.h b/src/Interpreters/ArrayJoinAction.h index aac5aac7b9db..141c9e440d5b 100644 --- a/src/Interpreters/ArrayJoinAction.h +++ b/src/Interpreters/ArrayJoinAction.h @@ -2,6 +2,7 @@ #include #include +#include #include @@ -58,12 +59,18 @@ class ArrayJoinResultIterator bool hasNext() const; private: + void initAnyArray(); + const PaddedPODArray & anyOffsets() const; + ColumnPtr cutAnyArray(size_t start, size_t length) const; + const ArrayJoinAction * array_join; Block block; bool enable_lazy_columns_replication; ColumnPtr any_array_map_ptr; - const ColumnArray * any_array; + /// Null if the joined column is replicated, then replicated_offsets is used instead. + const ColumnArray * any_array = nullptr; + PaddedPODArray replicated_offsets; /// If LEFT ARRAY JOIN, then we create columns in which empty arrays are replaced by arrays with one element - the default value. std::map non_empty_array_columns; diff --git a/tests/performance/array_join.xml b/tests/performance/array_join.xml index b3fc24d31665..af586967907b 100644 --- a/tests/performance/array_join.xml +++ b/tests/performance/array_join.xml @@ -1,4 +1,6 @@ + CREATE TABLE array_join_wide_chain (A Array(String), B Array(String), C Array(String)) ENGINE = MergeTree ORDER BY tuple() + INSERT INTO array_join_wide_chain SELECT arrayMap(i -> concat('a', toString(number), '_', toString(i)), range(100)), arrayMap(i -> concat('b', toString(number), '_', toString(i)), range(100)), arrayMap(i -> concat('c', toString(number), '_', toString(i)), range(100)) FROM numbers(100000) @@ -10,4 +12,10 @@ SELECT count() FROM (SELECT [number] a, [number * 2, number] b FROM numbers(10000000)) AS t LEFT ARRAY JOIN a, b WHERE NOT ignore(a + b) SETTINGS enable_unaligned_array_join = 1 with 'clickhouse' as str select arrayJoin(range(number % 10)), materialize(str) from numbers(10000000) + + + SELECT a, b, c FROM array_join_wide_chain ARRAY JOIN A AS a ARRAY JOIN B AS b ARRAY JOIN C AS c LIMIT 31 SETTINGS max_threads = 1 FORMAT Null + SELECT a, b, c FROM array_join_wide_chain LEFT ARRAY JOIN A AS a LEFT ARRAY JOIN B AS b LEFT ARRAY JOIN C AS c LIMIT 31 SETTINGS max_threads = 1 FORMAT Null + + DROP TABLE IF EXISTS array_join_wide_chain diff --git a/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference new file mode 100644 index 000000000000..6072b25b7968 --- /dev/null +++ b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference @@ -0,0 +1,25 @@ +1 a1_0 0 ('k',1) +1 a1_0 0 ('kk',2) +2 a2_0 0 ('k',2) +2 a2_0 0 ('kk',4) +2 a2_0 1 ('k',2) +2 a2_0 1 ('kk',4) +2 a2_1 0 ('k',2) +2 a2_1 0 ('kk',4) +2 a2_1 1 ('k',2) +2 a2_1 1 ('kk',4) +5 a5_0 0 ('k',5) +5 a5_0 0 ('kk',10) +5 a5_1 0 ('k',5) +5 a5_1 0 ('kk',10) +14 7208362719481167226 +14 7208362719481167226 +28 17533497658690827015 +28 17533497658690827015 +7 17387321633734044581 +7 17387321633734044581 +5 11221959250764763722 +5 11221959250764763722 +10 +10 +10 diff --git a/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql new file mode 100644 index 000000000000..6155f4f853ef --- /dev/null +++ b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql @@ -0,0 +1,30 @@ +-- The second and third joins get their arrays lazily replicated from the previous one. +SET enable_lazy_columns_replication = 1; + +DROP TABLE IF EXISTS t_lazy_arrays; +CREATE TABLE t_lazy_arrays (id UInt32, a Array(String), b Array(UInt32), c Array(String), m Map(String, UInt32)) ENGINE = MergeTree ORDER BY id; +INSERT INTO t_lazy_arrays SELECT number, arrayMap(i -> concat('a', toString(number), '_', toString(i)), range(number % 3)), range(number % 4), arrayMap(i -> concat('c', toString(i)), range(number % 4)), map('k', number, 'kk', number * 2) FROM numbers(7); + +-- Results must not depend on lazy replication. +SELECT id, x, y, kv FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y ARRAY JOIN m AS kv ORDER BY ALL SETTINGS max_block_size = 3; +SELECT count(), sum(cityHash64(id, x, y, kv)) FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y ARRAY JOIN m AS kv SETTINGS max_block_size = 3; +SELECT count(), sum(cityHash64(id, x, y, kv)) FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y ARRAY JOIN m AS kv SETTINGS max_block_size = 3, enable_lazy_columns_replication = 0; +SELECT count(), sum(cityHash64(id, x, y, kv)) FROM t_lazy_arrays LEFT ARRAY JOIN a AS x LEFT ARRAY JOIN b AS y LEFT ARRAY JOIN m AS kv SETTINGS max_block_size = 3; +SELECT count(), sum(cityHash64(id, x, y, kv)) FROM t_lazy_arrays LEFT ARRAY JOIN a AS x LEFT ARRAY JOIN b AS y LEFT ARRAY JOIN m AS kv SETTINGS max_block_size = 3, enable_lazy_columns_replication = 0; +SELECT count(), sum(cityHash64(id, x, y, z)) FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y, c AS z SETTINGS max_block_size = 3; +SELECT count(), sum(cityHash64(id, x, y, z)) FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y, c AS z SETTINGS max_block_size = 3, enable_lazy_columns_replication = 0; +SELECT count(), sum(cityHash64(id, x, y)) FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y WHERE y % 2 = 0 SETTINGS max_block_size = 3; +SELECT count(), sum(cityHash64(id, x, y)) FROM t_lazy_arrays ARRAY JOIN a AS x ARRAY JOIN b AS y WHERE y % 2 = 0 SETTINGS max_block_size = 3, enable_lazy_columns_replication = 0; + +DROP TABLE t_lazy_arrays; + +DROP TABLE IF EXISTS t_lazy_arrays_wide; +CREATE TABLE t_lazy_arrays_wide (id UInt32, a Array(String), b Array(String), c Array(String)) ENGINE = MergeTree ORDER BY id; +INSERT INTO t_lazy_arrays_wide SELECT number, arrayMap(i -> concat('a', toString(number), '_', toString(i), 'xxxxxxxxxx'), range(100)), arrayMap(i -> concat('b', toString(number), '_', toString(i), 'xxxxxxxxxx'), range(100)), arrayMap(i -> concat('c', toString(number), '_', toString(i), 'xxxxxxxxxx'), range(100)) FROM numbers(1000); + +-- Used to materialize the whole block of arrays in every join, way above this limit. +SELECT count() FROM (SELECT x, y, z FROM t_lazy_arrays_wide ARRAY JOIN a AS x ARRAY JOIN b AS y ARRAY JOIN c AS z LIMIT 10) SETTINGS max_block_size = 65409, max_threads = 1, enable_lazy_columns_replication = 1, max_memory_usage = 100000000; +SELECT count() FROM (SELECT x, y, z FROM t_lazy_arrays_wide LEFT ARRAY JOIN a AS x LEFT ARRAY JOIN b AS y LEFT ARRAY JOIN c AS z LIMIT 10) SETTINGS max_block_size = 65409, max_threads = 1, enable_lazy_columns_replication = 1, max_memory_usage = 100000000; +SELECT count() FROM (SELECT arrayJoin(a) AS x, arrayJoin(b) AS y, arrayJoin(c) AS z FROM t_lazy_arrays_wide WHERE id >= 0 LIMIT 10) SETTINGS max_block_size = 65409, max_threads = 1, enable_lazy_columns_replication = 1, max_memory_usage = 100000000, query_plan_lower_array_join_function = 1; + +DROP TABLE t_lazy_arrays_wide; From afada3c3f66c41148663c05f7347fe63278ccdf0 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 28 Sep 2026 10:43:53 +0000 Subject: [PATCH 064/185] Backport #119097 to 26.8: Bump `zxc` from 0.13.1 to 0.14.0 --- .gitmodules | 2 +- contrib/zxc | 2 +- contrib/zxc-cmake/CMakeLists.txt | 38 +++++++++++++++++++------------- 3 files changed, 25 insertions(+), 17 deletions(-) diff --git a/.gitmodules b/.gitmodules index 2f3dd2930a0a..db76496d84cb 100644 --- a/.gitmodules +++ b/.gitmodules @@ -416,7 +416,7 @@ url = https://github.com/ClickHouse/geometry.hpp [submodule "contrib/zxc"] path = contrib/zxc - url = https://github.com/ClickHouse/zxc + url = https://github.com/hellobertrand/zxc [submodule "contrib/libucontext"] path = contrib/libucontext url = https://github.com/kaniini/libucontext diff --git a/contrib/zxc b/contrib/zxc index b9890cfe3466..e568d1967309 160000 --- a/contrib/zxc +++ b/contrib/zxc @@ -1 +1 @@ -Subproject commit b9890cfe3466b19d0eb5005874ce81728efb9565 +Subproject commit e568d1967309fd5d160f45f2fdabcbff6ab3f058 diff --git a/contrib/zxc-cmake/CMakeLists.txt b/contrib/zxc-cmake/CMakeLists.txt index 98c52cb1111b..70451492f30d 100644 --- a/contrib/zxc-cmake/CMakeLists.txt +++ b/contrib/zxc-cmake/CMakeLists.txt @@ -33,9 +33,14 @@ set(ZXC_CORE_SOURCES # emitted symbols (e.g. zxc_compress_chunk_wrapper_avx2). zxc_dispatch.c declares # these suffixed symbols unconditionally per architecture, so the exact set of # variants below must match what the dispatcher expects for each target: -# x86-64 -> _default, _sse2, _avx2, _avx512 -# aarch64 -> _default, _neon +# x86-64 -> _default, _avx2, _avx512 +# aarch64 -> _default only (NEON is the AArch64 baseline) # other -> _default only +# SSE2 is the x86-64 baseline and NEON the AArch64 one, so `_default` already +# compiles those code paths; zxc dropped the dedicated `_sse2` and `_neon` +# variants in v0.14.0 and the dispatcher no longer references them. The +# remaining `_neon32` variant is for 32-bit ARM, which ClickHouse does not +# target. set(ZXC_VARIANT_OBJECTS "") macro(zxc_add_variant suffix) @@ -54,11 +59,14 @@ macro(zxc_add_variant suffix) endmacro() if (OS_DARWIN AND ARCH_AMD64) - # zxc's x86-64 runtime dispatch calls `__builtin_cpu_init` / + # Historically zxc's x86-64 runtime dispatch called `__builtin_cpu_init` / # `__builtin_cpu_supports`, which reference compiler-rt's `__cpu_model`. That - # symbol is not linked into ClickHouse's macOS cross-build, so linking fails - # with "undefined ___cpu_model". macOS builds are for local development only, - # so restrict zxc to its portable scalar core there. + # symbol is not linked into ClickHouse's macOS cross-build, so linking failed + # with "undefined ___cpu_model". Since v0.14.0 the dispatcher issues `CPUID` + # and `XGETBV` directly and no longer needs `__cpu_model`, so this is now a + # conservative restriction rather than a required one: macOS builds are for + # local development only and are never test-run in CI, so there is nothing + # to gain from the x86 SIMD variants there. Lifting it is a separate change. set(ZXC_FORCE_SCALAR TRUE) elseif (SANITIZE STREQUAL "memory") # clang's MemorySanitizer has no precise shadow model for the x86 byte-shuffle @@ -67,7 +75,7 @@ elseif (SANITIZE STREQUAL "memory") # that the shuffle discards still poisons the logical result. zxc's SIMD # kernels deliberately do speculative 16/32-byte loads whose tail lanes fall # into malloc'ed scratch buffers or not-yet-written output and are then - # discarded by such shuffles (e.g. `zxc_decode_copy_overlap_run`, + # discarded by such shuffles (e.g. `zxc_decode_copy_overlap_run32`, # `zxc_pivco_merge`), which triggers false `use-of-uninitialized-value` # reports. The scalar core branches only on initialized bytes, so build only # that under MSan. (AArch64 NEON `tbl` gets a precise shadow, but keep all @@ -90,16 +98,16 @@ if (ZXC_FORCE_SCALAR) target_compile_definitions(${_tgt} PRIVATE ZXC_DISABLE_SIMD) endforeach() elseif (ARCH_AMD64) - # SSE2/AVX2 flags are redundant with the x86-64-v3 baseline but harmless; + # The AVX2 flags are redundant with the x86-64-v3 baseline but harmless; # AVX512 is NOT in the baseline, so those flags are required for that variant - # (only ever entered at runtime on AVX-512 capable CPUs). - zxc_add_variant(_sse2 -msse2) - zxc_add_variant(_avx2 -mavx2 -mfma -mbmi -mbmi2 -mlzcnt) + # (only ever entered at runtime on AVX-512 capable CPUs). The flag sets must + # stay in sync with `zxc_detect_cpu_features`, which admits a CPU to the AVX2 + # and AVX-512 tiers only after proving BMI1, BMI2 and LZCNT as well. Note + # that it does not probe FMA, so `-mfma` must not be added here: the + # compiler would be free to emit FMA instructions into a variant that runs + # on any AVX2 CPU. + zxc_add_variant(_avx2 -mavx2 -mbmi -mbmi2 -mlzcnt) zxc_add_variant(_avx512 -mavx512f -mavx512bw -mavx512vbmi -mavx512vbmi2 -mbmi -mbmi2 -mlzcnt) -elseif (ARCH_AARCH64) - # NEON is guaranteed by ClickHouse's armv8.2-a+simd baseline; no extra flags - # (adding -march=armv8-a would downgrade below the baseline). - zxc_add_variant(_neon) endif () add_library(_zxc ${ZXC_CORE_SOURCES} ${ZXC_VARIANT_OBJECTS}) From b2dd7951beafc18f4596a4e2695793e25e1c4cfc Mon Sep 17 00:00:00 2001 From: Pedro Ferreira Date: Mon, 28 Sep 2026 12:54:20 +0000 Subject: [PATCH 065/185] Drop the query needing a master-only setting from 05234 `query_plan_lower_array_join_function` does not exist on 26.8, so the last query of `05234_array_join_replicated_array_stays_lazy` fails with UNKNOWN_SETTING. Without the setting the `arrayJoin` function form is not lowered to ARRAY JOIN, so the query would not exercise this change anyway. This was meant to be part of the cherry-pick resolution and is already in the 26.7, 26.3 and the three private ones; on this branch alone the edit was left unstaged, so the commit kept the auto-merged test. Both files are taken verbatim from the 26.7 cherry-pick (03c6b226231). https://s3.amazonaws.com/clickhouse-test-reports/json.html?PR=122650&sha=9b80c8730866a0f6d6c7b11e5d298a58bda8eda3&name_0=BackportPR&name_1=Stateless%20tests%20%28amd_asan_ubsan%2C%20distributed%20plan%2C%20parallel%29 Related: https://github.com/ClickHouse/ClickHouse/pull/122650 Co-Authored-By: Claude Opus 5 (1M context) --- .../05234_array_join_replicated_array_stays_lazy.reference | 1 - .../0_stateless/05234_array_join_replicated_array_stays_lazy.sql | 1 - 2 files changed, 2 deletions(-) diff --git a/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference index 6072b25b7968..8ed3d5998023 100644 --- a/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference +++ b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.reference @@ -22,4 +22,3 @@ 5 11221959250764763722 10 10 -10 diff --git a/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql index 6155f4f853ef..d733b1857521 100644 --- a/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql +++ b/tests/queries/0_stateless/05234_array_join_replicated_array_stays_lazy.sql @@ -25,6 +25,5 @@ INSERT INTO t_lazy_arrays_wide SELECT number, arrayMap(i -> concat('a', toString -- Used to materialize the whole block of arrays in every join, way above this limit. SELECT count() FROM (SELECT x, y, z FROM t_lazy_arrays_wide ARRAY JOIN a AS x ARRAY JOIN b AS y ARRAY JOIN c AS z LIMIT 10) SETTINGS max_block_size = 65409, max_threads = 1, enable_lazy_columns_replication = 1, max_memory_usage = 100000000; SELECT count() FROM (SELECT x, y, z FROM t_lazy_arrays_wide LEFT ARRAY JOIN a AS x LEFT ARRAY JOIN b AS y LEFT ARRAY JOIN c AS z LIMIT 10) SETTINGS max_block_size = 65409, max_threads = 1, enable_lazy_columns_replication = 1, max_memory_usage = 100000000; -SELECT count() FROM (SELECT arrayJoin(a) AS x, arrayJoin(b) AS y, arrayJoin(c) AS z FROM t_lazy_arrays_wide WHERE id >= 0 LIMIT 10) SETTINGS max_block_size = 65409, max_threads = 1, enable_lazy_columns_replication = 1, max_memory_usage = 100000000, query_plan_lower_array_join_function = 1; DROP TABLE t_lazy_arrays_wide; From 452cc3d8b5d6ead3f6c803cc10963b5bf560b2a5 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 28 Sep 2026 16:10:33 +0000 Subject: [PATCH 066/185] Update autogenerated version to 26.8.14.3 and contributors --- cmake/autogenerated_versions.txt | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cmake/autogenerated_versions.txt b/cmake/autogenerated_versions.txt index af630df25c06..4e3df68592ef 100644 --- a/cmake/autogenerated_versions.txt +++ b/cmake/autogenerated_versions.txt @@ -2,11 +2,11 @@ # NOTE: VERSION_REVISION has nothing common with DBMS_TCP_PROTOCOL_VERSION, # only DBMS_TCP_PROTOCOL_VERSION should be incremented on protocol changes. -SET(VERSION_REVISION 54526) +SET(VERSION_REVISION 54527) SET(VERSION_MAJOR 26) SET(VERSION_MINOR 8) -SET(VERSION_PATCH 14) -SET(VERSION_GITHASH 3eac80eef9b1f81e88b581a6c107c59f21d9e15c) -SET(VERSION_DESCRIBE v26.8.14.1-lts) -SET(VERSION_STRING 26.8.14.1) +SET(VERSION_PATCH 15) +SET(VERSION_GITHASH f1d4a0d36450556c9e3c4f8b3244770b20345516) +SET(VERSION_DESCRIBE v26.8.15.1-lts) +SET(VERSION_STRING 26.8.15.1) # end of autochange From 0993565cf3b13c885892331beafdaad019d25d3a Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 28 Sep 2026 17:55:11 +0000 Subject: [PATCH 067/185] Backport #122035 to 26.8: Disable statistics pruning in case of row level policy --- .../MergeTree/MergeTreeDataSelectExecutor.cpp | 10 +++- ...atistics_part_pruning_row_policy.reference | 16 ++++++ ...255_statistics_part_pruning_row_policy.sql | 56 +++++++++++++++++++ 3 files changed, 81 insertions(+), 1 deletion(-) create mode 100644 tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.reference create mode 100644 tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.sql diff --git a/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp b/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp index 30930da722f4..03feebb56429 100644 --- a/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp +++ b/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp @@ -21,6 +21,7 @@ #include #include #include +#include #include #include #include @@ -821,10 +822,17 @@ RangesInDataParts MergeTreeDataSelectExecutor::filterPartsByStatistics( /// 3. There are on-the-fly mutations or patch parts (statistics only reflects original data) /// 4. A masking policy applies: it rewrites values at read time, so the statistics (like /// the on-the-fly mutations above) no longer describe the values the query sees. + /// 5. A row policy applies: the statistics describe all rows of a part, including the ones the + /// policy hides, so the number of rows left to read after pruning by the query's predicate, + /// which is reported to the client, reveals the values of the hidden rows. The policy is + /// either pushed into this read (possibly from a wrapper such as `Alias`) or belongs to + /// this table and is applied above the read (e.g. for a child of `Merge`). if (!settings[Setting::use_statistics_for_part_pruning] || query_info.isFinal() || (mutations_snapshot && (mutations_snapshot->hasDataMutations() || mutations_snapshot->hasPatchParts())) - || (!parts.empty() && parts.front().data_part->storage.hasEnabledMaskingPolicies(context))) + || (!parts.empty() && parts.front().data_part->storage.hasEnabledMaskingPolicies(context)) + || query_info.row_level_filter + || (!parts.empty() && getEffectiveRowPolicyFilter(parts.front().data_part->storage, context))) { return parts; } diff --git a/tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.reference b/tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.reference new file mode 100644 index 000000000000..39b64abfe204 --- /dev/null +++ b/tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.reference @@ -0,0 +1,16 @@ +1 +0 +0 +0 +0 +0 +0 +0 +row_policy_probe_1_no_policy_1998 1000 +row_policy_probe_2_no_policy_1999 0 +row_policy_probe_3_table_policy_1998 1000 +row_policy_probe_4_table_policy_1999 1000 +row_policy_probe_5_merge_child_policy_1998 1000 +row_policy_probe_6_merge_child_policy_1999 1000 +row_policy_probe_7_alias_policy_1998 1000 +row_policy_probe_8_alias_policy_1999 1000 diff --git a/tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.sql b/tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.sql new file mode 100644 index 000000000000..da69c5c4a485 --- /dev/null +++ b/tests/queries/0_stateless/05255_statistics_part_pruning_row_policy.sql @@ -0,0 +1,56 @@ +-- Statistics describe all rows of a part, including the ones a row policy hides. Pruning parts by them +-- would make the number of rows read an oracle over the values of the hidden rows, so statistics +-- pruning is disabled when a row policy applies, whether the policy belongs to the table, to a child +-- of `Merge`, or to a wrapper such as `Alias`. + +DROP ROW POLICY IF EXISTS payroll_policy ON payroll; +DROP ROW POLICY IF EXISTS payroll_alias_policy ON payroll_alias; +DROP TABLE IF EXISTS payroll_merge; +DROP TABLE IF EXISTS payroll_alias; +DROP TABLE IF EXISTS payroll; + +SET use_statistics_for_part_pruning = 1; +SET materialize_statistics_on_insert = 1; +SET use_query_condition_cache = 0; +SET enable_parallel_replicas = 0; + +CREATE TABLE payroll (id UInt64, dept String, salary UInt64) +ENGINE = MergeTree ORDER BY id +SETTINGS auto_statistics_types = 'basic'; + +-- Visible rows have `salary` below 150, hidden rows have `salary` up to 1999. +INSERT INTO payroll SELECT number, if(number % 2 = 0, 'public', 'exec'), if(number % 2 = 0, 100 + number % 50, 1000 + number) FROM numbers(1000); + +CREATE TABLE payroll_merge (id UInt64, dept String, salary UInt64) ENGINE = Merge(currentDatabase(), '^payroll$'); +CREATE TABLE payroll_alias ENGINE = Alias('payroll'); + +-- Without a row policy the part is pruned when the predicate is out of the range of the statistics. +SELECT count() FROM payroll WHERE salary > 1998 SETTINGS log_comment = 'row_policy_probe_1_no_policy_1998'; +SELECT count() FROM payroll WHERE salary > 1999 SETTINGS log_comment = 'row_policy_probe_2_no_policy_1999'; + +CREATE ROW POLICY payroll_policy ON payroll FOR SELECT USING dept = 'public' TO CURRENT_USER; + +SELECT count() FROM payroll WHERE salary > 1998 SETTINGS log_comment = 'row_policy_probe_3_table_policy_1998'; +SELECT count() FROM payroll WHERE salary > 1999 SETTINGS log_comment = 'row_policy_probe_4_table_policy_1999'; + +SELECT count() FROM payroll_merge WHERE salary > 1998 SETTINGS log_comment = 'row_policy_probe_5_merge_child_policy_1998'; +SELECT count() FROM payroll_merge WHERE salary > 1999 SETTINGS log_comment = 'row_policy_probe_6_merge_child_policy_1999'; + +DROP ROW POLICY payroll_policy ON payroll; +CREATE ROW POLICY payroll_alias_policy ON payroll_alias FOR SELECT USING dept = 'public' TO CURRENT_USER; + +SELECT count() FROM payroll_alias WHERE salary > 1998 SETTINGS log_comment = 'row_policy_probe_7_alias_policy_1998', enable_analyzer = 1; +SELECT count() FROM payroll_alias WHERE salary > 1999 SETTINGS log_comment = 'row_policy_probe_8_alias_policy_1999', enable_analyzer = 1; + +SYSTEM FLUSH LOGS query_log; + +-- The number of rows read must not depend on the values of the hidden rows. +SELECT log_comment, read_rows +FROM system.query_log +WHERE current_database = currentDatabase() AND type = 'QueryFinish' AND log_comment LIKE 'row\_policy\_probe\_%' +ORDER BY log_comment; + +DROP ROW POLICY payroll_alias_policy ON payroll_alias; +DROP TABLE payroll_merge; +DROP TABLE payroll_alias; +DROP TABLE payroll; From 5e64e9f373428efcc02c4e6c5d033a38530ddff2 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Mon, 28 Sep 2026 19:17:13 +0000 Subject: [PATCH 068/185] Backport #122668 to 26.8: Link with `--icf=safe`: `--icf=all` merges address-taken functions and crashes gRPC compression --- CMakeLists.txt | 5 ++-- ...05291_grpc_transport_compression.reference | 1 + .../05291_grpc_transport_compression.sh | 28 +++++++++++++++++++ 3 files changed, 32 insertions(+), 2 deletions(-) create mode 100644 tests/queries/0_stateless/05291_grpc_transport_compression.reference create mode 100755 tests/queries/0_stateless/05291_grpc_transport_compression.sh diff --git a/CMakeLists.txt b/CMakeLists.txt index 7860b6cd2ed8..68b194ad629c 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -173,7 +173,8 @@ if (OS_LINUX) if (LINKER_NAME MATCHES "lld" AND NOT SANITIZE AND NOT SANITIZE_COVERAGE AND NOT WITH_COVERAGE) # Fold identical code and data sections to reduce binary size. - set (CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -Wl,--icf=all") + # `safe` keeps address-taken functions distinct, as C and C++ require. + set (CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -Wl,--icf=safe") endif () # In Release builds, strip residual debug sections that may come from contrib @@ -345,7 +346,7 @@ if (NOT SANITIZE AND NOT SANITIZE_COVERAGE AND NOT WITH_COVERAGE) set(COMPILER_FLAGS "${COMPILER_FLAGS} -ffunction-sections -fdata-sections") # -faddrsig emits an address-significance table that lets lld's ICF # distinguish address-taken symbols from non-address-taken ones. - # Only useful on Linux where we use lld with --icf=all; Darwin's ld64 + # Only useful on Linux where we use lld with --icf=safe; Darwin's ld64 # does not consume these sections. if (OS_LINUX) set(COMPILER_FLAGS "${COMPILER_FLAGS} -faddrsig") diff --git a/tests/queries/0_stateless/05291_grpc_transport_compression.reference b/tests/queries/0_stateless/05291_grpc_transport_compression.reference new file mode 100644 index 000000000000..08839f6bb296 --- /dev/null +++ b/tests/queries/0_stateless/05291_grpc_transport_compression.reference @@ -0,0 +1 @@ +200 diff --git a/tests/queries/0_stateless/05291_grpc_transport_compression.sh b/tests/queries/0_stateless/05291_grpc_transport_compression.sh new file mode 100755 index 000000000000..b42c789d9bd1 --- /dev/null +++ b/tests/queries/0_stateless/05291_grpc_transport_compression.sh @@ -0,0 +1,28 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: In fasttest, ENABLE_LIBRARIES=0, so the grpc library is not built + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +# gRPC queries with gzip and deflate compression of both the request and the result. +# gRPC sends a message uncompressed if compression does not shrink it, so the query carries a long comment. +python3 - "$CURDIR/../../../utils/grpc-client/pb2" <<'EOF' +import sys +sys.path.insert(0, sys.argv[1]) +import grpc +import clickhouse_grpc_pb2 +import clickhouse_grpc_pb2_grpc + +padding = " -- " + "x" * 1000 +ok = 0 +for algorithm, compression in (("gzip", grpc.Compression.Gzip), ("deflate", grpc.Compression.Deflate)): + with grpc.insecure_channel("localhost:9100", compression=compression) as channel: + stub = clickhouse_grpc_pb2_grpc.ClickHouseStub(channel) + for i in range(100): + result = stub.ExecuteQuery(clickhouse_grpc_pb2.QueryInfo( + query=f"SELECT {i}{padding}", transport_compression_type=algorithm, transport_compression_level=3)) + ok += result.output == f"{i}\n".encode() and not result.HasField("exception") +print(ok) +EOF From a5aa9b7e9c17c1185aea17f3d7c23cfb9d08d31a Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 06:10:04 +0000 Subject: [PATCH 069/185] Backport #122451 to 26.8: Fix array and interval `plus`/`minus` producing a constant inside an `Array` or a `Tuple` --- src/Functions/FunctionBinaryArithmetic.h | 3 ++- src/Functions/vectorFunctions.cpp | 12 ++++++----- ...nstant_element_in_array_or_tuple.reference | 20 +++++++++++++++++++ ...tic_constant_element_in_array_or_tuple.sql | 15 ++++++++++++++ 4 files changed, 44 insertions(+), 6 deletions(-) create mode 100644 tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.reference create mode 100644 tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.sql diff --git a/src/Functions/FunctionBinaryArithmetic.h b/src/Functions/FunctionBinaryArithmetic.h index 96adfe2a3e82..206717bab661 100644 --- a/src/Functions/FunctionBinaryArithmetic.h +++ b/src/Functions/FunctionBinaryArithmetic.h @@ -1867,7 +1867,8 @@ class FunctionBinaryArithmetic : public IFunction, WithContext ? array_element_function->executeImpl(new_arguments, result_array_type, rows_count) : executeImpl(new_arguments, result_array_type, rows_count); - return ColumnArray::create(res, typeid_cast(arguments[0].column.get())->getOffsetsPtr()); + /// The element-wise result can be a constant (for example a NULL), the data of an array cannot. + return ColumnArray::create(res->convertToFullColumnIfConst(), typeid_cast(arguments[0].column.get())->getOffsetsPtr()); } ColumnPtr executeArrayWithNumericImpl(const ColumnsWithTypeAndName & args, const DataTypePtr & result_type, size_t input_rows_count) const diff --git a/src/Functions/vectorFunctions.cpp b/src/Functions/vectorFunctions.cpp index 1df1d5757abc..ff44f7aea860 100644 --- a/src/Functions/vectorFunctions.cpp +++ b/src/Functions/vectorFunctions.cpp @@ -630,7 +630,7 @@ struct FunctionTupleOperationInterval final : public ITupleFunction else tuple_columns.resize(2); - tuple_columns[0] = arguments[0].column->convertToFullColumnIfConst(); + tuple_columns[0] = arguments[0].column; } else if (first_tuple) { @@ -675,15 +675,13 @@ struct FunctionTupleOperationInterval final : public ITupleFunction { auto minus = FunctionFactory::instance().get("minus", context); auto elem_minus = minus->build({left, arguments[1]}); - last_column = elem_minus->execute({left, arguments[1]}, arguments[1].type, input_rows_count, /* dry_run = */ false) - ->convertToFullColumnIfConst(); + last_column = elem_minus->execute({left, arguments[1]}, arguments[1].type, input_rows_count, /* dry_run = */ false); } else { auto plus = FunctionFactory::instance().get("plus", context); auto elem_plus = plus->build({left, arguments[1]}); - last_column = elem_plus->execute({left, arguments[1]}, arguments[1].type, input_rows_count, /* dry_run = */ false) - ->convertToFullColumnIfConst(); + last_column = elem_plus->execute({left, arguments[1]}, arguments[1].type, input_rows_count, /* dry_run = */ false); } } else @@ -700,6 +698,10 @@ struct FunctionTupleOperationInterval final : public ITupleFunction } } + /// Either operand can be a constant, which a tuple cannot hold. + for (auto & column : tuple_columns) + column = column->convertToFullColumnIfConst(); + return ColumnTuple::create(tuple_columns); } }; diff --git a/tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.reference b/tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.reference new file mode 100644 index 000000000000..acea8230f172 --- /dev/null +++ b/tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.reference @@ -0,0 +1,20 @@ +[NULL] +[NULL] [null] +[NULL] [null] +[NULL,NULL] [] +[NULL,NULL] [] +[NULL] [NULL] +[NULL] [NULL] +[[NULL]] +[[NULL]] +[NULL] +[NULL] +[NULL] + +[] +(0,1) (0,-1) +(1,1) (1,-1) +(1,1) (1,1,0) +(1,2) (1,1,1) +2020-01-01 01:00:00 +2020-01-02 01:00:00 diff --git a/tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.sql b/tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.sql new file mode 100644 index 000000000000..28fa181272a7 --- /dev/null +++ b/tests/queries/0_stateless/05258_arithmetic_constant_element_in_array_or_tuple.sql @@ -0,0 +1,15 @@ +-- Arithmetic must not produce a constant inside an array or a tuple: an array operation whose element-wise +-- result is a constant (a date plus a tuple holding NULL is NULL), and a constant interval added to a +-- non-constant interval of another unit. + +SELECT toString([toDateTime64(1, 3)] + [(2, NULL)]); +SELECT toString([toDateTime64(number, 3)] + [(2, NULL)]), toJSONString([toDateTime64(number, 3)] + [(2, NULL)]) FROM numbers(2); +SELECT arrayPushBack([toDateTime64(number, 3)] + [(2, NULL)], NULL), arrayDistinct([toDateTime64(number, 3)] + [(2, NULL)]) FROM numbers(2); +SELECT toString([toDate('2020-01-01') + number] - [(1, NULL)]), toString([(1, NULL)] + [toDateTime(number)]) FROM numbers(2); +SELECT toString([[toDateTime64(number, 3)]] + [[(2, NULL)]]) FROM numbers(2); +SELECT [toDateTime64(number, 3)] + [(2, NULL)] FROM numbers(1) UNION ALL SELECT materialize([NULL]); +SELECT arrayPartialShuffle([toDateTime64(1, 3)] + [(2, NULL)]) AS a GROUP BY a WITH TOTALS; + +SELECT toIntervalDay(number) + INTERVAL 1 HOUR, toIntervalDay(number) - INTERVAL 1 HOUR FROM numbers(2); +SELECT (INTERVAL 1 DAY, INTERVAL 1 HOUR) + toIntervalHour(number), (INTERVAL 1 DAY, INTERVAL 1 MONTH) + toIntervalHour(number) FROM numbers(2); +SELECT toDateTime('2020-01-01 00:00:00', 'UTC') + (toIntervalDay(number) + INTERVAL 1 HOUR) FROM numbers(2); From 5354187eea442a65ea90e6b8ae8e00d6d3e1543f Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 07:11:31 +0000 Subject: [PATCH 070/185] Backport #114629 to 26.8: Fix DPsub join reordering silently dropping single-table ON-clause filters --- .../QueryPlan/Optimizations/joinOrder.cpp | 17 ++-- ...ub_single_table_filter_placement.reference | 12 +++ ...69_dpsub_single_table_filter_placement.sql | 86 +++++++++++++++++++ 3 files changed, 107 insertions(+), 8 deletions(-) create mode 100644 tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.reference create mode 100644 tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.sql diff --git a/src/Processors/QueryPlan/Optimizations/joinOrder.cpp b/src/Processors/QueryPlan/Optimizations/joinOrder.cpp index a7f654ac37ed..0083567295b9 100644 --- a/src/Processors/QueryPlan/Optimizations/joinOrder.cpp +++ b/src/Processors/QueryPlan/Optimizations/joinOrder.cpp @@ -842,9 +842,10 @@ const std::vector & JoinOrderOptimizer::collectJoinEdgesMask(UI if (dpsub_data.edge_pinned[i] && (dpsub_data.edge_pin_mask[i] & ~joined)) continue; - /// Relations that must all be present before the predicate is applicable: the relations it - /// references (`sources`) plus any relations it is pinned to. For a plain equi-predicate the - /// pin is empty, so this is just `sources`. For a single-table conjunct of an outer join's ON + /// Works much like Extended Eligibility List (EEL) in case of outerjoins: + /// encoding relations that must be present for the predicate to be applicable (in `pin` mask) + /// For innerjoins its just the sources of the predicate, i.e., NEL, here pin is empty. + /// For a single-table conjunct of an outer join's ON /// clause (e.g. `t2.value = 'x'` in `... LEFT JOIN t3 ON t2.id = t3.id AND t2.value = 'x'`), /// `sources` is only `{t2}` but the pin is `{t3}`: the predicate belongs to the ON condition of /// the join that brings in `t3`, not to `t2` as a base-table filter. Placing it by `sources` @@ -856,17 +857,17 @@ const std::vector & JoinOrderOptimizer::collectJoinEdgesMask(UI if (std::popcount(applicable) <= 1) { - /// Base-relation filter or constant predicate: it becomes applicable as soon as its single - /// relation is present, so attach it at the earliest (two-relation) join to filter as low - /// as possible. - if (two_relations && (edge.fromLeft() || edge.fromRight() || edge.fromNone())) + /// Base-relation filter or constant predicate (the edge references at most one relation). + const bool relation_introduced = applicable != 0 && (left_mask == applicable || right_mask == applicable); + const bool constant_at_earliest_join = applicable == 0 && two_relations; + if (relation_introduced || constant_at_earliest_join) out.push_back(&edge); } else if ((applicable & ~left_mask) && (applicable & ~right_mask)) { /// The predicate spans the split (a connecting equi-predicate, or a single-table ON-clause /// conjunct pinned to the opposite side): neither side alone contains all the relations it - /// needs. This join is the lowest one that makes it applicable, so attach it here — into the + /// needs. This join is the lowest one that makes it applicable, so attach it here: into the /// correct join's ON condition. out.push_back(&edge); } diff --git a/tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.reference b/tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.reference new file mode 100644 index 000000000000..3350bf0bdf33 --- /dev/null +++ b/tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.reference @@ -0,0 +1,12 @@ +reported greedy: +0 Join_1_Value_0 0 Join_2_Value_0 0 Join_3_Value_0 +reported dpsub: +0 Join_1_Value_0 0 Join_2_Value_0 0 Join_3_Value_0 +filter deep relation greedy: +0 Join_2_Value_0 0 Join_3_Value_0 0 Join_1_Value_0 +filter deep relation dpsub: +0 Join_2_Value_0 0 Join_3_Value_0 0 Join_1_Value_0 +four way filter last greedy: +0 0 0 0 Join_4_Value_0 +four way filter last dpsub: +0 0 0 0 Join_4_Value_0 diff --git a/tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.sql b/tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.sql new file mode 100644 index 000000000000..b0657008aedb --- /dev/null +++ b/tests/queries/0_stateless/04869_dpsub_single_table_filter_placement.sql @@ -0,0 +1,86 @@ +-- Tests that the DPsub join-order algorithm keeps single-table filter predicates that live in a +-- join's ON clause (e.g. `... JOIN t ON t.a = u.a AND t.b = 'x'`). DPsub attaches such a predicate +-- at the join that introduces its relation. Two earlier placement bugs silently dropped the +-- predicate, letting extra rows through: +-- 1. attaching only at two-relation joins dropped it when the relation was introduced against an +-- already-multi-relation subplan (e.g. `t1 JOIN (t2 JOIN t3)`); +-- 2. gating on fromLeft()/fromRight() dropped it for any filter on a relation whose id is >= 2, +-- because those helpers test relation ids 0 and 1 specifically. +-- For each shape we print the result with the 'greedy' algorithm (the reference, which places the +-- predicate correctly) and with 'dpsub'; the two must be identical row-for-row. +-- +-- `query_plan_enable_optimizations = 0` is required to expose the bug and does NOT disable join +-- reordering (that is controlled by `query_plan_optimize_join_order_algorithm`). It disables the +-- general plan-optimization passes, notably filter push-down; with push-down enabled the +-- single-table filter is applied independently of the join and masks the dropped ON-clause +-- conjunct, so the wrong-result would not surface. + +DROP TABLE IF EXISTS t1; +DROP TABLE IF EXISTS t2; +DROP TABLE IF EXISTS t3; +DROP TABLE IF EXISTS t4; + +CREATE TABLE t1 (id UInt64, value String) ENGINE = MergeTree ORDER BY tuple(); +CREATE TABLE t2 (id UInt64, value String) ENGINE = MergeTree ORDER BY tuple(); +CREATE TABLE t3 (id UInt64, value String) ENGINE = MergeTree ORDER BY tuple(); +CREATE TABLE t4 (id UInt64, value String) ENGINE = MergeTree ORDER BY tuple(); + +INSERT INTO t1 VALUES (0, 'Join_1_Value_0'), (1, 'Join_1_Value_1'), (2, 'Join_1_Value_2'); +INSERT INTO t2 VALUES (0, 'Join_2_Value_0'), (1, 'Join_2_Value_1'), (3, 'Join_2_Value_3'); +INSERT INTO t3 VALUES (0, 'Join_3_Value_0'), (1, 'Join_3_Value_1'), (4, 'Join_3_Value_4'); +INSERT INTO t4 VALUES (0, 'Join_4_Value_0'), (1, 'Join_4_Value_1'), (5, 'Join_4_Value_5'); + +SET enable_analyzer = 1; +SET query_plan_optimize_join_order_limit = 10; + +-- The reported case: filter on the first relation, INNER then LEFT join. +SELECT 'reported greedy:'; +SELECT t1.id, t1.value, t2.id, t2.value, t3.id, t3.value +FROM t1 INNER JOIN t2 ON t1.id = t2.id AND t1.value = 'Join_1_Value_0' +LEFT JOIN t3 ON t2.id = t3.id ORDER BY ALL +SETTINGS query_plan_optimize_join_order_algorithm = 'greedy', + query_plan_enable_optimizations = 0; + +SELECT 'reported dpsub:'; +SELECT t1.id, t1.value, t2.id, t2.value, t3.id, t3.value +FROM t1 INNER JOIN t2 ON t1.id = t2.id AND t1.value = 'Join_1_Value_0' +LEFT JOIN t3 ON t2.id = t3.id ORDER BY ALL +SETTINGS query_plan_optimize_join_order_algorithm = 'dpsub', + query_plan_enable_optimizations = 0; + +-- Filter on a relation that is introduced last, against an already-joined subplan (relation id >= 2). +SELECT 'filter deep relation greedy:'; +SELECT t2.id, t2.value, t3.id, t3.value, t1.id, t1.value +FROM t2 INNER JOIN t3 ON t2.id = t3.id +INNER JOIN t1 ON t1.id = t2.id AND t1.value = 'Join_1_Value_0' ORDER BY ALL +SETTINGS query_plan_optimize_join_order_algorithm = 'greedy', + query_plan_enable_optimizations = 0; + +SELECT 'filter deep relation dpsub:'; +SELECT t2.id, t2.value, t3.id, t3.value, t1.id, t1.value +FROM t2 INNER JOIN t3 ON t2.id = t3.id +INNER JOIN t1 ON t1.id = t2.id AND t1.value = 'Join_1_Value_0' ORDER BY ALL +SETTINGS query_plan_optimize_join_order_algorithm = 'dpsub', + query_plan_enable_optimizations = 0; + +-- Four-way all-inner chain with the filter on the last relation (id = 3). +SELECT 'four way filter last greedy:'; +SELECT t1.id, t2.id, t3.id, t4.id, t4.value +FROM t1 INNER JOIN t2 ON t1.id = t2.id +INNER JOIN t3 ON t2.id = t3.id +INNER JOIN t4 ON t3.id = t4.id AND t4.value = 'Join_4_Value_0' ORDER BY ALL +SETTINGS query_plan_optimize_join_order_algorithm = 'greedy', + query_plan_enable_optimizations = 0; + +SELECT 'four way filter last dpsub:'; +SELECT t1.id, t2.id, t3.id, t4.id, t4.value +FROM t1 INNER JOIN t2 ON t1.id = t2.id +INNER JOIN t3 ON t2.id = t3.id +INNER JOIN t4 ON t3.id = t4.id AND t4.value = 'Join_4_Value_0' ORDER BY ALL +SETTINGS query_plan_optimize_join_order_algorithm = 'dpsub', + query_plan_enable_optimizations = 0; + +DROP TABLE t1; +DROP TABLE t2; +DROP TABLE t3; +DROP TABLE t4; From a126632cbde5ab200a22bf4c8bac315dc1a736d8 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 07:21:06 +0000 Subject: [PATCH 071/185] Backport #117590 to 26.8: Fix wrong results from JIT-compiled float to integer and `Decimal` conversions --- src/DataTypes/Native.cpp | 2 + src/Functions/FunctionIfBase.h | 2 +- src/Functions/FunctionsConversion.cpp | 64 +++++++------- .../04072_jit_decimal_expressions.sql | 1 + .../04143_if_decimal_int_literal_jit.sql | 2 +- ...04205_jit_decimal_cast_branches_parity.sql | 8 +- ...40_jit_float_to_big_int_libcalls.reference | 16 +--- .../04240_jit_float_to_big_int_libcalls.sql | 41 +++++---- .../05055_jit_float_to_integer_cast.reference | 8 ++ .../05055_jit_float_to_integer_cast.sql | 88 +++++++++++++++++++ 10 files changed, 162 insertions(+), 70 deletions(-) create mode 100644 tests/queries/0_stateless/05055_jit_float_to_integer_cast.reference create mode 100644 tests/queries/0_stateless/05055_jit_float_to_integer_cast.sql diff --git a/src/DataTypes/Native.cpp b/src/DataTypes/Native.cpp index 3784b161efef..6c670eafe12b 100644 --- a/src/DataTypes/Native.cpp +++ b/src/DataTypes/Native.cpp @@ -311,6 +311,8 @@ llvm::Value * nativeCastWithDecimalScale( } if (from_w.isFloat32() || from_w.isFloat64()) { + /// A float source must not reach here: `fptosi` has no defined result outside the destination range. + chassert(false, "Float to Decimal must not be JIT-compiled"); /// Float → `Decimal`: multiply by `10^to_scale` in floating point first, /// then truncate to the target integer storage type. if (to_scale == 0) diff --git a/src/Functions/FunctionIfBase.h b/src/Functions/FunctionIfBase.h index e80cd3b75694..6ec99c209760 100644 --- a/src/Functions/FunctionIfBase.h +++ b/src/Functions/FunctionIfBase.h @@ -64,7 +64,7 @@ class FunctionIfBase : public IFunction b.CreateCondBr(nativeBoolCast(b, cond), then, next); b.SetInsertPoint(then); - /// Use `nativeCastWithDecimalScale` to correctly lift integer/float branches to a + /// Use `nativeCastWithDecimalScale` to correctly lift integer branches to a /// `Decimal` `result_type` (and to convert between `Decimal` types of different scales). /// Plain `nativeCast` reinterprets the integer bits without applying the `10^scale` /// factor, which silently produces wrong values when the analyzer leaves a non-`Decimal` diff --git a/src/Functions/FunctionsConversion.cpp b/src/Functions/FunctionsConversion.cpp index 726781d7ea8c..d20712d3f738 100644 --- a/src/Functions/FunctionsConversion.cpp +++ b/src/Functions/FunctionsConversion.cpp @@ -3351,6 +3351,34 @@ bool castBothTypes(const IDataType * left, const IDataType * right, F && f) return castType(left, [&](const auto & left_) { return castType(right, [&](const auto & right_) { return f(left_, right_); }); }); } +/// Whether a numeric conversion `from` -> `to` can be JIT-compiled. A float source is refused for an +/// integer or `Decimal` destination, because `fptosi` / `fptoui` have no defined result outside the +/// destination range. A `Bool` destination stays allowed, it is compiled through `nativeBoolCast`. +static bool isCompilableNumericConversion(const IDataType * from, const IDataType * to) +{ + return castBothTypes(from, to, [](const auto & left, const auto & right) + { + using LeftDataType = std::decay_t; + using RightDataType = std::decay_t; + + if constexpr (IsDataTypeDecimalOrNumber && IsDataTypeDecimalOrNumber) + { + if constexpr (IsDataTypeNumber && IsDataTypeNumber) + { + if constexpr (is_floating_point + && !is_floating_point) + return isBool(right.getPtr()); + return true; + } + else if constexpr (IsDataTypeNumber && IsDataTypeDecimal) + return !is_floating_point; + else if constexpr (IsDataTypeDecimal && IsDataTypeNumber) + return true; + } + return false; + }); +} + bool convertIsCompilableImpl(const DataTypes & types, const DataTypePtr & result_type) { if (types.empty()) @@ -3359,25 +3387,7 @@ bool convertIsCompilableImpl(const DataTypes & types, const DataTypePtr & result if (!canBeNativeType(types[0]) || !canBeNativeType(result_type)) return false; - return castBothTypes( - types[0].get(), - result_type.get(), - [](const auto & left, const auto & right) - { - using LeftDataType = std::decay_t; - using RightDataType = std::decay_t; - - if constexpr (IsDataTypeDecimalOrNumber && IsDataTypeDecimalOrNumber) - { - if constexpr (IsDataTypeNumber && IsDataTypeNumber) - return true; - else if constexpr (IsDataTypeNumber && IsDataTypeDecimal) - return true; - else if constexpr (IsDataTypeDecimal && IsDataTypeNumber) - return true; - } - return false; - }); + return isCompilableNumericConversion(types[0].get(), result_type.get()); } llvm::Value * convertCompileImpl(llvm::IRBuilderBase & builder, const ValuesWithType & arguments, const DataTypePtr & result_type) @@ -3479,21 +3489,7 @@ bool FunctionCast::isCompilable() const if (!canBeNativeType(denull_input_type) || !canBeNativeType(denull_result_type)) return false; - return castBothTypes(denull_input_type.get(), denull_result_type.get(), [](const auto & left, const auto & right) - { - using LeftDataType = std::decay_t; - using RightDataType = std::decay_t; - if constexpr (IsDataTypeDecimalOrNumber && IsDataTypeDecimalOrNumber) - { - if constexpr (IsDataTypeNumber && IsDataTypeNumber) - return true; - else if constexpr (IsDataTypeNumber && IsDataTypeDecimal) - return true; - else if constexpr (IsDataTypeDecimal && IsDataTypeNumber) - return true; - } - return false; - }); + return isCompilableNumericConversion(denull_input_type.get(), denull_result_type.get()); } llvm::Value * FunctionCast::compile(llvm::IRBuilderBase & builder, const ValuesWithType & arguments) const diff --git a/tests/queries/0_stateless/04072_jit_decimal_expressions.sql b/tests/queries/0_stateless/04072_jit_decimal_expressions.sql index e735f68165d7..eda70438d6ae 100644 --- a/tests/queries/0_stateless/04072_jit_decimal_expressions.sql +++ b/tests/queries/0_stateless/04072_jit_decimal_expressions.sql @@ -34,6 +34,7 @@ SELECT toFloat32(d32), toFloat64(d64) FROM test_jit_dec_expr ORDER BY d32; SELECT 'Test integer to Decimal conversions'; SELECT toDecimal32(i64, 2), toDecimal64(i64, 4) FROM test_jit_dec_expr ORDER BY i64; +-- The `Float ->` direction is evaluated by the interpreter: see #117442. SELECT 'Test float to Decimal conversions'; SELECT toDecimal64(f64, 4) FROM test_jit_dec_expr ORDER BY f64; diff --git a/tests/queries/0_stateless/04143_if_decimal_int_literal_jit.sql b/tests/queries/0_stateless/04143_if_decimal_int_literal_jit.sql index 0900b1e0533e..d54a37585d5a 100644 --- a/tests/queries/0_stateless/04143_if_decimal_int_literal_jit.sql +++ b/tests/queries/0_stateless/04143_if_decimal_int_literal_jit.sql @@ -45,7 +45,7 @@ WITH materialize(3::Decimal(38, 30)) AS r, materialize(2) AS k SELECT sum(multiI WITH materialize(1::Decimal(76, 60)) AS r, materialize(1) AS k SELECT sum(if(k != 1, r, 1)); WITH materialize(1::Decimal(76, 73)) AS r, materialize(1) AS k SELECT sum(if(k != 1, r, 1)); --- `Float` -> `Decimal` branch lift (the analyzer promotes the result to `Decimal`). +-- A float branch is never lifted: the result is `Variant(Decimal(18, 4), Float64)`, which is not compiled. WITH materialize(toDecimal64(1.5, 4)) AS r, materialize(1) AS k SELECT if(k != 1, r, 2.5::Float64); -- Sanity check: the non-JIT path produces the same answers. diff --git a/tests/queries/0_stateless/04205_jit_decimal_cast_branches_parity.sql b/tests/queries/0_stateless/04205_jit_decimal_cast_branches_parity.sql index ca1d5926f830..ac79ea0cb10b 100644 --- a/tests/queries/0_stateless/04205_jit_decimal_cast_branches_parity.sql +++ b/tests/queries/0_stateless/04205_jit_decimal_cast_branches_parity.sql @@ -40,11 +40,9 @@ -- the analyzer promotes the result type to `Variant` because -- `use_variant_as_common_type = 1` by default, and `canBeNativeType` -- excludes `Variant`. `FunctionIfBase::isCompilableImpl` returns `false` --- and the helper is never called. The helper still has correct code paths --- for these (the `pow10_fp_const` helper avoids 64-bit narrowing via --- `APFloat::convertFromAPInt` so high-bit-width `Decimal128`/`Decimal256` --- factors stay accurate) so it is forward-compatible if the analyzer ever --- stops promoting to `Variant`. +-- and the helper is never called. `Float` -> `Decimal` is asserted against +-- rather than supported: an out-of-range float to integer conversion has no +-- defined result once compiled. -- -- * `Decimal` -> `Decimal` with different scales (scale increase OR decrease): -- `FunctionIf::executeImpl` rejects this with `NOT_IMPLEMENTED: diff --git a/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.reference b/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.reference index 67a5306eb758..102257aae8b5 100644 --- a/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.reference +++ b/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.reference @@ -1,18 +1,5 @@ --- JIT --- ---- Float -> 128-bit / 256-bit integers --- -2 -2 -2 -2 -2 -2 -2 -2 ---- 128-bit / 256-bit integers -> Float --- -2 -2 -2 -2 +--- 128-bit integers -> Float --- 2 2 2 @@ -36,3 +23,4 @@ 2 2 2 +1 diff --git a/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.sql b/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.sql index 54923031e8c2..71acd02410de 100644 --- a/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.sql +++ b/tests/queries/0_stateless/04240_jit_float_to_big_int_libcalls.sql @@ -6,25 +6,15 @@ SELECT '--- JIT ---'; SET compile_expressions = 1, min_count_to_compile_expression = 0; -SELECT '--- Float -> 128-bit / 256-bit integers ---'; -SELECT toInt128 (materialize(1.5) + materialize(0.5)); -SELECT toUInt128(materialize(1.5) + materialize(0.5)); -SELECT toInt256 (materialize(1.5) + materialize(0.5)); -SELECT toUInt256(materialize(1.5) + materialize(0.5)); -SELECT toInt128 (materialize(1.5)::Float32 + materialize(0.5)::Float32); -SELECT toUInt128(materialize(1.5)::Float32 + materialize(0.5)::Float32); -SELECT toInt256 (materialize(1.5)::Float32 + materialize(0.5)::Float32); -SELECT toUInt256(materialize(1.5)::Float32 + materialize(0.5)::Float32); - -SELECT '--- 128-bit / 256-bit integers -> Float ---'; +-- Only the 128-bit integer to float direction is compiled, and it reaches `__floattisf` and +-- `__floattidf`. A float source with an integer destination is declined, and a 256-bit value is not +-- a native JIT type, so both have an interpreted arm only: the first is covered by +-- `05055_jit_float_to_integer_cast`, the second by the block below. +SELECT '--- 128-bit integers -> Float ---'; SELECT toFloat32(materialize(2::Int128) + materialize(0::Int128)); SELECT toFloat64(materialize(2::Int128) + materialize(0::Int128)); SELECT toFloat32(materialize(2::UInt128) + materialize(0::UInt128)); SELECT toFloat64(materialize(2::UInt128) + materialize(0::UInt128)); -SELECT toFloat32(materialize(2::Int256) + materialize(0::Int256)); -SELECT toFloat64(materialize(2::Int256) + materialize(0::Int256)); -SELECT toFloat32(materialize(2::UInt256) + materialize(0::UInt256)); -SELECT toFloat64(materialize(2::UInt256) + materialize(0::UInt256)); SELECT '--- no JIT ---'; SET compile_expressions = 0; @@ -48,3 +38,24 @@ SELECT toFloat32(materialize(2::Int256) + materialize(0::Int256)); SELECT toFloat64(materialize(2::Int256) + materialize(0::Int256)); SELECT toFloat32(materialize(2::UInt256) + materialize(0::UInt256)); SELECT toFloat64(materialize(2::UInt256) + materialize(0::UInt256)); + +-- Every row above compares a compiled value against an interpreted one, so all of them would still +-- pass if the JIT arm stopped compiling. This pins that the 128-bit conversion is compiled wherever +-- the control is, which also holds in a build without the embedded compiler. The control is plain +-- arithmetic, so it stays at 1 even if every conversion stops being compilable. +SELECT toFloat64(materialize(2::Int128) + materialize(0::Int128)) + SETTINGS compile_expressions = 1, min_count_to_compile_expression = 0, log_comment = '04240_int128' FORMAT Null; +SELECT materialize(2.0) + materialize(0.0) + materialize(1.0) + SETTINGS compile_expressions = 1, min_count_to_compile_expression = 0, log_comment = '04240_control' FORMAT Null; + +SYSTEM FLUSH LOGS query_log; + +WITH shapes AS +( + SELECT log_comment, argMax(ProfileEvents['CompiledFunctionExecute'] > 0, event_time_microseconds) AS compiled + FROM system.query_log + WHERE current_database = currentDatabase() AND type = 'QueryFinish' AND log_comment LIKE '04240_%' + GROUP BY log_comment +) +SELECT (SELECT compiled FROM shapes WHERE log_comment = '04240_int128') + = (SELECT compiled FROM shapes WHERE log_comment = '04240_control'); diff --git a/tests/queries/0_stateless/05055_jit_float_to_integer_cast.reference b/tests/queries/0_stateless/05055_jit_float_to_integer_cast.reference new file mode 100644 index 000000000000..1c4357e61ffd --- /dev/null +++ b/tests/queries/0_stateless/05055_jit_float_to_integer_cast.reference @@ -0,0 +1,8 @@ +230 1 230 +250 1 250 +true true +true true +true true +true true +0 +1 1 1 diff --git a/tests/queries/0_stateless/05055_jit_float_to_integer_cast.sql b/tests/queries/0_stateless/05055_jit_float_to_integer_cast.sql new file mode 100644 index 000000000000..26e66cc4e70c --- /dev/null +++ b/tests/queries/0_stateless/05055_jit_float_to_integer_cast.sql @@ -0,0 +1,88 @@ +-- https://github.com/ClickHouse/ClickHouse/issues/117442 +SET compile_expressions = 1; +SET min_count_to_compile_expression = 0; + +DROP TABLE IF EXISTS t_jit_float_cast; +CREATE TABLE t_jit_float_cast (c0 UInt8) ENGINE = Memory; +INSERT INTO t_jit_float_cast VALUES (230), (250); + +-- In-range conversions are exact and identical compiled or interpreted. +SELECT c0, + toFloat64(c0) / 10 <= CAST(toFloat64(c0) AS UInt8) AS in_range_by_cast, + CAST(toFloat64(c0) AS Decimal32(2)) AS in_range_decimal +FROM t_jit_float_cast +ORDER BY c0; + +-- A value the destination cannot hold raises. The non-finite one is built from the column so that +-- it is not constant folded before execution. +SELECT CAST(-toFloat64(c0) * 1e9 AS Decimal32(2)) FROM t_jit_float_cast; -- { serverError DECIMAL_OVERFLOW } +SELECT toDecimal32(-toFloat64(c0) * 1e9, 2) FROM t_jit_float_cast; -- { serverError DECIMAL_OVERFLOW } +SELECT toUInt8(toFloat64(c0) / (toFloat64(c0) - toFloat64(c0))) FROM t_jit_float_cast; -- { serverError CANNOT_CONVERT_TYPE } +SELECT CAST(toFloat64(c0) / (toFloat64(c0) - toFloat64(c0)) AS UInt8) FROM t_jit_float_cast; -- { serverError CANNOT_CONVERT_TYPE } + +-- `Bool` is the one float to integer destination that stays compiled, because it is lowered as +-- `value != 0` rather than as a conversion, which is exact for every value. +SELECT CAST(-toFloat64(c0) * 1e30 AS Bool), + CAST(toFloat64(c0) / (toFloat64(c0) - toFloat64(c0)) AS Bool) +FROM t_jit_float_cast +ORDER BY c0; + +SET compile_expressions = 0; +SELECT CAST(-toFloat64(c0) * 1e30 AS Bool), + CAST(toFloat64(c0) / (toFloat64(c0) - toFloat64(c0)) AS Bool) +FROM t_jit_float_cast +ORDER BY c0; +SET compile_expressions = 1, min_count_to_compile_expression = 0; + +-- The value an out-of-range float converts to is not defined by the language, so compare the compiled +-- and the interpreted evaluation of the same expression instead of pinning a literal. +CREATE TABLE t_jit_float_cast_arms (c0 UInt8, lte UInt8) ENGINE = Memory; + +INSERT INTO t_jit_float_cast_arms +SELECT c0, toFloat64(c0) / 10 <= CAST(-toFloat64(c0) AS UInt8) FROM t_jit_float_cast; + +SET compile_expressions = 0; +INSERT INTO t_jit_float_cast_arms +SELECT c0, toFloat64(c0) / 10 <= CAST(-toFloat64(c0) AS UInt8) FROM t_jit_float_cast; +SET compile_expressions = 1, min_count_to_compile_expression = 0; + +SELECT count() FROM (SELECT c0 FROM t_jit_float_cast_arms GROUP BY c0 HAVING uniqExact(lte) > 1); + +DROP TABLE t_jit_float_cast_arms; + +-- Every row above is a value oracle, so all of them would still pass if the conversions silently +-- stopped or started being compiled. The shapes below pin which of them compiles. Each has one +-- compilable child, so once its `CAST` is declined nothing is left to compile: a declined shape +-- reaches zero even where the control compiles. `CompiledFunctionExecute` counts executions of an +-- already-compiled node, so a warm compiled cache does not change any of them. +SELECT CAST(toFloat64(number) AS Bool) FROM numbers(2) + SETTINGS compile_expressions = 1, min_count_to_compile_expression = 0, log_comment = '05055_bool' FORMAT Null; +SELECT CAST(toFloat64(number) AS UInt8) FROM numbers(2) + SETTINGS compile_expressions = 1, min_count_to_compile_expression = 0, log_comment = '05055_declined' FORMAT Null; +SELECT toFloat64(number) + 1 FROM numbers(2) + SETTINGS compile_expressions = 1, min_count_to_compile_expression = 0, log_comment = '05055_control' FORMAT Null; +-- The reported shape: a comparison whose right side is a declined conversion. The conversion becomes +-- an input to the compiled expression, so the comparison around it must still compile. Neither side +-- can compile on its own, so the counter here belongs to the comparison and to nothing else. +SELECT toFloat64(number) <= CAST(toFloat64(number) AS UInt8) FROM numbers(2) + SETTINGS compile_expressions = 1, min_count_to_compile_expression = 0, log_comment = '05055_parent' FORMAT Null; + +SYSTEM FLUSH LOGS query_log; + +WITH shapes AS +( + SELECT log_comment, argMax(ProfileEvents['CompiledFunctionExecute'] > 0, event_time_microseconds) AS compiled + FROM system.query_log + WHERE current_database = currentDatabase() AND type = 'QueryFinish' AND log_comment LIKE '05055_%' + GROUP BY log_comment +) +-- The control keeps the first column honest in a build without the embedded compiler, where every +-- shape is interpreted and an absolute assertion would go green on nothing being compiled. +SELECT + (SELECT compiled FROM shapes WHERE log_comment = '05055_bool') + = (SELECT compiled FROM shapes WHERE log_comment = '05055_control'), + (SELECT compiled FROM shapes WHERE log_comment = '05055_declined') = 0, + (SELECT compiled FROM shapes WHERE log_comment = '05055_parent') + = (SELECT compiled FROM shapes WHERE log_comment = '05055_control'); + +DROP TABLE t_jit_float_cast; From 801e55472b925a983ff727089ec166f2da2a34ac Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 07:21:44 +0000 Subject: [PATCH 072/185] Backport #114521 to 26.8: Fix Iceberg query failure after MODIFY COLUMN to Nullable --- .../DataLakes/Iceberg/SchemaProcessor.cpp | 10 +- ..._prewhere_modify_column_nullable.reference | 39 +++++ ...iceberg_prewhere_modify_column_nullable.sh | 153 ++++++++++++++++++ 3 files changed, 201 insertions(+), 1 deletion(-) create mode 100644 tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.reference create mode 100755 tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.sh diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp index 51583a11987f..5eaaaaf07975 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/SchemaProcessor.cpp @@ -813,7 +813,15 @@ std::shared_ptr IcebergSchemaProcessor::getSchemaTransformationDag( /// a whitespace-only difference is the same type and needs only a rename, not a cast. if (canonicalizeTypeSpacing(old_type) == canonicalizeTypeSpacing(new_type)) { - if (old_json->getValue(f_name) != name) + /// Nullability is carried by the separate `required` key, so equal type strings + /// can still resolve to different types. Only relaxing required to optional is + /// legal evolution; the reverse keeps the plain passthrough. + const bool old_required = old_json->getValue(f_required); + if (old_required && !required && !old_node->result_type->equals(*type)) + { + node = &dag->addCast(*old_node, type, name, nullptr); + } + else if (old_json->getValue(f_name) != name) { node = &dag->addAlias(*old_node, name); } diff --git a/tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.reference b/tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.reference new file mode 100644 index 000000000000..1a595af2079a --- /dev/null +++ b/tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.reference @@ -0,0 +1,39 @@ +--- WHERE on the evolved column --- +4 4 +5 five +--- PREWHERE on the evolved column --- +4 4 +5 five +--- PREWHERE IS NULL / IS NOT NULL --- +none +6 +--- PREWHERE on an untouched column --- +4 4 +--- declared type and full scan --- +7 2 Nullable(Int64) +0 0 +1 1 +2 2 +3 3 +4 4 +5 five +\N none +--- MODIFY COLUMN to Nullable plus RENAME COLUMN --- +4 4 Nullable(Int64) +5 five Nullable(Int64) +--- widening plus nullability (int required to long optional) --- +4 Nullable(Int64) +--- optional to required is still rejected --- +Iceberg spec doesn't allow change type from nullable to non-nullable +--- String column made Nullable --- +4 Nullable(String) +--- externally authored optional to required stays a passthrough --- +id Int64 +s String +none +one +three +none +one +three +three diff --git a/tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.sh b/tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.sh new file mode 100755 index 000000000000..064a0c290e54 --- /dev/null +++ b/tests/queries/0_stateless/04737_iceberg_prewhere_modify_column_nullable.sh @@ -0,0 +1,153 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-parallel-replicas +# `no-parallel-replicas`: see comment in `04071_iceberg_orc_prewhere_crash.sh`. +# `StorageObjectStorageCluster` (used when `parallel_replicas_for_cluster_engines = 1`, +# default) does not delegate `supportsPrewhere` to its underlying configuration. +# +# Regression test for issue #85029: filtering a column that `ALTER TABLE ... MODIFY COLUMN` +# made `Nullable` fails on the Iceberg data files written before the `ALTER`. + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +# `optimize_move_to_prewhere=1` + `query_plan_optimize_prewhere=1` are pinned on every +# discriminating statement: the failure only appears when the predicate is pushed into the +# reader, the runner randomizes both, and with either one off the pre-fix result is already +# correct, so the test would stop exercising the fix. +PREWHERE_SETTINGS="--optimize_move_to_prewhere=1 --query_plan_optimize_prewhere=1" + +TABLE="t_null_${CLICKHOUSE_DATABASE}_${RANDOM}" +TABLE_REN="t_ren_${CLICKHOUSE_DATABASE}_${RANDOM}" +TABLE_WID="t_wid_${CLICKHOUSE_DATABASE}_${RANDOM}" +TABLE_REJ="t_rej_${CLICKHOUSE_DATABASE}_${RANDOM}" +TABLE_STR="t_str_${CLICKHOUSE_DATABASE}_${RANDOM}" + +drop_table() { + ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS $1" + rm -rf "${USER_FILES_PATH}/$1/" +} + +# Rows 0..4 are written while `id` is still required, rows 5 and NULL after the `ALTER`, so the +# table mixes pre- and post-evolution data files. Only the pre-evolution ones carry the defect. +create_mixed_nullability_table() { + local table="$1" + local table_path="${USER_FILES_PATH}/${table}/" + rm -rf "${table_path}" + ${CLICKHOUSE_CLIENT} --query "DROP TABLE IF EXISTS ${table}" + ${CLICKHOUSE_CLIENT} --query "CREATE TABLE ${table} (id Int64, s String) ENGINE = IcebergLocal('${table_path}', 'Parquet')" + ${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "INSERT INTO ${table} SELECT number, toString(number) FROM numbers(5)" + ${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "ALTER TABLE ${table} MODIFY COLUMN id Nullable(Int64)" + ${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "INSERT INTO ${table} SELECT * FROM values('id Nullable(Int64), s String', (5, 'five'), (NULL, 'none'))" +} + +create_mixed_nullability_table "${TABLE}" + +echo "--- WHERE on the evolved column ---" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT id, s FROM ${TABLE} WHERE id > 3 ORDER BY id" + +# Explicit PREWHERE probes the reader-side path directly, without depending on the +# WHERE->PREWHERE mover. +echo "--- PREWHERE on the evolved column ---" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT id, s FROM ${TABLE} PREWHERE id > 3 ORDER BY id" + +echo "--- PREWHERE IS NULL / IS NOT NULL ---" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT s FROM ${TABLE} PREWHERE id IS NULL ORDER BY s" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT count() FROM ${TABLE} PREWHERE id IS NOT NULL" + +echo "--- PREWHERE on an untouched column ---" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT id, s FROM ${TABLE} PREWHERE s = '4' ORDER BY id" + +echo "--- declared type and full scan ---" +${CLICKHOUSE_CLIENT} --query "SELECT count(), countIf(id > 3), toTypeName(any(id)) FROM ${TABLE}" +${CLICKHOUSE_CLIENT} --query "SELECT id, s FROM ${TABLE} ORDER BY id, s" + +# Composition with a rename: the transform must apply the new name and the new nullability in one +# node. Before the fix the rename branch renamed the column and dropped the type change. +echo "--- MODIFY COLUMN to Nullable plus RENAME COLUMN ---" +create_mixed_nullability_table "${TABLE_REN}" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "ALTER TABLE ${TABLE_REN} RENAME COLUMN id TO idx" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT idx, s, toTypeName(idx) FROM ${TABLE_REN} PREWHERE idx > 3 ORDER BY idx" + +# Widening and nullability at once already took the type-conversion branch and was already +# correct; assert it stays correct. +echo "--- widening plus nullability (int required to long optional) ---" +rm -rf "${USER_FILES_PATH}/${TABLE_WID}/" +${CLICKHOUSE_CLIENT} --query "CREATE TABLE ${TABLE_WID} (id Int32, s String) ENGINE = IcebergLocal('${USER_FILES_PATH}/${TABLE_WID}/', 'Parquet')" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "INSERT INTO ${TABLE_WID} SELECT toInt32(number), toString(number) FROM numbers(5)" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "ALTER TABLE ${TABLE_WID} MODIFY COLUMN id Nullable(Int64)" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT id, toTypeName(id) FROM ${TABLE_WID} PREWHERE id > 3 ORDER BY id" + +# The reverse direction is not legal evolution and must keep being rejected. +echo "--- optional to required is still rejected ---" +rm -rf "${USER_FILES_PATH}/${TABLE_REJ}/" +${CLICKHOUSE_CLIENT} --query "CREATE TABLE ${TABLE_REJ} (id Nullable(Int64), s String) ENGINE = IcebergLocal('${USER_FILES_PATH}/${TABLE_REJ}/', 'Parquet')" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "INSERT INTO ${TABLE_REJ} SELECT number, toString(number) FROM numbers(3)" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "ALTER TABLE ${TABLE_REJ} MODIFY COLUMN id Int64" 2>&1 \ + | grep -oF "Iceberg spec doesn't allow change type from nullable to non-nullable" | head -1 + +# A non-numeric type reaches the same branch through a different comparison function. +echo "--- String column made Nullable ---" +rm -rf "${USER_FILES_PATH}/${TABLE_STR}/" +${CLICKHOUSE_CLIENT} --query "CREATE TABLE ${TABLE_STR} (v String, s String) ENGINE = IcebergLocal('${USER_FILES_PATH}/${TABLE_STR}/', 'Parquet')" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "INSERT INTO ${TABLE_STR} SELECT toString(number), toString(number) FROM numbers(5)" +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "ALTER TABLE ${TABLE_STR} MODIFY COLUMN v Nullable(String)" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT v, toTypeName(v) FROM ${TABLE_STR} PREWHERE v = '4' ORDER BY v" + +# The reverse direction (optional -> required) cannot be produced by `ALTER`, which rejects it +# above, so the passthrough branch is reached only through metadata written by another engine. +# Appending a schema leaves schema 0 byte-identical, so the schema-id immutability check still +# passes, and the read selects the pair with `iceberg_metadata_file_path`. +echo "--- externally authored optional to required stays a passthrough ---" +TABLE_REV="t_rev_${CLICKHOUSE_DATABASE}_${RANDOM}" +REV_PATH="${USER_FILES_PATH}/${TABLE_REV}/" +rm -rf "${REV_PATH}" +${CLICKHOUSE_CLIENT} --query "CREATE TABLE ${TABLE_REV} (id Nullable(Int64), s String) ENGINE = IcebergLocal('${REV_PATH}', 'Parquet')" +# The NULL row is what makes the assertion non-vacuous: a cast to the required type only +# misbehaves when a NULL actually has to pass through it. +${CLICKHOUSE_CLIENT} --allow_insert_into_iceberg=1 --query "INSERT INTO ${TABLE_REV} SELECT * FROM values('id Nullable(Int64), s String', (1, 'one'), (NULL, 'none'), (3, 'three'))" + +LATEST_REV=$(ls "${REV_PATH}metadata/" | grep -E '^v[0-9]+\.metadata\.json$' | sort -t v -k2 -n | tail -1) +REV_META=$(python3 - "${REV_PATH}metadata" "${LATEST_REV}" <<'PYEOF' +import copy, json, os, re, sys + +metadata_dir, latest_file = sys.argv[1], sys.argv[2] + +with open(os.path.join(metadata_dir, latest_file)) as fh: + metadata = json.load(fh) + +current = next(s for s in metadata["schemas"] if s["schema-id"] == metadata["current-schema-id"]) +tightened = copy.deepcopy(current) +tightened["schema-id"] = max(s["schema-id"] for s in metadata["schemas"]) + 1 +for field in tightened["fields"]: + if field["name"] == "id": + field["required"] = True + +metadata["schemas"].append(tightened) +metadata["current-schema-id"] = tightened["schema-id"] +metadata["last-updated-ms"] = metadata.get("last-updated-ms", 0) + 60000 + +version = int(re.match(r"v(\d+)\.metadata\.json", latest_file).group(1)) + 1 +tmp_file = os.path.join(metadata_dir, ".tmp_next") +with open(tmp_file, "w") as fh: + json.dump(metadata, fh) +os.rename(tmp_file, os.path.join(metadata_dir, f"v{version}.metadata.json")) +print(f"metadata/v{version}.metadata.json") +PYEOF +) + +REV_TF="icebergLocal('${REV_PATH}', 'Parquet', SETTINGS iceberg_metadata_file_path = '${REV_META}')" +# The tightened schema is what the reader resolves against. +${CLICKHOUSE_CLIENT} --query "DESCRIBE ${REV_TF}" | cut -f1,2 +# Each of these three shapes fails if the transform casts the old optional column to the new +# required type instead of passing it through. +${CLICKHOUSE_CLIENT} --query "SELECT s FROM ${REV_TF} ORDER BY s" +${CLICKHOUSE_CLIENT} --query "SELECT s FROM ${REV_TF} WHERE id IS NOT NULL ORDER BY s" +${CLICKHOUSE_CLIENT} ${PREWHERE_SETTINGS} --query "SELECT s FROM ${REV_TF} PREWHERE id > 1 ORDER BY s" + +drop_table "${TABLE}" +drop_table "${TABLE_REN}" +drop_table "${TABLE_WID}" +drop_table "${TABLE_REJ}" +drop_table "${TABLE_STR}" +drop_table "${TABLE_REV}" From 99207640e4ae5fad2c59a6597d6a15dc3f08856f Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 09:07:13 +0000 Subject: [PATCH 073/185] Backport #121132 to 26.8: Add the `http_x_clickhouse_format_overrides_output_format` compatibility setting --- docs/concepts/features/interfaces/http.mdx | 4 +- src/Access/SettingsConstraints.cpp | 4 ++ src/Core/Settings.cpp | 11 ++++ src/Core/SettingsChangesHistory.cpp | 1 + src/Server/HTTPHandler.cpp | 43 +++++++++---- ..._http_x_clickhouse_format_compat.reference | 34 +++++++++++ .../05233_http_x_clickhouse_format_compat.sh | 61 +++++++++++++++++++ 7 files changed, 144 insertions(+), 14 deletions(-) create mode 100644 tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.reference create mode 100755 tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.sh diff --git a/docs/concepts/features/interfaces/http.mdx b/docs/concepts/features/interfaces/http.mdx index ff580401ef9c..dbadd8c4a7c9 100644 --- a/docs/concepts/features/interfaces/http.mdx +++ b/docs/concepts/features/interfaces/http.mdx @@ -190,7 +190,7 @@ wget -nv -O- 'http://localhost:8123/?query=SELECT 1, 2, 3 FORMAT JSON' } ``` -You can use the `default_format` URL parameter to specify a default format other than `TabSeparated`. The `X-ClickHouse-Format` header selects the format of the response explicitly: it is an alias for the `output_format` setting, so it also overrides a `FORMAT` clause in the query. It never changes how the request body of an `INSERT` is parsed — use `input_format` or `format` for that. +You can use the `default_format` URL parameter to specify a default format other than `TabSeparated`. The `X-ClickHouse-Format` header selects the format of the response explicitly: it is an alias for the `output_format` setting, so it also overrides a `FORMAT` clause in the query. It never changes how the request body of an `INSERT` is parsed — use `input_format` or `format` for that. To restore the behavior of versions before 26.8, where the header was an alias for `default_format` and did not override a `FORMAT` clause, disable the `http_x_clickhouse_format_overrides_output_format` setting (in a user profile or as a URL parameter; like the format settings themselves, it can be changed in read-only mode). ```bash $ echo 'SELECT 1 FORMAT Pretty' | curl 'http://localhost:8123/?' --data-binary @- @@ -316,7 +316,7 @@ INSERT INTO t SELECT number FROM numbers(10) SETTINGS limit = 2; | Setting | Effect | |---------|--------| -| `output_format` | Overrides the output format. It takes precedence over the query's `FORMAT` clause, the path extension, `format`, and `default_format`. It can also be set using the `X-ClickHouse-Format` header. | +| `output_format` | Overrides the output format. It takes precedence over the query's `FORMAT` clause, the path extension, `format`, and `default_format`. It can also be set using the `X-ClickHouse-Format` header. With `http_x_clickhouse_format_overrides_output_format = 0`, the header sets `default_format` instead, as in versions before 26.8. | | `input_format` | Overrides the input format for `INSERT`. It takes precedence over the query's `FORMAT` clause and `format`. | | `format` | Overrides the format in both directions unless a direction-specific setting is provided. | | `default_format` | Sets the output format when the query has no `FORMAT` clause, no path extension, and no other format override. | diff --git a/src/Access/SettingsConstraints.cpp b/src/Access/SettingsConstraints.cpp index fd68f0c6da25..fcf380632efe 100644 --- a/src/Access/SettingsConstraints.cpp +++ b/src/Access/SettingsConstraints.cpp @@ -77,6 +77,10 @@ bool isAlwaysChangeableInReadonly(std::string_view name) /// HTTP routing / session. if (name == "database" || name == "default_format") return true; + /// Selects which of `output_format` / `default_format` the `X-ClickHouse-Format` header aliases; + /// both targets are changeable here, so the switch between them must be too. + if (name == "http_x_clickhouse_format_overrides_output_format") + return true; /// Output format selection and response compression. if (name == "format" || name == "input_format" || name == "output_format" || name == "compression") return true; diff --git a/src/Core/Settings.cpp b/src/Core/Settings.cpp index 9b4f66380508..fa401792157d 100644 --- a/src/Core/Settings.cpp +++ b/src/Core/Settings.cpp @@ -7070,6 +7070,17 @@ If enabled, any URL parameter not recognized as a known parameter, setting, or ` - A plain `name=value` becomes the equality `` `name` = 'value' `` (the identifier is back-quoted, the value is quoted as a string literal). - A comparison operator (`!=`, `>`, `<`, `>=`, `<=`, `<>`) makes it a comparison: either split across the parameter (`?a!=2`, `?a>=2`) or written inline when the URL has no `=` to split on (`?a<>2`, `?f(x)>3`), in which case the reassembled `name[=value]` is parsed as a full SQL expression. +)", 0) \ + DECLARE(Bool, http_x_clickhouse_format_overrides_output_format, true, R"( +Controls which setting the `X-ClickHouse-Format` HTTP header maps to. + +If enabled (the default), the header is an alias for the `output_format` setting: it is an explicit override of the response format that wins over the `FORMAT` clause in the query and over the file extension in the URL path. + +If disabled, the header is an alias for the `default_format` setting, as it was before version 26.8: it only selects the format used when the query has no `FORMAT` clause and no other format override is applied. + +In both cases the header overrides the URL parameter of the same name (`output_format` or `default_format`, respectively) and never changes how the request body of an `INSERT` is parsed. + +This is a compatibility setting for the HTTP interface: the header is consumed before the query is parsed, so it must be supplied via a URL parameter or a user profile, not via an in-query `SETTINGS` clause. Like `output_format` and `default_format` themselves, it can always be changed in read-only mode (`readonly = 1`), so a read-only user can pass it as a URL parameter. )", 0) \ \ DECLARE(UInt64, function_range_max_elements_in_block, 500000000, R"( diff --git a/src/Core/SettingsChangesHistory.cpp b/src/Core/SettingsChangesHistory.cpp index 54044bf6c441..7ae4b8ec11d4 100644 --- a/src/Core/SettingsChangesHistory.cpp +++ b/src/Core/SettingsChangesHistory.cpp @@ -129,6 +129,7 @@ const VersionToSettingsChangesMap & getSettingsChangesHistory() {"http_allow_table_as_file", false, false, "New setting to recognize a table name in the URL path of HTTP requests, with optional format/compression extensions."}, {"http_allow_filters_as_path", false, false, "New setting to recognize hive-style `name=value` filters in the URL path of HTTP requests."}, {"http_allow_filters_as_unrecognized_url_parameters", false, false, "New setting to treat unrecognized URL parameters as filter expressions in HTTP requests."}, + {"http_x_clickhouse_format_overrides_output_format", false, true, "Controls whether the `X-ClickHouse-Format` HTTP header is an alias for `output_format` (overriding the query's `FORMAT` clause), as since 26.8, or for `default_format`, as before 26.8. The entry sits in the 26.8 block because that is the release whose behavior it restores, so `compatibility` with a version before 26.8 brings back the old header behavior."}, {"ignore_on_cluster_for_replicated_handler_queries", false, false, "New setting to ignore the ON CLUSTER clause for handler management queries when handlers are backed by replicated (Keeper) storage."}, {"materialized_views_populate_atomically", false, true, "New setting that makes plain `CREATE MATERIALIZED VIEW ... POPULATE` locally atomic: existing data is snapshotted and the view is subscribed to new inserts together, under a brief exclusive lock on the source, so rows inserted through the same server are neither missed nor duplicated. The guarantee covers the local insert path only - inserts arriving on another replica or through a distributed write are outside the cut - and it requires a source that can provide a pinned snapshot (the `MergeTree` family and `Memory`); other sources, as well as `CREATE OR REPLACE` / `REPLACE`, keep the legacy non-atomic population. Set to `false` for the legacy non-atomic behavior everywhere."}, {"input_format_json_max_object_size", 512 * 1024 * 1024, 512 * 1024 * 1024, "New setting to limit the maximum size of a single JSON object in bytes"}, diff --git a/src/Server/HTTPHandler.cpp b/src/Server/HTTPHandler.cpp index 3637fe61d4f3..df830316d51b 100644 --- a/src/Server/HTTPHandler.cpp +++ b/src/Server/HTTPHandler.cpp @@ -94,6 +94,7 @@ namespace Setting extern const SettingsBool http_allow_table_as_file; extern const SettingsBool http_allow_filters_as_path; extern const SettingsBool http_allow_filters_as_unrecognized_url_parameters; + extern const SettingsBool http_x_clickhouse_format_overrides_output_format; extern const SettingsString compression; extern const SettingsString filter; extern const SettingsString format; @@ -393,20 +394,15 @@ void HTTPHandler::processQuery( deferred_unrecognized_params.emplace_back(key, value); } - /// The `X-ClickHouse-Database` header is an alias for the `database` setting, and - /// `X-ClickHouse-Format` is an alias for the `output_format` setting. They override any matching - /// URL parameter (preserving the historical precedence). - /// - /// `X-ClickHouse-Format` maps to `output_format` rather than to `default_format`: sending this - /// header means the client definitely wants the response in that format, so it is an explicit - /// override (winning over the query's `FORMAT` clause and the path extension), not a fallback - /// used only when nothing else selects a format. It maps to `output_format` and not to the - /// bidirectional `format`, because the header has always described the response only: the same - /// header on `INSERT INTO t FORMAT JSONEachRow …` must not reinterpret the request body. + /// The `X-ClickHouse-Database` header is an alias for the `database` setting. It overrides a + /// matching URL parameter (preserving the historical precedence). if (auto header_value = request.get("X-ClickHouse-Database", ""); !header_value.empty()) settings_changes.setSetting("database", header_value); - if (auto header_value = request.get("X-ClickHouse-Format", ""); !header_value.empty()) - settings_changes.setSetting("output_format", header_value); + + /// The `X-ClickHouse-Format` header is applied below, once the settings from the URL and the user + /// profile are in effect: which setting it aliases depends on + /// `http_x_clickhouse_format_overrides_output_format`. + const String format_header_value = request.get("X-ClickHouse-Format", ""); ContextMutablePtr context; { @@ -480,6 +476,29 @@ void HTTPHandler::processQuery( context->checkSettingsConstraints(settings_changes, SettingSource::QUERY); context->applySettingsChanges(settings_changes); + /// The `X-ClickHouse-Format` header is an alias for the `output_format` setting, or - when + /// `http_x_clickhouse_format_overrides_output_format` is disabled - for the `default_format` + /// setting, which is what it meant before 26.8. Either way it overrides the URL parameter of the + /// same name (preserving the historical precedence). The choice is read from the context after + /// the URL parameters and the user profile have been applied, so the compatibility setting can + /// come from either of them. + /// + /// By default `X-ClickHouse-Format` maps to `output_format` rather than to `default_format`: + /// sending this header means the client definitely wants the response in that format, so it is + /// an explicit override (winning over the query's `FORMAT` clause and the path extension), not a + /// fallback used only when nothing else selects a format. It maps to `output_format` and not to + /// the bidirectional `format`, because the header has always described the response only: the + /// same header on `INSERT INTO t FORMAT JSONEachRow …` must not reinterpret the request body. + if (!format_header_value.empty()) + { + SettingsChanges format_header_changes; + format_header_changes.setSetting( + context->getSettingsRef()[Setting::http_x_clickhouse_format_overrides_output_format] ? "output_format" : "default_format", + format_header_value); + context->checkSettingsConstraints(format_header_changes, SettingSource::QUERY); + context->applySettingsChanges(format_header_changes); + } + const auto & settings = context->getSettingsRef(); /// === URL path parsing happens after settings are applied === diff --git a/tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.reference b/tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.reference new file mode 100644 index 000000000000..91530eb01b1c --- /dev/null +++ b/tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.reference @@ -0,0 +1,34 @@ +-- default: the header overrides the FORMAT clause of the query +{"x":1} +-- default: the header applies when the query has no FORMAT clause +{"x":1} +-- default: the header overrides the output_format URL parameter +{"x":1} +-- default: the response header reports the effective format +X-ClickHouse-Format: JSONEachRow +-- disabled: the header only sets default_format, so the FORMAT clause wins +1 +-- disabled: the header applies when the query has no FORMAT clause +{"x":1} +-- disabled: the header overrides the default_format URL parameter +{"x":1} +-- disabled: an explicit output_format URL parameter wins over the header +1 +-- disabled: the response header reports the effective format +X-ClickHouse-Format: CSV +-- disabled: works on a read-only GET request too +1 +-- disabled: the header does not change the input format of an INSERT body +{"s":"a","n":1} +-- disabled via the session (as a profile would): the FORMAT clause wins +1 +-- disabled as a URL parameter in a readonly = 1 session: the setting is always changeable in read-only mode +1 +{"x":1} +-- ... while an ordinary setting is rejected in the same session +Cannot modify 'max_rows_to_read' setting in readonly mode +-- compatibility with a version before 26.8 restores the old header behavior +1 +-- compatibility with 26.8 or 26.9 keeps the header an alias for output_format, as in those releases +{"x":1} +{"x":1} diff --git a/tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.sh b/tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.sh new file mode 100755 index 000000000000..ff816400aedd --- /dev/null +++ b/tests/queries/0_stateless/05233_http_x_clickhouse_format_compat.sh @@ -0,0 +1,61 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# `X-ClickHouse-Format` is an alias for `output_format` by default (since 26.8), and for `default_format` +# when `http_x_clickhouse_format_overrides_output_format` is disabled (the behavior of earlier versions). + +URL_COMPAT="${CLICKHOUSE_URL}&http_x_clickhouse_format_overrides_output_format=0" + +echo "-- default: the header overrides the FORMAT clause of the query" +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +echo "-- default: the header applies when the query has no FORMAT clause" +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x' +echo "-- default: the header overrides the output_format URL parameter" +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&output_format=CSV" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x' +echo "-- default: the response header reports the effective format" +${CLICKHOUSE_CURL} -sS -i "${CLICKHOUSE_URL}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' | grep -i '^X-ClickHouse-Format' | tr -d '\r' + +echo "-- disabled: the header only sets default_format, so the FORMAT clause wins" +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +echo "-- disabled: the header applies when the query has no FORMAT clause" +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x' +echo "-- disabled: the header overrides the default_format URL parameter" +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}&default_format=CSV" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x' +echo "-- disabled: an explicit output_format URL parameter wins over the header" +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}&output_format=CSV" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x' +echo "-- disabled: the response header reports the effective format" +${CLICKHOUSE_CURL} -sS -i "${URL_COMPAT}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' | grep -i '^X-ClickHouse-Format' | tr -d '\r' +echo "-- disabled: works on a read-only GET request too" +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}&query=SELECT+1+AS+x+FORMAT+CSV" -H 'X-ClickHouse-Format: JSONEachRow' +echo "-- disabled: the header does not change the input format of an INSERT body" +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}" -d 'CREATE TABLE t_05233 (s String, n UInt8) ENGINE = Memory' +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'INSERT INTO t_05233 FORMAT CSV +"a",1' +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}" -d 'SELECT * FROM t_05233 FORMAT JSONEachRow' +${CLICKHOUSE_CURL} -sS "${URL_COMPAT}" -d 'DROP TABLE t_05233' + +echo "-- disabled via the session (as a profile would): the FORMAT clause wins" +SESSION_ID="${CLICKHOUSE_DATABASE}_05233" +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&session_id=${SESSION_ID}" -d 'SET http_x_clickhouse_format_overrides_output_format = 0' +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&session_id=${SESSION_ID}" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&session_id=${SESSION_ID}&close_session=1" -d 'SELECT 1 AS x FORMAT Null' + +echo "-- disabled as a URL parameter in a readonly = 1 session: the setting is always changeable in read-only mode" +# A bare URL: the randomized settings the harness puts into CLICKHOUSE_URL are not changeable under readonly = 1. +SESSION_ID_RO="${CLICKHOUSE_DATABASE}_05233_ro" +URL_RO="${CLICKHOUSE_URL%%\?*}?database=${CLICKHOUSE_DATABASE}&session_id=${SESSION_ID_RO}" +${CLICKHOUSE_CURL} -sS "${URL_RO}" -d 'SET readonly = 1' +${CLICKHOUSE_CURL} -sS "${URL_RO}&http_x_clickhouse_format_overrides_output_format=0" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +${CLICKHOUSE_CURL} -sS "${URL_RO}&http_x_clickhouse_format_overrides_output_format=1" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +echo "-- ... while an ordinary setting is rejected in the same session" +${CLICKHOUSE_CURL} -sS "${URL_RO}&max_rows_to_read=1" -d 'SELECT 1 AS x FORMAT CSV' | grep -o "Cannot modify 'max_rows_to_read' setting in readonly mode" +${CLICKHOUSE_CURL} -sS "${URL_RO}&close_session=1" -d 'SELECT 1 AS x FORMAT Null' + +echo "-- compatibility with a version before 26.8 restores the old header behavior" +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&compatibility=26.7" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +echo "-- compatibility with 26.8 or 26.9 keeps the header an alias for output_format, as in those releases" +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&compatibility=26.8" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' +${CLICKHOUSE_CURL} -sS "${CLICKHOUSE_URL}&compatibility=26.9" -H 'X-ClickHouse-Format: JSONEachRow' -d 'SELECT 1 AS x FORMAT CSV' From 3291a11dee8ae6a021b8d3547cfe151ac9dc0f2c Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 09:21:03 +0000 Subject: [PATCH 074/185] Backport #115065 to 26.8: Do not pop hints on a line displayed programmatically --- .gitmodules | 2 +- contrib/CMakeLists.txt | 6 +- contrib/ai-sdk-cpp-cmake/CMakeLists.txt | 23 +- contrib/nlohmann-json | 2 +- contrib/nlohmann-json-cmake/CMakeLists.txt | 6 +- contrib/replxx | 2 +- src/Client/ReplxxLineReader.cpp | 121 +++++- src/Client/ReplxxLineReader.h | 26 ++ ...909_client_hints_history_navigation.python | 392 ++++++++++++++++++ ..._client_hints_history_navigation.reference | 11 + .../04909_client_hints_history_navigation.sh | 13 + ...025_client_hints_prepopulated_query.python | 338 +++++++++++++++ ..._client_hints_prepopulated_query.reference | 2 + .../05025_client_hints_prepopulated_query.sh | 14 + 14 files changed, 931 insertions(+), 27 deletions(-) create mode 100644 tests/queries/0_stateless/04909_client_hints_history_navigation.python create mode 100644 tests/queries/0_stateless/04909_client_hints_history_navigation.reference create mode 100755 tests/queries/0_stateless/04909_client_hints_history_navigation.sh create mode 100644 tests/queries/0_stateless/05025_client_hints_prepopulated_query.python create mode 100644 tests/queries/0_stateless/05025_client_hints_prepopulated_query.reference create mode 100755 tests/queries/0_stateless/05025_client_hints_prepopulated_query.sh diff --git a/.gitmodules b/.gitmodules index 2f3dd2930a0a..f110ecccc5f7 100644 --- a/.gitmodules +++ b/.gitmodules @@ -365,7 +365,7 @@ url = https://github.com/ClickHouse/ai-sdk-cpp [submodule "contrib/nlohmann-json"] path = contrib/nlohmann-json - url = https://github.com/nlohmann/json.git + url = https://github.com/ClickHouse/json [submodule "contrib/crc32c"] path = contrib/crc32c url = https://github.com/ClickHouse/crc32c diff --git a/contrib/CMakeLists.txt b/contrib/CMakeLists.txt index 4253bf5fad42..537a9f057984 100644 --- a/contrib/CMakeLists.txt +++ b/contrib/CMakeLists.txt @@ -266,8 +266,12 @@ option(ENABLE_GOOGLE_CLOUD_CPP "Enable Google Cloud Cpp" ${ENABLE_GOOGLE_CLOUD_C # `crc32c` is a small portable library used by both `google-cloud-cpp` and the snappy framing format # in `SnappyWriteBuffer`/`SnappyFramedReadBuffer`, so it is always built when libraries are enabled. add_contrib (crc32c-cmake crc32c) +# `nlohmann::json` is header-only and is used by both `google-cloud-cpp` and `ai-sdk-cpp`, which are +# enabled independently of each other. It is added for everyone so that there is one copy of it in +# the build: two copies of the same version define the same symbols with different layouts, which +# the linker resolves silently - see `contrib/ai-sdk-cpp-cmake/CMakeLists.txt`. +add_contrib (nlohmann-json-cmake nlohmann-json) if(ENABLE_GOOGLE_CLOUD_CPP) - add_contrib (nlohmann-json-cmake nlohmann-json) add_contrib (google-cloud-cpp-cmake google-cloud-cpp) # requires grpc, protobuf, absl, nlohmann's json, crc32c else() message(STATUS "Not using Google Cloud Cpp") diff --git a/contrib/ai-sdk-cpp-cmake/CMakeLists.txt b/contrib/ai-sdk-cpp-cmake/CMakeLists.txt index 6bcf697023ab..72230124d7e1 100644 --- a/contrib/ai-sdk-cpp-cmake/CMakeLists.txt +++ b/contrib/ai-sdk-cpp-cmake/CMakeLists.txt @@ -38,13 +38,16 @@ set(AI_SDK_ANTHROPIC_SOURCES "${AI_SDK_SOURCE_DIR}/src/providers/anthropic/anthropic_factory.cpp" ) -# Add nlohmann_json from submodule -set(NLOHMANN_JSON_SOURCE_DIR "${AI_SDK_THIRD_PARTY_DIR}/nlohmann_json_patched") -add_library(_ai_sdk_nlohmann_json INTERFACE) -target_include_directories(_ai_sdk_nlohmann_json SYSTEM INTERFACE - "${NLOHMANN_JSON_SOURCE_DIR}/include" -) -add_library(ai_sdk_nlohmann_json::ai_sdk_nlohmann_json ALIAS _ai_sdk_nlohmann_json) +# `nlohmann::json` comes from `contrib/nlohmann-json`, not from the `third_party/nlohmann_json_patched` +# copy this library vendors. Both are version 3.12.0, so both define the same +# `nlohmann::json_abi_v3_12_0` symbols - and the two definitions differ: the `contrib` copy is the +# ClickHouse fork that reads numbers without consulting the locale, whose `lexer` holds a +# `number_buffer` member that the vendored copy has no field for. Linking both is an ODR violation +# that the linker resolves by keeping one definition of each symbol, so a `parser` laid out by one +# header ends up in a `parse` compiled from the other, which reads and destroys members past the +# end of it: `AddressSanitizer` reports a stack-buffer-overflow and the client aborts on the first +# response of the model. The vendored copy exists to keep the number parsing away from the locale, +# which the fork does properly, so there is nothing left to keep it for. # Add httplib from submodule (header-only) set(HTTPLIB_SOURCE_DIR "${AI_SDK_THIRD_PARTY_DIR}/httplib-header-only") @@ -79,7 +82,7 @@ target_include_directories(_ai-sdk-cpp-core SYSTEM target_link_libraries(_ai-sdk-cpp-core PUBLIC - ai_sdk_nlohmann_json::ai_sdk_nlohmann_json + ch_contrib::nlohmann_json PRIVATE httplib::httplib concurrentqueue @@ -105,7 +108,7 @@ target_include_directories(_ai-sdk-cpp-openai SYSTEM target_link_libraries(_ai-sdk-cpp-openai PUBLIC _ai-sdk-cpp-core - ai_sdk_nlohmann_json::ai_sdk_nlohmann_json + ch_contrib::nlohmann_json PRIVATE httplib::httplib concurrentqueue @@ -133,7 +136,7 @@ target_include_directories(_ai-sdk-cpp-anthropic SYSTEM target_link_libraries(_ai-sdk-cpp-anthropic PUBLIC _ai-sdk-cpp-core - ai_sdk_nlohmann_json::ai_sdk_nlohmann_json + ch_contrib::nlohmann_json PRIVATE httplib::httplib concurrentqueue diff --git a/contrib/nlohmann-json b/contrib/nlohmann-json index 55f93686c015..4438a2a0cda7 160000 --- a/contrib/nlohmann-json +++ b/contrib/nlohmann-json @@ -1 +1 @@ -Subproject commit 55f93686c01528224f448c19128836e7df245f72 +Subproject commit 4438a2a0cda77bf4cd74652423e03363e003a0cc diff --git a/contrib/nlohmann-json-cmake/CMakeLists.txt b/contrib/nlohmann-json-cmake/CMakeLists.txt index 752b845bc819..863c35363cd1 100644 --- a/contrib/nlohmann-json-cmake/CMakeLists.txt +++ b/contrib/nlohmann-json-cmake/CMakeLists.txt @@ -1,8 +1,4 @@ set(JSON_DIR ${ClickHouse_SOURCE_DIR}/contrib/nlohmann-json) add_library(_nlohmann_json INTERFACE) -set_property( - TARGET _nlohmann_json - APPEND - PROPERTY INTERFACE_INCLUDE_DIRECTORIES - ${JSON_DIR}/single_include) +target_include_directories(_nlohmann_json SYSTEM INTERFACE ${JSON_DIR}/single_include) add_library(ch_contrib::nlohmann_json ALIAS _nlohmann_json) diff --git a/contrib/replxx b/contrib/replxx index c2de583a3cd4..4ab25d428dd0 160000 --- a/contrib/replxx +++ b/contrib/replxx @@ -1 +1 @@ -Subproject commit c2de583a3cd41f7b476cb625e954c191b6fcc448 +Subproject commit 4ab25d428dd0bfc0590cac0d15e1fb872794dd62 diff --git a/src/Client/ReplxxLineReader.cpp b/src/Client/ReplxxLineReader.cpp index 37af94675984..6a564e5e1965 100644 --- a/src/Client/ReplxxLineReader.cpp +++ b/src/Client/ReplxxLineReader.cpp @@ -438,6 +438,19 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) hint_completions_context.clear(); hint_completions_context_size = 0; + /// A line that was just displayed programmatically (recalled from history, found by a + /// history search, pasted, brought back from the editor) must not pop hints by itself: + /// with hints visible, the next Up/Down press would navigate the hints instead of the + /// history. The display armed the one-shot and pinned the displayed text (the same + /// display can regenerate the hints once more when replxx replays a throttled + /// refresh); the first run for an edited text unpins and shows the hints again. + if (suppress_hints_once || (!suppress_hints_for_text.empty() && suppress_hints_for_text == rx.get_state().text())) + { + suppress_hints_once = false; + return replxx::Replxx::hints_t{}; + } + suppress_hints_for_text.clear(); + /// Mirror `set_complete_on_empty(false)` *before* matching: an empty last word matches /// every suggestion, and this callback runs on every zero-delay repaint, so we must not /// fold and stable-sort the whole dictionary only to drop the result here. @@ -488,12 +501,43 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) /// The modify callback runs on every dispatched action, so reset the mirror here to track /// replxx; the hint-navigation keys re-set it *after* invoking, so a real navigation stays. rx.set_modify_callback([this] (std::string &, int &) { hint_selection = -1; }); + + /// A pasted query is also a whole new line displayed at once, so it does not pop hints + /// either (replxx's default binding for the paste marker just invokes the same action; + /// the action reads the whole paste, so the buffer holds the pasted text afterwards). + rx.bind_key(Replxx::KEY::PASTE_START, [this](char32_t code) + { + suppress_hints_once = true; + auto result = rx.invoke(Replxx::ACTION::BRACKETED_PASTE, code); + /// The paste action fills the buffer directly, without invalidating replxx's hint + /// cache, which is keyed by the buffer text and lives across prompts. Pasting the + /// exact text that carried a visible hint on an earlier prompt would therefore + /// redisplay the cached hints without ever asking our hint callback, and the + /// suppression below would have nothing to suppress. Re-setting the state is what + /// invalidates that cache (see openEditor). + rx.set_state(rx.get_state()); + suppressHintsForDisplayedLine(); + return result; + }); } /// By default C-p/C-n bound to COMPLETE_NEXT/COMPLETE_PREV, /// bind C-p/C-n to history-previous/history-next like readline. - rx.bind_key(Replxx::KEY::control('N'), [this](char32_t code) { return rx.invoke(Replxx::ACTION::HISTORY_NEXT, code); }); - rx.bind_key(Replxx::KEY::control('P'), [this](char32_t code) { return rx.invoke(Replxx::ACTION::HISTORY_PREVIOUS, code); }); + rx.bind_key(Replxx::KEY::control('N'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_NEXT, code); }); + rx.bind_key(Replxx::KEY::control('P'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_PREVIOUS, code); }); + rx.bind_key(Replxx::KEY::meta(Replxx::KEY::DOWN), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_NEXT, code); }); + rx.bind_key(Replxx::KEY::meta(Replxx::KEY::UP), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_PREVIOUS, code); }); + rx.bind_key(Replxx::KEY::meta('p'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_COMMON_PREFIX_SEARCH, code); }); + rx.bind_key(Replxx::KEY::meta('n'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_COMMON_PREFIX_SEARCH, code); }); + rx.bind_key(Replxx::KEY::meta('<'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_FIRST, code); }); + rx.bind_key(Replxx::KEY::PAGE_UP, [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_FIRST, code); }); + rx.bind_key(Replxx::KEY::meta('>'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_LAST, code); }); + rx.bind_key(Replxx::KEY::PAGE_DOWN, [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_LAST, code); }); + rx.bind_key(Replxx::KEY::control('G'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_RESTORE_CURRENT, code); }); + rx.bind_key(Replxx::KEY::meta('g'), [this](char32_t code) { return historyNavigate(Replxx::ACTION::HISTORY_RESTORE, code); }); + rx.bind_key(Replxx::KEY::control('R'), [this](char32_t code) { return historySearch(Replxx::ACTION::HISTORY_INCREMENTAL_SEARCH, code); }); + rx.bind_key(Replxx::KEY::control('S'), [this](char32_t code) { return historySearch(Replxx::ACTION::HISTORY_INCREMENTAL_SEARCH, code); }); + rx.bind_key(Replxx::KEY::meta('r'), [this](char32_t code) { return historySearch(Replxx::ACTION::HISTORY_SEEDED_INCREMENTAL_SEARCH, code); }); /// We don't want the default, "suspend" behavior, it confuses people. if (options.ignore_shell_suspend) @@ -562,7 +606,7 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) hint_selection = next; return result; } - return rx.invoke(Replxx::ACTION::LINE_NEXT, code); + return historyNavigate(Replxx::ACTION::LINE_NEXT, code); }; /// Up navigates the hints only once a hint is selected; before that it keeps recalling /// command history, so the hints do not shadow it. @@ -575,7 +619,7 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) hint_selection = next; return result; } - return rx.invoke(Replxx::ACTION::LINE_PREVIOUS, code); + return historyNavigate(Replxx::ACTION::LINE_PREVIOUS, code); }; rx.bind_key(Replxx::KEY::DOWN, hint_next); rx.bind_key(Replxx::KEY::UP, hint_previous); @@ -590,7 +634,7 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) hint_selection = next; return result; } - return rx.invoke(Replxx::ACTION::LINE_PREVIOUS, code); + return historyNavigate(Replxx::ACTION::LINE_PREVIOUS, code); }); /// Right accepts the chosen hint (the single one shown, or the one selected by navigating); @@ -670,14 +714,23 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) /// REPAINT before to avoid prompt overlap by the query rx.invoke(Replxx::ACTION::REPAINT, code); - if (!new_query.empty()) + const bool selected_query = !new_query.empty(); + if (selected_query) + { + /// The picked query is a whole new line displayed at once - do not pop hints on it + /// (see historyNavigate). + suppress_hints_once = true; rx.set_state(replxx::Replxx::State(new_query.c_str(), static_cast(new_query.size()))); + } if (bracketed_paste_enabled) enableBracketedPaste(); rx.invoke(Replxx::ACTION::CLEAR_SELF, code); - return rx.invoke(Replxx::ACTION::REPAINT, code); + auto result = rx.invoke(Replxx::ACTION::REPAINT, code); + if (selected_query) + suppressHintsForDisplayedLine(); + return result; }; rx.bind_key(Replxx::KEY::control(key_fuzzy), interactive_history_search); @@ -692,7 +745,7 @@ ReplxxLineReader::ReplxxLineReader(ReplxxLineReader::Options && options) { /// Reverse search is detected by C-R. uint32_t reverse_search = Replxx::KEY::control('R'); - return rx.invoke(Replxx::ACTION::HISTORY_INCREMENTAL_SEARCH, reverse_search); + return historySearch(Replxx::ACTION::HISTORY_INCREMENTAL_SEARCH, reverse_search); }); /// Change cursor style for overwrite mode to blinking (see console_codes(5)) @@ -736,6 +789,41 @@ bool ReplxxLineReader::hintChosen() return hintPopupActive() && (hint_selection >= 0 || hint_count == 1); } +replxx::Replxx::ACTION_RESULT ReplxxLineReader::historyNavigate(replxx::Replxx::ACTION action, char32_t code) +{ + /// The recalled entry is displayed (and its hints regenerated) inside the action, so the + /// suppression must be armed before it; the pin below keeps later regenerations of the + /// recalled text hintless (the refresh inside the action may be throttled and replayed after + /// this returns) and is cleared by the first edit. + suppress_hints_once = true; + auto result = rx.invoke(action, code); + if (rx.history_recalled()) + suppressHintsForDisplayedLine(); + else + suppress_hints_once = false; + return result; +} + +replxx::Replxx::ACTION_RESULT ReplxxLineReader::historySearch(replxx::Replxx::ACTION action, char32_t code) +{ + /// The selected entry is displayed (and its hints regenerated) inside the search action, so + /// the suppression must be armed before it. C-R, C-S, Meta-R, and the ClickHouse regular + /// history-search binding all use this wrapper. + suppress_hints_once = true; + auto result = rx.invoke(action, code); + if (rx.history_recalled()) + suppressHintsForDisplayedLine(); + else + suppress_hints_once = false; + return result; +} + +void ReplxxLineReader::suppressHintsForDisplayedLine() +{ + suppress_hints_once = false; + suppress_hints_for_text = rx.get_state().text(); +} + ReplxxLineReader::~ReplxxLineReader() { /// `Replxx::print` may fail with `std::runtime_error("write failed")` when e.g. the pty of the embedded @@ -841,8 +929,19 @@ void ReplxxLineReader::openEditor(bool format_query) rx.print("\n"); } + /// The repaint below displays the whole buffer at once on every return path - the edited + /// query, or the original one brought back when the editor exited unsuccessfully or the + /// round trip threw. All of them are programmatic displays, so none of them may pop hints + /// (see historyNavigate); otherwise the hints left over from before the editor was opened + /// would stay live and the next Down would navigate them instead of the history. + /// replxx caches the hints by the buffer text, which is unchanged unless the edited query was + /// accepted, so re-setting the state is what makes it ask the hint callback again (and get an + /// empty list) instead of redisplaying the stale cached ones. + rx.set_state(rx.get_state()); + suppress_hints_once = true; rx.invoke(replxx::Replxx::ACTION::CLEAR_SELF, 0); rx.invoke(replxx::Replxx::ACTION::REPAINT, 0); + suppressHintsForDisplayedLine(); if (bracketed_paste_enabled) enableBracketedPaste(); @@ -866,6 +965,12 @@ void ReplxxLineReader::setInitialText(const String & text) if (!text.empty()) { rx.set_preload_buffer(text); + /// The preloaded query is displayed at once - do not pop hints on it (see + /// historyNavigate). The one-shot is consumed at the first render of the line inside + /// input(); the pin is set to the raw text (replxx may normalize whitespace in the + /// preload, in which case it just stays inert). + suppress_hints_once = true; + suppress_hints_for_text = text; } } diff --git a/src/Client/ReplxxLineReader.h b/src/Client/ReplxxLineReader.h index 81958ae8ca35..f7813514374d 100644 --- a/src/Client/ReplxxLineReader.h +++ b/src/Client/ReplxxLineReader.h @@ -52,6 +52,18 @@ class ReplxxLineReader : public LineReader int executeEditor(const std::string & path); void openEditor(bool format_query); + /// Run a history-navigation action with the hint suppression armed (see + /// `suppress_hints_once`): the entry it recalls must not pop hints by itself. + replxx::Replxx::ACTION_RESULT historyNavigate(replxx::Replxx::ACTION action, char32_t code); + + /// Run a history-search action with the hint suppression armed (see + /// `suppress_hints_once`): a selected entry must not pop hints by itself. + replxx::Replxx::ACTION_RESULT historySearch(replxx::Replxx::ACTION action, char32_t code); + + /// After a line was displayed programmatically, pin its text so that any hint regeneration + /// for it shows nothing (see `suppress_hints_for_text`). + void suppressHintsForDisplayedLine(); + /// Whether the text cursor is at the very end of the input (where as-you-type hints render). bool isCursorAtEndOfInput(); /// Whether the as-you-type hint "popup" is currently navigable here: hints are shown and the @@ -89,6 +101,20 @@ class ReplxxLineReader : public LineReader int hint_count = 0; int hint_selection = -1; + /// Suppression of the as-you-type hints for a line that is displayed programmatically - + /// recalled from history, found by a history search, pasted, brought back from the editor. + /// Such a display must not pop hints by itself: with hints visible, the next Up/Down press + /// would navigate the hints instead of the history. An edit shows the hints again. + /// `suppress_hints_once` is armed before the action that displays the line (the action + /// repaints, and regenerates the hints, inside itself) and consumed by the next run of the + /// hint callback. `suppress_hints_for_text` then pins the displayed text after the action, + /// because the same display can regenerate the hints again later - replxx replays a + /// throttled refresh after the key handler returns (its "rapid refresh" of e.g. a held-down + /// Up key) - so any later callback run for exactly this text shows no hints either; the + /// first run for an edited text clears the pin. + bool suppress_hints_once = false; + std::string suppress_hints_for_text; + /// Snapshot of the completion words computed when the hints were last displayed, plus the /// context (prefix and its length) they were computed for. The completion callback reuses it /// so that accepting a hint inserts exactly the word that was shown: replxx accepts a hint by diff --git a/tests/queries/0_stateless/04909_client_hints_history_navigation.python b/tests/queries/0_stateless/04909_client_hints_history_navigation.python new file mode 100644 index 000000000000..54cd43293f00 --- /dev/null +++ b/tests/queries/0_stateless/04909_client_hints_history_navigation.python @@ -0,0 +1,392 @@ +import multiprocessing +import os +import pty +import re +import select +import shlex +import sys +import time + +TIMEOUT_SECONDS = 30 + +# How long to keep reading after the last keystroke while waiting for the expected output. The +# committed-query case only sees the result once the query has executed, which can take longer +# than the inter-keystroke drain on slow (sanitizer) builds. +FINAL_WAIT_SECONDS = 20 + +# How long to wait for the client to print the next `:) ` prompt after a committed query. This is +# the time a query of this test may take, and it is generous because the query does not have to be +# fast for the test to mean anything: in an MSan build a trivial `SELECT` that fails to resolve an +# identifier has been seen taking 36 seconds. +PROMPT_WAIT_SECONDS = 120 + +# Idle window after every keystroke, long enough for `replxx` to repaint the line. +DRAIN_IDLE_SECONDS = 0.4 + +# Extra slack for the supervisor on top of everything the worker itself may legitimately wait for. +SUPERVISOR_SLACK_SECONDS = 15 + +# A pseudo-keystroke: wait for the next prompt instead of writing anything. Sending the keys of the +# next line while the previous query is still running would let the terminal driver echo them +# instead of `replxx` interpreting them - a bracketed paste would then appear verbatim as +# `^[[200~...^[[201~` in the output and never reach the line editor. +WAIT_FOR_PROMPT = object() + +# `clickhouse-test` dumps this file into the failure report, but only under the name it derives +# from the testcase (the `.sh` wrapper), so take the path from there. +DEBUG_LOG = os.environ["CLICKHOUSE_TEST_DEBUG_LOG"] + +# The gray SGR sequence replxx uses for the as-you-type hint (Color::GRAY -> "0;90"). +HINT_COLOR = "\x1b[0;90m" + + +def make_history_file(name, entries): + """Write a replxx-format history file with the given entries, oldest first.""" + path = os.path.join( + os.environ["CLICKHOUSE_TMP"], + os.path.splitext(os.path.basename(os.path.abspath(__file__)))[0] + + f".{name}.history", + ) + with open(path, "w") as f: + for i, entry in enumerate(entries): + f.write(f"### 2020-01-01 00:00:{i:02}.000\n{entry}\n") + return path + + +def read_until(master, debug_log_fd, predicate, timeout): + output = "" + deadline = time.time() + timeout + while time.time() < deadline: + r, _, _ = select.select([master], [], [], 0.3) + if not r: + if predicate(output): + return output + continue + try: + chunk = os.read(master, 4096) + except OSError: + break + debug_log_fd.write(repr(chunk) + "\n") + debug_log_fd.flush() + output += chunk.decode(errors="replace") + if predicate(output): + return output + return output + + +def drain(master, debug_log_fd, idle): + output = "" + while True: + r, _, _ = select.select([master], [], [], idle) + if not r: + break + try: + chunk = os.read(master, 4096) + except OSError: + break + debug_log_fd.write(repr(chunk) + "\n") + debug_log_fd.flush() + output += chunk.decode(errors="replace") + return output + + +def report_failure(name, shell_pid, output, debug_log_fd): + """A PTY case that fails prints just `FAIL`, which says nothing about why. Record what the + terminal actually showed and whether the client is still alive - the client dying (the read + loop hitting EOF) looks exactly like the expected text never being printed.""" + try: + reaped_pid, wait_status = os.waitpid(shell_pid, os.WNOHANG) + except ChildProcessError: + reaped_pid, wait_status = shell_pid, -1 + running = reaped_pid == 0 + banner = ( + f"=== {name}: FAIL ===\n" + f"client still running: {running}, wait status: {wait_status}\n" + f"terminal output: {output[-8000:]!r}\n" + ) + debug_log_fd.write(banner) + debug_log_fd.flush() + sys.stderr.write(banner) + sys.stderr.flush() + + +def run_case(program, argv, name, keystrokes, check, state=None): + """Fork a PTY, wait for the prompt, send keystrokes (with a small pause between + them so replxx repaints), then run `check` over the collected output.""" + shell_pid, master = pty.fork() + if shell_pid == 0: + os.environ["TERM"] = "xterm" + os.execv(program, argv) + return + + debug_log_fd = open(DEBUG_LOG, "a") + try: + read_until(master, debug_log_fd, lambda o: ":)" in o, TIMEOUT_SECONDS) + + output = "" + # Only what arrived after the last keystroke - a prompt seen earlier in `output` says + # nothing about the prompt this step is waiting for. + since_last_keystroke = "" + for keys in keystrokes: + if keys is WAIT_FOR_PROMPT: + extra = read_until( + master, + debug_log_fd, + lambda more, seen=since_last_keystroke: ":)" in seen + more, + PROMPT_WAIT_SECONDS, + ) + output += extra + since_last_keystroke += extra + # `read_until` returns on its deadline as well as on a match, and going on from a + # deadline would write the next keys into a terminal that is still running the + # previous query: the driver echoes them instead of `replxx` interpreting them, so + # a bracketed paste ends up in the output verbatim and the case fails on its + # content, saying nothing about the prompt that never came. Stop here instead. + if ":)" not in since_last_keystroke: + print(f"{name}: FAIL") + report_failure(name, shell_pid, output, debug_log_fd) + state.value = 1 + return + continue + os.write(master, keys) + since_last_keystroke = drain(master, debug_log_fd, DRAIN_IDLE_SECONDS) + output += since_last_keystroke + + # The expected output may arrive only after the committed query executes, which can + # outlast the inter-keystroke drain on slow builds. Keep reading until the check passes + # instead of giving up after a single drain window. + if not check(output): + output += read_until( + master, + debug_log_fd, + lambda extra: check(output + extra), + FINAL_WAIT_SECONDS, + ) + + ok = check(output) + print(f"{name}: {'OK' if ok else 'FAIL'}") + if not ok: + report_failure(name, shell_pid, output, debug_log_fd) + state.value = 0 if ok else 1 + finally: + os.close(master) + debug_log_fd.close() + + +def worker_budget(keystrokes): + """Everything `run_case` may legitimately spend: the wait for the first prompt, the idle window + after every keystroke, every explicit prompt wait, and the final wait for the expected output. + The supervisor must not expire before that, or ordinary slowness looks like a failure.""" + prompt_waits = sum(1 for keys in keystrokes if keys is WAIT_FOR_PROMPT) + return ( + TIMEOUT_SECONDS + + PROMPT_WAIT_SECONDS * prompt_waits + + DRAIN_IDLE_SECONDS * len(keystrokes) + + FINAL_WAIT_SECONDS + + SUPERVISOR_SLACK_SECONDS + ) + + +def run_with_timeout(program, argv, name, history, keystrokes, check): + """Run a case with a history file holding the `history` entries. The file is written anew for + every attempt: an attempt killed after its final Enter may already have appended the query to + it, and a retry starting from that tail would recall a different entry than the case intends + while still printing the expected result.""" + for attempt in range(5): + history_file = make_history_file(name, history) + state = multiprocessing.Value("i", -1) + process = multiprocessing.Process( + target=run_case, + args=(program, argv + [f"--history_file={history_file}"], name, keystrokes, check), + kwargs={"state": state}, + ) + process.start() + process.join(worker_budget(keystrokes)) + if process.is_alive(): + process.terminate() + if state.value in (0, 1): + return + # transient timeout on a loaded machine - retry + print(f"{name}: FAIL") + sys.stderr.write(f"=== {name}: FAIL === the case timed out on every attempt\n") + + +def main(): + program = os.environ["CLICKHOUSE_LOCAL"] + base = shlex.split(program) + args = base + ["--wait_for_suggestions_to_load", "--hints", "1"] + + # A line displayed programmatically (recalled from history, pasted) must not pop the + # as-you-type hints by itself: with hints visible, the next Up/Down press would navigate + # the hints instead of the history. An edit shows the hints again. + + # 1. Up recalls an entry; no hint appears at its end ("ive" would be the ghost suffix of + # "concatAssumeInjective"). + run_with_timeout( + base[0], + args, + "up_recall_no_hints", + ["SELECT concatAssumeInject"], + [b"\x1b[A"], + lambda o: "concatAssumeInject" in o and (HINT_COLOR + "ive") not in o, + ) + + # Meta-Up is replxx's direct history-previous binding. It must use the same suppression as + # plain Up, rather than leaving the default action to regenerate hints. + run_with_timeout( + base[0], + args, + "meta_up_recall_no_hints", + ["SELECT concatAssumeInject"], + [b"\x1b\x1b[A"], + lambda o: "concatAssumeInject" in o and (HINT_COLOR + "ive") not in o, + ) + + # Ctrl-G restores the currently recalled history entry after an edit. It is another whole + # line display, so Down must return to the scratch line rather than navigate the hint list. + run_with_timeout( + base[0], + args, + "restore_current_down_navigates_history", + ["SELECT concatAssumeInjec"], + [b"SELECT 111 AS rand", b"\x1b[A", b"t", b"\x07", b"\x1b[B", b"\r"], + lambda o: re.search(r"(?:│\s*111\s*│|\x1b\[\?25h111\r?\n)", o) is not None, + ) + + # 2. Up, Up, Down walks the history (the entries end in "rand", which has many hint + # matches - Down must not step into the hint list); Enter then runs the recalled entry, + # so its result (222) is printed. The result can use either Pretty or TabSeparated output + # depending on the client's default format. + run_with_timeout( + base[0], + args, + "down_navigates_history", + ["SELECT 111 AS rand", "SELECT 222 AS rand"], + [b"\x1b[A", b"\x1b[A", b"\x1b[B", b"\r"], + lambda o: re.search(r"(?:│\s*222\s*│|\x1b\[\?25h222\r?\n)", o) is not None, + ) + + # 3. Recalling a history entry with the same text as the scratch line is still a programmatic + # display. Down must return from that entry to the scratch line instead of selecting a hint. + run_with_timeout( + base[0], + args, + "same_text_recall_down_navigates_history", + ["SELECT 222 AS rand", "SELECT 111 AS rand"], + [b"SELECT 111 AS rand", b"\x1b[A", b"\x1b[B", b"\r"], + lambda o: re.search(r"(?:│\s*111\s*│|\x1b\[\?25h111\r?\n)", o) is not None, + ) + + # 3b. An incremental history search (Ctrl-T, the ClickHouse binding for the regular non-fuzzy + # reverse search) that accepts an entry is another whole-line display: it goes through a + # different `replxx` code path than the plain history recall, so it needs its own proof that + # Down keeps walking the history. Ctrl-T, then the search text, then Ctrl-E leaves the search + # keeping the found line ("SELECT 111 AS rand") and puts the cursor at its end, where the hints + # would render. The accepted entry is committed as the most recent recall, so the first Down + # re-displays it (`replxx` emulates the Windows down-arrow there) and the second one moves on + # to "SELECT 222 AS rand", which Enter then runs. + run_with_timeout( + base[0], + args, + "incremental_search_down_navigates_history", + ["SELECT 111 AS rand", "SELECT 222 AS rand"], + [b"\x14", b"111", b"\x05", b"\x1b[B", b"\x1b[B", b"\r"], + lambda o: re.search(r"(?:│\s*222\s*│|\x1b\[\?25h222\r?\n)", o) is not None, + ) + + # 4. Editing a recalled entry shows the hints again: after Up, typing the "t" completes the + # prefix "concatAssumeInject" and the ghost "ive" appears. + run_with_timeout( + base[0], + args, + "edit_after_recall_shows_hints", + ["SELECT concatAssumeInjec"], + [b"\x1b[A", b"t"], + lambda o: (HINT_COLOR + "ive") in o, + ) + + # 5. A (bracketed) paste is also a whole new line displayed at once - no hint at its end. + run_with_timeout( + base[0], + args, + "paste_no_hints", + [], + [b"\x1b[200~SELECT concatAssumeInject\x1b[201~"], + lambda o: "concatAssumeInject" in o and (HINT_COLOR + "ive") not in o, + ) + + # 5b. Pasting the exact text that carried a visible hint on an earlier prompt: replxx caches + # the hints by the buffer text and the cache outlives the prompt, so without invalidating it + # the paste would redisplay the stale ghost without ever asking our hint callback. The first + # line is typed (so the hints for it are generated and cached) and committed - it is not a + # valid query, which is irrelevant here, only the cache state matters - and then the very same + # text is pasted on the next prompt, where no hint may appear. Only the output after the last + # prompt is examined: the first, typed line legitimately shows the ghost. + run_with_timeout( + base[0], + args, + "same_text_paste_no_hints", + [], + [ + b"SELECT concatAssumeInject", + b"\r", + WAIT_FOR_PROMPT, + b"\x1b[200~SELECT concatAssumeInject\x1b[201~", + ], + lambda o: "concatAssumeInject" in o.rsplit(":)", 1)[-1] + and (HINT_COLOR + "ive") not in o.rsplit(":)", 1)[-1], + ) + + # 6. An unchanged editor round-trip redisplays the line programmatically. It must invalidate + # replxx's cached hints, so Down remains history navigation. + editor = os.path.join(os.environ["CLICKHOUSE_TMP"], "client_hints_unchanged_editor.sh") + with open(editor, "w") as f: + f.write("#!/bin/sh\nexit 0\n") + os.chmod(editor, 0o755) + previous_editor = os.environ.get("EDITOR") + os.environ["EDITOR"] = editor + try: + run_with_timeout( + base[0], + args, + "unchanged_editor_down_navigates_history", + [], + [b"SELECT 111 AS rand", b"\x1bE", b"\x1b[B", b"\r"], + lambda o: re.search(r"(?:│\s*111\s*│|\x1b\[\?25h111\r?\n)", o) is not None, + ) + finally: + if previous_editor is None: + del os.environ["EDITOR"] + else: + os.environ["EDITOR"] = previous_editor + + # 7. An editor that exits unsuccessfully brings the original line back - also a whole-line + # programmatic display, and one that happens while the hints of the line typed before the + # editor was opened are still live. Down must remain history navigation there too. + editor = os.path.join(os.environ["CLICKHOUSE_TMP"], "client_hints_failing_editor.sh") + with open(editor, "w") as f: + f.write("#!/bin/sh\nexit 1\n") + os.chmod(editor, 0o755) + previous_editor = os.environ.get("EDITOR") + os.environ["EDITOR"] = editor + try: + run_with_timeout( + base[0], + args, + "failing_editor_down_navigates_history", + [], + [b"SELECT 111 AS rand", b"\x1bE", b"\x1b[B", b"\r"], + lambda o: re.search(r"(?:│\s*111\s*│|\x1b\[\?25h111\r?\n)", o) is not None, + ) + finally: + if previous_editor is None: + del os.environ["EDITOR"] + else: + os.environ["EDITOR"] = previous_editor + + +if __name__ == "__main__": + # The check lambdas are not picklable, so the subprocesses must be forked (not spawned, + # which became the default on newer Python). + multiprocessing.set_start_method("fork") + main() diff --git a/tests/queries/0_stateless/04909_client_hints_history_navigation.reference b/tests/queries/0_stateless/04909_client_hints_history_navigation.reference new file mode 100644 index 000000000000..bcd45174864a --- /dev/null +++ b/tests/queries/0_stateless/04909_client_hints_history_navigation.reference @@ -0,0 +1,11 @@ +up_recall_no_hints: OK +meta_up_recall_no_hints: OK +restore_current_down_navigates_history: OK +down_navigates_history: OK +same_text_recall_down_navigates_history: OK +incremental_search_down_navigates_history: OK +edit_after_recall_shows_hints: OK +paste_no_hints: OK +same_text_paste_no_hints: OK +unchanged_editor_down_navigates_history: OK +failing_editor_down_navigates_history: OK diff --git a/tests/queries/0_stateless/04909_client_hints_history_navigation.sh b/tests/queries/0_stateless/04909_client_hints_history_navigation.sh new file mode 100755 index 000000000000..04878aed0389 --- /dev/null +++ b/tests/queries/0_stateless/04909_client_hints_history_navigation.sh @@ -0,0 +1,13 @@ +#!/usr/bin/env bash +# Tags: long, no-debug + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# `clickhouse-test` dumps `$CLICKHOUSE_TMP/.debuglog` into the failure report, and the +# testcase name is the name of this shell script - the Python part cannot derive it from its own +# file name, so pass the path down. +export CLICKHOUSE_TEST_DEBUG_LOG="$CLICKHOUSE_TMP/$(basename "${BASH_SOURCE[0]}").debuglog" + +python3 "$CUR_DIR"/04909_client_hints_history_navigation.python diff --git a/tests/queries/0_stateless/05025_client_hints_prepopulated_query.python b/tests/queries/0_stateless/05025_client_hints_prepopulated_query.python new file mode 100644 index 000000000000..072594e6baef --- /dev/null +++ b/tests/queries/0_stateless/05025_client_hints_prepopulated_query.python @@ -0,0 +1,338 @@ +"""A query prepopulated into the next input line (the `??` AI SQL generation flow) is displayed +programmatically, so - like a recalled or pasted line - it must not pop the as-you-type hints: +with hints visible, the next Up/Down press would navigate the hints instead of the history. + +The AI provider is a local HTTP server speaking just enough of the OpenAI chat completion protocol +to answer with a canned query, so the test needs no network and no API key. +""" + +import http.server +import json +import multiprocessing +import os +import pty +import select +import shlex +import sys +import threading +import time + +TIMEOUT_SECONDS = 60 + +# How long to keep reading after the last keystroke while waiting for the expected output. +FINAL_WAIT_SECONDS = 30 + +# How long an explicit `wait_for` of a keystroke may take. It is generous because it covers the +# wait for the next `:) ` prompt after a committed query, and a query does not have to be fast for +# this test to mean anything: in an MSan build a trivial `SELECT` that fails to resolve an +# identifier has been seen taking 36 seconds. +STEP_WAIT_SECONDS = 120 + +# Idle window after a keystroke that has no explicit `wait_for`, long enough for `replxx` to +# repaint the line. +DRAIN_IDLE_SECONDS = 0.4 + +# Extra slack for the supervisor on top of everything the worker itself may legitimately wait for. +SUPERVISOR_SLACK_SECONDS = 15 + +# `clickhouse-test` dumps this file into the failure report, but only under the name it derives +# from the testcase (the `.sh` wrapper), so take the path from there. +DEBUG_LOG = os.environ["CLICKHOUSE_TEST_DEBUG_LOG"] + +# The gray SGR sequence replxx uses for the as-you-type hint (Color::GRAY -> "0;90"). +HINT_COLOR = "\x1b[0;90m" + +# The query the fake provider "generates". Its last word is a prefix of `concatAssumeInjective`, +# so an unsuppressed hint would render the gray ghost suffix "ive" right after it. +GENERATED_QUERY = "SELECT concatAssumeInject" + +# The prepopulated line is syntax-highlighted, so only its last word survives as a plain substring +# of the terminal output. +GENERATED_QUERY_LAST_WORD = "concatAssumeInject" + + +class CompletionHandler(http.server.BaseHTTPRequestHandler): + def do_POST(self): + length = int(self.headers.get("Content-Length", 0)) + self.rfile.read(length) + body = json.dumps( + { + "id": "chatcmpl-test", + "object": "chat.completion", + "created": 0, + "model": "test-model", + "choices": [ + { + "index": 0, + "message": { + "role": "assistant", + "content": f"{GENERATED_QUERY}", + }, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + } + ).encode() + self.send_response(200) + self.send_header("Content-Type", "application/json") + self.send_header("Content-Length", str(len(body))) + self.end_headers() + self.wfile.write(body) + + def log_message(self, *_args): + pass + + +def write_config(port): + path = os.path.join( + os.environ["CLICKHOUSE_TMP"], + os.path.splitext(os.path.basename(os.path.abspath(__file__)))[0] + ".config.xml", + ) + with open(path, "w") as f: + f.write( + "\n" + " \n" + " openai\n" + " test\n" + f" http://127.0.0.1:{port}\n" + " test-model\n" + " false\n" + " \n" + "\n" + ) + return path + + +def make_history_file(): + path = os.path.join( + os.environ["CLICKHOUSE_TMP"], + os.path.splitext(os.path.basename(os.path.abspath(__file__)))[0] + ".history", + ) + with open(path, "w") as f: + pass + return path + + +def read_until(master, debug_log_fd, predicate, timeout): + """Read until the predicate holds, the deadline passes, or the client is gone. Returns whether + the predicate matched along with what was read: a deadline is not a match, and treating it as + one lets a case go on from a state it never reached.""" + output = "" + deadline = time.time() + timeout + while time.time() < deadline: + r, _, _ = select.select([master], [], [], 0.3) + if not r: + if predicate(output): + return True, output + continue + try: + chunk = os.read(master, 4096) + except OSError: + break + debug_log_fd.write(repr(chunk) + "\n") + debug_log_fd.flush() + output += chunk.decode(errors="replace") + if predicate(output): + return True, output + return predicate(output), output + + +def drain(master, debug_log_fd, idle): + output = "" + while True: + r, _, _ = select.select([master], [], [], idle) + if not r: + break + try: + chunk = os.read(master, 4096) + except OSError: + break + debug_log_fd.write(repr(chunk) + "\n") + debug_log_fd.flush() + output += chunk.decode(errors="replace") + return output + + +def report_failure(name, shell_pid, output, debug_log_fd): + """A PTY case that fails prints just `FAIL`, which says nothing about why. Record what the + terminal actually showed and whether the client is still alive - the client dying (the read + loop hitting EOF) looks exactly like the expected text never being printed.""" + try: + reaped_pid, wait_status = os.waitpid(shell_pid, os.WNOHANG) + except ChildProcessError: + reaped_pid, wait_status = shell_pid, -1 + running = reaped_pid == 0 + banner = ( + f"=== {name}: FAIL ===\n" + f"client still running: {running}, wait status: {wait_status}\n" + f"terminal output: {output[-8000:]!r}\n" + ) + debug_log_fd.write(banner) + debug_log_fd.flush() + sys.stderr.write(banner) + sys.stderr.flush() + + +def run_case(program, argv, name, keystrokes, check, state=None, check_from=0): + """Each keystroke is a pair of the bytes to send and an optional predicate to wait for before + sending the next one (in place of a fixed idle drain). Only the output produced from the + keystroke with index `check_from` on is passed to `check`, so the earlier keystrokes may + legitimately pop hints without failing a "no hints" check.""" + shell_pid, master = pty.fork() + if shell_pid == 0: + os.environ["TERM"] = "xterm" + os.execv(program, argv) + return + + debug_log_fd = open(DEBUG_LOG, "a") + try: + matched, transcript = read_until( + master, debug_log_fd, lambda o: ":)" in o, TIMEOUT_SECONDS + ) + if not matched: + print(f"{name}: FAIL") + report_failure(name, shell_pid, transcript, debug_log_fd) + state.value = 1 + return + + # Everything the terminal showed, for the failure transcript - `output` holds only what + # the check is allowed to look at. + output = "" + for index, (keys, wait_for) in enumerate(keystrokes): + os.write(master, keys) + if wait_for is not None: + matched, segment = read_until( + master, debug_log_fd, wait_for, STEP_WAIT_SECONDS + ) + else: + matched, segment = True, drain(master, debug_log_fd, DRAIN_IDLE_SECONDS) + transcript += segment + if index >= check_from: + output += segment + # A `wait_for` that expired means the state the next keystroke assumes was never + # reached: the hint that arms the stale cache never appeared, or the prompt never came + # back and the next line would be typed into a still running query, where the terminal + # driver echoes the bytes instead of `replxx` interpreting them. Either way the case + # would prove nothing while still being able to pass, so end it here with its + # transcript. + if not matched: + print(f"{name}: FAIL") + report_failure(name, shell_pid, transcript, debug_log_fd) + state.value = 1 + return + + # The prepopulated line only appears after the (faked) generation round trip, which can + # outlast the inter-keystroke drain on slow builds. + if not check(output): + _, extra = read_until( + master, + debug_log_fd, + lambda more: check(output + more), + FINAL_WAIT_SECONDS, + ) + output += extra + transcript += extra + + ok = check(output) + print(f"{name}: {'OK' if ok else 'FAIL'}") + if not ok: + report_failure(name, shell_pid, transcript, debug_log_fd) + state.value = 0 if ok else 1 + finally: + os.close(master) + debug_log_fd.close() + + +def worker_budget(keystrokes): + """Everything `run_case` may legitimately spend: the wait for the first prompt, every explicit + `wait_for`, the idle window after every other keystroke, and the final wait for the expected + output. The supervisor must not expire before that, or ordinary slowness looks like a + failure.""" + waits = sum(1 for _, wait_for in keystrokes if wait_for is not None) + return ( + TIMEOUT_SECONDS + + STEP_WAIT_SECONDS * waits + + DRAIN_IDLE_SECONDS * (len(keystrokes) - waits) + + FINAL_WAIT_SECONDS + + SUPERVISOR_SLACK_SECONDS + ) + + +def run_with_timeout(program, argv, name, keystrokes, check, check_from=0): + for attempt in range(5): + state = multiprocessing.Value("i", -1) + process = multiprocessing.Process( + target=run_case, + args=(program, argv, name, keystrokes, check), + kwargs={"state": state, "check_from": check_from}, + ) + process.start() + process.join(worker_budget(keystrokes)) + if process.is_alive(): + process.terminate() + if state.value in (0, 1): + return + # transient timeout on a loaded machine - retry + print(f"{name}: FAIL") + sys.stderr.write(f"=== {name}: FAIL === the case timed out on every attempt\n") + + +def main(): + server = http.server.ThreadingHTTPServer(("127.0.0.1", 0), CompletionHandler) + server.daemon_threads = True + threading.Thread(target=server.serve_forever, daemon=True).start() + + config = write_config(server.server_address[1]) + history = make_history_file() + + program = os.environ["CLICKHOUSE_LOCAL"] + base = shlex.split(program) + args = base + [ + "--wait_for_suggestions_to_load", + "--hints", + "1", + f"--config-file={config}", + f"--history_file={history}", + ] + + no_hints = lambda o: GENERATED_QUERY_LAST_WORD in o and (HINT_COLOR + "ive") not in o + + try: + run_with_timeout( + base[0], + args, + "prepopulated_query_no_hints", + [(b"?? show me a query\r", None)], + no_hints, + ) + # Same-text regression: replxx keeps the hint cache across prompts, so when the generated + # query is byte-for-byte equal to the previously displayed line, the hint callback may not + # even be called - the stale cached hint must not be reused either. Type the query manually + # first (and wait for the hint ghost, which poisons the cache), execute it, then have the + # AI "generate" the very same text. The `??` request is kept as short as possible so that + # replxx's rapid-refresh throttling (1 ms) can swallow its keystrokes without regenerating + # the hint seed - only then does the preloaded line hit the stale cache. Only the output + # after the `??` keystroke is checked, as the manually typed line legitimately shows the + # hint. + run_with_timeout( + base[0], + args, + "same_text_prepopulated_query_no_hints", + [ + (GENERATED_QUERY.encode(), lambda o: (HINT_COLOR + "ive") in o), + (b"\r", lambda o: ":)" in o), + (b"??q\r", None), + ], + no_hints, + check_from=2, + ) + finally: + server.shutdown() + + +if __name__ == "__main__": + # The check lambdas are not picklable, so the subprocesses must be forked (not spawned, + # which became the default on newer Python). + multiprocessing.set_start_method("fork") + main() diff --git a/tests/queries/0_stateless/05025_client_hints_prepopulated_query.reference b/tests/queries/0_stateless/05025_client_hints_prepopulated_query.reference new file mode 100644 index 000000000000..fccb0eff5b81 --- /dev/null +++ b/tests/queries/0_stateless/05025_client_hints_prepopulated_query.reference @@ -0,0 +1,2 @@ +prepopulated_query_no_hints: OK +same_text_prepopulated_query_no_hints: OK diff --git a/tests/queries/0_stateless/05025_client_hints_prepopulated_query.sh b/tests/queries/0_stateless/05025_client_hints_prepopulated_query.sh new file mode 100755 index 000000000000..d48ace2e154e --- /dev/null +++ b/tests/queries/0_stateless/05025_client_hints_prepopulated_query.sh @@ -0,0 +1,14 @@ +#!/usr/bin/env bash +# Tags: long, no-debug, no-fasttest +# no-fasttest: needs the AI SQL generator (`ENABLE_CLIENT_AI`), which is not built in the fast test. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# `clickhouse-test` dumps `$CLICKHOUSE_TMP/.debuglog` into the failure report, and the +# testcase name is the name of this shell script - the Python part cannot derive it from its own +# file name, so pass the path down. +export CLICKHOUSE_TEST_DEBUG_LOG="$CLICKHOUSE_TMP/$(basename "${BASH_SOURCE[0]}").debuglog" + +python3 "$CUR_DIR"/05025_client_hints_prepopulated_query.python From 9308016b33cdc3bc79b785205fd72691cb770e10 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 11:58:03 +0000 Subject: [PATCH 075/185] Backport #122057 to 26.8: Mask credentials of an S3 URI in validation exceptions --- src/IO/S3/URI.cpp | 42 ++++++++++++++++--- ..._uri_exception_masks_credentials.reference | 9 ++++ ...256_s3_uri_exception_masks_credentials.sql | 33 +++++++++++++++ 3 files changed, 78 insertions(+), 6 deletions(-) create mode 100644 tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.reference create mode 100644 tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.sql diff --git a/src/IO/S3/URI.cpp b/src/IO/S3/URI.cpp index b671cb447531..43d5c47f271f 100644 --- a/src/IO/S3/URI.cpp +++ b/src/IO/S3/URI.cpp @@ -4,11 +4,13 @@ #include #include #include +#include #include #include #include #include +#include #include #include @@ -33,6 +35,34 @@ namespace ErrorCodes namespace S3 { +namespace +{ + +/// `Poco::URI::toString` renders the userinfo (`user:password@`) and the query parameters of a presigned +/// URL verbatim. Exception messages reach `system.query_log` and the server log, which, unlike the query +/// text, are not masked, so a URI must not be put into them as is. +String maskedURIString(const Poco::URI & uri) +{ + String result = uri.toString(); + maskURIUserinfo(result); + /// With `compatibility_s3_presigned_url_query_in_path` the constructor folds the query of a presigned URL + /// into the path by percent-encoding its '?', which `toString` renders as `%3F`, so `maskPresignedURLParameters` + /// would not see the parameters. Put the '?' back: `toString` encodes '%' itself as `%25`, so a `%3F` can only + /// come from a '?'. + boost::replace_all(result, "%3F", "?"); + maskPresignedURLParameters(result); + return result; +} + +/// With `compatibility_s3_presigned_url_query_in_path` the query of a presigned URL ends up in the bucket or the key. +String maskedQuotedString(String value) +{ + maskPresignedURLParameters(value); + return quoteString(value); +} + +} + URI::URI(const std::string & uri_, bool allow_archive_path_syntax, bool keep_presigned_query_parameters, S3UriStyle uri_style) { /// Case when AWS Private Link Interface is being used @@ -143,13 +173,13 @@ URI::URI(const std::string & uri_, bool allow_archive_path_syntax, bool keep_pre case S3UriStyle::VIRTUAL_HOSTED: { if (!tryInitVirtualHostedStyle(is_using_aws_private_link_interface, false)) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid S3 virtual-hosted-style uri: {}", !uri.empty() ? uri.toString() : ""); + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid S3 virtual-hosted-style uri: {}", !uri.empty() ? maskedURIString(uri) : ""); break; } case S3UriStyle::PATH: { if (!tryInitPathStyle()) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid S3 path-style uri: {}", !uri.empty() ? uri.toString() : ""); + throw Exception(ErrorCodes::BAD_ARGUMENTS, "Invalid S3 path-style uri: {}", !uri.empty() ? maskedURIString(uri) : ""); break; } } @@ -229,8 +259,8 @@ void URI::validateBucket(const String & bucket, const Poco::URI & uri) throw Exception( ErrorCodes::BAD_ARGUMENTS, "Bucket name length is out of bounds in virtual hosted style S3 URI: {}{}", - quoteString(bucket), - !uri.empty() ? " (" + uri.toString() + ")" : ""); + maskedQuotedString(bucket), + !uri.empty() ? " (" + maskedURIString(uri) + ")" : ""); } void URI::validateKey(const String & key, const Poco::URI & uri) @@ -240,8 +270,8 @@ void URI::validateKey(const String & key, const Poco::URI & uri) throw Exception( ErrorCodes::BAD_ARGUMENTS, "Invalid S3 key: {}{}", - quoteString(key), - !uri.empty() ? " (" + uri.toString() + ")" : ""); + maskedQuotedString(key), + !uri.empty() ? " (" + maskedURIString(uri) + ")" : ""); }; diff --git a/tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.reference b/tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.reference new file mode 100644 index 000000000000..d8c2eb8f0b46 --- /dev/null +++ b/tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.reference @@ -0,0 +1,9 @@ +ExceptionBeforeStart 0 Bucket name length is out of bounds in virtual hosted style S3 URI: \'b\' (http://[HIDDEN]@127.0.0.1:9/b/k.csv) +ExceptionBeforeStart 0 Invalid S3 key: \'/a//b.csv\' (http://[HIDDEN]@my-bucket-name.s3.amazonaws.com//a//b.csv) +ExceptionBeforeStart 0 Invalid S3 virtual-hosted-style uri: http://[HIDDEN]@127.0.0.1:9/bucketname/k.csv +ExceptionBeforeStart 0 Invalid S3 path-style uri: http://[HIDDEN]@127.0.0.1:9 +ExceptionBeforeStart 0 Bucket name length is out of bounds in virtual hosted style S3 URI: \'b\' (http://127.0.0.1:9/b/k.csv?X-Amz-Signature=[HIDDEN]) +ExceptionBeforeStart 0 Bucket name length is out of bounds in virtual hosted style S3 URI: \'b\' (http://127.0.0.1:9/b/k.csv?X-Amz-Signature=[HIDDEN]) +ExceptionBeforeStart 0 Invalid S3 key: \'/a//b.csv?X-Amz-Algorithm=[HIDDEN]&X-Amz-Signature=[HIDDEN]\' (http://my-bucket-name.s3.amazonaws.com//a//b.csv?X-Amz-Algorithm=[HIDDEN]&X-Amz-Signature=[HIDDEN]) +ExceptionBeforeStart 0 Bucket name length is out of bounds in virtual hosted style S3 URI: \'bucketname?X-Amz-Signature=[HIDDEN]\' (http://127.0.0.1:9/bucketname?X-Amz-Signature=[HIDDEN]) +ExceptionBeforeStart 0 Bucket name length is out of bounds in virtual hosted style S3 URI: \'b\' (http://[HIDDEN]@127.0.0.1:9/b/x.csv) diff --git a/tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.sql b/tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.sql new file mode 100644 index 000000000000..cecd87c3589f --- /dev/null +++ b/tests/queries/0_stateless/05256_s3_uri_exception_masks_credentials.sql @@ -0,0 +1,33 @@ +-- Tags: no-fasttest +-- no-fasttest: needs the `s3` table function + +-- Exceptions thrown while validating an S3 URI must not reveal the credentials embedded in it: +-- the exception text is stored in `system.query_log`, which, unlike the query text, is not masked. + +SET log_queries = 1; + +-- Bucket name is too short. +SELECT count() FROM s3('http://s3user:SECRET_bucket@127.0.0.1:9/b/k.csv', 'CSV'); -- { serverError BAD_ARGUMENTS } +-- Key with consecutive slashes. +SELECT count() FROM s3('http://ku:SECRET_key@my-bucket-name.s3.amazonaws.com//a//b.csv', 'CSV'); -- { serverError BAD_ARGUMENTS } +-- Not a virtual-hosted-style URI. +SELECT count() FROM s3('http://vu:SECRET_virtual@127.0.0.1:9/bucketname/k.csv', 'CSV') SETTINGS s3_uri_style = 'virtual_hosted'; -- { serverError BAD_ARGUMENTS } +-- Not a path-style URI. +SELECT count() FROM s3('http://pu:SECRET_path@127.0.0.1:9', 'CSV') SETTINGS s3_uri_style = 'path'; -- { serverError BAD_ARGUMENTS } +-- The query parameters of a presigned URL. +SELECT count() FROM s3('http://127.0.0.1:9/b/k.csv?X-Amz-Signature=SECRET_signature', 'CSV'); -- { serverError BAD_ARGUMENTS } +-- With `compatibility_s3_presigned_url_query_in_path` the query of a presigned URL is folded into the path... +SELECT count() FROM s3('http://127.0.0.1:9/b/k.csv?X-Amz-Signature=SECRET_compat_uri', 'CSV') SETTINGS compatibility_s3_presigned_url_query_in_path = 1; -- { serverError BAD_ARGUMENTS } +-- ... so it ends up in the key... +SELECT count() FROM s3('http://my-bucket-name.s3.amazonaws.com//a//b.csv?X-Amz-Algorithm=AWS4-HMAC-SHA256&X-Amz-Signature=SECRET_compat_key', 'CSV') SETTINGS compatibility_s3_presigned_url_query_in_path = 1; -- { serverError BAD_ARGUMENTS } +-- ... or in the bucket. +SELECT count() FROM s3('http://127.0.0.1:9/bucketname?X-Amz-Signature=SECRET_compat_bucket_0123456789abcdef0123456789abcdef', 'CSV') SETTINGS compatibility_s3_presigned_url_query_in_path = 1; -- { serverError BAD_ARGUMENTS } +-- The same through a table engine. +CREATE TABLE s3_engine (x UInt8) ENGINE = S3('http://duser:SECRET_engine@127.0.0.1:9/b/x.csv', 'CSV'); -- { serverError BAD_ARGUMENTS } + +SYSTEM FLUSH LOGS query_log; + +SELECT type, position(exception, 'SECRET_') > 0 AS leaked, extract(exception, 'DB::Exception: (.*)\\. \\(BAD_ARGUMENTS\\)') AS message +FROM system.query_log +WHERE current_database = currentDatabase() AND type != 'QueryStart' AND exception_code = 36 +ORDER BY event_time_microseconds; From 26726010e8ca0e02f612dfdd9fe53ea08d6a03d1 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 20:48:24 +0000 Subject: [PATCH 076/185] Backport #119244 to 26.8: Do not log kafka secrets --- contrib/librdkafka-cmake/CMakeLists.txt | 3 +- .../include/chrdkafka_conf_sensitive.h | 14 +++++ .../librdkafka-cmake/rdkafka_conf_sensitive.c | 31 +++++++++++ src/Storages/Kafka/KafkaConfigLoader.cpp | 41 ++++++++++++--- .../test_storage_kafka/test_batch_fast.py | 51 +++++++++++++++++++ 5 files changed, 131 insertions(+), 9 deletions(-) create mode 100644 contrib/librdkafka-cmake/include/chrdkafka_conf_sensitive.h create mode 100644 contrib/librdkafka-cmake/rdkafka_conf_sensitive.c diff --git a/contrib/librdkafka-cmake/CMakeLists.txt b/contrib/librdkafka-cmake/CMakeLists.txt index 57ccb00086b0..b2945e968d8d 100644 --- a/contrib/librdkafka-cmake/CMakeLists.txt +++ b/contrib/librdkafka-cmake/CMakeLists.txt @@ -32,7 +32,7 @@ set(SRCS "${RDKAFKA_SOURCE_DIR}/rdkafka.c" "${RDKAFKA_SOURCE_DIR}/rdkafka_cert.c" "${RDKAFKA_SOURCE_DIR}/rdkafka_cgrp.c" - "${RDKAFKA_SOURCE_DIR}/rdkafka_conf.c" +# "${RDKAFKA_SOURCE_DIR}/rdkafka_conf.c" # compiled via the rdkafka_conf_sensitive.c wrapper below "${RDKAFKA_SOURCE_DIR}/rdkafka_coord.c" "${RDKAFKA_SOURCE_DIR}/rdkafka_error.c" "${RDKAFKA_SOURCE_DIR}/rdkafka_event.c" @@ -84,6 +84,7 @@ set(SRCS "${RDKAFKA_SOURCE_DIR}/rdkafka_transport.c" "${RDKAFKA_SOURCE_DIR}/rdkafka_txnmgr.c" "${RDKAFKA_SOURCE_DIR}/rdkafka_zstd.c" # WITH_ZSTD + "${CMAKE_CURRENT_SOURCE_DIR}/rdkafka_conf_sensitive.c" # ClickHouse wrapper compiling rdkafka_conf.c, see the comment in the file "${RDKAFKA_SOURCE_DIR}/rdlist.c" "${RDKAFKA_SOURCE_DIR}/rdlog.c" "${RDKAFKA_SOURCE_DIR}/rdmap.c" diff --git a/contrib/librdkafka-cmake/include/chrdkafka_conf_sensitive.h b/contrib/librdkafka-cmake/include/chrdkafka_conf_sensitive.h new file mode 100644 index 000000000000..83b99dce955c --- /dev/null +++ b/contrib/librdkafka-cmake/include/chrdkafka_conf_sensitive.h @@ -0,0 +1,14 @@ +/// See contrib/librdkafka-cmake/rdkafka_conf_sensitive.c +#pragma once + +#ifdef __cplusplus +extern "C" { +#endif + +/// The names of the configuration properties that librdkafka marks with the _RK_SENSITIVE flag, +/// i.e. whose values must not appear in logs, as a NULL-terminated array of static strings. +const char * const * chrd_kafka_conf_sensitive_properties(void); + +#ifdef __cplusplus +} +#endif diff --git a/contrib/librdkafka-cmake/rdkafka_conf_sensitive.c b/contrib/librdkafka-cmake/rdkafka_conf_sensitive.c new file mode 100644 index 000000000000..740f444ad0a1 --- /dev/null +++ b/contrib/librdkafka-cmake/rdkafka_conf_sensitive.c @@ -0,0 +1,31 @@ +/// Wrapper around librdkafka's rdkafka_conf.c: compiles the file (this is the only place it is +/// compiled, see the SRCS list in contrib/librdkafka-cmake/CMakeLists.txt) and adds an accessor +/// for the static rd_kafka_properties table, which is the source of truth for the _RK_SENSITIVE +/// flag. ClickHouse uses the flag to hide the values of sensitive properties (e.g. sasl.password) +/// in its logs. No public librdkafka API exposes the flag: the redacting dump +/// (rd_kafka_anyconf_dump with redact_sensitive) is static, and the public rd_kafka_conf_dump +/// does not redact. + +#include "rdkafka_conf.c" + +#include + +/* The table ends with a terminator entry, so the count is an upper bound and the + * array is always NULL-terminated. The names point into the static table. */ +static const char * chrd_sensitive_names[sizeof(rd_kafka_properties) / sizeof(*rd_kafka_properties)]; +static pthread_once_t chrd_sensitive_names_once = PTHREAD_ONCE_INIT; + +static void chrd_fill_sensitive_names(void) +{ + const struct rd_kafka_property * prop = NULL; + size_t n = 0; + for (prop = rd_kafka_properties; prop->name; prop++) + if (prop->scope & _RK_SENSITIVE) + chrd_sensitive_names[n++] = prop->name; +} + +const char * const * chrd_kafka_conf_sensitive_properties(void) +{ + pthread_once(&chrd_sensitive_names_once, chrd_fill_sensitive_names); + return chrd_sensitive_names; +} diff --git a/src/Storages/Kafka/KafkaConfigLoader.cpp b/src/Storages/Kafka/KafkaConfigLoader.cpp index ded88d65b90a..cb26fad59504 100644 --- a/src/Storages/Kafka/KafkaConfigLoader.cpp +++ b/src/Storages/Kafka/KafkaConfigLoader.cpp @@ -18,6 +18,7 @@ #include #include #include +#include #include #include @@ -552,6 +553,36 @@ void updateConfigurationFromConfig( } +namespace +{ + +/// Sensitive properties must not be logged in cleartext: the log records can reach not only the +/// server log, but also clients that set `send_logs_level`. +bool isSensitiveProperty(std::string_view name) +{ + /// The properties librdkafka marks with the _RK_SENSITIVE flag, plus a substring safety net + /// for properties unknown to the vendored librdkafka version. + static const std::unordered_set sensitive_properties = [] + { + std::unordered_set res; + for (const char * const * prop_name = chrd_kafka_conf_sensitive_properties(); *prop_name; ++prop_name) + res.emplace(*prop_name); + return res; + }(); + return sensitive_properties.contains(name) || name.contains("password") || name.contains("secret"); +} + +/// Log all properties of a Kafka client configuration, replacing the values of sensitive +/// properties, e.g. `sasl.password` or `sasl.oauthbearer.client.secret`, with `[HIDDEN]`. +void logConfigProperties(const cppkafka::Configuration & conf, const LoggerPtr & log, std::string_view client_type) +{ + for (const auto & property : conf.get_all()) + LOG_TRACE(log, "{} set property {}:{}", client_type, property.first, + isSensitiveProperty(property.first) ? "[HIDDEN]" : property.second); +} + +} + template cppkafka::Configuration KafkaConfigLoader::getConsumerConfiguration(TKafkaStorage & storage, const ConsumerConfigParams & params, IKafkaExceptionInfoSinkPtr exception_info_sink_ptr) { @@ -582,12 +613,7 @@ cppkafka::Configuration KafkaConfigLoader::getConsumerConfiguration(TKafkaStorag conf.set("enable.auto.offset.store", "false"); // Update offset automatically - to commit them all at once. conf.set("enable.partition.eof", "false"); // Ignore EOF messages - for (auto & property : conf.get_all()) - { - if (property.first.contains("password")) - continue; - LOG_TRACE(params.log, "Consumer set property {}:{}", property.first, property.second); - } + logConfigProperties(conf, params.log, "Consumer"); return conf; } @@ -608,8 +634,7 @@ cppkafka::Configuration KafkaConfigLoader::getProducerConfiguration(TKafkaStorag updateConfigurationFromConfig(loadProducerConfig, conf, storage, params); - for (auto & property : conf.get_all()) - LOG_TRACE(params.log, "Producer set property {}:{}", property.first, property.second); + logConfigProperties(conf, params.log, "Producer"); /// compression.codec is a global and topic level property, however compression.level is only a topic level property. /// cppkafka::Configuration::get_all returns the global properties only, so we need to check compression.level separately. diff --git a/tests/integration/test_storage_kafka/test_batch_fast.py b/tests/integration/test_storage_kafka/test_batch_fast.py index c36d42a7b0df..f55249b5a7d6 100644 --- a/tests/integration/test_storage_kafka/test_batch_fast.py +++ b/tests/integration/test_storage_kafka/test_batch_fast.py @@ -1931,6 +1931,57 @@ def test_kafka_producer_consumer_separate_settings( assert property_in_log in kafka_producer_applied_properties +@pytest.mark.parametrize( + "create_query_generator", + [ + k.generate_old_create_table_query, + k.generate_new_create_table_query, + ], +) +def test_kafka_password_not_logged(kafka_cluster, create_query_generator): + suffix = k.random_string(6) + kafka_table = f"kafka_{suffix}" + username = f"kafka_user_{suffix}" + password = f"secret_kafka_password_{suffix}" + + instance.rotate_logs() + instance.query( + create_query_generator( + kafka_table, + "key UInt64", + topic_list="password_not_logged", + consumer_group="test", + settings={ + "kafka_sasl_username": username, + "kafka_sasl_password": password, + }, + ) + ) + + # Create an mv to initialize the librdkafka consumers + instance.query(f"CREATE MATERIALIZED VIEW test.{kafka_table}_view ENGINE=MergeTree ORDER BY tuple() AS SELECT * FROM test.{kafka_table}") + instance.wait_for_log_line(f"{kafka_table}.*Created #0 consumer") + instance.query(f"DROP TABLE test.{kafka_table}_view") + instance.query(f"INSERT INTO test.{kafka_table} VALUES (1)") + + assert instance.contains_in_log(f"{kafka_table}.*Kafka producer created") + + # The property-logging loops ran for both the consumer and the producer, + # but they hid the values of the sensitive properties. `sasl.username` is + # hidden because librdkafka marks it with the _RK_SENSITIVE flag, not + # because of the name, so it validates the generated blacklist. + for client_type in ["Consumer", "Producer"]: + for property_name in ["sasl.username", "sasl.password"]: + assert instance.contains_in_log( + f"{kafka_table}.*{client_type} set property {property_name}:\\[HIDDEN\\]" + ) + # The username still appears in the logged CREATE TABLE text (only + # kafka_sasl_password is masked there), so check only the password value. + assert not instance.contains_in_log(password) + + instance.query(f"DROP TABLE test.{kafka_table}") + + @pytest.mark.parametrize( "create_query_generator, log_line", [ From 0540a1e3da97ea7429f97f78a22a80a094fb9947 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Tue, 29 Sep 2026 20:57:46 +0000 Subject: [PATCH 077/185] Backport #122423 to 26.8: Mask 'role_session_name' in system.query_log --- .../accessing-s3-data-securely.mdx | 1 + src/Backups/BackupInfo.cpp | 31 ++++-- src/Databases/DataLake/DataLakeConstants.h | 3 + src/Parsers/FunctionSecretArgumentsFinder.h | 28 ++++-- .../test_backup_restore_s3_role_arn/test.py | 8 +- tests/integration/test_database_glue/test.py | 93 +++++++++++++++++ .../test_storage_s3_queue/test_sts_smoke.py | 6 +- ...3_explicit_url_named_secret_mask.reference | 2 +- ...4510_s3_explicit_url_named_secret_mask.sql | 4 +- ...ake_catalog_hide_aws_external_id.reference | 8 +- ...0_datalake_catalog_hide_aws_external_id.sh | 18 ++-- ...edentials_role_session_name_mask.reference | 38 +++++++ ...tra_credentials_role_session_name_mask.sql | 84 ++++++++++++++++ ...ackup_persists_role_session_name.reference | 20 ++++ ...91_s3_backup_persists_role_session_name.sh | 99 +++++++++++++++++++ 15 files changed, 402 insertions(+), 41 deletions(-) create mode 100644 tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.reference create mode 100644 tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.sql create mode 100644 tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.reference create mode 100755 tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.sh diff --git a/docs/products/cloud/guides/data-sources/accessing-s3-data-securely.mdx b/docs/products/cloud/guides/data-sources/accessing-s3-data-securely.mdx index d8002f31a5f2..5c13a3cb8a13 100644 --- a/docs/products/cloud/guides/data-sources/accessing-s3-data-securely.mdx +++ b/docs/products/cloud/guides/data-sources/accessing-s3-data-securely.mdx @@ -141,6 +141,7 @@ DESCRIBE TABLE s3('https://s3.amazonaws.com/BUCKETNAME/BUCKETOBJECT.csv','CSVWit Below is an example query that uses the `role_session_name` as a shared secret to query data from a bucket. If the `role_session_name` isn't correct, this operation will fail. +Because it can act as a shared secret, ClickHouse masks the value of `role_session_name` as `[HIDDEN]` in `system.query_log`, `SHOW CREATE TABLE`, and everywhere else the query text is shown, the same way as `external_id`. ```sql DESCRIBE TABLE s3('https://s3.amazonaws.com/BUCKETNAME/BUCKETOBJECT.csv','CSVWithNames',extra_credentials(role_arn = 'arn:aws:iam::111111111111:role/ClickHouseAccessRole-001', role_session_name = 'secret-role-name')) diff --git a/src/Backups/BackupInfo.cpp b/src/Backups/BackupInfo.cpp index 755027d19075..5b68b55a10e4 100644 --- a/src/Backups/BackupInfo.cpp +++ b/src/Backups/BackupInfo.cpp @@ -11,7 +11,6 @@ #include #include #include -#include #include #include @@ -152,11 +151,22 @@ namespace return evaluated_literal->value.safeGet(); } - /// Rebuilds an `extra_credentials(...)` keeping only its non-secret keys, or `nullptr` if none are - /// left. `role_arn` and `role_session_name` only name the role to assume, which grants nothing - /// without the server's own identity and a matching trust policy, so `isNonSecretExtraCredentialsKey` - /// keeps them -- the same predicate that keeps them visible in a logged query. `external_id` is the - /// shared secret of the triple, and anything unclassifiable is dropped. + /// The `extra_credentials(...)` keys that the `` locator keeps in the `.backup` metadata: + /// the role identifiers `role_arn` and `role_session_name`. Every backup of a chain must be able to + /// reopen its base with the stored locator alone, and a trust policy may pin the session name through + /// `sts:RoleSessionName`, so dropping it would break such restores (see #116223). This deliberately + /// differs from the logged query text, where `role_session_name` is masked and only `role_arn` stays + /// visible (`FunctionSecretArgumentsFinder::isNonSecretExtraCredentialsKey`): the metadata file lives + /// in the user's own backup bucket, next to the data it protects. `external_id` is the shared secret + /// of the triple and is dropped, as is anything unclassifiable. + bool isBaseBackupRoleIdentifierKey(std::string_view key) + { + return key == "role_arn" || key == "role_session_name"; + } + + /// Rebuilds an `extra_credentials(...)` keeping only its role identifiers, or `nullptr` if none are + /// left. They only name the role to assume and its session, which grants nothing without the server's + /// own identity and a matching trust policy. ASTPtr withoutSecretExtraCredentials(const ASTPtr & function_arg, const ContextPtr & context) { const auto * func = function_arg->as(); @@ -168,7 +178,7 @@ namespace for (const auto & child : func->arguments->children) { auto key = getEffectiveKeyValueArgName(child, context); - if (!key || !FunctionSecretArgumentsFinder::isNonSecretExtraCredentialsKey(*key)) + if (!key || !isBaseBackupRoleIdentifierKey(*key)) continue; /// The value is only classified, not rewritten: an expression the open path resolves stays as /// it was written, so a locator that loses nothing serializes byte for byte. @@ -418,9 +428,10 @@ BackupInfo BackupInfo::withoutS3Credentials(ContextPtr context) const /// S3(collection, secret_access_key = '...') -> S3(collection) /// The keys are the `S3` authentication arguments consumed by `registerBackupEngineS3` - /// and `S3StorageParsedArguments::collectCredentials`, minus the non-secret role identifiers, which - /// stay so that a role-authenticated base backup remains openable. The key is resolved with the - /// context, so that an expression key (e.g. concat('secret_', 'access_key')) is recognized as well. + /// and `S3StorageParsedArguments::collectCredentials`, minus the role identifiers (`role_arn`, + /// `role_session_name`), which stay so that a role-authenticated base backup remains openable (see + /// isBaseBackupRoleIdentifierKey). The key is resolved with the context, so that an expression key + /// (e.g. concat('secret_', 'access_key')) is recognized as well. res.kv_args.erase( std::remove_if( res.kv_args.begin(), diff --git a/src/Databases/DataLake/DataLakeConstants.h b/src/Databases/DataLake/DataLakeConstants.h index bc5c96ac093f..93156b46fd0e 100644 --- a/src/Databases/DataLake/DataLakeConstants.h +++ b/src/Databases/DataLake/DataLakeConstants.h @@ -29,11 +29,14 @@ static inline std::unordered_map SETTINGS_TO_HIDE = {"aws_access_key_id", DEFAULT_MASKING_RULE}, {"aws_secret_access_key", DEFAULT_MASKING_RULE}, {"aws_external_id", DEFAULT_MASKING_RULE}, + /// A trust policy can require a specific session name (`sts:RoleSessionName`), so it is a secret too. + {"aws_role_session_name", DEFAULT_MASKING_RULE}, /// Legacy storage_* aliases (declared in DataLakeStorageSettings.h, originally for the Glue catalog) {"storage_catalog_credential", DEFAULT_MASKING_RULE}, {"storage_auth_header", DEFAULT_MASKING_RULE}, {"storage_aws_access_key_id", DEFAULT_MASKING_RULE}, {"storage_aws_secret_access_key", DEFAULT_MASKING_RULE}, + {"storage_aws_role_session_name", DEFAULT_MASKING_RULE}, /// OneLake credentials {"onelake_client_secret", DEFAULT_MASKING_RULE}, {"onelake_bearer_token", DEFAULT_MASKING_RULE}, diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index ee6b4922923b..821c215939aa 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -87,11 +87,17 @@ class FunctionSecretArgumentsFinder FunctionSecretArgumentsFinder::Result getResult() const { return result; } /// Whether a key of the `extra_credentials(..)` nested map carries a non-secret identifier whose - /// value stays visible when the map is masked (`role_arn` and `role_session_name`; the map's - /// secret is `external_id`). Any other key - unknown, malformed or an expression - fails closed. + /// value stays visible when the map is masked. Only `role_arn` qualifies: it names the role to + /// assume, like `access_key_id` names a key. The other two keys of the assume-role triple are + /// secrets: `external_id` is its shared secret, and `role_session_name` can be one too, because a + /// trust policy can require a specific value through the `sts:RoleSessionName` condition (the + /// ClickHouse Cloud guide documents exactly this use). Any other key - unknown, malformed or an + /// expression - fails closed. + /// The `.backup` metadata is a different matter: its `` locator keeps `role_session_name` + /// on purpose, so that a role-authenticated backup chain stays restorable (see `BackupInfo.cpp`). static bool isNonSecretExtraCredentialsKey(std::string_view key) { - return key == "role_arn" || key == "role_session_name"; + return key == "role_arn"; } protected: @@ -99,11 +105,12 @@ class FunctionSecretArgumentsFinder Result result; /// Named arguments carrying S3 secrets, shared by every S3 form (explicit-url and named-collection). - /// `external_id` is the shared secret of the assume-role triple; the other two (`role_arn`, - /// `role_session_name`) are non-secret identifiers passed inside `extra_credentials` and stay - /// visible (see isNonSecretExtraCredentialsKey). + /// `external_id` and `role_session_name` are the secrets of the assume-role triple; the third key + /// (`role_arn`) is a non-secret identifier passed inside `extra_credentials` and stays visible + /// (see isNonSecretExtraCredentialsKey). static constexpr std::string_view s3_secret_keys[] - = {"secret_access_key", "session_token", "google_adc_client_secret", "google_adc_refresh_token", "external_id"}; + = {"secret_access_key", "session_token", "google_adc_client_secret", "google_adc_refresh_token", "external_id", + "role_session_name"}; /// Named arguments carrying TLS credentials as the literal contents of a certificate or a key file, /// rather than as a path to it. They are secret and have to be hidden the same way a password is. @@ -120,9 +127,10 @@ class FunctionSecretArgumentsFinder void markSecretArgument(size_t index, bool argument_is_named = false); /// `headers(..)` and `extra_credentials(..)` are nested maps whose values are secret auth material - /// (`extra_credentials` carries the assume-role secret `external_id`; its non-secret identifiers - /// stay visible, see isNonSecretExtraCredentialsKey). The parsers accept them at any position, not - /// just at the tail. Record them so their values are hidden with the keys kept. + /// (`extra_credentials` carries the assume-role secrets `external_id` and `role_session_name`; its + /// non-secret identifier `role_arn` stays visible, see isNonSecretExtraCredentialsKey). The parsers + /// accept them at any position, not just at the tail. Record them so their values are hidden with + /// the keys kept. /// Idempotent: each map is recorded at most once. void maskNestedSecretMaps(); diff --git a/tests/integration/test_backup_restore_s3_role_arn/test.py b/tests/integration/test_backup_restore_s3_role_arn/test.py index a4ed1c43ca7d..8c4126518e29 100644 --- a/tests/integration/test_backup_restore_s3_role_arn/test.py +++ b/tests/integration/test_backup_restore_s3_role_arn/test.py @@ -9,9 +9,11 @@ from helpers.cluster import ClickHouseCluster from helpers.mock_servers import start_mock_servers -# `role_arn` and `role_session_name` are not secrets -- assuming the role still needs the server's -# own identity and a matching trust policy -- so they survive into the `` locator and -# let every hop of a chain reopen its base. The mock STS accepts the role and session name below. +# `role_arn` and `role_session_name` survive into the `` locator and let every hop of a +# chain reopen its base: assuming the role still needs the server's own identity and a matching trust +# policy. The session name is masked in logged query text, because a trust policy can pin it, but the +# stored locator keeps it on purpose (see `BackupInfo.cpp`). The mock STS accepts the role and +# session name below. # # Metadata written by a version that stripped the identifiers names no credentials and carries no # marker to reconstruct them from, so its base backup opens unauthenticated. Such a chain must still diff --git a/tests/integration/test_database_glue/test.py b/tests/integration/test_database_glue/test.py index 0254ff11fa8a..50098dc9bae6 100644 --- a/tests/integration/test_database_glue/test.py +++ b/tests/integration/test_database_glue/test.py @@ -1371,11 +1371,104 @@ def test_sts_smoke(started_cluster): result = node.query(f"SELECT sum(value) FROM {db_name_success}.`{root_namespace}.{table_name}`") assert result.strip() == "60", f"Expected sum to be 60 but got: {result}" + # `aws_role_session_name` can act as a shared secret (the trust policy can pin it, which is what the + # STS mock does), so every display surface hides it while `aws_role_arn` stays visible. This database + # carries no other secret, so `[HIDDEN]` can only come from the session name. + show_create = node.query(f"SHOW CREATE DATABASE {db_name_success}") + assert "miniorole" not in show_create + assert "arn::role" in show_create + assert "[HIDDEN]" in show_create + + engine_full = node.query( + f"SELECT engine_full FROM system.databases WHERE name = '{db_name_success}'" + ) + assert "miniorole" not in engine_full + assert "[HIDDEN]" in engine_full + + node.query("SYSTEM FLUSH LOGS system.query_log") + logged_create = node.query( + f"SELECT arrayStringConcat(groupArray(query), '\\n') FROM system.query_log " + f"WHERE query_kind = 'Create' AND type = 'QueryFinish' AND query LIKE '%{db_name_success}%'" + ) + assert "[HIDDEN]" in logged_create + assert "miniorole" not in logged_create + # Cleanup node.query(f"DROP DATABASE IF EXISTS {db_name_fail} SYNC") node.query(f"DROP DATABASE IF EXISTS {db_name_success} SYNC") +def test_sts_backup_restore_keeps_role_session_name(started_cluster): + """A backup archives the database definition with the real `aws_role_session_name`, although every + display surface shows `[HIDDEN]`. The STS mock grants the role only for the session name `miniorole`, + so the restored catalog can read its table only if the archived value was the real one.""" + node = started_cluster.instances["node1"] + + test_ref = f"test_sts_backup_{uuid.uuid4()}" + table_name = f"{test_ref}_table" + root_namespace = f"{test_ref}_namespace" + + catalog = load_catalog_impl(started_cluster) + catalog.create_namespace(root_namespace) + + schema = Schema( + NestedField(field_id=1, name="id", field_type=StringType(), required=False), + NestedField(field_id=2, name="value", field_type=DoubleType(), required=False), + ) + table = create_table(catalog, root_namespace, table_name, schema, PartitionSpec(), DEFAULT_SORT_ORDER, dir=table_name) + table.append( + pa.Table.from_pylist( + [ + {"id": "row1", "value": 10.0}, + {"id": "row2", "value": 20.0}, + {"id": "row3", "value": 30.0}, + ] + ) + ) + + db_name = f"db_backup_{test_ref.replace('-', '_')}" + restored_db_name = f"{db_name}_restored" + create_clickhouse_glue_database( + started_cluster, + node, + db_name, + additional_settings={ + "aws_role_arn": "arn::role", + "aws_role_session_name": "miniorole", + }, + query_settings={"s3_allow_server_credentials_in_user_queries": 1}, + with_credentials=False, + ) + result = node.query(f"SELECT sum(value) FROM {db_name}.`{root_namespace}.{table_name}`") + assert result.strip() == "60", f"Expected sum to be 60 but got: {result}" + + # A DataLakeCatalog database owns no tables, so the backup holds just the database definition. The + # default server config allows File backups under its `backups` directory. + backup = f"File('{db_name}')" + node.query(f"BACKUP DATABASE {db_name} TO {backup}") + node.query(f"DROP DATABASE IF EXISTS {restored_db_name} SYNC") + node.query( + f"RESTORE DATABASE {db_name} AS {restored_db_name} FROM {backup}", + settings={ + "allow_database_glue_catalog": 1, + "s3_allow_server_credentials_in_user_queries": 1, + }, + ) + + # Readable only with the real session name: the STS mock rejects any other value. + result = node.query(f"SELECT sum(value) FROM {restored_db_name}.`{root_namespace}.{table_name}`") + assert result.strip() == "60", f"Expected sum to be 60 but got: {result}" + + # The archived value came back, and the display of the restored definition is masked like any other. + show_create = node.query(f"SHOW CREATE DATABASE {restored_db_name}") + assert "miniorole" not in show_create + assert "arn::role" in show_create + assert "[HIDDEN]" in show_create + + node.query(f"DROP DATABASE IF EXISTS {restored_db_name} SYNC") + node.query(f"DROP DATABASE IF EXISTS {db_name} SYNC") + + def test_sts_smoke_no_opt_in(started_cluster): """A Glue DataLakeCatalog with aws_role_arn set, no explicit aws_access_key_id/aws_secret_access_key, and no s3_allow_server_credentials_in_user_queries opt-in. GlueCatalog routes through the same diff --git a/tests/integration/test_storage_s3_queue/test_sts_smoke.py b/tests/integration/test_storage_s3_queue/test_sts_smoke.py index 761606c336af..5b02ffebace7 100644 --- a/tests/integration/test_storage_s3_queue/test_sts_smoke.py +++ b/tests/integration/test_storage_s3_queue/test_sts_smoke.py @@ -111,14 +111,14 @@ def get_count(node, table_name): assert get_count(node, dst_table_name) == 10 assert ( - "extra_credentials(\\'role_arn\\' = \\'arn::role\\', \\'role_session_name\\' = \\'miniorole\\')" + "extra_credentials(\\'role_arn\\' = \\'arn::role\\', \\'role_session_name\\' = \\'[HIDDEN]\\')" in node.query(f"SHOW CREATE TABLE {table_name}") ) node.restart_clickhouse() assert ( - "extra_credentials(\\'role_arn\\' = \\'arn::role\\', \\'role_session_name\\' = \\'miniorole\\')" + "extra_credentials(\\'role_arn\\' = \\'arn::role\\', \\'role_session_name\\' = \\'[HIDDEN]\\')" in node.query(f"SHOW CREATE TABLE {table_name}") ) @@ -181,7 +181,7 @@ def test_s3_queue_extra_credentials_backup(started_cluster): # are equally invalid for MinIO, so a restored table that silently dropped # the clause would fail with the same error. Pin the round trip explicitly. assert ( - "extra_credentials(\\'role_arn\\' = \\'arn::role\\', \\'role_session_name\\' = \\'miniorole\\')" + "extra_credentials(\\'role_arn\\' = \\'arn::role\\', \\'role_session_name\\' = \\'[HIDDEN]\\')" in node.query(f"SHOW CREATE TABLE {table_name}") ) diff --git a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference index 0b581d50ec4a..5aa390bf9bb0 100644 --- a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference +++ b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.reference @@ -138,7 +138,7 @@ UNION id: 0, union_mode: UNION_ALL CONSTANT id: 14, constant_value: UInt64_1, constant_value_type: UInt8 JOIN TREE IDENTIFIER id: 15, identifier: system.one --- session_token, the Google ADC secrets (google_adc_client_secret, google_adc_refresh_token) and\n-- the extra_credentials assume-role material (external_id) passed to the explicit-url or\n-- named-collection S3 form must be masked like secret_access_key. Every secret value below is tagged\n-- so the final assertion can prove none of them leaks. They used to leak in plaintext in SHOW CREATE\n-- and logged query text.\n\n-- Engine form: SHOW CREATE hides every secret; the non-secret extra_credentials identifiers\n-- (role_arn, role_session_name) stay visible while external_id is hidden.\nDROP TABLE IF EXISTS t_04510; +-- session_token, the Google ADC secrets (google_adc_client_secret, google_adc_refresh_token) and\n-- the extra_credentials assume-role material (external_id) passed to the explicit-url or\n-- named-collection S3 form must be masked like secret_access_key. Every secret value below is tagged\n-- so the final assertion can prove none of them leaks. They used to leak in plaintext in SHOW CREATE\n-- and logged query text.\n\n-- Engine form: SHOW CREATE hides every secret; the non-secret extra_credentials identifier\n-- (role_arn) stays visible while external_id is hidden.\nDROP TABLE IF EXISTS t_04510; CREATE TABLE t_04510 (`x` UInt8) ENGINE = S3(\'http://localhost:11111/test/04510\', \'ak\', \'[HIDDEN]\', session_token = \'[HIDDEN]\', google_adc_client_secret = \'[HIDDEN]\', google_adc_refresh_token = \'[HIDDEN]\', extra_credentials(role_arn = \'visible_role_arn\', external_id = \'[HIDDEN]\'), format = \'TSV\') SHOW CREATE TABLE t_04510 SETTINGS format_display_secrets_in_show_and_select = 0; DROP TABLE t_04510; diff --git a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql index f3258d0805df..4f7eee6fb510 100644 --- a/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql +++ b/tests/queries/0_stateless/04510_s3_explicit_url_named_secret_mask.sql @@ -7,8 +7,8 @@ -- so the final assertion can prove none of them leaks. They used to leak in plaintext in SHOW CREATE -- and logged query text. --- Engine form: SHOW CREATE hides every secret; the non-secret extra_credentials identifiers --- (role_arn, role_session_name) stay visible while external_id is hidden. +-- Engine form: SHOW CREATE hides every secret; the non-secret extra_credentials identifier +-- (role_arn) stays visible while external_id is hidden. DROP TABLE IF EXISTS t_04510; CREATE TABLE t_04510 (x UInt8) ENGINE = S3('http://localhost:11111/test/04510', 'ak', 'SEKRIT_SAK', diff --git a/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.reference b/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.reference index dff69852207d..ba2bb9215f91 100644 --- a/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.reference +++ b/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.reference @@ -1,13 +1,13 @@ ---- default: aws_external_id hidden, role identifiers visible +--- default: aws_external_id and aws_role_session_name hidden, aws_role_arn visible aws_external_id = '[HIDDEN]' aws_access_key_id = '[HIDDEN]' aws_secret_access_key = '[HIDDEN]' aws_role_arn = 'arn:aws:iam::1:role/r' -aws_role_session_name = 'sess' +aws_role_session_name = '[HIDDEN]' OK: no secret in formatted query ---- show_secrets: aws_external_id visible +--- show_secrets: aws_external_id and aws_role_session_name visible aws_external_id = 'SECRET_THAT_MUST_NOT_LEAK' aws_access_key_id = 'SECRET_THAT_MUST_NOT_LEAK' aws_secret_access_key = 'SECRET_THAT_MUST_NOT_LEAK' aws_role_arn = 'arn:aws:iam::1:role/r' -aws_role_session_name = 'sess' +aws_role_session_name = 'SECRET_THAT_MUST_NOT_LEAK' diff --git a/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.sh b/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.sh index f33348fb92d4..d846eca5fa41 100755 --- a/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.sh +++ b/tests/queries/0_stateless/05030_datalake_catalog_hide_aws_external_id.sh @@ -1,8 +1,10 @@ #!/usr/bin/env bash -# Regression test: aws_external_id is the shared secret of the AWS AssumeRole triple, so it must be -# redacted as [HIDDEN] when a DataLakeCatalog CREATE query is formatted (system.databases.engine_full, -# SHOW CREATE DATABASE), while aws_role_arn and aws_role_session_name are non-secret identifiers that -# stay visible. Uses clickhouse-format so it needs no live catalog and is safe to run in parallel. +# Regression test: aws_external_id is the shared secret of the AWS AssumeRole triple, and +# aws_role_session_name can be one too (a trust policy can require a specific value through the +# sts:RoleSessionName condition), so both must be redacted as [HIDDEN] when a DataLakeCatalog CREATE +# query is formatted (system.databases.engine_full, SHOW CREATE DATABASE), while aws_role_arn is a +# non-secret identifier that stays visible. Uses clickhouse-format so it needs no live catalog and is +# safe to run in parallel. CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) # shellcheck source=../shell_config.sh @@ -17,7 +19,7 @@ aws_access_key_id = '${SECRET}', aws_secret_access_key = '${SECRET}', aws_external_id = '${SECRET}', aws_role_arn = 'arn:aws:iam::1:role/r', -aws_role_session_name = 'sess'" +aws_role_session_name = '${SECRET}'" show_settings() { local formatted="$1" @@ -32,9 +34,9 @@ show_settings() { done } -# Arm A: at the default the secret is redacted; arms C (non-secret identifiers stay visible) and +# Arm A: at the default the secrets are redacted; arms C (the non-secret role ARN stays visible) and # D (sibling AWS keys still redacted) are asserted by the same output. -echo "--- default: aws_external_id hidden, role identifiers visible" +echo "--- default: aws_external_id and aws_role_session_name hidden, aws_role_arn visible" formatted=$(echo "$query" | $CLICKHOUSE_FORMAT --oneline) show_settings "$formatted" if echo "$formatted" | grep -q "$SECRET"; then @@ -44,6 +46,6 @@ else fi # Arm B: an authorized caller can still retrieve the value, so the fix redacts rather than destroys. -echo "--- show_secrets: aws_external_id visible" +echo "--- show_secrets: aws_external_id and aws_role_session_name visible" formatted=$(echo "$query" | $CLICKHOUSE_FORMAT --oneline --show_secrets) show_settings "$formatted" diff --git a/tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.reference b/tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.reference new file mode 100644 index 000000000000..2d56e8b43279 --- /dev/null +++ b/tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.reference @@ -0,0 +1,38 @@ +CREATE TABLE default.t_05255\n(\n `x` UInt8\n)\nENGINE = S3(\'http://localhost:11111/test/05255\', \'ak\', \'[HIDDEN]\', \'TSV\', format = \'TSV\', extra_credentials(\'role_arn\' = \'visible_role_arn\', \'role_session_name\' = \'[HIDDEN]\', \'external_id\' = \'[HIDDEN]\')) +QUERY id: 0 + PROJECTION + LIST id: 1, nodes: 1 + MATCHER id: 2, matcher_type: ASTERISK + JOIN TREE + TABLE_FUNCTION id: 3, table_function_name: s3 + ARGUMENTS + LIST id: 4, nodes: 3 + CONSTANT id: 5, constant_value: \'http://localhost:11111/test/05255qt\', constant_value_type: String + CONSTANT id: 6, constant_value: \'Parquet\', constant_value_type: String + FUNCTION id: 7, function_name: extra_credentials, function_type: ordinary + ARGUMENTS + LIST id: 8, nodes: 2 + FUNCTION id: 9, function_name: equals, function_type: ordinary + ARGUMENTS + LIST id: 10, nodes: 2 + IDENTIFIER id: 11, identifier: role_arn + CONSTANT id: 12, constant_value: \'visible_role_arn\', constant_value_type: String + FUNCTION id: 13, function_name: equals, function_type: ordinary + ARGUMENTS + LIST id: 14, nodes: 2 + IDENTIFIER id: 15, identifier: role_session_name + CONSTANT id: 16, constant_value: [HIDDEN], constant_value_type: String +-- role_session_name inside extra_credentials(...) can act as a shared secret: a role\'s trust policy\n-- can require a specific value through the sts:RoleSessionName condition, and the ClickHouse Cloud\n-- guide documents exactly this use. It must therefore be masked like external_id, while role_arn is\n-- a non-secret identifier that stays visible. Every secret value below is tagged so the final\n-- assertion can prove none of them leaks. role_session_name used to be logged in plaintext.\n\n-- Engine form: SHOW CREATE hides role_session_name and external_id and keeps role_arn.\nDROP TABLE IF EXISTS t_05255; +CREATE TABLE t_05255 (`x` UInt8) ENGINE = S3(\'http://localhost:11111/test/05255\', \'ak\', \'[HIDDEN]\', extra_credentials(role_arn = \'visible_role_arn\', role_session_name = \'[HIDDEN]\', external_id = \'[HIDDEN]\'), format = \'TSV\') +SHOW CREATE TABLE t_05255 SETTINGS format_display_secrets_in_show_and_select = 0; +DROP TABLE t_05255; +SELECT * FROM s3(\'url_rsn\', \'Parquet\', extra_credentials(role_arn = \'arn:aws:iam::123456789012:role/visible_role\', role_session_name = \'[HIDDEN]\')) +SELECT * FROM s3(\'url_rsn_triple\', \'ak\', \'[HIDDEN]\', extra_credentials(role_session_name = \'[HIDDEN]\', role_arn = \'visible_role_arn\', external_id = \'[HIDDEN]\'), format = \'TSV\', structure = \'x UInt8\') +SELECT * FROM s3(\'url_rsn_ident\', \'ak\', \'[HIDDEN]\', extra_credentials(role_arn = visible_role_arn, role_session_name = \'[HIDDEN]\'), format = \'TSV\', structure = \'x UInt8\') +SELECT * FROM s3(nc_05255_missing, role_session_name = \'[HIDDEN]\', format = \'TSV\', structure = \'x UInt8\') +SELECT * FROM s3(nc_05255_missing, extra_credentials(role_arn = \'visible_role_arn\', role_session_name = \'[HIDDEN]\'), format = \'TSV\', structure = \'x UInt8\') +BACKUP TABLE nonexistent_05255 TO S3(\'url_bkp_rsn\', \'ak\', \'[HIDDEN]\', extra_credentials(equals(role_arn, \'visible_role_arn\'), role_session_name = \'[HIDDEN]\')) +BACKUP TABLE nonexistent_05255 TO S3(nc_bkp_05255_missing, role_session_name = \'[HIDDEN]\') +CREATE DATABASE db_05255_rsn ENGINE = Backup(\'\', S3(\'url_dbrsn\', \'ak\', \'[HIDDEN]\', extra_credentials(role_arn = \'visible_role_arn\', role_session_name = \'[HIDDEN]\'))) +EXPLAIN QUERY TREE run_passes = 0 SELECT * FROM s3(\'http://localhost:11111/test/05255qt\', \'Parquet\', extra_credentials(role_arn = \'visible_role_arn\', role_session_name = \'[HIDDEN]\')) +1 0 diff --git a/tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.sql b/tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.sql new file mode 100644 index 000000000000..2dde039ff107 --- /dev/null +++ b/tests/queries/0_stateless/05255_s3_extra_credentials_role_session_name_mask.sql @@ -0,0 +1,84 @@ +-- Tags: no-fasttest +-- no-fasttest: the S3 table engine is not available in the fast test build. + +-- role_session_name inside extra_credentials(...) can act as a shared secret: a role's trust policy +-- can require a specific value through the sts:RoleSessionName condition, and the ClickHouse Cloud +-- guide documents exactly this use. It must therefore be masked like external_id, while role_arn is +-- a non-secret identifier that stays visible. Every secret value below is tagged so the final +-- assertion can prove none of them leaks. role_session_name used to be logged in plaintext. + +-- Engine form: SHOW CREATE hides role_session_name and external_id and keeps role_arn. +DROP TABLE IF EXISTS t_05255; +CREATE TABLE t_05255 (x UInt8) +ENGINE = S3('http://localhost:11111/test/05255', 'ak', 'SEKRIT_SAK', + extra_credentials(role_arn = 'visible_role_arn', role_session_name = 'SEKRIT_RSN', external_id = 'SEKRIT_EID'), + format = 'TSV'); +SHOW CREATE TABLE t_05255 SETTINGS format_display_secrets_in_show_and_select = 0; +DROP TABLE t_05255; + +-- The forms below all fail at analysis (empty host / missing collection) before any network access, +-- and are logged with the secret replaced. Each carries a unique marker checked by the final assertion. + +-- Explicit-url function form in the shape users write it: no access key, the role alone authenticates. +SELECT * FROM s3('url_rsn', 'Parquet', + extra_credentials(role_arn = 'arn:aws:iam::123456789012:role/visible_role', role_session_name = 'SEKRIT_RSN')); -- { serverError BAD_ARGUMENTS } + +-- Together with the other two keys of the triple, at any position inside the map. +SELECT * FROM s3('url_rsn_triple', 'ak', 'SEKRIT_SAK', + extra_credentials(role_session_name = 'SEKRIT_RSNFIRST', role_arn = 'visible_role_arn', external_id = 'SEKRIT_EID'), + format = 'TSV', structure = 'x UInt8'); -- { serverError BAD_ARGUMENTS } + +-- The parser evaluates an identifier value as a literal, so an identifier carries the secret too. +SELECT * FROM s3('url_rsn_ident', 'ak', 'SEKRIT_SAK', + extra_credentials(role_arn = visible_role_arn, role_session_name = SEKRIT_IDRSN), + format = 'TSV', structure = 'x UInt8'); -- { serverError BAD_ARGUMENTS } + +-- Named-collection form: role_session_name as a named override of the collection. +SELECT * FROM s3(nc_05255_missing, role_session_name = 'SEKRIT_NCRSN', + format = 'TSV', structure = 'x UInt8'); -- { serverError NAMED_COLLECTION_DOESNT_EXIST } + +-- Named-collection form with the nested map. +SELECT * FROM s3(nc_05255_missing, extra_credentials(role_arn = 'visible_role_arn', role_session_name = 'SEKRIT_NCMAPRSN'), + format = 'TSV', structure = 'x UInt8'); -- { serverError NAMED_COLLECTION_DOESNT_EXIST } + +-- BACKUP ... TO S3: the explicit-url locator with the nested map, and the named-collection locator +-- with a named override. +BACKUP TABLE nonexistent_05255 TO S3('url_bkp_rsn', 'ak', 'SEKRIT_SAK', + extra_credentials(role_arn = 'visible_role_arn', role_session_name = 'SEKRIT_BKPRSN')); -- { serverError BAD_ARGUMENTS } +BACKUP TABLE nonexistent_05255 TO S3(nc_bkp_05255_missing, + role_session_name = 'SEKRIT_BKPNCRSN'); -- { serverError BAD_ARGUMENTS } + +-- The Backup database engine reconstructs the nested S3 destination. +CREATE DATABASE db_05255_rsn ENGINE = Backup('', S3('url_dbrsn', 'ak', 'SEKRIT_SAK', + extra_credentials(role_arn = 'visible_role_arn', role_session_name = 'SEKRIT_DBRSN'))); -- { serverError BAD_ARGUMENTS } + +-- The query-tree surface (EXPLAIN QUERY TREE) must hide the same value while keeping role_arn. +-- run_passes = 0 keeps the table function unresolved, so the collection and the storage are not touched. +SET enable_analyzer = 1; +EXPLAIN QUERY TREE run_passes = 0 SELECT * FROM s3('http://localhost:11111/test/05255qt', 'Parquet', extra_credentials(role_arn = 'visible_role_arn', role_session_name = 'SEKRIT_QTRSN')); + +SYSTEM FLUSH LOGS query_log; + +-- The exact logged text of every query above, in execution order: role_session_name must appear as +-- '[HIDDEN]' while role_arn and every other non-secret part stay visible verbatim. Each query has +-- exactly one terminal event: QueryFinish for the successful ones, an exception event for the +-- rejected ones. +SELECT query +FROM system.query_log +WHERE current_database = currentDatabase() + AND type != 'QueryStart' + AND query_kind != 'Set' -- sent by the test harness, not by this test + AND query NOT ILIKE 'SYSTEM FLUSH%' -- its own terminal event races with the flush it performs + AND query_id = initial_query_id -- only the statements issued here: a Replicated database logs + -- each DDL again from the replay worker, which inherits the + -- initiator's initial_query_id but gets a fresh query_id + AND event_date >= yesterday() AND event_time > now() - INTERVAL 5 MINUTE +ORDER BY event_time_microseconds; + +-- Assert the masking property over every row this test produced, replay rows included. +-- count() > 0 keeps an empty row set from passing vacuously. +SELECT count() > 0, countIf(query LIKE '%SEKRIT%') +FROM system.query_log +WHERE current_database = currentDatabase() + AND type != 'QueryStart' + AND event_date >= yesterday() AND event_time > now() - INTERVAL 5 MINUTE; diff --git a/tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.reference b/tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.reference new file mode 100644 index 000000000000..6b8a708e6562 --- /dev/null +++ b/tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.reference @@ -0,0 +1,20 @@ +-- archived Backup database definition keeps role_session_name (must be 1) +1 +-- archived Backup database definition contains no [HIDDEN] (must be 0) +0 +-- archived S3 table definition keeps role_session_name (must be 1) +1 +-- archived S3 table definition keeps external_id (must be 1) +1 +-- archived S3 table definition contains no [HIDDEN] (must be 0) +0 +-- restored Backup database reads its source table (must be 10) +10 +-- secret occurrences in SHOW CREATE DATABASE of the restored Backup database (must be 0) +0 +-- [HIDDEN] present in SHOW CREATE DATABASE of the restored Backup database (must be 1) +1 +-- secret occurrences in SHOW CREATE TABLE of the restored S3 table (must be 0) +0 +-- [HIDDEN] present in SHOW CREATE TABLE of the restored S3 table (must be 1) +1 diff --git a/tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.sh b/tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.sh new file mode 100755 index 000000000000..606883238807 --- /dev/null +++ b/tests/queries/0_stateless/05291_s3_backup_persists_role_session_name.sh @@ -0,0 +1,99 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-encrypted-storage +# Tag no-fasttest: requires the S3 endpoint +# Tag no-encrypted-storage: a backup from an encrypted disk restores only to an encrypted disk, so the restored Backup database gets no parts. + +# `role_session_name` is shown as [HIDDEN] wherever a definition is displayed. A backup must still +# archive the definition with the real value, because RESTORE recreates the object from the archived +# text: an S3 table or a Backup database restored with the literal '[HIDDEN]' would assume its role with +# the wrong session name. The archived definitions are read back from the backup itself, so the check +# does not go through the display layer that masks them. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +client_opts=( + --allow_repeated_settings + --send_logs_level 'error' +) + +src=${CLICKHOUSE_DATABASE}_src +view=${CLICKHOUSE_DATABASE}_view +view_restored=${CLICKHOUSE_DATABASE}_view_restored +tbl=${CLICKHOUSE_DATABASE}.t_rsn +tbl_restored=${CLICKHOUSE_DATABASE}.t_rsn_restored + +inner_url="http://localhost:11111/test/backups/${CLICKHOUSE_DATABASE}/rsn_inner" +outer_url="http://localhost:11111/test/backups/${CLICKHOUSE_DATABASE}/rsn_outer" +# Access key id 'test', secret 'testtest': the credentials the stateless suite uses for S3. +inner="S3('${inner_url}', 'test', 'testtest')" +# The Backup database mounts the inner backup with the same key pair plus a role_session_name and no +# role_arn: nothing assumes a role, so no STS endpoint is needed, while the value still travels through +# every place a locator is archived. +inner_with_session="S3('${inner_url}', 'test', 'testtest', extra_credentials(role_session_name = 'SEKRIT_DBRSN'))" +outer="S3('${outer_url}', 'test', 'testtest')" + +# The S3 table carries the full assume-role triple. Nothing reads it here, so no STS endpoint is needed +# either: the table function connects lazily, on the first read. +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP DATABASE IF EXISTS ${src}; +DROP DATABASE IF EXISTS ${view}; +DROP DATABASE IF EXISTS ${view_restored}; +DROP TABLE IF EXISTS ${tbl}; +DROP TABLE IF EXISTS ${tbl_restored}; +CREATE DATABASE ${src}; +CREATE TABLE ${src}.t (id UInt64) ENGINE = MergeTree ORDER BY id; +INSERT INTO ${src}.t SELECT number FROM numbers(10); +BACKUP DATABASE ${src} TO ${inner} FORMAT Null; +CREATE DATABASE ${view} ENGINE = Backup('${src}', ${inner_with_session}); +CREATE TABLE ${tbl} (id UInt64) ENGINE = S3('http://localhost:11111/test/${CLICKHOUSE_DATABASE}/t_rsn', 'ak', 'sk', + extra_credentials(role_arn = 'arn::role', role_session_name = 'SEKRIT_TRSN', external_id = 'SEKRIT_TEID'), format = 'CSV'); +BACKUP DATABASE ${view}, TABLE ${tbl} TO ${outer} FORMAT Null; +" + +# Reads one archived definition out of the outer backup, bypassing every display surface. +function archived() +{ + local path_in_backup=$1 && shift + ${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q \ + "SELECT line FROM s3('${outer_url}/${path_in_backup}', 'test', 'testtest', 'LineAsString') FORMAT TSVRaw" +} + +echo '-- archived Backup database definition keeps role_session_name (must be 1)' +archived "metadata/${view}.sql" | grep -c SEKRIT_DBRSN || true +echo '-- archived Backup database definition contains no [HIDDEN] (must be 0)' +archived "metadata/${view}.sql" | grep -c HIDDEN || true +echo '-- archived S3 table definition keeps role_session_name (must be 1)' +archived "metadata/${CLICKHOUSE_DATABASE}/t_rsn.sql" | grep -c SEKRIT_TRSN || true +echo '-- archived S3 table definition keeps external_id (must be 1)' +archived "metadata/${CLICKHOUSE_DATABASE}/t_rsn.sql" | grep -c SEKRIT_TEID || true +echo '-- archived S3 table definition contains no [HIDDEN] (must be 0)' +archived "metadata/${CLICKHOUSE_DATABASE}/t_rsn.sql" | grep -c HIDDEN || true + +# The round trip: both objects come back from the archived text. The restored Backup database mounts +# the inner backup through its archived locator, so it is readable only if that locator was archived +# as written. +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q \ + "RESTORE DATABASE ${view} AS ${view_restored}, TABLE ${tbl} AS ${tbl_restored} FROM ${outer} FORMAT Null" + +echo '-- restored Backup database reads its source table (must be 10)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SELECT count() FROM ${view_restored}.t" + +# On display the restored definitions are masked like any other: the archived value never shows. +echo '-- secret occurrences in SHOW CREATE DATABASE of the restored Backup database (must be 0)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE DATABASE ${view_restored}" | grep -c SEKRIT || true +echo '-- [HIDDEN] present in SHOW CREATE DATABASE of the restored Backup database (must be 1)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE DATABASE ${view_restored}" | grep -c -m1 '\[HIDDEN\]' +echo '-- secret occurrences in SHOW CREATE TABLE of the restored S3 table (must be 0)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE TABLE ${tbl_restored}" | grep -c SEKRIT || true +echo '-- [HIDDEN] present in SHOW CREATE TABLE of the restored S3 table (must be 1)' +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -q "SHOW CREATE TABLE ${tbl_restored}" | grep -c -m1 '\[HIDDEN\]' + +${CLICKHOUSE_CLIENT} "${client_opts[@]}" -m -q " +DROP TABLE ${tbl_restored}; +DROP TABLE ${tbl}; +DROP DATABASE ${view_restored}; +DROP DATABASE ${view}; +DROP DATABASE ${src}; +" From ba36b14823e91e1a9a2f44ab1ab91baca05863a4 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 30 Sep 2026 07:53:08 +0000 Subject: [PATCH 078/185] Backport #122767 to 26.8: Keeper: reply session expired to a client that asks to continue its session --- src/Server/KeeperTCPHandler.cpp | 26 ++++++++---- src/Server/KeeperTCPHandler.h | 10 ++++- tests/integration/test_keeper_session/test.py | 42 ++++++++++++++++++- .../src/jepsen/clickhouse/keeper/counter.clj | 9 ++-- .../src/jepsen/clickhouse/keeper/queue.clj | 6 ++- .../src/jepsen/clickhouse/keeper/set.clj | 10 +++-- .../src/jepsen/clickhouse/keeper/utils.clj | 10 +++++ 7 files changed, 94 insertions(+), 19 deletions(-) diff --git a/src/Server/KeeperTCPHandler.cpp b/src/Server/KeeperTCPHandler.cpp index 3ae104728940..1235b44bb4a2 100644 --- a/src/Server/KeeperTCPHandler.cpp +++ b/src/Server/KeeperTCPHandler.cpp @@ -276,10 +276,10 @@ KeeperTCPHandler::KeeperTCPHandler( } } -void KeeperTCPHandler::sendHandshake(bool has_leader, bool & use_compression) +void KeeperTCPHandler::sendHandshake(HandshakeResult result, bool & use_compression) { Coordination::write(Coordination::SERVER_HANDSHAKE_LENGTH, *out); - if (has_leader) + if (result != HandshakeResult::Rejected) { if (expect_opentelemetry_tracing_context) Coordination::write(Coordination::ZOOKEEPER_PROTOCOL_VERSION_WITH_TRACING, *out); @@ -298,8 +298,11 @@ void KeeperTCPHandler::sendHandshake(bool has_leader, bool & use_compression) Coordination::write(Coordination::KEEPER_PROTOCOL_VERSION_CONNECTION_REJECT, *out); } - Coordination::write(static_cast(session_timeout.totalMilliseconds()), *out); - Coordination::write(session_id, *out); + /// A zero timeout with a zero session id tells a ZooKeeper client that its session has expired. + const bool expired = result == HandshakeResult::SessionExpired; + Coordination::write(expired ? int32_t{0} : static_cast(session_timeout.totalMilliseconds()), *out); + /// A rejected client has no session, and would send any non-zero id back as the session to continue. + Coordination::write(result == HandshakeResult::Accepted ? session_id : int64_t{0}, *out); std::array passwd{}; Coordination::write(passwd, *out); out->next(); @@ -315,7 +318,6 @@ Poco::Timespan KeeperTCPHandler::receiveHandshake(int32_t handshake_length, bool int32_t protocol_version = 0; int64_t last_zxid_seen = 0; int32_t timeout_ms = 0; - int64_t previous_session_id = 0; /// We don't support session restore. So previous session_id is always zero. std::array passwd {}; if (!isHandShake(handshake_length)) @@ -465,6 +467,14 @@ void KeeperTCPHandler::runImpl() if (keeper_dispatcher->isTCPConnectionDrainStarted() || keeper_dispatcher->isShuttingDown()) return; + /// Keeper cannot restore sessions, and a new session in place of the old one would go unnoticed by the client. + if (previous_session_id != 0) + { + LOG_INFO(log, "Client asked to continue session {}, which cannot be restored, replying that it has expired", previous_session_id); + sendHandshake(HandshakeResult::SessionExpired, use_compression); + return; + } + if (keeper_dispatcher->isServerActive()) { try @@ -476,7 +486,7 @@ void KeeperTCPHandler::runImpl() catch (const Exception & e) { LOG_WARNING(log, "Cannot receive session id {}", e.displayText()); - sendHandshake(/* has_leader */ false, use_compression); + sendHandshake(HandshakeResult::Rejected, use_compression); return; } @@ -489,12 +499,12 @@ void KeeperTCPHandler::runImpl() return; } - sendHandshake(/* has_leader */ true, use_compression); + sendHandshake(HandshakeResult::Accepted, use_compression); } else { LOG_WARNING(log, "Ignoring user request, because the server is not active yet"); - sendHandshake(/* has_leader */ false, use_compression); + sendHandshake(HandshakeResult::Rejected, use_compression); return; } diff --git a/src/Server/KeeperTCPHandler.h b/src/Server/KeeperTCPHandler.h index 820306a41fed..6e1b9e5e96b7 100644 --- a/src/Server/KeeperTCPHandler.h +++ b/src/Server/KeeperTCPHandler.h @@ -77,6 +77,8 @@ class KeeperTCPHandler : public Poco::Net::TCPServerConnection Poco::Timespan max_session_timeout; Poco::Timespan session_timeout; int64_t session_id{-1}; + /// Session the client asked to continue in its handshake, 0 for a new session. + int64_t previous_session_id{0}; Stopwatch session_stopwatch; SocketInterruptablePollWrapperPtr poll_wrapper; Poco::Timespan send_timeout; @@ -105,7 +107,13 @@ class KeeperTCPHandler : public Poco::Net::TCPServerConnection void cancelWriteBuffer() noexcept; ReadBuffer & getReadBuffer(); - void sendHandshake(bool has_leader, bool & use_compression); + enum class HandshakeResult + { + Accepted, + Rejected, + SessionExpired, + }; + void sendHandshake(HandshakeResult result, bool & use_compression); Poco::Timespan receiveHandshake(int32_t handshake_length, bool & use_compression); static bool isHandShake(int32_t handshake_length); diff --git a/tests/integration/test_keeper_session/test.py b/tests/integration/test_keeper_session/test.py index 07062f42f5e6..213e1f3b22bc 100644 --- a/tests/integration/test_keeper_session/test.py +++ b/tests/integration/test_keeper_session/test.py @@ -36,6 +36,9 @@ reply_header_struct = struct.Struct("!iqi") stat_struct = struct.Struct("!qqqqiiiqiiq") +# Protocol version Keeper sends when it rejects a connection. +KEEPER_PROTOCOL_VERSION_CONNECTION_REJECT = 42 + @pytest.fixture(scope="module") def started_cluster(): @@ -86,7 +89,9 @@ def read_buffer(bytes, offset): return bytes[index : index + length], offset -def handshake(node_name=node1.name, session_timeout=1000, session_id=0): +def handshake( + node_name=node1.name, session_timeout=1000, session_id=0, full_reply=False +): client = None try: client = get_keeper_socket(node_name) @@ -130,6 +135,12 @@ def handshake(node_name=node1.name, session_timeout=1000, session_id=0): read_only = False print("negotiated_timeout - session_id", negotiated_timeout, session_id) + if full_reply: + (reply_length,) = int_struct.unpack_from(data, 0) + rest = data[int_struct.size + reply_length :] + while chunk := client.recv(1_000): + rest += chunk + return proto_version, negotiated_timeout, session_id, rest return negotiated_timeout, session_id finally: if client is not None: @@ -149,6 +160,35 @@ def test_session_timeout(started_cluster): assert negotiated_timeout == 10000 +def test_handshake_to_continue_session_is_expired(started_cluster): + wait_nodes() + negotiated_timeout, session_id = handshake( + node1.name, session_timeout=8000, session_id=0 + ) + assert negotiated_timeout == 8000 and session_id > 0 + # Keeper cannot restore a session, so a request to continue one must not be answered with a new session. + # The expired reply is the last thing the server sends before it closes the connection. + assert handshake( + node1.name, session_timeout=8000, session_id=session_id, full_reply=True + ) == (0, 0, 0, b"") + + try: + node2.stop_clickhouse() + node3.stop_clickhouse() + keeper_utils.wait_until_quorum_lost(cluster, node1) + # A rejected client gets no session id that it would send back as the session to continue. + assert handshake( + node1.name, session_timeout=8000, session_id=0, full_reply=True + ) == (KEEPER_PROTOCOL_VERSION_CONNECTION_REJECT, 8000, 0, b"") + assert handshake( + node1.name, session_timeout=8000, session_id=session_id, full_reply=True + ) == (0, 0, 0, b"") + finally: + node2.start_clickhouse() + node3.start_clickhouse() + wait_nodes() + + def test_session_close_shutdown(started_cluster): wait_nodes() diff --git a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/counter.clj b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/counter.clj index 160599290aae..cc8e3f678a10 100644 --- a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/counter.clj +++ b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/counter.clj @@ -34,9 +34,12 @@ :type :ok :value (long (count (zk-list conn root-path)))) (catch Exception _ (assoc op :type :info, :error :connect-error))) - :final-read (chu/exec-with-retries 30 (fn [] (assoc op - :type :ok - :value (long (count (zk-list conn root-path)))))) + :final-read (chu/exec-with-retries 30 (fn [] + (with-fresh-conn nodename (:with-auth test) + (fn [conn] + (assoc op + :type :ok + :value (long (count (zk-list conn root-path)))))))) :add (try (do (zk-multi-create-many-seq-nodes conn (concat-path root-path "seq-") (:value op) :with-acl (:with-auth test)) diff --git a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/queue.clj b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/queue.clj index d26849ee127e..be4b3f1481a6 100644 --- a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/queue.clj +++ b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/queue.clj @@ -45,8 +45,10 @@ :drain ; drain via delete is to long, just list all nodes (chu/exec-with-retries 30 (fn [] - (zk-sync conn) - (assoc op :type :ok :value (into #{} (map #(str %1) (zk-list conn root-path)))))))) + (with-fresh-conn nodename (:with-auth test) + (fn [conn] + (zk-sync conn) + (assoc op :type :ok :value (into #{} (map #(str %1) (zk-list conn root-path)))))))))) (teardown! [_ test]) diff --git a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/set.clj b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/set.clj index 87b4f91b5863..afe25f28a809 100644 --- a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/set.clj +++ b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/set.clj @@ -25,10 +25,12 @@ (invoke! [this test op] (case (:f op) :read (chu/exec-with-retries 30 (fn [] - (zk-sync conn) - (assoc op - :type :ok - :value (read-string (:data (zk-get-str conn k)))))) + (with-fresh-conn nodename (:with-auth test) + (fn [conn] + (zk-sync conn) + (assoc op + :type :ok + :value (read-string (:data (zk-get-str conn k)))))))) :add (try (do (zk-add-to-set conn k (:value op)) diff --git a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/utils.clj b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/utils.clj index 9947973e51ae..cfdf7d1a829a 100644 --- a/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/utils.clj +++ b/tests/jepsen.clickhouse/src/jepsen/clickhouse/keeper/utils.clj @@ -32,6 +32,16 @@ conn) conn))) +(defn with-fresh-conn + "Calls (f conn) on a new connection to node and closes it. Keeper cannot continue a session, + so a worker's connection that dropped after its last operation is expired." + [node with-auth f] + (let [conn (zk-connect node 9181 30000 with-auth)] + (try + (f conn) + (finally + (zk/close conn))))) + (defn zk-create-range [conn n & {:keys [with-acl] :or {with-acl false}}] (dorun (map (fn [v] (zk/create-all conn v From d94783f341100c9c84a27cc5e11849a7b0deff20 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 30 Sep 2026 08:19:18 +0000 Subject: [PATCH 079/185] Backport #122829 to 26.8: Fix crash on invalid Iceberg manifest-list entries --- .../DataLakes/Iceberg/IcebergWrites.cpp | 66 +++++++++---- .../ObjectStorage/DataLakes/Iceberg/Utils.cpp | 7 ++ ...g_malformed_manifest_list_writer.reference | 4 + ..._iceberg_malformed_manifest_list_writer.sh | 96 +++++++++++++++++++ 4 files changed, 157 insertions(+), 16 deletions(-) create mode 100644 tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.reference create mode 100755 tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.sh diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp index 95e04d42d5d8..cb62e186cfbd 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/IcebergWrites.cpp @@ -932,32 +932,66 @@ void generateManifestList( forEachAvroEntry(resolved_manifest_list_path, object_storage, context, "IcebergWrites", [&](const avro::GenericDatum & datum) { + if (datum.type() != avro::AVRO_RECORD) + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Manifest list {} contains an entry with Avro type {}, but a record is required", + resolved_manifest_list_path, + static_cast(datum.type())); + const avro::GenericRecord & old_entry = datum.value(); + + auto validate_field_type = [&](const String & field_name, avro::Type expected_type) -> const avro::GenericDatum & + { + if (!old_entry.hasField(field_name)) + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Manifest list {} entry is missing required field '{}'", + resolved_manifest_list_path, + field_name); + + const avro::GenericDatum & field = old_entry.field(field_name); + if (field.type() != expected_type) + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Manifest list {} field '{}' has Avro type {}, but type {} is required", + resolved_manifest_list_path, + field_name, + static_cast(field.type()), + static_cast(expected_type)); + + return field; + }; + + const avro::GenericDatum & old_manifest_path = validate_field_type(Iceberg::f_manifest_path, avro::AVRO_STRING); + /// When a path filter is supplied, copy only the matching entries. if (!carry_forward_manifest_paths.empty() - && !carry_forward_manifest_paths.contains(old_entry.field(Iceberg::f_manifest_path).value())) + && !carry_forward_manifest_paths.contains(old_manifest_path.value())) return; avro::GenericDatum new_datum(schema.root()); avro::GenericRecord & new_entry = new_datum.value(); - new_entry.field(f_manifest_path) = old_entry.field(Iceberg::f_manifest_path); - new_entry.field(f_manifest_length) = old_entry.field(Iceberg::f_manifest_length); - new_entry.field(f_partition_spec_id) = old_entry.field(Iceberg::f_partition_spec_id); + + auto copy_required_field = [&](const String & field_name, avro::Type expected_type) + { + new_entry.field(field_name) = validate_field_type(field_name, expected_type); + }; + + new_entry.field(f_manifest_path) = old_manifest_path; + copy_required_field(Iceberg::f_manifest_length, avro::AVRO_LONG); + copy_required_field(Iceberg::f_partition_spec_id, avro::AVRO_INT); /// iceberg-spark changed `f_added_snapshot_id` from 'null, long' to 'long' (apache/iceberg#11626); rewrite with the new schema in case we read the old type. if (old_entry.hasField(Iceberg::f_added_snapshot_id)) { const avro::GenericDatum & old_added_snapshot_id_entry = old_entry.field(Iceberg::f_added_snapshot_id); - if (old_added_snapshot_id_entry.isUnion()) - { - if (old_added_snapshot_id_entry.unionBranch() == 0) /// it means add_snapshot_id is null - { - /// This only happens when we read data written by a old version of iceberg, which violates the spec of iceberg. - throw Exception( - ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, - "Manifest list {} has null value for field '{}', but it is required", - resolved_manifest_list_path, - Iceberg::f_added_snapshot_id); - } - } + if (old_added_snapshot_id_entry.type() != avro::AVRO_LONG) + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Manifest list {} field '{}' has Avro type {}, but a non-null long is required", + resolved_manifest_list_path, + Iceberg::f_added_snapshot_id, + static_cast(old_added_snapshot_id_entry.type())); + new_entry.field(f_added_snapshot_id) = old_added_snapshot_id_entry.value(); } else diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp index 497f559d7c95..57f604b78c41 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/Utils.cpp @@ -1610,6 +1610,13 @@ void forEachAvroEntry( auto reader_base = std::make_unique(std::move(input_stream), MAX_AVRO_SCHEMA_DEPTH); avro::DataFileReader reader(std::move(reader_base)); + if (reader.readerSchema().root()->type() != avro::AVRO_RECORD) + throw Exception( + ErrorCodes::ICEBERG_SPECIFICATION_VIOLATION, + "Avro file {} has root schema type {}, but Iceberg manifest-list entries must be records", + filename, + static_cast(reader.readerSchema().root()->type())); + avro::GenericDatum datum(reader.readerSchema()); while (reader.read(datum)) callback(datum); diff --git a/tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.reference b/tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.reference new file mode 100644 index 000000000000..2e9db112b775 --- /dev/null +++ b/tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.reference @@ -0,0 +1,4 @@ +--- malformed manifest list is rejected cleanly --- +ICEBERG_SPECIFICATION_VIOLATION +--- stateless server is still alive --- +1 diff --git a/tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.sh b/tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.sh new file mode 100755 index 000000000000..069f6e0f755f --- /dev/null +++ b/tests/queries/0_stateless/05292_iceberg_malformed_manifest_list_writer.sh @@ -0,0 +1,96 @@ +#!/usr/bin/env bash +# Tags: no-fasttest +# Tag no-fasttest: requires `IcebergLocal` (USE_AVRO build option) and Iceberg writes. + +# Regression test for a malformed manifest list reaching the Iceberg writer carry-forward path. +# `clickhouse local` contains the pre-fix fatal signal to a short-lived subprocess. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +WORK_DIR="${CLICKHOUSE_TMP}/iceberg_malformed_manifest_list_writer_${CLICKHOUSE_TEST_UNIQUE_NAME}" +LOCAL_PATH="${WORK_DIR}/db" +TABLE_ROOT="${WORK_DIR}/iceberg/t0" + +rm -rf "${WORK_DIR}" +mkdir -p "${LOCAL_PATH}" "${TABLE_ROOT}" +trap 'rm -rf "${WORK_DIR}"' EXIT + +${CLICKHOUSE_LOCAL} \ + --path "${LOCAL_PATH}" \ + --allow_insert_into_iceberg=1 \ + --multiquery \ + --query " + CREATE TABLE t0 (x Int32) ENGINE = IcebergLocal('${TABLE_ROOT}/'); + INSERT INTO t0 VALUES (1), (2); + " -- --user_files_path="${WORK_DIR}" + +MANIFEST_LIST=$(find "${TABLE_ROOT}/metadata" -maxdepth 1 -name 'snap-*.avro' -type f | sort | head -1) +if [ -z "${MANIFEST_LIST}" ]; then + echo "manifest list not found" + exit 1 +fi + +MALFORMED_MANIFEST_LIST_HEX=$(python3 - <<'PY' +def write_long(value): + value = (value << 1) ^ (value >> 63) + out = bytearray() + while True: + byte = value & 0x7F + value >>= 7 + out.append(byte | 0x80 if value else byte) + if not value: + return bytes(out) + + +def write_bytes(value): + return write_long(len(value)) + value + + +schema = b'{"type":"array","items":"int"}' +sync = bytes.fromhex('7d4f6998c268758353f227bc9dba7b7e') +metadata = [(b'avro.codec', b'null'), (b'avro.schema', schema)] + +header = bytearray(b'Obj\x01') +header += write_long(len(metadata)) +for key, value in metadata: + header += write_bytes(key) + write_bytes(value) +header += write_long(0) + sync + +# One root object: an Avro array containing one int value, `[1]`. +datum = write_long(1) + write_long(1) + write_long(0) +block = write_long(1) + write_long(len(datum)) + datum + sync +print((bytes(header) + block).hex()) +PY +) + +echo '--- malformed manifest list is rejected cleanly ---' +output=$( + ${CLICKHOUSE_LOCAL} \ + --path "${LOCAL_PATH}" \ + --allow_insert_into_iceberg=1 \ + --engine_file_truncate_on_insert=1 \ + --multiquery \ + --query " + SELECT count() FROM t0 FORMAT Null; + INSERT INTO FUNCTION file('${MANIFEST_LIST}', RawBLOB) SELECT unhex('${MALFORMED_MANIFEST_LIST_HEX}'); + INSERT INTO t0 VALUES (3); + " -- --user_files_path="${WORK_DIR}" 2>&1 +) +status=$? + +if echo "${output}" | grep -qF 'ICEBERG_SPECIFICATION_VIOLATION'; then + echo 'ICEBERG_SPECIFICATION_VIOLATION' +elif [ "${status}" -eq 139 ]; then + echo 'SIGSEGV' +elif echo "${output}" | grep -qF 'Received signal'; then + echo 'Received signal' +elif echo "${output}" | grep -qF 'Segmentation fault'; then + echo 'Segmentation fault' +else + echo "${output}" | grep -E -m1 'Code:|Exception|signal|Segmentation|Aborted' || echo "exit status ${status}" +fi + +echo '--- stateless server is still alive ---' +${CLICKHOUSE_CLIENT} --query "SELECT 1" From 695d3810bca80f994f351385f049a7fd9a12eeb7 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 30 Sep 2026 08:59:38 +0000 Subject: [PATCH 080/185] Backport #122693 to 26.8: Parse some parquet metadata lazily --- .../Formats/Impl/Parquet/Reader.cpp | 28 +++++++++++++------ 1 file changed, 19 insertions(+), 9 deletions(-) diff --git a/src/Processors/Formats/Impl/Parquet/Reader.cpp b/src/Processors/Formats/Impl/Parquet/Reader.cpp index 1ee01ab2bbd5..d5ba590006c6 100644 --- a/src/Processors/Formats/Impl/Parquet/Reader.cpp +++ b/src/Processors/Formats/Impl/Parquet/Reader.cpp @@ -432,16 +432,25 @@ void Reader::prefilterAndInitRowGroups(const std::optional clickhouse_to_parquet_name; - const auto * query_side_column_mapper = format_filter_info->current_schema_column_mapper - ? format_filter_info->current_schema_column_mapper.get() - : format_filter_info->column_mapper.get(); - if (query_side_column_mapper && format_filter_info->column_mapper) - clickhouse_to_parquet_name = - query_side_column_mapper->makeMapping(format_filter_info->column_mapper->getFieldIdToClickHouseName()).first; + std::optional> clickhouse_to_parquet_name; + auto get_clickhouse_to_parquet_name = [&]() -> const std::unordered_map & + { + if (!clickhouse_to_parquet_name) + { + clickhouse_to_parquet_name.emplace(); + const auto * query_side_column_mapper = format_filter_info->current_schema_column_mapper + ? format_filter_info->current_schema_column_mapper.get() + : format_filter_info->column_mapper.get(); + if (query_side_column_mapper && format_filter_info->column_mapper) + *clickhouse_to_parquet_name + = query_side_column_mapper->makeMapping(format_filter_info->column_mapper->getFieldIdToClickHouseName()).first; + } + return *clickhouse_to_parquet_name; + }; auto resolve_geo_meta = [&](const String & ch_name) -> std::unordered_map::const_iterator { - if (auto it = clickhouse_to_parquet_name.find(ch_name); it != clickhouse_to_parquet_name.end()) + const auto & mapping = get_clickhouse_to_parquet_name(); + if (auto it = mapping.find(ch_name); it != mapping.end()) return geo_meta->find(it->second); return geo_meta->find(ch_name); }; @@ -453,7 +462,8 @@ void Reader::prefilterAndInitRowGroups(const std::optional String { - if (auto it = clickhouse_to_parquet_name.find(ch_name); it != clickhouse_to_parquet_name.end()) + const auto & mapping = get_clickhouse_to_parquet_name(); + if (auto it = mapping.find(ch_name); it != mapping.end()) return it->second; return ch_name; }; From da76d8ae6b11da5444889e260318bf3a97a025ce Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 30 Sep 2026 10:49:33 +0000 Subject: [PATCH 081/185] Backport #122904 to 26.8: Silence `--icf=safe` warnings for the stripped Rust libraries --- cmake/strip_rust_symbols.sh | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/cmake/strip_rust_symbols.sh b/cmake/strip_rust_symbols.sh index cb91889088bf..4987b189590f 100755 --- a/cmake/strip_rust_symbols.sh +++ b/cmake/strip_rust_symbols.sh @@ -50,7 +50,8 @@ for sym in "$@"; do done # Localize all symbols except the public ones, then strip unneeded locals -"$OBJCOPY" $KEEP_FLAGS --strip-unneeded "$WORK_DIR/combined.o" "$WORK_DIR/stripped.o" +# Drop .llvm_addrsig: after ld -r and objcopy its sh_link is 0, so lld --icf=safe ignores it with a warning +"$OBJCOPY" $KEEP_FLAGS --strip-unneeded --remove-section=.llvm_addrsig "$WORK_DIR/combined.o" "$WORK_DIR/stripped.o" # Repackage as .a (replace original) rm -f "$LIB_PATH" From e8f84dd97537f6f95f86087c1aa9f2b1902d8ea6 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 30 Sep 2026 17:44:23 +0000 Subject: [PATCH 082/185] Backport #123060 to 26.8: Hide queue engine secrets written with the legacy `s3queue_` prefix --- src/Parsers/ASTSetQuery.cpp | 8 +++- ...eue_legacy_prefix_secret_masking.reference | 7 ++++ ...4_s3queue_legacy_prefix_secret_masking.sql | 40 +++++++++++++++++++ 3 files changed, 53 insertions(+), 2 deletions(-) create mode 100644 tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.reference create mode 100644 tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.sql diff --git a/src/Parsers/ASTSetQuery.cpp b/src/Parsers/ASTSetQuery.cpp index 4c440a87852a..43eb30fcc937 100644 --- a/src/Parsers/ASTSetQuery.cpp +++ b/src/Parsers/ASTSetQuery.cpp @@ -58,12 +58,16 @@ static std::array engineSettingsToHide() /// disagree on what is secret. static std::optional renderSecretChangeValue(const SettingChange & change) { - if (auto masked = CoreSettings::renderSecretSettingValue(change.name, change.value)) + /// The queue engines also take every setting, a format setting included, with the legacy `s3queue_` prefix. + static constexpr std::string_view s3queue_prefix = "s3queue_"; + const String setting_name = change.name.starts_with(s3queue_prefix) ? change.name.substr(s3queue_prefix.size()) : change.name; + + if (auto masked = CoreSettings::renderSecretSettingValue(setting_name, change.value)) return masked; for (const auto * settings_to_hide : engineSettingsToHide()) { - auto it = settings_to_hide->find(change.name); + auto it = settings_to_hide->find(setting_name); if (it != settings_to_hide->end()) return it->second(change.value); } diff --git a/tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.reference b/tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.reference new file mode 100644 index 000000000000..e2849f54c782 --- /dev/null +++ b/tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.reference @@ -0,0 +1,7 @@ +CREATE TABLE default.t_legacy\n(\n `a` String\n)\nENGINE = S3Queue(\'http://whatever-we-dont-care:9001/root/t_legacy/*\', \'NOSIGN\', \'CSV\')\nSETTINGS mode = \'unordered\', s3queue_loading_retries = 7, s3queue_after_processing_move_secret_access_key = \'[HIDDEN]\', s3queue_after_processing_move_connection_string = \'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=[HIDDEN];\', s3queue_format_avro_schema_registry_url = \'http://[HIDDEN]@registry:8080/\' +CREATE TABLE default.t_canonical\n(\n `a` String\n)\nENGINE = S3Queue(\'http://whatever-we-dont-care:9001/root/t_canonical/*\', \'NOSIGN\', \'CSV\')\nSETTINGS mode = \'unordered\', after_processing_move_secret_access_key = \'[HIDDEN]\', after_processing_move_connection_string = \'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=[HIDDEN];\', format_avro_schema_registry_url = \'http://[HIDDEN]@registry:8080/\' +t_canonical 1 1 +t_legacy 1 1 +Create 1 +Create 1 +Alter 1 diff --git a/tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.sql b/tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.sql new file mode 100644 index 000000000000..fe2670ed8790 --- /dev/null +++ b/tests/queries/0_stateless/05314_s3queue_legacy_prefix_secret_masking.sql @@ -0,0 +1,40 @@ +-- Tags: no-fasttest, zookeeper +-- no-fasttest: the `S3Queue` engine is not built in the fast test. + +-- The queue engines also take every setting with the legacy `s3queue_` prefix, and a secret written that way is +-- hidden in `SHOW CREATE TABLE`, `system.tables` and `system.query_log` as it is without the prefix. + +DROP TABLE IF EXISTS t_legacy; +DROP TABLE IF EXISTS t_canonical; + +CREATE TABLE t_legacy (a String) +ENGINE = S3Queue('http://whatever-we-dont-care:9001/root/t_legacy/*', NOSIGN, 'CSV') +SETTINGS mode = 'unordered', s3queue_loading_retries = 7, + s3queue_after_processing_move_secret_access_key = 'SEKRIT_05314_1', + s3queue_after_processing_move_connection_string = 'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_05314_2;', + s3queue_format_avro_schema_registry_url = 'http://user:SEKRIT_05314_3@registry:8080/'; + +CREATE TABLE t_canonical (a String) +ENGINE = S3Queue('http://whatever-we-dont-care:9001/root/t_canonical/*', NOSIGN, 'CSV') +SETTINGS mode = 'unordered', + after_processing_move_secret_access_key = 'SEKRIT_05314_4', + after_processing_move_connection_string = 'DefaultEndpointsProtocol=https;AccountName=a;AccountKey=SEKRIT_05314_5;', + format_avro_schema_registry_url = 'http://user:SEKRIT_05314_6@registry:8080/'; + +SHOW CREATE TABLE t_legacy; +SHOW CREATE TABLE t_canonical; + +SELECT name, position(create_table_query, 'SEKRIT') = 0, position(engine_full, 'SEKRIT') = 0 +FROM system.tables WHERE database = currentDatabase() ORDER BY name; + +ALTER TABLE t_legacy MODIFY SETTING s3queue_after_processing_move_secret_access_key = 'SEKRIT_05314_7'; + +SYSTEM FLUSH LOGS query_log; +SELECT query_kind, position(query, 'SEKRIT') = 0 +FROM system.query_log +WHERE event_date >= yesterday() AND current_database = currentDatabase() AND is_initial_query + AND type = 'QueryFinish' AND query_kind IN ('Create', 'Alter') +ORDER BY event_time_microseconds; + +DROP TABLE t_legacy; +DROP TABLE t_canonical; From d76d7856587d9f6f21d406e50affcf64085216a2 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Wed, 30 Sep 2026 22:27:12 +0000 Subject: [PATCH 083/185] Backport #123114 to 26.8: S3 client check initial url in backup client --- src/Backups/BackupIO_S3.cpp | 3 + .../configs/remote_url_allow_hosts.xml | 5 ++ .../test_remote_host_filter.py | 56 +++++++++++++++++++ 3 files changed, 64 insertions(+) create mode 100644 tests/integration/test_backup_restore_s3/configs/remote_url_allow_hosts.xml create mode 100644 tests/integration/test_backup_restore_s3/test_remote_host_filter.py diff --git a/src/Backups/BackupIO_S3.cpp b/src/Backups/BackupIO_S3.cpp index 57e9c45cbfd3..2b7c176d0ecd 100644 --- a/src/Backups/BackupIO_S3.cpp +++ b/src/Backups/BackupIO_S3.cpp @@ -3,6 +3,7 @@ #if USE_AWS_S3 #include #include +#include #include #include #include @@ -131,6 +132,8 @@ class S3BackupClientCreator const S3Settings & settings, const ContextPtr & context) { + context->getGlobalContext()->getRemoteHostFilter().checkURL(s3_uri.uri); + Aws::Auth::AWSCredentials credentials(access_key_id, secret_access_key); HTTPHeaderEntries headers; String session_token = settings.auth_settings[S3AuthSetting::session_token]; diff --git a/tests/integration/test_backup_restore_s3/configs/remote_url_allow_hosts.xml b/tests/integration/test_backup_restore_s3/configs/remote_url_allow_hosts.xml new file mode 100644 index 000000000000..294b4bc7dcb4 --- /dev/null +++ b/tests/integration/test_backup_restore_s3/configs/remote_url_allow_hosts.xml @@ -0,0 +1,5 @@ + + + minio1:9001 + + diff --git a/tests/integration/test_backup_restore_s3/test_remote_host_filter.py b/tests/integration/test_backup_restore_s3/test_remote_host_filter.py new file mode 100644 index 000000000000..977d02c84e73 --- /dev/null +++ b/tests/integration/test_backup_restore_s3/test_remote_host_filter.py @@ -0,0 +1,56 @@ +import uuid + +import pytest + +from helpers.cluster import ClickHouseCluster +from helpers.config_cluster import minio_secret_key + +cluster = ClickHouseCluster(__file__) +node = cluster.add_instance( + "node", + main_configs=["configs/remote_url_allow_hosts.xml"], + with_minio=True, +) + + +@pytest.fixture(scope="module", autouse=True) +def start_cluster(): + try: + cluster.start() + node.query( + "CREATE TABLE t (id UInt64, s String) ENGINE = MergeTree ORDER BY id" + ) + node.query("INSERT INTO t VALUES (1, 'a'), (2, 'b')") + yield cluster + finally: + cluster.shutdown() + + +def test_backup_to_allowed_host(): + name = uuid.uuid4().hex + destination = f"S3('http://minio1:9001/root/data/backups/{name}', 'minio', '{minio_secret_key}')" + node.query(f"BACKUP TABLE t TO {destination}") + node.query(f"RESTORE TABLE t AS t_{name} FROM {destination}") + assert node.query(f"SELECT count() FROM t_{name}") == "2\n" + node.query(f"DROP TABLE t_{name} SYNC") + + +@pytest.mark.parametrize( + "url", + [ + "http://minio1:9002/root/data/backups/x", + "http://resolver:8080/root/data/backups/x", + "http://127.0.0.1:9/root/data/backups/x", + ], +) +def test_backup_to_disallowed_host(url): + destination = f"S3('{url}', 'minio', '{minio_secret_key}')" + settings = "SETTINGS backup_restore_s3_retry_attempts = 0" + + error = node.query_and_get_error(f"BACKUP TABLE t TO {destination} {settings}") + assert "UNACCEPTABLE_URL" in error, error + + error = node.query_and_get_error( + f"RESTORE TABLE t AS t_restored FROM {destination} {settings}" + ) + assert "UNACCEPTABLE_URL" in error, error From 85989f16e86349a9b1254c073efac3b9b46f3136 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 08:07:34 +0000 Subject: [PATCH 084/185] Backport #121621 to 26.8: Fix `NO_COMMON_TYPE` in case of `use_variant_as_common_type=0` for jemalloc fragmentation dashboard --- docs/concepts/features/performance/allocation-profiling.mdx | 2 +- programs/server/jemalloc.html | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/docs/concepts/features/performance/allocation-profiling.mdx b/docs/concepts/features/performance/allocation-profiling.mdx index 7c51dfc01f93..383f8bca8cb8 100644 --- a/docs/concepts/features/performance/allocation-profiling.mdx +++ b/docs/concepts/features/performance/allocation-profiling.mdx @@ -354,7 +354,7 @@ The same query can split the waste of each size class between its backtraces, in WITH 60e9 AS min_age_ns SELECT format('{} {}', if(s.est_old_objects > 0, s.stack, format('[unattributed];arena_{}_class_{}', toString(b.arena), toString(b.size))), - toString(toUInt64(if(s.est_old_objects > 0, b.waste * s.est_old_objects / sum(s.est_old_objects) OVER (PARTITION BY b.arena, b.index), b.waste)))) + toString(toUInt64(if(s.est_old_objects > 0, b.waste * s.est_old_objects / sum(s.est_old_objects) OVER (PARTITION BY b.arena, b.index), toFloat64(b.waste))))) FROM system.jemalloc_arena_bins AS b LEFT JOIN ( diff --git a/programs/server/jemalloc.html b/programs/server/jemalloc.html index fd785fa66f28..8505ae77e245 100644 --- a/programs/server/jemalloc.html +++ b/programs/server/jemalloc.html @@ -2922,7 +2922,7 @@

Mutex Statistics

/// Both branches must come from the same scan of the sampled allocations: every /// scan flushes a fresh profile dump, so two independent subqueries could observe /// different dumps and double-count or drop a cell. - const query = `SELECT format('{} {}', if(s.est_old_objects > 0, s.stack, format('[unattributed];arena_{}_class_{}', toString(b.arena), toString(b.size))), toString(toUInt64(if(s.est_old_objects > 0, b.waste * s.est_old_objects / sum(s.est_old_objects) OVER (PARTITION BY b.arena, b.index), b.waste)))) FROM system.jemalloc_arena_bins AS b LEFT JOIN (${fragmentationPinnersSubquery(minAgeNs)}) AS s ON s.size_class = b.index AND s.arena = b.arena WHERE b.large = 0 AND b.waste >= ${minWaste}${purposeCond} SETTINGS allow_introspection_functions = 1`; + const query = `SELECT format('{} {}', if(s.est_old_objects > 0, s.stack, format('[unattributed];arena_{}_class_{}', toString(b.arena), toString(b.size))), toString(toUInt64(if(s.est_old_objects > 0, b.waste * s.est_old_objects / sum(s.est_old_objects) OVER (PARTITION BY b.arena, b.index), toFloat64(b.waste))))) FROM system.jemalloc_arena_bins AS b LEFT JOIN (${fragmentationPinnersSubquery(minAgeNs)}) AS s ON s.size_class = b.index AND s.arena = b.arena WHERE b.large = 0 AND b.waste >= ${minWaste}${purposeCond} SETTINGS allow_introspection_functions = 1`; const response = await fetch(buildUrl(query), fetchOptions); if (!response.ok) { throw new Error(`HTTP ${response.status}: ${response.statusText}`); From c87233d6f0932cb3cdbb3d2851e75dd84aa94c75 Mon Sep 17 00:00:00 2001 From: Pedro Ferreira Date: Thu, 1 Oct 2026 08:38:38 +0000 Subject: [PATCH 085/185] Fix the 26.8 build: replace the master-only row policy helper `Storages/getEffectiveRowPolicyFilter.h` does not exist on 26.8. The call site always passes the data part's MergeTree, which has no underlying storages, so the walk it does on master reduces to the table's own policy; the local helper here checks exactly that. The `Alias` case still holds: on 26.8 the alias pushes its policy into `query_info.row_level_filter`. Co-Authored-By: Claude Opus 5.5 --- .../MergeTree/MergeTreeDataSelectExecutor.cpp | 14 ++++++++++++-- 1 file changed, 12 insertions(+), 2 deletions(-) diff --git a/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp b/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp index 03feebb56429..862a95339e5a 100644 --- a/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp +++ b/src/Storages/MergeTree/MergeTreeDataSelectExecutor.cpp @@ -21,7 +21,8 @@ #include #include #include -#include +#include +#include #include #include #include @@ -805,6 +806,15 @@ std::expected MergeTreeDataSelectExecutor::canUseInde } +static bool hasEffectiveRowPolicy(const IStorage & storage, const ContextPtr & context) +{ + auto storage_id = storage.getStorageID(); + if (!storage_id.hasDatabase()) + return false; + auto filter = context->getRowPolicyFilter(storage_id.getDatabaseName(), storage_id.getTableName(), RowPolicyFilterType::SELECT_FILTER); + return filter && !filter->isAlwaysTrue(); +} + RangesInDataParts MergeTreeDataSelectExecutor::filterPartsByStatistics( const RangesInDataParts & parts, const StorageMetadataPtr & metadata_snapshot, @@ -832,7 +842,7 @@ RangesInDataParts MergeTreeDataSelectExecutor::filterPartsByStatistics( || (mutations_snapshot && (mutations_snapshot->hasDataMutations() || mutations_snapshot->hasPatchParts())) || (!parts.empty() && parts.front().data_part->storage.hasEnabledMaskingPolicies(context)) || query_info.row_level_filter - || (!parts.empty() && getEffectiveRowPolicyFilter(parts.front().data_part->storage, context))) + || (!parts.empty() && hasEffectiveRowPolicy(parts.front().data_part->storage, context))) { return parts; } From 258ea89e2d2c1dfc4d43fa61c86a1987785982fb Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 09:06:22 +0000 Subject: [PATCH 086/185] Backport #122681 to 26.8: Build Iceberg min/max hyperrectangles only for columns the filter uses --- .../Iceberg/ManifestFileIterator.cpp | 27 +++++++++---------- .../Iceberg/ManifestFilesPruning.cpp | 13 +++++---- .../DataLakes/Iceberg/ManifestFilesPruning.h | 3 +++ 3 files changed, 21 insertions(+), 22 deletions(-) diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp index fe5553bcc118..01c067bc9009 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFileIterator.cpp @@ -511,20 +511,18 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) PruningReturnStatus pruning_status = PruningReturnStatus::NOT_PRUNED; if (filter_dag) { + const ManifestFilesPruner * current_pruner = getOrCreatePruner(entry->resolved_schema_id); + /// Compute per-column hyperrectangles for DATA files std::unordered_map hyperrectangles; if (parsed_entry->content_type == FileContentType::DATA) { - for (const auto & [column_id, bounds] : parsed_entry->value_bounds) + for (const auto & [column_id, column_type] : current_pruner->getMinMaxColumnTypes()) { - auto field_characteristics = schema_processor_ptr->tryGetFieldCharacteristics(resolved_schema_id, column_id); - /// If we don't have column characteristics, bounds don't have any sense. - /// This happens if the subfield is inside map or array, because we don't support - /// name generation for such subfields (we support names of nested subfields in structs only). - if (!field_characteristics) + auto bounds_it = parsed_entry->value_bounds.find(column_id); + if (bounds_it == parsed_entry->value_bounds.end()) continue; - - const auto & name_and_type = *field_characteristics; + const auto & bounds = bounds_it->second; String left_str; String right_str; @@ -532,13 +530,13 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) if (!bounds.first.tryGet(left_str) || !bounds.second.tryGet(right_str)) continue; - if (const auto type_id = name_and_type.type->getTypeId(); + if (const auto type_id = column_type->getTypeId(); type_id == DB::TypeIndex::Tuple || type_id == DB::TypeIndex::Map || type_id == DB::TypeIndex::Array || type_id == DB::TypeIndex::Variant) continue; - auto left = deserializeFieldFromBinaryRepr(left_str, name_and_type.type, true); - auto right = deserializeFieldFromBinaryRepr(right_str, name_and_type.type, false); + auto left = deserializeFieldFromBinaryRepr(left_str, column_type, true); + auto right = deserializeFieldFromBinaryRepr(right_str, column_type, false); if (!left || !right) { /// Pruning is skipped either way, but at scale 38 a bound that only loses its widened @@ -559,12 +557,12 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) /// declared expose that inversion, which is why they are read again here. std::optional declared_left = left; std::optional declared_right = right; - if (DB::WhichDataType(DB::removeNullable(name_and_type.type)).isDecimal()) + if (DB::WhichDataType(DB::removeNullable(column_type)).isDecimal()) { declared_left = deserializeFieldFromBinaryRepr( - left_str, name_and_type.type, true, /*compensate_rounding=*/false); + left_str, column_type, true, /*compensate_rounding=*/false); declared_right = deserializeFieldFromBinaryRepr( - right_str, name_and_type.type, false, /*compensate_rounding=*/false); + right_str, column_type, false, /*compensate_rounding=*/false); } /// A pair inverted as declared means the manifest's statistics are untrustworthy, so no @@ -586,7 +584,6 @@ ProcessedManifestFileEntryPtr ManifestFileIterator::processRow(size_t row_index) } } - const ManifestFilesPruner * current_pruner = getOrCreatePruner(entry->resolved_schema_id); pruning_status = current_pruner->canBePruned(entry, hyperrectangles); } insertRowToLogTable( diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp index c7c76cbd1483..ab8e95a358b6 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.cpp @@ -137,6 +137,7 @@ ManifestFilesPruner::ManifestFilesPruner( if (!name_and_type.has_value()) continue; + min_max_column_types.emplace(used_column_id, name_and_type->type); name_and_type->name = DB::backQuote(DB::toString(used_column_id)); ExpressionActionsPtr expression @@ -394,19 +395,17 @@ PruningReturnStatus ManifestFilesPruner::canBePruned( for (const auto & [column_id, key_condition] : min_max_key_conditions) { - std::optional name_and_type = schema_processor.tryGetFieldCharacteristics(initial_schema_id, column_id); - /// There is no such column in this manifest file - if (!name_and_type.has_value()) - { + auto type_it = min_max_column_types.find(column_id); + if (type_it == min_max_column_types.end()) continue; - } + const auto & column_type = type_it->second; auto info_it = entry->parsed_entry->columns_infos.find(column_id); bool has_no_nulls = info_it != entry->parsed_entry->columns_infos.end() && info_it->second.nulls_count.has_value() && *info_it->second.nulls_count == 0; - const DataTypes data_types{name_and_type->type}; + const DataTypes data_types{column_type}; if (entry->common_partition_specification) { @@ -419,7 +418,7 @@ PruningReturnStatus ManifestFilesPruner::canBePruned( auto range = rangeOfPartitionValue( partition_field.transform_name, partition_value[partition_field.tuple_index], - *removeNullable(name_and_type->type)); + *removeNullable(column_type)); if (range && !key_condition.mayBeTrueInRange(1, &range->left, &range->right, data_types)) return PruningReturnStatus::PARTITION_PRUNED; diff --git a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h index 78c136167a88..e9aa14f97e57 100644 --- a/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h +++ b/src/Storages/ObjectStorage/DataLakes/Iceberg/ManifestFilesPruning.h @@ -43,6 +43,7 @@ class ManifestFilesPruner std::optional partition_key_condition; std::unordered_map min_max_key_conditions; + std::unordered_map min_max_column_types; /// NOTE: tricky part to support RENAME column. /// Takes ActionDAG representation of user's WHERE expression and /// rename columns to the their origina numeric ID's in iceberg @@ -58,6 +59,8 @@ class ManifestFilesPruner DB::ContextPtr context); PruningReturnStatus canBePruned(const ProcessedManifestFileEntryPtr & entry, const std::unordered_map & entry_hyperrectangles) const; + + const std::unordered_map & getMinMaxColumnTypes() const { return min_max_column_types; } }; } From 74f2e6b59474fb572dacf76d9da3a5a9bfb0db18 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 09:55:06 +0000 Subject: [PATCH 087/185] Backport #120516 to 26.8: Fix access checks in PREWHERE and indexHint() that used aliases --- src/Planner/CollectTableExpressionData.cpp | 27 +++- src/Planner/PlannerJoinTree.cpp | 17 +++ src/Planner/TableExpressionData.h | 23 ++++ src/Planner/Utils.cpp | 42 +++--- src/Planner/Utils.h | 4 + .../collectSelectedColumnsFromTable.cpp | 30 +--- src/Storages/MergeTree/MergeTreeData.cpp | 3 +- ...ewhere_alias_column_access_check.reference | 60 ++++++++ ...5218_prewhere_alias_column_access_check.sh | 129 ++++++++++++++++++ .../05293_index_hint_column_access.reference | 24 ++++ .../05293_index_hint_column_access.sh | 48 +++++++ ...nly_alter_guard_alias_dependency.reference | 1 + ...data_only_alter_guard_alias_dependency.sql | 28 ++++ 13 files changed, 387 insertions(+), 49 deletions(-) create mode 100644 tests/queries/0_stateless/05218_prewhere_alias_column_access_check.reference create mode 100755 tests/queries/0_stateless/05218_prewhere_alias_column_access_check.sh create mode 100644 tests/queries/0_stateless/05293_index_hint_column_access.reference create mode 100755 tests/queries/0_stateless/05293_index_hint_column_access.sh create mode 100644 tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.reference create mode 100644 tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.sql diff --git a/src/Planner/CollectTableExpressionData.cpp b/src/Planner/CollectTableExpressionData.cpp index 996b518cb17f..c3d7831aeeaa 100644 --- a/src/Planner/CollectTableExpressionData.cpp +++ b/src/Planner/CollectTableExpressionData.cpp @@ -1,5 +1,7 @@ #include +#include + #include #include #include @@ -18,6 +20,7 @@ #include #include #include +#include namespace DB @@ -62,7 +65,16 @@ class CollectSourceColumnsVisitor : public InDepthQueryTreeVisitorWithContextgetOrCreateTableExpressionData(column_node->getColumnSource()); + index_hint_table_expression_data.markColumnForAccessCheck(column_node->getColumnName()); + } return; + } auto column_source_node = column_node->getColumnSource(); auto column_source_node_type = column_source_node->getNodeType(); @@ -106,6 +118,9 @@ class CollectSourceColumnsVisitor : public InDepthQueryTreeVisitorWithContextgetExpression(); + /// The visitor above has registered the expression's source columns as read but not selected, and the + /// children walk that follows must not select them either: a grant on the ALIAS name is sufficient. + inlined_alias_expressions.insert(node.get()); return; } @@ -174,6 +189,8 @@ class CollectSourceColumnsVisitor : public InDepthQueryTreeVisitorWithContextas()) return child_node != table_node->getMaterializedCTESubquery(); - return !(checkSubquery(child_node) || isAliasColumn(parent_node)); + return !(checkSubquery(child_node) || isAliasColumn(parent_node) || inlined_alias_expressions.contains(parent_node.get())); } static bool isIndexHintFunction(const QueryTreeNodePtr & node) @@ -263,6 +280,9 @@ class CollectSourceColumnsVisitor : public InDepthQueryTreeVisitorWithContext inlined_alias_expressions; }; class CollectPrewhereTableExpressionVisitor : public ConstInDepthQueryTreeVisitor @@ -440,6 +460,11 @@ void collectTableExpressionData(QueryTreeNodePtr & query_node, PlannerContextPtr const auto & selected_column_names = table_expression_data.getSelectedColumnsNames(); required_column_names_without_prewhere.insert(selected_column_names.begin(), selected_column_names.end()); + /// The visit below inlines ALIAS columns, which would hide their names from the access check. + /// Record what PREWHERE references first, so the same names are checked as for WHERE. + for (const auto & column_name : collectReferencedColumnNames(query_node_typed.getPrewhere(), prewhere_table_expression)) + table_expression_data.markColumnForAccessCheck(column_name); + collect_source_columns_visitor.setKeepAliasColumns(false); collect_source_columns_visitor.visit(query_node_typed.getPrewhere()); diff --git a/src/Planner/PlannerJoinTree.cpp b/src/Planner/PlannerJoinTree.cpp index 67ca7fb5945a..52e09cafb0b3 100644 --- a/src/Planner/PlannerJoinTree.cpp +++ b/src/Planner/PlannerJoinTree.cpp @@ -722,6 +722,21 @@ bool applyTrivialCountWithSparsityFilterIfPossible( return true; } +/** Check the SELECT privilege for the columns that the planner resolved "away": `indexHint` arguments and ALIAS + * columns inlined into PREWHERE. Checked separately from the selected columns on purpose: a trivial query such as + * `SELECT count() FROM t` passes with a grant on any one column, while these names are always required. + */ +void checkAccessRightsForColumnsResolvedAway( + const TableNode & table_node, const TableExpressionData & table_expression_data, const ContextPtr & query_context) +{ + const auto & column_names = table_expression_data.getAccessCheckedColumnsNames(); + if (column_names.empty()) + return; + + checkAccessRights( + table_node.getStorage(), table_node.getStorageID(), table_node.getStorageSnapshot(), column_names, query_context); +} + void prepareBuildQueryPlanForTableExpression(const QueryTreeNodePtr & table_expression, const SelectQueryOptions & select_query_options, PlannerContextPtr & planner_context) { const auto & query_context = planner_context->getQueryContext(); @@ -744,6 +759,8 @@ void prepareBuildQueryPlanForTableExpression(const QueryTreeNodePtr & table_expr const auto & column_names_with_aliases = table_expression_data.getSelectedColumnsNames(); columns_names_allowed_to_select = checkAccessRights( table_node->getStorage(), table_node->getStorageID(), table_node->getStorageSnapshot(), column_names_with_aliases, query_context); + + checkAccessRightsForColumnsResolvedAway(*table_node, table_expression_data, query_context); } else if (table_function_node) { diff --git a/src/Planner/TableExpressionData.h b/src/Planner/TableExpressionData.h index 29b765a27dc7..b3943d38feb4 100644 --- a/src/Planner/TableExpressionData.h +++ b/src/Planner/TableExpressionData.h @@ -93,6 +93,24 @@ class TableExpressionData selected_column_names.push_back(column_name); } + /** Mark a column that the user references explicitly, but that never becomes a selected column. + * + * This is needed for columns that the planner resolves away before the access check runs : + * an ALIAS column inlined into PREWHERE and a column used only as an indexHint argument. + */ + void markColumnForAccessCheck(const std::string & column_name) + { + auto [_, inserted] = access_checked_column_names_set.emplace(column_name); + if (inserted) + access_checked_column_names.push_back(column_name); + } + + /// Get columns that are not selected, but still require a SELECT privilege check + const Names & getAccessCheckedColumnsNames() const + { + return access_checked_column_names; + } + /// Get columns that are requested from table expression, including ALIAS columns const Names & getSelectedColumnsNames() const { @@ -292,6 +310,11 @@ class TableExpressionData /// To deduplicate columns in `selected_column_names` NameSet selected_column_names_set; + /// Columns that the user references explicitly, but that are resolved away before access check. + Names access_checked_column_names; + /// To deduplicate columns in above + NameSet access_checked_column_names_set; + /// Expression to calculate ALIAS columns /// Keep alias name (String) + expression (ActionsDAG) pairs; vector preserves insertion order. AliasColumnExpressions alias_column_expressions; diff --git a/src/Planner/Utils.cpp b/src/Planner/Utils.cpp index a54f859f69fb..32c307f61773 100644 --- a/src/Planner/Utils.cpp +++ b/src/Planner/Utils.cpp @@ -620,6 +620,29 @@ NameSet checkAccessRights( return {}; } +NameSet collectReferencedColumnNames(const QueryTreeNodePtr & node, const QueryTreeNodePtr & table_expression) +{ + NameSet column_names; + traverseQueryTree( + node, + [](const QueryTreeNodePtr & parent, const QueryTreeNodePtr &) + { + /// Don't go inside an ALIAS column expression: a grant on the alias name is sufficient. + const auto * column_node = parent->as(); + if (!column_node || !column_node->hasExpression()) + return true; + const auto & column_source = column_node->getColumnSourceOrNull(); + return !(column_source && column_source->getNodeType() == QueryTreeNodeType::TABLE); + }, + [&](const QueryTreeNodePtr & current) + { + const auto * column_node = current->as(); + if (column_node && column_node->getColumnSourceOrNull().get() == table_expression.get()) + column_names.insert(column_node->getColumnName()); + }); + return column_names; +} + static void checkAccessRightsForFilter(const QueryTreeNodePtr & filter_query_tree, const QueryTreeNodePtr & table_expression, const ContextPtr & query_context) @@ -650,24 +673,7 @@ static void checkAccessRightsForFilter(const QueryTreeNodePtr & filter_query_tre return; } - NameSet column_names; - traverseQueryTree( - filter_query_tree, - [](const QueryTreeNodePtr & parent, const QueryTreeNodePtr &) - { - /// Don't go inside an ALIAS column expression: a grant on the alias name is sufficient. - const auto * column_node = parent->as(); - if (!column_node || !column_node->hasExpression()) - return true; - const auto & column_source = column_node->getColumnSourceOrNull(); - return !(column_source && column_source->getNodeType() == QueryTreeNodeType::TABLE); - }, - [&](const QueryTreeNodePtr & node) - { - const auto * column_node = node->as(); - if (column_node && column_node->getColumnSourceOrNull().get() == table_expression.get()) - column_names.insert(column_node->getColumnName()); - }); + NameSet column_names = collectReferencedColumnNames(filter_query_tree, table_expression); if (column_names.empty()) return; diff --git a/src/Planner/Utils.h b/src/Planner/Utils.h index 3b504afb5286..032a614a73d7 100644 --- a/src/Planner/Utils.h +++ b/src/Planner/Utils.h @@ -92,6 +92,10 @@ QueryTreeNodePtr replaceTableExpressionsWithDummyTables( SelectQueryInfo buildSelectQueryInfo(const QueryTreeNodePtr & query_tree, const PlannerContextPtr & planner_context); +/// Names of `table_expression` columns referenced from `node`, with ALIAS columns under their own name: +/// their expressions are not entered, because a grant on the alias name is sufficient to use it. +NameSet collectReferencedColumnNames(const QueryTreeNodePtr & node, const QueryTreeNodePtr & table_expression); + /// Check if current user has privileges to SELECT columns from table /// Throws an exception if access to any column from `column_names` is not granted /// If `column_names` is empty, check access to any columns and return names of accessible columns diff --git a/src/Planner/collectSelectedColumnsFromTable.cpp b/src/Planner/collectSelectedColumnsFromTable.cpp index 25d1ae549e96..6fecac2fd34b 100644 --- a/src/Planner/collectSelectedColumnsFromTable.cpp +++ b/src/Planner/collectSelectedColumnsFromTable.cpp @@ -2,7 +2,6 @@ #include #include #include -#include #include @@ -21,12 +20,6 @@ class CollectSelectedColumnsFromTableVisitor : public InDepthQueryTreeVisitorWit void enterImpl(QueryTreeNodePtr & node) { - if (isIndexHintFunction(node)) - { - is_inside_index_hint_function = true; - return; - } - auto * column_node = node->as(); if (!column_node) return; @@ -38,24 +31,10 @@ class CollectSelectedColumnsFromTableVisitor : public InDepthQueryTreeVisitorWit if (!source_table || source_table->getStorageID() != storage_id) return; - /// A special case for the "indexHint" function. We don't need its arguments for execution if column's source table is MergeTree. - /// Instead, we prepare an ActionsDAG for its arguments and store it inside a function (see ActionsDAG::buildFilterActionsDAG). - /// So this optimization allows not to read arguments of "indexHint" (if not needed in other contexts) but only to use index analysis for them. - if (is_inside_index_hint_function && source_table->getStorage()->isMergeTree()) - return; - + /// Note that arguments of the "indexHint" function need to be checked for SELECT privilege selected_columns.insert(column_node->getColumnName()); } - void leaveImpl(QueryTreeNodePtr & node) - { - if (isIndexHintFunction(node)) - { - is_inside_index_hint_function = false; - return; - } - } - bool isAliasColumn(const QueryTreeNodePtr & node) const { const auto * column_node = node->as(); @@ -73,19 +52,12 @@ class CollectSelectedColumnsFromTableVisitor : public InDepthQueryTreeVisitorWit return !isAliasColumn(parent_node); } - bool isIndexHintFunction(const QueryTreeNodePtr & node) const - { - return node->as() && node->as()->getFunctionName() == "indexHint"; - } - std::vector getSelectedColumns() const { return std::vector(selected_columns.begin(), selected_columns.end()); } private: - /// True if we are traversing arguments of function "indexHint". - bool is_inside_index_hint_function = false; const StorageID & storage_id; std::unordered_set selected_columns; }; diff --git a/src/Storages/MergeTree/MergeTreeData.cpp b/src/Storages/MergeTree/MergeTreeData.cpp index 592d37ed175b..eb0a0814121a 100644 --- a/src/Storages/MergeTree/MergeTreeData.cpp +++ b/src/Storages/MergeTree/MergeTreeData.cpp @@ -5071,8 +5071,9 @@ Names expressionSourceColumns(const ASTPtr & ast, const ColumnsDescription & col auto planner_context = std::make_shared(analysis_context, global_planner_context, SelectQueryOptions{}); collectSetsAndSourceColumns(expression, planner_context, /*keep_alias_columns=*/ false); + /// ALIAS columns are inlined above, so the physical columns their expressions read are the dependencies. if (const auto * table_expression_data = planner_context->getTableExpressionDataOrNull(table_node)) - return table_expression_data->getSelectedColumnsNames(); + return table_expression_data->getColumnNames(); return {}; } diff --git a/tests/queries/0_stateless/05218_prewhere_alias_column_access_check.reference b/tests/queries/0_stateless/05218_prewhere_alias_column_access_check.reference new file mode 100644 index 000000000000..018a9a02fa76 --- /dev/null +++ b/tests/queries/0_stateless/05218_prewhere_alias_column_access_check.reference @@ -0,0 +1,60 @@ +-- user granted only pub +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias +[0,1,2,3,4,5,6,7] +SELECT pub FROM t_prewhere_alias PREWHERE pub = 2 +2 +SELECT pub FROM t_prewhere_alias PREWHERE secret = 200 +ACCESS_DENIED +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias = 200 +ACCESS_DENIED +SELECT count() FROM t_prewhere_alias PREWHERE secret_alias = 200 +ACCESS_DENIED +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression = 201 +ACCESS_DENIED +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret >= 500) +ACCESS_DENIED +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret_alias >= 500) +ACCESS_DENIED +SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(secret < 500) +ACCESS_DENIED +SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(pub < 5) +1 +-- user granted pub and the alias columns, but not their source +SELECT secret_alias FROM t_prewhere_alias PREWHERE pub = 2 +200 +SELECT pub FROM t_prewhere_alias WHERE secret_alias = 200 +2 +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias = 200 +2 +SELECT pub FROM t_prewhere_alias WHERE secret_alias_expression = 201 +2 +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression = 201 +2 +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression + 0 = 201 +2 +SELECT pub FROM t_prewhere_alias PREWHERE secret = 200 +ACCESS_DENIED +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret >= 500) +ACCESS_DENIED +-- user granted pub and the source column, but not the alias columns +SELECT pub FROM t_prewhere_alias PREWHERE secret = 200 +2 +SELECT pub FROM t_prewhere_alias WHERE secret_alias = 200 +ACCESS_DENIED +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias = 200 +ACCESS_DENIED +SELECT pub FROM t_prewhere_alias WHERE secret_alias_expression = 201 +ACCESS_DENIED +SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression = 201 +ACCESS_DENIED +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret >= 500) +[4,5,6,7] +SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(secret < 500) +0 +SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(secret < 501) +1 +-- row policy over a column the user is not granted +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias +[0,1,3,4,5,6,7] +SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias PREWHERE pub > 1 +[3,4,5,6,7] diff --git a/tests/queries/0_stateless/05218_prewhere_alias_column_access_check.sh b/tests/queries/0_stateless/05218_prewhere_alias_column_access_check.sh new file mode 100755 index 000000000000..89efe0535064 --- /dev/null +++ b/tests/queries/0_stateless/05218_prewhere_alias_column_access_check.sh @@ -0,0 +1,129 @@ +#!/usr/bin/env bash +# Tags: no-old-analyzer, no-parallel-replicas +# no-parallel-replicas: reading an ALIAS column with a grant on only that ALIAS column is denied under +# parallel replicas. That is an unrelated pre-existing bug (reproducible on master without this test's +# subject), so exclude the setting rather than encode the wrong behaviour in the reference. + +# Column-level SELECT grants must be enforced for columns that the planner resolves away before the +# access check runs, so that PREWHERE cannot be used as an oracle over a column the user cannot read: +# +# 1. An ALIAS column referenced in PREWHERE is replaced by its expression, so it used to never reach +# the list of selected columns and was not access checked at all, while the same alias in SELECT or +# WHERE was correctly denied. +# 2. A column used only as an `indexHint` argument is never read, only used for index analysis, but +# which granules survive the analysis is observable in the result. +# +# In both cases the required privilege is the one on the name written in the query, exactly as in +# WHERE: referencing an ALIAS column requires a grant on the ALIAS column itself, and a grant on the +# physical columns its expression reads is neither sufficient nor required. Those source columns are +# still read from disk to compute the ALIAS - the administrator authored the ALIAS expression and so +# chose what it exposes - but their values never reach the user. + +# Most of the queries below are expected to be denied, and each denial would otherwise add an +# log line and a stack trace to the output, because the harness streams server-side logs to the client. +CLICKHOUSE_CLIENT_SERVER_LOGS_LEVEL=none + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +USER_PUB="user_pub_${CLICKHOUSE_DATABASE}" +USER_ALIAS="user_alias_${CLICKHOUSE_DATABASE}" +USER_SOURCE="user_source_${CLICKHOUSE_DATABASE}" + +${CLICKHOUSE_CLIENT} --multiquery --query " + DROP USER IF EXISTS ${USER_PUB}, ${USER_ALIAS}, ${USER_SOURCE}; + + CREATE TABLE t_prewhere_alias + ( + secret Int32, + pub Int32, + secret_alias Int32 ALIAS secret, + secret_alias_expression Int32 ALIAS secret_alias + 1 + ) + ENGINE = MergeTree ORDER BY secret SETTINGS index_granularity = 1, min_bytes_for_wide_part = 0; + + INSERT INTO t_prewhere_alias (secret, pub) SELECT number * 100, number FROM numbers(8); + + CREATE USER ${USER_PUB}, ${USER_ALIAS}, ${USER_SOURCE}; + GRANT SELECT(pub) ON ${CLICKHOUSE_DATABASE}.t_prewhere_alias TO ${USER_PUB}; + GRANT SELECT(pub, secret_alias, secret_alias_expression) ON ${CLICKHOUSE_DATABASE}.t_prewhere_alias TO ${USER_ALIAS}; + GRANT SELECT(pub, secret) ON ${CLICKHOUSE_DATABASE}.t_prewhere_alias TO ${USER_SOURCE}; +" + +# Run all the queries of one user in a single session, echoing each query before its own result so +# that every line of the reference is attributable. --ignore-error keeps the session going past the +# ACCESS_DENIED answers that most of these queries are expected to produce; without it the first +# denial would end the session and silently drop every query after it. +# +# An exception is printed by the client as exactly three lines - the "Received exception" banner, the +# "Code: ..." message and the echoed query - of which only the error name is stable: the banner holds +# the server version and the message holds the user name, which embeds ${CLICKHOUSE_DATABASE}. Reduce +# the block to that name, keeping it distinct per error so that a change of error shows up as a diff +# rather than being masked. +run_all() +{ + local user=$1 + shift + + local sql="" + local query + for query in "$@" + do + # The echoed query is derived from the query itself, so the two cannot drift apart. + sql+="SELECT '${query//\'/\'\'}'; +${query}; +" + done + + ${CLICKHOUSE_CLIENT} --user "${user}" --multiquery --ignore-error --query "${sql}" 2>&1 \ + | sed -E '/^Received exception/d; /^\(query:/d; s/^Code: [0-9]+\..*\(([A-Z_]+)\)$/\1/' +} + +echo "-- user granted only pub" +run_all "${USER_PUB}" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias" \ + "SELECT pub FROM t_prewhere_alias PREWHERE pub = 2" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret = 200" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias = 200" \ + "SELECT count() FROM t_prewhere_alias PREWHERE secret_alias = 200" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression = 201" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret >= 500)" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret_alias >= 500)" \ + "SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(secret < 500)" \ + "SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(pub < 5)" + +echo "-- user granted pub and the alias columns, but not their source" +run_all "${USER_ALIAS}" \ + "SELECT secret_alias FROM t_prewhere_alias PREWHERE pub = 2" \ + "SELECT pub FROM t_prewhere_alias WHERE secret_alias = 200" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias = 200" \ + "SELECT pub FROM t_prewhere_alias WHERE secret_alias_expression = 201" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression = 201" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression + 0 = 201" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret = 200" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret >= 500)" + +echo "-- user granted pub and the source column, but not the alias columns" +run_all "${USER_SOURCE}" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret = 200" \ + "SELECT pub FROM t_prewhere_alias WHERE secret_alias = 200" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias = 200" \ + "SELECT pub FROM t_prewhere_alias WHERE secret_alias_expression = 201" \ + "SELECT pub FROM t_prewhere_alias PREWHERE secret_alias_expression = 201" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias WHERE indexHint(secret >= 500)" \ + "SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(secret < 500)" \ + "SELECT count() FROM t_prewhere_alias WHERE pub = 5 AND indexHint(secret < 501)" + +# A row policy is defined by an administrator, so it may reference columns the user cannot read. +echo "-- row policy over a column the user is not granted" +${CLICKHOUSE_CLIENT} --query "CREATE ROW POLICY p_prewhere_alias ON ${CLICKHOUSE_DATABASE}.t_prewhere_alias USING secret_alias_expression != 201 TO ${USER_PUB}" +run_all "${USER_PUB}" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias" \ + "SELECT arraySort(groupArray(pub)) FROM t_prewhere_alias PREWHERE pub > 1" + +${CLICKHOUSE_CLIENT} --multiquery --query " + DROP ROW POLICY p_prewhere_alias ON ${CLICKHOUSE_DATABASE}.t_prewhere_alias; + DROP TABLE t_prewhere_alias; + DROP USER ${USER_PUB}, ${USER_ALIAS}, ${USER_SOURCE}; +" diff --git a/tests/queries/0_stateless/05293_index_hint_column_access.reference b/tests/queries/0_stateless/05293_index_hint_column_access.reference new file mode 100644 index 000000000000..030142563560 --- /dev/null +++ b/tests/queries/0_stateless/05293_index_hint_column_access.reference @@ -0,0 +1,24 @@ +=== granted: id +--- SELECT count() FROM t_index_hint_access WHERE id = 3 AND indexHint(secret > 5000) +ACCESS_DENIED +--- SELECT count() FROM t_index_hint_access WHERE id = 3 AND indexHint(secret < 5000) +ACCESS_DENIED +--- SELECT count() FROM t_index_hint_access WHERE indexHint(secret < 5000) +ACCESS_DENIED +--- SELECT count() FROM t_index_hint_access PREWHERE indexHint(secret < 5000) +ACCESS_DENIED +--- SELECT count() FROM (SELECT id FROM t_index_hint_access WHERE indexHint(secret < 5000)) +ACCESS_DENIED +--- SELECT count() FROM t_index_hint_access WHERE id = 3 AND indexHint(secret_alias > 5000) +ACCESS_DENIED +--- SELECT count() FROM t_index_hint_access AS l JOIN t_index_hint_access AS r ON l.id = r.id WHERE indexHint(r.secret < 5000) +ACCESS_DENIED +--- SELECT count() FROM t_index_hint_access WHERE indexHint(id < 5) +10 +=== granted: id, secret, secret_alias +--- SELECT count() FROM t_index_hint_access WHERE id = 3 AND indexHint(secret > 5000) +0 +--- SELECT count() FROM t_index_hint_access WHERE id = 3 AND indexHint(secret < 5000) +1 +--- SELECT count() FROM t_index_hint_access WHERE id = 3 AND indexHint(secret_alias > 5000) +1 diff --git a/tests/queries/0_stateless/05293_index_hint_column_access.sh b/tests/queries/0_stateless/05293_index_hint_column_access.sh new file mode 100755 index 000000000000..ab29b226a3c6 --- /dev/null +++ b/tests/queries/0_stateless/05293_index_hint_column_access.sh @@ -0,0 +1,48 @@ +#!/usr/bin/env bash + +# Tags: no-parallel-replicas + +# A column referenced only inside `indexHint` is not read, but index analysis prunes granules by it, +# so `count()` acts as an oracle for its values. It must require the same SELECT grant as a regular predicate. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +user="user_05293_${CLICKHOUSE_DATABASE}" +table="${CLICKHOUSE_DATABASE}.t_index_hint_access" + +${CLICKHOUSE_CLIENT} --query " + DROP TABLE IF EXISTS $table; + CREATE TABLE $table (id UInt64, secret UInt64, secret_alias ALIAS secret) + ENGINE = MergeTree ORDER BY secret SETTINGS index_granularity = 1; + INSERT INTO $table SELECT number, number * 1000 FROM numbers(10); + DROP USER IF EXISTS $user; + CREATE USER $user; + GRANT SELECT(id) ON $table TO $user; +" + +function query_as_user() +{ + echo "--- ${1//${CLICKHOUSE_DATABASE}./}" + ${CLICKHOUSE_CLIENT} --user "$user" --query "$1" 2>&1 | grep -oE '^[0-9]+$|ACCESS_DENIED' | uniq +} + +echo "=== granted: id" +query_as_user "SELECT count() FROM $table WHERE id = 3 AND indexHint(secret > 5000)" +query_as_user "SELECT count() FROM $table WHERE id = 3 AND indexHint(secret < 5000)" +query_as_user "SELECT count() FROM $table WHERE indexHint(secret < 5000)" +query_as_user "SELECT count() FROM $table PREWHERE indexHint(secret < 5000)" +query_as_user "SELECT count() FROM (SELECT id FROM $table WHERE indexHint(secret < 5000))" +query_as_user "SELECT count() FROM $table WHERE id = 3 AND indexHint(secret_alias > 5000)" +query_as_user "SELECT count() FROM $table AS l JOIN $table AS r ON l.id = r.id WHERE indexHint(r.secret < 5000)" +query_as_user "SELECT count() FROM $table WHERE indexHint(id < 5)" + +${CLICKHOUSE_CLIENT} --query "GRANT SELECT(secret, secret_alias) ON $table TO $user" + +echo "=== granted: id, secret, secret_alias" +query_as_user "SELECT count() FROM $table WHERE id = 3 AND indexHint(secret > 5000)" +query_as_user "SELECT count() FROM $table WHERE id = 3 AND indexHint(secret < 5000)" +query_as_user "SELECT count() FROM $table WHERE id = 3 AND indexHint(secret_alias > 5000)" + +${CLICKHOUSE_CLIENT} --query "DROP USER $user; DROP TABLE $table" diff --git a/tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.reference b/tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.reference new file mode 100644 index 000000000000..e2c68f15f4ec --- /dev/null +++ b/tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.reference @@ -0,0 +1 @@ +(1,'') (1) diff --git a/tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.sql b/tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.sql new file mode 100644 index 000000000000..f92699c6879d --- /dev/null +++ b/tests/queries/0_stateless/05294_metadata_only_alter_guard_alias_dependency.sql @@ -0,0 +1,28 @@ +-- A stored MATERIALIZED column that reads the altered named tuple only through an ALIAS column +-- must block a metadata-only ALTER exactly like a direct reference does: nothing would recompute it. + +DROP TABLE IF EXISTS t_guard_plain_alias; +DROP TABLE IF EXISTS t_guard_function_alias; + +CREATE TABLE t_guard_plain_alias (t Tuple(a Int64), t_alias ALIAS t, m String MATERIALIZED toString(t_alias)) +ENGINE = MergeTree ORDER BY tuple(); +INSERT INTO t_guard_plain_alias (t) VALUES ((1)); + +CREATE TABLE t_guard_function_alias (t Tuple(a Int64), s ALIAS toString(t), m String MATERIALIZED s) +ENGINE = MergeTree ORDER BY tuple(); +INSERT INTO t_guard_function_alias (t) VALUES ((1)); + +ALTER TABLE t_guard_plain_alias MODIFY COLUMN t Tuple(a Int64, b String) +SETTINGS allow_metadata_only_named_tuple_alter = 1; -- { serverError ALTER_OF_COLUMN_IS_FORBIDDEN } + +ALTER TABLE t_guard_function_alias MODIFY COLUMN t Tuple(a Int64, b String) +SETTINGS allow_metadata_only_named_tuple_alter = 1; -- { serverError ALTER_OF_COLUMN_IS_FORBIDDEN } + +-- Without the lazy conversion the change runs as a full mutation that recomputes the column. +ALTER TABLE t_guard_function_alias MODIFY COLUMN t Tuple(a Int64, b String) +SETTINGS allow_metadata_only_named_tuple_alter = 0, mutations_sync = 2; + +SELECT t, m FROM t_guard_function_alias; + +DROP TABLE t_guard_plain_alias; +DROP TABLE t_guard_function_alias; From cd12432877aba6ae598a9dc5218a9876c2e2a7f0 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 10:30:31 +0000 Subject: [PATCH 088/185] Backport #121398 to 26.8: Skip the direct join when the right table has a row policy --- src/Functions/FunctionJoinGet.cpp | 12 +- src/Interpreters/JoinedTables.cpp | 20 ++ .../Access/ParserCreateRowPolicyQuery.cpp | 302 ++++++++++++++++++ src/Planner/PlannerJoinTree.cpp | 20 +- src/Storages/StorageJoin.cpp | 2 + .../05233_direct_join_row_policy.reference | 43 +++ .../05233_direct_join_row_policy.sql | 63 ++++ .../05234_row_policy_storage_join.reference | 12 + .../05234_row_policy_storage_join.sql | 31 ++ 9 files changed, 500 insertions(+), 5 deletions(-) create mode 100644 tests/queries/0_stateless/05233_direct_join_row_policy.reference create mode 100644 tests/queries/0_stateless/05233_direct_join_row_policy.sql create mode 100644 tests/queries/0_stateless/05234_row_policy_storage_join.reference create mode 100644 tests/queries/0_stateless/05234_row_policy_storage_join.sql diff --git a/src/Functions/FunctionJoinGet.cpp b/src/Functions/FunctionJoinGet.cpp index e3d7b87f97d8..3c71c00821c1 100644 --- a/src/Functions/FunctionJoinGet.cpp +++ b/src/Functions/FunctionJoinGet.cpp @@ -11,6 +11,8 @@ #include #include #include +#include +#include namespace DB { @@ -21,6 +23,7 @@ namespace Setting namespace ErrorCodes { + extern const int ACCESS_DENIED; extern const int ILLEGAL_TYPE_OF_ARGUMENT; extern const int NUMBER_OF_ARGUMENTS_DOESNT_MATCH; } @@ -146,7 +149,14 @@ ExecutableFunctionPtr FunctionJoinGet::prepare(const ColumnsWithTypeAndName &) c Names column_names = storage_join->getKeyNames(); column_names.push_back(attr_name); - context->checkAccess(AccessType::SELECT, storage_join->getStorageID(), column_names); + const auto storage_id = storage_join->getStorageID(); + context->checkAccess(AccessType::SELECT, storage_id, column_names); + + /// The hash table is read as is, so a row policy on the table cannot be applied here any more than in a JOIN. + auto row_policy_filter = context->getRowPolicyFilter(storage_id.getDatabaseName(), storage_id.getTableName(), RowPolicyFilterType::SELECT_FILTER); + if (row_policy_filter && !row_policy_filter->isAlwaysTrue()) + throw Exception(ErrorCodes::ACCESS_DENIED, + "Cannot use {} because a row policy is applied on table {} with the Join engine", function_name, storage_id.getNameForLogs()); return std::make_unique(function_name, context, table_lock, storage_join, result_columns); } diff --git a/src/Interpreters/JoinedTables.cpp b/src/Interpreters/JoinedTables.cpp index ebdc9bd48d15..c64b2735268a 100644 --- a/src/Interpreters/JoinedTables.cpp +++ b/src/Interpreters/JoinedTables.cpp @@ -1,5 +1,7 @@ #include +#include +#include #include #include @@ -40,6 +42,7 @@ namespace Setting namespace ErrorCodes { + extern const int ACCESS_DENIED; extern const int ALIAS_REQUIRED; extern const int AMBIGUOUS_COLUMN_NAME; extern const int LOGICAL_ERROR; @@ -344,6 +347,23 @@ std::shared_ptr JoinedTables::makeTableJoin(const ASTSelectQuery & se { auto joined_table_id = context->resolveStorageID(table_to_join.database_and_table_name); StoragePtr storage = DatabaseCatalog::instance().tryGetTable(joined_table_id, context); + + /// A special storage replaces the right-side plan, and with it the `FilterStep` carrying the + /// table's row policy, so such a table has to be joined as an ordinary stream. A `Join` table + /// is a prebuilt hash table read as is, so it cannot be filtered at all. + if (storage) + { + auto row_policy_filter = context->getRowPolicyFilter( + joined_table_id.getDatabaseName(), joined_table_id.getTableName(), RowPolicyFilterType::SELECT_FILTER); + if (row_policy_filter && !row_policy_filter->isAlwaysTrue()) + { + if (typeid_cast(storage.get())) + throw Exception(ErrorCodes::ACCESS_DENIED, + "Cannot join table {} with the Join engine because a row policy is applied on it", joined_table_id.getNameForLogs()); + storage = nullptr; + } + } + if (storage) { if (auto storage_join = std::dynamic_pointer_cast(storage); storage_join) diff --git a/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp b/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp index 63dcf9769961..e6e8bea7217d 100644 --- a/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp +++ b/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp @@ -315,3 +315,305 @@ bool ParserCreateRowPolicyQuery::parseImpl(Pos & pos, ASTPtr & node, Expected & return true; } } + +namespace DB +{ + +void registerStatementRowPolicy(StatementFactory & factory) +{ + factory.registerStatement("CREATE ROW POLICY", + { + .description = R"DOCS_MD( +Creates a [row policy](/concepts/features/security/access-rights#row-policy-management), i.e. a filter used to determine which rows a user can read from a table. + + +Row policies make sense only for users with readonly access. If a user can modify a table or copy partitions between tables, it defeats the restrictions of row policies. + + +Syntax: + +```sql +-- Multiple names on one table target +CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] policy_name [, ...] + [ON CLUSTER cluster_name] + ON { [db.]table | db.* } + [IN access_storage_type] + [[FOR SELECT] USING {condition | NONE}] + [AS {PERMISSIVE | RESTRICTIVE}] + [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] + +-- One name on multiple table targets +CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] policy_name + [ON CLUSTER cluster_name] + ON { [db.]table | db.* } [, ...] + [IN access_storage_type] + [[FOR SELECT] USING {condition | NONE}] + [AS {PERMISSIVE | RESTRICTIVE}] + [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] + +-- Mixed packing: each name paired with its own table target +CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] + policy_name ON { [db.]table | db.* } [, policy_name ON { [db.]table | db.* } ...] + [ON CLUSTER cluster_name] + [IN access_storage_type] + [[FOR SELECT] USING {condition | NONE}] + [AS {PERMISSIVE | RESTRICTIVE}] + [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] +``` + +`ParserRowPolicyNames` accepts **three** packing forms (not a full Cartesian product): + +1. **Multiple names, one target** — `pol1, pol2 ON table1` creates each listed name on that single table (or `db.*`). +2. **One name, multiple targets** — `pol1 ON table1, table2` creates the same short name on each listed target. +3. **Mixed pairs** — `p1 ON t1, p2 ON t2` creates each name only on its paired target. + +A multi-name list **cannot** be combined with a multi-table `ON` list in one group: `p1, p2 ON t1, t2` is rejected. After a multi-name group, you also cannot append another comma-separated `name ON target` group in the same statement. + +Optional `ON CLUSTER` applies to the whole statement (one cluster name). ClickHouse does **not** accept a different `ON CLUSTER` per policy name packed into a single create — run separate `CREATE ROW POLICY` statements when policies must be created on different clusters. + +`CREATE ROW POLICY` requires the [CREATE ROW POLICY](/reference/statements/grant#access-management) privilege on the table the policy is created on. `OR REPLACE` throws away an existing policy of the same name, including which roles it applies to, so it additionally requires the [DROP ROW POLICY](/reference/statements/grant#access-management) privilege on that table. The `DROP ROW POLICY` privilege is required whether or not the policy already exists, so the statement cannot be used to find out which policies exist. + +## Multiple names and tables {#multiple-names-and-tables} + +Valid: + +```sql +-- Several policy names, one table +CREATE ROW POLICY pol1, pol2, pol3 ON table1 + FOR SELECT USING id = 1 + TO accountant; + +-- One policy name, several tables +CREATE ROW POLICY IF NOT EXISTS pol1 ON table1, table2, table3 + FOR SELECT USING id = 1 + TO accountant; + +-- Mixed packing: different name per table +CREATE ROW POLICY p4 ON db.table, p5 ON db2.table2 + USING a = b; + +-- Same policy on several tables, on a cluster +CREATE ROW POLICY IF NOT EXISTS pol1 ON CLUSTER replicated_cluster ON table1, table2 + FOR SELECT USING id = 1 + TO accountant; +``` + +Invalid: + +```sql +-- Multi-name × multi-table in one ON-group (not a Cartesian product) +CREATE ROW POLICY p1, p2 ON t1, t2 + FOR SELECT USING id = 1 + TO accountant; + +-- Different clusters per name in one statement +CREATE ROW POLICY pol1 ON CLUSTER cluster1 ON table1, pol2 ON CLUSTER cluster2 ON table2 +``` + +## USING clause {#using-clause} + +Defines a filter condition for a table. A user can only see rows for which the condition is true (evaluates to a non-zero value). This is similar to adding an extra `WHERE` condition to every query the user runs against the table. + +For example, the following policy limits `analyst_role` to rows from the EU: + +```sql +CREATE ROW POLICY region_filter ON db.orders +USING region = 'EU' +TO analyst_role; +``` + +With this policy, `SELECT * FROM db.orders` returns the same rows as `SELECT * FROM db.orders WHERE region = 'EU'` would. + +## TO Clause {#to-clause} + +In the `TO` section you can provide a list of users and roles this policy should work for. For example, `CREATE ROW POLICY ... TO accountant, john@localhost`. + +Keyword `ALL` means all the ClickHouse users, including current user. Keyword `ALL EXCEPT` allows excluding some users from the all users list, for example, `CREATE ROW POLICY ... TO ALL EXCEPT accountant, john@localhost` + +Roles named in the `TO` section, including those after `ALL EXCEPT`, are matched against the current user's enabled roles ([`system.enabled_roles`](/reference/system-tables/enabled_roles)), not against every role granted to the user, so [`SET ROLE`](/reference/statements/set-role) can change which policies apply. + +## AS Clause {#as-clause} + +It's allowed to have more than one policy enabled on the same table for the same user at one time. So we need a way to combine the conditions from multiple policies. + +By default, policies are combined using the boolean `OR` operator. For example, the following policies: + +```sql +CREATE ROW POLICY pol1 ON mydb.table1 USING b=1 TO mira, peter +CREATE ROW POLICY pol2 ON mydb.table1 USING c=2 TO peter, antonio +``` + +enable the user `peter` to see rows with either `b=1` or `c=2`. + +The `AS` clause specifies how policies should be combined with other policies. Policies can be either permissive or restrictive. By default, policies are permissive, which means they are combined using the boolean `OR` operator. + +A policy can be defined as restrictive as an alternative. Restrictive policies are combined using the boolean `AND` operator. + +Here is the general formula: + +```text +row_is_visible = (one or more of the conditions from the permissive policies that apply to the current user and their enabled roles are non-zero) AND + (all of the conditions from the restrictive policies that apply to the current user and their enabled roles are non-zero) +``` + +If no permissive condition applies, the first condition has no effect and only the restrictive policies decide, because `access_control_improvements.users_without_row_policies_can_read_rows` is enabled by default. A user to whom no condition applies therefore sees every row, and `access_control_improvements.throw_on_unmatched_row_policies`, disabled by default, raises an exception instead when the table does have conditions and none of them apply. + +For example, the following policies: + +```sql +CREATE ROW POLICY pol1 ON mydb.table1 USING b=1 TO mira, peter +CREATE ROW POLICY pol2 ON mydb.table1 USING c=2 AS RESTRICTIVE TO peter, antonio +``` + +enable the user `peter` to see rows only if both `b=1` AND `c=2`. + +Database policies are combined with table policies. + +For example, the following policies: + +```sql +CREATE ROW POLICY pol1 ON mydb.* USING b=1 TO mira, peter +CREATE ROW POLICY pol2 ON mydb.table1 USING c=2 AS RESTRICTIVE TO peter, antonio +``` + +enable the user `peter` to see table1 rows only if both `b=1` AND `c=2`, although +any other table in mydb would have only `b=1` policy applied for the user. + +## Tables that read from other tables {#tables-that-read-from-other-tables} + +A row policy filters rows where the data is actually read. An `Alias` table returns the rows of its target table as its own, so the row policies of the target apply to reads through the alias as well, combined with the policies of the alias itself using a logical `AND`. A `Merge` table applies the policies of the tables it reads from. One exception: when a matched table reads remotely, such as a `Distributed` table, the remote server processes the query before the policy is applied, because the policy runs above that table's read rather than at the read. Such a query can fail, when it aggregates without selecting the policy's columns, or return fewer rows than the policy allows, when the remote server applies an `ORDER BY ... LIMIT` to rows the policy would have hidden. Define the policy on the underlying local tables of each remote server instead. + +This does not extend to every table that reads from another table. A `Buffer` table and a materialized view read through their destination or target table do **not** inherit that table's row policies: the policy is written against the target's schema and, for a view with `SQL SECURITY DEFINER`, is evaluated for a different user than the one running the read. Define the policy on the table users actually query in those cases. + +## Distributed and remote-backed tables {#distributed-and-remote-backed-tables} + +A row policy filters rows where the table data is actually read. A table that delegates reading to remote servers, such as a [Distributed](/reference/engines/table-engines/special/distributed) table or a wrapper over one (for example, a materialized view with a `Distributed` target), only ships the query text to the remote servers and cannot apply the policy filter to the remote read. To keep the filter from being silently dropped, queries to such a table by users the policy applies to are rejected with an `ILLEGAL_PREWHERE` error. + +Instead, define the policy on the underlying local tables on each remote server; it is applied there when the shipped query reads them: + +```sql +-- Filters reads of local_table on this server, including reads shipped by a Distributed table over it. +CREATE ROW POLICY filter ON mydb.local_table USING a < 1000 TO john; +``` + + +This works while the query is shipped as text, which is the default. With [`serialize_query_plan = 1`](/reference/settings/session-settings/serialize#serialize_query_plan) the initiator ships an already-built read plan instead, and a remote server executing such a plan does not apply its own row policies, so a read of a `Distributed` table over `local_table` returns unfiltered rows. Keep `serialize_query_plan = 0` for users whose row policies must be enforced. See [issue #112891](https://github.com/ClickHouse/ClickHouse/issues/112891). + + +## Join tables {#join-tables} + +A [Join](/reference/engines/table-engines/special/join) table is a prepared hash table that a `JOIN` or `joinGet` reads as is, so its rows cannot be filtered there. A policy on such a table, including a database-wide `ON db.*` policy, filters a plain `SELECT` from the table, but while it applies, `JOIN` and `joinGet` queries against the table fail with `ACCESS_DENIED`. + +## ON CLUSTER Clause {#on-cluster-clause} + +Allows creating row policies on a cluster, see [Distributed DDL](/reference/statements/distributed-ddl). This is also the convenient way to create the policy on the local tables of every server of the cluster. + +## Examples {#examples} + +`CREATE ROW POLICY filter1 ON mydb.mytable USING a<1000 TO accountant, john@localhost` + +`CREATE ROW POLICY filter2 ON mydb.mytable USING a<1000 AND b=5 TO ALL EXCEPT mira` + +`CREATE ROW POLICY filter3 ON mydb.mytable USING 1 TO admin` + +`CREATE ROW POLICY filter4 ON mydb.* USING 1 TO admin` +)DOCS_MD", + .syntax = R"( +CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] policy_name [, ...] + [ON CLUSTER cluster_name] + ON { [db.]table | db.* } [, ...] + [IN access_storage_type] + [[FOR SELECT] USING {condition | NONE}] + [AS {PERMISSIVE | RESTRICTIVE}] + [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] +)", + .parent = "CREATE", + .related = {"ALTER ROW POLICY", "CREATE MASKING POLICY", "CREATE ROLE", "DROP", "SHOW"}, + }); + + factory.registerStatement("ALTER ROW POLICY", + { + .description = R"DOCS_MD( +Changes row policy. + +Syntax: + +```sql +-- Rename: exactly one fully qualified policy (one name on one target). +-- RENAME TO may be combined with the same optional alteration clauses as below. +ALTER [ROW] POLICY [IF EXISTS] name + ON { [database.]table | database.* } + RENAME TO new_name + [ON CLUSTER cluster_name] + [AS {PERMISSIVE | RESTRICTIVE}] + [FOR SELECT] + [USING {condition | NONE}][,...] + [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] + +-- Multiple names on one table target (no RENAME) +ALTER [ROW] POLICY [IF EXISTS] name [, ...] + [ON CLUSTER cluster_name] + ON { [database.]table | database.* } + [AS {PERMISSIVE | RESTRICTIVE}] + [FOR SELECT] + [USING {condition | NONE}][,...] + [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] + +-- One name on multiple table targets (no RENAME) +ALTER [ROW] POLICY [IF EXISTS] name + [ON CLUSTER cluster_name] + ON { [database.]table | database.* } [, ...] + [AS {PERMISSIVE | RESTRICTIVE}] + [FOR SELECT] + [USING {condition | NONE}][,...] + [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] + +-- Mixed packing: each name paired with its own table target (no RENAME) +ALTER [ROW] POLICY [IF EXISTS] + name ON { [database.]table | database.* } [, name ON { [database.]table | database.* } ...] + [ON CLUSTER cluster_name] + [AS {PERMISSIVE | RESTRICTIVE}] + [FOR SELECT] + [USING {condition | NONE}][,...] + [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] +``` + +`RENAME TO` is only accepted when the statement names **exactly one** policy on **one** table target. Packed multi-name or multi-table lists cannot include `RENAME TO` — rename those policies in separate statements. On that single-policy form, `RENAME TO` can still be combined with other alterations such as `AS`, `USING`, and `TO` in the same statement. + +Without `RENAME TO`, packing matches `ParserRowPolicyNames`: multiple names on **one** target, one name on **multiple** targets, or mixed `name ON target` pairs. Multi-name × multi-table (`p1, p2 ON t1, t2`) is **not** accepted. Optional `ON CLUSTER` applies once to the whole statement; run separate `ALTER ROW POLICY` statements when different clusters are required. + +Examples: + +```sql +-- Single-policy rename +ALTER ROW POLICY p1 ON db.table RENAME TO p1_new; + +-- Rename plus other alterations on the same single policy +ALTER POLICY old_name ON db.table RENAME TO new_name USING id > 10; + +-- Multiple names, one table +ALTER POLICY p1, p2 ON db.table TO ALL; + +-- One name, multiple tables +ALTER POLICY p1 ON db.table, db.table2 USING NONE; + +-- Mixed targets without rename +ALTER POLICY p1 ON db.table, p2 ON db2.table2 TO ALL; +``` +)DOCS_MD", + .syntax = R"( +ALTER [ROW] POLICY [IF EXISTS] name [, ...] + ON { [database.]table | database.* } [, ...] + [RENAME TO new_name] + [ON CLUSTER cluster_name] + [AS {PERMISSIVE | RESTRICTIVE}] + [FOR SELECT] + [USING {condition | NONE}][,...] + [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] +)", + .parent = "ALTER", + .related = {"CREATE ROW POLICY", "ALTER", "SHOW"}, + }); +} + +} diff --git a/src/Planner/PlannerJoinTree.cpp b/src/Planner/PlannerJoinTree.cpp index 67ca7fb5945a..2e0ee3dbd986 100644 --- a/src/Planner/PlannerJoinTree.cpp +++ b/src/Planner/PlannerJoinTree.cpp @@ -31,6 +31,7 @@ #include #include #include +#include #include #include #include @@ -184,6 +185,7 @@ namespace ErrorCodes extern const int PARAMETER_OUT_OF_BOUND; extern const int TOO_MANY_COLUMNS; extern const int UNSUPPORTED_METHOD; + extern const int ACCESS_DENIED; } namespace @@ -3121,12 +3123,22 @@ JoinTreeQueryPlan buildQueryPlanForJoinNode( join_node, planner_context); - PreparedJoinStorage prepared_join; - bool allow_storage_join = right_join_tree_query_plan.used_row_policies.empty() + /// A prepared storage replaces the right-side plan, and with it the `FilterStep` applying the + /// table's row policy, so such a table is joined as a stream. A `Join` table is a prebuilt + /// hash table read as is, so it cannot be filtered at all. + bool right_table_has_row_policy = !right_join_tree_query_plan.used_row_policies.empty(); + + PreparedJoinStorage prepared_join = tryGetStorageInTableJoin(join_node.getRightTableExpressionNode(), planner_context); + if (prepared_join.storage_join && right_table_has_row_policy) + throw Exception(ErrorCodes::ACCESS_DENIED, + "Cannot join table {} with the Join engine because a row policy is applied on it", + prepared_join.storage_join->getStorageID().getNameForLogs()); + + bool allow_storage_join = !right_table_has_row_policy && right_join_tree_query_plan.stage == QueryProcessingStage::FetchColumns && right_join_tree_query_plan.useful_sets.empty(); - if (allow_storage_join) - prepared_join = tryGetStorageInTableJoin(join_node.getRightTableExpressionNode(), planner_context); + if (!allow_storage_join) + prepared_join = {}; if (prepared_join) { bool use_nulls = settings[Setting::join_use_nulls] && isLeftOrFull(join_node.getKind()); diff --git a/src/Storages/StorageJoin.cpp b/src/Storages/StorageJoin.cpp index cbbb62afde5a..cdc7a5c14c2f 100644 --- a/src/Storages/StorageJoin.cpp +++ b/src/Storages/StorageJoin.cpp @@ -641,6 +641,8 @@ Default value: `1`. The `Join`-engine tables can't be used in `GLOBAL JOIN` operations. +A [row policy](/reference/statements/create/row-policy) on a `Join`-engine table filters a plain `SELECT` from it, but a `JOIN` or `joinGet` reads the prepared hash table as is and cannot filter its rows, so while a policy applies to the table such queries fail with `ACCESS_DENIED`. + The `Join`-engine allows to specify [join_use_nulls](/reference/settings/session-settings/join#join_use_nulls) setting in the `CREATE TABLE` statement. [SELECT](/reference/statements/select/index) query should have the same `join_use_nulls` value. ## Usage examples {#example} diff --git a/tests/queries/0_stateless/05233_direct_join_row_policy.reference b/tests/queries/0_stateless/05233_direct_join_row_policy.reference new file mode 100644 index 000000000000..523709ccc754 --- /dev/null +++ b/tests/queries/0_stateless/05233_direct_join_row_policy.reference @@ -0,0 +1,43 @@ +-- plain select +1 public public-1 +3 public public-3 +-- inner +1 public public-1 +3 public public-3 +-- inner, policy column not selected +1 public-1 +3 public-3 +-- left +1 1 public-1 +2 0 +3 3 public-3 +5 0 +-- left, join_use_nulls +1 1 public-1 +2 \N \N +3 3 public-3 +5 \N \N +-- left any +1 public-1 +2 +3 public-3 +5 +-- left semi +1 +3 +-- left anti +2 +5 +-- key space enumeration +1 public-1 +3 public-3 +-- using +1 public-1 +3 public-3 +-- no direct join under a row policy +0 +-- direct join without a policy +1 public public-1 +2 hidden hidden-2 +3 public public-3 +Algorithm: DirectKeyValueJoin diff --git a/tests/queries/0_stateless/05233_direct_join_row_policy.sql b/tests/queries/0_stateless/05233_direct_join_row_policy.sql new file mode 100644 index 000000000000..e2a073ca13a2 --- /dev/null +++ b/tests/queries/0_stateless/05233_direct_join_row_policy.sql @@ -0,0 +1,63 @@ +-- Tags: use-rocksdb + +DROP TABLE IF EXISTS kv_rls; +DROP TABLE IF EXISTS probe_rls; + +CREATE TABLE kv_rls (key UInt64, tenant String, secret String) ENGINE = EmbeddedRocksDB PRIMARY KEY key; +INSERT INTO kv_rls VALUES (1, 'public', 'public-1'), (2, 'hidden', 'hidden-2'), (3, 'public', 'public-3'), (4, 'hidden', 'hidden-4'); + +CREATE TABLE probe_rls (key UInt64) ENGINE = TinyLog; +INSERT INTO probe_rls VALUES (1), (2), (3), (5); + +CREATE ROW POLICY kv_rls_public ON kv_rls FOR SELECT USING tenant = 'public' TO CURRENT_USER; + +SET join_algorithm = 'direct,hash'; +SET join_use_nulls = 0; + +SELECT '-- plain select'; +SELECT key, tenant, secret FROM kv_rls ORDER BY key; + +SELECT '-- inner'; +SELECT p.key, kv.tenant, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; + +SELECT '-- inner, policy column not selected'; +SELECT p.key, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; + +SELECT '-- left'; +SELECT p.key, kv.key, kv.secret FROM probe_rls AS p LEFT JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; + +SELECT '-- left, join_use_nulls'; +SELECT p.key, kv.key, kv.secret FROM probe_rls AS p LEFT JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key SETTINGS join_use_nulls = 1; + +SELECT '-- left any'; +SELECT p.key, kv.secret FROM probe_rls AS p LEFT ANY JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; + +SELECT '-- left semi'; +SELECT p.key FROM probe_rls AS p LEFT SEMI JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; + +SELECT '-- left anti'; +SELECT p.key FROM probe_rls AS p LEFT ANTI JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; + +SELECT '-- key space enumeration'; +SELECT kv.key, kv.secret FROM numbers(10) AS n INNER JOIN kv_rls AS kv ON kv.key = n.number ORDER BY kv.key; + +SELECT '-- using'; +SELECT key, secret FROM probe_rls INNER JOIN kv_rls USING (key) ORDER BY key; + +SELECT '-- no direct join under a row policy'; +SELECT count() FROM ( + EXPLAIN actions = 1 + SELECT p.key, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key +) WHERE explain LIKE '%DirectKeyValueJoin%'; + +DROP ROW POLICY kv_rls_public ON kv_rls; + +SELECT '-- direct join without a policy'; +SELECT p.key, kv.tenant, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; +SELECT extract(explain, 'Algorithm: \\w+') FROM ( + EXPLAIN actions = 1 + SELECT p.key, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key +) WHERE explain LIKE '%Algorithm:%'; + +DROP TABLE kv_rls; +DROP TABLE probe_rls; diff --git a/tests/queries/0_stateless/05234_row_policy_storage_join.reference b/tests/queries/0_stateless/05234_row_policy_storage_join.reference new file mode 100644 index 000000000000..2ac892e8f505 --- /dev/null +++ b/tests/queries/0_stateless/05234_row_policy_storage_join.reference @@ -0,0 +1,12 @@ +-- without a policy +1 a +2 b +3 +b \N +-- with a policy +1 a +-- after dropping the policy +1 a +2 b +3 +b \N diff --git a/tests/queries/0_stateless/05234_row_policy_storage_join.sql b/tests/queries/0_stateless/05234_row_policy_storage_join.sql new file mode 100644 index 000000000000..91d6f948686d --- /dev/null +++ b/tests/queries/0_stateless/05234_row_policy_storage_join.sql @@ -0,0 +1,31 @@ +DROP TABLE IF EXISTS join_rls; +DROP TABLE IF EXISTS probe_join_rls; +DROP ROW POLICY IF EXISTS join_rls_policy ON join_rls; + +CREATE TABLE join_rls (key UInt64, value String) ENGINE = Join(ANY, LEFT, key); +INSERT INTO join_rls VALUES (1, 'a'), (2, 'b'); + +CREATE TABLE probe_join_rls (key UInt64) ENGINE = TinyLog; +INSERT INTO probe_join_rls VALUES (1), (2), (3); + +SELECT '-- without a policy'; +SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; +SELECT joinGet(join_rls, 'value', toUInt64(2)), joinGetOrNull(join_rls, 'value', toUInt64(3)); + +CREATE ROW POLICY join_rls_policy ON join_rls FOR SELECT USING value = 'a' TO CURRENT_USER; + +SELECT '-- with a policy'; +SELECT key, value FROM join_rls ORDER BY key; +SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; -- { serverError ACCESS_DENIED } +SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key SETTINGS join_algorithm = 'hash'; -- { serverError ACCESS_DENIED } +SELECT joinGet(join_rls, 'value', toUInt64(2)); -- { serverError ACCESS_DENIED } +SELECT joinGetOrNull(join_rls, 'value', toUInt64(2)); -- { serverError ACCESS_DENIED } + +DROP ROW POLICY join_rls_policy ON join_rls; + +SELECT '-- after dropping the policy'; +SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; +SELECT joinGet(join_rls, 'value', toUInt64(2)), joinGetOrNull(join_rls, 'value', toUInt64(3)); + +DROP TABLE join_rls; +DROP TABLE probe_join_rls; From 4a0ed782f3d39bec3303a94858308d87ba5fc989 Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 06:35:24 +0000 Subject: [PATCH 089/185] Hide credentials in NATS/RabbitMQ addresses, XDBC connection strings and URL settings Six call sites still masked a credential embedded in a URL-shaped value with maskURIPassword, an unanchored scan that reproduces the old regular expression ([^:]+://[^:]*):([^@]*)@(.*) exactly: it masks from the first ':' after any '://' to the first '@'. The masked extent has to cover whatever the value's consumer treats as a credential, and that scan matches no consumer in the tree, so it is at once too narrow (a password containing '@' keeps its tail, and a userinfo with no password is not masked at all) and too wide (a credential-free URL whose query string contains '@' is rewritten into a host that is not the real one). All three show up in SHOW CREATE, system.tables and system.query_log. Swapping in maskURIUserinfo at every site, which is what the issue asks for, would replace a partial leak with a full one at five of the six. That function implements the RFC 3986 bound, which is Poco::URI's bound, and three of these consumers are not Poco::URI: - libnats trims the value, takes the scheme as optional and ends the userinfo at the LAST '@' of the whole remainder, unbounded by the '/?#' that closes an authority (contrib/nats-io/src/url.c, natsUrl_Create; both halves go into the CONNECT frame in conn.c). So nats://u:pa/ss@h:4222 authenticates with password pa/ss, which maskURIUserinfo leaves untouched, and user:pass@h:4222 authenticates while neither masker touches it. - AMQP-CPP ends the login at the FIRST '@' after the scheme, again unbounded (contrib/AMQP-CPP/include/amqpcpp/address.h). - an XDBC connection string is never parsed by ClickHouse: it is read verbatim in ITableFunctionXDBC and forwarded to the bridge as one opaque parameter, so its grammar is the driver's. This repository's own examples put the password in a query parameter, and the majority spelling is DSN=...;Pwd=... . No URI scan can bound a credential in that. A password containing '/' is an ordinary shape, so this is not a corner case. Apply one rule instead: a value is echoed with only its credential hidden where ClickHouse owns its grammar, and a value whose grammar belongs to someone else is not echoed. The second half is already the policy in DatabaseURL.cpp and in the old XDBC fallback. Concretely, the two broker addresses and the XDBC connection string are hidden whole (for the brokers, exactly when the value carries an '@', which is a superset of anything either parser can read as a credential and is empty precisely when there is nothing to hide, so the documented forms such as localhost:4222 and amqp://h:5672/v stay fully visible). The three query-level URL settings are read through Poco::URI, so there the partial mask is right and is kept: it moves from maskURIPassword to maskURIUserinfo, which is what fixes the reporter's three cases. They gain a fail-closed leg, because nothing validates them at SET time, resolveURLBase accepts a base whose '://' is anywhere rather than at the start, and a statement is masked for logging before its settings are validated: a value with no scheme in front of it therefore reaches a log, and no parser can say which part of it is the credential. url_base additionally reaches the Azure URL parser, where abfs and abfss put a non-secret container in front of the '@', so it gets its own rule that skips the userinfo leg for those two schemes; az and azure have no '@' in their grammar and need no exception. Two holes in the XDBC named-collection branch close as a side effect of dropping its bespoke helper for findSecretNamedArgument: only the first occurrence of a duplicated datasource key used to be masked, and a key written as a constant expression (concat('data', 'source') = ...) was applied by the named-collection parser while being invisible to this finder, so the whole connection string was echoed. The fail-closed scan that already covered that hazard for the TLS keys is extracted and reused, leaving one carrier for it. The anchored-scheme test is likewise reduced to one carrier, findURIAuthority, which also replaces a private copy in DatabaseURL.cpp that used the locale-dependent std::isalpha and std::isalnum. Deliberate consequences: a broker address carrying a stray '@' and any URI-shaped connection string are now hidden whole, and format_display_secrets_in_show_and_select still shows the real value to a privileged reader. maskURIUserinfo itself is unchanged, which the re2 equivalence test proves by passing unedited. Closes: https://github.com/ClickHouse/ClickHouse/issues/121257 --- src/Common/maskURIPassword.h | 40 ++++++--- src/Core/SettingsSecrets.h | 40 ++++++++- src/Databases/DatabaseURL.cpp | 25 +----- src/Parsers/FunctionSecretArgumentsFinder.cpp | 72 ++++++---------- src/Parsers/FunctionSecretArgumentsFinder.h | 12 +-- src/Parsers/tests/gtest_Parser.cpp | 7 +- src/Storages/NATS/NATS_fwd.h | 8 +- src/Storages/RabbitMQ/RabbitMQ_fwd.h | 8 +- .../test_mask_sensitive_info/test.py | 20 ++--- .../integration/test_storage_rabbitmq/test.py | 2 +- ...etting_ast_json_and_secret_parts.reference | 2 +- .../05056_settings_secret_masking.reference | 10 +-- ...oker_and_xdbc_credential_masking.reference | 28 ++++++ ...5233_broker_and_xdbc_credential_masking.sh | 85 +++++++++++++++++++ 14 files changed, 244 insertions(+), 115 deletions(-) create mode 100644 tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference create mode 100755 tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh diff --git a/src/Common/maskURIPassword.h b/src/Common/maskURIPassword.h index afc505b185e8..d1f88c5e43b7 100644 --- a/src/Common/maskURIPassword.h +++ b/src/Common/maskURIPassword.h @@ -73,15 +73,11 @@ inline bool maskURIPassword(std::string * uri) return false; } -/** Mask the userinfo part of a URL: `scheme://anything@rest` becomes `scheme://[HIDDEN]@rest`. - * Returns whether anything was masked. - * - * This used to be the regular expression `^([a-zA-Z][a-zA-Z0-9+.-]*://)[^/?#]+@` rewritten to - * `\1[HIDDEN]@`. Only a match at the start of the string counts, and the userinfo is taken - * greedily up to the last '@' before the path, so a password that itself contains an at-sign is - * masked whole. `src/Common/tests/gtest_mask_uri_password.cpp` checks this against re2. +/** The offset just past the `://` of a value that starts with an RFC 3986 scheme, and `npos` when it + * does not start with one. A value with no scheme at its start has no authority that a URI parser + * would recognise, so nothing in it can be located as a credential by position. */ -inline bool maskURIUserinfo(std::string & url) +inline size_t findURIAuthority(std::string_view uri) { static constexpr std::string_view SEPARATOR = "://"; @@ -90,19 +86,35 @@ inline bool maskURIUserinfo(std::string & url) auto is_letter = [](char c) { return ('a' <= c && c <= 'z') || ('A' <= c && c <= 'Z'); }; auto is_letter_or_digit = [&](char c) { return is_letter(c) || ('0' <= c && c <= '9'); }; - if (url.empty() || !is_letter(url[0])) - return false; + if (uri.empty() || !is_letter(uri[0])) + return std::string_view::npos; size_t scheme_end = 1; - while (scheme_end < url.length() - && (is_letter_or_digit(url[scheme_end]) || url[scheme_end] == '+' || url[scheme_end] == '.' || url[scheme_end] == '-')) + while (scheme_end < uri.length() + && (is_letter_or_digit(uri[scheme_end]) || uri[scheme_end] == '+' || uri[scheme_end] == '.' || uri[scheme_end] == '-')) ++scheme_end; - if (url.compare(scheme_end, SEPARATOR.length(), SEPARATOR) != 0) + if (uri.compare(scheme_end, SEPARATOR.length(), SEPARATOR) != 0) + return std::string_view::npos; + + return scheme_end + SEPARATOR.length(); +} + +/** Mask the userinfo part of a URL: `scheme://anything@rest` becomes `scheme://[HIDDEN]@rest`. + * Returns whether anything was masked. + * + * This used to be the regular expression `^([a-zA-Z][a-zA-Z0-9+.-]*://)[^/?#]+@` rewritten to + * `\1[HIDDEN]@`. Only a match at the start of the string counts, and the userinfo is taken + * greedily up to the last '@' before the path, so a password that itself contains an at-sign is + * masked whole. `src/Common/tests/gtest_mask_uri_password.cpp` checks this against re2. + */ +inline bool maskURIUserinfo(std::string & url) +{ + size_t authority_begin = findURIAuthority(url); + if (authority_begin == std::string::npos) return false; /// `[^/?#]+@` - the userinfo, greedy, so it ends at the last '@' before the path. - size_t authority_begin = scheme_end + SEPARATOR.length(); size_t authority_end = url.find_first_of("/?#", authority_begin); if (authority_end == std::string::npos) authority_end = url.length(); diff --git a/src/Core/SettingsSecrets.h b/src/Core/SettingsSecrets.h index dd6ad89674a7..77d883ee75ec 100644 --- a/src/Core/SettingsSecrets.h +++ b/src/Core/SettingsSecrets.h @@ -27,14 +27,48 @@ using ValueMaskingFunc = std::function; /// precondition holds by construction and needs no check. inline bool maskURLCredentials(String & value) { - bool masked = maskURIPassword(&value); + /// Nothing validates these settings when they are set, `StorageURL::resolveURLBase` accepts a base + /// whose `://` is anywhere rather than at the start, and a statement is masked for logging before + /// its settings are validated. So a value with no scheme in front still reaches a log, and no URI + /// parser can locate the credential inside it: hide such a value whole. + if (findURIAuthority(value) == String::npos && value.contains('@')) + { + value = "[HIDDEN]"; + return true; + } + + bool masked = maskURIUserinfo(value); masked |= maskPresignedURLParameters(value); return masked; } +/// Under `abfs`/`abfss` the part in front of the `@` is the container and not a credential (the +/// Hadoop grammar `abfss://@.dfs.core.windows.net/`), and the credential +/// is the SAS in the query string. `az`/`azure` have no `@` in their grammar and need no exception. +/// Keep the two names in sync with `parseAzureURL` in `src/Storages/StorageURL.cpp`, which compares +/// them after lowercasing; `Core` cannot depend on `Storages` to share the list. +inline bool maskURLBaseCredentials(String & value) +{ + static constexpr std::string_view azure_container_schemes[] = {"abfs", "abfss"}; + + if (size_t authority = findURIAuthority(value); authority != String::npos) + { + String scheme = value.substr(0, authority - 3); + for (auto & c : scheme) + if ('A' <= c && c <= 'Z') + c += 'a' - 'A'; + + for (auto azure_scheme : azure_container_schemes) + if (scheme == azure_scheme) + return maskPresignedURLParameters(value); + } + + return maskURLCredentials(value); +} + /// The settings of the query-level `Settings` collection whose value can carry a credential, and how /// each one is masked. `system.query_log.query` shows -/// `format_avro_schema_registry_url = 'http://user:[HIDDEN]@registry:8080'`, so every other place that +/// `format_avro_schema_registry_url = 'http://[HIDDEN]@registry:8080'`, so every other place that /// prints the same value hides the same secret through this map. /// /// Mirrors the per-engine `SETTINGS_TO_HIDE` maps (`Kafka_fwd.h`, `NATS_fwd.h`, ...), which do this @@ -42,7 +76,7 @@ inline bool maskURLCredentials(String & value) static inline std::unordered_map SETTINGS_TO_HIDE = { {"format_avro_schema_registry_url", maskURLCredentials}, - {"url_base", maskURLCredentials}, + {"url_base", maskURLBaseCredentials}, {"s3_base", maskURLCredentials}, }; diff --git a/src/Databases/DatabaseURL.cpp b/src/Databases/DatabaseURL.cpp index b08e1332830f..5ff8e6312e0f 100644 --- a/src/Databases/DatabaseURL.cpp +++ b/src/Databases/DatabaseURL.cpp @@ -17,6 +17,7 @@ #include #include #include +#include #include #include @@ -42,26 +43,6 @@ namespace ErrorCodes namespace { -/// Check that the string starts with a valid RFC 3986 scheme followed by "://". -bool hasURLScheme(const String & url) -{ - auto scheme_end = url.find("://"); - if (scheme_end == String::npos || scheme_end == 0) - return false; - - if (!std::isalpha(static_cast(url[0]))) - return false; - - for (size_t i = 1; i < scheme_end; ++i) - { - char c = url[i]; - if (!std::isalnum(static_cast(c)) && c != '+' && c != '-' && c != '.') - return false; - } - - return true; -} - /// A table of a `URL` database is a thin wrapper over the `url` table function, which dispatches /// the URL to the matching backend (`file`, `s3`, `azureBlobStorage`, ...). struct URLTableDelegate @@ -265,7 +246,7 @@ DatabaseURL::DatabaseURL(const String & name_, const String & base_url_, Context : IDatabase(name_), WithContext(context_->getGlobalContext()), base_url(base_url_) { /// Not echoed back: password masking anchors on the `://` this value lacks, so it would log the password. - if (!base_url.empty() && !hasURLScheme(base_url)) + if (!base_url.empty() && findURIAuthority(base_url) == String::npos) throw Exception(ErrorCodes::BAD_ARGUMENTS, "The base URL of a URL database must contain a scheme (e.g. https://)"); } @@ -274,7 +255,7 @@ String DatabaseURL::getTableURL(const String & name) const { String resolved = StorageURL::resolveURLBase(name, base_url, "base URL of the URL database"); - if (!hasURLScheme(resolved)) + if (findURIAuthority(resolved) == String::npos) return {}; return resolved; } diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 95bfbc06b57e..6e184d513122 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -396,11 +396,8 @@ void FunctionSecretArgumentsFinder::findMySQLFunctionSecretArguments() } } -void FunctionSecretArgumentsFinder::findTLSCredentialsSecretArguments(size_t start) +void FunctionSecretArgumentsFinder::markNamedArgumentsWithUnreadableKeys(size_t start) { - for (const auto & key : tls_credentials_secret_keys) - findSecretNamedArgument(key, start); - /// The named-collection parser does not require the key of a `key = value` argument to be a plain /// literal or identifier: `getKeyValueFromASTImpl` evaluates it as a constant expression, so /// `mysql(creds, concat('ssl_ca', '_pem') = 'SECRET', table = 't')` passes a TLS credential too. @@ -423,6 +420,14 @@ void FunctionSecretArgumentsFinder::findTLSCredentialsSecretArguments(size_t sta } } +void FunctionSecretArgumentsFinder::findTLSCredentialsSecretArguments(size_t start) +{ + for (const auto & key : tls_credentials_secret_keys) + findSecretNamedArgument(key, start); + + markNamedArgumentsWithUnreadableKeys(start); +} + void FunctionSecretArgumentsFinder::findMongoDBSecretArguments() { String uri; @@ -493,12 +498,17 @@ void FunctionSecretArgumentsFinder::findArrowFlightSecretArguments() void FunctionSecretArgumentsFinder::findXDBCSecretArguments() { + /// The connection string is never parsed by ClickHouse: `ITableFunctionXDBC` forwards it verbatim + /// to the bridge, so its grammar belongs to the JDBC/ODBC driver. It can carry the password in a + /// query parameter (`jdbc('mysql://host:3306/?user=root&password=root', ...)`, from this function's + /// own documentation) or as `Pwd=` in a `KEY=value;` list, neither of which a URI scan locates. + /// There is no extent that can be kept visible, so the value is hidden whole; + /// `format_display_secrets_in_show_and_select` still shows the real one to a privileged reader. if (isNamedCollectionName(0)) { /// jdbc(named_collection, ..., datasource = 'DSN', ...) /// odbc(named_collection, ..., connection_settings = 'DSN', ...) /// `datasource` and `connection_settings` are mutually exclusive aliases. - /// If the value is a URI, mask only the password; otherwise hide the whole value. /// If somehow both are present (invalid query), hide all named arguments. ssize_t ds_idx = findNamedArgument(nullptr, "datasource", 1); ssize_t cs_idx = findNamedArgument(nullptr, "connection_settings", 1); @@ -509,56 +519,22 @@ void FunctionSecretArgumentsFinder::findXDBCSecretArguments() result.start = 1; result.count = function->arguments->size() - 1; result.are_named = true; + return; } - else if (ds_idx >= 0) - maskXDBCSecretNamedArgument("datasource", 1); - else if (cs_idx >= 0) - maskXDBCSecretNamedArgument("connection_settings", 1); + + findSecretNamedArgument("datasource", 1); + findSecretNamedArgument("connection_settings", 1); + markNamedArgumentsWithUnreadableKeys(1); } else { /// jdbc('DSN', schema, table) / jdbc('DSN', table) /// odbc('DSN', schema, table) / odbc('DSN', table) /// JDBC('DSN', database, table) / ODBC('DSN', database, table) - /// The connection string may be a URI with credentials embedded, - /// e.g. scheme://username:password@host:port/dbname - /// If so, mask only the password part; otherwise hide the whole argument. - String uri; - if (tryGetStringFromArgument(0, &uri)) - { - if (maskURIPassword(&uri)) - { - chassert(result.count == 0); - result.start = 0; - result.count = 1; - result.replacement = std::move(uri); - return; - } - } markSecretArgument(0, false); } } -void FunctionSecretArgumentsFinder::maskXDBCSecretNamedArgument(std::string_view key, size_t start) -{ - String value; - ssize_t arg_idx = findNamedArgument(&value, key, start); - if (arg_idx < 0) - return; - - if (!value.empty() && maskURIPassword(&value)) - { - result.are_named = true; - result.start = arg_idx; - result.count = 1; - result.replacement = std::move(value); - } - else - { - markSecretArgument(arg_idx, /* argument_is_named= */ true); - } -} - void FunctionSecretArgumentsFinder::findS3FunctionSecretArguments(bool is_cluster_function) { /// s3Cluster('cluster_name', 'url', ...) has 'url' as its second argument. @@ -1014,7 +990,7 @@ void FunctionSecretArgumentsFinder::findNATSTableEngineSecretArguments() /// The only positional argument the engine accepts is the name of a named collection, so the /// credentials can only appear as named overrides. The `SETTINGS` clause form is masked /// separately by `NATS::SETTINGS_TO_HIDE`, and this function masks the same keys the same way: - /// the secrets are hidden whole, while `nats_url` keeps everything but its userinfo password. + /// the secrets are hidden whole, and so is a `nats_url` that carries an '@'. /// `nats_server_list` is hidden whole because each list entry can carry userinfo credentials. /// Fail closed on a key we cannot read as a plain literal: it can name a secret setting. for (size_t i = 0; i < function->arguments->size(); ++i) @@ -1042,8 +1018,10 @@ void FunctionSecretArgumentsFinder::findNATSTableEngineSecretArguments() String url; if (equals_func->arguments->at(1)->tryGetString(&url, /* allow_identifier= */ false)) { - if (maskURIPassword(&url)) - result.replaced_arguments[i] = "nats_url = " + quoteString(url); + /// An '@' is the only reliable sign of a credential here, and there is no extent to + /// keep visible; see the `nats_url` rule in `NATS_fwd.h` for why. + if (url.contains('@')) + markSecretArgument(i, /* argument_is_named= */ true); } else { diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index 821c215939aa..3c86e22c3b3b 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -120,6 +120,8 @@ class FunctionSecretArgumentsFinder /// Named arguments carrying NATS credentials. They are the setting names, because the `NATS` engine /// takes its arguments as overrides of a named collection (`NATS(collection, nats_token = '...')`). /// `nats_server_list` is a destination and can carry URI userinfo credentials, so hide it whole. + /// `nats_url` is not here because the documented form (`'localhost:4222'`) carries no credential and + /// stays visible; it is handled separately, by the presence of an '@'. /// Keep in sync with `NATS::SETTINGS_TO_HIDE`, which masks the same secrets in the `SETTINGS` clause. static constexpr std::string_view nats_secret_keys[] = {"nats_password", "nats_token", "nats_credential_file", "nats_credentials", "nats_server_list"}; @@ -171,11 +173,6 @@ class FunctionSecretArgumentsFinder void findRedisTableEngineSecretArguments(); void findArrowFlightSecretArguments(); void findXDBCSecretArguments(); - - /// Similar to `findSecretNamedArgument`, but if the value is a URI with credentials, - /// masks only the password part instead of hiding the entire value. - void maskXDBCSecretNamedArgument(std::string_view key, size_t start); - void findS3FunctionSecretArguments(bool is_cluster_function); void findAzureBlobStorageFunctionSecretArguments(bool is_cluster_function); bool maskAzureConnectionString(ssize_t url_arg_idx, bool argument_is_named = false, size_t start = 0); @@ -248,6 +245,11 @@ class FunctionSecretArgumentsFinder /// duplicate-key validation runs, so `session_token = 'a', session_token = 'b'` must hide both. bool findSecretNamedArgument(std::string_view key, size_t start = 0); + /// Hides the value of every `key = value` argument from `start` on whose key this finder cannot read + /// as a plain literal or identifier. The named-collection parser evaluates such a key as a constant + /// expression, so it can name a secret argument that `findNamedArgument` never sees. + void markNamedArgumentsWithUnreadableKeys(size_t start); + /// Masks the secrets of an S3 named-collection form: the secret named overrides (every occurrence, /// in any order; the span covering them may hide a non-secret argument in between, which is safe) /// and the `headers(...)` / `extra_credentials(...)` map overrides. diff --git a/src/Parsers/tests/gtest_Parser.cpp b/src/Parsers/tests/gtest_Parser.cpp index 588067e1751f..9e9a699dc4b3 100644 --- a/src/Parsers/tests/gtest_Parser.cpp +++ b/src/Parsers/tests/gtest_Parser.cpp @@ -296,8 +296,8 @@ TEST(ParserCreateQuery, MaskNATSTableEngineCredentials) TEST(ParserCreateQuery, MaskNATSTableEngineURLPassword) { - /// A `nats_url` override can carry the credentials in its userinfo. Only the password is hidden, - /// keeping the rest of the url visible, the same way the `SETTINGS` clause form is masked. + /// A `nats_url` override carrying an '@' is hidden whole, the same way the `SETTINGS` clause form + /// is masked: libnats reads a credential that no URI masker can bound. const String query = "CREATE TABLE test_nats (key UInt64) " "ENGINE = NATS(nats1, nats_url = 'nats://plain_user:plain_password@example.com:4222')"; @@ -308,7 +308,8 @@ TEST(ParserCreateQuery, MaskNATSTableEngineURLPassword) const String masked = ast->formatForLogging(); EXPECT_EQ(masked.find("plain_password"), String::npos); - EXPECT_NE(masked.find("nats://plain_user:[HIDDEN]@example.com:4222"), String::npos); + EXPECT_EQ(masked.find("plain_user"), String::npos); + EXPECT_NE(masked.find("nats_url = '[HIDDEN]'"), String::npos); } TEST(ParserCreateQuery, MaskNATSTableEngineServerListPassword) diff --git a/src/Storages/NATS/NATS_fwd.h b/src/Storages/NATS/NATS_fwd.h index f9b1043f7b7d..a74a9d3687c4 100644 --- a/src/Storages/NATS/NATS_fwd.h +++ b/src/Storages/NATS/NATS_fwd.h @@ -1,7 +1,6 @@ #pragma once #include #include -#include #include namespace NATS @@ -26,7 +25,12 @@ static inline std::unordered_map SETTINGS_TO_HIDE = std::string masked_value; if (!value.tryGet(masked_value)) return {}; - DB::maskURIPassword(&masked_value); + /// libnats trims the value, takes the scheme as optional, and ends the userinfo at the LAST + /// '@' of the whole value (`contrib/nats-io/src/url.c`, `natsUrl_Create`), so its credential + /// can run past anything an RFC 3986 authority covers. An '@' is therefore the only reliable + /// sign that this address carries one, and there is no extent to keep visible. + if (masked_value.contains('@')) + masked_value = "[HIDDEN]"; return fmt::format("'{}'", masked_value); }} }; diff --git a/src/Storages/RabbitMQ/RabbitMQ_fwd.h b/src/Storages/RabbitMQ/RabbitMQ_fwd.h index af4a3c6fd2bb..d0f5ccf96090 100644 --- a/src/Storages/RabbitMQ/RabbitMQ_fwd.h +++ b/src/Storages/RabbitMQ/RabbitMQ_fwd.h @@ -1,7 +1,6 @@ #pragma once #include #include -#include #include namespace RabbitMQ @@ -21,7 +20,12 @@ static inline std::unordered_map SETTINGS_TO_HIDE = std::string masked_value; if (!value.tryGet(masked_value)) return {}; - DB::maskURIPassword(&masked_value); + /// AMQP-CPP ends the login at the FIRST '@' after the scheme, unbounded by the `/?#` that + /// closes an RFC 3986 authority (`contrib/AMQP-CPP/include/amqpcpp/address.h`, `Address`), so + /// its credential can run past anything a URI masker covers. An '@' is therefore the only + /// reliable sign that this address carries one, and there is no extent to keep visible. + if (masked_value.contains('@')) + masked_value = "[HIDDEN]"; return fmt::format("'{}'", masked_value); }} }; diff --git a/tests/integration/test_mask_sensitive_info/test.py b/tests/integration/test_mask_sensitive_info/test.py index 4ea3878add59..24b95f076493 100644 --- a/tests/integration/test_mask_sensitive_info/test.py +++ b/tests/integration/test_mask_sensitive_info/test.py @@ -396,8 +396,8 @@ def generate_create_table_numbered(tail): generate_create_table_numbered(f"(`x` int) ENGINE = AzureQueue('{azure_storage_account_url}', 'cont', '*', '{azure_account_name}', '[HIDDEN]', 'CSV', 'none') SETTINGS mode = 'unordered'"), generate_create_table_numbered(f"(`x` int) ENGINE = AzureBlobStorage('{masked_sas_conn_string}', 'exampledatasets', 'example.csv')"), generate_create_table_numbered("(`x` int) ENGINE = S3('https://my-s3-endpoint/bucket/data.csv', 'myaccess', '[HIDDEN]', 'CSV')"), - generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://schema_user:[HIDDEN]@'"), - generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://schema_user:[HIDDEN]@domain.com'"), + generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://[HIDDEN]@'"), + generate_create_table_numbered("(`x` int) ENGINE = Kafka SETTINGS kafka_broker_list = '127.0.0.1', kafka_topic_list = 'topic', kafka_group_name = 'group', kafka_format = 'JSONEachRow', kafka_security_protocol = 'sasl_ssl', kafka_sasl_mechanism = 'PLAIN', kafka_sasl_username = 'user', kafka_sasl_password = '[HIDDEN]', format_avro_schema_registry_url = 'http://[HIDDEN]@domain.com'"), generate_create_table_numbered("(`x` int) ENGINE = S3('http://minio1:9001/root/data/test5.csv.gz', 'CSV', access_key_id = 'minio', secret_access_key = '[HIDDEN]', compression_method = 'gzip')"), generate_create_table_numbered("(`x` int) ENGINE = ArrowFlight('arrowflight1:5006', 'dataset', 'arrowflight_user', '[HIDDEN]')"), generate_create_table_numbered("(`x` int) ENGINE = ArrowFlight(named_collection_1, host = 'arrowflight1', port = 5006, dataset = 'dataset', username = 'arrowflight_user', password = '[HIDDEN]')"), @@ -405,12 +405,12 @@ def generate_create_table_numbered(tail): generate_create_table_numbered("(`x` int) ENGINE = Redis('localhost', 0, '[HIDDEN]') PRIMARY KEY x"), generate_create_table_numbered("(`x` int) ENGINE = JDBC('[HIDDEN]', 'mydb', 'mytable')"), generate_create_table_numbered("(`x` int) ENGINE = ODBC('[HIDDEN]', 'mydb', 'mytable')"), - generate_create_table_numbered("(`x` int) ENGINE = JDBC('jdbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')"), - generate_create_table_numbered("(`x` int) ENGINE = ODBC('odbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')"), + generate_create_table_numbered("(`x` int) ENGINE = JDBC('[HIDDEN]', 'mydb', 'mytable')"), + generate_create_table_numbered("(`x` int) ENGINE = ODBC('[HIDDEN]', 'mydb', 'mytable')"), + generate_create_table_numbered("(`x` int) ENGINE = JDBC(named_collection_1, datasource = '[HIDDEN]', external_database = 'mydb', external_table = 'mytable')"), + generate_create_table_numbered("(`x` int) ENGINE = ODBC(named_collection_1, connection_settings = '[HIDDEN]', external_database = 'mydb', external_table = 'mytable')"), generate_create_table_numbered("(`x` int) ENGINE = JDBC(named_collection_1, datasource = '[HIDDEN]', external_database = 'mydb', external_table = 'mytable')"), generate_create_table_numbered("(`x` int) ENGINE = ODBC(named_collection_1, connection_settings = '[HIDDEN]', external_database = 'mydb', external_table = 'mytable')"), - generate_create_table_numbered("(`x` int) ENGINE = JDBC(named_collection_1, datasource = 'jdbc://user:[HIDDEN]@localhost:5432/mydb', external_database = 'mydb', external_table = 'mytable')"), - generate_create_table_numbered("(`x` int) ENGINE = ODBC(named_collection_1, connection_settings = 'odbc://user:[HIDDEN]@localhost:5432/mydb', external_database = 'mydb', external_table = 'mytable')"), generate_create_table_numbered("(`x` int) ENGINE = JDBC(named_collection_1, datasource = '[HIDDEN]', connection_settings = '[HIDDEN]', external_database = '[HIDDEN]', external_table = '[HIDDEN]')"), generate_create_table_numbered("(`x` int) ENGINE = JDBC(named_collection_1, connection_settings = '[HIDDEN]', external_database = '[HIDDEN]', datasource = '[HIDDEN]', external_table = '[HIDDEN]')"), generate_create_table_numbered("(`x` int) ENGINE = NATS SETTINGS nats_url = 'localhost:4222', nats_subjects = 'subject', nats_format = 'JSONEachRow', nats_token = '[HIDDEN]'"), @@ -656,12 +656,12 @@ def make_test_case(i): "CREATE TABLE tablefunc49 (`x` int) AS redis('localhost', 'key', 'key Int64', 0, '[HIDDEN]')", "CREATE TABLE tablefunc50 (`x` int) AS jdbc('[HIDDEN]', 'mydb', 'mytable')", "CREATE TABLE tablefunc51 (`x` int) AS odbc('[HIDDEN]', 'mydb', 'mytable')", - "CREATE TABLE tablefunc52 (`x` int) AS jdbc('jdbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')", - "CREATE TABLE tablefunc53 (`x` int) AS odbc('odbc://user:[HIDDEN]@localhost:5432/mydb', 'mydb', 'mytable')", + "CREATE TABLE tablefunc52 (`x` int) AS jdbc('[HIDDEN]', 'mydb', 'mytable')", + "CREATE TABLE tablefunc53 (`x` int) AS odbc('[HIDDEN]', 'mydb', 'mytable')", "CREATE TABLE tablefunc54 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]')", "CREATE TABLE tablefunc55 (`x` int) AS odbc(named_collection_1, connection_settings = '[HIDDEN]')", - "CREATE TABLE tablefunc56 (`x` int) AS jdbc(named_collection_1, datasource = 'jdbc://user:[HIDDEN]@localhost:5432/mydb')", - "CREATE TABLE tablefunc57 (`x` int) AS odbc(named_collection_1, connection_settings = 'odbc://user:[HIDDEN]@localhost:5432/mydb')", + "CREATE TABLE tablefunc56 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]')", + "CREATE TABLE tablefunc57 (`x` int) AS odbc(named_collection_1, connection_settings = '[HIDDEN]')", "CREATE TABLE tablefunc58 (`x` int) AS jdbc(named_collection_1, datasource = '[HIDDEN]', connection_settings = '[HIDDEN]')", "CREATE TABLE tablefunc59 (`x` int) AS jdbc(named_collection_1, connection_settings = '[HIDDEN]', external_database = '[HIDDEN]', datasource = '[HIDDEN]')", "CREATE TABLE tablefunc60 (`x` int) AS deltaLakeS3('http://minio1:9001/root/data/test11.csv.gz', 'minio', '[HIDDEN]')", diff --git a/tests/integration/test_storage_rabbitmq/test.py b/tests/integration/test_storage_rabbitmq/test.py index 252fc892b9e4..9f02e5fbe90e 100644 --- a/tests/integration/test_storage_rabbitmq/test.py +++ b/tests/integration/test_storage_rabbitmq/test.py @@ -3602,7 +3602,7 @@ def test_hiding_credentials(rabbitmq_cluster, db, unique): instance.query("SYSTEM FLUSH LOGS") message = instance.query(f"SELECT message FROM system.text_log WHERE message ILIKE '%CREATE TABLE {db}.{table_name}%'") assert "rabbitmq_password = \\'[HIDDEN]\\'" in message - assert "rabbitmq_address = \\'amqp://root:[HIDDEN]@rabbitmq1:5672/\\'" in message + assert "rabbitmq_address = \\'[HIDDEN]\\'" in message def test_rabbitmq_default_mode_nack_on_parse_error(rabbitmq_cluster, db, unique): diff --git a/tests/queries/0_stateless/04665_valueless_setting_ast_json_and_secret_parts.reference b/tests/queries/0_stateless/04665_valueless_setting_ast_json_and_secret_parts.reference index 209f37b46ed4..f726717f90db 100644 --- a/tests/queries/0_stateless/04665_valueless_setting_ast_json_and_secret_parts.reference +++ b/tests/queries/0_stateless/04665_valueless_setting_ast_json_and_secret_parts.reference @@ -8,7 +8,7 @@ TYPE_MISMATCH 1 SELECT 1 SETTINGS format_avro_schema_registry_url = \'http://user:pass@localhost\' TYPE_MISMATCH -SELECT 1 SETTINGS format_avro_schema_registry_url = \'http://user:[HIDDEN]@localhost\' 1 +SELECT 1 SETTINGS format_avro_schema_registry_url = \'http://[HIDDEN]@localhost\' 1 TYPE_MISMATCH TYPE_MISMATCH SELECT 1 SETTINGS format_avro_schema_registry_url = \'http://user:pass@localhost\' diff --git a/tests/queries/0_stateless/05056_settings_secret_masking.reference b/tests/queries/0_stateless/05056_settings_secret_masking.reference index ed9415436a87..e1fa25e506f9 100644 --- a/tests/queries/0_stateless/05056_settings_secret_masking.reference +++ b/tests/queries/0_stateless/05056_settings_secret_masking.reference @@ -1,13 +1,13 @@ 1 1 1 1 1 1 1 -http://u:[HIDDEN]@reg:8080/ -https://u:[HIDDEN]@example.com/d/ +http://[HIDDEN]@reg:8080/ +https://[HIDDEN]@example.com/d/ 1 1 -CREATE SETTINGS PROFILE `profile` SETTINGS format_avro_schema_registry_url = \'http://u:[HIDDEN]@reg:8080/\' -http://u:[HIDDEN]@reg:8080/ +CREATE SETTINGS PROFILE `profile` SETTINGS format_avro_schema_registry_url = \'http://[HIDDEN]@reg:8080/\' +http://[HIDDEN]@reg:8080/ 1 1 1 1 1 1 https://h/f?X-Amz-Signature=c05056presignedsignature -http://u:[HIDDEN]@reg:8080/ http://u:[HIDDEN]@reg:8080/ +http://[HIDDEN]@reg:8080/ http://[HIDDEN]@reg:8080/ s3://bucket/prefix/ diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference new file mode 100644 index 000000000000..3f7a95b33b5a --- /dev/null +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference @@ -0,0 +1,28 @@ +a1_nats_slash_password SET nats_url = '[HIDDEN]' +a2_nats_no_scheme SET nats_url = '[HIDDEN]' +a3_nats_leading_space SET nats_url = '[HIDDEN]' +a4_nats_control SET nats_url = 'localhost:4222' +b1_amqp_slash_password SET rabbitmq_address = '[HIDDEN]' +b2_amqp_control SET rabbitmq_address = 'amqp://h:5672/v' +c1_nats_engine_argument CREATE TABLE t05233 (`x` UInt8) ENGINE = NATS(nc05233, nats_url = '[HIDDEN]') +c2_nats_engine_control CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233, nats_url = 'localhost:4222') +d1_jdbc_query_parameter CREATE TABLE db05233absent.t05233 (`x` UInt8) ENGINE = JDBC('[HIDDEN]', 'db', 't') +d2_jdbc_at_in_password CREATE TABLE db05233absent.t05233 (`x` UInt8) ENGINE = JDBC('[HIDDEN]', 'db', 't') +d3_odbc_key_value_form CREATE TABLE db05233absent.t05233 (`x` UInt8) ENGINE = ODBC('[HIDDEN]', 'db', 't') +d4_odbc_named_argument SELECT * FROM odbc(nc05233, connection_settings = '[HIDDEN]') +d5_jdbc_duplicate_key SELECT * FROM jdbc(nc05233, datasource = '[HIDDEN]', datasource = '[HIDDEN]') +d6_jdbc_computed_key SELECT * FROM jdbc(nc05233, concat('data', 'source') = '[HIDDEN]') +d7_jdbc_both_aliases SELECT * FROM jdbc(nc05233, datasource = '[HIDDEN]', connection_settings = '[HIDDEN]') +d8_mysql_computed_key SELECT * FROM mysql(nc05233, concat('ssl_ca', '_pem') = '[HIDDEN]', `table` = 't') +e10_url_base_abfss_space SELECT 1 SETTINGS url_base = '[HIDDEN]' +e11_avro_in_engine CREATE TABLE t05233 (`x` UInt8) ENGINE = NATS(nc05233) SETTINGS format_avro_schema_registry_url = 'http://[HIDDEN]@reg:8080/' +e1_avro_at_in_password SELECT 1 SETTINGS format_avro_schema_registry_url = 'http://[HIDDEN]@reg:8080/' +e2_avro_userinfo_only SELECT 1 SETTINGS format_avro_schema_registry_url = 'http://[HIDDEN]@reg:8080/' +e3_url_base_no_userinfo SELECT 1 SETTINGS url_base = 'http://h:8080/d/?email=a@b.com' +e4_url_base_no_scheme SELECT 1 SETTINGS url_base = '[HIDDEN]' +e5_url_base_control SELECT 1 SETTINGS url_base = 'https://h/d/' +e6_s3_base_presigned SELECT 1 SETTINGS s3_base = 'https://b/f.csv?X-Amz-Signature=[HIDDEN]' +e7_url_base_abfss SELECT 1 SETTINGS url_base = 'abfss://container@account.dfs.core.windows.net/d/' +e8_avro_abfss_control SELECT 1 SETTINGS format_avro_schema_registry_url = 'abfss://[HIDDEN]@h/' +e9_url_base_az_control SELECT 1 SETTINGS url_base = 'az://[HIDDEN]@account.blob.core.windows.net/c/' +0 diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh new file mode 100755 index 000000000000..6dd462de82f2 --- /dev/null +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh @@ -0,0 +1,85 @@ +#!/usr/bin/env bash + +# A credential embedded in a broker address, in an XDBC connection string or in a URL-valued setting +# must not reach `system.query_log`. Every statement below is rejected, and none of them needs a +# broker, a bridge or an existing named collection: masking runs when the statement is formatted for +# logging, before it is validated. Each positive case is paired with the control that has to stay +# fully visible, and each credential is a distinct `leak05233*` canary, so a leak points straight at +# the site that leaked it. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +PREFIX="05233_${CLICKHOUSE_DATABASE}_$(random_str 8)" +ARMS=0 + +arm() { # arm + ARMS=$((ARMS + 1)) + $CLICKHOUSE_CLIENT --query_id="${PREFIX}_$1" --log_queries=1 -q "$2" > /dev/null 2>&1 +} + +# `nats_url`: libnats ends the userinfo at the last '@' of the whole value, with the scheme optional +# and the value trimmed, so a password containing '/' and a scheme-less address both authenticate. +arm a1_nats_slash_password "SET nats_url = 'nats://u:pa/leak05233nats@h:4222'" +arm a2_nats_no_scheme "SET nats_url = 'u:leak05233noscheme@h:4222'" +arm a3_nats_leading_space "SET nats_url = ' nats://u:pa/leak05233space@h:4222'" +arm a4_nats_control "SET nats_url = 'localhost:4222'" + +# `rabbitmq_address`: AMQP-CPP ends the login at the first '@' after the scheme, also unbounded by the +# '/' that closes an RFC 3986 authority. +arm b1_amqp_slash_password "SET rabbitmq_address = 'amqp://u:pa/leak05233amqp@h:5672/v'" +arm b2_amqp_control "SET rabbitmq_address = 'amqp://h:5672/v'" + +# The engine-argument form has its own masking site and must agree with the `SETTINGS` clause above. +arm c1_nats_engine_argument "CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233, nats_url = 'nats://u:pa/leak05233arg@h:4222')" +arm c2_nats_engine_control "CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233, nats_url = 'localhost:4222')" + +# An XDBC connection string is forwarded to the driver verbatim, so its grammar is the driver's: the +# password can sit in a query parameter or in a `KEY=value;` list, and no URI scan bounds it. The +# positional form is written against a database this test never creates, because a `jdbc(...)` table +# function that reaches argument validation then spends 30 seconds trying to start the bridge. +arm d1_jdbc_query_parameter "CREATE TABLE db05233absent.t05233 (x UInt8) ENGINE = JDBC('mysql://u:p@h/?password=leak05233param', 'db', 't')" +arm d2_jdbc_at_in_password "CREATE TABLE db05233absent.t05233 (x UInt8) ENGINE = JDBC('jdbc://user:pa@leak05233at@h:5432/db', 'db', 't')" +arm d3_odbc_key_value_form "CREATE TABLE db05233absent.t05233 (x UInt8) ENGINE = ODBC('DSN=x;Uid=u;Pwd=leak05233kv', 'db', 't')" +arm d4_odbc_named_argument "SELECT * FROM odbc(nc05233, connection_settings = 'odbc://u:pa@leak05233odbc@h/db')" +arm d5_jdbc_duplicate_key "SELECT * FROM jdbc(nc05233, datasource = 'a://u:leak05233dup1@h/1', datasource = 'b://u:leak05233dup2@h/2')" +arm d6_jdbc_computed_key "SELECT * FROM jdbc(nc05233, concat('data', 'source') = 'jdbc://u:leak05233computed@h/db')" +arm d7_jdbc_both_aliases "SELECT * FROM jdbc(nc05233, datasource = 'a://u:leak05233both1@h/1', connection_settings = 'b://u:leak05233both2@h/2')" +# Control for the fail-closed unreadable-key scan the XDBC branch now shares with the TLS keys. +arm d8_mysql_computed_key "SELECT * FROM mysql(nc05233, concat('ssl_ca', '_pem') = 'leak05233mysql', table = 't')" + +# The query-level URL settings are read through `Poco::URI`, so host and path stay visible and only +# the userinfo is hidden. A value with no scheme in front of it is hidden whole instead. +arm e1_avro_at_in_password "SELECT 1 SETTINGS format_avro_schema_registry_url = 'http://user:pa@leak05233avro@reg:8080/'" +arm e2_avro_userinfo_only "SELECT 1 SETTINGS format_avro_schema_registry_url = 'http://leak05233token@reg:8080/'" +arm e3_url_base_no_userinfo "SELECT 1 SETTINGS url_base = 'http://h:8080/d/?email=a@b.com'" +arm e4_url_base_no_scheme "SELECT 1 SETTINGS url_base = ' http://u:leak05233space@h/d/'" +arm e5_url_base_control "SELECT 1 SETTINGS url_base = 'https://h/d/'" +arm e6_s3_base_presigned "SELECT 1 SETTINGS s3_base = 'https://b/f.csv?X-Amz-Signature=leak05233sig'" +# `abfs`/`abfss` put a non-secret container in front of the '@', and only for `url_base`, the one of +# the three settings that reaches the Azure URL parser. +arm e7_url_base_abfss "SELECT 1 SETTINGS url_base = 'abfss://container@account.dfs.core.windows.net/d/'" +arm e8_avro_abfss_control "SELECT 1 SETTINGS format_avro_schema_registry_url = 'abfss://u:leak05233avroabfss@h/'" +arm e9_url_base_az_control "SELECT 1 SETTINGS url_base = 'az://u:leak05233az@account.blob.core.windows.net/c/'" +arm e10_url_base_abfss_space "SELECT 1 SETTINGS url_base = ' abfss://c@a.dfs.core.windows.net/d/'" +# The same setting reached through an `ENGINE = ... SETTINGS` clause rather than through a query. +arm e11_avro_in_engine "CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233) SETTINGS format_avro_schema_registry_url = 'http://user:pa@leak05233engine@reg:8080/'" + +for _ in {1..120}; do + $CLICKHOUSE_CLIENT -q "SYSTEM FLUSH LOGS query_log" + LOGGED=$($CLICKHOUSE_CLIENT -q "SELECT uniqExact(query_id) FROM system.query_log + WHERE event_date >= yesterday() AND current_database = currentDatabase() AND query_id LIKE '${PREFIX}\_%'") + [ "$LOGGED" -ge "$ARMS" ] && break + sleep 0.5 +done + +$CLICKHOUSE_CLIENT -q "SELECT replaceOne(query_id, '${PREFIX}_', '') AS arm, max(query) + FROM system.query_log + WHERE event_date >= yesterday() AND current_database = currentDatabase() AND query_id LIKE '${PREFIX}\_%' + GROUP BY arm ORDER BY arm FORMAT TSVRaw" + +# No canary in any column that prints the statement or its settings, in any arm. +$CLICKHOUSE_CLIENT -q "SELECT countIf(position(concat(query, formatted_query, toString(Settings)), 'leak05233') > 0) + FROM system.query_log + WHERE event_date >= yesterday() AND current_database = currentDatabase() AND query_id LIKE '${PREFIX}\_%'" From 66942793b54423bd444c3bd4d4d2f108c517d387 Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 09:19:18 +0000 Subject: [PATCH 090/185] Mask broker engine-argument overrides and drop the abfs container exception Three defects found reviewing the previous commit. The abfs/abfss exception in CoreSettings::SETTINGS_TO_HIDE masked strictly LESS than the code it replaced. maskURLBaseCredentials returned maskPresignedURLParameters(value) alone for those two schemes, so url_base = 'abfss://c:pw@a.dfs.core.windows.net/d/' logged pw verbatim where maskURIPassword had logged abfss://c:[HIDDEN]@a... . That is a masking regression in one of the settings this change set names, and it also contradicted the rule the rest of the work applies: the exception's own comment said the credential of such a base is the SAS in the query string, and then delegated to a helper that carries only the AWS and GCS parameter names and so cannot mask a SAS. A value whose credential ClickHouse cannot bound is not echoed with a partial mask. The exception is therefore deleted rather than widened, url_base goes back to maskURLCredentials, all three URL settings share one rule for every scheme, and Core no longer mirrors a copy of the Azure grammar that had to be kept in sync with parseAzureURL. This supersedes the paragraph of the previous commit message that described the container as staying visible. findTableEngineSecretArguments dispatched on the engine name and had branches for NATS and XDBC but none for RabbitMQ or Kafka, although exactly three engines read named-collection overrides out of their engine arguments (tryGetNamedCollectionWithOverrides(args.engine_args, ...) in StorageKafkaUtils.cpp, StorageNATS.cpp and StorageRabbitMQ.cpp) and only NATS had a finder. So ENGINE = RabbitMQ(nc, rabbitmq_address = 'amqp://u:secret@h/') ENGINE = RabbitMQ(nc, rabbitmq_password = 'secret') ENGINE = Kafka(nc, kafka_sasl_password = 'secret') reached SHOW CREATE, system.tables.create_table_query and system.query_log.query in cleartext, while the same settings written in a SETTINGS clause were hidden by the engine's own SETTINGS_TO_HIDE. Rather than copy the NATS loop, its body becomes findBrokerTableEngineSecretArguments(secret_keys, address_key), so the requirement that the finder and SETTINGS_TO_HIDE agree has one carrier; NATS and RabbitMQ are two thin wrappers over it. Kafka deliberately does not use it, because Kafka still accepts the legacy positional form Kafka('brokers', 'topics', 'group', 'format', ...) whose arguments are all non-secret, so failing closed on a positional argument there would hide information a user is entitled to see. NATS and RabbitMQ accept no positional argument other than the collection name, so for those two a positional is hidden. In the XDBC named-collection branch every helper walks equals arguments only, so a positional argument after the collection name was never masked and jdbc(nc, 'jdbc://u:pw@h/db') echoed the connection string. Such a statement is invalid and XDBC rejects it, but rejection happens after the statement has been formatted for logging. The branch now hides any argument after the collection name that is not a readable key = value pair, which is the treatment the NATS sibling already gave that shape. Tests: seven arms added to 05233_broker_and_xdbc_credential_masking, each positive paired with the control that has to stay fully visible, in particular an '@'-free rabbitmq_address and Kafka's legacy positional form. Three gtests added next to the NATS ones, each asserting the engine-argument spelling, the SETTINGS spelling and that control. Every existing NATS assertion, the re2 equivalence test and the TLS expectations pass unedited. --- src/Core/SettingsSecrets.h | 26 +--- src/Parsers/FunctionSecretArgumentsFinder.cpp | 56 +++++++-- src/Parsers/FunctionSecretArgumentsFinder.h | 17 +++ src/Parsers/tests/gtest_Parser.cpp | 119 ++++++++++++++++++ ...oker_and_xdbc_credential_masking.reference | 9 +- ...5233_broker_and_xdbc_credential_masking.sh | 16 ++- 6 files changed, 207 insertions(+), 36 deletions(-) diff --git a/src/Core/SettingsSecrets.h b/src/Core/SettingsSecrets.h index 77d883ee75ec..11c49b1f8290 100644 --- a/src/Core/SettingsSecrets.h +++ b/src/Core/SettingsSecrets.h @@ -42,30 +42,6 @@ inline bool maskURLCredentials(String & value) return masked; } -/// Under `abfs`/`abfss` the part in front of the `@` is the container and not a credential (the -/// Hadoop grammar `abfss://@.dfs.core.windows.net/`), and the credential -/// is the SAS in the query string. `az`/`azure` have no `@` in their grammar and need no exception. -/// Keep the two names in sync with `parseAzureURL` in `src/Storages/StorageURL.cpp`, which compares -/// them after lowercasing; `Core` cannot depend on `Storages` to share the list. -inline bool maskURLBaseCredentials(String & value) -{ - static constexpr std::string_view azure_container_schemes[] = {"abfs", "abfss"}; - - if (size_t authority = findURIAuthority(value); authority != String::npos) - { - String scheme = value.substr(0, authority - 3); - for (auto & c : scheme) - if ('A' <= c && c <= 'Z') - c += 'a' - 'A'; - - for (auto azure_scheme : azure_container_schemes) - if (scheme == azure_scheme) - return maskPresignedURLParameters(value); - } - - return maskURLCredentials(value); -} - /// The settings of the query-level `Settings` collection whose value can carry a credential, and how /// each one is masked. `system.query_log.query` shows /// `format_avro_schema_registry_url = 'http://[HIDDEN]@registry:8080'`, so every other place that @@ -76,7 +52,7 @@ inline bool maskURLBaseCredentials(String & value) static inline std::unordered_map SETTINGS_TO_HIDE = { {"format_avro_schema_registry_url", maskURLCredentials}, - {"url_base", maskURLBaseCredentials}, + {"url_base", maskURLCredentials}, {"s3_base", maskURLCredentials}, }; diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 6e184d513122..b21297dbcef5 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -525,6 +525,17 @@ void FunctionSecretArgumentsFinder::findXDBCSecretArguments() findSecretNamedArgument("datasource", 1); findSecretNamedArgument("connection_settings", 1); markNamedArgumentsWithUnreadableKeys(1); + + /// After the collection name every argument must be a named override. A positional one is + /// invalid, and the statement is formatted for logging before validation rejects it, so it + /// can carry the connection string itself: hide it whole (fail closed). + for (size_t i = 1; i < function->arguments->size(); ++i) + { + const auto equals_func = function->arguments->at(i)->getFunction(); + if (!equals_func || equals_func->name() != "equals" || !equals_func->hasArguments() + || equals_func->arguments->size() != 2) + markSecretArgument(i, /* argument_is_named= */ false); + } } else { @@ -972,6 +983,21 @@ void FunctionSecretArgumentsFinder::findTableEngineSecretArguments() /// NATS(named_collection, nats_password = 'password', nats_credentials = '...', ...) findNATSTableEngineSecretArguments(); } + else if (engine_name == "RabbitMQ") + { + /// RabbitMQ(named_collection, rabbitmq_address = '...', rabbitmq_password = '...') + findRabbitMQTableEngineSecretArguments(); + } + else if (engine_name == "Kafka") + { + /// Kafka(named_collection, kafka_sasl_password = '...') - `registerStorageKafka` reads named + /// overrides of the collection, so an override carries the secret the `SETTINGS` clause form + /// hides through `Kafka::SETTINGS_TO_HIDE`. The legacy positional form + /// (`Kafka('brokers', 'topics', 'group', 'format', ...)`) carries no secret and stays visible, + /// so this form does not fail closed on a positional argument. + findSecretNamedArgument("kafka_sasl_password", 1); + markNamedArgumentsWithUnreadableKeys(1); + } else if ((engine_name == "JDBC") || (engine_name == "ODBC")) { /// JDBC('DSN', database, table) @@ -981,17 +1007,21 @@ void FunctionSecretArgumentsFinder::findTableEngineSecretArguments() } } -void FunctionSecretArgumentsFinder::findNATSTableEngineSecretArguments() +void FunctionSecretArgumentsFinder::findBrokerTableEngineSecretArguments( + std::span secret_keys, std::string_view address_key) { /// NATS(named_collection [, nats_password = 'password'] [, nats_token = 'token'] /// [, nats_credential_file = '/path'] [, nats_credentials = 'user JWT and seed'] /// [, nats_url = 'nats://user:password@host:4222'] /// [, nats_server_list = 'nats://user:password@host:4222,...'], ...) - /// The only positional argument the engine accepts is the name of a named collection, so the + /// RabbitMQ(named_collection [, rabbitmq_password = 'password'] + /// [, rabbitmq_address = 'amqp://user:password@host:5672/vhost'], ...) + /// The only positional argument these engines accept is the name of a named collection, so the /// credentials can only appear as named overrides. The `SETTINGS` clause form is masked - /// separately by `NATS::SETTINGS_TO_HIDE`, and this function masks the same keys the same way: - /// the secrets are hidden whole, and so is a `nats_url` that carries an '@'. - /// `nats_server_list` is hidden whole because each list entry can carry userinfo credentials. + /// separately by the engine's own `SETTINGS_TO_HIDE`, and this function masks the same keys the + /// same way: the secrets are hidden whole, and so is an address that carries an '@'. + /// A key list can hold a destination (`nats_server_list`), which is hidden whole because each + /// list entry can carry userinfo credentials. /// Fail closed on a key we cannot read as a plain literal: it can name a secret setting. for (size_t i = 0; i < function->arguments->size(); ++i) { @@ -1013,13 +1043,13 @@ void FunctionSecretArgumentsFinder::findNATSTableEngineSecretArguments() { markSecretArgument(i, /* argument_is_named= */ true); } - else if (key == "nats_url") + else if (key == address_key) { String url; if (equals_func->arguments->at(1)->tryGetString(&url, /* allow_identifier= */ false)) { /// An '@' is the only reliable sign of a credential here, and there is no extent to - /// keep visible; see the `nats_url` rule in `NATS_fwd.h` for why. + /// keep visible; see the address rule in the engine's `_fwd.h` for why. if (url.contains('@')) markSecretArgument(i, /* argument_is_named= */ true); } @@ -1030,13 +1060,23 @@ void FunctionSecretArgumentsFinder::findNATSTableEngineSecretArguments() markSecretArgument(i, /* argument_is_named= */ true); } } - else if (std::find(std::begin(nats_secret_keys), std::end(nats_secret_keys), key) != std::end(nats_secret_keys)) + else if (std::find(secret_keys.begin(), secret_keys.end(), key) != secret_keys.end()) { markSecretArgument(i, /* argument_is_named= */ true); } } } +void FunctionSecretArgumentsFinder::findNATSTableEngineSecretArguments() +{ + findBrokerTableEngineSecretArguments(nats_secret_keys, "nats_url"); +} + +void FunctionSecretArgumentsFinder::findRabbitMQTableEngineSecretArguments() +{ + findBrokerTableEngineSecretArguments(rabbitmq_secret_keys, "rabbitmq_address"); +} + void FunctionSecretArgumentsFinder::findExternalDistributedTableEngineSecretArguments() { if (isNamedCollectionName(1)) diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index 3c86e22c3b3b..9aee325aafd3 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -3,6 +3,7 @@ #include #include #include +#include #include #include #include @@ -126,6 +127,12 @@ class FunctionSecretArgumentsFinder static constexpr std::string_view nats_secret_keys[] = {"nats_password", "nats_token", "nats_credential_file", "nats_credentials", "nats_server_list"}; + /// Named arguments carrying RabbitMQ credentials, as the setting names, for the same reason as + /// `nats_secret_keys`. `rabbitmq_address` is not here because the documented form carries no + /// credential and stays visible; it is handled by the presence of an '@'. + /// Keep in sync with `RabbitMQ::SETTINGS_TO_HIDE`. + static constexpr std::string_view rabbitmq_secret_keys[] = {"rabbitmq_password"}; + void markSecretArgument(size_t index, bool argument_is_named = false); /// `headers(..)` and `extra_credentials(..)` are nested maps whose values are secret auth material @@ -221,7 +228,17 @@ class FunctionSecretArgumentsFinder void findRedisFunctionSecretArguments(); void findYTsaurusStorageTableEngineSecretArguments(); void findBigQuerySecretArguments(); + /// Masks the named-collection override form of a broker engine + /// (`NATS(collection, nats_token = '...')`, `RabbitMQ(collection, rabbitmq_address = '...')`). + /// `secret_keys` are hidden whole; `address_key` is hidden only when its value carries an '@', + /// because the documented address form carries no credential. Fails closed on a positional + /// argument and on a key this finder cannot read as a plain literal. + /// Keep the key lists in sync with the engine's own `SETTINGS_TO_HIDE`, which masks the same + /// settings in the `SETTINGS` clause. + void findBrokerTableEngineSecretArguments( + std::span secret_keys, std::string_view address_key); void findNATSTableEngineSecretArguments(); + void findRabbitMQTableEngineSecretArguments(); void findDatabaseEngineSecretArguments(); void findMySQLDatabaseSecretArguments(); void findS3DatabaseSecretArguments(); diff --git a/src/Parsers/tests/gtest_Parser.cpp b/src/Parsers/tests/gtest_Parser.cpp index 9e9a699dc4b3..73253f5d6c46 100644 --- a/src/Parsers/tests/gtest_Parser.cpp +++ b/src/Parsers/tests/gtest_Parser.cpp @@ -380,6 +380,125 @@ TEST(ParserCreateQuery, MaskNATSTableEnginePositionalArguments) EXPECT_NE(masked.find("[HIDDEN]"), String::npos); } +TEST(ParserCreateQuery, MaskXDBCTableEnginePositionalAfterCollection) +{ + /// After a collection name every XDBC argument must be a named override, but the statement is + /// formatted for logging before validation rejects a positional one, and that positional can be + /// the connection string itself. The engine spelling takes more positional arguments than the + /// table function does, so it is asserted separately. + const String query = + "CREATE TABLE test_jdbc (key UInt64) " + "ENGINE = JDBC(jdbc1, 'DSN=mydb;Uid=user;Pwd=plain_password', 'mydb', 'mytable')"; + + DB::ParserCreateQuery parser; + DB::ASTPtr ast = DB::parseQuery(parser, query, 0, 0, 0); + + const String masked = ast->formatForLogging(); + + EXPECT_EQ(masked.find("plain_password"), String::npos); + EXPECT_EQ(masked.find("Uid=user"), String::npos); + /// The collection name is the one legitimate positional argument and stays visible. + EXPECT_NE(masked.find("jdbc1"), String::npos); + EXPECT_NE(masked.find("[HIDDEN]"), String::npos); + + /// A named override of the same connection string keeps its key visible, as before. + const String named_query = + "CREATE TABLE test_jdbc (key UInt64) " + "ENGINE = JDBC(jdbc1, datasource = 'DSN=mydb;Uid=user;Pwd=plain_named_password', " + "external_database = 'mydb', external_table = 'mytable')"; + + DB::ASTPtr named_ast = DB::parseQuery(parser, named_query, 0, 0, 0); + const String named_masked = named_ast->formatForLogging(); + + EXPECT_EQ(named_masked.find("plain_named_password"), String::npos); + EXPECT_NE(named_masked.find("datasource = '[HIDDEN]'"), String::npos); + /// The non-secret named arguments stay visible: the positional scan must not widen to them. + EXPECT_NE(named_masked.find("external_table = 'mytable'"), String::npos); +} + +TEST(ParserCreateQuery, MaskRabbitMQTableEngineCredentials) +{ + /// `RabbitMQ` also takes its settings as overrides of a named collection, so the same credentials + /// reach `SHOW CREATE TABLE` through the engine arguments and through the `SETTINGS` clause. + const String query = + "CREATE TABLE test_rabbitmq (key UInt64) ENGINE = RabbitMQ(rabbitmq1, " + "rabbitmq_password = 'plain_password', " + "rabbitmq_address = 'amqp://plain_user:plain_address_password@example.com:5672/vhost')"; + + DB::ParserCreateQuery parser; + DB::ASTPtr ast = DB::parseQuery(parser, query, 0, 0, 0); + + const String masked = ast->formatForLogging(); + + EXPECT_EQ(masked.find("plain_password"), String::npos); + EXPECT_EQ(masked.find("plain_address_password"), String::npos); + EXPECT_EQ(masked.find("plain_user"), String::npos); + /// The keys of the named overrides are not secrets and stay visible, as does the collection name. + EXPECT_NE(masked.find("rabbitmq1"), String::npos); + EXPECT_NE(masked.find("rabbitmq_password = '[HIDDEN]'"), String::npos); + EXPECT_NE(masked.find("rabbitmq_address = '[HIDDEN]'"), String::npos); + + /// An address with no '@' carries no credential and stays fully visible. + const String control_query = + "CREATE TABLE test_rabbitmq (key UInt64) " + "ENGINE = RabbitMQ(rabbitmq1, rabbitmq_address = 'amqp://example.com:5672/vhost')"; + + DB::ASTPtr control_ast = DB::parseQuery(parser, control_query, 0, 0, 0); + EXPECT_NE(control_ast->formatForLogging().find("amqp://example.com:5672/vhost"), String::npos); + + /// The `SETTINGS` clause form is masked by `RabbitMQ::SETTINGS_TO_HIDE` and must agree. + const String settings_query = + "CREATE TABLE test_rabbitmq_settings (key UInt64) ENGINE = RabbitMQ " + "SETTINGS rabbitmq_password = 'plain_settings_password'"; + + DB::ASTPtr settings_ast = DB::parseQuery(parser, settings_query, 0, 0, 0); + const String settings_masked = settings_ast->formatForLogging(); + + EXPECT_EQ(settings_masked.find("plain_settings_password"), String::npos); + EXPECT_NE(settings_masked.find("rabbitmq_password = '[HIDDEN]'"), String::npos); +} + +TEST(ParserCreateQuery, MaskKafkaTableEngineCredentials) +{ + /// `Kafka` reads named overrides of a collection too, so `kafka_sasl_password` needs masking in the + /// engine arguments and not only in the `SETTINGS` clause. + const String query = + "CREATE TABLE test_kafka (key UInt64) ENGINE = Kafka(kafka1, kafka_sasl_password = 'plain_password')"; + + DB::ParserCreateQuery parser; + DB::ASTPtr ast = DB::parseQuery(parser, query, 0, 0, 0); + + const String masked = ast->formatForLogging(); + + EXPECT_EQ(masked.find("plain_password"), String::npos); + EXPECT_NE(masked.find("kafka1"), String::npos); + EXPECT_NE(masked.find("kafka_sasl_password = '[HIDDEN]'"), String::npos); + + /// Unlike `NATS` and `RabbitMQ`, `Kafka` accepts a legacy positional form whose arguments are all + /// non-secret, so a positional argument must stay visible rather than fail closed. + const String positional_query = + "CREATE TABLE test_kafka (key UInt64) " + "ENGINE = Kafka('broker:9092', 'topic', 'group', 'JSONEachRow')"; + + DB::ASTPtr positional_ast = DB::parseQuery(parser, positional_query, 0, 0, 0); + const String positional_masked = positional_ast->formatForLogging(); + + EXPECT_NE(positional_masked.find("broker:9092"), String::npos); + EXPECT_NE(positional_masked.find("group"), String::npos); + EXPECT_EQ(positional_masked.find("[HIDDEN]"), String::npos); + + /// The `SETTINGS` clause form is masked by `Kafka::SETTINGS_TO_HIDE` and must agree. + const String settings_query = + "CREATE TABLE test_kafka_settings (key UInt64) ENGINE = Kafka " + "SETTINGS kafka_sasl_password = 'plain_settings_password'"; + + DB::ASTPtr settings_ast = DB::parseQuery(parser, settings_query, 0, 0, 0); + const String settings_masked = settings_ast->formatForLogging(); + + EXPECT_EQ(settings_masked.find("plain_settings_password"), String::npos); + EXPECT_NE(settings_masked.find("kafka_sasl_password = '[HIDDEN]'"), String::npos); +} + TEST_P(ParserTest, parseQuery) { const auto & parser = std::get<0>(GetParam()); diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference index 3f7a95b33b5a..e4fbeaba3b3e 100644 --- a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference @@ -14,15 +14,22 @@ d5_jdbc_duplicate_key SELECT * FROM jdbc(nc05233, datasource = '[HIDDEN]', datas d6_jdbc_computed_key SELECT * FROM jdbc(nc05233, concat('data', 'source') = '[HIDDEN]') d7_jdbc_both_aliases SELECT * FROM jdbc(nc05233, datasource = '[HIDDEN]', connection_settings = '[HIDDEN]') d8_mysql_computed_key SELECT * FROM mysql(nc05233, concat('ssl_ca', '_pem') = '[HIDDEN]', `table` = 't') +d9_jdbc_positional_after_collection SELECT * FROM jdbc(nc05233, '[HIDDEN]') e10_url_base_abfss_space SELECT 1 SETTINGS url_base = '[HIDDEN]' e11_avro_in_engine CREATE TABLE t05233 (`x` UInt8) ENGINE = NATS(nc05233) SETTINGS format_avro_schema_registry_url = 'http://[HIDDEN]@reg:8080/' +e12_url_base_abfss_password SELECT 1 SETTINGS url_base = 'abfss://[HIDDEN]@a.dfs.core.windows.net/d/' e1_avro_at_in_password SELECT 1 SETTINGS format_avro_schema_registry_url = 'http://[HIDDEN]@reg:8080/' e2_avro_userinfo_only SELECT 1 SETTINGS format_avro_schema_registry_url = 'http://[HIDDEN]@reg:8080/' e3_url_base_no_userinfo SELECT 1 SETTINGS url_base = 'http://h:8080/d/?email=a@b.com' e4_url_base_no_scheme SELECT 1 SETTINGS url_base = '[HIDDEN]' e5_url_base_control SELECT 1 SETTINGS url_base = 'https://h/d/' e6_s3_base_presigned SELECT 1 SETTINGS s3_base = 'https://b/f.csv?X-Amz-Signature=[HIDDEN]' -e7_url_base_abfss SELECT 1 SETTINGS url_base = 'abfss://container@account.dfs.core.windows.net/d/' +e7_url_base_abfss SELECT 1 SETTINGS url_base = 'abfss://[HIDDEN]@account.dfs.core.windows.net/d/' e8_avro_abfss_control SELECT 1 SETTINGS format_avro_schema_registry_url = 'abfss://[HIDDEN]@h/' e9_url_base_az_control SELECT 1 SETTINGS url_base = 'az://[HIDDEN]@account.blob.core.windows.net/c/' +f1_rabbitmq_engine_argument CREATE TABLE t05233 (`x` UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = '[HIDDEN]') +f2_rabbitmq_engine_password CREATE TABLE t05233 (`x` UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_password = '[HIDDEN]') +f3_rabbitmq_engine_control CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://h:5672/v') +f4_kafka_engine_password CREATE TABLE t05233 (`x` UInt8) ENGINE = Kafka(nc05233, kafka_sasl_password = '[HIDDEN]') +f5_kafka_positional_control CREATE TABLE t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow') 0 diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh index 6dd462de82f2..1a2e17f0e1ab 100755 --- a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh @@ -35,6 +35,15 @@ arm b2_amqp_control "SET rabbitmq_address = 'amqp://h:5672/v'" arm c1_nats_engine_argument "CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233, nats_url = 'nats://u:pa/leak05233arg@h:4222')" arm c2_nats_engine_control "CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233, nats_url = 'localhost:4222')" +# `Kafka`, `NATS` and `RabbitMQ` all read their settings as named overrides of the collection, so an +# override of a secret setting needs the same masking as the `SETTINGS` clause. Kafka's legacy +# positional form carries no secret, so its positional arguments stay visible. +arm f1_rabbitmq_engine_argument "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://u:leak05233rmqarg@h:5672/v')" +arm f2_rabbitmq_engine_password "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_password = 'leak05233rmqpw')" +arm f3_rabbitmq_engine_control "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://h:5672/v')" +arm f4_kafka_engine_password "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka(nc05233, kafka_sasl_password = 'leak05233kafkapw')" +arm f5_kafka_positional_control "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow')" + # An XDBC connection string is forwarded to the driver verbatim, so its grammar is the driver's: the # password can sit in a query parameter or in a `KEY=value;` list, and no URI scan bounds it. The # positional form is written against a database this test never creates, because a `jdbc(...)` table @@ -48,6 +57,8 @@ arm d6_jdbc_computed_key "SELECT * FROM jdbc(nc05233, concat('data', 'source arm d7_jdbc_both_aliases "SELECT * FROM jdbc(nc05233, datasource = 'a://u:leak05233both1@h/1', connection_settings = 'b://u:leak05233both2@h/2')" # Control for the fail-closed unreadable-key scan the XDBC branch now shares with the TLS keys. arm d8_mysql_computed_key "SELECT * FROM mysql(nc05233, concat('ssl_ca', '_pem') = 'leak05233mysql', table = 't')" +# After the collection name every argument must be named, so a positional one is hidden whole. +arm d9_jdbc_positional_after_collection "SELECT * FROM jdbc(nc05233, 'jdbc://u:leak05233pos@h/db')" # The query-level URL settings are read through `Poco::URI`, so host and path stay visible and only # the userinfo is hidden. A value with no scheme in front of it is hidden whole instead. @@ -57,12 +68,13 @@ arm e3_url_base_no_userinfo "SELECT 1 SETTINGS url_base = 'http://h:8080/d/?ema arm e4_url_base_no_scheme "SELECT 1 SETTINGS url_base = ' http://u:leak05233space@h/d/'" arm e5_url_base_control "SELECT 1 SETTINGS url_base = 'https://h/d/'" arm e6_s3_base_presigned "SELECT 1 SETTINGS s3_base = 'https://b/f.csv?X-Amz-Signature=leak05233sig'" -# `abfs`/`abfss` put a non-secret container in front of the '@', and only for `url_base`, the one of -# the three settings that reaches the Azure URL parser. +# All three URL settings follow the one rule, for every scheme: the userinfo is hidden and the rest +# stays visible, and a value whose scheme is not at the start is hidden whole. arm e7_url_base_abfss "SELECT 1 SETTINGS url_base = 'abfss://container@account.dfs.core.windows.net/d/'" arm e8_avro_abfss_control "SELECT 1 SETTINGS format_avro_schema_registry_url = 'abfss://u:leak05233avroabfss@h/'" arm e9_url_base_az_control "SELECT 1 SETTINGS url_base = 'az://u:leak05233az@account.blob.core.windows.net/c/'" arm e10_url_base_abfss_space "SELECT 1 SETTINGS url_base = ' abfss://c@a.dfs.core.windows.net/d/'" +arm e12_url_base_abfss_password "SELECT 1 SETTINGS url_base = 'abfss://c:leak05233abfs@a.dfs.core.windows.net/d/'" # The same setting reached through an `ENGINE = ... SETTINGS` clause rather than through a query. arm e11_avro_in_engine "CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233) SETTINGS format_avro_schema_registry_url = 'http://user:pa@leak05233engine@reg:8080/'" From 5c93ecf8f19075d14ae18cb005c33539929bdd0e Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 11:43:00 +0000 Subject: [PATCH 091/185] Start the Kafka and XDBC named-argument scans at index 0 Two defects found reviewing the previous commit, both the same shape one argument index further out: a scan that starts at index 1 is only correct when index 0 is proven to be a named-collection name. The Kafka branch the previous commit added scanned from index 1, but Kafka is the one engine here whose collection name is optional: registerStorageKafka falls through to the legacy positional path when there is no identifier at index 0, and rejects a malformed call in checkAndGetLiteralArgument only after the statement has been formatted for logging. So ENGINE = Kafka(kafka_sasl_password = 'secret', 'clickhouse') was masked nowhere and reached system.query_log.query and the server log in cleartext. The shape is one users write; 03305_fix_kafka_table_with_kw_arguments exists for exactly ENGINE = Kafka(a = '1', 'clickhouse'), and kafka_sasl_password is the engine's only secret setting. Both scans now start at 0. That cannot widen the mask, because findNamedArgument and markNamedArgumentsWithUnreadableKeys skip any argument that is not an equals function with two arguments, which is what keeps the legacy positional form fully visible. findXDBCSecretArguments had the mirror image in its other branch. The else branch is taken whenever index 0 is not an identifier, and it hid only index 0, so in a keyword-argument call with no collection the connection string sits at another index and was echoed: jdbc(external_table = 't', datasource = 'DSN=x;Uid=u;Pwd=secret') jdbc and odbc have no validity precondition and ITableFunctionXDBC validates after formatting, so the call is logged. Whether the credential leaked depended only on the order the keys were typed in, while the named-argument style without a collection is what a user coming from s3(url = ..., access_key_id = ...) writes. The two aliases and the unreadable-key scan now run from index 1 in that branch too, on top of the unchanged markSecretArgument(0, false). For a valid positional call all three additions are no-ops, since no equals argument exists, so the schema and table arguments stay visible. The class is now closed rather than the two examples: every other named-argument scan in the file that starts at a non-zero index sits inside an isNamedCollectionName test, so index 0 is proven to be an identifier there, and these two branches were the only ones without that property. Two arms are added to 05233_broker_and_xdbc_credential_masking, each keeping its non-secret neighbour visible so it doubles as the over-masking control, plus one gtest for the XDBC engine spelling and one assertion block for the Kafka one. --- src/Parsers/FunctionSecretArgumentsFinder.cpp | 14 ++++++-- src/Parsers/tests/gtest_Parser.cpp | 33 +++++++++++++++++++ ...oker_and_xdbc_credential_masking.reference | 2 ++ ...5233_broker_and_xdbc_credential_masking.sh | 6 ++++ 4 files changed, 52 insertions(+), 3 deletions(-) diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index b21297dbcef5..4038ac7ed5d0 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -543,6 +543,13 @@ void FunctionSecretArgumentsFinder::findXDBCSecretArguments() /// odbc('DSN', schema, table) / odbc('DSN', table) /// JDBC('DSN', database, table) / ODBC('DSN', database, table) markSecretArgument(0, false); + + /// A named argument means this is not the positional form at all, and the connection string + /// can then sit at any index under either alias. Validation rejects such a call only after the + /// statement has been formatted for logging, so hide those values too (fail closed). + findSecretNamedArgument("datasource", 1); + findSecretNamedArgument("connection_settings", 1); + markNamedArgumentsWithUnreadableKeys(1); } } @@ -994,9 +1001,10 @@ void FunctionSecretArgumentsFinder::findTableEngineSecretArguments() /// overrides of the collection, so an override carries the secret the `SETTINGS` clause form /// hides through `Kafka::SETTINGS_TO_HIDE`. The legacy positional form /// (`Kafka('brokers', 'topics', 'group', 'format', ...)`) carries no secret and stays visible, - /// so this form does not fail closed on a positional argument. - findSecretNamedArgument("kafka_sasl_password", 1); - markNamedArgumentsWithUnreadableKeys(1); + /// so this form does not fail closed on a positional argument. That form also makes the + /// collection name optional, so the scan starts at index 0: a named argument can be the first. + findSecretNamedArgument("kafka_sasl_password", 0); + markNamedArgumentsWithUnreadableKeys(0); } else if ((engine_name == "JDBC") || (engine_name == "ODBC")) { diff --git a/src/Parsers/tests/gtest_Parser.cpp b/src/Parsers/tests/gtest_Parser.cpp index 73253f5d6c46..634465edcde6 100644 --- a/src/Parsers/tests/gtest_Parser.cpp +++ b/src/Parsers/tests/gtest_Parser.cpp @@ -416,6 +416,25 @@ TEST(ParserCreateQuery, MaskXDBCTableEnginePositionalAfterCollection) EXPECT_NE(named_masked.find("external_table = 'mytable'"), String::npos); } +TEST(ParserCreateQuery, MaskXDBCNamedArgumentsWithoutCollection) +{ + /// A named argument at index 0 is not a collection name, so this call is not the positional form: + /// the connection string can be under either alias at any index, and the statement is formatted + /// for logging before validation rejects it. + const String query = + "CREATE TABLE test_jdbc (key UInt64) ENGINE = JDBC(external_database = 'mydb', " + "datasource = 'DSN=mydb;Uid=user;Pwd=plain_password')"; + + DB::ParserCreateQuery parser; + DB::ASTPtr ast = DB::parseQuery(parser, query, 0, 0, 0); + + const String masked = ast->formatForLogging(); + + EXPECT_EQ(masked.find("plain_password"), String::npos); + EXPECT_EQ(masked.find("Uid=user"), String::npos); + EXPECT_NE(masked.find("datasource = '[HIDDEN]'"), String::npos); +} + TEST(ParserCreateQuery, MaskRabbitMQTableEngineCredentials) { /// `RabbitMQ` also takes its settings as overrides of a named collection, so the same credentials @@ -487,6 +506,20 @@ TEST(ParserCreateQuery, MaskKafkaTableEngineCredentials) EXPECT_NE(positional_masked.find("group"), String::npos); EXPECT_EQ(positional_masked.find("[HIDDEN]"), String::npos); + /// The legacy positional form makes the collection name optional, so a named argument can be the + /// first one, and the statement is formatted for logging before it is rejected. + const String first_arg_query = + "CREATE TABLE test_kafka (key UInt64) " + "ENGINE = Kafka(kafka_sasl_password = 'plain_first_password', 'clickhouse')"; + + DB::ASTPtr first_arg_ast = DB::parseQuery(parser, first_arg_query, 0, 0, 0); + const String first_arg_masked = first_arg_ast->formatForLogging(); + + EXPECT_EQ(first_arg_masked.find("plain_first_password"), String::npos); + EXPECT_NE(first_arg_masked.find("kafka_sasl_password = '[HIDDEN]'"), String::npos); + /// The positional argument beside it is not a secret and stays visible. + EXPECT_NE(first_arg_masked.find("'clickhouse'"), String::npos); + /// The `SETTINGS` clause form is masked by `Kafka::SETTINGS_TO_HIDE` and must agree. const String settings_query = "CREATE TABLE test_kafka_settings (key UInt64) ENGINE = Kafka " diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference index e4fbeaba3b3e..104821b8bd64 100644 --- a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference @@ -6,6 +6,7 @@ b1_amqp_slash_password SET rabbitmq_address = '[HIDDEN]' b2_amqp_control SET rabbitmq_address = 'amqp://h:5672/v' c1_nats_engine_argument CREATE TABLE t05233 (`x` UInt8) ENGINE = NATS(nc05233, nats_url = '[HIDDEN]') c2_nats_engine_control CREATE TABLE t05233 (x UInt8) ENGINE = NATS(nc05233, nats_url = 'localhost:4222') +d10_jdbc_named_without_collection SELECT * FROM jdbc('[HIDDEN]', datasource = '[HIDDEN]') d1_jdbc_query_parameter CREATE TABLE db05233absent.t05233 (`x` UInt8) ENGINE = JDBC('[HIDDEN]', 'db', 't') d2_jdbc_at_in_password CREATE TABLE db05233absent.t05233 (`x` UInt8) ENGINE = JDBC('[HIDDEN]', 'db', 't') d3_odbc_key_value_form CREATE TABLE db05233absent.t05233 (`x` UInt8) ENGINE = ODBC('[HIDDEN]', 'db', 't') @@ -32,4 +33,5 @@ f2_rabbitmq_engine_password CREATE TABLE t05233 (`x` UInt8) ENGINE = RabbitMQ(nc f3_rabbitmq_engine_control CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://h:5672/v') f4_kafka_engine_password CREATE TABLE t05233 (`x` UInt8) ENGINE = Kafka(nc05233, kafka_sasl_password = '[HIDDEN]') f5_kafka_positional_control CREATE TABLE t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow') +f6_kafka_first_argument_secret CREATE TABLE t05233 (`x` UInt8) ENGINE = Kafka(kafka_sasl_password = '[HIDDEN]', 'clickhouse') 0 diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh index 1a2e17f0e1ab..556dfbbd3a48 100755 --- a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh @@ -43,6 +43,9 @@ arm f2_rabbitmq_engine_password "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitM arm f3_rabbitmq_engine_control "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://h:5672/v')" arm f4_kafka_engine_password "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka(nc05233, kafka_sasl_password = 'leak05233kafkapw')" arm f5_kafka_positional_control "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow')" +# That positional form also makes the collection name optional, so a named override can be the first +# argument. The positional beside it stays visible, so this arm is its own over-masking control. +arm f6_kafka_first_argument_secret "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka(kafka_sasl_password = 'leak05233kafkafirst', 'clickhouse')" # An XDBC connection string is forwarded to the driver verbatim, so its grammar is the driver's: the # password can sit in a query parameter or in a `KEY=value;` list, and no URI scan bounds it. The @@ -59,6 +62,9 @@ arm d7_jdbc_both_aliases "SELECT * FROM jdbc(nc05233, datasource = 'a://u:le arm d8_mysql_computed_key "SELECT * FROM mysql(nc05233, concat('ssl_ca', '_pem') = 'leak05233mysql', table = 't')" # After the collection name every argument must be named, so a positional one is hidden whole. arm d9_jdbc_positional_after_collection "SELECT * FROM jdbc(nc05233, 'jdbc://u:leak05233pos@h/db')" +# A named argument at index 0 is not a collection name, so the connection string can then sit at any +# index under either alias. +arm d10_jdbc_named_without_collection "SELECT * FROM jdbc(external_table = 't05233', datasource = 'DSN=x;Uid=u;Pwd=leak05233nocoll')" # The query-level URL settings are read through `Poco::URI`, so host and path stay visible and only # the userinfo is hidden. A value with no scheme in front of it is hidden whole instead. From ba517e21e331c40625ea8d8d790bdfeb424b5a07 Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 14:40:41 +0000 Subject: [PATCH 092/185] Start the positional MySQL TLS scans at index 0 findMySQLFunctionSecretArguments and findMySQLDatabaseSecretArguments scanned for TLS credentials from index 5 and index 4 in their positional branches, one past the last positional slot. The keys those scans hide are PEM material (ssl_ca_pem, ssl_cert_pem, ssl_key_pem and the three PostgreSQL spellings), the contents of a certificate or a private key, so a credential written before the positional arguments was masked nowhere: mysql(ssl_key_pem = '', '127.0.0.1:3306', 'db', 't', 'u') CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '', '127.0.0.1:3306', 'db', 'u') Such a call is invalid, but ASTFunction::formatImpl runs the finder whenever show_secrets is false, which is before any argument is validated, so the statement reaches system.query_log.query and the server log in cleartext first. The first function serves the mysql and postgresql table functions and the MySQL, PostgreSQL and MaterializedPostgreSQL table engines; the second serves the MySQL, PostgreSQL, MaterializedPostgreSQL, Remote and RemoteSecure database engines. Both scans now start at 0. That cannot widen the mask, because findNamedArgument and markNamedArgumentsWithUnreadableKeys skip any argument that is not an equals function with exactly two arguments, so every positional literal stays visible. The pre-existing arms of 04648_mysql_tls_credentials keep asserting that byte for byte. Two corrections to the previous commit's message. It claimed the class was closed, on the grounds that every other named-argument scan in the file starting at a non-zero index sits inside an isNamedCollectionName test and that the two branches it fixed were the only ones without that property. That was wrong. These two positional MySQL branches have the same shape and sit in the negative branch of the same test, where index 0 is proven not to be a collection name, and they are closed here. The statement that holds now is that no named-argument scan in the file starts past index 0 unless index 0 is proven to be a collection name. It also said the two arms it added each keep their non-secret neighbour visible, "so it doubles as the over-masking control". That holds for the Kafka arm, whose positional neighbour 'clickhouse' stays visible. It does not hold for the jdbc one: that neighbour sits at index 0 and is hidden whole by the untouched markSecretArgument(0, false). The over-masking control for that branch is d1 to d3, which keep their schema and table arguments visible. Two arms are added to 04648_mysql_tls_credentials, the test that already owns these keys, one per code path, through the formatter that both share. --- src/Parsers/FunctionSecretArgumentsFinder.cpp | 4 ++-- .../0_stateless/04648_mysql_tls_credentials.reference | 4 ++++ tests/queries/0_stateless/04648_mysql_tls_credentials.sh | 7 +++++++ 3 files changed, 13 insertions(+), 2 deletions(-) diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 4038ac7ed5d0..698086605c73 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -392,7 +392,7 @@ void FunctionSecretArgumentsFinder::findMySQLFunctionSecretArguments() { /// mysql('host:port', 'database', 'table', 'user', 'password', ...) markSecretArgument(4); - findTLSCredentialsSecretArguments(5); + findTLSCredentialsSecretArguments(0); } } @@ -1321,7 +1321,7 @@ void FunctionSecretArgumentsFinder::findMySQLDatabaseSecretArguments() { /// MySQL('host:port', 'database', 'user', 'password') markSecretArgument(3); - findTLSCredentialsSecretArguments(4); + findTLSCredentialsSecretArguments(0); } } diff --git a/tests/queries/0_stateless/04648_mysql_tls_credentials.reference b/tests/queries/0_stateless/04648_mysql_tls_credentials.reference index cd9aa6da303c..f82c14fb17fa 100644 --- a/tests/queries/0_stateless/04648_mysql_tls_credentials.reference +++ b/tests/queries/0_stateless/04648_mysql_tls_credentials.reference @@ -12,6 +12,10 @@ SELECT * FROM mysql('127.0.0.1:3306', 'db', 't', 'u', '[HIDDEN]', ssl_ca_pem = ' CREATE TABLE t (`x` Int32) ENGINE = MySQL('127.0.0.1:3306', 'db', 't', 'u', '[HIDDEN]', ssl_cert_pem = '[HIDDEN]') --- MySQL database engine, positional arguments CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', 'u', '[HIDDEN]', ssl_ca_pem = '[HIDDEN]') +--- mysql table function, credentials before the positional arguments +SELECT * FROM mysql(ssl_key_pem = '[HIDDEN]', '127.0.0.1:3306', 'db', 't', '[HIDDEN]') +--- MySQL database engine, credentials before the positional arguments +CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '[HIDDEN]', '127.0.0.1:3306', 'db', '[HIDDEN]') --- key given as a constant expression SELECT * FROM mysql(creds, concat('ssl_ca', '_pem') = '[HIDDEN]', `table` = 't') --- key given as a constant expression, positional arguments diff --git a/tests/queries/0_stateless/04648_mysql_tls_credentials.sh b/tests/queries/0_stateless/04648_mysql_tls_credentials.sh index 0fcb5994fd45..a6f214d3a93a 100755 --- a/tests/queries/0_stateless/04648_mysql_tls_credentials.sh +++ b/tests/queries/0_stateless/04648_mysql_tls_credentials.sh @@ -49,6 +49,13 @@ format "MySQL table engine, positional arguments" \ format "MySQL database engine, positional arguments" \ "CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', 'u', '${SECRET}', ssl_ca_pem = '${SECRET}')" +# The same credentials written before the positional arguments: the call is invalid, but it is +# formatted for logging before it is rejected, so the scan must not start past the first argument. +format "mysql table function, credentials before the positional arguments" \ + "SELECT * FROM mysql(ssl_key_pem = '${SECRET}', '127.0.0.1:3306', 'db', 't', 'u')" +format "MySQL database engine, credentials before the positional arguments" \ + "CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '${SECRET}', '127.0.0.1:3306', 'db', 'u')" + # The key of a named argument is not required to be a plain identifier or literal: the named # collection parser evaluates it as a constant expression, so `concat('ssl_ca', '_pem')` names a TLS # credential too. The formatter cannot evaluate it, so it hides the value of every argument whose key From ab89730870ac17c9483afe2fe2fff43f84fc323b Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 17:36:16 +0000 Subject: [PATCH 093/185] Locate the positional MySQL password among the positional arguments The positional branches of findMySQLFunctionSecretArguments and findMySQLDatabaseSecretArguments got the password wrong in one way with two faces: a positional slot is not a raw argument index, and a positional branch owes the same named-argument scan its named-collection sibling runs. A `key = value` argument written before the positional block shifts every positional one index along, so markSecretArgument(4) and markSecretArgument(3) hid the wrong argument and echoed the password; and because the branch scanned named arguments only for the TLS keys, a `password =` override written there was masked nowhere at all: mysql(ssl_key_pem = '', '127.0.0.1:3306', 'db', 't', 'u', '') mysql('127.0.0.1:3306', 'db', 't', password = '', 'u') CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '', '127.0.0.1:3306', 'db', 'u', '') CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', password = '', 'u') Such calls are invalid, but ASTFunction::formatImpl runs the finder whenever show_secrets is false, which is before any argument is validated, so the statement reaches system.query_log.query and the server log in cleartext first. Both faces are fixed together. classifyPositionalArguments returns the raw indexes of the arguments that are not `key = value` pairs, so the branch reads its slot out of that list instead of counting raw arguments, and a positional argument written after the first named one is hidden rather than listed: the parsers reject that mix, the statement is formatted before they do, and the slot it was meant to fill is then unknowable. classifyS3Arguments already applies that rule to the s3 signatures, and the helper reuses its shape rather than its body, which carries S3-specific key handling. The branch then runs findSecretNamedArgument("password", 0), the call its named-collection sibling already makes. A well-formed call is unchanged byte for byte: its named arguments come after the positional block, so the returned list is [0, 1, 2, ...] and the slot lookup yields the same index as before. Twenty of the twenty-two formatting assertions of 04648_mysql_tls_credentials are untouched; the two that move are the arms that previously recorded a credential as visible, and they now carry the password so the file's own leak oracle can see it. The invariant this closes is narrow, and only the part I enumerated is claimed. A named-argument scan in this file starts past index 0 in twenty-four places. It is safe where index 0 is proven to be a collection name, and also where index 0 is hidden whole: findXDBCSecretArguments's else branch scans from 1 with isNamedCollectionName(0) false, and is safe because markSecretArgument(0, false) above it hides that argument entirely. No claim is made for the file as a whole. A scan that starts past index 0 behind isNamedCollectionName(N) for N greater than zero still skips a secret named argument written first, which ExternalDistributed, azureBlobStorage and remote were measured to echo. Those are untouched here. --- src/Parsers/FunctionSecretArgumentsFinder.cpp | 35 +++++++++++++++++-- src/Parsers/FunctionSecretArgumentsFinder.h | 7 ++++ .../04648_mysql_tls_credentials.reference | 4 +-- .../04648_mysql_tls_credentials.sh | 8 +++-- ...oker_and_xdbc_credential_masking.reference | 2 +- ...5233_broker_and_xdbc_credential_masking.sh | 2 +- 6 files changed, 49 insertions(+), 9 deletions(-) diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index 698086605c73..b0642fd2cae5 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -391,11 +391,39 @@ void FunctionSecretArgumentsFinder::findMySQLFunctionSecretArguments() else { /// mysql('host:port', 'database', 'table', 'user', 'password', ...) - markSecretArgument(4); + const auto positional = classifyPositionalArguments(); + if (positional.size() > 4) + markSecretArgument(positional[4]); + findSecretNamedArgument("password", 0); findTLSCredentialsSecretArguments(0); } } +std::vector FunctionSecretArgumentsFinder::classifyPositionalArguments(size_t start) +{ + std::vector positional; + bool seen_named = false; + for (size_t i = start; i < function->arguments->size(); ++i) + { + const auto equals_func = function->arguments->at(i)->getFunction(); + if (equals_func && equals_func->name() == "equals" && equals_func->hasArguments() + && equals_func->arguments->size() == 2) + { + seen_named = true; + continue; + } + + if (seen_named) + { + markSecretArgument(i); + continue; + } + + positional.push_back(i); + } + return positional; +} + void FunctionSecretArgumentsFinder::markNamedArgumentsWithUnreadableKeys(size_t start) { /// The named-collection parser does not require the key of a `key = value` argument to be a plain @@ -1320,7 +1348,10 @@ void FunctionSecretArgumentsFinder::findMySQLDatabaseSecretArguments() else { /// MySQL('host:port', 'database', 'user', 'password') - markSecretArgument(3); + const auto positional = classifyPositionalArguments(); + if (positional.size() > 3) + markSecretArgument(positional[3]); + findSecretNamedArgument("password", 0); findTLSCredentialsSecretArguments(0); } } diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index 9aee325aafd3..2f72472f5f17 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -267,6 +267,13 @@ class FunctionSecretArgumentsFinder /// expression, so it can name a secret argument that `findNamedArgument` never sees. void markNamedArgumentsWithUnreadableKeys(size_t start); + /// The raw indexes of the arguments from `start` on that are not `key = value` pairs, in order, so a + /// branch that knows a secret's positional slot can find the argument that actually holds it. A + /// positional argument after the first named one is hidden instead of listed: the parsers reject that + /// mix, the statement is formatted for logging before they do, and the slot it was meant to fill is + /// then unknowable. `classifyS3Arguments` applies the same rule to the `s3` signatures. + std::vector classifyPositionalArguments(size_t start = 0); + /// Masks the secrets of an S3 named-collection form: the secret named overrides (every occurrence, /// in any order; the span covering them may hide a non-secret argument in between, which is safe) /// and the `headers(...)` / `extra_credentials(...)` map overrides. diff --git a/tests/queries/0_stateless/04648_mysql_tls_credentials.reference b/tests/queries/0_stateless/04648_mysql_tls_credentials.reference index f82c14fb17fa..237a8929e956 100644 --- a/tests/queries/0_stateless/04648_mysql_tls_credentials.reference +++ b/tests/queries/0_stateless/04648_mysql_tls_credentials.reference @@ -13,9 +13,9 @@ CREATE TABLE t (`x` Int32) ENGINE = MySQL('127.0.0.1:3306', 'db', 't', 'u', '[HI --- MySQL database engine, positional arguments CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', 'u', '[HIDDEN]', ssl_ca_pem = '[HIDDEN]') --- mysql table function, credentials before the positional arguments -SELECT * FROM mysql(ssl_key_pem = '[HIDDEN]', '127.0.0.1:3306', 'db', 't', '[HIDDEN]') +SELECT * FROM mysql(ssl_key_pem = '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]') --- MySQL database engine, credentials before the positional arguments -CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '[HIDDEN]', '127.0.0.1:3306', 'db', '[HIDDEN]') +CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]') --- key given as a constant expression SELECT * FROM mysql(creds, concat('ssl_ca', '_pem') = '[HIDDEN]', `table` = 't') --- key given as a constant expression, positional arguments diff --git a/tests/queries/0_stateless/04648_mysql_tls_credentials.sh b/tests/queries/0_stateless/04648_mysql_tls_credentials.sh index a6f214d3a93a..4cb5e6ac04cf 100755 --- a/tests/queries/0_stateless/04648_mysql_tls_credentials.sh +++ b/tests/queries/0_stateless/04648_mysql_tls_credentials.sh @@ -50,11 +50,13 @@ format "MySQL database engine, positional arguments" \ "CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', 'u', '${SECRET}', ssl_ca_pem = '${SECRET}')" # The same credentials written before the positional arguments: the call is invalid, but it is -# formatted for logging before it is rejected, so the scan must not start past the first argument. +# formatted for logging before it is rejected, so the scan must not start past the first argument, +# and the positional password must be located among the positional arguments rather than at a fixed +# argument index, which a named argument written first moves. format "mysql table function, credentials before the positional arguments" \ - "SELECT * FROM mysql(ssl_key_pem = '${SECRET}', '127.0.0.1:3306', 'db', 't', 'u')" + "SELECT * FROM mysql(ssl_key_pem = '${SECRET}', '127.0.0.1:3306', 'db', 't', 'u', '${SECRET}')" format "MySQL database engine, credentials before the positional arguments" \ - "CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '${SECRET}', '127.0.0.1:3306', 'db', 'u')" + "CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '${SECRET}', '127.0.0.1:3306', 'db', 'u', '${SECRET}')" # The key of a named argument is not required to be a plain identifier or literal: the named # collection parser evaluates it as a constant expression, so `concat('ssl_ca', '_pem')` names a TLS diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference index 104821b8bd64..126d8989eee7 100644 --- a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.reference @@ -32,6 +32,6 @@ f1_rabbitmq_engine_argument CREATE TABLE t05233 (`x` UInt8) ENGINE = RabbitMQ(nc f2_rabbitmq_engine_password CREATE TABLE t05233 (`x` UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_password = '[HIDDEN]') f3_rabbitmq_engine_control CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://h:5672/v') f4_kafka_engine_password CREATE TABLE t05233 (`x` UInt8) ENGINE = Kafka(nc05233, kafka_sasl_password = '[HIDDEN]') -f5_kafka_positional_control CREATE TABLE t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow') +f5_kafka_positional_control CREATE TABLE db05233absent.t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow') f6_kafka_first_argument_secret CREATE TABLE t05233 (`x` UInt8) ENGINE = Kafka(kafka_sasl_password = '[HIDDEN]', 'clickhouse') 0 diff --git a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh index 556dfbbd3a48..85ff60ae41da 100755 --- a/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh +++ b/tests/queries/0_stateless/05233_broker_and_xdbc_credential_masking.sh @@ -42,7 +42,7 @@ arm f1_rabbitmq_engine_argument "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitM arm f2_rabbitmq_engine_password "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_password = 'leak05233rmqpw')" arm f3_rabbitmq_engine_control "CREATE TABLE t05233 (x UInt8) ENGINE = RabbitMQ(nc05233, rabbitmq_address = 'amqp://h:5672/v')" arm f4_kafka_engine_password "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka(nc05233, kafka_sasl_password = 'leak05233kafkapw')" -arm f5_kafka_positional_control "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow')" +arm f5_kafka_positional_control "CREATE TABLE db05233absent.t05233 (x UInt8) ENGINE = Kafka('broker05233:9092', 'topic05233', 'group05233', 'JSONEachRow')" # That positional form also makes the collection name optional, so a named override can be the first # argument. The positional beside it stays visible, so this arm is its own over-masking control. arm f6_kafka_first_argument_secret "CREATE TABLE t05233 (x UInt8) ENGINE = Kafka(kafka_sasl_password = 'leak05233kafkafirst', 'clickhouse')" From 4c6ceaa199766baa777776c9c893b303738fb7c2 Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 19:29:20 +0000 Subject: [PATCH 094/185] Do not echo the rejected url_base or s3_base value StorageURL::resolveURLBase rejects a base with no "://" by interpolating the raw value into its BAD_ARGUMENTS message. The value is a setting that can carry a credential, so the credential is disclosed in full through the client error, the exception column of the query log and the server log: SELECT * FROM url('data.csv', CSV, 'c String') SETTINGS url_base = 'user:@example.invalid/def/' Code: 36. The `url_base` setting must contain a scheme (e.g. https://), got: user:@example.invalid/def/ Nothing gates that. url_base is settable in a settings profile, so the value belongs to whoever wrote the profile while the error goes to whichever user trips it, and the message is not behind format_display_secrets_in_show_and_select. The same message carries s3_base and format_avro_schema_registry_url: the callers that reach it with a raw user value are TableFunctionURL.cpp for url_base and ObjectStorage/S3/Configuration.cpp in two places for s3_base. The value is dropped rather than masked. A masker only finds a credential it can locate: maskURLCredentials fires on a value containing '@', so s3_base = 'DSN=x;Pwd=' would still be echoed whole. The setting name stays in the message, which is what a user needs in order to find their own typo. The sibling carrier already has this treatment. DatabaseURL refuses a scheme-less base without echoing it, and 04627_url_database_engine asserts that with a leak count. The url_base and s3_base arms of the same invariant grep only the message prefix, which is why nothing in tree caught this. They now carry the same pair of assertions, so all three carriers are pinned the same way. The text before the removed ", got:" is byte identical, so those pre-existing prefix arms and their reference lines do not move. The new arms run clickhouse-local rather than clickhouse-client: the client appends the query it was given to an exception, and that echo is the caller's own input rather than anything the message disclosed, so a leak count taken through the client can never reach zero. 04648_mysql_tls_credentials gains two arms as well. The two findSecretNamedArgument("password", 0) calls added to the positional MySQL branches in the previous commit were exercised by no test: every existing named password arm goes through the collection branch, and the arms that carry a named credential beside a positional block use a TLS key. The new arms are the collection-less named password shape for each branch, and deleting either call reddens exactly one of them. --- src/Storages/StorageURL.cpp | 3 ++- tests/queries/0_stateless/04070_url_base_setting.reference | 2 ++ tests/queries/0_stateless/04070_url_base_setting.sh | 7 +++++++ tests/queries/0_stateless/04626_s3_base_setting.reference | 2 ++ tests/queries/0_stateless/04626_s3_base_setting.sh | 7 +++++++ .../0_stateless/04648_mysql_tls_credentials.reference | 4 ++++ tests/queries/0_stateless/04648_mysql_tls_credentials.sh | 7 +++++++ 7 files changed, 31 insertions(+), 1 deletion(-) diff --git a/src/Storages/StorageURL.cpp b/src/Storages/StorageURL.cpp index 7f3e7f6dca7a..5ebb3cec411e 100644 --- a/src/Storages/StorageURL.cpp +++ b/src/Storages/StorageURL.cpp @@ -1882,8 +1882,9 @@ String StorageURL::resolveURLBase(const String & url, const String & base, const } auto scheme_end = base.find("://"); + /// Not echoed back: the value can carry a credential, and password masking anchors on the `://` it lacks. if (scheme_end == String::npos) - throw Exception(ErrorCodes::BAD_ARGUMENTS, "The `{}` setting must contain a scheme (e.g. https://), got: {}", base_setting_name, base); + throw Exception(ErrorCodes::BAD_ARGUMENTS, "The `{}` setting must contain a scheme (e.g. https://)", base_setting_name); /// Find the boundary of the path component in the base URL (before '?' or '#'). auto authority_start = scheme_end + 3; /// skip "://" diff --git a/tests/queries/0_stateless/04070_url_base_setting.reference b/tests/queries/0_stateless/04070_url_base_setting.reference index 8c23601338af..958e5fdfc6c7 100644 --- a/tests/queries/0_stateless/04070_url_base_setting.reference +++ b/tests/queries/0_stateless/04070_url_base_setting.reference @@ -28,3 +28,5 @@ url = 'http://base.invalid/dir/persist.csv' 0 0 must contain a scheme +must contain a scheme +0 diff --git a/tests/queries/0_stateless/04070_url_base_setting.sh b/tests/queries/0_stateless/04070_url_base_setting.sh index 46289ca8b2c1..9127cb87d066 100755 --- a/tests/queries/0_stateless/04070_url_base_setting.sh +++ b/tests/queries/0_stateless/04070_url_base_setting.sh @@ -172,3 +172,10 @@ DROP TABLE IF EXISTS ${CLICKHOUSE_TEST_UNIQUE_NAME}_url_pos_creds; # Invalid url_base (no scheme) should produce an error $CLICKHOUSE_CLIENT --query "SELECT * FROM url('data.csv', CSV, 'c String') SETTINGS url_base = 'example.invalid/def/', $FAST" 2>&1 | grep -oF 'must contain a scheme' | head -1 + +# The rejected value is not echoed back: it can carry a credential, and the message reaches the client, +# the exception column of the query log and the server log, none of which the display-secrets setting +# gates. (last line counts the leaks; clickhouse-local is used because the client also prints back the +# query it was given, which is the caller's own input rather than something the message disclosed) +$CLICKHOUSE_LOCAL --query "SELECT * FROM url('data.csv', CSV, 'c String') SETTINGS url_base = 'user:SEKRIT_PW@example.invalid/def/', $FAST" 2>&1 | grep -oF 'must contain a scheme' | head -1 +$CLICKHOUSE_LOCAL --query "SELECT * FROM url('data.csv', CSV, 'c String') SETTINGS url_base = 'user:SEKRIT_PW@example.invalid/def/', $FAST" 2>&1 | grep -c SEKRIT_PW ||: diff --git a/tests/queries/0_stateless/04626_s3_base_setting.reference b/tests/queries/0_stateless/04626_s3_base_setting.reference index 6b40f3b84623..97a52a1f9a5c 100644 --- a/tests/queries/0_stateless/04626_s3_base_setting.reference +++ b/tests/queries/0_stateless/04626_s3_base_setting.reference @@ -20,3 +20,5 @@ 5 --- s3_base without a scheme is an error must contain a scheme +must contain a scheme +0 diff --git a/tests/queries/0_stateless/04626_s3_base_setting.sh b/tests/queries/0_stateless/04626_s3_base_setting.sh index e8793ce77cab..88ef59858add 100755 --- a/tests/queries/0_stateless/04626_s3_base_setting.sh +++ b/tests/queries/0_stateless/04626_s3_base_setting.sh @@ -61,3 +61,10 @@ ${CLICKHOUSE_CLIENT} -q "DROP TABLE test_s3_base_nc" && ${CLICKHOUSE_CLIENT} -q echo '--- s3_base without a scheme is an error' ${CLICKHOUSE_CLIENT} -q "SELECT * FROM s3('${FILE}', 'test', 'testtest', 'TSV', 'n UInt32, s String') SETTINGS s3_base = 'localhost:11111/test/'" 2>&1 | grep -oF 'must contain a scheme' | head -1 + +# The rejected value is not echoed back: it can carry a credential, and the message reaches the client, +# the exception column of the query log and the server log, none of which the display-secrets setting +# gates. (last line counts the leaks; clickhouse-local is used because the client also prints back the +# query it was given, which is the caller's own input rather than something the message disclosed) +${CLICKHOUSE_LOCAL} -q "SELECT * FROM s3('${FILE}', 'test', 'testtest', 'TSV', 'n UInt32, s String') SETTINGS s3_base = 'user:SEKRIT_PW@localhost:11111/test/'" 2>&1 | grep -oF 'must contain a scheme' | head -1 +${CLICKHOUSE_LOCAL} -q "SELECT * FROM s3('${FILE}', 'test', 'testtest', 'TSV', 'n UInt32, s String') SETTINGS s3_base = 'user:SEKRIT_PW@localhost:11111/test/'" 2>&1 | grep -c SEKRIT_PW ||: diff --git a/tests/queries/0_stateless/04648_mysql_tls_credentials.reference b/tests/queries/0_stateless/04648_mysql_tls_credentials.reference index 237a8929e956..daa2aafc0de8 100644 --- a/tests/queries/0_stateless/04648_mysql_tls_credentials.reference +++ b/tests/queries/0_stateless/04648_mysql_tls_credentials.reference @@ -16,6 +16,10 @@ CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', 'u', '[HIDDEN]', ssl_ca SELECT * FROM mysql(ssl_key_pem = '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]') --- MySQL database engine, credentials before the positional arguments CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]', '[HIDDEN]') +--- mysql table function, named password without a collection +SELECT * FROM mysql('127.0.0.1:3306', 'db', 't', password = '[HIDDEN]') +--- MySQL database engine, named password without a collection +CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', password = '[HIDDEN]') --- key given as a constant expression SELECT * FROM mysql(creds, concat('ssl_ca', '_pem') = '[HIDDEN]', `table` = 't') --- key given as a constant expression, positional arguments diff --git a/tests/queries/0_stateless/04648_mysql_tls_credentials.sh b/tests/queries/0_stateless/04648_mysql_tls_credentials.sh index 4cb5e6ac04cf..0bd7978918bb 100755 --- a/tests/queries/0_stateless/04648_mysql_tls_credentials.sh +++ b/tests/queries/0_stateless/04648_mysql_tls_credentials.sh @@ -58,6 +58,13 @@ format "mysql table function, credentials before the positional arguments" \ format "MySQL database engine, credentials before the positional arguments" \ "CREATE DATABASE d ENGINE = MySQL(ssl_ca_pem = '${SECRET}', '127.0.0.1:3306', 'db', 'u', '${SECRET}')" +# A named `password` override with no named collection: the parsers reject the mix, but the statement is +# formatted for logging first, and the key is readable, so only the named scan can hide this value. +format "mysql table function, named password without a collection" \ + "SELECT * FROM mysql('127.0.0.1:3306', 'db', 't', password = '${SECRET}')" +format "MySQL database engine, named password without a collection" \ + "CREATE DATABASE d ENGINE = MySQL('127.0.0.1:3306', 'db', password = '${SECRET}')" + # The key of a named argument is not required to be a plain identifier or literal: the named # collection parser evaluates it as a constant expression, so `concat('ssl_ca', '_pem')` names a TLS # credential too. The formatter cannot evaluate it, so it hides the value of every argument whose key From 6d6f859750146f752f56771ec09cf365f85f9b94 Mon Sep 17 00:00:00 2001 From: Groene AI <270696204+groeneai@users.noreply.github.com> Date: Mon, 21 Sep 2026 20:08:07 +0000 Subject: [PATCH 095/185] Trim comments Delete comments narrating this change's own design, keep the third-party masker semantics (libnats, AMQP-CPP, XDBC driver grammar) and the cross-file sync invariants as standalone properties. --- src/Common/maskURIPassword.h | 4 +- src/Core/SettingsSecrets.h | 6 +-- src/Parsers/FunctionSecretArgumentsFinder.cpp | 37 ++++++------------- src/Parsers/FunctionSecretArgumentsFinder.h | 25 +++---------- src/Storages/NATS/NATS_fwd.h | 6 +-- src/Storages/RabbitMQ/RabbitMQ_fwd.h | 6 +-- 6 files changed, 23 insertions(+), 61 deletions(-) diff --git a/src/Common/maskURIPassword.h b/src/Common/maskURIPassword.h index d1f88c5e43b7..29038c229fb7 100644 --- a/src/Common/maskURIPassword.h +++ b/src/Common/maskURIPassword.h @@ -73,9 +73,7 @@ inline bool maskURIPassword(std::string * uri) return false; } -/** The offset just past the `://` of a value that starts with an RFC 3986 scheme, and `npos` when it - * does not start with one. A value with no scheme at its start has no authority that a URI parser - * would recognise, so nothing in it can be located as a credential by position. +/** The offset just past the `://` of a value that starts with an RFC 3986 scheme, `npos` otherwise. */ inline size_t findURIAuthority(std::string_view uri) { diff --git a/src/Core/SettingsSecrets.h b/src/Core/SettingsSecrets.h index 11c49b1f8290..41dcce0c253b 100644 --- a/src/Core/SettingsSecrets.h +++ b/src/Core/SettingsSecrets.h @@ -27,10 +27,8 @@ using ValueMaskingFunc = std::function; /// precondition holds by construction and needs no check. inline bool maskURLCredentials(String & value) { - /// Nothing validates these settings when they are set, `StorageURL::resolveURLBase` accepts a base - /// whose `://` is anywhere rather than at the start, and a statement is masked for logging before - /// its settings are validated. So a value with no scheme in front still reaches a log, and no URI - /// parser can locate the credential inside it: hide such a value whole. + /// A statement is masked for logging before its settings are validated, so a value that no URI + /// parser can read still reaches a log. if (findURIAuthority(value) == String::npos && value.contains('@')) { value = "[HIDDEN]"; diff --git a/src/Parsers/FunctionSecretArgumentsFinder.cpp b/src/Parsers/FunctionSecretArgumentsFinder.cpp index b0642fd2cae5..8fc8edcbfd84 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.cpp +++ b/src/Parsers/FunctionSecretArgumentsFinder.cpp @@ -526,12 +526,11 @@ void FunctionSecretArgumentsFinder::findArrowFlightSecretArguments() void FunctionSecretArgumentsFinder::findXDBCSecretArguments() { - /// The connection string is never parsed by ClickHouse: `ITableFunctionXDBC` forwards it verbatim - /// to the bridge, so its grammar belongs to the JDBC/ODBC driver. It can carry the password in a - /// query parameter (`jdbc('mysql://host:3306/?user=root&password=root', ...)`, from this function's - /// own documentation) or as `Pwd=` in a `KEY=value;` list, neither of which a URI scan locates. - /// There is no extent that can be kept visible, so the value is hidden whole; - /// `format_display_secrets_in_show_and_select` still shows the real one to a privileged reader. + /// The connection string goes verbatim to the bridge, so its grammar is the JDBC/ODBC driver's: the + /// password can sit in a query parameter (`?password=`) or as `Pwd=` in a `KEY=value;` list. + /// An invalid call is formatted for logging before validation rejects it, so both branches below + /// fail closed: after a collection name a positional argument can be the connection string, and a + /// named argument means the call is not the positional form at all. if (isNamedCollectionName(0)) { /// jdbc(named_collection, ..., datasource = 'DSN', ...) @@ -554,9 +553,6 @@ void FunctionSecretArgumentsFinder::findXDBCSecretArguments() findSecretNamedArgument("connection_settings", 1); markNamedArgumentsWithUnreadableKeys(1); - /// After the collection name every argument must be a named override. A positional one is - /// invalid, and the statement is formatted for logging before validation rejects it, so it - /// can carry the connection string itself: hide it whole (fail closed). for (size_t i = 1; i < function->arguments->size(); ++i) { const auto equals_func = function->arguments->at(i)->getFunction(); @@ -572,9 +568,6 @@ void FunctionSecretArgumentsFinder::findXDBCSecretArguments() /// JDBC('DSN', database, table) / ODBC('DSN', database, table) markSecretArgument(0, false); - /// A named argument means this is not the positional form at all, and the connection string - /// can then sit at any index under either alias. Validation rejects such a call only after the - /// statement has been formatted for logging, so hide those values too (fail closed). findSecretNamedArgument("datasource", 1); findSecretNamedArgument("connection_settings", 1); markNamedArgumentsWithUnreadableKeys(1); @@ -1025,12 +1018,8 @@ void FunctionSecretArgumentsFinder::findTableEngineSecretArguments() } else if (engine_name == "Kafka") { - /// Kafka(named_collection, kafka_sasl_password = '...') - `registerStorageKafka` reads named - /// overrides of the collection, so an override carries the secret the `SETTINGS` clause form - /// hides through `Kafka::SETTINGS_TO_HIDE`. The legacy positional form - /// (`Kafka('brokers', 'topics', 'group', 'format', ...)`) carries no secret and stays visible, - /// so this form does not fail closed on a positional argument. That form also makes the - /// collection name optional, so the scan starts at index 0: a named argument can be the first. + /// Kafka(named_collection, kafka_sasl_password = '...'); the legacy positional form carries no + /// secret and makes the collection name optional, so a named argument can be the first one. findSecretNamedArgument("kafka_sasl_password", 0); markNamedArgumentsWithUnreadableKeys(0); } @@ -1050,14 +1039,11 @@ void FunctionSecretArgumentsFinder::findBrokerTableEngineSecretArguments( /// [, nats_credential_file = '/path'] [, nats_credentials = 'user JWT and seed'] /// [, nats_url = 'nats://user:password@host:4222'] /// [, nats_server_list = 'nats://user:password@host:4222,...'], ...) - /// RabbitMQ(named_collection [, rabbitmq_password = 'password'] - /// [, rabbitmq_address = 'amqp://user:password@host:5672/vhost'], ...) + /// RabbitMQ(named_collection [, rabbitmq_password = '...'] [, rabbitmq_address = 'amqp://user:pass@host'], ...) /// The only positional argument these engines accept is the name of a named collection, so the /// credentials can only appear as named overrides. The `SETTINGS` clause form is masked - /// separately by the engine's own `SETTINGS_TO_HIDE`, and this function masks the same keys the - /// same way: the secrets are hidden whole, and so is an address that carries an '@'. - /// A key list can hold a destination (`nats_server_list`), which is hidden whole because each - /// list entry can carry userinfo credentials. + /// separately by the engine's own `SETTINGS_TO_HIDE`, which this function must stay in sync with. + /// A destination key (`nats_server_list`) is hidden whole: each list entry can carry userinfo. /// Fail closed on a key we cannot read as a plain literal: it can name a secret setting. for (size_t i = 0; i < function->arguments->size(); ++i) { @@ -1084,8 +1070,7 @@ void FunctionSecretArgumentsFinder::findBrokerTableEngineSecretArguments( String url; if (equals_func->arguments->at(1)->tryGetString(&url, /* allow_identifier= */ false)) { - /// An '@' is the only reliable sign of a credential here, and there is no extent to - /// keep visible; see the address rule in the engine's `_fwd.h` for why. + /// An '@' is the only reliable sign of a credential here; see the engine's `_fwd.h`. if (url.contains('@')) markSecretArgument(i, /* argument_is_named= */ true); } diff --git a/src/Parsers/FunctionSecretArgumentsFinder.h b/src/Parsers/FunctionSecretArgumentsFinder.h index 2f72472f5f17..79894454c833 100644 --- a/src/Parsers/FunctionSecretArgumentsFinder.h +++ b/src/Parsers/FunctionSecretArgumentsFinder.h @@ -121,15 +121,12 @@ class FunctionSecretArgumentsFinder /// Named arguments carrying NATS credentials. They are the setting names, because the `NATS` engine /// takes its arguments as overrides of a named collection (`NATS(collection, nats_token = '...')`). /// `nats_server_list` is a destination and can carry URI userinfo credentials, so hide it whole. - /// `nats_url` is not here because the documented form (`'localhost:4222'`) carries no credential and - /// stays visible; it is handled separately, by the presence of an '@'. + /// `nats_url` is not here: it is hidden only when its value carries an '@'. /// Keep in sync with `NATS::SETTINGS_TO_HIDE`, which masks the same secrets in the `SETTINGS` clause. static constexpr std::string_view nats_secret_keys[] = {"nats_password", "nats_token", "nats_credential_file", "nats_credentials", "nats_server_list"}; - /// Named arguments carrying RabbitMQ credentials, as the setting names, for the same reason as - /// `nats_secret_keys`. `rabbitmq_address` is not here because the documented form carries no - /// credential and stays visible; it is handled by the presence of an '@'. + /// As `nats_secret_keys`, for RabbitMQ; `rabbitmq_address` is hidden only when it carries an '@'. /// Keep in sync with `RabbitMQ::SETTINGS_TO_HIDE`. static constexpr std::string_view rabbitmq_secret_keys[] = {"rabbitmq_password"}; @@ -228,13 +225,6 @@ class FunctionSecretArgumentsFinder void findRedisFunctionSecretArguments(); void findYTsaurusStorageTableEngineSecretArguments(); void findBigQuerySecretArguments(); - /// Masks the named-collection override form of a broker engine - /// (`NATS(collection, nats_token = '...')`, `RabbitMQ(collection, rabbitmq_address = '...')`). - /// `secret_keys` are hidden whole; `address_key` is hidden only when its value carries an '@', - /// because the documented address form carries no credential. Fails closed on a positional - /// argument and on a key this finder cannot read as a plain literal. - /// Keep the key lists in sync with the engine's own `SETTINGS_TO_HIDE`, which masks the same - /// settings in the `SETTINGS` clause. void findBrokerTableEngineSecretArguments( std::span secret_keys, std::string_view address_key); void findNATSTableEngineSecretArguments(); @@ -262,16 +252,11 @@ class FunctionSecretArgumentsFinder /// duplicate-key validation runs, so `session_token = 'a', session_token = 'b'` must hide both. bool findSecretNamedArgument(std::string_view key, size_t start = 0); - /// Hides the value of every `key = value` argument from `start` on whose key this finder cannot read - /// as a plain literal or identifier. The named-collection parser evaluates such a key as a constant - /// expression, so it can name a secret argument that `findNamedArgument` never sees. + /// Hides the value of every `key = value` argument from `start` on whose key is not a plain literal. void markNamedArgumentsWithUnreadableKeys(size_t start); - /// The raw indexes of the arguments from `start` on that are not `key = value` pairs, in order, so a - /// branch that knows a secret's positional slot can find the argument that actually holds it. A - /// positional argument after the first named one is hidden instead of listed: the parsers reject that - /// mix, the statement is formatted for logging before they do, and the slot it was meant to fill is - /// then unknowable. `classifyS3Arguments` applies the same rule to the `s3` signatures. + /// The raw indexes of the arguments from `start` on that are not `key = value` pairs, in order. A + /// positional argument after the first named one is hidden instead of listed: its slot is unknowable. std::vector classifyPositionalArguments(size_t start = 0); /// Masks the secrets of an S3 named-collection form: the secret named overrides (every occurrence, diff --git a/src/Storages/NATS/NATS_fwd.h b/src/Storages/NATS/NATS_fwd.h index a74a9d3687c4..a471bffbbd1d 100644 --- a/src/Storages/NATS/NATS_fwd.h +++ b/src/Storages/NATS/NATS_fwd.h @@ -25,10 +25,8 @@ static inline std::unordered_map SETTINGS_TO_HIDE = std::string masked_value; if (!value.tryGet(masked_value)) return {}; - /// libnats trims the value, takes the scheme as optional, and ends the userinfo at the LAST - /// '@' of the whole value (`contrib/nats-io/src/url.c`, `natsUrl_Create`), so its credential - /// can run past anything an RFC 3986 authority covers. An '@' is therefore the only reliable - /// sign that this address carries one, and there is no extent to keep visible. + /// libnats takes the scheme as optional and ends the userinfo at the LAST '@' of the whole + /// value (`contrib/nats-io/src/url.c`, `natsUrl_Create`), so no URI authority bounds it. if (masked_value.contains('@')) masked_value = "[HIDDEN]"; return fmt::format("'{}'", masked_value); diff --git a/src/Storages/RabbitMQ/RabbitMQ_fwd.h b/src/Storages/RabbitMQ/RabbitMQ_fwd.h index d0f5ccf96090..6dc017ae1e25 100644 --- a/src/Storages/RabbitMQ/RabbitMQ_fwd.h +++ b/src/Storages/RabbitMQ/RabbitMQ_fwd.h @@ -20,10 +20,8 @@ static inline std::unordered_map SETTINGS_TO_HIDE = std::string masked_value; if (!value.tryGet(masked_value)) return {}; - /// AMQP-CPP ends the login at the FIRST '@' after the scheme, unbounded by the `/?#` that - /// closes an RFC 3986 authority (`contrib/AMQP-CPP/include/amqpcpp/address.h`, `Address`), so - /// its credential can run past anything a URI masker covers. An '@' is therefore the only - /// reliable sign that this address carries one, and there is no extent to keep visible. + /// AMQP-CPP ends the login at the FIRST '@' after the scheme, unbounded by the `/?#` that closes + /// an RFC 3986 authority (`contrib/AMQP-CPP/include/amqpcpp/address.h`), so no URI masker bounds it. if (masked_value.contains('@')) masked_value = "[HIDDEN]"; return fmt::format("'{}'", masked_value); From f5fd959ecd20c72268a41f7c63c66a177f1b341a Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 12:48:04 +0000 Subject: [PATCH 096/185] Backport #122345 to 26.8: Fix LOGICAL_ERROR on lambda over an `Array(Tuple(...))` --- src/Analyzer/Resolve/IdentifierResolver.cpp | 11 ++++ .../evaluateScalarSubqueryIfNeeded.cpp | 4 +- src/Analyzer/Resolve/resolveFunction.cpp | 5 +- src/Interpreters/Context.cpp | 6 +++ src/Interpreters/Context.h | 1 + .../InterpreterSelectQueryAnalyzer.cpp | 37 +++++++++++++- .../QueryPlan/DistributedCreateLocalPlan.cpp | 5 ++ src/Storages/StorageView.cpp | 3 ++ ...lter_source_column_during_insert.reference | 1 + ...nsert_alter_source_column_during_insert.sh | 43 ++++++++++++++++ ...ary_view_over_source_reads_table.reference | 5 ++ ..._ordinary_view_over_source_reads_table.sql | 51 +++++++++++++++++++ ...292_mv_source_substitution_scope.reference | 16 ++++++ .../05292_mv_source_substitution_scope.sql | 51 +++++++++++++++++++ ...rdinary_view_over_source_inlined.reference | 4 ++ ...3_mv_ordinary_view_over_source_inlined.sql | 37 ++++++++++++++ ...e_remote_local_shard_reads_table.reference | 2 + ...v_source_remote_local_shard_reads_table.sh | 45 ++++++++++++++++ 18 files changed, 323 insertions(+), 4 deletions(-) create mode 100644 tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.reference create mode 100755 tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.sh create mode 100644 tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.reference create mode 100644 tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.sql create mode 100644 tests/queries/0_stateless/05292_mv_source_substitution_scope.reference create mode 100644 tests/queries/0_stateless/05292_mv_source_substitution_scope.sql create mode 100644 tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.reference create mode 100644 tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.sql create mode 100644 tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.reference create mode 100755 tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.sh diff --git a/src/Analyzer/Resolve/IdentifierResolver.cpp b/src/Analyzer/Resolve/IdentifierResolver.cpp index c14879e0afb2..73cd1c3c40a9 100644 --- a/src/Analyzer/Resolve/IdentifierResolver.cpp +++ b/src/Analyzer/Resolve/IdentifierResolver.cpp @@ -305,6 +305,17 @@ std::shared_ptr IdentifierResolver::tryResolveTableIdentifier(const I StorageID storage_id(database_name, table_name); storage_id = context->resolveStorageID(storage_id); + + /// The view source carries the inserted block and its types. For a MV, return this source + /// directly as a table node instead of swapping it later for the storage from the catalog + /// which may have been changed by a concurrent ALTER (the MV types must match the + /// snapshot at the start of the INSERT, not the current types). + /// For an inner query of an ordinary view, keep the normal flow that resolves from the catalog. + if (auto view_source = context->getViewSource(); + view_source && !context->isViewInnerQuery() + && view_source->getStorageID().getFullNameNotQuoted() == storage_id.getFullNameNotQuoted()) + return std::make_shared(view_source, context); + bool is_temporary_table = storage_id.getDatabaseName() == DatabaseCatalog::TEMPORARY_DATABASE; StoragePtr storage; diff --git a/src/Analyzer/Resolve/evaluateScalarSubqueryIfNeeded.cpp b/src/Analyzer/Resolve/evaluateScalarSubqueryIfNeeded.cpp index 2b9c6355a330..93d4dfb653cb 100644 --- a/src/Analyzer/Resolve/evaluateScalarSubqueryIfNeeded.cpp +++ b/src/Analyzer/Resolve/evaluateScalarSubqueryIfNeeded.cpp @@ -215,7 +215,9 @@ void QueryAnalyzer::evaluateScalarSubqueryIfNeeded(QueryTreeNodePtr & node, Iden addQueryTreePasses(query_tree_pass_manager, options.only_analyze); query_tree_pass_manager.run(query_tree); - if (auto storage = subquery_context->getViewSource()) + /// The inner query of an ordinary view referenced by the view query reads the table itself, + /// not the inserted block. + if (auto storage = subquery_context->getViewSource(); storage && !subquery_context->isViewInnerQuery()) replaceStorageInQueryTree(query_tree, subquery_context, storage); auto interpreter = std::make_unique(query_tree, subquery_context, options); diff --git a/src/Analyzer/Resolve/resolveFunction.cpp b/src/Analyzer/Resolve/resolveFunction.cpp index bb3e74a48993..aacf493888b7 100644 --- a/src/Analyzer/Resolve/resolveFunction.cpp +++ b/src/Analyzer/Resolve/resolveFunction.cpp @@ -2141,8 +2141,9 @@ ProjectionNames QueryAnalyzer::resolveFunction(QueryTreeNodePtr & node, Identifi } else { - /// Replace storage with values storage of insertion block - if (StoragePtr storage = scope.context->getViewSource()) + /// Replace storage with values storage of insertion block. + /// The inner query of an ordinary view referenced by the view query reads the table itself. + if (StoragePtr storage = scope.context->getViewSource(); storage && !scope.context->isViewInnerQuery()) { QueryTreeNodePtr table_expression = in_second_argument; diff --git a/src/Interpreters/Context.cpp b/src/Interpreters/Context.cpp index 4bcc3b6bc767..54cc24c6f638 100644 --- a/src/Interpreters/Context.cpp +++ b/src/Interpreters/Context.cpp @@ -3388,6 +3388,12 @@ StoragePtr Context::getViewSource() const return view_source; } + +void Context::clearViewSource() +{ + view_source.reset(); +} + bool Context::displaySecretsInShowAndSelect() const { return shared->server_settings[ServerSetting::display_secrets_in_show_and_select]; diff --git a/src/Interpreters/Context.h b/src/Interpreters/Context.h index 6c2dcdcff8f5..522eef1fa017 100644 --- a/src/Interpreters/Context.h +++ b/src/Interpreters/Context.h @@ -1156,6 +1156,7 @@ class Context: public ContextData, public std::enable_shared_from_this void addViewSource(const StoragePtr & storage); StoragePtr getViewSource() const; + void clearViewSource(); String getCurrentDatabase() const; String getCurrentQueryId() const { return client_info.current_query_id; } diff --git a/src/Interpreters/InterpreterSelectQueryAnalyzer.cpp b/src/Interpreters/InterpreterSelectQueryAnalyzer.cpp index 4f39711e5f35..d8355d065b16 100644 --- a/src/Interpreters/InterpreterSelectQueryAnalyzer.cpp +++ b/src/Interpreters/InterpreterSelectQueryAnalyzer.cpp @@ -219,9 +219,44 @@ QueryPlanPtr buildQueryPlanForAutomaticParallelReplicas( } } +/// Like `extractAllTableReferences`, but does not descend into the inner queries of views inlined +/// by the analyzer (`analyzer_inline_views`) into a query that is not itself inside a view: +/// they read their own tables, just like a view that is not inlined. +static bool isViewInnerQueryNode(const QueryTreeNodePtr & node) +{ + if (const auto * query_node = node->as()) + return query_node->getContext()->isViewInnerQuery(); + if (const auto * union_node = node->as()) + return union_node->getContext()->isViewInnerQuery(); + return false; +} + +static void extractTableReferencesOutsideViews(const QueryTreeNodePtr & node, bool outer_is_view_inner, QueryTreeNodes & result) +{ + bool is_view_inner = isViewInnerQueryNode(node); + if (is_view_inner && !outer_is_view_inner) + return; + + if (node->getNodeType() == QueryTreeNodeType::TABLE) + { + result.push_back(node); + } + else if (const auto * query_node = node->as()) + { + for (const auto & table_expression : extractTableExpressions(query_node->getJoinTreeNodeTyped(), /*add_array_join=*/ false, /*recursive=*/ false)) + extractTableReferencesOutsideViews(table_expression, is_view_inner, result); + } + else if (const auto * union_node = node->as()) + { + for (const auto & query : union_node->getQueries().getNodes()) + extractTableReferencesOutsideViews(query, is_view_inner, result); + } +} + void replaceStorageInQueryTree(QueryTreeNodePtr & query_tree, const ContextPtr & context, const StoragePtr & storage) { - auto nodes = extractAllTableReferences(query_tree); + QueryTreeNodes nodes; + extractTableReferencesOutsideViews(query_tree, isViewInnerQueryNode(query_tree), nodes); IQueryTreeNode::ReplacementMap replacement_map; for (auto & node : nodes) diff --git a/src/Processors/QueryPlan/DistributedCreateLocalPlan.cpp b/src/Processors/QueryPlan/DistributedCreateLocalPlan.cpp index 7697b0e5ed0c..e98c055c074a 100644 --- a/src/Processors/QueryPlan/DistributedCreateLocalPlan.cpp +++ b/src/Processors/QueryPlan/DistributedCreateLocalPlan.cpp @@ -29,6 +29,11 @@ std::unique_ptr createLocalPlan( auto query_plan = std::make_unique(); auto new_context = Context::createCopy(context); + /// The local shard reads the table itself, like the remote shards do: the inserted block + /// of a materialized view (whose query contains this `remote` or `Distributed` read) must not + /// replace the source table here. + new_context->clearViewSource(); + if (build_logical_plan && !default_database.empty()) new_context->setCurrentDatabase(default_database); diff --git a/src/Storages/StorageView.cpp b/src/Storages/StorageView.cpp index b98dfce1db65..1d27a13b7d6e 100644 --- a/src/Storages/StorageView.cpp +++ b/src/Storages/StorageView.cpp @@ -771,6 +771,9 @@ ContextPtr StorageView::getViewSubqueryContext(ContextPtr context, const Storage view_settings[Setting::max_result_bytes] = 0; view_settings[Setting::extremes] = false; view_context->setSettings(view_settings); + /// The inlined view body is the inner query of the view, just like in `getViewContext`: + /// e.g. it must read the table itself, not the inserted block of a materialized view. + view_context->setIsViewInnerQuery(true); return view_context; } diff --git a/tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.reference b/tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.reference new file mode 100644 index 000000000000..64bb6b746dce --- /dev/null +++ b/tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.reference @@ -0,0 +1 @@ +30 diff --git a/tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.sh b/tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.sh new file mode 100755 index 000000000000..ca4d82b589af --- /dev/null +++ b/tests/queries/0_stateless/05241_mv_insert_alter_source_column_during_insert.sh @@ -0,0 +1,43 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# ALTER of the source column type between two blocks of one INSERT. The view query must be +# resolved against the pushed block, not against the table's current metadata. + +$CLICKHOUSE_CLIENT -q " + DROP TABLE IF EXISTS src; + DROP TABLE IF EXISTS dst; + DROP TABLE IF EXISTS mv; + CREATE TABLE src (arr Array(Tuple(a UInt32))) ENGINE = Null; + CREATE TABLE dst (cnt UInt64) ENGINE = MergeTree ORDER BY tuple(); + CREATE MATERIALIZED VIEW mv TO dst AS SELECT count() AS cnt FROM src WHERE arrayExists(x -> x.a >= 0, arr); +" + +# One row per block, 0.2 s apart, so the INSERT lasts about 6 s. +$CLICKHOUSE_CLIENT -q " + INSERT INTO src + SELECT [tuple(toUInt32(number))] + FROM numbers(30) + WHERE NOT ignore(sleepEachRow(0.2)) + SETTINGS max_block_size = 1, min_insert_block_size_rows = 1, min_insert_block_size_bytes = 1, max_threads = 1, max_insert_threads = 1 +" & + +# Wait until the first block has passed through the view, so the INSERT surely started with the old type. +while [[ $($CLICKHOUSE_CLIENT -q "SELECT count() FROM dst") -lt 1 ]] +do + sleep 0.1 +done + +$CLICKHOUSE_CLIENT -q "ALTER TABLE src MODIFY COLUMN arr Array(Tuple(a UInt32, b Nullable(UInt32)))" +wait + +$CLICKHOUSE_CLIENT -q "SELECT sum(cnt) FROM dst" + +$CLICKHOUSE_CLIENT -q " + DROP TABLE mv; + DROP TABLE dst; + DROP TABLE src; +" diff --git a/tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.reference b/tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.reference new file mode 100644 index 000000000000..09e49688a648 --- /dev/null +++ b/tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.reference @@ -0,0 +1,5 @@ +3 1 +1 1 +2 1 +3 1 +3 1 1 diff --git a/tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.sql b/tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.sql new file mode 100644 index 000000000000..52aecd93f385 --- /dev/null +++ b/tests/queries/0_stateless/05291_mv_ordinary_view_over_source_reads_table.sql @@ -0,0 +1,51 @@ +-- The query of a materialized view sees only the inserted block of its source table, +-- but an ordinary view over the same table referenced from that query reads the whole table. + +DROP TABLE IF EXISTS mv_scalar; +DROP TABLE IF EXISTS mv_join; +DROP TABLE IF EXISTS mv_inner; +DROP TABLE IF EXISTS dst_scalar; +DROP TABLE IF EXISTS dst_join; +DROP TABLE IF EXISTS dst_inner; +DROP VIEW IF EXISTS v; +DROP VIEW IF EXISTS v_in; +DROP VIEW IF EXISTS v_scalar; +DROP TABLE IF EXISTS src; + +CREATE TABLE src (x UInt64) ENGINE = MergeTree ORDER BY x; +CREATE VIEW v AS SELECT x FROM src; +-- The subqueries inside an ordinary view read the table as well. +CREATE VIEW v_in AS SELECT count() AS c FROM numbers(200) WHERE number IN (SELECT x FROM src); +CREATE VIEW v_scalar AS SELECT (SELECT count() FROM src) AS c; +-- Rows inserted before the materialized views exist: the view sees them, the inserted block does not contain them. +INSERT INTO src SELECT number + 100 FROM numbers(10); + +CREATE TABLE dst_scalar (block_rows UInt64, view_rows UInt64) ENGINE = MergeTree ORDER BY tuple(); +CREATE MATERIALIZED VIEW mv_scalar TO dst_scalar AS + SELECT count() AS block_rows, (SELECT count() FROM v) AS view_rows FROM src; + +CREATE TABLE dst_join (x UInt64, view_rows UInt64) ENGINE = MergeTree ORDER BY x; +CREATE MATERIALIZED VIEW mv_join TO dst_join AS + SELECT s.x AS x, w.c AS view_rows FROM src AS s CROSS JOIN (SELECT count() AS c FROM v) AS w; + +CREATE TABLE dst_inner (block_rows UInt64, in_rows Nullable(UInt64), scalar_rows Nullable(UInt64)) ENGINE = MergeTree ORDER BY tuple(); +CREATE MATERIALIZED VIEW mv_inner TO dst_inner AS + SELECT count() AS block_rows, (SELECT c FROM v_in) AS in_rows, (SELECT c FROM v_scalar) AS scalar_rows FROM src; + +INSERT INTO src VALUES (1), (2), (3); + +-- Whether the inserted block itself is already visible through the view depends on when the part is committed. +SELECT block_rows, view_rows >= 10 FROM dst_scalar; +SELECT x, view_rows >= 10 FROM dst_join ORDER BY x; +SELECT block_rows, in_rows >= 10, scalar_rows >= 10 FROM dst_inner; + +DROP TABLE mv_scalar; +DROP TABLE mv_join; +DROP TABLE mv_inner; +DROP TABLE dst_scalar; +DROP TABLE dst_join; +DROP TABLE dst_inner; +DROP VIEW v; +DROP VIEW v_in; +DROP VIEW v_scalar; +DROP TABLE src; diff --git a/tests/queries/0_stateless/05292_mv_source_substitution_scope.reference b/tests/queries/0_stateless/05292_mv_source_substitution_scope.reference new file mode 100644 index 000000000000..bf0a7a17cc33 --- /dev/null +++ b/tests/queries/0_stateless/05292_mv_source_substitution_scope.reference @@ -0,0 +1,16 @@ +view in scalar subquery +1 2 +2 2 +3 3 +view in join +1 2 +2 2 +3 3 +source table in subquery +1 2 +2 2 +3 1 +source table in IN +1 2 +2 2 +3 1 diff --git a/tests/queries/0_stateless/05292_mv_source_substitution_scope.sql b/tests/queries/0_stateless/05292_mv_source_substitution_scope.sql new file mode 100644 index 000000000000..81523cd807b8 --- /dev/null +++ b/tests/queries/0_stateless/05292_mv_source_substitution_scope.sql @@ -0,0 +1,51 @@ +-- The pushed block stands in for the source table in the view query, including its subqueries. +-- A view over the source table keeps reading the whole table. + +DROP TABLE IF EXISTS src; +DROP TABLE IF EXISTS v; +DROP TABLE IF EXISTS dst_scalar; +DROP TABLE IF EXISTS dst_join; +DROP TABLE IF EXISTS dst_subquery; +DROP TABLE IF EXISTS dst_in; +DROP TABLE IF EXISTS mv_scalar; +DROP TABLE IF EXISTS mv_join; +DROP TABLE IF EXISTS mv_subquery; +DROP TABLE IF EXISTS mv_in; + +CREATE TABLE src (k UInt32) ENGINE = MergeTree ORDER BY k; +CREATE VIEW v AS SELECT count() AS c FROM src; + +CREATE TABLE dst_scalar (k UInt32, c UInt64) ENGINE = MergeTree ORDER BY k; +CREATE MATERIALIZED VIEW mv_scalar TO dst_scalar AS SELECT k, (SELECT c FROM v) AS c FROM src; + +CREATE TABLE dst_join (k UInt32, c UInt64) ENGINE = MergeTree ORDER BY k; +CREATE MATERIALIZED VIEW mv_join TO dst_join AS SELECT k, c FROM src CROSS JOIN v; + +CREATE TABLE dst_subquery (k UInt32, c UInt64) ENGINE = MergeTree ORDER BY k; +CREATE MATERIALIZED VIEW mv_subquery TO dst_subquery AS SELECT k, c FROM src CROSS JOIN (SELECT count() AS c FROM src) AS t; + +CREATE TABLE dst_in (k UInt32, c UInt64) ENGINE = MergeTree ORDER BY k; +CREATE MATERIALIZED VIEW mv_in TO dst_in AS SELECT k, c FROM src CROSS JOIN (SELECT count() AS c FROM numbers(10) WHERE number IN (SELECT k FROM src)) AS t; + +INSERT INTO src VALUES (1), (2); +INSERT INTO src VALUES (3); + +SELECT 'view in scalar subquery'; +SELECT * FROM dst_scalar ORDER BY k; +SELECT 'view in join'; +SELECT * FROM dst_join ORDER BY k; +SELECT 'source table in subquery'; +SELECT * FROM dst_subquery ORDER BY k; +SELECT 'source table in IN'; +SELECT * FROM dst_in ORDER BY k; + +DROP TABLE mv_scalar; +DROP TABLE mv_join; +DROP TABLE mv_subquery; +DROP TABLE mv_in; +DROP TABLE dst_scalar; +DROP TABLE dst_join; +DROP TABLE dst_subquery; +DROP TABLE dst_in; +DROP TABLE v; +DROP TABLE src; diff --git a/tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.reference b/tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.reference new file mode 100644 index 000000000000..f551dab923cb --- /dev/null +++ b/tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.reference @@ -0,0 +1,4 @@ +3 1 +1 1 +2 1 +3 1 diff --git a/tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.sql b/tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.sql new file mode 100644 index 000000000000..0641478a90ae --- /dev/null +++ b/tests/queries/0_stateless/05293_mv_ordinary_view_over_source_inlined.sql @@ -0,0 +1,37 @@ +-- An ordinary view over the source table of a materialized view reads the whole table +-- also when the analyzer inlines the view body into the view query. + +SET analyzer_inline_views = 1; + +DROP TABLE IF EXISTS mv_scalar; +DROP TABLE IF EXISTS mv_join; +DROP TABLE IF EXISTS dst_scalar; +DROP TABLE IF EXISTS dst_join; +DROP VIEW IF EXISTS v; +DROP TABLE IF EXISTS src; + +CREATE TABLE src (x UInt64) ENGINE = MergeTree ORDER BY x; +CREATE VIEW v AS SELECT x FROM src; +-- Rows inserted before the materialized views exist: the view sees them, the inserted block does not contain them. +INSERT INTO src SELECT number + 100 FROM numbers(10); + +CREATE TABLE dst_scalar (block_rows UInt64, view_rows UInt64) ENGINE = MergeTree ORDER BY tuple(); +CREATE MATERIALIZED VIEW mv_scalar TO dst_scalar AS + SELECT count() AS block_rows, (SELECT count() FROM v) AS view_rows FROM src; + +CREATE TABLE dst_join (x UInt64, view_rows UInt64) ENGINE = MergeTree ORDER BY x; +CREATE MATERIALIZED VIEW mv_join TO dst_join AS + SELECT s.x AS x, w.c AS view_rows FROM src AS s CROSS JOIN (SELECT count() AS c FROM v) AS w; + +INSERT INTO src VALUES (1), (2), (3); + +-- Whether the inserted block itself is already visible through the view depends on when the part is committed. +SELECT block_rows, view_rows >= 10 FROM dst_scalar; +SELECT x, view_rows >= 10 FROM dst_join ORDER BY x; + +DROP TABLE mv_scalar; +DROP TABLE mv_join; +DROP TABLE dst_scalar; +DROP TABLE dst_join; +DROP VIEW v; +DROP TABLE src; diff --git a/tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.reference b/tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.reference new file mode 100644 index 000000000000..ae2506b46c84 --- /dev/null +++ b/tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.reference @@ -0,0 +1,2 @@ +distributed 3 10 +remote 3 10 diff --git a/tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.sh b/tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.sh new file mode 100755 index 000000000000..c1e513627cac --- /dev/null +++ b/tests/queries/0_stateless/05294_mv_source_remote_local_shard_reads_table.sh @@ -0,0 +1,45 @@ +#!/usr/bin/env bash + +# The local shard of a `remote` or `Distributed` read inside a materialized view query +# reads the source table itself, not the inserted block, like the remote shards do. +# The database name is spelled out: `currentDatabase()` in the view query does not resolve +# to the test database while the block is pushed to the view. + +CURDIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CURDIR"/../shell_config.sh + +$CLICKHOUSE_CLIENT -m -q " +DROP TABLE IF EXISTS mv; +DROP TABLE IF EXISTS mv_dist; +DROP TABLE IF EXISTS dst; +DROP TABLE IF EXISTS dist; +DROP TABLE IF EXISTS src; + +CREATE TABLE src (x UInt64) ENGINE = MergeTree ORDER BY x; +INSERT INTO src SELECT number + 100 FROM numbers(10); + +CREATE TABLE dist AS src ENGINE = Distributed(test_shard_localhost, ${CLICKHOUSE_DATABASE}, src); + +CREATE TABLE dst (name String, block_rows UInt64, table_rows UInt64) ENGINE = MergeTree ORDER BY name; + +CREATE MATERIALIZED VIEW mv TO dst AS + SELECT 'remote' AS name, count() AS block_rows, + (SELECT count() FROM remote('127.0.0.1', ${CLICKHOUSE_DATABASE}, src) WHERE x >= 100) AS table_rows + FROM src; + +CREATE MATERIALIZED VIEW mv_dist TO dst AS + SELECT 'distributed' AS name, count() AS block_rows, + (SELECT count() FROM ${CLICKHOUSE_DATABASE}.dist WHERE x >= 100) AS table_rows + FROM src; + +INSERT INTO src VALUES (1), (2), (3); + +SELECT * FROM dst ORDER BY name; + +DROP TABLE mv; +DROP TABLE mv_dist; +DROP TABLE dst; +DROP TABLE dist; +DROP TABLE src; +" From 306e3a25edb2a2693ea8aebe1d1f1428aae025c7 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 12:54:46 +0000 Subject: [PATCH 097/185] Backport #122673 to 26.8: Hide HTTP headers of `HTTP` dictionary sources --- src/Dictionaries/HTTPDictionarySource.cpp | 2 + .../ASTFunctionWithKeyValueArguments.cpp | 12 +++- ...tp_dictionary_hide_header_values.reference | 9 +++ ...141_http_dictionary_hide_header_values.sql | 68 +++++++++++++++++++ ...ctionary_hide_header_values_json.reference | 3 + ...http_dictionary_hide_header_values_json.sh | 46 +++++++++++++ 6 files changed, 137 insertions(+), 3 deletions(-) create mode 100644 tests/queries/0_stateless/05141_http_dictionary_hide_header_values.reference create mode 100644 tests/queries/0_stateless/05141_http_dictionary_hide_header_values.sql create mode 100644 tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.reference create mode 100755 tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.sh diff --git a/src/Dictionaries/HTTPDictionarySource.cpp b/src/Dictionaries/HTTPDictionarySource.cpp index 3e70047a8906..6d7970523bef 100644 --- a/src/Dictionaries/HTTPDictionarySource.cpp +++ b/src/Dictionaries/HTTPDictionarySource.cpp @@ -413,6 +413,8 @@ Setting fields: | `value` | Value set for a specific identifier name. | When creating a dictionary using the DDL command (`CREATE DICTIONARY ...`) remote hosts for HTTP dictionaries are checked against the contents of `remote_url_allow_hosts` section from config to prevent database users to access arbitrary HTTP server. + +The headers, including their names, are shown as `HEADERS ('[HIDDEN]')` in the output of `SHOW CREATE DICTIONARY`, in `system.tables` and in the query logs, the same way as the password. As with the password, a query that cannot be parsed is logged as is, with only [`query_masking_rules`](/reference/settings/server-settings/settings/query#query_masking_rules) applied. To display the headers in `SHOW CREATE DICTIONARY` and `system.tables`, enable the server setting [`display_secrets_in_show_and_select`](/reference/settings/server-settings/settings/other#display_secrets_in_show_and_select) and the format setting [`format_display_secrets_in_show_and_select`](/reference/settings/formats/format#format_display_secrets_in_show_and_select); the user also needs the `displaySecretsInShowAndSelect` privilege. These settings do not affect the query logs. )DOCS_MD", .syntax = "SOURCE(HTTP(url 'https://host/path' format 'CSV'))", .related = {"file"}}); diff --git a/src/Parsers/ASTFunctionWithKeyValueArguments.cpp b/src/Parsers/ASTFunctionWithKeyValueArguments.cpp index ea1b63f82b41..17156bf476ef 100644 --- a/src/Parsers/ASTFunctionWithKeyValueArguments.cpp +++ b/src/Parsers/ASTFunctionWithKeyValueArguments.cpp @@ -20,12 +20,16 @@ namespace { /// Keys of a dictionary source whose value must not be shown. Besides the password, this covers /// the TLS credentials that are given as the contents of a certificate or a key file (a path is - /// not accepted from a `CREATE DICTIONARY` query in the first place). + /// not accepted from a `CREATE DICTIONARY` query in the first place), and the custom HTTP headers + /// of the `HTTP` source, whose values often carry API tokens. The headers are hidden as a whole, + /// names included: the query is logged before the dictionary source validates its structure, + /// so a malformed definition must not leak either. bool isSecretKey(const String & key) { return key == "password" || key == "ssl_ca_pem" || key == "ssl_cert_pem" || key == "ssl_key_pem" - || key == "sslrootcert_pem" || key == "sslcert_pem" || key == "sslkey_pem"; + || key == "sslrootcert_pem" || key == "sslcert_pem" || key == "sslkey_pem" + || key == "headers" || key == "header"; } } @@ -56,7 +60,9 @@ void ASTPair::readJSON(const Poco::JSON::Object & json) { JSONObjectReader r(json); - first = r.getString("first"); + /// The SQL parser lower-cases the key (see `ParserKeyValuePair`), and the checks for secret keys in + /// `formatImpl` and `hasSecretParts` rely on it, so canonicalize it the same way here. + first = Poco::toLower(r.getString("first")); if (first.empty()) throw Exception(ErrorCodes::BAD_ARGUMENTS, "Missing or empty 'first' in ASTPair during AST JSON deserialization"); diff --git a/tests/queries/0_stateless/05141_http_dictionary_hide_header_values.reference b/tests/queries/0_stateless/05141_http_dictionary_hide_header_values.reference new file mode 100644 index 000000000000..ca36e0892c1a --- /dev/null +++ b/tests/queries/0_stateless/05141_http_dictionary_hide_header_values.reference @@ -0,0 +1,9 @@ +CREATE DICTIONARY default.d_05141\n(\n `id` UInt64,\n `v` String\n)\nPRIMARY KEY id\nSOURCE(HTTP(URL \'http://localhost:11111/x.tsv\' FORMAT \'TabSeparated\' CREDENTIALS (USER \'user\' PASSWORD \'[HIDDEN]\') HEADERS (\'[HIDDEN]\')))\nLIFETIME(MIN 0 MAX 0)\nLAYOUT(FLAT()) +0 0 1 +d_05141_flat HEADERS \'[HIDDEN]\')) LIFETIME(MIN 0 MAX 0) LAYOUT(FLAT()) +d_05141_foo HEADERS (\'[HIDDEN]\'))) LIFETIME(MIN 0 MAX 0) LAYOUT(FLAT()) +d_05141_key HEADERS (\'[HIDDEN]\'))) LIFETIME(MIN 0 MAX 0) LAYOUT(FLAT()) +d_05141_nested HEADERS (\'[HIDDEN]\'))) LIFETIME(MIN 0 MAX 0) LAYOUT(FLAT()) +d_05141_nobr HEADERS (\'[HIDDEN]\'))) LIFETIME(MIN 0 MAX 0) LAYOUT(FLAT()) +d_05141_typo HEADERS (\'[HIDDEN]\'))) LIFETIME(MIN 0 MAX 0) LAYOUT(FLAT()) +2 14 0 diff --git a/tests/queries/0_stateless/05141_http_dictionary_hide_header_values.sql b/tests/queries/0_stateless/05141_http_dictionary_hide_header_values.sql new file mode 100644 index 000000000000..0e1468fdee93 --- /dev/null +++ b/tests/queries/0_stateless/05141_http_dictionary_hide_header_values.sql @@ -0,0 +1,68 @@ +-- The values of custom HTTP headers of an `HTTP` dictionary source often carry credentials, +-- so they must be hidden in `SHOW CREATE DICTIONARY`, `system.tables` and `system.query_log`, +-- the same way as the password. They are hidden as a whole, header names included. + +SET format_display_secrets_in_show_and_select = 0; + +DROP DICTIONARY IF EXISTS d_05141; +CREATE DICTIONARY d_05141 (id UInt64, v String) +PRIMARY KEY id +SOURCE(HTTP( + url 'http://localhost:11111/x.tsv' + format 'TabSeparated' + credentials(user 'user' password 'SEKRIT_PW') + headers( + header(name 'API-KEY' value 'SEKRIT_TOKEN_1') + header(name 'X-Other' value 'SEKRIT_TOKEN_2') + ) +)) +LIFETIME(0) LAYOUT(FLAT()); + +SHOW CREATE DICTIONARY d_05141; + +SELECT create_table_query LIKE concat('%', 'SEKRIT', '%'), create_table_query LIKE '%API-KEY%', create_table_query LIKE '%HEADERS (\'[HIDDEN]\')%' +FROM system.tables WHERE database = currentDatabase() AND name = 'd_05141'; + +DROP DICTIONARY d_05141; + +-- The query is logged before the dictionary source validates its structure, so malformed header +-- definitions must not leak either. +CREATE DICTIONARY d_05141_typo (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header(name 'API-KEY' vaule 'SEKRIT_TYPO')))) +LIFETIME(0) LAYOUT(FLAT()); +CREATE DICTIONARY d_05141_key (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header(secret 'SEKRIT_KEY')))) +LIFETIME(0) LAYOUT(FLAT()); +CREATE DICTIONARY d_05141_nested (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header(name(foo 'SEKRIT_NESTED'))))) +LIFETIME(0) LAYOUT(FLAT()); +CREATE DICTIONARY d_05141_func (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header(name concat('X-', 'SEKRIT_FUNC') value 'SEKRIT_FUNC_VALUE')))) +LIFETIME(0) LAYOUT(FLAT()); -- { serverError INCORRECT_DICTIONARY_DEFINITION } +CREATE DICTIONARY d_05141_array (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header(name ['SEKRIT_ARRAY'])))) +LIFETIME(0) LAYOUT(FLAT()); -- { serverError BAD_ARGUMENTS } +CREATE DICTIONARY d_05141_nobr (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header 'SEKRIT_NOBR'))) +LIFETIME(0) LAYOUT(FLAT()); +CREATE DICTIONARY d_05141_foo (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(foo 'SEKRIT_FOO'))) +LIFETIME(0) LAYOUT(FLAT()); +CREATE DICTIONARY d_05141_flat (id UInt64, v String) PRIMARY KEY id +SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers 'SEKRIT_HEADERS')) +LIFETIME(0) LAYOUT(FLAT()); + +SELECT name, extract(create_table_query, 'HEADERS.*$') +FROM system.tables WHERE database = currentDatabase() AND name LIKE 'd\\_05141\\_%' ORDER BY name; + +SYSTEM FLUSH LOGS query_log; +SELECT countIf(query LIKE '%d\\_05141 %'), countIf(query LIKE '%d\\_05141\\_%'), countIf(query LIKE concat('%', 'SEKRIT', '%')) +FROM system.query_log +WHERE current_database = currentDatabase() AND query_kind = 'Create' AND event_date >= yesterday(); + +DROP DICTIONARY d_05141_typo; +DROP DICTIONARY d_05141_key; +DROP DICTIONARY d_05141_nested; +DROP DICTIONARY d_05141_nobr; +DROP DICTIONARY d_05141_foo; +DROP DICTIONARY d_05141_flat; diff --git a/tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.reference b/tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.reference new file mode 100644 index 000000000000..f7c7e2e15325 --- /dev/null +++ b/tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.reference @@ -0,0 +1,3 @@ +HEADERS (\'[HIDDEN]\'))) +CREDENTIALS (USER \'user\' PASSWORD \'[HIDDEN]\') HEADERS (\'[HIDDEN]\'))) +1 0 diff --git a/tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.sh b/tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.sh new file mode 100755 index 000000000000..11eae7fc24b4 --- /dev/null +++ b/tests/queries/0_stateless/05142_http_dictionary_hide_header_values_json.sh @@ -0,0 +1,46 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + + +# The HTTP headers of an `HTTP` dictionary source must also be hidden when the query is submitted +# as a JSON AST (`dialect = 'clickhouse_json'`). + +CLICKHOUSE_CLIENT_JSON="${CLICKHOUSE_CLIENT} --enable_json_ast_dialect 1 --dialect clickhouse_json" + +function to_json() +{ + ${CLICKHOUSE_CLIENT} --query "SELECT parseQueryToJSON(\$\$$1\$\$) FORMAT TSVRaw" +} + +# A definition submitted as a JSON AST is masked like the SQL one. +JSON=$(to_json "CREATE DICTIONARY ${CLICKHOUSE_DATABASE}.d_05142 (id UInt64, v String) PRIMARY KEY id + SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' headers(header(name 'API-KEY' value 'SEKRIT_JSON')))) + LIFETIME(0) LAYOUT(FLAT())") +${CLICKHOUSE_CLIENT_JSON} --query "$JSON" +${CLICKHOUSE_CLIENT} --query "SELECT extract(create_table_query, 'HEADERS.*\\)\\)\\)') FROM system.tables WHERE database = currentDatabase() AND name = 'd_05142' SETTINGS format_display_secrets_in_show_and_select = 0" + +# The SQL parser lower-cases the keys, but a JSON AST can spell them in any case; the secret keys must +# be recognized anyway. +JSON=$(to_json "CREATE DICTIONARY ${CLICKHOUSE_DATABASE}.d_05142_case (id UInt64, v String) PRIMARY KEY id + SOURCE(HTTP(url 'http://localhost:11111/x.tsv' format 'TabSeparated' credentials(user 'user' password 'SEKRIT_JSON_CASE_PW') + headers(header(name 'API-KEY' value 'SEKRIT_JSON_CASE')))) + LIFETIME(0) LAYOUT(FLAT())") +for key in headers header value password +do + JSON=${JSON//\"first\":\"$key\"/\"first\":\"${key^^}\"} +done +${CLICKHOUSE_CLIENT_JSON} --query "$JSON" +${CLICKHOUSE_CLIENT} --query "SELECT extract(create_table_query, 'CREDENTIALS.*\\)\\)\\)') FROM system.tables WHERE database = currentDatabase() AND name = 'd_05142_case' SETTINGS format_display_secrets_in_show_and_select = 0" + +# The JSON queries are logged without the headers and the password. +${CLICKHOUSE_CLIENT} --query "SYSTEM FLUSH LOGS query_log" +${CLICKHOUSE_CLIENT} --query " + SELECT count() > 0, countIf(query LIKE concat('%', 'SEKRIT', '%')) + FROM system.query_log + WHERE current_database = currentDatabase() AND query_kind = 'Create' AND query LIKE '%d_05142%' AND event_date >= yesterday()" + +${CLICKHOUSE_CLIENT} --query "DROP DICTIONARY d_05142" +${CLICKHOUSE_CLIENT} --query "DROP DICTIONARY d_05142_case" From e02ecceb99a30785413f5a74251acaae78196e84 Mon Sep 17 00:00:00 2001 From: vdimir Date: Thu, 1 Oct 2026 14:02:18 +0000 Subject: [PATCH 098/185] Keep only the legacy analyzer part of #121398 A right table with a row policy is joined through its query plan instead of its special storage, so the `FilterStep` carrying the policy is kept. The `ACCESS_DENIED` for `JOIN` and `joinGet` on a `Join` table and its docs are not backported. This also drops the `registerStatementRowPolicy` block that the conflict resolution pulled into the parser: `StatementFactory` does not exist in 26.8. Co-Authored-By: Claude Fable 5.1 --- src/Functions/FunctionJoinGet.cpp | 12 +- src/Interpreters/JoinedTables.cpp | 20 +- .../Access/ParserCreateRowPolicyQuery.cpp | 302 ------------------ src/Planner/PlannerJoinTree.cpp | 20 +- src/Storages/StorageJoin.cpp | 2 - .../05233_direct_join_row_policy.reference | 24 +- .../05233_direct_join_row_policy.sql | 40 +-- .../05234_row_policy_storage_join.reference | 12 - .../05234_row_policy_storage_join.sql | 31 -- 9 files changed, 33 insertions(+), 430 deletions(-) delete mode 100644 tests/queries/0_stateless/05234_row_policy_storage_join.reference delete mode 100644 tests/queries/0_stateless/05234_row_policy_storage_join.sql diff --git a/src/Functions/FunctionJoinGet.cpp b/src/Functions/FunctionJoinGet.cpp index 3c71c00821c1..e3d7b87f97d8 100644 --- a/src/Functions/FunctionJoinGet.cpp +++ b/src/Functions/FunctionJoinGet.cpp @@ -11,8 +11,6 @@ #include #include #include -#include -#include namespace DB { @@ -23,7 +21,6 @@ namespace Setting namespace ErrorCodes { - extern const int ACCESS_DENIED; extern const int ILLEGAL_TYPE_OF_ARGUMENT; extern const int NUMBER_OF_ARGUMENTS_DOESNT_MATCH; } @@ -149,14 +146,7 @@ ExecutableFunctionPtr FunctionJoinGet::prepare(const ColumnsWithTypeAndName &) c Names column_names = storage_join->getKeyNames(); column_names.push_back(attr_name); - const auto storage_id = storage_join->getStorageID(); - context->checkAccess(AccessType::SELECT, storage_id, column_names); - - /// The hash table is read as is, so a row policy on the table cannot be applied here any more than in a JOIN. - auto row_policy_filter = context->getRowPolicyFilter(storage_id.getDatabaseName(), storage_id.getTableName(), RowPolicyFilterType::SELECT_FILTER); - if (row_policy_filter && !row_policy_filter->isAlwaysTrue()) - throw Exception(ErrorCodes::ACCESS_DENIED, - "Cannot use {} because a row policy is applied on table {} with the Join engine", function_name, storage_id.getNameForLogs()); + context->checkAccess(AccessType::SELECT, storage_join->getStorageID(), column_names); return std::make_unique(function_name, context, table_lock, storage_join, result_columns); } diff --git a/src/Interpreters/JoinedTables.cpp b/src/Interpreters/JoinedTables.cpp index c64b2735268a..b1abb5e4a478 100644 --- a/src/Interpreters/JoinedTables.cpp +++ b/src/Interpreters/JoinedTables.cpp @@ -42,7 +42,6 @@ namespace Setting namespace ErrorCodes { - extern const int ACCESS_DENIED; extern const int ALIAS_REQUIRED; extern const int AMBIGUOUS_COLUMN_NAME; extern const int LOGICAL_ERROR; @@ -349,20 +348,11 @@ std::shared_ptr JoinedTables::makeTableJoin(const ASTSelectQuery & se StoragePtr storage = DatabaseCatalog::instance().tryGetTable(joined_table_id, context); /// A special storage replaces the right-side plan, and with it the `FilterStep` carrying the - /// table's row policy, so such a table has to be joined as an ordinary stream. A `Join` table - /// is a prebuilt hash table read as is, so it cannot be filtered at all. - if (storage) - { - auto row_policy_filter = context->getRowPolicyFilter( - joined_table_id.getDatabaseName(), joined_table_id.getTableName(), RowPolicyFilterType::SELECT_FILTER); - if (row_policy_filter && !row_policy_filter->isAlwaysTrue()) - { - if (typeid_cast(storage.get())) - throw Exception(ErrorCodes::ACCESS_DENIED, - "Cannot join table {} with the Join engine because a row policy is applied on it", joined_table_id.getNameForLogs()); - storage = nullptr; - } - } + /// table's row policy, so such a table has to be joined as an ordinary stream. + auto joined_table_row_policy = context->getRowPolicyFilter( + joined_table_id.getDatabaseName(), joined_table_id.getTableName(), RowPolicyFilterType::SELECT_FILTER); + if (joined_table_row_policy && !joined_table_row_policy->isAlwaysTrue()) + storage = nullptr; if (storage) { diff --git a/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp b/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp index e6e8bea7217d..63dcf9769961 100644 --- a/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp +++ b/src/Parsers/Access/ParserCreateRowPolicyQuery.cpp @@ -315,305 +315,3 @@ bool ParserCreateRowPolicyQuery::parseImpl(Pos & pos, ASTPtr & node, Expected & return true; } } - -namespace DB -{ - -void registerStatementRowPolicy(StatementFactory & factory) -{ - factory.registerStatement("CREATE ROW POLICY", - { - .description = R"DOCS_MD( -Creates a [row policy](/concepts/features/security/access-rights#row-policy-management), i.e. a filter used to determine which rows a user can read from a table. - - -Row policies make sense only for users with readonly access. If a user can modify a table or copy partitions between tables, it defeats the restrictions of row policies. - - -Syntax: - -```sql --- Multiple names on one table target -CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] policy_name [, ...] - [ON CLUSTER cluster_name] - ON { [db.]table | db.* } - [IN access_storage_type] - [[FOR SELECT] USING {condition | NONE}] - [AS {PERMISSIVE | RESTRICTIVE}] - [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] - --- One name on multiple table targets -CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] policy_name - [ON CLUSTER cluster_name] - ON { [db.]table | db.* } [, ...] - [IN access_storage_type] - [[FOR SELECT] USING {condition | NONE}] - [AS {PERMISSIVE | RESTRICTIVE}] - [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] - --- Mixed packing: each name paired with its own table target -CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] - policy_name ON { [db.]table | db.* } [, policy_name ON { [db.]table | db.* } ...] - [ON CLUSTER cluster_name] - [IN access_storage_type] - [[FOR SELECT] USING {condition | NONE}] - [AS {PERMISSIVE | RESTRICTIVE}] - [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] -``` - -`ParserRowPolicyNames` accepts **three** packing forms (not a full Cartesian product): - -1. **Multiple names, one target** — `pol1, pol2 ON table1` creates each listed name on that single table (or `db.*`). -2. **One name, multiple targets** — `pol1 ON table1, table2` creates the same short name on each listed target. -3. **Mixed pairs** — `p1 ON t1, p2 ON t2` creates each name only on its paired target. - -A multi-name list **cannot** be combined with a multi-table `ON` list in one group: `p1, p2 ON t1, t2` is rejected. After a multi-name group, you also cannot append another comma-separated `name ON target` group in the same statement. - -Optional `ON CLUSTER` applies to the whole statement (one cluster name). ClickHouse does **not** accept a different `ON CLUSTER` per policy name packed into a single create — run separate `CREATE ROW POLICY` statements when policies must be created on different clusters. - -`CREATE ROW POLICY` requires the [CREATE ROW POLICY](/reference/statements/grant#access-management) privilege on the table the policy is created on. `OR REPLACE` throws away an existing policy of the same name, including which roles it applies to, so it additionally requires the [DROP ROW POLICY](/reference/statements/grant#access-management) privilege on that table. The `DROP ROW POLICY` privilege is required whether or not the policy already exists, so the statement cannot be used to find out which policies exist. - -## Multiple names and tables {#multiple-names-and-tables} - -Valid: - -```sql --- Several policy names, one table -CREATE ROW POLICY pol1, pol2, pol3 ON table1 - FOR SELECT USING id = 1 - TO accountant; - --- One policy name, several tables -CREATE ROW POLICY IF NOT EXISTS pol1 ON table1, table2, table3 - FOR SELECT USING id = 1 - TO accountant; - --- Mixed packing: different name per table -CREATE ROW POLICY p4 ON db.table, p5 ON db2.table2 - USING a = b; - --- Same policy on several tables, on a cluster -CREATE ROW POLICY IF NOT EXISTS pol1 ON CLUSTER replicated_cluster ON table1, table2 - FOR SELECT USING id = 1 - TO accountant; -``` - -Invalid: - -```sql --- Multi-name × multi-table in one ON-group (not a Cartesian product) -CREATE ROW POLICY p1, p2 ON t1, t2 - FOR SELECT USING id = 1 - TO accountant; - --- Different clusters per name in one statement -CREATE ROW POLICY pol1 ON CLUSTER cluster1 ON table1, pol2 ON CLUSTER cluster2 ON table2 -``` - -## USING clause {#using-clause} - -Defines a filter condition for a table. A user can only see rows for which the condition is true (evaluates to a non-zero value). This is similar to adding an extra `WHERE` condition to every query the user runs against the table. - -For example, the following policy limits `analyst_role` to rows from the EU: - -```sql -CREATE ROW POLICY region_filter ON db.orders -USING region = 'EU' -TO analyst_role; -``` - -With this policy, `SELECT * FROM db.orders` returns the same rows as `SELECT * FROM db.orders WHERE region = 'EU'` would. - -## TO Clause {#to-clause} - -In the `TO` section you can provide a list of users and roles this policy should work for. For example, `CREATE ROW POLICY ... TO accountant, john@localhost`. - -Keyword `ALL` means all the ClickHouse users, including current user. Keyword `ALL EXCEPT` allows excluding some users from the all users list, for example, `CREATE ROW POLICY ... TO ALL EXCEPT accountant, john@localhost` - -Roles named in the `TO` section, including those after `ALL EXCEPT`, are matched against the current user's enabled roles ([`system.enabled_roles`](/reference/system-tables/enabled_roles)), not against every role granted to the user, so [`SET ROLE`](/reference/statements/set-role) can change which policies apply. - -## AS Clause {#as-clause} - -It's allowed to have more than one policy enabled on the same table for the same user at one time. So we need a way to combine the conditions from multiple policies. - -By default, policies are combined using the boolean `OR` operator. For example, the following policies: - -```sql -CREATE ROW POLICY pol1 ON mydb.table1 USING b=1 TO mira, peter -CREATE ROW POLICY pol2 ON mydb.table1 USING c=2 TO peter, antonio -``` - -enable the user `peter` to see rows with either `b=1` or `c=2`. - -The `AS` clause specifies how policies should be combined with other policies. Policies can be either permissive or restrictive. By default, policies are permissive, which means they are combined using the boolean `OR` operator. - -A policy can be defined as restrictive as an alternative. Restrictive policies are combined using the boolean `AND` operator. - -Here is the general formula: - -```text -row_is_visible = (one or more of the conditions from the permissive policies that apply to the current user and their enabled roles are non-zero) AND - (all of the conditions from the restrictive policies that apply to the current user and their enabled roles are non-zero) -``` - -If no permissive condition applies, the first condition has no effect and only the restrictive policies decide, because `access_control_improvements.users_without_row_policies_can_read_rows` is enabled by default. A user to whom no condition applies therefore sees every row, and `access_control_improvements.throw_on_unmatched_row_policies`, disabled by default, raises an exception instead when the table does have conditions and none of them apply. - -For example, the following policies: - -```sql -CREATE ROW POLICY pol1 ON mydb.table1 USING b=1 TO mira, peter -CREATE ROW POLICY pol2 ON mydb.table1 USING c=2 AS RESTRICTIVE TO peter, antonio -``` - -enable the user `peter` to see rows only if both `b=1` AND `c=2`. - -Database policies are combined with table policies. - -For example, the following policies: - -```sql -CREATE ROW POLICY pol1 ON mydb.* USING b=1 TO mira, peter -CREATE ROW POLICY pol2 ON mydb.table1 USING c=2 AS RESTRICTIVE TO peter, antonio -``` - -enable the user `peter` to see table1 rows only if both `b=1` AND `c=2`, although -any other table in mydb would have only `b=1` policy applied for the user. - -## Tables that read from other tables {#tables-that-read-from-other-tables} - -A row policy filters rows where the data is actually read. An `Alias` table returns the rows of its target table as its own, so the row policies of the target apply to reads through the alias as well, combined with the policies of the alias itself using a logical `AND`. A `Merge` table applies the policies of the tables it reads from. One exception: when a matched table reads remotely, such as a `Distributed` table, the remote server processes the query before the policy is applied, because the policy runs above that table's read rather than at the read. Such a query can fail, when it aggregates without selecting the policy's columns, or return fewer rows than the policy allows, when the remote server applies an `ORDER BY ... LIMIT` to rows the policy would have hidden. Define the policy on the underlying local tables of each remote server instead. - -This does not extend to every table that reads from another table. A `Buffer` table and a materialized view read through their destination or target table do **not** inherit that table's row policies: the policy is written against the target's schema and, for a view with `SQL SECURITY DEFINER`, is evaluated for a different user than the one running the read. Define the policy on the table users actually query in those cases. - -## Distributed and remote-backed tables {#distributed-and-remote-backed-tables} - -A row policy filters rows where the table data is actually read. A table that delegates reading to remote servers, such as a [Distributed](/reference/engines/table-engines/special/distributed) table or a wrapper over one (for example, a materialized view with a `Distributed` target), only ships the query text to the remote servers and cannot apply the policy filter to the remote read. To keep the filter from being silently dropped, queries to such a table by users the policy applies to are rejected with an `ILLEGAL_PREWHERE` error. - -Instead, define the policy on the underlying local tables on each remote server; it is applied there when the shipped query reads them: - -```sql --- Filters reads of local_table on this server, including reads shipped by a Distributed table over it. -CREATE ROW POLICY filter ON mydb.local_table USING a < 1000 TO john; -``` - - -This works while the query is shipped as text, which is the default. With [`serialize_query_plan = 1`](/reference/settings/session-settings/serialize#serialize_query_plan) the initiator ships an already-built read plan instead, and a remote server executing such a plan does not apply its own row policies, so a read of a `Distributed` table over `local_table` returns unfiltered rows. Keep `serialize_query_plan = 0` for users whose row policies must be enforced. See [issue #112891](https://github.com/ClickHouse/ClickHouse/issues/112891). - - -## Join tables {#join-tables} - -A [Join](/reference/engines/table-engines/special/join) table is a prepared hash table that a `JOIN` or `joinGet` reads as is, so its rows cannot be filtered there. A policy on such a table, including a database-wide `ON db.*` policy, filters a plain `SELECT` from the table, but while it applies, `JOIN` and `joinGet` queries against the table fail with `ACCESS_DENIED`. - -## ON CLUSTER Clause {#on-cluster-clause} - -Allows creating row policies on a cluster, see [Distributed DDL](/reference/statements/distributed-ddl). This is also the convenient way to create the policy on the local tables of every server of the cluster. - -## Examples {#examples} - -`CREATE ROW POLICY filter1 ON mydb.mytable USING a<1000 TO accountant, john@localhost` - -`CREATE ROW POLICY filter2 ON mydb.mytable USING a<1000 AND b=5 TO ALL EXCEPT mira` - -`CREATE ROW POLICY filter3 ON mydb.mytable USING 1 TO admin` - -`CREATE ROW POLICY filter4 ON mydb.* USING 1 TO admin` -)DOCS_MD", - .syntax = R"( -CREATE [ROW] POLICY [IF NOT EXISTS | OR REPLACE] policy_name [, ...] - [ON CLUSTER cluster_name] - ON { [db.]table | db.* } [, ...] - [IN access_storage_type] - [[FOR SELECT] USING {condition | NONE}] - [AS {PERMISSIVE | RESTRICTIVE}] - [TO {role1 [, role2 ...] | ALL | ALL EXCEPT role1 [, role2 ...]}] -)", - .parent = "CREATE", - .related = {"ALTER ROW POLICY", "CREATE MASKING POLICY", "CREATE ROLE", "DROP", "SHOW"}, - }); - - factory.registerStatement("ALTER ROW POLICY", - { - .description = R"DOCS_MD( -Changes row policy. - -Syntax: - -```sql --- Rename: exactly one fully qualified policy (one name on one target). --- RENAME TO may be combined with the same optional alteration clauses as below. -ALTER [ROW] POLICY [IF EXISTS] name - ON { [database.]table | database.* } - RENAME TO new_name - [ON CLUSTER cluster_name] - [AS {PERMISSIVE | RESTRICTIVE}] - [FOR SELECT] - [USING {condition | NONE}][,...] - [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] - --- Multiple names on one table target (no RENAME) -ALTER [ROW] POLICY [IF EXISTS] name [, ...] - [ON CLUSTER cluster_name] - ON { [database.]table | database.* } - [AS {PERMISSIVE | RESTRICTIVE}] - [FOR SELECT] - [USING {condition | NONE}][,...] - [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] - --- One name on multiple table targets (no RENAME) -ALTER [ROW] POLICY [IF EXISTS] name - [ON CLUSTER cluster_name] - ON { [database.]table | database.* } [, ...] - [AS {PERMISSIVE | RESTRICTIVE}] - [FOR SELECT] - [USING {condition | NONE}][,...] - [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] - --- Mixed packing: each name paired with its own table target (no RENAME) -ALTER [ROW] POLICY [IF EXISTS] - name ON { [database.]table | database.* } [, name ON { [database.]table | database.* } ...] - [ON CLUSTER cluster_name] - [AS {PERMISSIVE | RESTRICTIVE}] - [FOR SELECT] - [USING {condition | NONE}][,...] - [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] -``` - -`RENAME TO` is only accepted when the statement names **exactly one** policy on **one** table target. Packed multi-name or multi-table lists cannot include `RENAME TO` — rename those policies in separate statements. On that single-policy form, `RENAME TO` can still be combined with other alterations such as `AS`, `USING`, and `TO` in the same statement. - -Without `RENAME TO`, packing matches `ParserRowPolicyNames`: multiple names on **one** target, one name on **multiple** targets, or mixed `name ON target` pairs. Multi-name × multi-table (`p1, p2 ON t1, t2`) is **not** accepted. Optional `ON CLUSTER` applies once to the whole statement; run separate `ALTER ROW POLICY` statements when different clusters are required. - -Examples: - -```sql --- Single-policy rename -ALTER ROW POLICY p1 ON db.table RENAME TO p1_new; - --- Rename plus other alterations on the same single policy -ALTER POLICY old_name ON db.table RENAME TO new_name USING id > 10; - --- Multiple names, one table -ALTER POLICY p1, p2 ON db.table TO ALL; - --- One name, multiple tables -ALTER POLICY p1 ON db.table, db.table2 USING NONE; - --- Mixed targets without rename -ALTER POLICY p1 ON db.table, p2 ON db2.table2 TO ALL; -``` -)DOCS_MD", - .syntax = R"( -ALTER [ROW] POLICY [IF EXISTS] name [, ...] - ON { [database.]table | database.* } [, ...] - [RENAME TO new_name] - [ON CLUSTER cluster_name] - [AS {PERMISSIVE | RESTRICTIVE}] - [FOR SELECT] - [USING {condition | NONE}][,...] - [TO {role [,...] | ALL | ALL EXCEPT role [,...]}] -)", - .parent = "ALTER", - .related = {"CREATE ROW POLICY", "ALTER", "SHOW"}, - }); -} - -} diff --git a/src/Planner/PlannerJoinTree.cpp b/src/Planner/PlannerJoinTree.cpp index 2e0ee3dbd986..67ca7fb5945a 100644 --- a/src/Planner/PlannerJoinTree.cpp +++ b/src/Planner/PlannerJoinTree.cpp @@ -31,7 +31,6 @@ #include #include #include -#include #include #include #include @@ -185,7 +184,6 @@ namespace ErrorCodes extern const int PARAMETER_OUT_OF_BOUND; extern const int TOO_MANY_COLUMNS; extern const int UNSUPPORTED_METHOD; - extern const int ACCESS_DENIED; } namespace @@ -3123,22 +3121,12 @@ JoinTreeQueryPlan buildQueryPlanForJoinNode( join_node, planner_context); - /// A prepared storage replaces the right-side plan, and with it the `FilterStep` applying the - /// table's row policy, so such a table is joined as a stream. A `Join` table is a prebuilt - /// hash table read as is, so it cannot be filtered at all. - bool right_table_has_row_policy = !right_join_tree_query_plan.used_row_policies.empty(); - - PreparedJoinStorage prepared_join = tryGetStorageInTableJoin(join_node.getRightTableExpressionNode(), planner_context); - if (prepared_join.storage_join && right_table_has_row_policy) - throw Exception(ErrorCodes::ACCESS_DENIED, - "Cannot join table {} with the Join engine because a row policy is applied on it", - prepared_join.storage_join->getStorageID().getNameForLogs()); - - bool allow_storage_join = !right_table_has_row_policy + PreparedJoinStorage prepared_join; + bool allow_storage_join = right_join_tree_query_plan.used_row_policies.empty() && right_join_tree_query_plan.stage == QueryProcessingStage::FetchColumns && right_join_tree_query_plan.useful_sets.empty(); - if (!allow_storage_join) - prepared_join = {}; + if (allow_storage_join) + prepared_join = tryGetStorageInTableJoin(join_node.getRightTableExpressionNode(), planner_context); if (prepared_join) { bool use_nulls = settings[Setting::join_use_nulls] && isLeftOrFull(join_node.getKind()); diff --git a/src/Storages/StorageJoin.cpp b/src/Storages/StorageJoin.cpp index cdc7a5c14c2f..cbbb62afde5a 100644 --- a/src/Storages/StorageJoin.cpp +++ b/src/Storages/StorageJoin.cpp @@ -641,8 +641,6 @@ Default value: `1`. The `Join`-engine tables can't be used in `GLOBAL JOIN` operations. -A [row policy](/reference/statements/create/row-policy) on a `Join`-engine table filters a plain `SELECT` from it, but a `JOIN` or `joinGet` reads the prepared hash table as is and cannot filter its rows, so while a policy applies to the table such queries fail with `ACCESS_DENIED`. - The `Join`-engine allows to specify [join_use_nulls](/reference/settings/session-settings/join#join_use_nulls) setting in the `CREATE TABLE` statement. [SELECT](/reference/statements/select/index) query should have the same `join_use_nulls` value. ## Usage examples {#example} diff --git a/tests/queries/0_stateless/05233_direct_join_row_policy.reference b/tests/queries/0_stateless/05233_direct_join_row_policy.reference index 523709ccc754..f196664e9f2c 100644 --- a/tests/queries/0_stateless/05233_direct_join_row_policy.reference +++ b/tests/queries/0_stateless/05233_direct_join_row_policy.reference @@ -4,9 +4,6 @@ -- inner 1 public public-1 3 public public-3 --- inner, policy column not selected -1 public-1 -3 public-3 -- left 1 1 public-1 2 0 @@ -17,27 +14,22 @@ 2 \N \N 3 3 public-3 5 \N \N --- left any -1 public-1 -2 -3 public-3 -5 -- left semi 1 3 -- left anti 2 5 --- key space enumeration -1 public-1 -3 public-3 --- using +-- join table 1 public-1 +2 3 public-3 --- no direct join under a row policy -0 --- direct join without a policy +5 +-- without a policy 1 public public-1 2 hidden hidden-2 3 public public-3 -Algorithm: DirectKeyValueJoin +1 public-1 +2 hidden-2 +3 public-3 +5 diff --git a/tests/queries/0_stateless/05233_direct_join_row_policy.sql b/tests/queries/0_stateless/05233_direct_join_row_policy.sql index e2a073ca13a2..38db3f227d2d 100644 --- a/tests/queries/0_stateless/05233_direct_join_row_policy.sql +++ b/tests/queries/0_stateless/05233_direct_join_row_policy.sql @@ -1,18 +1,24 @@ -- Tags: use-rocksdb +SET enable_analyzer = 0; +SET join_algorithm = 'direct,hash'; +SET join_use_nulls = 0; + DROP TABLE IF EXISTS kv_rls; +DROP TABLE IF EXISTS join_rls; DROP TABLE IF EXISTS probe_rls; CREATE TABLE kv_rls (key UInt64, tenant String, secret String) ENGINE = EmbeddedRocksDB PRIMARY KEY key; INSERT INTO kv_rls VALUES (1, 'public', 'public-1'), (2, 'hidden', 'hidden-2'), (3, 'public', 'public-3'), (4, 'hidden', 'hidden-4'); +CREATE TABLE join_rls (key UInt64, tenant String, secret String) ENGINE = Join(ANY, LEFT, key); +INSERT INTO join_rls VALUES (1, 'public', 'public-1'), (2, 'hidden', 'hidden-2'), (3, 'public', 'public-3'), (4, 'hidden', 'hidden-4'); + CREATE TABLE probe_rls (key UInt64) ENGINE = TinyLog; INSERT INTO probe_rls VALUES (1), (2), (3), (5); CREATE ROW POLICY kv_rls_public ON kv_rls FOR SELECT USING tenant = 'public' TO CURRENT_USER; - -SET join_algorithm = 'direct,hash'; -SET join_use_nulls = 0; +CREATE ROW POLICY join_rls_public ON join_rls FOR SELECT USING tenant = 'public' TO CURRENT_USER; SELECT '-- plain select'; SELECT key, tenant, secret FROM kv_rls ORDER BY key; @@ -20,44 +26,28 @@ SELECT key, tenant, secret FROM kv_rls ORDER BY key; SELECT '-- inner'; SELECT p.key, kv.tenant, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; -SELECT '-- inner, policy column not selected'; -SELECT p.key, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; - SELECT '-- left'; SELECT p.key, kv.key, kv.secret FROM probe_rls AS p LEFT JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; SELECT '-- left, join_use_nulls'; SELECT p.key, kv.key, kv.secret FROM probe_rls AS p LEFT JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key SETTINGS join_use_nulls = 1; -SELECT '-- left any'; -SELECT p.key, kv.secret FROM probe_rls AS p LEFT ANY JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; - SELECT '-- left semi'; SELECT p.key FROM probe_rls AS p LEFT SEMI JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; SELECT '-- left anti'; SELECT p.key FROM probe_rls AS p LEFT ANTI JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; -SELECT '-- key space enumeration'; -SELECT kv.key, kv.secret FROM numbers(10) AS n INNER JOIN kv_rls AS kv ON kv.key = n.number ORDER BY kv.key; - -SELECT '-- using'; -SELECT key, secret FROM probe_rls INNER JOIN kv_rls USING (key) ORDER BY key; - -SELECT '-- no direct join under a row policy'; -SELECT count() FROM ( - EXPLAIN actions = 1 - SELECT p.key, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key -) WHERE explain LIKE '%DirectKeyValueJoin%'; +SELECT '-- join table'; +SELECT p.key, j.secret FROM probe_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; DROP ROW POLICY kv_rls_public ON kv_rls; +DROP ROW POLICY join_rls_public ON join_rls; -SELECT '-- direct join without a policy'; +SELECT '-- without a policy'; SELECT p.key, kv.tenant, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key ORDER BY p.key; -SELECT extract(explain, 'Algorithm: \\w+') FROM ( - EXPLAIN actions = 1 - SELECT p.key, kv.secret FROM probe_rls AS p INNER JOIN kv_rls AS kv ON kv.key = p.key -) WHERE explain LIKE '%Algorithm:%'; +SELECT p.key, j.secret FROM probe_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; DROP TABLE kv_rls; +DROP TABLE join_rls; DROP TABLE probe_rls; diff --git a/tests/queries/0_stateless/05234_row_policy_storage_join.reference b/tests/queries/0_stateless/05234_row_policy_storage_join.reference deleted file mode 100644 index 2ac892e8f505..000000000000 --- a/tests/queries/0_stateless/05234_row_policy_storage_join.reference +++ /dev/null @@ -1,12 +0,0 @@ --- without a policy -1 a -2 b -3 -b \N --- with a policy -1 a --- after dropping the policy -1 a -2 b -3 -b \N diff --git a/tests/queries/0_stateless/05234_row_policy_storage_join.sql b/tests/queries/0_stateless/05234_row_policy_storage_join.sql deleted file mode 100644 index 91d6f948686d..000000000000 --- a/tests/queries/0_stateless/05234_row_policy_storage_join.sql +++ /dev/null @@ -1,31 +0,0 @@ -DROP TABLE IF EXISTS join_rls; -DROP TABLE IF EXISTS probe_join_rls; -DROP ROW POLICY IF EXISTS join_rls_policy ON join_rls; - -CREATE TABLE join_rls (key UInt64, value String) ENGINE = Join(ANY, LEFT, key); -INSERT INTO join_rls VALUES (1, 'a'), (2, 'b'); - -CREATE TABLE probe_join_rls (key UInt64) ENGINE = TinyLog; -INSERT INTO probe_join_rls VALUES (1), (2), (3); - -SELECT '-- without a policy'; -SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; -SELECT joinGet(join_rls, 'value', toUInt64(2)), joinGetOrNull(join_rls, 'value', toUInt64(3)); - -CREATE ROW POLICY join_rls_policy ON join_rls FOR SELECT USING value = 'a' TO CURRENT_USER; - -SELECT '-- with a policy'; -SELECT key, value FROM join_rls ORDER BY key; -SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; -- { serverError ACCESS_DENIED } -SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key SETTINGS join_algorithm = 'hash'; -- { serverError ACCESS_DENIED } -SELECT joinGet(join_rls, 'value', toUInt64(2)); -- { serverError ACCESS_DENIED } -SELECT joinGetOrNull(join_rls, 'value', toUInt64(2)); -- { serverError ACCESS_DENIED } - -DROP ROW POLICY join_rls_policy ON join_rls; - -SELECT '-- after dropping the policy'; -SELECT p.key, j.value FROM probe_join_rls AS p LEFT ANY JOIN join_rls AS j ON j.key = p.key ORDER BY p.key; -SELECT joinGet(join_rls, 'value', toUInt64(2)), joinGetOrNull(join_rls, 'value', toUInt64(3)); - -DROP TABLE join_rls; -DROP TABLE probe_join_rls; From 2d4cececd7280f8578672149aa471f95b595633b Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 19:02:35 +0000 Subject: [PATCH 099/185] Update autogenerated version to 26.8.15.10 and contributors --- cmake/autogenerated_versions.txt | 10 +++++----- 1 file changed, 5 insertions(+), 5 deletions(-) diff --git a/cmake/autogenerated_versions.txt b/cmake/autogenerated_versions.txt index 4e3df68592ef..9789a0bd275e 100644 --- a/cmake/autogenerated_versions.txt +++ b/cmake/autogenerated_versions.txt @@ -2,11 +2,11 @@ # NOTE: VERSION_REVISION has nothing common with DBMS_TCP_PROTOCOL_VERSION, # only DBMS_TCP_PROTOCOL_VERSION should be incremented on protocol changes. -SET(VERSION_REVISION 54527) +SET(VERSION_REVISION 54528) SET(VERSION_MAJOR 26) SET(VERSION_MINOR 8) -SET(VERSION_PATCH 15) -SET(VERSION_GITHASH f1d4a0d36450556c9e3c4f8b3244770b20345516) -SET(VERSION_DESCRIBE v26.8.15.1-lts) -SET(VERSION_STRING 26.8.15.1) +SET(VERSION_PATCH 16) +SET(VERSION_GITHASH 3c6755634c252e1031f2db8bfdfed39f86038237) +SET(VERSION_DESCRIBE v26.8.16.1-lts) +SET(VERSION_STRING 26.8.16.1) # end of autochange From f358962f416eea773e8bb5d39a6220270321c59d Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 22:31:13 +0000 Subject: [PATCH 100/185] Backport #120220 to 26.8: Look through lossless conversions of the indexed column and of constants in the text index --- .../optimizeDirectReadFromTextIndex.cpp | 6 +- .../MergeTreeIndexBloomFilterText.cpp | 13 +- .../MergeTree/MergeTreeIndexConditionText.cpp | 149 ++++++++++++----- .../MergeTree/MergeTreeIndexConditionText.h | 6 +- src/Storages/MergeTree/RPNBuilder.cpp | 33 ++-- src/Storages/MergeTree/RPNBuilder.h | 13 +- ...ap_lowcardinality_subcolumn_cast.reference | 80 +++++++++ ...ndex_map_lowcardinality_subcolumn_cast.sql | 156 ++++++++++++++++++ ...t_index_map_key_wrapped_constant.reference | 30 ++++ ...20_text_index_map_key_wrapped_constant.sql | 66 ++++++++ ...ap_absent_key_fixedstring_needle.reference | 14 ++ ...enbf_map_absent_key_fixedstring_needle.sql | 40 +++++ ...ow_cardinality_nullable_constant.reference | 20 +++ ...alue_low_cardinality_nullable_constant.sql | 40 +++++ ...sor_lossless_conversion_haystack.reference | 40 +++++ ...processor_lossless_conversion_haystack.sql | 73 ++++++++ 16 files changed, 709 insertions(+), 70 deletions(-) create mode 100644 tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.reference create mode 100644 tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.sql create mode 100644 tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.reference create mode 100644 tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.sql create mode 100644 tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.reference create mode 100644 tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.sql create mode 100644 tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.reference create mode 100644 tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.sql create mode 100644 tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.reference create mode 100644 tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.sql diff --git a/src/Processors/QueryPlan/Optimizations/optimizeDirectReadFromTextIndex.cpp b/src/Processors/QueryPlan/Optimizations/optimizeDirectReadFromTextIndex.cpp index a1f99567b656..765a0ffde1dc 100644 --- a/src/Processors/QueryPlan/Optimizations/optimizeDirectReadFromTextIndex.cpp +++ b/src/Processors/QueryPlan/Optimizations/optimizeDirectReadFromTextIndex.cpp @@ -703,11 +703,15 @@ class TextIndexDAGReplacer const auto & preprocessor_dag = preprocessor->getOriginalActionsDAG(); chassert(preprocessor_dag.getOutputs().size() == 1); const auto & preprocessor_output = preprocessor_dag.getOutputs().front(); - auto haystack_name = getNameWithoutAliases(arg_haystack); + /// The index was analyzed on the expression under lossless conversions, e.g. `s` in `hasToken(toNullable(s), 'Foo')`. + const auto * haystack = unwrapLosslessConversion(arg_haystack); + auto haystack_name = getNameWithoutAliases(haystack); /// Check that preprocessor contains current expression as its argument. if (hasSubexpression(preprocessor_output, haystack_name)) { + new_children[0] = haystack; + if (apply_postprocessor) { preprocessor_source_ast = preprocessor->getExpressionAST(new_children[0]->result_name); diff --git a/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp b/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp index e8e299d1a699..4a257d3e8c56 100644 --- a/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexBloomFilterText.cpp @@ -464,6 +464,15 @@ bool MergeTreeConditionBloomFilterText::traverseTreeEquals( Field const_value = value_field; + /// An absent map key reads ''. `String = FixedString(N)` ignores the zero padding of the constant, while + /// the default value of `FixedString(N)` is reported as '', so the constant is compared without its padding. + Field value_without_padding = value_field; + if (isFixedString(value_type) && value_field.getType() == Field::Types::String) + { + auto & value = value_without_padding.safeGet(); + value.resize(value.find_last_not_of('\0') + 1); + } + const auto column_name = key_node.getColumnName(); auto key_index = getKeyIndex(column_name); const auto map_key_index = getKeyIndex(fmt::format("mapKeys({})", column_name)); @@ -483,7 +492,7 @@ bool MergeTreeConditionBloomFilterText::traverseTreeEquals( * We cannot skip keys that does not exist in map if comparison is with default type value because * that way we skip necessary granules where map key does not exist. */ - if (value_field == value_type->getDefault()) + if (value_without_padding == value_type->getDefault()) return false; auto first_argument = key_function_node.getArgumentAt(0); @@ -526,7 +535,7 @@ bool MergeTreeConditionBloomFilterText::traverseTreeEquals( /// Same as arrayElement: skip when comparing with default value because /// the subcolumn returns default for keys that don't exist in the map. - if (value_field == value_type->getDefault()) + if (value_without_padding == value_type->getDefault()) return false; if (const auto map_keys_index = getKeyIndex(fmt::format("mapKeys({})", map_column_name))) diff --git a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp index 1640e2b731c1..331256a048a4 100644 --- a/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp +++ b/src/Storages/MergeTree/MergeTreeIndexConditionText.cpp @@ -898,6 +898,83 @@ static void validateRegexpPatterns(const Array & patterns, const Settings & sett #endif } +namespace +{ + +/// Whether converting a value of type `from` to type `to` never changes it and never throws. +/// `LowCardinality` may be added or dropped and `Nullable` may be added, at any depth of `Array`. +/// `Nullable` cannot be dropped, because it may throw on NULL. +bool isLosslessConversion(const DataTypePtr & from, const DataTypePtr & to) +{ + auto from_type = removeLowCardinality(from); + auto to_type = removeLowCardinality(to); + + if (to_type->isNullable()) + { + from_type = removeNullable(from_type); + to_type = removeNullable(to_type); + } + else if (from_type->isNullable()) + { + return false; + } + + if (from_type->equals(*to_type)) + return true; + + const auto * from_array = typeid_cast(from_type.get()); + const auto * to_array = typeid_cast(to_type.get()); + return from_array && to_array && isLosslessConversion(from_array->getNestedType(), to_array->getNestedType()); +} + +/// Whether the node is `CAST`, `_CAST`, `toNullable` or `toLowCardinality` with a lossless conversion (see above). +bool isLosslessConversionFunction(const ActionsDAG::Node & node) +{ + if (node.type != ActionsDAG::ActionType::FUNCTION || !node.function_base) + return false; + + const auto function_name = node.function_base->getName(); + const size_t arguments_size = node.children.size(); + + const bool is_cast = (function_name == "CAST" || function_name == "_CAST") && arguments_size == 2; + const bool is_wrapper = (function_name == "toNullable" || function_name == "toLowCardinality") && arguments_size == 1; + + if (!is_cast && !is_wrapper) + return false; + + return isLosslessConversion(node.children.front()->result_type, node.result_type); +} + +/// Strips lossless conversions from the node (see above). +RPNBuilderTreeNode unwrapLosslessConversion(const RPNBuilderTreeNode & node) +{ + if (!node.isFunction()) + return node; + + /// Only the DAG form carries the types; the AST form is left as is. + const auto function = node.toFunctionNode(); + const auto * function_dag_node = function.getDAGNode(); + + if (!function_dag_node || !isLosslessConversionFunction(*function_dag_node)) + return node; + + return unwrapLosslessConversion(function.getArgumentAt(0)); +} + +} + +const ActionsDAG::Node * unwrapLosslessConversion(const ActionsDAG::Node * node) +{ + const auto * node_without_alias = node; + while (node_without_alias->type == ActionsDAG::ActionType::ALIAS) + node_without_alias = node_without_alias->children.front(); + + if (!isLosslessConversionFunction(*node_without_alias)) + return node; + + return unwrapLosslessConversion(node_without_alias->children.front()); +} + /// The value an absent map key reads: `''`, or all NUL when the value type is `FixedString`. /// `mapValues` stores neither. static bool isMapValueDefault(std::string_view value, const Block & header) @@ -913,7 +990,7 @@ static bool isMapValueDefault(std::string_view value, const Block & header) bool MergeTreeIndexConditionText::traverseFunctionNode( const RPNBuilderFunctionTreeNode & function_node, - const RPNBuilderTreeNode & index_column_node, + const RPNBuilderTreeNode & argument_node, DataTypePtr value_type, Field value_field, RPNElement & out) const @@ -921,6 +998,9 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( const String function_name = function_node.getFunctionName(); auto direct_read_mode = getDirectReadMode(function_name); + /// The index knows the expression under the conversion, e.g. `m.key_` in `equals(_CAST(m.key_, 'String'), 'value')`. + const auto index_column_node = unwrapLosslessConversion(argument_node); + auto index_column_name = index_column_node.getColumnName(); bool has_index_column = hasIndexForColumn(index_column_name); bool has_map_keys_column = hasIndexForColumn(fmt::format("mapKeys({})", index_column_name)); @@ -1157,50 +1237,30 @@ bool MergeTreeIndexConditionText::traverseFunctionNode( } if (function_name == "hasToken" || function_name == "hasTokenOrNull") { - // hasToken and hasTokenOrNull are legacy functions which assume splitByNonAlpha as - /// tokenizer. The text index can answer it only correctly if this is the index tokenizer. - /// In all other cases, bypass the index. + /// `hasToken` splits by non-alphanumeric characters, so only an index with the same tokenizer can answer it. if (tokenizer->getType() != ITokenizer::Type::SplitByNonAlpha) return false; - /// Unlike hasToken, hasTokenOrNull is never rewritten to direct-read, so the pre/postprocessor - /// is also not applied to its needle. Using the index here (where stringToTokens does apply them, - /// e.g. mapping a dropped token to the empty sentinel that prunes every granule) would disagree - /// with the scan result. Bail out so the index is not used for hasTokenOrNull when a - /// pre/postprocessor is configured; the plain index path is unaffected. + /// `hasTokenOrNull` is never rewritten to a direct read, so its needle never goes through the + /// pre/postprocessor, while `stringToTokens` applies them. The index would disagree with the scan. if (function_name == "hasTokenOrNull" && (has_preprocessor || has_postprocessor)) return false; - /// A needle containing a token separator is invalid for `hasToken` and the brute-force scan raises - /// BAD_ARGUMENTS for this. hasToken uses Exact direct read, so the index would tokenize the needle and - /// silently replace the predicate (or prune the granule that would have thrown), hiding the exception. - /// Therefore bypass the index and do a brute-force scan. hasTokenOrNull is not affected: it returns NULL - /// - /// A separator is any ASCII non-alphanumeric character. - if (function_name == "hasToken" - && std::ranges::any_of(value_field.safeGet(), [](unsigned char c) { return isASCII(c) && !isAlphaNumericASCII(c); })) + /// The scan raises BAD_ARGUMENTS for a needle with a separator, and the exact direct read of `hasToken` + /// would replace the predicate and hide it. `hasTokenOrNull` returns NULL for such a needle instead. + if (function_name == "hasToken" && std::ranges::any_of(value_field.safeGet(), isTokenSeparator)) return false; auto tokens = stringToTokens(value_field); if (tokens.empty()) { + /// A needle without a word character is invalid: leave it to the scan, which raises or returns NULL. + /// Otherwise the pre/postprocessor dropped the needle (e.g. a stop word), so it is not in the index: + /// push the empty sentinel to prune every granule. const String & string_needle = value_field.safeGet(); - if (!string_needle.empty()) - { - /// hasToken uses splitByNonAlpha as its tokenizer, so: - /// - A needle without any word character (alphanumeric or non-ASCII) is invalid. - /// - Bypass the index in that case so the row-level evaluation throws BAD_ARGUMENTS (or returns NULL for hasTokenOrNull) - /// -- Consistent with the no-index behaviour. - /// If the needle does contain word characters (e.g. "abc" with ngrams(4)): - /// - It is valid but too short for the index's tokenizer: - /// -- Fall through to push "" so all granules are pruned and the query returns 0 rows. - /// If the postprocessor filters the needle (e.g. stop-word): - /// -- The needle is not in the index; push "" sentinel so the condition evaluates to false. - if (std::ranges::none_of(string_needle, [](unsigned char c) { return !isASCII(c) || isAlphaNumericASCII(c); })) - return false; - } - /// - If the needle does contain word characters (e.g. "abc" with ngrams(4)), it is valid but too short for the index's tokenizer: - /// Fall through but push "" so all granules are pruned and the query returns 0 rows. + if (!string_needle.empty() && std::ranges::all_of(string_needle, isTokenSeparator)) + return false; + tokens.push_back(""); } @@ -1523,9 +1583,13 @@ bool MergeTreeIndexConditionText::traverseMapElementKeyNode(const RPNBuilderFunc /// It can be an arbitrary function that returns 0 for the default value of the map value type. /// It is true because `arrayElement` (and the equivalent subcolumn access) returns default value if key doesn't exist in the map, /// therefore we can use index to skip granules and use direct read as a hint for the original condition. + /// The result may be `Nullable` or `LowCardinality`, e.g. for a `Nullable` key: NULL reads as false below, as in WHERE. const auto * dag_node = function_node.getDAGNode(); - if (!dag_node || !dag_node->function_base || !dag_node->isDeterministic() || !WhichDataType(dag_node->result_type).isUInt8()) + if (!dag_node + || !dag_node->function_base + || !dag_node->isDeterministic() + || !WhichDataType(removeLowCardinalityAndNullable(dag_node->result_type)).isUInt8()) return false; auto subdag = ActionsDAG::cloneSubDAG({dag_node}, true); @@ -1565,10 +1629,14 @@ bool MergeTreeIndexConditionText::traverseMapElementKeyNode(const RPNBuilderFunc if (map_argument->type != ActionsDAG::ActionType::INPUT || map_argument->result_name != required_column.name) return false; - if (const_key_argument->type != ActionsDAG::ActionType::COLUMN || !isStringOrFixedString(const_key_argument->result_type)) + /// A NULL key is declined: `arrayElement` returns NULL for it, not the default value. + Field key_field; + DataTypePtr key_type; + if (!RPNBuilderTreeNode(const_key_argument, function_node.getTreeContext()).tryGetConstant(key_field, key_type) + || key_field.getType() != Field::Types::String) return false; - key_const_value = std::string{const_key_argument->column->getDataAt(0)}; + key_const_value = key_field.safeGet(); } else { @@ -1672,8 +1740,10 @@ bool MergeTreeIndexConditionText::traverseJSONSubcolumnKeyNode( /// Similar to traverseMapElementKeyNode but for JSON subcolumns. const auto * dag_node = function_node.getDAGNode(); - if (!dag_node || !dag_node->function_base || !dag_node->isDeterministic() - || !WhichDataType(removeNullable(dag_node->result_type)).isUInt8()) + if (!dag_node + || !dag_node->function_base + || !dag_node->isDeterministic() + || !WhichDataType(removeLowCardinalityAndNullable(dag_node->result_type)).isUInt8()) return false; auto subdag = ActionsDAG::cloneSubDAG({dag_node}, true); @@ -1741,8 +1811,9 @@ bool MergeTreeIndexConditionText::tryPrepareSetForTextSearch( /// `m['key']` answered by a `mapValues(m)` index: an absent key reads the value type's default. bool has_index_for_map_element_value = false; - auto has_index = [&](const RPNBuilderTreeNode & node) + auto has_index = [&](const RPNBuilderTreeNode & argument) { + const auto node = unwrapLosslessConversion(argument); if (hasIndexForMapElementValue(node)) { has_index_for_map_element_value = true; diff --git a/src/Storages/MergeTree/MergeTreeIndexConditionText.h b/src/Storages/MergeTree/MergeTreeIndexConditionText.h index 5ac4da009fc1..49f37d4ece26 100644 --- a/src/Storages/MergeTree/MergeTreeIndexConditionText.h +++ b/src/Storages/MergeTree/MergeTreeIndexConditionText.h @@ -163,7 +163,7 @@ class MergeTreeIndexConditionText final : public IMergeTreeIndexCondition, publi bool traverseFunctionNode( const RPNBuilderFunctionTreeNode & function_node, - const RPNBuilderTreeNode & index_column_node, + const RPNBuilderTreeNode & argument_node, DataTypePtr value_type, Field value_field, RPNElement & out) const; @@ -237,4 +237,8 @@ class MergeTreeIndexConditionText final : public IMergeTreeIndexCondition, publi static constexpr std::string_view TEXT_INDEX_VIRTUAL_COLUMN_PREFIX = "__text_index_"; bool isTextIndexVirtualColumn(const String & column_name); +/// Strips `CAST`, `_CAST`, `toNullable` and `toLowCardinality` from the node while the conversion never +/// changes the value and never throws. The index is analyzed on the expression under such conversions. +const ActionsDAG::Node * unwrapLosslessConversion(const ActionsDAG::Node * node); + } diff --git a/src/Storages/MergeTree/RPNBuilder.cpp b/src/Storages/MergeTree/RPNBuilder.cpp index 47ed74cc8ff1..dfacb0126fc1 100644 --- a/src/Storages/MergeTree/RPNBuilder.cpp +++ b/src/Storages/MergeTree/RPNBuilder.cpp @@ -248,14 +248,14 @@ const Settings & RPNBuilderTreeContext::getSettings() const return query_context->getSettingsRef(); } -RPNBuilderTreeNode::RPNBuilderTreeNode(const ActionsDAG::Node * dag_node_, RPNBuilderTreeContext & tree_context_) +RPNBuilderTreeNode::RPNBuilderTreeNode(const ActionsDAG::Node * dag_node_, const RPNBuilderTreeContext & tree_context_) : dag_node(dag_node_) , tree_context(tree_context_) { chassert(dag_node); } -RPNBuilderTreeNode::RPNBuilderTreeNode(const IAST * ast_node_, RPNBuilderTreeContext & tree_context_) +RPNBuilderTreeNode::RPNBuilderTreeNode(const IAST * ast_node_, const RPNBuilderTreeContext & tree_context_) : ast_node(ast_node_) , tree_context(tree_context_) { @@ -371,6 +371,16 @@ ColumnWithTypeAndName RPNBuilderTreeNode::getConstantColumn() const return result; } +/// A `Field` is the plain value of a constant: `LowCardinality` is only an encoding of the column, +/// and a value that is not NULL has no `Nullable` type. +static DataTypePtr getTypeOfConstantValue(const Field & value, const DataTypePtr & type) +{ + auto value_type = removeLowCardinality(type); + if (!value.isNull()) + value_type = removeNullable(value_type); + return value_type; +} + bool RPNBuilderTreeNode::tryGetConstant(Field & output_value, DataTypePtr & output_type) const { if (ast_node) @@ -389,12 +399,7 @@ bool RPNBuilderTreeNode::tryGetConstant(Field & output_value, DataTypePtr & outp /// Simple literal output_value = literal->value; - output_type = block_with_constants.getByName(column_name).type; - - /// If constant is not Null, we can assume it's type is not Nullable as well. - if (!output_value.isNull()) - output_type = removeNullable(output_type); - + output_type = getTypeOfConstantValue(output_value, block_with_constants.getByName(column_name).type); return true; } if (block_with_constants.has(column_name) && isColumnConst(*block_with_constants.getByName(column_name).column)) @@ -402,11 +407,7 @@ bool RPNBuilderTreeNode::tryGetConstant(Field & output_value, DataTypePtr & outp /// An expression which is dependent on constants only const auto & constant_column = block_with_constants.getByName(column_name); output_value = (*constant_column.column)[0]; - output_type = constant_column.type; - - if (!output_value.isNull()) - output_type = removeNullable(output_type); - + output_type = getTypeOfConstantValue(output_value, constant_column.type); return true; } } @@ -417,11 +418,7 @@ bool RPNBuilderTreeNode::tryGetConstant(Field & output_value, DataTypePtr & outp if (node_without_alias->column) { output_value = node_without_alias->column->getField(); - output_type = node_without_alias->result_type; - - if (!output_value.isNull()) - output_type = removeNullable(output_type); - + output_type = getTypeOfConstantValue(output_value, node_without_alias->result_type); return true; } } diff --git a/src/Storages/MergeTree/RPNBuilder.h b/src/Storages/MergeTree/RPNBuilder.h index a785294769b2..1a38ec1e3496 100644 --- a/src/Storages/MergeTree/RPNBuilder.h +++ b/src/Storages/MergeTree/RPNBuilder.h @@ -76,10 +76,10 @@ class RPNBuilderTreeNode { public: /// Construct RPNBuilderTreeNode with non null dag node and tree context - explicit RPNBuilderTreeNode(const ActionsDAG::Node * dag_node_, RPNBuilderTreeContext & tree_context_); + explicit RPNBuilderTreeNode(const ActionsDAG::Node * dag_node_, const RPNBuilderTreeContext & tree_context_); /// Construct RPNBuilderTreeNode with non null ast node and tree context - explicit RPNBuilderTreeNode(const IAST * ast_node_, RPNBuilderTreeContext & tree_context_); + explicit RPNBuilderTreeNode(const IAST * ast_node_, const RPNBuilderTreeContext & tree_context_); /// Get AST node const IAST * getASTNode() const { return ast_node; } @@ -113,6 +113,7 @@ class RPNBuilderTreeNode /** Try get constant from node. If node is constant returns true, and constant value and constant type output parameters are set. * Otherwise false is returned. + * The output type is the type of the value: `LowCardinality` is removed, and `Nullable` is removed when the value is not NULL. */ bool tryGetConstant(Field & output_value, DataTypePtr & output_type) const; @@ -141,16 +142,10 @@ class RPNBuilderTreeNode return tree_context; } - /// Get tree context - RPNBuilderTreeContext & getTreeContext() - { - return tree_context; - } - protected: const IAST * ast_node = nullptr; const ActionsDAG::Node * dag_node = nullptr; - RPNBuilderTreeContext & tree_context; + const RPNBuilderTreeContext & tree_context; }; /** RPNBuilderFunctionTreeNode is wrapper around RPNBuilderTreeNode with function type. diff --git a/tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.reference b/tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.reference new file mode 100644 index 000000000000..97d96022f020 --- /dev/null +++ b/tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.reference @@ -0,0 +1,80 @@ +-- the analyzer wraps the subcolumn into a cast to String +1 +-- mapValues: the index prunes granules through the cast +Granules: 2/2 +Name: idx +Granules: 1/2 +Granules: 2/2 +Name: idx +Granules: 1/2 +-- mapValues: results match the scan +idx 1 +idx 3 +scan 1 +scan 3 +idx 1 +idx 2 +scan 1 +scan 2 +idx 3 +scan 3 +-- mapValues is only a hint: a value that belongs to another key does not match +idx 0 +-- mapValues on LowCardinality(Nullable) values: the cast to Nullable(String) is looked through +Granules: 2/2 +Name: idx +Granules: 1/2 +idx 3 +scan 3 +-- String column: conversions that only add Nullable or LowCardinality are looked through +CAST Nullable Granules: 2/2 +CAST Nullable Name: idx +CAST Nullable Granules: 1/2 +CAST LowCardinality Granules: 2/2 +CAST LowCardinality Name: idx +CAST LowCardinality Granules: 1/2 +toNullable Granules: 2/2 +toNullable Name: idx +toNullable Granules: 1/2 +toLowCardinality Granules: 2/2 +toLowCardinality Name: idx +toLowCardinality Granules: 1/2 +nested Granules: 2/2 +nested Name: idx +nested Granules: 1/2 +idx 1 +idx 2 +scan 1 +scan 2 +not idx 3 +not idx 4 +not scan 3 +not scan 4 +in idx 1 +in idx 3 +in scan 1 +in scan 3 +-- join_use_nulls pushes the filter down to the outer side as toNullable(column) +1 +3 +-- Nullable column: adding LowCardinality is looked through +Granules: 2/2 +Name: idx +Granules: 1/2 +idx 1 +idx 2 +scan 1 +scan 2 +-- Nullable column: dropping Nullable is not looked through, the cast still throws on the NULL row +0 +-- Array column: a cast that only wraps the elements is looked through +Array(Nullable) Granules: 2/2 +Array(Nullable) Name: idx +Array(Nullable) Granules: 1/2 +Array(LowCardinality) Granules: 2/2 +Array(LowCardinality) Name: idx +Array(LowCardinality) Granules: 1/2 +idx 1 +idx 2 +scan 1 +scan 2 diff --git a/tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.sql b/tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.sql new file mode 100644 index 000000000000..9c2f4b4409a6 --- /dev/null +++ b/tests/queries/0_stateless/05217_text_index_map_lowcardinality_subcolumn_cast.sql @@ -0,0 +1,156 @@ +-- Tags: no-parallel-replicas +-- Tag no-parallel-replicas -- direct read is not compatible with parallel replicas + +-- `arrayElement` on a `Map(K, LowCardinality(V))` returns `V`, while the subcolumn `m.key_` is `LowCardinality(V)`. +-- Hence `optimize_functions_to_subcolumns` rewrites `m['key'] = 'value'` into `_CAST(m.key_, 'V') = 'value'`, +-- and the text index has to look through the cast: for an index on `mapValues(m)`. +-- The same applies to every conversion that cannot change a value or throw: adding or dropping `LowCardinality`, +-- adding `Nullable` (also as `toNullable`, which `join_use_nulls` emits in pushed-down filters), at any depth of `Array`. + +SET explain_query_plan_default = 'legacy'; +SET enable_analyzer = 1; +SET optimize_functions_to_subcolumns = 1; +SET use_skip_indexes = 1; +SET query_plan_direct_read_from_text_index = 1; + +DROP TABLE IF EXISTS tab_values; + +CREATE TABLE tab_values +( + id UInt32, + m Map(String, LowCardinality(String)), + INDEX idx mapValues(m) TYPE text(tokenizer = 'splitByNonAlpha') GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY id +SETTINGS index_granularity = 2, min_bytes_for_wide_part = 0; + +INSERT INTO tab_values VALUES (1, {'level':'error','msg':'disk is full'}), (2, {'level':'warn','msg':'disk is slow'}), (3, {'level':'error','msg':'network is down'}), (4, {}); + +SELECT '-- the analyzer wraps the subcolumn into a cast to String'; +SELECT count() FROM (EXPLAIN QUERY TREE run_passes = 1 SELECT id FROM tab_values WHERE m['level'] = 'warn') WHERE explain LIKE '%function_name: _CAST%'; + +SELECT '-- mapValues: the index prunes granules through the cast'; +SELECT trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_values WHERE m['level'] = 'warn') WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_values WHERE hasToken(m['msg'], 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; + +SELECT '-- mapValues: results match the scan'; +SELECT 'idx', id FROM tab_values WHERE m['level'] = 'error' ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_values WHERE m['level'] = 'error' ORDER BY id SETTINGS use_skip_indexes = 0; +SELECT 'idx', id FROM tab_values WHERE hasToken(m['msg'], 'disk') ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_values WHERE hasToken(m['msg'], 'disk') ORDER BY id SETTINGS use_skip_indexes = 0; +SELECT 'idx', id FROM tab_values WHERE m['msg'] LIKE '%down%' ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_values WHERE m['msg'] LIKE '%down%' ORDER BY id SETTINGS use_skip_indexes = 0; +SELECT '-- mapValues is only a hint: a value that belongs to another key does not match'; +SELECT 'idx', count() FROM tab_values WHERE m['level'] = 'disk' SETTINGS force_data_skipping_indices = 'idx'; + +DROP TABLE tab_values; + +-- A `mapValues` index on `LowCardinality(Nullable(String))` values: the analyzer casts to `Nullable(String)`. +DROP TABLE IF EXISTS tab_values_n; + +CREATE TABLE tab_values_n +( + id UInt32, + m Map(String, LowCardinality(Nullable(String))), + INDEX idx mapValues(m) TYPE text(tokenizer = 'splitByNonAlpha') GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY id +SETTINGS index_granularity = 2, min_bytes_for_wide_part = 0; + +INSERT INTO tab_values_n VALUES (1, {'level':'error'}), (2, {'level':'error'}), (3, {'level':'warn'}), (4, {'level':NULL}); + +SELECT '-- mapValues on LowCardinality(Nullable) values: the cast to Nullable(String) is looked through'; +SELECT trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_values_n WHERE m['level'] = 'warn') WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'idx', id FROM tab_values_n WHERE m['level'] = 'warn' ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_values_n WHERE m['level'] = 'warn' ORDER BY id SETTINGS use_skip_indexes = 0; + +DROP TABLE tab_values_n; + +-- Plain columns with explicit conversions. +DROP TABLE IF EXISTS tab_s; + +CREATE TABLE tab_s +( + id UInt32, + s String, + INDEX idx s TYPE text(tokenizer = 'splitByNonAlpha') GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY id +SETTINGS index_granularity = 2, min_bytes_for_wide_part = 0; + +INSERT INTO tab_s VALUES (1, 'disk is full'), (2, 'disk is slow'), (3, 'network is down'), (4, ''); + +SELECT '-- String column: conversions that only add Nullable or LowCardinality are looked through'; +SELECT 'CAST Nullable', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_s WHERE hasToken(CAST(s, 'Nullable(String)'), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'CAST LowCardinality', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_s WHERE hasToken(CAST(s, 'LowCardinality(String)'), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'toNullable', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_s WHERE hasToken(toNullable(s), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'toLowCardinality', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_s WHERE hasToken(toLowCardinality(s), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'nested', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_s WHERE hasToken(toNullable(toLowCardinality(s)), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'idx', id FROM tab_s WHERE hasToken(toNullable(s), 'disk') ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_s WHERE hasToken(toNullable(s), 'disk') ORDER BY id SETTINGS use_skip_indexes = 0; +SELECT 'not idx', id FROM tab_s WHERE NOT hasToken(toNullable(s), 'disk') ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'not scan', id FROM tab_s WHERE NOT hasToken(toNullable(s), 'disk') ORDER BY id SETTINGS use_skip_indexes = 0; +SELECT 'in idx', id FROM tab_s WHERE toNullable(s) IN ('disk is full', 'network is down') ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'in scan', id FROM tab_s WHERE toNullable(s) IN ('disk is full', 'network is down') ORDER BY id SETTINGS use_skip_indexes = 0; + +SELECT '-- join_use_nulls pushes the filter down to the outer side as toNullable(column)'; +DROP TABLE IF EXISTS tab_ids; +CREATE TABLE tab_ids (id UInt32) ENGINE = MergeTree ORDER BY id; +INSERT INTO tab_ids SELECT number FROM numbers(1, 4); +-- `query_plan_convert_outer_join_to_inner_join` is pinned because the filter is pushed below the join only +-- when the LEFT JOIN becomes an INNER one; otherwise it stays above the join and no index is consulted. +SELECT count() FROM (EXPLAIN indexes = 1 SELECT l.id FROM tab_ids AS l LEFT JOIN tab_s AS r ON l.id = r.id WHERE hasToken(r.s, 'network') SETTINGS join_use_nulls = 1, query_plan_convert_outer_join_to_inner_join = 1) WHERE explain LIKE '%Name: idx%'; +SELECT l.id FROM tab_ids AS l LEFT JOIN tab_s AS r ON l.id = r.id WHERE hasToken(r.s, 'network') ORDER BY l.id SETTINGS join_use_nulls = 1; +DROP TABLE tab_ids; + +DROP TABLE tab_s; + +DROP TABLE IF EXISTS tab_ns; + +CREATE TABLE tab_ns +( + id UInt32, + s Nullable(String), + INDEX idx s TYPE text(tokenizer = 'splitByNonAlpha') GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY id +SETTINGS index_granularity = 2, min_bytes_for_wide_part = 0; + +INSERT INTO tab_ns VALUES (1, 'disk is full'), (2, 'disk is slow'), (3, 'network is down'), (4, NULL); + +SELECT '-- Nullable column: adding LowCardinality is looked through'; +SELECT trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_ns WHERE hasToken(CAST(s, 'LowCardinality(Nullable(String))'), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'idx', id FROM tab_ns WHERE hasToken(CAST(s, 'LowCardinality(Nullable(String))'), 'disk') ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_ns WHERE hasToken(CAST(s, 'LowCardinality(Nullable(String))'), 'disk') ORDER BY id SETTINGS use_skip_indexes = 0; + +SELECT '-- Nullable column: dropping Nullable is not looked through, the cast still throws on the NULL row'; +SELECT count() FROM (EXPLAIN indexes = 1 SELECT id FROM tab_ns WHERE hasToken(CAST(s, 'String'), 'network')) WHERE explain LIKE '%Name: idx%'; +SELECT id FROM tab_ns WHERE hasToken(CAST(s, 'String'), 'network'); -- { serverError CANNOT_INSERT_NULL_IN_ORDINARY_COLUMN } + +DROP TABLE tab_ns; + +DROP TABLE IF EXISTS tab_arr; + +CREATE TABLE tab_arr +( + id UInt32, + arr Array(String), + INDEX idx arr TYPE text(tokenizer = 'array') GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY id +SETTINGS index_granularity = 2, min_bytes_for_wide_part = 0; + +INSERT INTO tab_arr VALUES (1, ['disk', 'full']), (2, ['disk', 'slow']), (3, ['network', 'down']), (4, []); + +SELECT '-- Array column: a cast that only wraps the elements is looked through'; +SELECT 'Array(Nullable)', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_arr WHERE has(CAST(arr, 'Array(Nullable(String))'), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'Array(LowCardinality)', trimLeft(explain) FROM (EXPLAIN indexes = 1 SELECT id FROM tab_arr WHERE has(CAST(arr, 'Array(LowCardinality(String))'), 'network')) WHERE explain LIKE '%Name:%' OR explain LIKE '%Granules:%'; +SELECT 'idx', id FROM tab_arr WHERE has(CAST(arr, 'Array(Nullable(String))'), 'disk') ORDER BY id SETTINGS force_data_skipping_indices = 'idx'; +SELECT 'scan', id FROM tab_arr WHERE has(CAST(arr, 'Array(Nullable(String))'), 'disk') ORDER BY id SETTINGS use_skip_indexes = 0; + +DROP TABLE tab_arr; diff --git a/tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.reference b/tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.reference new file mode 100644 index 000000000000..97f89747069d --- /dev/null +++ b/tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.reference @@ -0,0 +1,30 @@ +-- results are correct and independent of the key type wrapper +2 +2 +2 +2 +0 +0 +0 +-- mapValues index prunes granules for a wrapped constant key, arrayElement form +1 +1 +1 +-- mapValues index prunes granules for a wrapped constant key, subcolumn form +1 +1 +1 +-- mapKeys index prunes granules for a wrapped constant key, arrayElement form +1 +1 +1 +-- mapKeys index prunes granules for a wrapped constant key, subcolumn form +1 +1 +1 +-- a NULL constant key does not use the mapKeys index: arrayElement returns NULL, not the default +0 +-- FixedString cast on the value does not use the mapValues index, Nullable cast does +2 +0 +1 diff --git a/tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.sql b/tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.sql new file mode 100644 index 000000000000..c6fa27dd91fe --- /dev/null +++ b/tests/queries/0_stateless/05220_text_index_map_key_wrapped_constant.sql @@ -0,0 +1,66 @@ +-- A text index on `mapKeys(m)` / `mapValues(m)` is used for `m[key] = value` also when the constant key +-- is wrapped in `Nullable`, `LowCardinality` or `LowCardinality(Nullable)`. A non-NULL constant key +-- reads the value-type default ('') for an absent key, not NULL, so the index can still prune granules. +-- The `arrayElement(m, key)` form is exercised with `materialize` on the key, which defeats the rewrite +-- to a map subcolumn; without it the analyzer produces `_CAST(m.key_, 'Nullable(String)')`. + +SET enable_analyzer = 1; +SET optimize_functions_to_subcolumns = 1; +-- Keep a `ReadFromMergeTree` step in the plan so the granule count is reported. +SET query_plan_optimize_count_from_text_index = 0; + +DROP TABLE IF EXISTS tab; + +CREATE TABLE tab +( + m Map(String, String), + INDEX idx_keys mapKeys(m) TYPE text(tokenizer = 'splitByNonAlpha') GRANULARITY 1, + INDEX idx_vals mapValues(m) TYPE text(tokenizer = 'splitByNonAlpha') GRANULARITY 1 +) +ENGINE = MergeTree +ORDER BY tuple() +SETTINGS index_granularity = 8; + +INSERT INTO tab SELECT map('key' || toString(number % 1000), 'value' || toString(number % 1000)) FROM numbers(1024); + +SELECT '-- results are correct and independent of the key type wrapper'; +SELECT count() FROM tab WHERE m['key5'] = 'value5'; +SELECT count() FROM tab WHERE m[CAST('key5' AS Nullable(String))] = 'value5'; +SELECT count() FROM tab WHERE m[CAST('key5' AS LowCardinality(String))] = 'value5'; +SELECT count() FROM tab WHERE m[CAST('key5' AS LowCardinality(Nullable(String)))] = 'value5'; +SELECT count() FROM tab WHERE m[CAST('missing' AS Nullable(String))] = 'value5'; +SELECT count() FROM tab WHERE m[CAST(NULL AS Nullable(String))] = 'value5'; +SELECT count() FROM tab WHERE m[CAST(NULL AS LowCardinality(Nullable(String)))] = 'value5'; + +SELECT '-- mapValues index prunes granules for a wrapped constant key, arrayElement form'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST('key5' AS Nullable(String)))] = 'value5') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST('key5' AS LowCardinality(String)))] = 'value5') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST('key5' AS LowCardinality(Nullable(String))))] = 'value5') WHERE explain ILIKE '%Granules: 2/128%'; + +SELECT '-- mapValues index prunes granules for a wrapped constant key, subcolumn form'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[CAST('key5' AS Nullable(String))] = 'value5') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[CAST('key5' AS LowCardinality(String))] = 'value5') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[CAST('key5' AS LowCardinality(Nullable(String)))] = 'value5') WHERE explain ILIKE '%Granules: 2/128%'; + +SELECT '-- mapKeys index prunes granules for a wrapped constant key, arrayElement form'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST('key5' AS Nullable(String)))] != '') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST('key5' AS LowCardinality(String)))] != '') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST('key5' AS LowCardinality(Nullable(String))))] != '') WHERE explain ILIKE '%Granules: 2/128%'; + +SELECT '-- mapKeys index prunes granules for a wrapped constant key, subcolumn form'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[CAST('key5' AS Nullable(String))] != '') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[CAST('key5' AS LowCardinality(String))] != '') WHERE explain ILIKE '%Granules: 2/128%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[CAST('key5' AS LowCardinality(Nullable(String)))] != '') WHERE explain ILIKE '%Granules: 2/128%'; + +SELECT '-- a NULL constant key does not use the mapKeys index: arrayElement returns NULL, not the default'; +SELECT count() FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE m[materialize(CAST(NULL AS Nullable(String)))] != '') WHERE explain ILIKE '%idx_keys%'; + +-- A cast of the map value that changes its bytes must not reuse the mapValues index: the index stores +-- the raw String token, while `CAST(m[key], 'FixedString(N)')` pads the value with zero bytes. Only the +-- casts that keep the bytes (adding `Nullable` or `LowCardinality`) may reuse it. +SELECT '-- FixedString cast on the value does not use the mapValues index, Nullable cast does'; +SELECT count() FROM tab WHERE CAST(m['key5'], 'FixedString(6)') = toFixedString('value5', 6); +SELECT count() FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE CAST(m[materialize('key5')], 'FixedString(6)') = toFixedString('value5', 6)) WHERE explain ILIKE '%idx_vals%'; +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM tab WHERE CAST(m[materialize('key5')], 'Nullable(String)') = 'value5') WHERE explain ILIKE '%idx_vals%'; + +DROP TABLE tab; diff --git a/tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.reference b/tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.reference new file mode 100644 index 000000000000..f0490c810534 --- /dev/null +++ b/tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.reference @@ -0,0 +1,14 @@ +-- absent key, empty FixedString needle: all rows match +1024 +1024 +1024 +1024 +1024 +1024 +1024 +1024 +-- present key, padded FixedString needle: the index is used and the row is found +1 500 +1 +1 500 +1 diff --git a/tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.sql b/tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.sql new file mode 100644 index 000000000000..306b49765f51 --- /dev/null +++ b/tests/queries/0_stateless/05221_tokenbf_map_absent_key_fixedstring_needle.sql @@ -0,0 +1,40 @@ +-- `m['missing'] = CAST('', 'FixedString(N)')` matches every row without the key: `arrayElement` returns '' for +-- an absent key, and `String = FixedString(N)` ignores the zero padding of the constant. The `mapKeys` bloom +-- filter index must not be used for such a comparison, otherwise every granule is pruned and the rows are lost. +-- The default value of `FixedString(N)` is reported as '', while the constant holds N zero bytes, so the guard +-- has to compare the constant without its padding. + +SET enable_analyzer = 1; + +DROP TABLE IF EXISTS t_tokenbf; +DROP TABLE IF EXISTS t_ngrambf; + +CREATE TABLE t_tokenbf (id UInt64, attrs Map(String, String), INDEX idx mapKeys(attrs) TYPE tokenbf_v1(256, 2, 0) GRANULARITY 1) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 8; +INSERT INTO t_tokenbf SELECT number, map(if(number = 500, 'entity', 'other'), 'v') FROM numbers(1024); + +CREATE TABLE t_ngrambf (id UInt64, attrs Map(String, String), INDEX idx mapKeys(attrs) TYPE ngrambf_v1(3, 256, 2, 0) GRANULARITY 1) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 8; +INSERT INTO t_ngrambf SELECT number, map(if(number = 500, 'entityword', 'otherword'), 'v') FROM numbers(1024); + +SELECT '-- absent key, empty FixedString needle: all rows match'; +SET optimize_functions_to_subcolumns = 0; +SELECT count() FROM t_tokenbf WHERE attrs['missing'] = CAST('', 'FixedString(3)'); +SELECT count() FROM t_tokenbf WHERE attrs['missing'] = CAST('', 'LowCardinality(FixedString(3))'); +SELECT count() FROM t_tokenbf WHERE attrs['missing'] = CAST('', 'Nullable(FixedString(3))'); +SELECT count() FROM t_ngrambf WHERE attrs['missing'] = CAST('', 'FixedString(3)'); +SET optimize_functions_to_subcolumns = 1; +SELECT count() FROM t_tokenbf WHERE attrs['missing'] = CAST('', 'FixedString(3)'); +SELECT count() FROM t_tokenbf WHERE attrs['missing'] = CAST('', 'LowCardinality(FixedString(3))'); +SELECT count() FROM t_tokenbf WHERE attrs['missing'] = CAST('', 'Nullable(FixedString(3))'); +SELECT count() FROM t_ngrambf WHERE attrs['missing'] = CAST('', 'FixedString(3)'); + +SELECT '-- present key, padded FixedString needle: the index is used and the row is found'; +SELECT count(), min(id) FROM t_tokenbf WHERE attrs['entity'] = CAST('v', 'FixedString(3)'); +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM t_tokenbf WHERE attrs['entity'] = CAST('v', 'FixedString(3)')) WHERE explain ILIKE '%Granules: 1/128%'; +SET optimize_functions_to_subcolumns = 0; +SELECT count(), min(id) FROM t_tokenbf WHERE attrs['entity'] = CAST('v', 'FixedString(3)'); +SELECT count() > 0 FROM (EXPLAIN indexes = 1 SELECT count() FROM t_tokenbf WHERE attrs['entity'] = CAST('v', 'FixedString(3)')) WHERE explain ILIKE '%Granules: 1/128%'; + +DROP TABLE t_tokenbf; +DROP TABLE t_ngrambf; diff --git a/tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.reference b/tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.reference new file mode 100644 index 000000000000..450deb365d13 --- /dev/null +++ b/tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.reference @@ -0,0 +1,20 @@ +-- { echo } + +SELECT count() FROM t_map_keys WHERE m['k'] = ''; +5 +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS Nullable(String)); +5 +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS LowCardinality(String)); +5 +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))); +5 +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))) SETTINGS use_skip_indexes = 0; +5 +SELECT count() FROM t_map_values WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))); +5 +SELECT count() FROM t_map_values WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))) SETTINGS use_skip_indexes = 0; +5 +SELECT count() FROM t_map_keys_tokenbf WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))); +5 +SELECT count() FROM t_map_keys_tokenbf WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))) SETTINGS use_skip_indexes = 0; +5 diff --git a/tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.sql b/tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.sql new file mode 100644 index 000000000000..0d5529aaac3e --- /dev/null +++ b/tests/queries/0_stateless/05241_bloom_filter_map_default_value_low_cardinality_nullable_constant.sql @@ -0,0 +1,40 @@ +-- `m['k']` returns the default value for rows without the key `k`, so an index on the map cannot be used +-- when the constant equals that default. The default must be taken from the value of the constant, not from +-- its type: the default of `LowCardinality(Nullable(String))` is NULL, while the constant is ''. + +DROP TABLE IF EXISTS t_map_keys; +DROP TABLE IF EXISTS t_map_values; +DROP TABLE IF EXISTS t_map_keys_tokenbf; + +CREATE TABLE t_map_keys (id UInt64, m Map(String, String), INDEX idx mapKeys(m) TYPE bloom_filter GRANULARITY 1) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; + +CREATE TABLE t_map_values (id UInt64, m Map(String, String), INDEX idx mapValues(m) TYPE bloom_filter GRANULARITY 1) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; + +CREATE TABLE t_map_keys_tokenbf (id UInt64, m Map(String, String), INDEX idx mapKeys(m) TYPE tokenbf_v1(512, 3, 0) GRANULARITY 1) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; + +INSERT INTO t_map_keys SELECT number, if(number % 2 = 0, map('k', 'v'), map('x', 'v')) FROM numbers(10); +INSERT INTO t_map_values SELECT number, if(number % 2 = 0, map('k', 'v'), map('x', 'v')) FROM numbers(10); +INSERT INTO t_map_keys_tokenbf SELECT number, if(number % 2 = 0, map('k', 'v'), map('x', 'v')) FROM numbers(10); + +-- { echo } + +SELECT count() FROM t_map_keys WHERE m['k'] = ''; +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS Nullable(String)); +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS LowCardinality(String)); +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))); +SELECT count() FROM t_map_keys WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))) SETTINGS use_skip_indexes = 0; + +SELECT count() FROM t_map_values WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))); +SELECT count() FROM t_map_values WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))) SETTINGS use_skip_indexes = 0; + +SELECT count() FROM t_map_keys_tokenbf WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))); +SELECT count() FROM t_map_keys_tokenbf WHERE m['k'] = CAST('' AS LowCardinality(Nullable(String))) SETTINGS use_skip_indexes = 0; + +-- { echoOff } + +DROP TABLE t_map_keys; +DROP TABLE t_map_values; +DROP TABLE t_map_keys_tokenbf; diff --git a/tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.reference b/tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.reference new file mode 100644 index 000000000000..674e55afe3ac --- /dev/null +++ b/tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.reference @@ -0,0 +1,40 @@ +-- { echo } + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(s, 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(s, 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(CAST(s, 'Nullable(String)'), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(CAST(s, 'Nullable(String)'), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toLowCardinality(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toLowCardinality(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAnyTokens(toNullable(s), 'Foo Qux') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAnyTokens(toNullable(s), 'Foo Qux') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAllTokens(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAllTokens(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasPhrase(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasPhrase(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab_postprocessor WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2] +SELECT arraySort(groupArray(id)) FROM tab_postprocessor WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2] +-- A part without the materialized index evaluates the row-level function. +INSERT INTO tab SETTINGS materialize_skip_indexes_on_insert = 0 VALUES (4, 'Foo qux'), (5, 'foo qux'); +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +[1,2,4,5] +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; +[1,2,4,5] diff --git a/tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.sql b/tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.sql new file mode 100644 index 000000000000..beec5f63dc11 --- /dev/null +++ b/tests/queries/0_stateless/05242_text_index_preprocessor_lossless_conversion_haystack.sql @@ -0,0 +1,73 @@ +-- Tags: no-parallel-replicas +-- Tag no-parallel-replicas -- direct read is not compatible with parallel replicas + +-- The text index is analyzed on the expression under lossless conversions of the haystack, e.g. `s` in +-- `hasToken(toNullable(s), 'Foo')`, and its preprocessor is applied to the needle. The row-level function has +-- to apply the preprocessor to the unwrapped haystack as well, so the result doesn't depend on whether the +-- index is read directly or the part has the index materialized. + +SET enable_analyzer = 1; +SET use_skip_indexes = 1; +SET use_skip_indexes_on_data_read = 1; + +DROP TABLE IF EXISTS tab; +DROP TABLE IF EXISTS tab_postprocessor; + +CREATE TABLE tab +( + id UInt64, + s String, + INDEX idx(s) TYPE text(tokenizer = splitByNonAlpha, preprocessor = lower(s)) +) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; + +CREATE TABLE tab_postprocessor +( + id UInt64, + s String, + INDEX idx(s) TYPE text(tokenizer = splitByNonAlpha, preprocessor = lower(s), postprocessor = if(s = 'bar', '', s)) +) +ENGINE = MergeTree ORDER BY id SETTINGS index_granularity = 1; + +SYSTEM STOP MERGES tab; +SYSTEM STOP MERGES tab_postprocessor; + +INSERT INTO tab VALUES (1, 'Foo bar'), (2, 'foo bar'), (3, 'baz'); +INSERT INTO tab_postprocessor VALUES (1, 'Foo bar'), (2, 'foo bar'), (3, 'baz'); + +-- { echo } + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(s, 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(s, 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(CAST(s, 'Nullable(String)'), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(CAST(s, 'Nullable(String)'), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toLowCardinality(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toLowCardinality(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAnyTokens(toNullable(s), 'Foo Qux') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAnyTokens(toNullable(s), 'Foo Qux') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAllTokens(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasAllTokens(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasPhrase(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasPhrase(toNullable(s), 'Foo Bar') SETTINGS query_plan_direct_read_from_text_index = 0; + +SELECT arraySort(groupArray(id)) FROM tab_postprocessor WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab_postprocessor WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; + +-- A part without the materialized index evaluates the row-level function. +INSERT INTO tab SETTINGS materialize_skip_indexes_on_insert = 0 VALUES (4, 'Foo qux'), (5, 'foo qux'); + +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 1; +SELECT arraySort(groupArray(id)) FROM tab WHERE hasToken(toNullable(s), 'Foo') SETTINGS query_plan_direct_read_from_text_index = 0; + +-- { echoOff } + +DROP TABLE tab; +DROP TABLE tab_postprocessor; From 0914f104970876bd4093ab4dc652b3d78ebd24b7 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Thu, 1 Oct 2026 22:51:07 +0000 Subject: [PATCH 101/185] Backport #122272 to 26.8: Fix `CREATE OR REPLACE` bypassing the drop privilege of the replaced table via `Replicated` databases and `ON CLUSTER` --- src/Interpreters/InterpreterCreateQuery.cpp | 27 +++++++++++---- ...or_replace_remote_drop_privilege.reference | 4 +++ ...create_or_replace_remote_drop_privilege.sh | 33 +++++++++++++++++++ 3 files changed, 58 insertions(+), 6 deletions(-) create mode 100644 tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.reference create mode 100755 tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.sh diff --git a/src/Interpreters/InterpreterCreateQuery.cpp b/src/Interpreters/InterpreterCreateQuery.cpp index bac302032c8a..19de34177dd7 100644 --- a/src/Interpreters/InterpreterCreateQuery.cpp +++ b/src/Interpreters/InterpreterCreateQuery.cpp @@ -1668,6 +1668,16 @@ bool isReplicated(const ASTStorage & storage) return storage_name.starts_with("Replicated") || storage_name.starts_with("Shared"); } +/// The drop privilege matching the kind of an existing table. +AccessType getDropAccessType(const IStorage & table) +{ + if (table.isView()) + return AccessType::DROP_VIEW; + if (table.isDictionary()) + return AccessType::DROP_DICTIONARY; + return AccessType::DROP_TABLE; +} + } BlockIO InterpreterCreateQuery::createTable(ASTCreateQuery & create) @@ -2840,12 +2850,7 @@ BlockIO InterpreterCreateQuery::doCreateOrReplaceTable(ASTCreateQuery & create, { /// The replaced table is dropped after the swap, under an internal temporary name that /// grants cannot cover, so check the drop privilege for its kind here, on its real name. - AccessType drop_access = AccessType::DROP_TABLE; - if (to_drop->isView()) - drop_access = AccessType::DROP_VIEW; - else if (to_drop->isDictionary()) - drop_access = AccessType::DROP_DICTIONARY; - current_context->checkAccess(drop_access, to_drop_id); + current_context->checkAccess(getDropAccessType(*to_drop), to_drop_id); to_drop->checkTableSizeBelowDropLimit(current_context); } }); @@ -3510,6 +3515,16 @@ AccessRightsElements InterpreterCreateQuery::getRequiredAccess() const } } + /// Replicated and ON CLUSTER replays run with full access, so the drop privilege for the replaced + /// table's kind must be required here, on its real name, while the query still runs as the user. + if ((create.replace_table || create.create_or_replace || create.replace_view) && !create.isTemporary()) + { + String database_name = getContext()->resolveDatabase(create.getDatabase()); + if (auto database = DatabaseCatalog::instance().tryGetDatabase(database_name)) + if (auto table = database->tryGetTable(create.getTable(), getContext())) + required_access.emplace_back(getDropAccessType(*table), database_name, create.getTable()); + } + if (create.targets) { for (const auto & target : create.targets->targets) diff --git a/tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.reference b/tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.reference new file mode 100644 index 000000000000..9971dc825b31 --- /dev/null +++ b/tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.reference @@ -0,0 +1,4 @@ +ACCESS_DENIED +ACCESS_DENIED +Dictionary +Dictionary diff --git a/tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.sh b/tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.sh new file mode 100755 index 000000000000..975c23566f92 --- /dev/null +++ b/tests/queries/0_stateless/05238_create_or_replace_remote_drop_privilege.sh @@ -0,0 +1,33 @@ +#!/usr/bin/env bash +# Tags: zookeeper +# `CREATE OR REPLACE` sent to a Replicated database or ON CLUSTER is replayed with full access, so the drop +# privilege for the replaced table's kind must be checked on the initiator, as the user, before it is sent. + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +user="user_${CLICKHOUSE_TEST_UNIQUE_NAME}" +atomic_db="atomic_${CLICKHOUSE_DATABASE}" +repl_db="repl_${CLICKHOUSE_DATABASE}" + +${CLICKHOUSE_CLIENT} --distributed_ddl_output_mode=none --query " +CREATE DATABASE ${atomic_db} ENGINE = Atomic; +CREATE DATABASE ${repl_db} ENGINE = Replicated('/test/${CLICKHOUSE_TEST_ZOOKEEPER_PREFIX}/repl', '1', '1'); +CREATE DICTIONARY ${atomic_db}.d (k UInt64) PRIMARY KEY k SOURCE(NULL()) LAYOUT(FLAT()) LIFETIME(0); +CREATE DICTIONARY ${repl_db}.d (k UInt64) PRIMARY KEY k SOURCE(NULL()) LAYOUT(FLAT()) LIFETIME(0); +CREATE USER ${user} IDENTIFIED WITH plaintext_password BY '${user}'; +GRANT SELECT, CREATE VIEW, DROP VIEW ON ${atomic_db}.* TO ${user}; +GRANT SELECT, CREATE VIEW, DROP VIEW ON ${repl_db}.* TO ${user}; +GRANT CLUSTER ON *.* TO ${user}; +" + +${CLICKHOUSE_CLIENT} --user "${user}" --password "${user}" --query "CREATE OR REPLACE VIEW ${repl_db}.d AS SELECT 1" 2>&1 | grep -Fo ACCESS_DENIED | uniq +${CLICKHOUSE_CLIENT} --user "${user}" --password "${user}" --query "CREATE OR REPLACE VIEW ${atomic_db}.d ON CLUSTER test_shard_localhost AS SELECT 1" 2>&1 | grep -Fo ACCESS_DENIED | uniq + +${CLICKHOUSE_CLIENT} --query " +SELECT engine FROM system.tables WHERE database IN ('atomic_$CLICKHOUSE_DATABASE', 'repl_$CLICKHOUSE_DATABASE') AND name = 'd'; +DROP DATABASE ${atomic_db}; +DROP DATABASE ${repl_db}; +DROP USER ${user}; +" From 0e9782821f827ac5925d063f812beb2f467c2c08 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 2 Oct 2026 00:26:21 +0000 Subject: [PATCH 102/185] Backport #117403 to 26.8: Fix ssl_certificate authentication bypass via embedded NUL byte in certificate CN --- src/Access/Authentication.cpp | 5 ++ src/Common/Crypto/X509Certificate.cpp | 47 ++++++++---- .../certs/client13-cert.pem | 31 ++++++++ .../certs/client13-key.pem | 51 +++++++++++++ .../certs/client14-cert.pem | 31 ++++++++ .../certs/client14-key.pem | 51 +++++++++++++ .../certs/generate_certs.sh | 73 +++++++++++++++++++ .../test_ssl_cert_authentication/test.py | 35 +++++++++ 8 files changed, 310 insertions(+), 14 deletions(-) create mode 100644 tests/integration/test_ssl_cert_authentication/certs/client13-cert.pem create mode 100644 tests/integration/test_ssl_cert_authentication/certs/client13-key.pem create mode 100644 tests/integration/test_ssl_cert_authentication/certs/client14-cert.pem create mode 100644 tests/integration/test_ssl_cert_authentication/certs/client14-key.pem diff --git a/src/Access/Authentication.cpp b/src/Access/Authentication.cpp index 3b5e4424239a..a0bcd3479bec 100644 --- a/src/Access/Authentication.cpp +++ b/src/Access/Authentication.cpp @@ -298,6 +298,11 @@ namespace for (const auto & certificate_subject : ssl_certificate_credentials->getSSLCertificateSubjects().at(type)) { + // Subjects are extracted with their exact bytes, so an embedded NUL byte survives. No valid + // hostname or URI contains one, and '*' must not match a span like "evil\0" in + // "evil\0.corp.example.com", so such a subject never matches a wildcard. + if (certificate_subject.contains('\0')) + continue; // Checked before the substr below so its length cannot underflow when prefix and suffix overlap. if (certificate_subject.size() < prefix.size() + suffix.size()) continue; diff --git a/src/Common/Crypto/X509Certificate.cpp b/src/Common/Crypto/X509Certificate.cpp index a2009ddf3f4f..8efa49f8e527 100644 --- a/src/Common/Crypto/X509Certificate.cpp +++ b/src/Common/Crypto/X509Certificate.cpp @@ -173,26 +173,45 @@ std::string X509Certificate::subjectName() const return buffer; } +/// Extract the value of the first entry with the given NID from an X509 name as a length-delimited +/// string. We read the ASN1_STRING bytes directly instead of X509_NAME_get_text_by_NID because that +/// function copies into a fixed C buffer and NUL-terminates: an embedded NUL byte (e.g. a CN of +/// "admin\0.evil.com") would be silently truncated to "admin", letting a certificate impersonate a +/// different subject during authentication. Preserving the exact bytes makes such a value compare +/// unequal to any NUL-free configured subject, and also avoids silent truncation of long names. +static std::string extractNameEntry(X509_NAME * name, uint nid) +{ + if (!name) + return {}; + + const int index = X509_NAME_get_index_by_NID(name, static_cast(nid), -1); + if (index < 0) + return {}; + + const X509_NAME_ENTRY * entry = X509_NAME_get_entry(name, index); + if (!entry) + return {}; + + const ASN1_STRING * data = X509_NAME_ENTRY_get_data(entry); + if (!data) + return {}; + + const unsigned char * bytes = ASN1_STRING_get0_data(data); + const int length = ASN1_STRING_length(data); + if (!bytes || length < 0) + return {}; + + return std::string(reinterpret_cast(bytes), static_cast(length)); +} + std::string X509Certificate::issuerName(uint nid) const { - if (X509_NAME * issuer = X509_get_issuer_name(certificate)) - { - char buffer[X509Certificate::NAME_BUFFER_SIZE]; - if (X509_NAME_get_text_by_NID(issuer, nid, buffer, sizeof(buffer)) >= 0) - return std::string(buffer); - } - return {}; + return extractNameEntry(X509_get_issuer_name(certificate), nid); } std::string X509Certificate::subjectName(uint nid) const { - if (X509_NAME * subj = X509_get_subject_name(certificate)) - { - char buffer[X509Certificate::NAME_BUFFER_SIZE]; - if (X509_NAME_get_text_by_NID(subj, nid, buffer, sizeof(buffer)) >= 0) - return std::string(buffer); - } - return {}; + return extractNameEntry(X509_get_subject_name(certificate), nid); } std::string X509Certificate::commonName() const diff --git a/tests/integration/test_ssl_cert_authentication/certs/client13-cert.pem b/tests/integration/test_ssl_cert_authentication/certs/client13-cert.pem new file mode 100644 index 000000000000..82687debfcde --- /dev/null +++ b/tests/integration/test_ssl_cert_authentication/certs/client13-cert.pem @@ -0,0 +1,31 @@ +-----BEGIN CERTIFICATE----- +MIIFPzCCAyegAwIBAgIUb8W/doRsWnXgJSqDgDz8Yb+cLVwwDQYJKoZIhvcNAQEL +BQAwUjELMAkGA1UEBhMCUlUxEzARBgNVBAgMClNvbWUtU3RhdGUxITAfBgNVBAoM +GEludGVybmV0IFdpZGdpdHMgUHR5IEx0ZDELMAkGA1UEAwwCY2EwHhcNMjAwMTAx +MDAwMDAwWhcNMjkxMjI5MDAwMDAwWjBhMQswCQYDVQQGEwJSVTETMBEGA1UECAwK +U29tZS1TdGF0ZTEhMB8GA1UECgwYSW50ZXJuZXQgV2lkZ2l0cyBQdHkgTHRkMRow +GAYDVQQDDBFjbGllbnQxAC5ldmlsLmNvbTCCAiIwDQYJKoZIhvcNAQEBBQADggIP +ADCCAgoCggIBAOMBJ2yqK72eew557aeGID0RXFzVs6lLTN4/71C4/309gAcIoqsw +Phulwzj3IfH7v2PPntsROxGtRCVS5LWarXUTEviEaiWUp1MwoFlrhn2Dm78hIgYC +7NrdtI48tb9t9y6eTio99WdIiXGY2KNlctfnTyCxKeanZi7JZUmb9VW95gh/5dl3 +sajZPjUDrFxqUQbzsSzbgElFkjvOzN5xEoUM1qbjHcyML7aKweXJKhSswdi4K3oA +v/FbFleJQgoRPB43JDqc5YjpPARKJlt5VZqpkgiKFAm0R+o4TYeCBd0kZFtL+7Y3 +TO27ZdQiOFKR/DINPCdA3/nxtQ7GJ17g/nwXLQRKUDxXain7jo+87OHghMA8WfwQ +lMSkQ+sqJdxA/VunQEHNCljw64BHeuU6CqQDU9eEnFNq9+3Qb0INiMRn6EGj20LK +5+q/fJqGm/N8BcuoYHUymbtojAjnJmCIWwvTzo7OYaG84nRwOPkZZvcSOrzn2CbT +zhmXHiSnP7ZAL/r24ZW5oikjw6Ie40iqQ2MyknTv3Ko1LWIKkjAbMcE6fo3SlzMA +fVsm9wAyVa/UyjO1TcnPNNta1TdqSgdRsEPDvR/IMPQYJ7Be0H0SZLj2/v8ZmHmZ +KVfGKdWnFBQKiTSHVEBHBdXADcXi2grYjnOBKnQ7H2UQK9jfGMu4FPYlAgMBAAEw +DQYJKoZIhvcNAQELBQADggIBAF4piZQxvyxwfhQxmNbyc8EtUo5qBIWE026e5Hsl +4QX3glKhAo+xLEQbKAImsMdXXYNdGaBJDTEZRbGSvyEgVa8e4YF0FP+CkFDcM0oF +k7YUgQHkZvlR/lsd5idMM7JJ1WXnux6jay9hpF9GEi2QMB2QkwuRwBHkgtjEun8C +qdyvApSecY6BXQllAJyoalSVR4ZzaMGVgSGL99RKyn320icevO6WDOVocEiohJTK +KOwf8q9G6riKWsRbNNPhkt9QwZzC0/RuKV6r25QR7H+GSjmllcBnbyqW3+DGGuJL +iMtKrCmx4xHedvDXRxi8ZxsfIrWVI7f5hgcWCXFWFMeqkU9nD3TGJrAuamBBQMZn +aC3q9AyVudroJs/1ojrSD92+H3qSzhqoLZ4bd531ShPasQ/EHIiOeN794j3Rxo0d +ituD2LSboe2Zr4TNhi7ou/dSfDXPEwJSb5V1Xgp9OyhQCKnFa+RkbQVtME5dVBMR +oO4HMTqaCgYSzCKhrxIO/vid2p6NQp7ZI8V2hAhUR4aBwnWFFspmuWaXpZqXxE6+ +a46rTA4HLpQ7TKwbi4fS2zWSh64nCHmv72GOHQ86dlXDNvfqLo8vyz79chKPo+31 +fsTZJks9wBzeSsq0tqAWflN/hQQCZJT1IpIc5VwU45FLSraKBGYRpCP4FnciXUuJ +RiyF +-----END CERTIFICATE----- diff --git a/tests/integration/test_ssl_cert_authentication/certs/client13-key.pem b/tests/integration/test_ssl_cert_authentication/certs/client13-key.pem new file mode 100644 index 000000000000..452e63223383 --- /dev/null +++ b/tests/integration/test_ssl_cert_authentication/certs/client13-key.pem @@ -0,0 +1,51 @@ +-----BEGIN RSA PRIVATE KEY----- +MIIJKAIBAAKCAgEA4wEnbKorvZ57Dnntp4YgPRFcXNWzqUtM3j/vULj/fT2ABwii +qzA+G6XDOPch8fu/Y8+e2xE7Ea1EJVLktZqtdRMS+IRqJZSnUzCgWWuGfYObvyEi +BgLs2t20jjy1v233Lp5OKj31Z0iJcZjYo2Vy1+dPILEp5qdmLsllSZv1Vb3mCH/l +2XexqNk+NQOsXGpRBvOxLNuASUWSO87M3nEShQzWpuMdzIwvtorB5ckqFKzB2Lgr +egC/8VsWV4lCChE8HjckOpzliOk8BEomW3lVmqmSCIoUCbRH6jhNh4IF3SRkW0v7 +tjdM7btl1CI4UpH8Mg08J0Df+fG1DsYnXuD+fBctBEpQPFdqKfuOj7zs4eCEwDxZ +/BCUxKRD6yol3ED9W6dAQc0KWPDrgEd65ToKpANT14ScU2r37dBvQg2IxGfoQaPb +Qsrn6r98moab83wFy6hgdTKZu2iMCOcmYIhbC9POjs5hobzidHA4+Rlm9xI6vOfY +JtPOGZceJKc/tkAv+vbhlbmiKSPDoh7jSKpDYzKSdO/cqjUtYgqSMBsxwTp+jdKX +MwB9Wyb3ADJVr9TKM7VNyc8021rVN2pKB1GwQ8O9H8gw9BgnsF7QfRJkuPb+/xmY +eZkpV8Yp1acUFAqJNIdUQEcF1cANxeLaCtiOc4EqdDsfZRAr2N8Yy7gU9iUCAwEA +AQKCAgAhsUz9M2q866tUaY4iXxAk5ZFBvtQ6NztzAfv5YrlFCPOaLP1xOX6rvYLd +eczCLP0/Xr8IwSTKdF7uHUC7Svhzgn++fsTUdse5Bg7JXDdregphUr/JMvZIa+n9 +xzlV7t07wEerZw59Txuwo8uFFlchKtX97QhiPDa4kLAlve4AKmiHRW3DmnY2MKcA +2Qk5nuwPg0Go/fFWBBX20N8WevjmjOfFKxDZXctCeujZkW3bVg2K0ZcJC4McWx/U +oUI52qIDfWgKrQ6NL3T20VyibL64bx0GLccDQ7dS0RrXbISF9Kin5lNiEbuAzjTd +yLB3DO+TgY8L58n2lUSsKzFdGI7yFjJTr6yzmKIayS11c00TCBQ0w/l+oCWOhrvM +zn/k3mrORcRA1aqX5mcqAdzs6hYFIyFbKXNH8hlE7JnUUbcvR76WbzMl6IxPxMT+ +ikbTIQnG+T2PMgmaxVbe9YhyHT2YbMHF7igocz5mI0PHsdNaB1IBSV1G2Knq8QJi +rbYUdCShTJksrNIXghyjbCPx5VIo05fx8NleCuNND18WPs32tI2rceLlF3KSk8GF +moCnPJFTtEgk5IvzWavOGO+abkU1A9xGOhNHCMZVxOca/ZGlhHSRBCuhagpLPx9e +QprxfFVIFG8jNrN0/Mhgv3ivKmbYZuv03wdcWDtKCdMrxkah9wKCAQEA8Xp+V4zk +Ta4nqkjL4TTv+uczszGt/Kb5OYZyb1EzUseabRyOFkb2iMUf+Ol/92UMBlh/FzTa +RFHE1lRed45l8Jhgsbaq9r8gmJZ1WZM2IUByUI++z+USlEtG8O+8wlniyTn9Fc+n +y+wkZwSeJSLWZAWIzeygJs/ceKG+4TpnLZnC2rQqOoKGc+0Dimc1E5HV3mOkhwe1 +CtiYOUvgNf0rEV7E5D7smjpAeiJ8xPhYRg0XXLztN+9rQA+6AvwpbI3VnyAKbPUQ +/J+ZVg0cVMH7pTDXMj0jB2wevrIa7xj2Pn4w/jR8i/tNjFauMMvnofr6RW6KqYk/ +opXhfSE1Pk9w6wKCAQEA8KfWOQEmKlC0KJrN1/D1ZchNZf57pNFtjyWdjLUah05k +a0QLUe1yme7lTw59tsFIfrd/ozeiWKSvhum9/mUNvwBgPjnqp73pmrEEu1bgQ8Ib +cCm55GAmBWlNMO52iD+EPdsYvj+KgAWxu6W8B6XsyjoVLBYqHwNzitZlW9D1PAbY +KWBppmntVAhj7oNvk40uWEFIuTU0o0RIrvcYmX795sr2PY69ozOdiY6UziR4oRAs +DhJ+0a8Sjct0/RMq+sOpBONrpOBHqJzoCX+/alOjbJ8FYIMuc6OU2cNtcxjL3tcZ +cJKInsMl7bMYhi0SIEnttuAgh6mzD4Iox1WX3hbxLwKCAQEAySWPHmCnQCnEsqzG +38aX7DkwsKC+XEm+KnPa2O0bwmWwNhlmJhpgfBcwBciDJtHODW8kFnGZKvWt8BcB +RbehJKPZT89oY/dbMJ+MCtx8Z4BmeML0X/ph2pNF+abJZl83cQVT0xpRnKUMwZ9w +GGEHkvOlPFtSIGJfNUEOXlCm565ASKtwzaIyW1hf7acA6Fc/fmsj/rKl1O4NBxU0 +I+TKPkLh4Xqk3eeOE+6sXeq8pUV8Y2ygcUqG3Sr8eEYSP3F6M32eEZql1rkfIjXz +loqrkrO3XgrdJe6jQZfcpbP2UqINL4MLCBOCSldd7Bm9zgjg9nsZGBXSox0UYoYJ +8uh7lwKCAQA+I3gi+/L41iHOojooWeVjRJcHkPAoHJNndNT7cf/JlCpFsCokG2WN +7at0AE/hkoK/hW4FnOXkcZGJCm2udDVabiRrrNS0P0tEUBTisonxtPsUuRFwsIrg +ttHhopEkmRHyTtJSvWFrsQy1YRPt/Z/oj5rL6WUy4NdCsB032fqYZ0QFWwmsmIlZ +O1liSrnSpY+j5id6+wv+ZDFITDEj4TB6GUn/lw3MSBWTYSd3Gt+y5tQZRhlM1yG5 +TyGD/yEH4uGPi5FN30NhfDJF0aCBOdtmvqDKzNR/s2tJ1zY5k9uATJYbBRsVs1nl +yGq6qoSVpcEliTWdEepURM12utkd1VqtAoIBAFLriLD2AF0+ayheiDFgsbRo78V3 +dygUPBZi7b6nPWv3gugnwvWrTFaCAndACxNZPWNwEcING2Utjp8PTvPTzad9yibl +3xqNSKR9gaqmDYPEY1DwYGCpHRc969PIYcyK6z3dICj1ySEo0RwTa+Ab0OHeRWzS +ROnmmy9a9Vpx7KyMZcmOGhSXL3q7nr2imI+qZ4tGiEJts85d2fXAVhGeBOagzy5n +uD00wYzBc8rZ9u+d7ZOyQetgoJBnhyQNzW7IBCN+Z3EWW5lPlHQJpTXjHCdXgiZU +28pX/4J/mIcaURA0xhexuuLqTEjLlFoAnjNjWCCXv0u5uP8Ba35g4BSAz0o= +-----END RSA PRIVATE KEY----- diff --git a/tests/integration/test_ssl_cert_authentication/certs/client14-cert.pem b/tests/integration/test_ssl_cert_authentication/certs/client14-cert.pem new file mode 100644 index 000000000000..25c2687c0edf --- /dev/null +++ b/tests/integration/test_ssl_cert_authentication/certs/client14-cert.pem @@ -0,0 +1,31 @@ +-----BEGIN CERTIFICATE----- +MIIFazCCA1OgAwIBAgIUTgWFBpS4cJ/+Fgg5jAfOubkwokowDQYJKoZIhvcNAQEL +BQAwUjELMAkGA1UEBhMCUlUxEzARBgNVBAgMClNvbWUtU3RhdGUxITAfBgNVBAoM +GEludGVybmV0IFdpZGdpdHMgUHR5IEx0ZDELMAkGA1UEAwwCY2EwHhcNMjAwMTAx +MDAwMDAwWhcNMjkxMjI5MDAwMDAwWjBmMQswCQYDVQQGEwJSVTETMBEGA1UECAwK +U29tZS1TdGF0ZTEhMB8GA1UECgwYSW50ZXJuZXQgV2lkZ2l0cyBQdHkgTHRkMR8w +HQYDVQQDDBZldmlsAC5jb3JwLmV4YW1wbGUuY29tMIICIjANBgkqhkiG9w0BAQEF +AAOCAg8AMIICCgKCAgEAs6uGg+vk4GlmN+ZlJb9gUI6AS6TJVQvLCSw7RPHt0k8x +qMXeoHid3mmVIC7tWK6BDEbAWJB814cWzOviXy5GbuUZB5Yq1MR0bQV968PBa/jp +/X6eDAW3KDfvgsk2eBo+aZqbFh1EJqXzN04d255X2mlAGpwW97T+XhXa/ahxj1Ey +G5smL6jaybcrKj/Xce4qVgSQk0kVlun/35Rfd6kcFTdq+Laap3nO0JKBYwfcoeMU +vVfRZV0iaP7sbOpiZlv0+zFDDGX2Xi18VH2xeclICN4ar8K0h+VNE1uqf27r+3NA +iZbZphA2huN60KYP+IV09KirWhFM3Xcd5LlqbluRUwWxp4xXZNWFIa6AeSbuC+43 +hAa0vBfN9+CE03ifX6stqfUy/eiJsghuPEZOUXqJpBpz4VHpxXVqiLX11hrGw60S +rEQJ30SVahULl4v1/zhFaRgFnZE2yRdQG0h1AejA5OUykKTH5vAkfZRxPmL3hnBB +fg+iti93EFvdC2bXNMB04X+VkvF8CcqRUr2Ka7ZpfleJuO9ps2e0LBtjSJOLxYRA +N13RUMp9m/45F5iGE4QXdUFzOrasrclg8KfP0Zl+6/FQcygV4bLQywqm/zgCX1zC +SAb88JLEElIUmyMFT/cvst1uscy1fvoLseuKcz+9woAF1LA8XrZzMPRA22xSABMC +AwEAAaMlMCMwIQYDVR0RBBowGIIWZXZpbAAuY29ycC5leGFtcGxlLmNvbTANBgkq +hkiG9w0BAQsFAAOCAgEAFuyPgncpYKGYffQPoOWR2yTn6L56XxvQcFNrchyKkwAn +zsBQFLOX4T4z1FuIQTwwKiSRwz+hwuIfcK2++I1yyxgYzw8Cxw7uqpvRi6AgN9pk +LEsXbSOvqaQyFEukVFenbkIejBVexF74gk2LKJYnT7MRVgs/k8C5Lqy5lKpB/2Gk +7xnzVe6mLhdZrMRbtCmt+Pr2m5l+XMGucnoaNCK87zSIGEt8fCYNB//vQ61GeANv +hRYiWgdWcOCnl2ip+EWbq0+ZK2JvJ8Ih2d0S+Qtbhek8XQsuCoa5P3YjLyAfzhpk +/CZb+VbOH41qZ2uWi05xVE3IRbJBlRDnX/bXC4Nx2ojzOnWAqzmrJffsKal/04oP +BcA6dOzm+CvT7AlzImlRI/6ZEC6DlrXSkvm5yRNwdUhTZ5ZjQnSyZBF4lNY3u9Sa +G1ipxyR9wRcA7uac8cFv/MBAGP5godwi3WPvqC/7noGDbZqDxcsZOeVFdP+NFtFg +dilITLxlgaQimN2nxhPvAAz6EcJ2owFgJJk6J3f6ICq8gdbJzFq2uVJrKBBL35St +/aOpflDRhQkMdJwrCuI/+4NBcXG5fQywjH8H+9dvbUONIlttbwE2zZJgXtY0gb+A +kY8fmy5pWBYnYAB8lFdAZxW0Sx6ehkmLA/ExtH/7jz/GwnwS12/of6wSvQHj8so= +-----END CERTIFICATE----- diff --git a/tests/integration/test_ssl_cert_authentication/certs/client14-key.pem b/tests/integration/test_ssl_cert_authentication/certs/client14-key.pem new file mode 100644 index 000000000000..10d1b89a2701 --- /dev/null +++ b/tests/integration/test_ssl_cert_authentication/certs/client14-key.pem @@ -0,0 +1,51 @@ +-----BEGIN RSA PRIVATE KEY----- +MIIJKAIBAAKCAgEAs6uGg+vk4GlmN+ZlJb9gUI6AS6TJVQvLCSw7RPHt0k8xqMXe +oHid3mmVIC7tWK6BDEbAWJB814cWzOviXy5GbuUZB5Yq1MR0bQV968PBa/jp/X6e +DAW3KDfvgsk2eBo+aZqbFh1EJqXzN04d255X2mlAGpwW97T+XhXa/ahxj1EyG5sm +L6jaybcrKj/Xce4qVgSQk0kVlun/35Rfd6kcFTdq+Laap3nO0JKBYwfcoeMUvVfR +ZV0iaP7sbOpiZlv0+zFDDGX2Xi18VH2xeclICN4ar8K0h+VNE1uqf27r+3NAiZbZ +phA2huN60KYP+IV09KirWhFM3Xcd5LlqbluRUwWxp4xXZNWFIa6AeSbuC+43hAa0 +vBfN9+CE03ifX6stqfUy/eiJsghuPEZOUXqJpBpz4VHpxXVqiLX11hrGw60SrEQJ +30SVahULl4v1/zhFaRgFnZE2yRdQG0h1AejA5OUykKTH5vAkfZRxPmL3hnBBfg+i +ti93EFvdC2bXNMB04X+VkvF8CcqRUr2Ka7ZpfleJuO9ps2e0LBtjSJOLxYRAN13R +UMp9m/45F5iGE4QXdUFzOrasrclg8KfP0Zl+6/FQcygV4bLQywqm/zgCX1zCSAb8 +8JLEElIUmyMFT/cvst1uscy1fvoLseuKcz+9woAF1LA8XrZzMPRA22xSABMCAwEA +AQKCAgAEaIj8Y6VR/EQNyxFgQ7nRQC3VrU1jUM7CgttRbb4wEtFdGr3DojH9awnF +qGEac+2mp3XAtorZnu7oSEFdpH0F64kZro2OeuOAaUoVps/wHkNffOPT17AOxJCT +3OwBNmOho7F6cW1ipV+6U6hX4yK0sTBpdrr5iO9Uz6R35NIkehGIq93b/YCgwmXE +u5xFp1pSkfoaIwjskwE8Mx/Eh9mwi5OMVq6kvVBdvbp++4pmTnQL0UPKAOb/PIIA +ih+v80GniCXk//tzhBow2ISqQE4MKabt+REE5JNnjjA4wDf6C3Hh7lmYwX0VAi/Z +PrnVlzCvcBQEObhxFqMdIY+C9awzGrf0iJ1UGZms5IrcKwvEL7VsrQqO2COEv42K +KtbVNpRMYG5n+VwtylChmh2g6oX4wMLbYIykDfXZVv/FIjP3vgPxs19t/CRVUEer +x6GztwhAmKvp4WzerVH1xJ7QKcaIuy4cjhNLVUQCcjPbFbtHsHzvuScXHfppCM67 +6rpNQ5DrQ0Wj0a1kCal/57aKQQdD37k4aTTEoCbDJkbD10SoaljJYElHleB26xCV +AwbyXz6gwJ80CxbXS1uco3c86keDekrGV5f73XX53MZouRR90Ii/TJvvdGutXFkX +mpAIaOZlfeDSFMJoFaSpOj3UaY4RBwcXXDB69QpFMDe2ND0gwQKCAQEA6jw1G0dP +HwDYfQSMzTQkhle7hHqxPoyuooBe4Vl6MJ64S22tgvRLE+FuLfubR9tMBYYr/Ub4 +CmSBE/XMbNGpH31yct5I/8rc9OnRg0bcCw2dz/Q7b84LKas6fkBFaf1mMlPvH2zi +nwtS3Bw3Q6DAky1ra6rqjiZDtDCN+vQOh0zXVdOyVjApDBbngX/x8hVhFLs3SuwS +jm3P8YloEQ6aQ0wCjvsemJs+IpDlH7cHwYlgxBREqQ1wOKdo8IZ/MTTvlPOXW04F +pP995LCvtVacjhfOLaWdHTs442yuFvFBoPWFEvx2CGYTQe+dk8zmzLFdYVNpEvbV +dh/m1hMrA3lUdwKCAQEAxF1d+QuDanP39u+16u4+u61p3gzYv2f9m+bjViiYMv6N +kSDzdXrKidD1I+iEgIy1QjRfdHlQMCIpP8D4zuWCrNY57QKdV4bRKs0RSgpOXPv+ +FGeVf9taPIH/lUH8FUJ1gqfoLX+6W2RbZxcNPOSOf13o6jTjBbB6oVetljNjsnDS +us7CJe3+MDk20pO2+E49Q0TuyeALDPRSciSvJrVCVmoPs6rpu22z5zMCxd6+WztL +YBWfFM38QpIMUg1miz7hU7m7lQHlYE9/p9RSI5ePD9mgepPQN+bQg0fGQaD0tzad +MJr/BI0OvHLdwFQNLya66nWK7s85aD/UjDRO/A6kRQKCAQB1zGes21ToM6WsYeBp +xsJjqbWNb6K54UhmQwb0b+pqjzgB/xuW00L6sZGWoIW8QoZd9Ncknk9Z8qeToTb4 +twxF4PHw4Od3dM9ggEK0sasyB9wI3DwUA1xLzWgyXCJMpnqB7wJAHKNv9uLp/Wqx +oSOYIOx4DlG9wXKlKRIOVjUESFm3OSrj+355LP+qeez0oVnccjbhgA3pAULlpwPm +KCDenVhgDdyaROCfw5znMUY+R9eZZNQO7Mo2Q8Mby5gl6AhhMYw6B+gAzdjDbTRA +j1lWgJRZEoQMUl9OyLZYpWYrC66sGLlHigY/T8FAtniQEtbyfl9GgUpjCLIvkR49 +tgQLAoIBAF/WefD9D4y7QQDCifU5hmCvCIaZmogAxyR6EeaRNYdd+dYlUO27mnKd +C6gU6eabxjOjwBrmwp5bbepx0n2YQqj8fZURu51mbVwIbjHGyexUCPQIgky+0FHL +2OQOKmxt3VCBhq3+MwQ7/OhZtdpMasf7G5yDZ3H1akSouE4gkr4alp8aHmPIvlDm ++7zW32xdM0VLtYfN01blP//5q4qm2NO4PCWieyVBK5bhrK7KQfng/K7Onq/WwRH0 +mhLJ+4xmii8E7WqSXFMfOdy9ocFBTU+dFdf9oJhIDOil9Ts+xXFONHXukBy2g8Sy +A0zFORIUQxH/gGmBtjENRj2PoiUfOEkCggEBAMiBLM8L+k7ovGd7KVoeSgk7FjsV +5gq0GDwXGov9HEyeY/mmmuS5mfxZUYvRgzbpQutqy88MvbMsCS7VDtT9m0IFs8zR +qXIUQmOrYJuI37BrA16K5EDTj6wY/7DKRSM4guNqQjYnJ2bYBrXCZLWFUF/OVl3w +VF5reN7BCMOTXGc4aAeJzcmvhGU7M3WnGwL7yGkiX8BCDXo+ddLuf1SyDUttfrmo +Kbb4h4LAe2ATtRO0F9SfEetdmHUwO6iRJGQNQQWO7pWOjUStSry3Z8AM0uVDBRDG +Ol3DmKQlxbFAhLD62KkcWt30eNnNMlYaq8huKY4Y2juyi5ZQ31sDij4worA= +-----END RSA PRIVATE KEY----- diff --git a/tests/integration/test_ssl_cert_authentication/certs/generate_certs.sh b/tests/integration/test_ssl_cert_authentication/certs/generate_certs.sh index 19354a5a3e36..b8bbd34567a2 100755 --- a/tests/integration/test_ssl_cert_authentication/certs/generate_certs.sh +++ b/tests/integration/test_ssl_cert_authentication/certs/generate_certs.sh @@ -60,3 +60,76 @@ openssl x509 -req -days 36525 -in client_far_future-req.pem -CA ca-cert.pem -CAk # 6. Generate one more self-signed certificate and private key for using as wrong certificate (because it's not signed by CA) openssl req -newkey rsa:4096 -x509 -days 3650 -nodes -batch -keyout wrong-key.pem -out wrong-cert.pem -subj "/C=RU/ST=Some-State/O=Internet Widgits Pty Ltd/CN=client" + +# 7. Generate a CA-signed certificate whose CN carries an embedded NUL byte ("client1\0.evil.com"), +# to test that server-side CN extraction does not truncate at the NUL. User 'john' is configured with +# client1, so a server that truncated the CN at the NUL would extract +# "client1" and wrongly authenticate this certificate as 'john'; the full CN must be preserved so the +# match fails. openssl's CLI cannot place a NUL inside a -subj field, so we build this certificate +# with the 'cryptography' library instead. +python3 - <<'PY' +from cryptography import x509 +from cryptography.x509.oid import NameOID +from cryptography.hazmat.primitives import hashes, serialization +from cryptography.hazmat.primitives.asymmetric import rsa +import datetime + +with open("ca-key.pem", "rb") as f: + ca_key = serialization.load_pem_private_key(f.read(), password=None) +with open("ca-cert.pem", "rb") as f: + ca_cert = x509.load_pem_x509_certificate(f.read()) + +key = rsa.generate_private_key(public_exponent=65537, key_size=4096) +subject = x509.Name([ + x509.NameAttribute(NameOID.COUNTRY_NAME, "RU"), + x509.NameAttribute(NameOID.STATE_OR_PROVINCE_NAME, "Some-State"), + x509.NameAttribute(NameOID.ORGANIZATION_NAME, "Internet Widgits Pty Ltd"), + x509.NameAttribute(NameOID.COMMON_NAME, "client1\x00.evil.com"), +]) +now = datetime.datetime(2020, 1, 1) +cert = (x509.CertificateBuilder() + .subject_name(subject).issuer_name(ca_cert.subject) + .public_key(key.public_key()).serial_number(x509.random_serial_number()) + .not_valid_before(now).not_valid_after(now + datetime.timedelta(days=3650)) + .sign(ca_key, hashes.SHA256())) +with open("client13-key.pem", "wb") as f: + f.write(key.private_bytes(serialization.Encoding.PEM, serialization.PrivateFormat.TraditionalOpenSSL, serialization.NoEncryption())) +with open("client13-cert.pem", "wb") as f: + f.write(cert.public_bytes(serialization.Encoding.PEM)) +PY + +# 8. Generate a CA-signed certificate whose CN and DNS SAN both carry an embedded NUL byte +# ("evil\0.corp.example.com"), to test that wildcard matching rejects it. Users 'wildcard_cn' and +# 'wildcard_dns' are configured with '*.corp.example.com' and 'DNS:*.corp.example.com': without an +# explicit check, '*' would match the single "label" "evil\0" and authenticate this certificate. +python3 - <<'PY' +from cryptography import x509 +from cryptography.x509.oid import NameOID +from cryptography.hazmat.primitives import hashes, serialization +from cryptography.hazmat.primitives.asymmetric import rsa +import datetime + +with open("ca-key.pem", "rb") as f: + ca_key = serialization.load_pem_private_key(f.read(), password=None) +with open("ca-cert.pem", "rb") as f: + ca_cert = x509.load_pem_x509_certificate(f.read()) + +key = rsa.generate_private_key(public_exponent=65537, key_size=4096) +subject = x509.Name([ + x509.NameAttribute(NameOID.COUNTRY_NAME, "RU"), + x509.NameAttribute(NameOID.STATE_OR_PROVINCE_NAME, "Some-State"), + x509.NameAttribute(NameOID.ORGANIZATION_NAME, "Internet Widgits Pty Ltd"), + x509.NameAttribute(NameOID.COMMON_NAME, "evil\x00.corp.example.com"), +]) +now = datetime.datetime(2020, 1, 1) +cert = (x509.CertificateBuilder() + .subject_name(subject).issuer_name(ca_cert.subject) + .public_key(key.public_key()).serial_number(x509.random_serial_number()) + .not_valid_before(now).not_valid_after(now + datetime.timedelta(days=3650)) + .add_extension(x509.SubjectAlternativeName([x509.DNSName("evil\x00.corp.example.com")]), critical=False) + .sign(ca_key, hashes.SHA256())) +with open("client14-key.pem", "wb") as f: + f.write(key.private_bytes(serialization.Encoding.PEM, serialization.PrivateFormat.TraditionalOpenSSL, serialization.NoEncryption())) +with open("client14-cert.pem", "wb") as f: + f.write(cert.public_bytes(serialization.Encoding.PEM)) +PY diff --git a/tests/integration/test_ssl_cert_authentication/test.py b/tests/integration/test_ssl_cert_authentication/test.py index 977bc82ea276..118172905b9a 100644 --- a/tests/integration/test_ssl_cert_authentication/test.py +++ b/tests/integration/test_ssl_cert_authentication/test.py @@ -147,6 +147,18 @@ def test_native_fallback_to_password(): assert "AUTHENTICATION_FAILED" in str(err.value) +def test_native_cn_nul_byte_no_bypass(): + # Authentication bypass: client13's CN is "client1\0.evil.com" and user 'john' is configured with + # client1. If server-side CN extraction truncated at the embedded NUL + # byte, the CN would collapse to "client1" and the certificate would authenticate as user 'john'. + # The full CN must be preserved, so the match must fail. + with pytest.raises(Exception) as err: + execute_query_native( + instance, "SELECT currentUser()", user="john", cert_name="client13" + ) + assert "AUTHENTICATION_FAILED" in str(err.value) + + def get_ssl_context(cert_name): context = WrapSSLContextWithSNI(SSL_HOST, ssl.PROTOCOL_TLS_CLIENT) context.load_verify_locations(cafile=f"{SCRIPT_DIR}/certs/ca-cert.pem") @@ -221,6 +233,14 @@ def test_https_wrong_cert(): ) +def test_https_cn_nul_byte_no_bypass(): + # Same bypass as test_native_cn_nul_byte_no_bypass, over the HTTPS interface: client13's CN + # "client1\0.evil.com" must not be truncated to "client1" and authenticate as user 'john'. + with pytest.raises(Exception) as err: + execute_query_https("SELECT currentUser()", user="john", cert_name="client13") + assert "403" in str(err.value) + + def test_https_non_ssl_auth(): # Users with non-SSL authentication are allowed, in this case we can skip sending a client certificate at all (because "verificationMode" is set to "relaxed"). # assert execute_query_https("SELECT currentUser()", user="peter", enable_ssl_auth=False) == "peter\n" @@ -546,6 +566,21 @@ def test_x509_cn_wildcard_single_label(): assert "403" in str(err.value) +def test_x509_wildcard_nul_byte_no_bypass(): + # Authentication bypass: client14's CN and DNS SAN are both "evil\0.corp.example.com". Users + # 'wildcard_cn' and 'wildcard_dns' are configured with '*.corp.example.com' and + # 'DNS:*.corp.example.com'. The '*' must not match the span "evil\0", so both must fail. + for user in ["wildcard_cn", "wildcard_dns"]: + with pytest.raises(Exception) as err: + execute_query_native( + instance, "SELECT currentUser()", user=user, cert_name="client14" + ) + assert "AUTHENTICATION_FAILED" in str(err.value) + with pytest.raises(Exception) as err: + execute_query_https("SELECT currentUser()", user=user, cert_name="client14") + assert "403" in str(err.value) + + def test_x509_uri_san_wildcard_dot_in_segment(): # Non-regression: '.' separates labels for DNS/CN but is NOT a separator for URI SANs, # whose separator is '/'. A wildcard URI path segment may legitimately contain dots, so From 625bcff8eac7cb0a8e8a12508dc3f7d26d8328ae Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 2 Oct 2026 01:33:14 +0000 Subject: [PATCH 103/185] Backport #123099 to 26.8: Add more path validations --- .../DataLakes/Paimon/PaimonClient.cpp | 2 + .../DataLakes/Paimon/PaimonMetadata.cpp | 1 + .../ObjectStorage/DataLakes/Paimon/Utils.cpp | 14 +++++ .../ObjectStorage/DataLakes/Paimon/Utils.h | 1 + ...1_paimon_manifest_path_traversal.reference | 5 ++ .../05301_paimon_manifest_path_traversal.sh | 63 +++++++++++++++++++ 6 files changed, 86 insertions(+) create mode 100644 tests/queries/0_stateless/05301_paimon_manifest_path_traversal.reference create mode 100755 tests/queries/0_stateless/05301_paimon_manifest_path_traversal.sh diff --git a/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonClient.cpp b/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonClient.cpp index 50484b2e20bd..08fac3e0d656 100644 --- a/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonClient.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonClient.cpp @@ -298,6 +298,7 @@ PaimonSnapshot PaimonTableClient::getSnapshot(const std::pair & s std::pair, size_t> PaimonTableClient::getManifestMeta(String manifest_list_path, bool disable_filesystem_cache) { /// read manifest list file + Paimon::checkPathIsRelativeToTable(manifest_list_path, "manifest list"); auto context = getContext(); RelativePathWithMetadata relative_path(std::filesystem::path(table_location) / PAIMON_MANIFEST_DIR / manifest_list_path); auto read_settings = getPaimonMetadataReadSettings(disable_filesystem_cache); @@ -320,6 +321,7 @@ std::pair, size_t> PaimonTableClient::getMan PaimonManifest PaimonTableClient::getDataManifest(String manifest_path, const PaimonTableSchema & table_schema, const String & partition_default_name, bool disable_filesystem_cache) { + Paimon::checkPathIsRelativeToTable(manifest_path, "manifest"); String manifest_file_name(manifest_path.begin() + manifest_path.find_last_of('/') + 1, manifest_path.end()); if (manifest_file_name.starts_with("index-manifest-")) return {}; diff --git a/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonMetadata.cpp b/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonMetadata.cpp index b97affbdeff5..89e441c7ea7c 100644 --- a/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonMetadata.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Paimon/PaimonMetadata.cpp @@ -896,6 +896,7 @@ Strings PaimonMetadata::collectDataFilesFromManifests( auto manifest = getManifest(meta.file_name, snapshot_state->schema_id); for (const auto & entry : manifest->entries) { + Paimon::checkPathIsRelativeToTable(entry.file.file_name, "data file"); String file_path = (std::filesystem::path(persistent_components.table_path) / entry.file.bucket_path / entry.file.file_name); diff --git a/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.cpp b/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.cpp index a3885f3c55f6..bc1fef9cbf51 100644 --- a/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.cpp +++ b/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.cpp @@ -1,4 +1,6 @@ +#include #include +#include #include #include #include @@ -29,6 +31,7 @@ namespace ErrorCodes extern const int CANNOT_PRINT_FLOAT_OR_DOUBLE_NUMBER; extern const int BAD_ARGUMENTS; extern const int VALUE_IS_OUT_OF_RANGE_OF_DATA_TYPE; +extern const int PATH_ACCESS_DENIED; } } namespace Paimon @@ -371,6 +374,17 @@ String getBucketPath(const String & partition, Int32 bucket, const PaimonTableSc return bucket_path; } +void checkPathIsRelativeToTable(const String & path, std::string_view kind) +{ + const std::filesystem::path fs_path(path); + if (fs_path.has_root_path() || std::ranges::any_of(fs_path, [](const auto & component) { return component == ".."; })) + throw Exception( + ErrorCodes::PATH_ACCESS_DENIED, + "Paimon {} path `{}` must be relative to the table directory and must not contain `..`", + kind, + path); +} + String concatPath(std::initializer_list paths) { if (paths.size() == 1) diff --git a/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.h b/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.h index b01d8d33a9eb..fb1171b109be 100644 --- a/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.h +++ b/src/Storages/ObjectStorage/DataLakes/Paimon/Utils.h @@ -73,6 +73,7 @@ class PathEscape DB::Row getPartitionFields(const String & partition, const PaimonTableSchema & table_schema); String getBucketPath(const String & partition, Int32 bucket, const PaimonTableSchema & table_schema, const String & partition_default_name); String concatPath(std::initializer_list paths); +void checkPathIsRelativeToTable(const String & path, std::string_view kind); template void getValueFromJSON(T & t, const Poco::JSON::Object::Ptr & json, const String & key) diff --git a/tests/queries/0_stateless/05301_paimon_manifest_path_traversal.reference b/tests/queries/0_stateless/05301_paimon_manifest_path_traversal.reference new file mode 100644 index 000000000000..ea980fefbba0 --- /dev/null +++ b/tests/queries/0_stateless/05301_paimon_manifest_path_traversal.reference @@ -0,0 +1,5 @@ +INSIDE_TABLE +PATH_ACCESS_DENIED +PATH_ACCESS_DENIED +PATH_ACCESS_DENIED +PATH_ACCESS_DENIED diff --git a/tests/queries/0_stateless/05301_paimon_manifest_path_traversal.sh b/tests/queries/0_stateless/05301_paimon_manifest_path_traversal.sh new file mode 100755 index 000000000000..6ec04988ef6b --- /dev/null +++ b/tests/queries/0_stateless/05301_paimon_manifest_path_traversal.sh @@ -0,0 +1,63 @@ +#!/usr/bin/env bash +# Tags: no-fasttest + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +BASE_DIR="${CLICKHOUSE_USER_FILES_UNIQUE}/paimon_path_traversal" +rm -rf "${BASE_DIR}" +mkdir -p "${BASE_DIR}/outside" +echo "SECRET_OUTSIDE_TABLE" > "${BASE_DIR}/outside/secret.txt" + +# $1 - table name, $2 - manifest file name in the manifest list, $3 - data file name in the manifest +create_table() +{ + local table_dir="${BASE_DIR}/$1" + mkdir -p "${table_dir}/schema" "${table_dir}/snapshot" "${table_dir}/manifest" "${table_dir}/bucket-0" + echo "INSIDE_TABLE" > "${table_dir}/bucket-0/data.txt" + echo '{"version":3,"id":0,"highestFieldId":0,"partitionKeys":[],"primaryKeys":[],"options":{},"timeMillis":0,"fields":[{"id":0,"name":"data","type":"STRING NOT NULL"}]}' > "${table_dir}/schema/schema-0" + echo -n '1' > "${table_dir}/snapshot/LATEST" + echo '{"id":1,"schemaId":0,"baseManifestList":"manifest-list-1","deltaManifestList":"manifest-list-1","commitUser":"test","commitIdentifier":0,"commitKind":"APPEND","timeMillis":0}' > "${table_dir}/snapshot/snapshot-1" + + ${CLICKHOUSE_CLIENT} -q " + INSERT INTO FUNCTION file('${table_dir}/manifest/manifest-list-1', 'Avro', + '_FILE_NAME String, _FILE_SIZE Int64, _NUM_ADDED_FILES Int64, _NUM_DELETED_FILES Int64, + _PARTITION_STATS Tuple(_MAX_VALUES String, _MIN_VALUES String, _NULL_COUNTS Array(Int64)), _SCHEMA_ID Int64') + VALUES ('$2', 1, 1, 0, ('', '', [0]), 0)" + + ${CLICKHOUSE_CLIENT} -q " + INSERT INTO FUNCTION file('${table_dir}/manifest/manifest-1', 'Avro', + '_KIND Int32, _PARTITION String, _BUCKET Int32, _TOTAL_BUCKETS Int32, + _FILE Tuple(_FILE_NAME String, _FILE_SIZE Int64, _ROW_COUNT Int64, _MIN_KEY String, _MAX_KEY String, + _KEY_STATS Tuple(_MAX_VALUES String, _MIN_VALUES String, _NULL_COUNTS Array(Int64)), + _VALUE_STATS Tuple(_MAX_VALUES String, _MIN_VALUES String, _NULL_COUNTS Array(Int64)), + _MIN_SEQUENCE_NUMBER Int64, _MAX_SEQUENCE_NUMBER Int64, _SCHEMA_ID Int64, _LEVEL Int32, + _EXTRA_FILES Array(String), _CREATION_TIME Nullable(DateTime64(6)), _DELETE_ROW_COUNT Nullable(Int64), + _EMBEDDED_FILE_INDEX Nullable(String), _FILE_SOURCE Nullable(Int8), _VALUE_STATS_COLS Array(Int64))') + VALUES (0, unhex('00000000'), 0, 1, + ('$3', 13, 1, '', '', ('', '', [0]), ('', '', [0]), 0, 0, 0, 0, [], NULL, NULL, NULL, NULL, []))" +} + +read_table() +{ + ${CLICKHOUSE_CLIENT} -q "SELECT data FROM paimonLocal('${BASE_DIR}/$1', 'RawBLOB', 'data String')" 2>&1 \ + | grep -o -m1 -E "INSIDE_TABLE|SECRET_OUTSIDE_TABLE|PATH_ACCESS_DENIED" +} + +create_table valid 'manifest-1' 'data.txt' +read_table valid + +create_table data_file_dotdot 'manifest-1' '../../outside/secret.txt' +read_table data_file_dotdot + +create_table data_file_absolute 'manifest-1' "${BASE_DIR}/outside/secret.txt" +read_table data_file_absolute + +create_table manifest_dotdot '../../valid/manifest/manifest-1' 'data.txt' +read_table manifest_dotdot + +create_table manifest_absolute "${BASE_DIR}/valid/manifest/manifest-1" 'data.txt' +read_table manifest_absolute + +rm -rf "${BASE_DIR}" From 3d750e67d81a8b0e1e957e6374b5a181985cd3cf Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 2 Oct 2026 04:28:23 +0000 Subject: [PATCH 104/185] Backport #122868 to 26.8: Fix a TTL check crash on a constant `arrayMap` result, reject a constant nested column in `ColumnArray` --- src/Columns/ColumnArray.cpp | 58 ++------------- src/Columns/ColumnArray.h | 8 --- src/Columns/ColumnTuple.cpp | 7 +- src/Columns/tests/gtest_column_array.cpp | 25 +++++++ src/Columns/tests/gtest_column_tuple.cpp | 70 +++++++++++++++++++ src/Storages/TTLDescription.cpp | 3 +- ...l_array_map_constant_lambda_body.reference | 3 + ...292_ttl_array_map_constant_lambda_body.sql | 20 ++++++ 8 files changed, 129 insertions(+), 65 deletions(-) create mode 100644 tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.reference create mode 100644 tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.sql diff --git a/src/Columns/ColumnArray.cpp b/src/Columns/ColumnArray.cpp index e08fcc942e81..983e8647d420 100644 --- a/src/Columns/ColumnArray.cpp +++ b/src/Columns/ColumnArray.cpp @@ -59,6 +59,9 @@ ColumnArray::ColumnArray(MutableColumnPtr && nested_column, MutableColumnPtr && if (data) { + if (isColumnConst(*data)) + throw Exception(ErrorCodes::LOGICAL_ERROR, "ColumnArray cannot have ColumnConst as its nested column"); + Offset last_offset = offsets_data.empty() ? 0 : offsets_data.back(); /// This will also prevent possible overflow in offset. @@ -79,16 +82,14 @@ ColumnArray::ColumnArray(MutableColumnPtr && nested_column, MutableColumnPtr && "offsets_column is not monotonically increasing: the offset {} at position {} is greater than the next offset {}", *non_monotonic, non_monotonic - offsets_data.begin(), *(non_monotonic + 1)); #endif - - /** NOTE - * Arrays with constant value are possible and used in implementation of higher order functions (see FunctionReplicate). - * But in most cases, arrays with constant value are unexpected and code will work wrong. Use with caution. - */ } ColumnArray::ColumnArray(MutableColumnPtr && nested_column) : data(std::move(nested_column)) { + if (isColumnConst(*data)) + throw Exception(ErrorCodes::LOGICAL_ERROR, "ColumnArray cannot have ColumnConst as its nested column"); + if (!data->empty()) throw Exception(ErrorCodes::LOGICAL_ERROR, "Not empty data passed to ColumnArray, but no offsets passed"); @@ -638,13 +639,6 @@ bool ColumnArray::hasEqualOffsets(const ColumnArray & other) const } -ColumnPtr ColumnArray::convertToFullColumnIfConst() const -{ - /// It is possible to have an array with constant data and non-constant offsets. - /// Example is the result of expression: replicate('hello', [1]) - return ColumnArray::create(data->convertToFullColumnIfConst(), offsets); -} - void ColumnArray::getExtremes(Field & min, Field & max, size_t start, size_t end) const { min = Array(); @@ -1444,8 +1438,6 @@ ColumnPtr ColumnArray::replicate(const Offsets & replicate_offsets) const return replicateNumber(replicate_offsets); if (typeid_cast(data.get())) return replicateString(replicate_offsets); - if (typeid_cast(data.get())) - return replicateConst(replicate_offsets); if (typeid_cast(data.get())) return replicateNullable(replicate_offsets); if (typeid_cast(data.get())) @@ -1586,44 +1578,6 @@ ColumnPtr ColumnArray::replicateString(const Offsets & replicate_offsets) const } -ColumnPtr ColumnArray::replicateConst(const Offsets & replicate_offsets) const -{ - size_t col_size = size(); - if (col_size != replicate_offsets.size()) - throw Exception(ErrorCodes::SIZES_OF_COLUMNS_DOESNT_MATCH, "Size of offsets doesn't match size of column."); - - if (0 == col_size) - return cloneEmpty(); - - const Offsets & src_offsets = getOffsets(); - - auto res_column_offsets = ColumnOffsets::create(); - Offsets & res_offsets = res_column_offsets->getData(); - res_offsets.reserve_exact(replicate_offsets.back()); - - Offset prev_replicate_offset = 0; - Offset prev_data_offset = 0; - Offset current_new_offset = 0; - - for (size_t i = 0; i < col_size; ++i) - { - size_t size_to_replicate = replicate_offsets[i] - prev_replicate_offset; - size_t value_size = src_offsets[i] - prev_data_offset; - - for (size_t j = 0; j < size_to_replicate; ++j) - { - current_new_offset += value_size; - res_offsets.push_back(current_new_offset); - } - - prev_replicate_offset = replicate_offsets[i]; - prev_data_offset = src_offsets[i]; - } - - return ColumnArray::create(getData().cloneResized(current_new_offset), std::move(res_column_offsets)); -} - - ColumnPtr ColumnArray::replicateGeneric(const Offsets & replicate_offsets) const { size_t col_size = size(); diff --git a/src/Columns/ColumnArray.h b/src/Columns/ColumnArray.h index 1d4fade3dc32..50e68b18e551 100644 --- a/src/Columns/ColumnArray.h +++ b/src/Columns/ColumnArray.h @@ -133,7 +133,6 @@ class ColumnArray final : public COWHelper, ColumnArr size_t allocatedBytes() const override; void protect() override; ColumnPtr replicate(const Offsets & replicate_offsets) const override; - ColumnPtr convertToFullColumnIfConst() const override; void getExtremes(Field & min, Field & max, size_t start, size_t end) const override; bool hasEqualOffsets(const ColumnArray & other) const; @@ -244,13 +243,6 @@ class ColumnArray final : public COWHelper, ColumnArr /// Multiply the values if the nested column is ColumnString. The code is too complicated. ColumnPtr replicateString(const Offsets & replicate_offsets) const; - /** Non-constant arrays of constant values are quite rare. - * Most functions can not work with them, and does not create such columns as a result. - * An exception is the function `replicate` (see FunctionsMiscellaneous.h), which has service meaning for the implementation of lambda functions. - * Only for its sake is the implementation of the `replicate` method for ColumnArray(ColumnConst). - */ - ColumnPtr replicateConst(const Offsets & replicate_offsets) const; - /** The following is done by simply replicating of nested columns. */ ColumnPtr replicateTuple(const Offsets & replicate_offsets) const; diff --git a/src/Columns/ColumnTuple.cpp b/src/Columns/ColumnTuple.cpp index 7e178d368a11..ba512dc19290 100644 --- a/src/Columns/ColumnTuple.cpp +++ b/src/Columns/ColumnTuple.cpp @@ -21,7 +21,6 @@ namespace DB namespace ErrorCodes { - extern const int ILLEGAL_COLUMN; extern const int NOT_IMPLEMENTED; extern const int CANNOT_INSERT_VALUE_OF_DIFFERENT_SIZE_INTO_TUPLE; extern const int LOGICAL_ERROR; @@ -54,7 +53,7 @@ ColumnTuple::ColumnTuple(MutableColumns && mutable_columns) for (auto & column : mutable_columns) { if (isColumnConst(*column)) - throw Exception(ErrorCodes::ILLEGAL_COLUMN, "ColumnTuple cannot have ColumnConst as its element"); + throw Exception(ErrorCodes::LOGICAL_ERROR, "ColumnTuple cannot have ColumnConst as its element"); columns.push_back(std::move(column)); } @@ -70,7 +69,7 @@ ColumnTuple::Ptr ColumnTuple::create(const Columns & columns) for (const auto & column : columns) if (isColumnConst(*column)) - throw Exception(ErrorCodes::ILLEGAL_COLUMN, "ColumnTuple cannot have ColumnConst as its element"); + throw Exception(ErrorCodes::LOGICAL_ERROR, "ColumnTuple cannot have ColumnConst as its element"); auto column_tuple = ColumnTuple::create(columns[0]->size()); column_tuple->columns.assign(columns.begin(), columns.end()); @@ -85,7 +84,7 @@ ColumnTuple::Ptr ColumnTuple::create(const TupleColumns & columns) for (const auto & column : columns) if (isColumnConst(*column)) - throw Exception(ErrorCodes::ILLEGAL_COLUMN, "ColumnTuple cannot have ColumnConst as its element"); + throw Exception(ErrorCodes::LOGICAL_ERROR, "ColumnTuple cannot have ColumnConst as its element"); auto column_tuple = ColumnTuple::create(columns[0]->size()); column_tuple->columns = columns; diff --git a/src/Columns/tests/gtest_column_array.cpp b/src/Columns/tests/gtest_column_array.cpp index a9b55a582735..c2b95118b018 100644 --- a/src/Columns/tests/gtest_column_array.cpp +++ b/src/Columns/tests/gtest_column_array.cpp @@ -1,4 +1,5 @@ #include +#include #include #include @@ -23,6 +24,14 @@ ColumnArray::MutablePtr createArray(std::vector data_values, std::vector return ColumnArray::create(std::move(data), std::move(offsets)); } +/// One array of two rows of a constant: the offsets match the nested column, so only the constant is wrong. +ColumnArray::MutablePtr createArrayOverConst() +{ + auto offsets = ColumnArray::ColumnOffsets::create(); + offsets->getData().push_back(2); + return ColumnArray::create(ColumnConst::create(ColumnUInt64::create(1, 42), 2), std::move(offsets)); +} + } /// A ColumnArray is created from an already populated nested column and offsets, so its offsets @@ -64,6 +73,12 @@ TEST(ColumnArray, InconsistentOffsetsAreRejected) EXPECT_THROW(createArray({10}, {}), Exception); } +TEST(ColumnArray, ConstNestedColumnIsRejected) +{ + EXPECT_THROW(createArrayOverConst(), Exception); + EXPECT_THROW(ColumnArray::create(ColumnConst::create(ColumnUInt64::create(1, 42), 0)), Exception); +} + #endif /// A decreasing offset makes `sizeAt` underflow to a huge value even when the last offset matches @@ -86,4 +101,14 @@ TEST(ColumnArrayDeathTest, NonMonotonicOffsetsAreRejected) EXPECT_DEATH((createArray({10, 20, 30}, {3, 0, 3})), "not monotonically increasing"); } +TEST(ColumnArrayDeathTest, ConstNestedColumnIsRejected) +{ + ::testing::FLAGS_gtest_death_test_style = "threadsafe"; + + EXPECT_DEATH(createArrayOverConst(), "ColumnArray cannot have ColumnConst as its nested column"); + EXPECT_DEATH( + ColumnArray::create(ColumnConst::create(ColumnUInt64::create(1, 42), 0)), + "ColumnArray cannot have ColumnConst as its nested column"); +} + #endif diff --git a/src/Columns/tests/gtest_column_tuple.cpp b/src/Columns/tests/gtest_column_tuple.cpp index cc784d9ff53e..bf073267ae49 100644 --- a/src/Columns/tests/gtest_column_tuple.cpp +++ b/src/Columns/tests/gtest_column_tuple.cpp @@ -1,4 +1,5 @@ #include +#include #include #include @@ -18,6 +19,7 @@ namespace DB namespace ErrorCodes { extern const int SIZES_OF_COLUMNS_DOESNT_MATCH; +extern const int LOGICAL_ERROR; } } @@ -142,3 +144,71 @@ TEST(ColumnTuple, EmptyTuplePermute) ASSERT_EQ(e.code(), ErrorCodes::SIZES_OF_COLUMNS_DOESNT_MATCH); } } + +namespace +{ + +/// Two rows of a constant next to a full element of two rows: the sizes match, so only the constant is wrong. +ColumnPtr createTupleFromMutableColumnsWithConst() +{ + MutableColumns elements; + elements.push_back(ColumnUInt64::create(2, 1)); + elements.push_back(ColumnConst::create(ColumnUInt64::create(1, 42), 2)); + return ColumnTuple::create(std::move(elements)); +} + +ColumnPtr createTupleFromColumnsWithConst() +{ + return ColumnTuple::create(Columns{ColumnUInt64::create(2, 1), ColumnConst::create(ColumnUInt64::create(1, 42), 2)}); +} + +ColumnPtr createTupleFromTupleColumnsWithConst() +{ + VectorWithMemoryTracking elements; + elements.emplace_back(ColumnUInt64::create(2, 1)); + elements.emplace_back(ColumnConst::create(ColumnUInt64::create(1, 42), 2)); + return ColumnTuple::create(elements); +} + +} + +/// Skipped under debug/sanitizers: LOGICAL_ERROR aborts there, so the exception can't be caught. +#ifndef DEBUG_OR_SANITIZER_BUILD + +namespace +{ + +int errorCodeOf(ColumnPtr (*create)()) +{ + try + { + (void)create(); + } + catch (const Exception & e) + { + return e.code(); + } + return 0; +} + +} + +TEST(ColumnTuple, ConstElementIsRejected) +{ + EXPECT_EQ(errorCodeOf(createTupleFromMutableColumnsWithConst), ErrorCodes::LOGICAL_ERROR); + EXPECT_EQ(errorCodeOf(createTupleFromColumnsWithConst), ErrorCodes::LOGICAL_ERROR); + EXPECT_EQ(errorCodeOf(createTupleFromTupleColumnsWithConst), ErrorCodes::LOGICAL_ERROR); +} + +#else + +TEST(ColumnTupleDeathTest, ConstElementIsRejected) +{ + ::testing::FLAGS_gtest_death_test_style = "threadsafe"; + + EXPECT_DEATH(createTupleFromMutableColumnsWithConst(), "ColumnTuple cannot have ColumnConst as its element"); + EXPECT_DEATH(createTupleFromColumnsWithConst(), "ColumnTuple cannot have ColumnConst as its element"); + EXPECT_DEATH(createTupleFromTupleColumnsWithConst(), "ColumnTuple cannot have ColumnConst as its element"); +} + +#endif diff --git a/src/Storages/TTLDescription.cpp b/src/Storages/TTLDescription.cpp index 3c06a812616b..086002a38a3e 100644 --- a/src/Storages/TTLDescription.cpp +++ b/src/Storages/TTLDescription.cpp @@ -715,7 +715,8 @@ std::vector checkActionsDAGForAggregateFunctions( { auto offsets = ColumnArray::ColumnOffsets::create(); offsets->getData().push_back(1); - candidates.push_back(ColumnArray::create(element->cloneResized(1), std::move(offsets))); + candidates.push_back( + ColumnArray::create(element->convertToFullColumnIfConst()->cloneResized(1), std::move(offsets))); } } else diff --git a/tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.reference b/tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.reference new file mode 100644 index 000000000000..e8183f05f5db --- /dev/null +++ b/tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.reference @@ -0,0 +1,3 @@ +1 +1 +1 diff --git a/tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.sql b/tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.sql new file mode 100644 index 000000000000..85d82c894c13 --- /dev/null +++ b/tests/queries/0_stateless/05292_ttl_array_map_constant_lambda_body.sql @@ -0,0 +1,20 @@ +-- A TTL expression that reads the result of `arrayMap` whose lambda returns a constant `Dynamic` value. + +-- With `cast_keep_nullable`, `toUInt8` of a `Dynamic` value is `Nullable`, and a TTL expression cannot be `Nullable`. +SET cast_keep_nullable = 0; + +CREATE TABLE t_ttl_array_map_length (d DateTime, arr Array(UInt32)) ENGINE = MergeTree ORDER BY tuple() +TTL d + toIntervalDay(length(arrayMap(x -> CAST(1, 'Dynamic'), arr))); +INSERT INTO t_ttl_array_map_length VALUES ('2100-01-01 00:00:00', [1]); +SELECT count() FROM t_ttl_array_map_length; + +CREATE TABLE t_ttl_array_map_element (d DateTime, arr Array(UInt32)) ENGINE = MergeTree ORDER BY tuple() +TTL d + toIntervalDay(toUInt8(arrayMap(x -> CAST(1, 'Dynamic'), arr)[1])); +INSERT INTO t_ttl_array_map_element VALUES ('2100-01-01 00:00:00', [1]); +SELECT count() FROM t_ttl_array_map_element; + +ALTER TABLE t_ttl_array_map_element MODIFY TTL d + toIntervalDay(toUInt8(arrayMap(x -> CAST(2, 'Dynamic'), arr)[1])); +SELECT count() FROM t_ttl_array_map_element; + +DROP TABLE t_ttl_array_map_length; +DROP TABLE t_ttl_array_map_element; From e25638152729d67db9ef78dc0e7d1a144221ef4b Mon Sep 17 00:00:00 2001 From: Mikhail Koviazin Date: Fri, 2 Oct 2026 09:23:37 +0200 Subject: [PATCH 105/185] Fix analyzer setting in views over `Distributed` Keep the analyzer disabled in `InterpreterSelectQuery` after applying a stored `SELECT`'s settings. This prevents `Distributed` reads from entering analyzer-only paths without a planner context. --- src/Interpreters/InterpreterSelectQuery.cpp | 9 ++++ ...nalyzer_setting_over_distributed.reference | 16 ++++++ ...view_analyzer_setting_over_distributed.sql | 49 +++++++++++++++++++ 3 files changed, 74 insertions(+) create mode 100644 tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.reference create mode 100644 tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.sql diff --git a/src/Interpreters/InterpreterSelectQuery.cpp b/src/Interpreters/InterpreterSelectQuery.cpp index 4de94dd46ea0..578fb66e392b 100644 --- a/src/Interpreters/InterpreterSelectQuery.cpp +++ b/src/Interpreters/InterpreterSelectQuery.cpp @@ -139,6 +139,7 @@ namespace Setting extern const SettingsUInt64 aggregation_in_order_max_block_bytes; extern const SettingsUInt64 aggregation_memory_efficient_merge_threads; extern const SettingsBool allow_calculating_subcolumns_sizes_for_merge_tree_reading; + extern const SettingsBool allow_experimental_analyzer; extern const SettingsUInt64 allow_experimental_parallel_reading_from_replicas; extern const SettingsUInt64 automatic_parallel_replicas_mode; extern const SettingsBool async_socket_for_remote; @@ -3949,8 +3950,16 @@ void InterpreterSelectQuery::initSettings() { auto & query = getSelectQuery(); if (query.settings()) + { InterpreterSetQuery(query.settings(), context).executeForCurrentContext(options.ignore_setting_constraints); + /// The old interpreter disabled the analyzer in `IInterpreterUnionOrSelectQuery`, but a `SELECT` + /// stored in a `VIEW` may enable it again here. Storages must not use analyzer-only query tree + /// paths (for example, `StorageDistributed::read`) when this interpreter built the query. + if (context->getSettingsRef()[Setting::allow_experimental_analyzer]) + context->setSetting("allow_experimental_analyzer", false); + } + const auto & client_info = context->getClientInfo(); if (client_info.query_kind == ClientInfo::QueryKind::SECONDARY_QUERY && diff --git a/tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.reference b/tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.reference new file mode 100644 index 000000000000..071137aa6b5d --- /dev/null +++ b/tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.reference @@ -0,0 +1,16 @@ +view analyzer 1, read analyzer 0 +2 +s1 2.5 +s2 7 +view analyzer 1, read analyzer 0, remote shard +s1 2.5 +s2 7 +view analyzer 1, read analyzer 1 +s1 2.5 +s2 7 +view analyzer 0, read analyzer 1 +s1 2.5 +s2 7 +view analyzer 0, read analyzer 0 +s1 2.5 +s2 7 diff --git a/tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.sql b/tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.sql new file mode 100644 index 000000000000..a85ac964d159 --- /dev/null +++ b/tests/queries/0_stateless/05315_view_analyzer_setting_over_distributed.sql @@ -0,0 +1,49 @@ +-- Tags: distributed + +-- The old interpreter must not use the analyzer-only `Distributed` path when a view's +-- stored `SELECT` has `SETTINGS enable_analyzer = 1`. + +DROP VIEW IF EXISTS v_new; +DROP VIEW IF EXISTS v_old; +DROP TABLE IF EXISTS dist; +DROP TABLE IF EXISTS src; + +CREATE TABLE src (trade_date Date, strategy String, pnl Float64, ts DateTime) ENGINE = Memory; +INSERT INTO src VALUES + ('2026-09-17', 's1', 1.5, '2026-09-17 10:00:00'), + ('2026-09-17', 's1', 2.5, '2026-09-17 11:00:00'), + ('2026-09-17', 's2', 7, '2026-09-17 09:00:00'), + ('2026-09-16', 's3', 9, '2026-09-16 10:00:00'); + +CREATE TABLE dist AS src ENGINE = Distributed(test_shard_localhost, currentDatabase(), src); + +CREATE VIEW v_new AS + SELECT strategy, argMax(pnl, ts) AS pnl + FROM dist WHERE trade_date = '2026-09-17' + GROUP BY strategy SETTINGS enable_analyzer = 1; + +CREATE VIEW v_old AS + SELECT strategy, argMax(pnl, ts) AS pnl + FROM dist WHERE trade_date = '2026-09-17' + GROUP BY strategy SETTINGS enable_analyzer = 0; + +SELECT 'view analyzer 1, read analyzer 0'; +SELECT count() FROM v_new SETTINGS enable_analyzer = 0; +SELECT strategy, pnl FROM v_new ORDER BY strategy SETTINGS enable_analyzer = 0; + +SELECT 'view analyzer 1, read analyzer 0, remote shard'; +SELECT strategy, pnl FROM v_new ORDER BY strategy SETTINGS enable_analyzer = 0, prefer_localhost_replica = 0; + +SELECT 'view analyzer 1, read analyzer 1'; +SELECT strategy, pnl FROM v_new ORDER BY strategy SETTINGS enable_analyzer = 1; + +SELECT 'view analyzer 0, read analyzer 1'; +SELECT strategy, pnl FROM v_old ORDER BY strategy SETTINGS enable_analyzer = 1; + +SELECT 'view analyzer 0, read analyzer 0'; +SELECT strategy, pnl FROM v_old ORDER BY strategy SETTINGS enable_analyzer = 0; + +DROP VIEW v_new; +DROP VIEW v_old; +DROP TABLE dist; +DROP TABLE src; From 42edff16bcd01db2e7b285abe47592e82e0f8844 Mon Sep 17 00:00:00 2001 From: Mikhail Koviazin Date: Fri, 2 Oct 2026 09:39:12 +0200 Subject: [PATCH 106/185] Fix `ALIAS` row policies on legacy distributed reads Expand table aliases before building row-policy filter actions on shards. Keep row-policy inputs separate from `PREWHERE` inputs so columns needed by selected aliases remain available. --- src/Interpreters/InterpreterSelectQuery.cpp | 26 +++--- ..._policy_alias_column_distributed.reference | 38 ++++++++ ...16_row_policy_alias_column_distributed.sql | 92 +++++++++++++++++++ 3 files changed, 143 insertions(+), 13 deletions(-) create mode 100644 tests/queries/0_stateless/05316_row_policy_alias_column_distributed.reference create mode 100644 tests/queries/0_stateless/05316_row_policy_alias_column_distributed.sql diff --git a/src/Interpreters/InterpreterSelectQuery.cpp b/src/Interpreters/InterpreterSelectQuery.cpp index 4de94dd46ea0..ba1062256cba 100644 --- a/src/Interpreters/InterpreterSelectQuery.cpp +++ b/src/Interpreters/InterpreterSelectQuery.cpp @@ -303,10 +303,13 @@ try ASTs select_expressions; - /// The first column is our filter expression. - /// the row_policy_filter_expression should be cloned, because it may be changed by TreeRewriter. - /// which make it possible an invalid expression, although it may be valid in whole select. - select_expressions.push_back(row_policy_filter_expression->clone()); + /// The first column is our filter expression. Clone it because `TreeRewriter` can change the AST. + auto filter_expression = row_policy_filter_expression->clone(); + + /// `TreeRewriter` expands table aliases only on the initiator, but these filters are also + /// created on shards. An `ALIAS` used by a row policy must be evaluated before the read filter. + replaceAliasColumnsInQuery(filter_expression, metadata_snapshot->getColumns(), {}, context); + select_expressions.push_back(std::move(filter_expression)); /// Keep columns that are required after the filter actions. for (const auto & column_str : prerequisite_columns) @@ -2626,7 +2629,7 @@ void InterpreterSelectQuery::addPrewhereAliasActions() } /// Set of all (including ALIAS) required columns for PREWHERE - auto get_prewhere_columns = [&]() + auto get_prewhere_columns = [&](bool include_row_level_filter) { NameSet columns; @@ -2639,7 +2642,7 @@ void InterpreterSelectQuery::addPrewhereAliasActions() /// A row policy that will not be pushed into the storage read is applied as an ordinary /// FilterStep above it, so its columns are read normally and are not PREWHERE columns. - if (row_level_filter && shouldPushRowLevelFilterToStorage()) + if (row_level_filter && include_row_level_filter && shouldPushRowLevelFilterToStorage()) { auto row_level_required_columns = row_level_filter->actions.getRequiredColumns().getNames(); columns.insert(row_level_required_columns.begin(), row_level_required_columns.end()); @@ -2657,15 +2660,14 @@ void InterpreterSelectQuery::addPrewhereAliasActions() /// before any other executions. if (alias_columns_required) { - NameSet required_columns_from_prewhere = get_prewhere_columns(); + /// The row-level filter runs before `PREWHERE`, but its inputs are not produced by + /// `PREWHERE`. Keep them in the alias actions for queries that still need them. + NameSet required_columns_from_prewhere = get_prewhere_columns(/*include_row_level_filter=*/ false); NameSet required_aliases_from_prewhere; /// Set of ALIAS required columns for PREWHERE /// Expression, that contains all raw required columns ASTPtr required_columns_all_expr = make_intrusive(); - /// Expression, that contains raw required columns for PREWHERE - ASTPtr required_columns_from_prewhere_expr = make_intrusive(); - /// Sort out already known required columns between expressions, /// also populate `required_aliases_from_prewhere`. for (const auto & column : required_columns) @@ -2690,8 +2692,6 @@ void InterpreterSelectQuery::addPrewhereAliasActions() if (required_columns_from_prewhere.contains(column)) { - required_columns_from_prewhere_expr->children.emplace_back(std::move(column_expr)); - if (is_alias) required_aliases_from_prewhere.insert(column); } @@ -2759,7 +2759,7 @@ void InterpreterSelectQuery::addPrewhereAliasActions() const auto & supported_prewhere_columns = storage->supportedPrewhereColumns(); if (supported_prewhere_columns.has_value()) { - NameSet required_columns_from_prewhere = get_prewhere_columns(); + NameSet required_columns_from_prewhere = get_prewhere_columns(/*include_row_level_filter=*/ true); const auto & table_columns = metadata_snapshot->getColumns(); const bool include_subcolumns = storage->supportedPrewhereColumnsIncludeSubcolumns(); diff --git a/tests/queries/0_stateless/05316_row_policy_alias_column_distributed.reference b/tests/queries/0_stateless/05316_row_policy_alias_column_distributed.reference new file mode 100644 index 000000000000..565151239d36 --- /dev/null +++ b/tests/queries/0_stateless/05316_row_policy_alias_column_distributed.reference @@ -0,0 +1,38 @@ +baseline local +1 +2 +3 +4 +baseline distributed +1 +2 +3 +4 +policy local +2 +policy distributed local replica +2 +policy distributed remote replica +2 +policy distributed with PREWHERE +2 +policy local without alias optimization +2 +policy local with aliases selected +2 20 ok20 +policy local with aliases selected and PREWHERE +2 20 ok20 +policy distributed with aliases selected +2 20 ok20 +policy distributed with aliases selected and PREWHERE +2 20 ok20 +policy distributed new analyzer +2 +policy with only an ALIAS predicate +1 2 +policy with inline expression +2 +additional filter on ALIAS +2 +3 +4 diff --git a/tests/queries/0_stateless/05316_row_policy_alias_column_distributed.sql b/tests/queries/0_stateless/05316_row_policy_alias_column_distributed.sql new file mode 100644 index 000000000000..91e7d1360b03 --- /dev/null +++ b/tests/queries/0_stateless/05316_row_policy_alias_column_distributed.sql @@ -0,0 +1,92 @@ +-- Tags: distributed + +SET enable_analyzer = 0; + +DROP ROW POLICY IF EXISTS rp ON t; +DROP TABLE IF EXISTS t_dist; +DROP TABLE IF EXISTS t; + +CREATE TABLE t +( + k UInt64, + team String, + a UInt64 ALIAS k * 10, + a2 String ALIAS concat(team, toString(a)) +) ENGINE = MergeTree ORDER BY k; + +INSERT INTO t VALUES (1, 'ok'), (2, 'ok'), (3, 'ok'), (4, 'no'); + +CREATE TABLE t_dist AS t +ENGINE = Distributed('test_shard_localhost', currentDatabase(), 't', rand()); + +SELECT 'baseline local'; +SELECT k FROM t ORDER BY k; + +SELECT 'baseline distributed'; +SELECT k FROM t_dist ORDER BY k SETTINGS prefer_localhost_replica = 0; + +-- 1 fails the `ALIAS` condition; 3 fails the nested `ALIAS` condition; +-- 4 passes both `ALIAS` conditions but fails the physical team condition. +CREATE ROW POLICY rp ON t FOR SELECT +USING team = 'ok' AND a >= 20 AND a2 != 'ok30' TO ALL; + +SELECT 'policy local'; +SELECT k FROM t ORDER BY k; + +SELECT 'policy distributed local replica'; +SELECT k FROM t_dist ORDER BY k SETTINGS prefer_localhost_replica = 1; + +SELECT 'policy distributed remote replica'; +SELECT k FROM t_dist ORDER BY k SETTINGS prefer_localhost_replica = 0; + +SELECT 'policy distributed with PREWHERE'; +SELECT k FROM t_dist PREWHERE k >= 1 WHERE k <= 4 ORDER BY k +SETTINGS prefer_localhost_replica = 0; + +SELECT 'policy local without alias optimization'; +SELECT k FROM t ORDER BY k SETTINGS optimize_respect_aliases = 0; + +SELECT 'policy local with aliases selected'; +SELECT k, a, a2 FROM t ORDER BY k SETTINGS optimize_respect_aliases = 0; + +-- `PREWHERE` uses team, while both the policy and the selected aliases need k. +-- The alias step must preserve k even when a separate `PREWHERE` step exists. +SELECT 'policy local with aliases selected and PREWHERE'; +SELECT k, a, a2 FROM t PREWHERE team != 'no' ORDER BY k +SETTINGS optimize_respect_aliases = 0; + +SELECT 'policy distributed with aliases selected'; +SELECT k, a, a2 FROM t_dist ORDER BY k SETTINGS prefer_localhost_replica = 0; + +SELECT 'policy distributed with aliases selected and PREWHERE'; +SELECT k, a, a2 FROM t_dist PREWHERE team != 'no' ORDER BY k +SETTINGS prefer_localhost_replica = 0, optimize_respect_aliases = 0; + +SELECT 'policy distributed new analyzer'; +SELECT k FROM t_dist ORDER BY k +SETTINGS enable_analyzer = 1, prefer_localhost_replica = 0; + +-- A policy whose only predicate references an `ALIAS` still needs its value +-- before the row-level filter, even when the `SELECT` only needs `count`. +CREATE ROW POLICY OR REPLACE rp ON t FOR SELECT +USING a2 = 'ok20' TO ALL; + +SELECT 'policy with only an ALIAS predicate'; +SELECT count(), sum(k) FROM t_dist SETTINGS prefer_localhost_replica = 0; + +-- The equivalent expression is a control for the alias substitution. +CREATE ROW POLICY OR REPLACE rp ON t FOR SELECT +USING team = 'ok' AND k * 10 >= 20 AND concat(team, toString(k * 10)) != 'ok30' TO ALL; + +SELECT 'policy with inline expression'; +SELECT k FROM t_dist ORDER BY k SETTINGS prefer_localhost_replica = 0; + +DROP ROW POLICY rp ON t; + +-- Additional table filters use the same filter-action construction path. +SELECT 'additional filter on ALIAS'; +SELECT k FROM t ORDER BY k +SETTINGS optimize_respect_aliases = 0, additional_table_filters = {'t': 'a >= 20'}; + +DROP TABLE t_dist; +DROP TABLE t; From 89532b91a3329e9e8882c6efe553066257de5699 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 2 Oct 2026 08:05:09 +0000 Subject: [PATCH 107/185] Backport #123337 to 26.8: Fix a crash when reading an ORC file whose root type is not a struct --- .../Impl/NativeORCBlockInputFormat.cpp | 15 ++++++- .../Formats/Impl/NativeORCBlockInputFormat.h | 3 ++ src/Storages/Hive/HiveFile.cpp | 2 +- .../05316_orc_non_struct_root.reference | 6 +++ .../0_stateless/05316_orc_non_struct_root.sh | 38 ++++++++++++++++++ .../data_orc/non_struct_root_array.orc | Bin 0 -> 230 bytes .../data_orc/non_struct_root_bigint.orc | Bin 0 -> 164 bytes 7 files changed, 61 insertions(+), 3 deletions(-) create mode 100644 tests/queries/0_stateless/05316_orc_non_struct_root.reference create mode 100755 tests/queries/0_stateless/05316_orc_non_struct_root.sh create mode 100644 tests/queries/0_stateless/data_orc/non_struct_root_array.orc create mode 100644 tests/queries/0_stateless/data_orc/non_struct_root_bigint.orc diff --git a/src/Processors/Formats/Impl/NativeORCBlockInputFormat.cpp b/src/Processors/Formats/Impl/NativeORCBlockInputFormat.cpp index 20096b6f91d1..aa7115021e73 100644 --- a/src/Processors/Formats/Impl/NativeORCBlockInputFormat.cpp +++ b/src/Processors/Formats/Impl/NativeORCBlockInputFormat.cpp @@ -1015,6 +1015,17 @@ std::unique_ptr buildORCSearchArgument( return builder->build(); } +std::unique_ptr createORCReader(std::unique_ptr stream, const orc::ReaderOptions & options) +{ + auto reader = orc::createReader(std::move(stream), options); + if (reader->getType().getKind() != orc::STRUCT) + throw Exception( + ErrorCodes::INCORRECT_DATA, + "ORC files whose root type is not a struct are not supported, the file has root type {}", + reader->getType().toString()); + return reader; +} + static void getFileReader( ReadBuffer & in, std::unique_ptr & file_reader, @@ -1036,7 +1047,7 @@ static void getFileReader( options.setCacheOptions(orc::CacheOptions{.holeSizeLimit = hole_size_limit, .rangeSizeLimit = range_size_limit}); auto input_stream = asORCInputStream(in, format_settings, use_prefetch, is_stopped); - file_reader = orc::createReader(std::move(input_stream), options); + file_reader = createORCReader(std::move(input_stream), options); } static const orc::Type * @@ -1519,7 +1530,7 @@ void ORCColumnToCHColumn::orcTableToCHChunk( { const auto * struct_batch = dynamic_cast(table); if (!struct_batch) - throw Exception(ErrorCodes::LOGICAL_ERROR, "ORC table must be StructVectorBatch but is {}", struct_batch->toString()); + throw Exception(ErrorCodes::LOGICAL_ERROR, "ORC table must be StructVectorBatch but is {}", table->toString()); if (schema->getSubtypeCount() != struct_batch->fields.size()) throw Exception( diff --git a/src/Processors/Formats/Impl/NativeORCBlockInputFormat.h b/src/Processors/Formats/Impl/NativeORCBlockInputFormat.h index 70df42d54ead..49a29a7d8517 100644 --- a/src/Processors/Formats/Impl/NativeORCBlockInputFormat.h +++ b/src/Processors/Formats/Impl/NativeORCBlockInputFormat.h @@ -63,6 +63,9 @@ std::unique_ptr asORCInputStreamLoadIntoMemory(ReadBuffer & in /// instead of returning a null pointer that the library dereferences. orc::MemoryPool & getORCMemoryPool(); +/// Creates an ORC file reader; throws INCORRECT_DATA if the root type of the file is not a struct. +std::unique_ptr createORCReader(std::unique_ptr stream, const orc::ReaderOptions & options); + std::unique_ptr buildORCSearchArgument( const KeyCondition & key_condition, const Block & header, const orc::Type & schema, const FormatSettings & format_settings); diff --git a/src/Storages/Hive/HiveFile.cpp b/src/Storages/Hive/HiveFile.cpp index 6dfd1f78e73f..378bb0840ef7 100644 --- a/src/Storages/Hive/HiveFile.cpp +++ b/src/Storages/Hive/HiveFile.cpp @@ -161,7 +161,7 @@ void HiveORCFile::prepareReader() std::atomic is_stopped{0}; orc::ReaderOptions options; options.setMemoryPool(getORCMemoryPool()); - reader = orc::createReader(asORCInputStream(*in, format_settings, /*use_prefetch=*/false, is_stopped), options); + reader = createORCReader(asORCInputStream(*in, format_settings, /*use_prefetch=*/false, is_stopped), options); } void HiveORCFile::prepareColumnMapping() diff --git a/tests/queries/0_stateless/05316_orc_non_struct_root.reference b/tests/queries/0_stateless/05316_orc_non_struct_root.reference new file mode 100644 index 000000000000..dd8297685ff4 --- /dev/null +++ b/tests/queries/0_stateless/05316_orc_non_struct_root.reference @@ -0,0 +1,6 @@ +INCORRECT_DATA +INCORRECT_DATA +ORC files whose root type is not a struct are not supported +INCORRECT_DATA +the file has root type bigint +10 diff --git a/tests/queries/0_stateless/05316_orc_non_struct_root.sh b/tests/queries/0_stateless/05316_orc_non_struct_root.sh new file mode 100755 index 000000000000..ccc6d5b8079b --- /dev/null +++ b/tests/queries/0_stateless/05316_orc_non_struct_root.sh @@ -0,0 +1,38 @@ +#!/usr/bin/env bash +# Tags: no-fasttest + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Reading an ORC file whose root type is not a struct must fail with an error instead of crashing. +# The fixtures have the root types `bigint` and `array`. +BIGINT_ROOT="$CUR_DIR/data_orc/non_struct_root_bigint.orc" +ARRAY_ROOT="$CUR_DIR/data_orc/non_struct_root_array.orc" + +check() +{ + local query="$1" expected="$2" + local out + out=$($CLICKHOUSE_LOCAL --query "$query" 2>&1) + if [ -z "$out" ]; then + echo "no output at all" + else + echo "$out" | grep -o -m1 "$expected" || echo "unexpected: $out" + fi +} + +check "SELECT * FROM file('$BIGINT_ROOT', ORC, 'x Int64') FORMAT Null" 'INCORRECT_DATA' +check "SELECT count() FROM file('$BIGINT_ROOT', ORC, 'x Int64')" 'INCORRECT_DATA' +check "DESCRIBE file('$ARRAY_ROOT', ORC)" 'ORC files whose root type is not a struct are not supported' +check "SELECT * FROM file('$ARRAY_ROOT', ORC, 'x Array(Int64)') FORMAT Null" 'INCORRECT_DATA' + +# The message names the root type of the file. +check "SELECT * FROM file('$BIGINT_ROOT', ORC, 'x Int64') FORMAT Null" 'the file has root type bigint' + +# A regular ORC file, whose root type is a struct, is still read. +STRUCT_ROOT="$CLICKHOUSE_TMP/05316_struct_root_${CLICKHOUSE_DATABASE}.orc" +rm -f "$STRUCT_ROOT" +$CLICKHOUSE_LOCAL --query "INSERT INTO FUNCTION file('$STRUCT_ROOT', ORC) SELECT number AS x FROM numbers(5)" +$CLICKHOUSE_LOCAL --query "SELECT sum(x) FROM file('$STRUCT_ROOT', ORC)" +rm -f "$STRUCT_ROOT" diff --git a/tests/queries/0_stateless/data_orc/non_struct_root_array.orc b/tests/queries/0_stateless/data_orc/non_struct_root_array.orc new file mode 100644 index 0000000000000000000000000000000000000000..e1f883326d83f31274dc9e7a67be5cd18a975574 GIT binary patch literal 230 zcmeYda#m(w;Ns_EW?*0t;^1HoU`S$;V9*AN2}8tqIM{^PI2Z)jB!mJOSb?gTxEUA@ zu`)0TdN43NPI{amoG?M)A=AYkAt8rZoF~#*FZDRswF_`}@GCBz;p{wz(G+Nw1Q*&diFplDJ3M~LV`g;z@rBVjT?AothKP1!7MRp zhKj_R3X2;pH!LJrZm6^{Nyw=Es5DU9DD_hE)l3Ek4nF}#iG~IiB_;+Q4Os)WrZ3FQ I{z1;-0DM?Fr2qf` literal 0 HcmV?d00001 diff --git a/tests/queries/0_stateless/data_orc/non_struct_root_bigint.orc b/tests/queries/0_stateless/data_orc/non_struct_root_bigint.orc new file mode 100644 index 0000000000000000000000000000000000000000..77e2dbe03c2b2149294bc57687cb2ab89a6e1ca1 GIT binary patch literal 164 zcmeYda@J;G;1cFyW?*0t;^ANwV&h;C;E)grVBln6IKaZdWDb;O<6skDkdOe8xsAs}qDZnVv(7>X^#K5B=Y{1s^g_+qu$XOf!R1Og3 literal 0 HcmV?d00001 From 5d99a3b2747fcbd6d2711aaf590776a5b5ef24a1 Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 2 Oct 2026 08:07:25 +0000 Subject: [PATCH 108/185] Backport #123336 to 26.8: Do not dereference a null `logs_queue` in `OwnSplitChannel::log` --- src/Loggers/OwnSplitChannel.cpp | 3 ++- ...05316_local_sync_logger_without_channels.reference | 2 ++ .../05316_local_sync_logger_without_channels.sh | 11 +++++++++++ 3 files changed, 15 insertions(+), 1 deletion(-) create mode 100644 tests/queries/0_stateless/05316_local_sync_logger_without_channels.reference create mode 100755 tests/queries/0_stateless/05316_local_sync_logger_without_channels.sh diff --git a/src/Loggers/OwnSplitChannel.cpp b/src/Loggers/OwnSplitChannel.cpp index f35fbb8f8b65..547fdd7f4f73 100644 --- a/src/Loggers/OwnSplitChannel.cpp +++ b/src/Loggers/OwnSplitChannel.cpp @@ -65,7 +65,8 @@ void OwnSplitChannel::log(Poco::Message && msg) return; const auto & logs_queue = CurrentThread::getInternalTextLogsQueue(); - if (channels.empty() && (logs_queue == nullptr && !logs_queue->isNeeded(msg.getPriority(), msg.getSource()))) + if (channels.empty() && !text_log_max_priority.load(std::memory_order_relaxed) + && (logs_queue == nullptr || !logs_queue->isNeeded(msg.getPriority(), msg.getSource()))) return; if (const auto & masker = SensitiveDataMasker::getInstance()) diff --git a/tests/queries/0_stateless/05316_local_sync_logger_without_channels.reference b/tests/queries/0_stateless/05316_local_sync_logger_without_channels.reference new file mode 100644 index 000000000000..6ed281c757a9 --- /dev/null +++ b/tests/queries/0_stateless/05316_local_sync_logger_without_channels.reference @@ -0,0 +1,2 @@ +1 +1 diff --git a/tests/queries/0_stateless/05316_local_sync_logger_without_channels.sh b/tests/queries/0_stateless/05316_local_sync_logger_without_channels.sh new file mode 100755 index 000000000000..3ef19f954157 --- /dev/null +++ b/tests/queries/0_stateless/05316_local_sync_logger_without_channels.sh @@ -0,0 +1,11 @@ +#!/usr/bin/env bash + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# Synchronous logger with no log destination and one logger level raised explicitly. +$CLICKHOUSE_LOCAL --query "SELECT 1" -- --logger.async=0 --logger.levels.Application=trace + +# Client logs requested with send_logs_level still arrive when the logger has no destination. +$CLICKHOUSE_LOCAL --query "SELECT 1" --send_logs_level=trace -- --logger.async=0 2>&1 >/dev/null | grep -F ' executeQuery: ' | grep -c -F 'SELECT 1' From 32a05457cefab196e2852fcb83fb1d1afcab8fcf Mon Sep 17 00:00:00 2001 From: robot-clickhouse Date: Fri, 2 Oct 2026 08:48:22 +0000 Subject: [PATCH 109/185] Backport #120313 to 26.8: Hide secret arguments in `EXPLAIN AST` output --- src/Interpreters/InterpreterExplainQuery.cpp | 168 ++++++++++++++++++ .../05219_explain_ast_hide_secrets.reference | 101 +++++++++++ .../05219_explain_ast_hide_secrets.sh | 64 +++++++ 3 files changed, 333 insertions(+) create mode 100644 tests/queries/0_stateless/05219_explain_ast_hide_secrets.reference create mode 100755 tests/queries/0_stateless/05219_explain_ast_hide_secrets.sh diff --git a/src/Interpreters/InterpreterExplainQuery.cpp b/src/Interpreters/InterpreterExplainQuery.cpp index 1f2b0e376ad3..239cf6c75a23 100644 --- a/src/Interpreters/InterpreterExplainQuery.cpp +++ b/src/Interpreters/InterpreterExplainQuery.cpp @@ -28,9 +28,11 @@ #include #include #include +#include #include #include #include +#include #include #include @@ -61,10 +63,13 @@ #include #include +#include +#include #include #include #include #include +#include #include #include #include @@ -336,6 +341,161 @@ namespace } }; + /// Replace a node with a single `'[HIDDEN]'` literal, keeping its alias. + void hideWholeNode(ASTPtr & node) + { + auto hidden = make_intrusive(Field("[HIDDEN]")); + hidden->setAlias(node->tryGetAlias()); + node = std::move(hidden); + } + + /// Replace every literal inside a node with `'[HIDDEN]'`, keeping the expression structure. Only + /// for the secret arguments of `encrypt` / `HMAC`, where the shape is not a secret (a key built as + /// `leftPad('...', 16, '*')` stays readable as such); every other secret slot is hidden whole. + void hideLiteralsInSubtree(ASTPtr & node) + { + if (node->as()) + { + hideWholeNode(node); + return; + } + for (auto & child : node->children) + hideLiteralsInSubtree(child); + } + + /// Keep in sync with the names `FunctionSecretArgumentsFinder` sends to `findEncryptionFunctionSecretArguments` + /// and `findHMACSecretArguments`. A name missing here only makes the dump stricter: its span is hidden whole. + bool isEncryptionOrHMACFunction(const ASTFunction & function) + { + return function.name == "encrypt" || function.name == "decrypt" || function.name == "aes_encrypt_mysql" + || function.name == "aes_decrypt_mysql" || function.name == "tryDecrypt" || equalsCaseInsensitive(function.name, "HMAC"); + } + + bool isKeyValueArgument(const IAST & node) + { + const auto * function = node.as(); + return function && function->name == "equals" && function->arguments && function->arguments->children.size() == 2; + } + + /// The secret value of a `key = value` argument is its second child; anything else carries the + /// secret in the node itself. + ASTPtr & secretValueSlot(ASTPtr & node) + { + if (isKeyValueArgument(*node)) + return node->as()->arguments->children[1]; + return node; + } + + /// Replace an argument with the partially masked SQL the formatter prints for it: a URL with its + /// credentials removed, or the reconstructed `S3(...)` destination of a `Backup` database. The + /// finder builds the text from literals it read, so it parses. If it does not, the original node + /// must not stay in the tree; the argument is hidden whole (fail closed). + void replaceWithMaskedText(ASTPtr & node, const String & text) + { + ParserExpression parser; + const char * pos = text.data(); + String error; + ASTPtr parsed = tryParseQuery( + parser, + pos, + text.data() + text.size(), + error, + /* hilite= */ false, + "masked secret argument", + /* allow_multi_statements= */ false, + /* max_query_size= */ 0, + DBMS_DEFAULT_MAX_PARSER_DEPTH, + DBMS_DEFAULT_MAX_PARSER_BACKTRACKS, + /* skip_insignificant= */ true); + if (!parsed) + { + hideWholeNode(node); + return; + } + parsed->setAlias(node->tryGetAlias()); + node = std::move(parsed); + } + + /// `DumpASTNode` prints a literal through `IAST::getID`, value included, so the dump cannot hide + /// secrets while formatting as `ASTFunction::formatImpl` does. Hide them in the tree instead. As in the + /// formatter, a secret slot becomes one `'[HIDDEN]'` literal. That includes the slots the finder could + /// not inspect (a url built by `concat(...)`, an identifier in a password slot): their expression is + /// part of the secret and must not be dumped node by node. Only the `encrypt` / `HMAC` span keeps its + /// structure (see `hideLiteralsInSubtree`). All values of a nested map (`headers(...)`, + /// `extra_credentials(...)`) are hidden; the formatter keeps the non-secret `extra_credentials` + /// values, so the dump is stricter. + struct HideSecretArgumentsMatcher + { + struct Data + { + }; + + static bool needChildVisit(const ASTPtr &, const ASTPtr &) { return true; } + + static void visit(ASTPtr & ast, Data &) + { + auto * function = ast->as(); + if (!function || !function->arguments) + return; + + auto secret_arguments = FunctionSecretArgumentsFinderAST(*function).getResult(); + if (!secret_arguments.hasSecrets()) + return; + + auto & arguments = function->arguments->children; + for (size_t i = 0; i < arguments.size(); ++i) + { + if (auto * map = arguments[i]->as(); + map && map->arguments && std::ranges::contains(secret_arguments.nested_maps, map->name)) + { + for (auto & entry : map->arguments->children) + hideWholeNode(secretValueSlot(entry)); + continue; + } + + if (auto replaced = secret_arguments.replaced_arguments.find(i); replaced != secret_arguments.replaced_arguments.end()) + { + replaceWithMaskedText(arguments[i], replaced->second); + continue; + } + + /// An individually masked argument: only the named `key = value` form keeps its key. + if (auto masked = secret_arguments.masked_arguments.find(i); masked != secret_arguments.masked_arguments.end()) + { + hideWholeNode(masked->second ? secretValueSlot(arguments[i]) : arguments[i]); + continue; + } + + if (!(secret_arguments.start <= i && i < secret_arguments.start + secret_arguments.count)) + continue; + + if (!secret_arguments.replacement.empty()) + { + const auto text + = secret_arguments.quote_replacement ? quoteString(secret_arguments.replacement) : secret_arguments.replacement; + replaceWithMaskedText(secret_arguments.are_named ? secretValueSlot(arguments[i]) : arguments[i], text); + continue; + } + + if (secret_arguments.are_named) + { + hideWholeNode(secretValueSlot(arguments[i])); + continue; + } + + /// Only the span of `encrypt` / `HMAC` keeps its structure. Any other unnamed span without a + /// replacement, such as an unreadable url in `mongodb(concat(...), 'c')`, is hidden whole. So is a + /// `key = value` in the span: it is a positional secret written as a comparison. + if (isEncryptionOrHMACFunction(*function) && !isKeyValueArgument(*arguments[i])) + hideLiteralsInSubtree(arguments[i]); + else + hideWholeNode(arguments[i]); + } + } + }; + + using HideSecretArgumentsVisitor = InDepthNodeVisitor; + bool hasSecretsInActionsDAG(const ActionsDAG & dag) { for (const auto & node : dag.getNodes()) @@ -996,6 +1156,14 @@ QueryPipeline InterpreterExplainQuery::executeImpl() ExplainAnalyzedSyntaxVisitor(data).visit(query); } + /// `optimize = 1` inlines views the user may read but whose secrets they may not see. + /// Hide them under the same gate as `SHOW CREATE`. + if (!canDisplaySecrets(query_context)) + { + HideSecretArgumentsVisitor::Data data; + HideSecretArgumentsVisitor(data).visit(query); + } + if (settings.graph) dumpASTInDotFormat(*ast.getExplainedQuery(), buf); else diff --git a/tests/queries/0_stateless/05219_explain_ast_hide_secrets.reference b/tests/queries/0_stateless/05219_explain_ast_hide_secrets.reference new file mode 100644 index 000000000000..405f09ad316d --- /dev/null +++ b/tests/queries/0_stateless/05219_explain_ast_hide_secrets.reference @@ -0,0 +1,101 @@ +-- the inlined view body hides the key +SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 2) + ExpressionList (children 1) + Identifier encrypted_secret + TablesInSelectQuery (children 1) + TablesInSelectQueryElement (children 1) + TableExpression (children 1) + SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 2) + ExpressionList (children 1) + Function hex (alias encrypted_secret) (children 1) + ExpressionList (children 1) + Function encrypt (children 1) + ExpressionList (children 3) + Literal \'aes-128-ecb\' + Identifier secret + Literal \'[HIDDEN]\' + TablesInSelectQuery (children 1) + TablesInSelectQueryElement (children 1) + TableExpression (children 1) + TableIdentifier default.private_plaintext +-- the session setting alone does not disclose it + Literal \'aes-128-ecb\' + Literal \'[HIDDEN]\' +-- the graph dump hides it too +Literal \'[HIDDEN]\' +-- a secret typed into the explained query itself is hidden as well +SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 1) + ExpressionList (children 1) + Function encrypt (children 1) + ExpressionList (children 4) + Literal \'aes-128-ecb\' + Literal \'[HIDDEN]\' + Literal \'[HIDDEN]\' + Function leftPad (children 1) + ExpressionList (children 3) + Literal \'[HIDDEN]\' + Literal \'[HIDDEN]\' + Literal \'[HIDDEN]\' +-- a positional secret written as a comparison collapses to one node, not to 'Function equals' +SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 1) + ExpressionList (children 1) + Function encrypt (children 1) + ExpressionList (children 3) + Literal \'aes-128-ecb\' + Literal \'[HIDDEN]\' + Literal \'[HIDDEN]\' +-- a url the finder cannot read collapses to one node, not to 'Function concat' +SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 2) + ExpressionList (children 1) + Asterisk + TablesInSelectQuery (children 1) + TablesInSelectQueryElement (children 1) + TableExpression (children 1) + Function url (children 1) + ExpressionList (children 1) + Literal \'[HIDDEN]\' +-- a mongodb url the finder cannot read collapses to one node, not to 'Function concat' +SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 2) + ExpressionList (children 1) + Asterisk + TablesInSelectQuery (children 1) + TablesInSelectQueryElement (children 1) + TableExpression (children 1) + Function mongodb (children 1) + ExpressionList (children 2) + Literal \'[HIDDEN]\' + Literal \'c\' +-- the same url as a named override keeps its key, the value collapses +SelectWithUnionQuery (children 1) + ExpressionList (children 1) + SelectQuery (children 2) + ExpressionList (children 1) + Asterisk + TablesInSelectQuery (children 1) + TablesInSelectQueryElement (children 1) + TableExpression (children 1) + Function url (children 1) + ExpressionList (children 2) + Identifier creds + Function equals (children 1) + ExpressionList (children 2) + Identifier url + Literal \'[HIDDEN]\' +-- a nested map keeps its keys and hides its values + Literal \'http://x/f\' + Literal \'Authorization\' + Literal \'[HIDDEN]\' +-- the view itself stays usable for the restricted user +1 diff --git a/tests/queries/0_stateless/05219_explain_ast_hide_secrets.sh b/tests/queries/0_stateless/05219_explain_ast_hide_secrets.sh new file mode 100755 index 000000000000..dcc3ca50f8c0 --- /dev/null +++ b/tests/queries/0_stateless/05219_explain_ast_hide_secrets.sh @@ -0,0 +1,64 @@ +#!/usr/bin/env bash +# Tags: no-fasttest, no-replicated-database +# Tag no-fasttest: the encryption functions are not available in the fast test build +# Tag no-replicated-database: SQL SECURITY DEFINER views and users are set up per-test + +CUR_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd) +# shellcheck source=../shell_config.sh +. "$CUR_DIR"/../shell_config.sh + +# `EXPLAIN AST optimize = 1` inlines the body of a view the user may only SELECT from, and the dump +# prints every literal verbatim. The secret arguments must be hidden as `SHOW CREATE` hides them. +# The stateless test server keeps `display_secrets_in_show_and_select` off, so the gate always hides here. + +user="user_05219_${CLICKHOUSE_DATABASE}_$RANDOM" +db=${CLICKHOUSE_DATABASE} +key='Sixteen byte key' + +${CLICKHOUSE_CLIENT} <