tinybird-toolset 2.6.2__tar.gz → 2.6.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {tinybird_toolset-2.6.2/src/tinybird_toolset.egg-info → tinybird_toolset-2.6.3}/PKG-INFO +1 -1
  2. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/setup.py +1 -1
  3. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3/src/tinybird_toolset.egg-info}/PKG-INFO +1 -1
  4. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_convert_to_row_binary.py +79 -0
  5. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/LICENSE +0 -0
  6. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/MANIFEST.in +0 -0
  7. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/README.md +0 -0
  8. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/conf.py +0 -0
  9. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/setup.cfg +0 -0
  10. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/src/chtoolset/__init__.py +0 -0
  11. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/src/chtoolset/query.py +0 -0
  12. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/src/tinybird_toolset.egg-info/SOURCES.txt +0 -0
  13. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/src/tinybird_toolset.egg-info/dependency_links.txt +0 -0
  14. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/src/tinybird_toolset.egg-info/requires.txt +0 -0
  15. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/src/tinybird_toolset.egg-info/top_level.txt +0 -0
  16. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_check_compatible_types.py +0 -0
  17. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_check_ttl_partition_compatibility.py +0 -0
  18. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_check_write_query.py +0 -0
  19. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_chquery.py +0 -0
  20. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_explain_ast.py +0 -0
  21. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_get_columns_from_create_query.py +0 -0
  22. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_get_left_table.py +0 -0
  23. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_internal_cache.py +0 -0
  24. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_normalize_query_keep_names.py +0 -0
  25. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_normalize_table_column.py +0 -0
  26. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_parse_create_materialized_view_target_table.py +0 -0
  27. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_replace_tables.py +0 -0
  28. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_replace_tables_backward_compat.py +0 -0
  29. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_rewrite_aggregation_states.py +0 -0
  30. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_tables.py +0 -0
  31. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_validate_partition_key.py +0 -0
  32. {tinybird_toolset-2.6.2 → tinybird_toolset-2.6.3}/tests/test_validate_ttl.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tinybird-toolset
3
- Version: 2.6.2
3
+ Version: 2.6.3
4
4
  Home-page: https://gitlab.com/tinybird/clickhouse-toolset
5
5
  Author: Tinybird.co
6
6
  Author-email: support@tinybird.co
@@ -1,7 +1,7 @@
1
1
  from setuptools import setup, Extension
2
2
 
3
3
  NAME = 'tinybird-toolset'
4
- VERSION = '2.6.2'
4
+ VERSION = '2.6.3'
5
5
 
6
6
  # Shared metadata for both the full (extension) build and the metadata-only
7
7
  # fallback below, so the two setup() calls can't drift apart.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: tinybird-toolset
3
- Version: 2.6.2
3
+ Version: 2.6.3
4
4
  Home-page: https://gitlab.com/tinybird/clickhouse-toolset
5
5
  Author: Tinybird.co
6
6
  Author-email: support@tinybird.co
@@ -1,6 +1,7 @@
1
1
  import json
2
2
  import logging
3
3
  import re
4
+ import shutil
4
5
  import pytest
5
6
  from chtoolset import query as chquery
6
7
  import random
@@ -26,6 +27,8 @@ omit_quarantine_changes = False
26
27
  logging.basicConfig(level=logging.DEBUG)
27
28
 
28
29
  fix_q_tests = os.getenv('FIXQTESTS', 'False').lower() == '1'
30
+ # clickhouse local is used to decode encoder output where available (it is not in the CI image)
31
+ clickhouse_available = shutil.which('clickhouse') is not None
29
32
  MB = 1024**2
30
33
 
31
34
 
@@ -8678,6 +8681,14 @@ class TestRowBinaryEncoder:
8678
8681
  ('c2', 'LowCardinality(String)', '', False),
8679
8682
  ('c3', 'Array(Tuple(String, Int16, Date))', '.x.y.m[:]', False)],
8680
8683
  "Invalid JSONPath"),
8684
+ # LowCardinality can never sit below a Nullable: ClickHouse rejects Nullable around composite and
8685
+ # LowCardinality types. The encoder relies on this when it caches a LowCardinality-free serialization
8686
+ # per column (recursiveRemoveLowCardinality does not descend through Nullable).
8687
+ (ConversionMode.ALWAYS, [('c1', 'Nullable(LowCardinality(String))', '$.c1', False)], "cannot be inside Nullable"),
8688
+ (ConversionMode.ALWAYS, [('c1', 'Nullable(Array(LowCardinality(String)))', '$.c1', False)], "cannot be inside Nullable"),
8689
+ (ConversionMode.ALWAYS, [('c1', 'Array(Nullable(Array(LowCardinality(String))))', '$.c1', False)], "cannot be inside Nullable"),
8690
+ (ConversionMode.ALWAYS, [('c1', 'Nullable(Tuple(LowCardinality(String)))', '$.c1', False)], "cannot be inside Nullable"),
8691
+ (ConversionMode.ALWAYS, [('c1', 'Nullable(Map(String, LowCardinality(String)))', '$.c1', False)], "cannot be inside Nullable"),
8681
8692
  (ConversionMode.ALWAYS, '[{"invalid-json"}]', "Invalid schema"),
8682
8693
  (ConversionMode.ALWAYS, '[{"name":"v","type":"String","has_default":false}]', "Invalid schema"),
8683
8694
  (ConversionMode.ALWAYS, '{"name":"v","type":"String","jsonpath":"$.v","has_default":false}', "Invalid schema"),
@@ -8999,6 +9010,74 @@ class TestRowBinaryEncoder:
8999
9010
  enc = chquery.RowBinaryEncoder(right_schema, max_json_row_size=100)
9000
9011
  enc.close()
9001
9012
 
9013
+ # The encoder writes LowCardinality columns with the serialization of the equivalent
9014
+ # LowCardinality-free type (see ColumnDefinition::field_serialization). RowBinary has no
9015
+ # dictionary encoding, so the bytes must be identical to the plain type at every nesting level,
9016
+ # and ClickHouse must decode them under the LowCardinality schema to the same value.
9017
+ LOWCARDINALITY_EQUIVALENCE_CASES = [
9018
+ ('LowCardinality(String)', 'String', 'es'),
9019
+ ('LowCardinality(String)', 'String', ''),
9020
+ ('LowCardinality(Nullable(String))', 'Nullable(String)', 'x'),
9021
+ ('LowCardinality(Nullable(String))', 'Nullable(String)', None),
9022
+ ('LowCardinality(FixedString(2))', 'FixedString(2)', 'ab'),
9023
+ ('LowCardinality(UInt16)', 'UInt16', 7),
9024
+ ('LowCardinality(Nullable(Int64))', 'Nullable(Int64)', -5),
9025
+ ('LowCardinality(DateTime)', 'DateTime', '2026-09-08 12:00:00'),
9026
+ ('LowCardinality(Date)', 'Date', '2026-09-08'),
9027
+ ('Array(LowCardinality(String))', 'Array(String)', ['a', 'b']),
9028
+ ('Array(LowCardinality(String))', 'Array(String)', []),
9029
+ ('Array(LowCardinality(Nullable(String)))', 'Array(Nullable(String))', ['a', None, 'c']),
9030
+ ('Array(Array(LowCardinality(String)))', 'Array(Array(String))', [['a'], [], ['b', 'c']]),
9031
+ ('Map(LowCardinality(String), LowCardinality(String))', 'Map(String, String)', {'k1': 'v1', 'k2': 'v2'}),
9032
+ ('Map(String, LowCardinality(Nullable(String)))', 'Map(String, Nullable(String))', {'k1': 'v1', 'k2': None}),
9033
+ ('Map(String, Array(LowCardinality(String)))', 'Map(String, Array(String))', {'k': ['a', 'b']}),
9034
+ ('Tuple(LowCardinality(String), Nullable(Int32))', 'Tuple(String, Nullable(Int32))', ['a', 1]),
9035
+ ('Tuple(a LowCardinality(String), b LowCardinality(Nullable(String)))', 'Tuple(a String, b Nullable(String))', ['a', None]),
9036
+ ('Array(Tuple(LowCardinality(String), Int8))', 'Array(Tuple(String, Int8))', [['a', 1], ['b', 2]]),
9037
+ ('Map(String, Tuple(LowCardinality(String), Array(LowCardinality(UInt8))))', 'Map(String, Tuple(String, Array(UInt8)))', {'k': ['a', [1, 2]]}),
9038
+ ]
9039
+
9040
+ @pytest.mark.parametrize("legacy_conversion_mode", [True, False])
9041
+ @pytest.mark.parametrize("lc_type, plain_type, value", LOWCARDINALITY_EQUIVALENCE_CASES)
9042
+ def test_lowcardinality_encodes_identically_to_plain_type(self, legacy_conversion_mode, lc_type, plain_type, value):
9043
+ data = json.dumps({'v': value})
9044
+ # Array and Tuple columns are addressed with the collection operator
9045
+ jsonpath = '$.v[:]' if plain_type.startswith(('Array', 'Tuple')) else '$.v'
9046
+ outputs = {}
9047
+ for type_name in (lc_type, plain_type):
9048
+ schema = json.dumps(get_schema([('v', type_name, jsonpath, False)]))
9049
+ with chquery.RowBinaryEncoder(schema, legacy_conversion_mode=legacy_conversion_mode) as encoder:
9050
+ result, quarantine, nrows, nquarantine = encoder.encode(data)
9051
+ assert (nrows, nquarantine, quarantine) == (1, 0, b''), f"{type_name} did not encode {data}"
9052
+ outputs[type_name] = result
9053
+ assert outputs[lc_type] == outputs[plain_type]
9054
+
9055
+ # The bytes must also be valid RowBinary for the LowCardinality schema itself
9056
+ if not clickhouse_available:
9057
+ pytest.skip("clickhouse binary is not available for RowBinary decoding")
9058
+ decoded = {}
9059
+ for type_name in (lc_type, plain_type):
9060
+ query = validationQuery([('v', type_name)], outputs[type_name]).rstrip(';')
9061
+ query += ', allow_suspicious_low_cardinality_types=1 FORMAT JSON'
9062
+ row, error = RunQuery(query)
9063
+ assert error is None, f"clickhouse could not decode {type_name}: {error}"
9064
+ decoded[type_name] = row
9065
+ assert decoded[lc_type] == decoded[plain_type]
9066
+
9067
+ # A JSON value next to a LowCardinality element takes the column-based write path
9068
+ # (binary_json_as_string), which must keep using the serialization of the full column type.
9069
+ @pytest.mark.parametrize("binary_json_as_string", [True, False])
9070
+ def test_lowcardinality_next_to_json_encodes_identically_to_plain_type(self, binary_json_as_string):
9071
+ data = json.dumps({'v': [{'a': 1, 'b': 'x'}, 'tag']})
9072
+ outputs = {}
9073
+ for type_name in ('Tuple(JSON, LowCardinality(String))', 'Tuple(JSON, String)'):
9074
+ schema = json.dumps(get_schema([('v', type_name, '$.v[:]', False)]))
9075
+ with chquery.RowBinaryEncoder(schema, binary_json_as_string=binary_json_as_string, legacy_conversion_mode=False) as encoder:
9076
+ result, quarantine, nrows, nquarantine = encoder.encode(data)
9077
+ assert (nrows, nquarantine, quarantine) == (1, 0, b''), f"{type_name} did not encode {data}"
9078
+ outputs[type_name] = result
9079
+ assert outputs['Tuple(JSON, LowCardinality(String))'] == outputs['Tuple(JSON, String)']
9080
+
9002
9081
  def test_getRaw_whitespace_trimming_for_integers(self):
9003
9082
  """Test that getRaw method properly trims whitespace from integers, especially when they are the last element in JSON objects."""
9004
9083
  # Schema with integer and string fields