From 9dcd320d4c80bcd9e7b3b7020ef320f20e9bc715 Mon Sep 17 00:00:00 2001 From: Gabriel Date: Sun, 26 Jul 2026 13:19:27 +0800 Subject: [PATCH] [test](regression) Complete Paimon read/write P0 coverage Add merge-engine semantic coverage for Parquet and ORC across automatic and forced JNI readers. Verify unsupported Paimon DML remains atomic, document the Paimon/Iceberg P0 matrix, and isolate the known CTAS metadata atomicity defect. --- .../test_paimon_merge_engine_matrix.out | 105 +++++++++ .../paimon/test_paimon_write_boundary.out | 15 ++ .../PAIMON_ICEBERG_READ_WRITE_P0_COVERAGE.md | 102 ++++++++ ...test_paimon_ctas_atomicity_negative.groovy | 74 ++++++ .../test_paimon_merge_engine_matrix.groovy | 223 ++++++++++++++++++ .../paimon/test_paimon_write_boundary.groovy | 107 +++++++++ 6 files changed, 626 insertions(+) create mode 100644 regression-test/data/external_table_p0/paimon/test_paimon_merge_engine_matrix.out create mode 100644 regression-test/data/external_table_p0/paimon/test_paimon_write_boundary.out create mode 100644 regression-test/suites/external_table_p0/PAIMON_ICEBERG_READ_WRITE_P0_COVERAGE.md create mode 100644 regression-test/suites/external_table_p0/paimon/test_paimon_ctas_atomicity_negative.groovy create mode 100644 regression-test/suites/external_table_p0/paimon/test_paimon_merge_engine_matrix.groovy create mode 100644 regression-test/suites/external_table_p0/paimon/test_paimon_write_boundary.groovy diff --git a/regression-test/data/external_table_p0/paimon/test_paimon_merge_engine_matrix.out b/regression-test/data/external_table_p0/paimon/test_paimon_merge_engine_matrix.out new file mode 100644 index 00000000000000..5b7470831c19eb --- /dev/null +++ b/regression-test/data/external_table_p0/paimon/test_paimon_merge_engine_matrix.out @@ -0,0 +1,105 @@ +-- This file is automatically generated. You should know what you did if you want to edit this +-- !auto_parquet_deduplicate -- +1 11 new-1 +2 22 new-2 +3 30 only-3 + +-- !auto_parquet_partial_update -- +1 15 2 filled-1 +2 30 4 base-2 + +-- !auto_parquet_aggregation -- +1 7 35 new-1 +2 9 300 new-2 + +-- !auto_parquet_first_row -- +1 10 first-1 +2 20 first-2 +3 30 first-3 + +-- !auto_parquet_engine_aggregates -- +63 51 16 60 + +-- !auto_orc_deduplicate -- +1 11 new-1 +2 22 new-2 +3 30 only-3 + +-- !auto_orc_partial_update -- +1 15 2 filled-1 +2 30 4 base-2 + +-- !auto_orc_aggregation -- +1 7 35 new-1 +2 9 300 new-2 + +-- !auto_orc_first_row -- +1 10 first-1 +2 20 first-2 +3 30 first-3 + +-- !auto_orc_engine_aggregates -- +63 51 16 60 + +-- !auto_dynamic_cross_partition -- +1 new-part 11 +2 stable-part 20 + +-- !auto_dynamic_old_partition -- +0 + +-- !auto_dynamic_new_partition -- +1 + +-- !forced_jni_parquet_deduplicate -- +1 11 new-1 +2 22 new-2 +3 30 only-3 + +-- !forced_jni_parquet_partial_update -- +1 15 2 filled-1 +2 30 4 base-2 + +-- !forced_jni_parquet_aggregation -- +1 7 35 new-1 +2 9 300 new-2 + +-- !forced_jni_parquet_first_row -- +1 10 first-1 +2 20 first-2 +3 30 first-3 + +-- !forced_jni_parquet_engine_aggregates -- +63 51 16 60 + +-- !forced_jni_orc_deduplicate -- +1 11 new-1 +2 22 new-2 +3 30 only-3 + +-- !forced_jni_orc_partial_update -- +1 15 2 filled-1 +2 30 4 base-2 + +-- !forced_jni_orc_aggregation -- +1 7 35 new-1 +2 9 300 new-2 + +-- !forced_jni_orc_first_row -- +1 10 first-1 +2 20 first-2 +3 30 first-3 + +-- !forced_jni_orc_engine_aggregates -- +63 51 16 60 + +-- !forced_jni_dynamic_cross_partition -- +1 new-part 11 +2 stable-part 20 + +-- !forced_jni_dynamic_old_partition -- +0 + +-- !forced_jni_dynamic_new_partition -- +1 + diff --git a/regression-test/data/external_table_p0/paimon/test_paimon_write_boundary.out b/regression-test/data/external_table_p0/paimon/test_paimon_write_boundary.out new file mode 100644 index 00000000000000..f1118d0bd7069e --- /dev/null +++ b/regression-test/data/external_table_p0/paimon/test_paimon_write_boundary.out @@ -0,0 +1,15 @@ +-- This file is automatically generated. You should know what you did if you want to edit this +-- !before_rows -- +1 10 base-1 +2 20 base-2 + +-- !before_snapshots -- +1 + +-- !after_rows -- +1 10 base-1 +2 20 base-2 + +-- !after_snapshots -- +1 + diff --git a/regression-test/suites/external_table_p0/PAIMON_ICEBERG_READ_WRITE_P0_COVERAGE.md b/regression-test/suites/external_table_p0/PAIMON_ICEBERG_READ_WRITE_P0_COVERAGE.md new file mode 100644 index 00000000000000..f3760f4db68ced --- /dev/null +++ b/regression-test/suites/external_table_p0/PAIMON_ICEBERG_READ_WRITE_P0_COVERAGE.md @@ -0,0 +1,102 @@ + + +# Paimon / Iceberg Read and Write P0 Coverage + +## Scope + +This matrix compares the supported Doris surface with the format capabilities documented by +[Apache Paimon](https://paimon.apache.org/docs/1.0/), its +[ecosystem matrix](https://paimon.apache.org/docs/master/ecosystem/), and Iceberg's +[write](https://iceberg.apache.org/docs/latest/spark-writes/) and +[evolution](https://iceberg.apache.org/docs/latest/evolution/) documentation. A format feature is a Doris P0 contract +only when Doris supports it. Unsupported write or format boundaries require deterministic negative +coverage and no metadata or data mutation. + +Doris currently documents Paimon data access as read-only. Master also supports Paimon catalog, +database and table metadata creation/deletion, but it has no Paimon data sink. Iceberg supports read, +DDL, INSERT, INSERT OVERWRITE, CTAS and, for compatible V2/V3 tables, DELETE, UPDATE and MERGE INTO. + +## Audit result + +The master inventory before this change contained 41 Paimon suites and 145 Iceberg suites, including +35 Iceberg write suites and four dedicated Iceberg DML suites. Iceberg P0 is complete for the Doris- +supported read/write surface: its existing matrices cover file-format versions, delete encodings, +evolution, time travel, DDL, distributed writes, row-level DML, failure atomicity and unsupported +boundaries. + +Paimon P0 was incomplete. No suite selected a `merge-engine`, and write rejection was not checked +across all DML shapes with pre/post data and snapshot invariants. PM01-PM03 below close those gaps. +PM04 exposes one remaining product defect as an opt-in negative regression: failed Paimon CTAS leaves +target metadata. Streaming writes, overwrite, delete/update and merge listed as unsupported by the +Paimon ecosystem matrix are boundary tests rather than positive Doris P0 contracts. + +## Risks + +| ID | Risk | Source | Impact | Priority | +| --- | --- | --- | --- | --- | +| R01 | Append and primary-key tables are planned with the wrong merge semantics | Black box: Paimon table models | Silent wrong results | P0 | +| R02 | Deduplicate, partial-update, aggregation and first-row produce the same result for duplicate keys | White box: LSM sorted-run merge | Silent wrong results | P0 | +| R03 | Fixed and dynamic bucket tables expose duplicate or stale cross-partition keys | Paimon data distribution | Silent wrong results | P0 | +| R04 | Automatic and forced-JNI routing disagree on Paimon merged rows, or native raw-file reads disagree where conversion is supported | Doris split routing | Query correctness | P0 | +| R05 | A rejected Paimon write creates a snapshot, changes data or leaves a CTAS table | Doris read-only boundary | Data or metadata mutation | P0 | +| R06 | Iceberg schema or partition evolution binds old field/spec IDs during current or historical reads | Iceberg evolution | Silent wrong results | P0 | +| R07 | Position/equality deletes or V3 deletion vectors are applied to the wrong file or snapshot | Iceberg row-level deletes | Deleted data visible or live data lost | P0 | +| R08 | Iceberg writes lose rows, route them to the wrong transform, or publish partial failed commits | Doris distributed Iceberg sink | Data loss or corruption | P0 | +| R09 | Snapshot, tag or branch reads/writes leak schema or data across references | Both formats | Historical data corruption | P0 | +| R10 | Unsupported format, DML mode or CTAS mutates state before rejection | Capability boundary | Partial commits or orphan metadata | P0 | + +## Feature matrix + +| Format | Capability | P0 status | Main suites | +| --- | --- | --- | --- | +| Paimon | Append table, partitioned table, primitive and nested types | Covered | `test_paimon_catalog`, `test_paimon_partition_table`, `test_paimon_full_schema_change` | +| Paimon | Primary-key deduplicate, partial-update, aggregation and first-row | Covered | `test_paimon_merge_engine_matrix` | +| Paimon | Fixed bucket, dynamic bucket and cross-partition update | Covered | `test_paimon_merge_engine_matrix`, `test_paimon_partition_pk_delete_refs` | +| Paimon | Parquet/ORC and mixed-format reads; JNI/native parity | Covered | `test_paimon_merge_engine_matrix`, `paimon_tb_mix_format`, `test_paimon_cpp_reader` | +| Paimon | Snapshot/timestamp/tag/branch and incremental modes | Covered | `paimon_time_travel`, `paimon_incr_read`, `test_paimon_schema_time_travel_matrix` | +| Paimon | Schema evolution, partition-key restrictions and historical schema binding | Covered | `test_paimon_schema_time_travel_matrix`, `test_paimon_partition_mutation_atomicity` | +| Paimon | Deletion vectors, upsert/delete visibility and data/system tables | Covered | `test_paimon_deletion_vector`, `paimon_data_system_table`, `paimon_system_table` | +| Paimon | Catalog/database/table create and drop | Covered | `test_create_paimon_table` | +| Paimon | Doris data write-back | Negative boundary covered | `test_paimon_write_boundary` | +| Paimon | Failed CTAS metadata atomicity | Isolated known-bug regression | `test_paimon_ctas_atomicity_negative` | +| Iceberg | V1/V2/V3, Parquet/ORC, position/equality deletes and deletion vectors | Covered | `test_iceberg_position_delete`, `test_iceberg_equality_delete`, `test_iceberg_deletion_vector` | +| Iceberg | Schema, partition and sort-order evolution | Covered | `test_iceberg_schema_time_travel_matrix`, `test_iceberg_partition_evolution_format_scanner`, `iceberg_schema_change_ddl` | +| Iceberg | Snapshot/timestamp/tag/branch reads and reference actions | Covered | `test_iceberg_time_travel`, `iceberg_query_tag_branch`, `test_iceberg_schema_ref_actions_matrix` | +| Iceberg | INSERT, INSERT OVERWRITE, static/hybrid partition and CTAS | Covered | `write/test_iceberg_write_insert`, `write/test_iceberg_write_overwrite_evolution`, `write/test_iceberg_write_ctas_format_boundary` | +| Iceberg | DELETE, UPDATE and MERGE in supported MOR modes | Covered | `dml/test_iceberg_update_delete_advanced`, `dml/test_iceberg_merge_into_advanced`, `write/test_iceberg_write_dml_modes_evolution` | +| Iceberg | Distributed/concurrent commits, failure atomicity and Spark interoperability | Covered | `write/test_iceberg_write_concurrent_merge_invariants`, `write/test_iceberg_write_overwrite_atomicity` | +| Iceberg | System tables, views, caches and catalog variants | Covered | `test_iceberg_sys_table`, `test_iceberg_view_query_p0`, `test_iceberg_table_cache`, catalog-specific suites | +| Iceberg | COW row DML, tag writes and unsupported file writes | Negative boundary covered | `write/test_iceberg_write_dml_modes_evolution`, `write/test_iceberg_write_branch_dml_boundary`, `write/test_iceberg_write_ctas_format_boundary` | + +Detailed schema/time-travel and Iceberg write combinations are maintained in +`iceberg_paimon_schema_time_travel_coverage.md` and `iceberg/write/ICEBERG_WRITE_P0_COVERAGE.md`. + +## Added test design + +| Case | Goal | Risks | Dimension | Preconditions | Load | Expected | +| --- | --- | --- | --- | --- | --- | --- | +| PM01 | Distinguish all four primary-key merge engines | R01, R02, R04 | Functional, correctness, compatibility | Paimon Parquet/ORC tables | Duplicate keys across several commits | Each engine returns its documented merged row under automatic and forced-JNI routing | +| PM02 | Validate dynamic-bucket cross-partition deduplication | R03, R04 | Correctness | Primary key excludes partition key, bucket=-1 | Move one key between partitions | Exactly one current row remains in the new partition | +| PM03 | Preserve the Paimon read-only boundary | R05, R10 | Negative, atomicity | Existing Paimon PK table | VALUES, SELECT, OVERWRITE, UPDATE, DELETE, MERGE | Every statement fails before a snapshot or data change | +| PM04 | Reject or roll back Paimon CTAS atomically | R05, R10 | Isolated negative, atomicity | Paimon catalog with no target table | CREATE TABLE AS SELECT | Desired contract: the command fails and no target table remains; current bug leaves the table | + +Every P0 risk maps to at least one deterministic positive, boundary, or isolated known-bug +regression. Catalog authentication and cloud storage permutations stay in their existing connector +suites because they do not add format semantics to this matrix. diff --git a/regression-test/suites/external_table_p0/paimon/test_paimon_ctas_atomicity_negative.groovy b/regression-test/suites/external_table_p0/paimon/test_paimon_ctas_atomicity_negative.groovy new file mode 100644 index 00000000000000..6d449f4af1e1b4 --- /dev/null +++ b/regression-test/suites/external_table_p0/paimon/test_paimon_ctas_atomicity_negative.groovy @@ -0,0 +1,74 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +suite("test_paimon_ctas_atomicity_negative", + "p0,external,paimon,external_docker,external_docker_paimon") { + String enabled = context.config.otherConfigs.get("enablePaimonTest") + if (enabled == null || !enabled.equalsIgnoreCase("true")) { + logger.info("disable paimon test") + return + } + + // CTAS currently creates Paimon metadata before discovering that no Paimon data sink exists. + // Keep this opt-in until the product rejects or rolls back the statement atomically. + String knownBugEnabled = context.config.otherConfigs.get("enablePaimonKnownBugTest") + if (knownBugEnabled == null || !knownBugEnabled.equalsIgnoreCase("true")) { + logger.info("skip isolated Paimon known-bug regression") + return + } + + String minioPort = context.config.otherConfigs.get("iceberg_minio_port") + String externalEnvIp = context.config.otherConfigs.get("externalEnvIp") + String catalogName = "test_paimon_ctas_atomicity_negative" + String dbName = "paimon_ctas_atomicity_negative_db" + + sql """drop catalog if exists ${catalogName}""" + sql """ + create catalog ${catalogName} properties ( + 'type'='paimon', + 'warehouse'='s3://warehouse/wh', + 's3.endpoint'='http://${externalEnvIp}:${minioPort}', + 's3.access_key'='admin', + 's3.secret_key'='password', + 's3.path.style.access'='true', + 'meta.cache.paimon.table.ttl-second'='0' + ) + """ + + try { + spark_paimon_multi """ + create database if not exists paimon.${dbName}; + drop table if exists paimon.${dbName}.ctas_target; + """ + + sql """switch ${catalogName}""" + sql """use ${dbName}""" + + // A failed CTAS must not leave metadata that makes a retry fail with TABLE ALREADY EXISTS. + test { + sql """ + create table ctas_target engine=paimon + as select cast(1 as int) as id, cast('candidate' as string) as payload + """ + exception "PaimonExternalCatalog" + } + assertEquals(0, (sql """show tables like 'ctas_target'""").size()) + } finally { + spark_paimon """drop table if exists paimon.${dbName}.ctas_target""" + sql """drop catalog if exists ${catalogName}""" + } +} diff --git a/regression-test/suites/external_table_p0/paimon/test_paimon_merge_engine_matrix.groovy b/regression-test/suites/external_table_p0/paimon/test_paimon_merge_engine_matrix.groovy new file mode 100644 index 00000000000000..266bc7c458b8ab --- /dev/null +++ b/regression-test/suites/external_table_p0/paimon/test_paimon_merge_engine_matrix.groovy @@ -0,0 +1,223 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +suite("test_paimon_merge_engine_matrix", + "p0,external,paimon,external_docker,external_docker_paimon") { + String enabled = context.config.otherConfigs.get("enablePaimonTest") + if (enabled == null || !enabled.equalsIgnoreCase("true")) { + logger.info("disable paimon test") + return + } + + String minioPort = context.config.otherConfigs.get("iceberg_minio_port") + String externalEnvIp = context.config.otherConfigs.get("externalEnvIp") + String catalogName = "test_paimon_merge_engine_matrix" + String dbName = "paimon_merge_engine_matrix_db" + + sql """drop catalog if exists ${catalogName}""" + sql """ + create catalog ${catalogName} properties ( + 'type'='paimon', + 'warehouse'='s3://warehouse/wh', + 's3.endpoint'='http://${externalEnvIp}:${minioPort}', + 's3.access_key'='admin', + 's3.secret_key'='password', + 's3.path.style.access'='true', + 'meta.cache.paimon.table.ttl-second'='0' + ) + """ + + try { + spark_paimon """create database if not exists paimon.${dbName}""" + + ["parquet", "orc"].each { String format -> + String deduplicateTable = "deduplicate_${format}" + String partialUpdateTable = "partial_update_${format}" + String aggregationTable = "aggregation_${format}" + String firstRowTable = "first_row_${format}" + + spark_paimon_multi """ + drop table if exists paimon.${dbName}.${deduplicateTable}; + create table paimon.${dbName}.${deduplicateTable} ( + id int, + score int, + note string + ) using paimon tblproperties ( + 'primary-key'='id', + 'bucket'='1', + 'merge-engine'='deduplicate', + 'file.format'='${format}' + ); + insert into paimon.${dbName}.${deduplicateTable} values + (1, 10, 'old-1'), + (2, 20, 'old-2'), + (3, 30, 'only-3'); + insert into paimon.${dbName}.${deduplicateTable} values + (1, 11, 'new-1'), + (2, 22, 'new-2'); + + drop table if exists paimon.${dbName}.${partialUpdateTable}; + create table paimon.${dbName}.${partialUpdateTable} ( + id int, + score int, + quantity int, + note string + ) using paimon tblproperties ( + 'primary-key'='id', + 'bucket'='1', + 'merge-engine'='partial-update', + 'file.format'='${format}' + ); + insert into paimon.${dbName}.${partialUpdateTable} values + (1, 10, 2, cast(null as string)), + (2, 30, cast(null as int), 'base-2'); + insert into paimon.${dbName}.${partialUpdateTable} values + (1, cast(null as int), cast(null as int), 'filled-1'), + (2, cast(null as int), 4, cast(null as string)); + insert into paimon.${dbName}.${partialUpdateTable} values + (1, 15, cast(null as int), cast(null as string)); + + drop table if exists paimon.${dbName}.${aggregationTable}; + create table paimon.${dbName}.${aggregationTable} ( + id int, + max_score int, + total bigint, + note string + ) using paimon tblproperties ( + 'primary-key'='id', + 'bucket'='1', + 'merge-engine'='aggregation', + 'fields.max_score.aggregate-function'='max', + 'fields.total.aggregate-function'='sum', + 'file.format'='${format}' + ); + insert into paimon.${dbName}.${aggregationTable} values + (1, 5, 10, 'old-1'), + (2, 9, 100, 'old-2'); + insert into paimon.${dbName}.${aggregationTable} values + (1, 7, 20, cast(null as string)), + (2, 8, 200, 'new-2'); + insert into paimon.${dbName}.${aggregationTable} values + (1, 6, 5, 'new-1'); + + drop table if exists paimon.${dbName}.${firstRowTable}; + create table paimon.${dbName}.${firstRowTable} ( + id int, + score int, + note string + ) using paimon tblproperties ( + 'primary-key'='id', + 'bucket'='1', + 'merge-engine'='first-row', + 'file.format'='${format}' + ); + insert into paimon.${dbName}.${firstRowTable} values + (1, 10, 'first-1'), + (2, 20, 'first-2'); + insert into paimon.${dbName}.${firstRowTable} values + (1, 99, 'ignored-1'), + (2, 88, 'ignored-2'), + (3, 30, 'first-3'); + """ + spark_paimon """ + call paimon.sys.compact( + table => '${dbName}.${firstRowTable}', + compact_strategy => 'full' + ) + """ + } + + // Dynamic buckets maintain a global key-to-partition mapping. Updating one key in a new + // partition must remove the old logical row instead of exposing both physical versions. + spark_paimon_multi """ + drop table if exists paimon.${dbName}.dynamic_cross_partition; + create table paimon.${dbName}.dynamic_cross_partition ( + id int, + part string, + score int + ) using paimon + partitioned by (part) + tblproperties ( + 'primary-key'='id', + 'bucket'='-1', + 'merge-engine'='deduplicate', + 'file.format'='parquet' + ); + insert into paimon.${dbName}.dynamic_cross_partition values + (1, 'old-part', 10), + (2, 'stable-part', 20); + insert into paimon.${dbName}.dynamic_cross_partition values + (1, 'new-part', 11); + """ + + sql """switch ${catalogName}""" + sql """use ${dbName}""" + + [false, true].each { boolean forceJni -> + // MOR primary-key splits normally stay on Paimon's merge-aware reader. Keep both + // automatic routing and the explicit JNI override so a future routing change cannot + // silently bypass the merge-engine contract. + String reader = forceJni ? "forced_jni" : "auto" + sql """set force_jni_scanner=${forceJni}""" + + ["parquet", "orc"].each { String format -> + // PM-ME01: last-write-wins must merge overlapping LSM sorted runs. + "order_qt_${reader}_${format}_deduplicate" """ + select id, score, note from deduplicate_${format} order by id + """ + + // PM-ME02: NULL means "field not supplied" for partial-update, not "erase value". + "order_qt_${reader}_${format}_partial_update" """ + select id, score, quantity, note from partial_update_${format} order by id + """ + + // PM-ME03: aggregation functions apply across files while the default value field + // keeps last_non_null_value semantics. + "order_qt_${reader}_${format}_aggregation" """ + select id, max_score, total, note from aggregation_${format} order by id + """ + + // PM-ME04: first-row is intentionally different from deduplicate for duplicate keys. + "order_qt_${reader}_${format}_first_row" """ + select id, score, note from first_row_${format} order by id + """ + + "qt_${reader}_${format}_engine_aggregates" """ + select + (select sum(score) from deduplicate_${format}), + (select sum(score + quantity) from partial_update_${format}), + (select sum(max_score) from aggregation_${format}), + (select sum(score) from first_row_${format}) + """ + } + + // PM-DD01: cross-partition deduplication must expose one current row per primary key. + "order_qt_${reader}_dynamic_cross_partition" """ + select id, part, score from dynamic_cross_partition order by id + """ + "qt_${reader}_dynamic_old_partition" """ + select count(*) from dynamic_cross_partition where part = 'old-part' + """ + "qt_${reader}_dynamic_new_partition" """ + select count(*) from dynamic_cross_partition where part = 'new-part' + """ + } + } finally { + sql """set force_jni_scanner=false""" + sql """drop catalog if exists ${catalogName}""" + } +} diff --git a/regression-test/suites/external_table_p0/paimon/test_paimon_write_boundary.groovy b/regression-test/suites/external_table_p0/paimon/test_paimon_write_boundary.groovy new file mode 100644 index 00000000000000..2f41229debcc17 --- /dev/null +++ b/regression-test/suites/external_table_p0/paimon/test_paimon_write_boundary.groovy @@ -0,0 +1,107 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +suite("test_paimon_write_boundary", + "p0,external,paimon,external_docker,external_docker_paimon") { + String enabled = context.config.otherConfigs.get("enablePaimonTest") + if (enabled == null || !enabled.equalsIgnoreCase("true")) { + logger.info("disable paimon test") + return + } + + String minioPort = context.config.otherConfigs.get("iceberg_minio_port") + String externalEnvIp = context.config.otherConfigs.get("externalEnvIp") + String catalogName = "test_paimon_write_boundary" + String dbName = "paimon_write_boundary_db" + sql """drop catalog if exists ${catalogName}""" + sql """ + create catalog ${catalogName} properties ( + 'type'='paimon', + 'warehouse'='s3://warehouse/wh', + 's3.endpoint'='http://${externalEnvIp}:${minioPort}', + 's3.access_key'='admin', + 's3.secret_key'='password', + 's3.path.style.access'='true', + 'meta.cache.paimon.table.ttl-second'='0' + ) + """ + + try { + spark_paimon_multi """ + create database if not exists paimon.${dbName}; + drop table if exists paimon.${dbName}.write_boundary; + create table paimon.${dbName}.write_boundary ( + id int, + score int, + note string + ) using paimon tblproperties ( + 'primary-key'='id', + 'bucket'='1', + 'file.format'='parquet' + ); + insert into paimon.${dbName}.write_boundary values + (1, 10, 'base-1'), + (2, 20, 'base-2'); + """ + + sql """switch ${catalogName}""" + sql """use ${dbName}""" + + qt_before_rows """select id, score, note from write_boundary order by id""" + qt_before_snapshots """select count(*) from write_boundary\$snapshots""" + + // WB01-WB06 preserve the documented data-write boundary at analysis time. The source table + // and its snapshot list must stay unchanged after every rejected write shape. + test { + sql """insert into write_boundary values (3, 30, 'insert-values')""" + exception "PaimonExternalCatalog" + } + test { + sql """insert into write_boundary select 3, 30, 'insert-select'""" + exception "PaimonExternalCatalog" + } + test { + sql """insert overwrite table write_boundary values (3, 30, 'overwrite')""" + exception "PaimonExternalCatalog" + } + test { + sql """update write_boundary set score = score + 1 where id = 1""" + exception "target table in update command should be an olapTable" + } + test { + sql """delete from write_boundary where id = 1""" + exception "delete command could be only used on olap table" + } + test { + sql """ + merge into write_boundary target + using (select 1 as id, 99 as score, 'merge' as note) source + on target.id = source.id + when matched then update set score = source.score, note = source.note + when not matched then insert (id, score, note) + values (source.id, source.score, source.note) + """ + exception "merge into command only support MOW unique key olapTable" + } + + sql """refresh table write_boundary""" + qt_after_rows """select id, score, note from write_boundary order by id""" + qt_after_snapshots """select count(*) from write_boundary\$snapshots""" + } finally { + sql """drop catalog if exists ${catalogName}""" + } +}