From e7e1f9b1c6624880252b8a98a6dddd3bef41728e Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Sun, 28 Jun 2026 20:03:23 +0530 Subject: [PATCH 01/20] feat(connectors): add generic JDBC source connector Adds a JDBC source connector that polls a configured SQL query against any JDBC-compliant database (PostgreSQL, MySQL, Oracle, SQL Server, H2) through an embedded JVM and produces each row as a JSON message. Supports bulk and incremental (offset-tracked) modes with persisted offset state. Also adds the shared connector integration-test harness and a small TCP listener change so that harness can read the server's written runtime config when bound to a fixed port. The matching JDBC sink connector follows in a separate PR. --- .config/nextest.toml | 12 + .github/workflows/_build_rust_artifacts.yml | 2 +- .github/workflows/edge-release.yml | 1 + Cargo.lock | 55 +- Cargo.toml | 1 + core/connectors/README.md | 1 + .../connectors/jdbc_bulk_mode.toml | 81 + .../example_config/connectors/jdbc_h2.toml | 60 + .../example_config/connectors/jdbc_mysql.toml | 85 + .../connectors/jdbc_oracle.toml | 82 + .../connectors/jdbc_sqlserver.toml | 66 + .../connectors/test_jdbc_h2.toml | 45 + core/connectors/sources/README.md | 1 + .../connectors/sources/jdbc_source/Cargo.toml | 67 + core/connectors/sources/jdbc_source/README.md | 539 ++++ .../sources/jdbc_source/config.toml | 50 + .../connectors/sources/jdbc_source/src/lib.rs | 2843 +++++++++++++++++ .../connectors/jdbc/config_postgres.toml | 22 + .../connectors_config_postgres/jdbc_pg.toml | 49 + core/integration/tests/connectors/jdbc/mod.rs | 21 + .../connectors/jdbc/test_with_postgres.rs | 624 ++++ core/integration/tests/connectors/mod.rs | 190 +- core/server/src/tcp/tcp_listener.rs | 8 +- 23 files changed, 4898 insertions(+), 7 deletions(-) create mode 100644 core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml create mode 100644 core/connectors/runtime/example_config/connectors/jdbc_h2.toml create mode 100644 core/connectors/runtime/example_config/connectors/jdbc_mysql.toml create mode 100644 core/connectors/runtime/example_config/connectors/jdbc_oracle.toml create mode 100644 core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml create mode 100644 core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml create mode 100644 core/connectors/sources/jdbc_source/Cargo.toml create mode 100644 core/connectors/sources/jdbc_source/README.md create mode 100644 core/connectors/sources/jdbc_source/config.toml create mode 100644 core/connectors/sources/jdbc_source/src/lib.rs create mode 100644 core/integration/tests/connectors/jdbc/config_postgres.toml create mode 100644 core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml create mode 100644 core/integration/tests/connectors/jdbc/mod.rs create mode 100644 core/integration/tests/connectors/jdbc/test_with_postgres.rs diff --git a/.config/nextest.toml b/.config/nextest.toml index 3070df8bdc..916c27ecd0 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -48,6 +48,18 @@ max-threads = 1 filter = 'package(integration) and test(/connectors::elasticsearch::/)' test-group = "elasticsearch" +# JDBC tests each start their own iggy-server plus a Postgres testcontainer and +# an embedded JVM, and they share a JDBC driver JAR downloaded to a single path +# on first run. nextest runs each test in its own process, so the in-source +# `#[serial]` guard does not serialize them here; `max-threads = 1` does, which +# avoids resource contention and a race to download the same driver JAR. +[test-groups.jdbc] +max-threads = 1 + +[[profile.default.overrides]] +filter = 'package(integration) and test(/connectors::jdbc::/)' +test-group = "jdbc" + [profile.default] slow-timeout = { period = "60s", terminate-after = 5 } diff --git a/.github/workflows/_build_rust_artifacts.yml b/.github/workflows/_build_rust_artifacts.yml index 5dfe055149..316bfc2ade 100644 --- a/.github/workflows/_build_rust_artifacts.yml +++ b/.github/workflows/_build_rust_artifacts.yml @@ -46,7 +46,7 @@ on: connector_plugins: type: string required: false - default: "iggy_connector_elasticsearch_sink,iggy_connector_elasticsearch_source,iggy_connector_iceberg_sink,iggy_connector_postgres_sink,iggy_connector_postgres_source,iggy_connector_quickwit_sink,iggy_connector_random_source,iggy_connector_s3_sink,iggy_connector_stdout_sink,iggy_connector_surrealdb_sink" + default: "iggy_connector_elasticsearch_sink,iggy_connector_elasticsearch_source,iggy_connector_iceberg_sink,iggy_connector_jdbc_source,iggy_connector_postgres_sink,iggy_connector_postgres_source,iggy_connector_quickwit_sink,iggy_connector_random_source,iggy_connector_s3_sink,iggy_connector_stdout_sink,iggy_connector_surrealdb_sink" description: "Comma-separated list of connector plugin crates to build as shared libraries" outputs: artifact_name: diff --git a/.github/workflows/edge-release.yml b/.github/workflows/edge-release.yml index 87b7eb3b84..613d6a4b95 100644 --- a/.github/workflows/edge-release.yml +++ b/.github/workflows/edge-release.yml @@ -104,6 +104,7 @@ jobs: - `iggy_connector_elasticsearch_sink` - `iggy_connector_elasticsearch_source` - `iggy_connector_iceberg_sink` + - `iggy_connector_jdbc_source` - `iggy_connector_postgres_sink` - `iggy_connector_postgres_source` - `iggy_connector_quickwit_sink` diff --git a/Cargo.lock b/Cargo.lock index 22cbf3cfe2..e274442948 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2742,7 +2742,7 @@ checksum = "0b023947811758c97c59bf9d1c188fd619ad4718dcaa767947df1cadb14f39f4" dependencies = [ "glob", "libc", - "libloading", + "libloading 0.8.9", ] [[package]] @@ -6203,6 +6203,16 @@ version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424" +[[package]] +name = "humantime-serde" +version = "1.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "57a3db5ea5923d99402c94e9feb261dc5ee9b4efa158b0315f788cf549cc200c" +dependencies = [ + "humantime", + "serde", +] + [[package]] name = "hwlocality" version = "1.0.0-alpha.12" @@ -7024,6 +7034,28 @@ dependencies = [ "uuid", ] +[[package]] +name = "iggy_connector_jdbc_source" +version = "0.1.0" +dependencies = [ + "async-trait", + "base64", + "chrono", + "dashmap", + "humantime-serde", + "iggy_common", + "iggy_connector_sdk", + "jni 0.21.1", + "regex", + "secrecy", + "serde", + "serde_json", + "tokio", + "toml 1.1.2+spec-1.1.0", + "tracing", + "uuid", +] + [[package]] name = "iggy_connector_mongodb_sink" version = "0.4.1-edge.1" @@ -7526,6 +7558,15 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "java-locator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09c46c1fe465c59b1474e665e85e1256c3893dd00927b8d55f63b09044c1e64f" +dependencies = [ + "glob", +] + [[package]] name = "jiff" version = "0.2.29" @@ -7577,7 +7618,9 @@ dependencies = [ "cesu8", "cfg-if", "combine", + "java-locator", "jni-sys 0.3.1", + "libloading 0.7.4", "log", "thiserror 1.0.69", "walkdir", @@ -7906,6 +7949,16 @@ dependencies = [ "pkg-config", ] +[[package]] +name = "libloading" +version = "0.7.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67380fd3b2fbe7527a606e18729d21c6f3951633d0500574c4dc22d2d638b9f" +dependencies = [ + "cfg-if", + "winapi", +] + [[package]] name = "libloading" version = "0.8.9" diff --git a/Cargo.toml b/Cargo.toml index 87a96add3e..543b15ae83 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -48,6 +48,7 @@ members = [ "core/connectors/sinks/surrealdb_sink", "core/connectors/sources/elasticsearch_source", "core/connectors/sources/influxdb_source", + "core/connectors/sources/jdbc_source", "core/connectors/sources/postgres_source", "core/connectors/sources/random_source", "core/consensus", diff --git a/core/connectors/README.md b/core/connectors/README.md index 64bfc9a1f8..d93d22d99e 100644 --- a/core/connectors/README.md +++ b/core/connectors/README.md @@ -98,6 +98,7 @@ Please refer to the **[Source documentation](https://github.com/apache/iggy/tree ### Available Sources - **Elasticsearch Source** - polls documents from Elasticsearch indices +- **JDBC Source** - reads rows from any JDBC-compliant database (PostgreSQL, MySQL, Oracle, SQL Server, H2) via an embedded JVM; bulk and incremental modes - **PostgreSQL Source** - reads rows from PostgreSQL tables with multiple consumption strategies (delete after read, mark as processed, timestamp tracking) - **Random Source** - generates random test messages (useful for testing/development) diff --git a/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml new file mode 100644 index 0000000000..25e2b458ae --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml @@ -0,0 +1,81 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration - BULK MODE +# Bulk mode works with ALL JDBC databases without any special requirements +# No tracking column needed - just executes your query and fetches results + +type = "source" +key = "jdbc_bulk_example" +enabled = true +version = 0 +name = "JDBC Bulk Mode Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# This example uses PostgreSQL, but bulk mode works identically with: +# MySQL, Oracle, SQL Server, H2, Derby, DB2, etc. +jdbc_url = "jdbc:postgresql://localhost:5432/warehouse" + +# Database credentials can be in URL or separate +# jdbc_url = "jdbc:postgresql://localhost:5432/warehouse?user=myuser&password=mypass" +username = "warehouse_user" +password = "secret" + +driver_class = "org.postgresql.Driver" +driver_jar_path = "/opt/jdbc-drivers/postgresql-42.6.0.jar" + +# Bulk mode: Any valid SELECT query +# Can include JOINs, aggregations, complex WHERE clauses, etc. +query = """ +SELECT + p.product_id,\ + p.product_name, + p.category, + p.price, + COUNT(o.order_id) as total_orders, + SUM(o.quantity) as total_quantity +FROM products p +LEFT JOIN orders o ON p.product_id = o.product_id +GROUP BY p.product_id, p.product_name, p.category, p.price +""" + +# Poll once per hour for daily snapshots +poll_interval = "1h" + +# Large batch size for full table scans +batch_size = 10000 + +# BULK MODE - no tracking column needed! +mode = "bulk" + +# Bulk mode benefits: +# - No tracking column required +# - Works with any SELECT query +# - Supports complex queries (JOINs, aggregations, window functions) +# - Perfect for periodic snapshots +# - Universal compatibility with all databases + +snake_case_columns = true +include_metadata = false + +[[streams]] +stream = "warehouse" +topic = "product_summary" +partition_id = 1 +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml new file mode 100644 index 0000000000..82ee459ff5 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml @@ -0,0 +1,60 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for H2 Database +# H2 is useful for testing and development as it's an embedded Java database + +type = "source" +key = "jdbc_h2_example" +enabled = true +version = 0 +name = "JDBC H2 Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# H2 connection URL (in-memory database) +jdbc_url = "jdbc:h2:mem:testdb;DB_CLOSE_DELAY=-1" + +# H2 JDBC driver +driver_class = "org.h2.Driver" + +# Path to H2 driver JAR +# Download from: https://repo1.maven.org/maven2/com/h2database/h2/2.2.224/h2-2.2.224.jar +# Note: Update this path to match where you downloaded the JAR file +driver_jar_path = "/tmp/jdbc-drivers/h2-2.2.224.jar" + +# H2 credentials (default) +username = "sa" +password = "" + +# Simple query for testing +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" + +poll_interval = "10s" +batch_size = 100 +tracking_column = "id" +initial_offset = "0" +mode = "incremental" +snake_case_columns = false +include_metadata = true + +[[streams]] +stream = "test" +topic = "users" +partition_id = 1 +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml b/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml new file mode 100644 index 0000000000..fd2ed5fc48 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml @@ -0,0 +1,85 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for MySQL +# This file demonstrates how to configure the JDBC source connector +# to read data from a MySQL database and publish to Iggy streams. + +type = "source" +key = "jdbc_mysql_example" +enabled = true +version = 0 +name = "JDBC MySQL Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# JDBC connection URL +# Option 1: Separate credentials (recommended) +jdbc_url = "jdbc:mysql://localhost:3306/ecommerce?useSSL=false&serverTimezone=UTC" + +# Option 2: Embedded credentials in URL (alternative) +# jdbc_url = "jdbc:mysql://iggy_user:iggy_password@localhost:3306/ecommerce?useSSL=false&serverTimezone=UTC" + +# JDBC driver class name +driver_class = "com.mysql.cj.jdbc.Driver" + +# Path to JDBC driver JAR file +# Download from: https://repo1.maven.org/maven2/com/mysql/mysql-connector-j/8.0.33/mysql-connector-j-8.0.33.jar +driver_jar_path = "/opt/jdbc-drivers/mysql-connector-j-8.0.33.jar" + +# Database credentials (optional if included in jdbc_url) +username = "iggy_user" +password = "iggy_password" + +# SQL query to execute +# Use {last_offset} placeholder for incremental reads +query = "SELECT * FROM orders WHERE updated_at > {last_offset} ORDER BY updated_at ASC" + +# How often to poll the database +poll_interval = "30s" + +# Maximum number of rows to fetch per poll +batch_size = 1000 + +# Column to track for incremental reads (must be in query result) +tracking_column = "updated_at" + +# Initial offset value for the first poll +initial_offset = "2024-01-01 00:00:00" + +# Source mode: "incremental" or "bulk" +# Note: Both modes work with ALL JDBC databases (MySQL, Oracle, PostgreSQL, etc.) +# - incremental: Tracks last offset, avoids duplicate reads +# - bulk: Full table scan, no offset tracking +mode = "incremental" + +# Convert column names to snake_case (e.g., OrderDate -> order_date) +snake_case_columns = true + +# Include metadata wrapper in output messages +include_metadata = true + +# Custom JVM options (optional) +jvm_options = ["-Xmx512m", "-Xms128m"] + +# Target Iggy stream and topic +[[streams]] +stream = "ecommerce" +topic = "orders" +partition_id = 1 +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml new file mode 100644 index 0000000000..04478bf4ab --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml @@ -0,0 +1,82 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for Oracle Database + +type = "source" +key = "jdbc_oracle_example" +enabled = true +version = 0 +name = "JDBC Oracle Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# JDBC connection URL +# Option 1: Separate credentials +jdbc_url = "jdbc:oracle:thin:@localhost:1521:XE" + +# Option 2: Embedded credentials in URL (Oracle uses / separator) +# jdbc_url = "jdbc:oracle:thin:system/oracle@localhost:1521:XE" + +# JDBC driver class name +driver_class = "oracle.jdbc.OracleDriver" + +# Path to JDBC driver JAR file +# Download from: https://www.oracle.com/database/technologies/appdev/jdbc-downloads.html +driver_jar_path = "/opt/jdbc-drivers/ojdbc11.jar" + +# Database credentials (optional if included in jdbc_url) +username = "system" +password = "oracle" + +# SQL query to execute +# Oracle example with ROWNUM or use a numeric/timestamp column +query = "SELECT * FROM CUSTOMERS WHERE ID > {last_offset} ORDER BY ID" + +# How often to poll the database +poll_interval = "1m" + +# Maximum number of rows to fetch per poll +batch_size = 500 + +# Column to track for incremental reads (must be in query result) +tracking_column = "ID" + +# Initial offset value for the first poll +initial_offset = "0" + +# Source mode: "incremental" or "bulk" +# Works with ALL JDBC databases universally +mode = "incremental" + +# Convert column names to snake_case (e.g., OrderDate -> order_date) +snake_case_columns = true + +# Include metadata wrapper in output messages +include_metadata = true + + +# Custom JVM options (optional) +jvm_options = ["-Xmx512m", "-Xms256m"] + +# Target Iggy stream and topic +[[streams]] +stream = "crm" +topic = "customers" +partition_id = 1 +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml b/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml new file mode 100644 index 0000000000..9e28d08a03 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml @@ -0,0 +1,66 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for Microsoft SQL Server + +type = "source" +key = "jdbc_sqlserver_example" +enabled = true +version = 0 +name = "JDBC SQL Server Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# JDBC connection URL +# Option 1: Separate credentials +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=Sales;encrypt=false" + +# Option 2: Embedded credentials in URL +# jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=Sales;user=sa;password=YourPassword123;encrypt=false" + +# JDBC driver class name +driver_class = "com.microsoft.sqlserver.jdbc.SQLServerDriver" + +# Path to JDBC driver JAR file +# Download from: https://repo1.maven.org/maven2/com/microsoft/sqlserver/mssql-jdbc/ +driver_jar_path = "/opt/jdbc-drivers/mssql-jdbc-12.4.1.jre11.jar" + +# Database credentials (optional if included in jdbc_url) +username = "sa" +password = "YourPassword123" + +# SQL query to execute +query = "SELECT * FROM Orders WHERE OrderDate > {last_offset} ORDER BY OrderDate" + +poll_interval = "15s" +batch_size = 2000 +tracking_column = "OrderDate" +initial_offset = "2024-01-01" +mode = "incremental" + +# Convert SQL Server naming to snake_case +snake_case_columns = true +include_metadata = true + +connection_timeout_ms = 30000 + +[[streams]] +stream = "sales" +topic = "orders" +partition_id = 1 +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml new file mode 100644 index 0000000000..06b65c21b2 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml @@ -0,0 +1,45 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "source" +key = "jdbc_h2_test" +enabled = true +version = 0 +name = "JDBC H2 Test Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +jdbc_url = "jdbc:h2:mem:testdb;DB_CLOSE_DELAY=-1;INIT=CREATE TABLE IF NOT EXISTS users (id INT PRIMARY KEY AUTO_INCREMENT, name VARCHAR(100), email VARCHAR(100), created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP);INSERT INTO users (name, email) SELECT 'User ' || x, 'user' || x || '@test.com' FROM SYSTEM_RANGE(1, 100) WHERE NOT EXISTS (SELECT 1 FROM users);" +driver_class = "org.h2.Driver" +driver_jar_path = "/tmp/jdbc-drivers/h2-2.2.224.jar" +username = "sa" +password = "" +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" +poll_interval = "5s" +batch_size = 10 +tracking_column = "id" +initial_offset = "0" +mode = "incremental" +snake_case_columns = true +include_metadata = true + +[[streams]] +stream = "test" +topic = "users" +partition_id = 1 +schema = "json" diff --git a/core/connectors/sources/README.md b/core/connectors/sources/README.md index 34989aef00..cea774355a 100644 --- a/core/connectors/sources/README.md +++ b/core/connectors/sources/README.md @@ -10,6 +10,7 @@ Source connectors are responsible for ingesting data from external sources into | ------ | ----------- | | **elasticsearch_source** | Polls documents from Elasticsearch indices with timestamp-based tracking | | **influxdb_source** | Polls InfluxDB with cursor-based timestamp tracking; supports V2 (Flux, annotated CSV) and V3 (SQL, JSONL) | +| **jdbc_source** | Reads rows from any JDBC-compliant database (PostgreSQL, MySQL, Oracle, SQL Server, H2) via an embedded JVM; bulk and incremental modes | | **postgres_source** | Reads rows from PostgreSQL tables with multiple strategies: delete after read, mark as processed, or timestamp tracking | | **random_source** | Generates random test messages (useful for testing and development) | diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml new file mode 100644 index 0000000000..c3eabd9e61 --- /dev/null +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -0,0 +1,67 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +[package] +name = "iggy_connector_jdbc_source" +version = "0.1.0" +edition = "2024" +license = "Apache-2.0" +keywords = ["iggy", "messaging", "streaming", "jdbc", "source"] +categories = ["database"] +description = "Generic JDBC source connector for Iggy - supports MySQL, Oracle, SQL Server, H2, and any JDBC-compliant database" +readme = "README.md" + +[package.metadata.cargo-machete] +ignored = ["dashmap", "humantime-serde"] + +[lib] +crate-type = ["cdylib", "rlib"] + +[features] +default = [] + +[dependencies] +async-trait = { workspace = true } +base64 = { workspace = true } +chrono = { workspace = true } + +# Required by source_connector! macro +dashmap = { workspace = true } + +# For parsing duration strings +humantime-serde = "1.1" + +# Shared serde helpers (SecretString serialization) +iggy_common = { workspace = true } + +# Connector SDK +iggy_connector_sdk = { workspace = true } + +# JNI for Java interop with invocation support +jni = { version = "0.21", features = ["invocation"] } + +# For sanitizing passwords in logs +regex = { workspace = true } +secrecy = { workspace = true } +serde = { workspace = true, features = ["derive"] } +serde_json = { workspace = true } +tokio = { workspace = true, features = ["full"] } +tracing = { workspace = true } +uuid = { workspace = true, features = ["v4"] } + +[dev-dependencies] +toml = { workspace = true } diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md new file mode 100644 index 0000000000..c2e4d979b4 --- /dev/null +++ b/core/connectors/sources/jdbc_source/README.md @@ -0,0 +1,539 @@ +# JDBC Source Connector + +A generic JDBC source connector for Iggy that supports any JDBC-compliant database including MySQL, PostgreSQL, Oracle, SQL Server, H2, Derby, and more. + +## Overview + +This connector reads data from relational databases using JDBC (Java Database Connectivity) and publishes it as messages to Iggy streams. It supports both bulk and incremental data synchronization modes. + +## Features + +- **Universal Database Support**: Works with any database that has a JDBC driver +- **Incremental Sync**: Track changes using timestamps or auto-increment IDs +- **Bulk Mode**: Re-runs the query each poll for snapshots (capped at `batch_size` rows; see limitations) +- **Type Mapping**: Automatic conversion of SQL types to JSON +- **Configurable Polling**: Control how frequently data is fetched +- **State Management**: Automatically tracks offsets to prevent duplicate reads +- **Flexible Queries**: Support for custom SQL queries with placeholders + +## Supported Databases + +**ALL JDBC-compliant databases are supported for both bulk and incremental modes:** + +- MySQL / MariaDB +- PostgreSQL +- Oracle Database +- Microsoft SQL Server +- H2 Database +- Apache Derby +- IBM DB2 +- SQLite (via JDBC) +- SAP HANA +- Teradata +- Snowflake +- Amazon Redshift +- Google BigQuery +- Any other JDBC-compliant database + +**Key Point:** The JDBC connector provides a **single, universal implementation** that works with all these databases. You don't need separate connectors for MySQL, Oracle, etc. Just swap the JDBC driver JAR and connection string! + +## Prerequisites + +1. **Java Runtime Environment (JRE)**: JRE 8 or later must be installed +2. **JDBC Driver**: Download the appropriate JDBC driver JAR for your database + +### Downloading JDBC Drivers + +**MySQL:** + +```bash +wget https://repo1.maven.org/maven2/com/mysql/mysql-connector-j/8.0.33/mysql-connector-j-8.0.33.jar +``` + +**PostgreSQL:** + +```bash +wget https://jdbc.postgresql.org/download/postgresql-42.6.0.jar +``` + +**Oracle:** + +- Download from [Oracle JDBC Driver Downloads](https://www.oracle.com/database/technologies/appdev/jdbc-downloads.html) + +**SQL Server:** + +```bash +wget https://repo1.maven.org/maven2/com/microsoft/sqlserver/mssql-jdbc/12.4.1.jre11/mssql-jdbc-12.4.1.jre11.jar +``` + +**H2:** + +```bash +wget https://repo1.maven.org/maven2/com/h2database/h2/2.2.224/h2-2.2.224.jar +``` + +## Configuration + +### Basic Configuration (Incremental Sync) + +```toml +type = "source" +key = "jdbc_mysql_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:mysql://localhost:3306/ecommerce" +driver_class = "com.mysql.cj.jdbc.Driver" +driver_jar_path = "/opt/jdbc-drivers/mysql-connector-j-8.0.33.jar" +username = "iggy_user" +password = "secret_password" +query = "SELECT * FROM orders WHERE updated_at > {last_offset} ORDER BY updated_at ASC" +poll_interval = "30s" +batch_size = 1000 +tracking_column = "updated_at" +initial_offset = "2024-01-01 00:00:00" +mode = "incremental" +snake_case_columns = true +include_metadata = true + +[[streams]] +stream = "ecommerce" +topic = "orders" +partition_id = 1 +``` + +### Bulk Mode Configuration + +```toml +type = "source" +key = "jdbc_bulk_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:postgresql://localhost:5432/warehouse" +driver_class = "org.postgresql.Driver" +driver_jar_path = "/opt/jdbc-drivers/postgresql-42.6.0.jar" +username = "warehouse_user" +password = "secret" +query = "SELECT * FROM product_catalog" +poll_interval = "1h" +batch_size = 5000 +mode = "bulk" +snake_case_columns = false +include_metadata = true + +[[streams]] +stream = "warehouse" +topic = "products" +``` + +### Oracle Database Example + +```toml +type = "source" +key = "jdbc_oracle_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:oracle:thin:@localhost:1521:XE" +driver_class = "oracle.jdbc.OracleDriver" +driver_jar_path = "/opt/jdbc-drivers/ojdbc11.jar" +username = "system" +password = "oracle" +query = "SELECT * FROM CUSTOMERS WHERE ID > {last_offset} ORDER BY ID" +poll_interval = "1m" +batch_size = 500 +tracking_column = "ID" +initial_offset = "0" +mode = "incremental" +jvm_options = ["-Xmx256m", "-Xms128m"] + +[[streams]] +stream = "crm" +topic = "customers" +``` + +### SQL Server Example + +```toml +type = "source" +key = "jdbc_sqlserver_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=Sales;encrypt=false" +driver_class = "com.microsoft.sqlserver.jdbc.SQLServerDriver" +driver_jar_path = "/opt/jdbc-drivers/mssql-jdbc-12.4.1.jre11.jar" +username = "sa" +password = "YourPassword123" +query = "SELECT * FROM Orders WHERE OrderDate > {last_offset} ORDER BY OrderDate" +poll_interval = "15s" +batch_size = 2000 +tracking_column = "OrderDate" +initial_offset = "2024-01-01" +mode = "incremental" + +[[streams]] +stream = "sales" +topic = "orders" +``` + +## Configuration Parameters + +| Parameter | Type | Required | Default | Description | +| ----------- | ------ | ---------- | --------- | ------------- | +| `jdbc_url` | string | Yes | - | JDBC connection URL (can include credentials) | +| `driver_class` | string | Yes | - | JDBC driver class name | +| `driver_jar_path` | string | Yes | - | Path to the JDBC driver JAR (checked to exist at startup; passed to the embedded JVM as `-Djava.class.path`) | +| `username` | string | No | - | Database username (optional if in jdbc_url) | +| `password` | string | No | - | Database password (optional if in jdbc_url) | +| `query` | string | Yes | - | SQL query to execute (supports `{last_offset}` and `{tracking_column}` placeholders) | +| `poll_interval` | duration | Yes | - | How often to poll (e.g., "30s", "5m", "1h") | +| `batch_size` | u32 | No | 1000 | Maximum rows to fetch per poll | +| `tracking_column` | string | Incremental | - | Column to track for incremental reads (required in incremental mode; the query must also `ORDER BY` it) | +| `initial_offset` | string | No | - | Starting offset value for first poll | +| `mode` | string | No | "incremental" | Sync mode: "incremental" or "bulk" (bulk works with ALL databases) | +| `connection_timeout_ms` | u64 | No | 30000 | Timeout (ms) for the per-poll connection liveness check | +| `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | +| `snake_case_columns` | bool | No | false | Convert column names to snake_case | +| `include_metadata` | bool | No | true | Include metadata (table, operation, timestamp) | + +## Query Placeholders + +The `query` parameter supports placeholders for dynamic queries: + +- `{last_offset}`: Replaced with the last tracked offset value, wrapped in quotes and escaped +- `{tracking_column}`: Replaced with the configured `tracking_column` (validated as a plain SQL identifier) + +Incremental mode is validated at `open()` and enforces the following (the +connector refuses to start otherwise): + +- **`tracking_column` is required.** Without it the offset can never advance and + every poll re-reads the same rows. +- **The query must order by the tracking column, ascending, as the first + `ORDER BY` term.** Row limiting uses `setMaxRows`, so an unordered (or + otherwise-ordered) query returns an arbitrary subset; advancing the offset to + that subset's max would permanently skip the unread lower keys. The validator + therefore requires the tracking column to be the **first** ordering term of the + outer `ORDER BY` and rejects a descending (`DESC`) direction. Write either + `ORDER BY {tracking_column}` or the column name (optionally table-qualified, + e.g. `ORDER BY t.updated_at`). A composite order such as `ORDER BY other, id` + (tracking column not first) or a `DESC` order is rejected at `open()`. + +The tracking column must also be: + +- **Homogeneously typed and monotonic.** Offsets are compared as integers, then + floats, then lexicographically; a column mixing numeric-looking and + non-numeric strings can make the connector's comparison disagree with the + database's `>` and skip or re-read rows. Prefer an auto-increment ID or a + timestamp. +- **`NOT NULL`.** A NULL tracking value cannot be watermarked, so the connector + errors the poll if it reads one and keeps erroring (making no progress) until + the query is fixed. Exclude NULLs in the query, e.g. `AND {tracking_column} IS + NOT NULL`. +- **Lexicographically ordered, for timestamps.** Timestamp columns are read via + the driver's string form; ensure that form is ordered (ISO-8601 is). A locale + format such as `MM/DD/YYYY` is not monotonic as text and will misorder. + +**Example:** + +```sql +-- Configuration +tracking_column = "id" +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" + +-- First poll (no offset yet) +SELECT * FROM users WHERE id > '0' ORDER BY id + +-- After processing rows up to id=100 +SELECT * FROM users WHERE id > '100' ORDER BY id +``` + +## Output Format + +Each database row is converted to a JSON message: + +### With Metadata (default) + +```json +{ + "table_name": null, + "operation_type": "SELECT", + "timestamp": "2024-01-09T10:30:00Z", + "data": { + "id": 123, + "name": "John Doe", + "email": "john@example.com", + "created_at": "2024-01-08T15:20:00" + } +} +``` + +### Without Metadata + +```json +{ + "id": 123, + "name": "John Doe", + "email": "john@example.com", + "created_at": "2024-01-08T15:20:00" +} +``` + +## Type Mapping + +JDBC SQL types are automatically mapped to JSON: + +| SQL Type | JSON Type | Notes | +| ---------- | ----------- | ------- | +| BIT, BOOLEAN | boolean | - | +| TINYINT, SMALLINT, INTEGER | number | Integer | +| BIGINT | number | Long integer (values above 2^53 may lose precision in JSON consumers that parse numbers as f64) | +| FLOAT, REAL | number | Float | +| DOUBLE | number | Double | +| NUMERIC, DECIMAL | string | Emitted as a string to preserve arbitrary precision (e.g. money) | +| CHAR, VARCHAR, TEXT | string | - | +| DATE, TIME, TIMESTAMP | string | Driver string form | +| BINARY, VARBINARY, LONGVARBINARY | string | Base64 encoded | +| NULL | null | - | + +## Runtime notes & limitations + +- **Embedded JVM, one per process.** JNI permits a single `JavaVM` per OS + process. All JDBC *source* instances in the connectors runtime share one JVM + (the first instance's `jvm_options`/classpath win). A JDBC source and a JDBC + sink are separate shared libraries and **cannot both create a JVM in the same + runtime process** — run them in separate connectors-runtime processes. +- **Blocking I/O.** JDBC calls go through JNI and are synchronous. The fetch in + `poll()` (and the close in `close()`) runs under `tokio::task::block_in_place` + so it does not monopolize a shared async-runtime worker, but the work is still + blocking; size the runtime and `poll_interval`/`batch_size` accordingly. +- **Bulk mode has no pagination beyond `batch_size`, and fails closed.** Row + limiting uses JDBC `setMaxRows`. In bulk mode the connector probes with + `batch_size + 1` rows and, if the result set is larger than `batch_size`, + **errors the poll instead of syncing a truncated subset**. For tables larger + than a batch, raise `batch_size` to cover the full result, or use incremental + mode with an ordered `tracking_column`. (Full cross-database OFFSET pagination + is a planned follow-up.) +- **Delivery semantics.** State (the incremental offset) is persisted by the + runtime only after a batch is successfully sent, so offsets are **at-least-once + across restarts**: a crash or restart never skips rows. One narrower gap + remains: a *transient in-process send failure without a restart* can skip the + batch it happened on, because the in-memory offset is not rolled back. This + matches the other offset-tracking source connectors and is a runtime-level + limitation (there is no per-poll delivery ack to the connector); a stronger + guarantee is tracked as a follow-up. +- **Connection recovery.** The connection is validated with `Connection.isValid` + each poll and transparently re-established (closing the old handle) if it has + dropped. +- **`SQLState` classification is informational today.** Query failures are + classified into transient vs permanent error variants, but the runtime does + not yet apply differentiated backoff based on that distinction; it currently + shapes the error variant and the log message only. +- **Credentials reach the JVM heap unzeroed.** `password` is held as a + `SecretString` on the Rust side, but the JDBC API takes a `java.lang.String`, + so the password is copied onto the JVM heap as an ordinary (non-zeroed) string + for the lifetime of the connection. This is inherent to the JDBC surface and + is an accepted risk. + +### Credential precedence + +Provide credentials **either** via `username` + `password` **or** embedded in the +`jdbc_url`, not both: + +- `username` and `password` must be **both set** (separate-credential auth) or + **both unset** (URL-embedded credentials). A half-set pair is rejected at + `open()`. +- When both `username`/`password` and URL-embedded credentials are present, the + driver decides precedence (typically the explicit `getConnection(url, user, + pass)` arguments win). Avoid the ambiguity by using only one method. + +## Troubleshooting + +### Connection Failures + +**Error**: "Failed to create JDBC connection" + +**Solution**: + +- Verify JDBC URL format for your database +- Check username/password +- Ensure database server is accessible +- Verify firewall rules + +### Driver Not Found + +**Error**: "Failed to find driver class" + +**Solution**: + +- Verify `driver_jar_path` points to correct JAR file +- Check `driver_class` name matches your JDBC driver +- Ensure JAR file has read permissions + +### JVM Issues + +**Error**: "Failed to create JVM" + +**Solution**: + +- Ensure Java is installed: `java -version` +- Increase JVM memory: + + ```toml + jvm_options = ["-Xmx1g", "-Xms512m"] + ``` + +### No Data Being Fetched + +**Check**: + +- Verify query returns results when run directly in database +- Check `initial_offset` value +- Review connector logs for errors +- Ensure `tracking_column` exists in query result + +## Performance Tuning + +### Optimize Batch Size + +```toml +# Small batches for low latency +batch_size = 100 +poll_interval = "5s" + +# Large batches for throughput +batch_size = 10000 +poll_interval = "1m" +``` + +### JVM Memory Tuning + +```toml +jvm_options = [ + "-Xmx1g", # Maximum heap size + "-Xms512m", # Initial heap size + "-XX:+UseG1GC" # Use G1 garbage collector +] +``` + +### Query Optimization + +- Add indexes on tracking columns +- Use efficient WHERE clauses +- Avoid SELECT * in production (specify columns) +- Consider database-specific optimizations + +## Connection String Formats + +### MySQL + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:mysql://localhost:3306/mydb" +username = "user" +password = "pass" + +# Option 2: Embedded in URL +jdbc_url = "jdbc:mysql://user:pass@localhost:3306/mydb" +``` + +### PostgreSQL + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:postgresql://localhost:5432/mydb" +username = "user" +password = "pass" + +# Option 2: Embedded in URL +jdbc_url = "jdbc:postgresql://localhost:5432/mydb?user=myuser&password=mypass" +``` + +### Oracle + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:oracle:thin:@localhost:1521:XE" +username = "system" +password = "oracle" + +# Option 2: Embedded in URL (Oracle uses @ for host) +jdbc_url = "jdbc:oracle:thin:system/oracle@localhost:1521:XE" +``` + +### SQL Server + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=mydb" +username = "sa" +password = "YourPassword123" + +# Option 2: Embedded in URL +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=mydb;user=sa;password=YourPassword123" +``` + +### H2 (In-Memory) + +```toml +# No credentials needed for in-memory +jdbc_url = "jdbc:h2:mem:testdb" + +# Or with file-based +jdbc_url = "jdbc:h2:file:/data/mydb;USER=sa;PASSWORD=sa" +``` + +## Mode Comparison + +### Incremental Mode (Universal) + +**Works with ALL databases** - requires only a tracking column: + +```toml +mode = "incremental" +tracking_column = "updated_at" # or "id", "created_at", etc. +query = "SELECT * FROM table WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" +``` + +**Benefits:** + +- Prevents duplicate reads +- Tracks offset automatically +- Efficient for large tables +- Works with timestamps, IDs, or any orderable column + +**Database Examples** (the query must order by the tracking column): + +- MySQL: `WHERE updated_at > {last_offset} ORDER BY updated_at` +- Oracle: `WHERE id > {last_offset} ORDER BY id` (use a monotonic key; `ROWNUM` is not a valid tracking column) +- SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` +- PostgreSQL: `WHERE id > {last_offset} ORDER BY id` + +### Bulk Mode (Universal) + +**Works with ALL databases** - no special requirements: + +```toml +mode = "bulk" +query = "SELECT * FROM table" # Any valid SELECT query +``` + +**Benefits:** + +- No tracking column needed +- Works with any SELECT query +- Good for snapshots +- Supports complex queries with JOINs, aggregations, etc. + +**Limitation:** the result set is capped at `batch_size` rows (`setMaxRows`) with +no pagination beyond it. Rather than sync a truncated subset, bulk mode **fails +closed**: if the result set is larger than `batch_size` the poll errors. Raise +`batch_size` to cover the full table, or use incremental mode, for large tables. + +**Use Cases:** + +- Initial data load +- Periodic full snapshots (that fit within `batch_size`) +- Complex analytical queries +- Tables without tracking columns diff --git a/core/connectors/sources/jdbc_source/config.toml b/core/connectors/sources/jdbc_source/config.toml new file mode 100644 index 0000000000..8cb3f06441 --- /dev/null +++ b/core/connectors/sources/jdbc_source/config.toml @@ -0,0 +1,50 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "source" +key = "jdbc" +enabled = true +version = 0 +name = "JDBC source" +path = "../../target/release/libiggy_connector_jdbc_source" +verbose = false + +[[streams]] +stream = "user_events" +topic = "users" +schema = "json" +batch_length = 100 + +[plugin_config] +jdbc_url = "jdbc:postgresql://localhost:5432/database" +driver_class = "org.postgresql.Driver" +driver_jar_path = "/tmp/jdbc-drivers/postgresql-42.7.1.jar" +# Credential precedence: provide credentials EITHER via username + password here +# OR embedded in jdbc_url, not both. username and password must be both set or +# both unset (a half-set pair is rejected at startup). If both this pair and +# URL-embedded credentials are present, the driver decides which wins; avoid the +# ambiguity by using only one method. +username = "postgres" +password = "postgres" +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" +poll_interval = "1s" +batch_size = 1000 +tracking_column = "id" +initial_offset = "0" +mode = "incremental" +snake_case_columns = false +include_metadata = true diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs new file mode 100644 index 0000000000..ea19ab0916 --- /dev/null +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -0,0 +1,2843 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use async_trait::async_trait; +use chrono::{DateTime, Utc}; +use iggy_common::serde_secret; +use iggy_connector_sdk::{ + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source_connector, +}; +use jni::objects::{GlobalRef, JByteArray, JObject, JString, JThrowable, JValue}; +use jni::{JNIEnv, JavaVM}; +use regex::Regex; +use secrecy::{ExposeSecret, SecretString}; +use serde::{Deserialize, Serialize}; +use std::sync::{Arc, Mutex, MutexGuard}; +use std::time::{Duration, Instant}; +use tracing::{debug, info, warn}; +use uuid::Uuid; + +/// Clear any pending Java exception on the current thread. The JNI spec forbids +/// making most calls while an exception is pending; doing so aborts the whole +/// embedded JVM (the entire connectors-runtime process). Every fallible JNI +/// call in this connector clears on error before returning so a thrown Java +/// exception is never left pending for the next call on this thread. No-op when +/// nothing is pending. +fn clear_pending_exception(env: &mut JNIEnv) { + let _ = env.exception_clear(); +} + +/// Best-effort `close()` on a JDBC handle used in error/cleanup paths. Clears +/// any pending exception first (`close()` is a `CallVoidMethod`, which JNI +/// forbids while an exception is pending) and again afterwards in case the +/// close itself throws. +fn best_effort_close(env: &mut JNIEnv, handle: &JObject) { + clear_pending_exception(env); + let _ = env.call_method(handle, "close", "()V", &[]); + clear_pending_exception(env); +} + +/// Evaluate a fallible JNI call; on error clear any pending Java exception and +/// return `Error::Connection` with context. Keeps a thrown Java exception from +/// being left pending for the next JNI call on this thread. +macro_rules! jni { + ($env:expr, $call:expr, $ctx:expr) => { + match $call { + Ok(value) => value, + Err(err) => { + clear_pending_exception(&mut *$env); + return Err(Error::Connection(format!("{}: {err}", $ctx))); + } + } + }; +} + +/// Like [`jni!`] but returns `Error::InitError`, for the connection-setup path. +macro_rules! jni_init { + ($env:expr, $call:expr, $ctx:expr) => { + match $call { + Ok(value) => value, + Err(err) => { + clear_pending_exception(&mut *$env); + return Err(Error::InitError(format!("{}: {err}", $ctx))); + } + } + }; +} + +/// Lock a mutex, mapping a poisoned lock to a returned `Error` instead of +/// panicking. A thread that panicked while holding the lock then surfaces as a +/// logged, recoverable failure rather than a permanent panic loop on every +/// subsequent poll. +fn lock_mutex<'a, T>(mutex: &'a Mutex, what: &str) -> Result, Error> { + mutex + .lock() + .map_err(|_| Error::Connection(format!("{what} mutex poisoned"))) +} + +/// Cached compiled regex patterns for password sanitization +static RE_USER_PASS_AT: std::sync::LazyLock = + std::sync::LazyLock::new(|| Regex::new(r"://([^:]+):([^@?;/]+)@").unwrap()); +static RE_PASSWORD_PARAM: std::sync::LazyLock = + std::sync::LazyLock::new(|| Regex::new(r"(?i)(password|pwd|pass)=([^;&\s]+)").unwrap()); +static RE_ORACLE_PASS: std::sync::LazyLock = + std::sync::LazyLock::new(|| Regex::new(r"thin:([^/]+)/([^@]+)@").unwrap()); + +/// Matches the canonical incremental predicate so it can be removed on the first +/// (no-offset) poll. Case- and whitespace-tolerant so `where {tracking_column}>{last_offset}` +/// and `WHERE {tracking_column} > {last_offset}` are all recognized, rather +/// than only one exact literal spelling. +static RE_INCREMENTAL_PREDICATE: std::sync::LazyLock = std::sync::LazyLock::new(|| { + Regex::new(r"(?i)\bWHERE\s+\{tracking_column\}\s*>\s*\{last_offset\}").unwrap() +}); + +const CONNECTOR_NAME: &str = "JDBC source"; + +/// Source mode for the JDBC connector +#[derive(Debug, Clone, Deserialize, Serialize, PartialEq)] +#[serde(rename_all = "lowercase")] +pub enum Mode { + /// Re-run the query on every poll, capped at `batch_size` rows via JDBC + /// `setMaxRows`. There is no pagination beyond that cap, so rather than sync a + /// truncated subset, bulk mode fails closed: the poll probes with + /// `batch_size + 1` rows and errors if the result set is larger than + /// `batch_size`. Raise `batch_size` to cover the full table, or use + /// incremental mode, for tables larger than a batch. + Bulk, + /// Track the last offset and fetch only rows beyond it on each poll. + Incremental, +} + +/// Configuration for JDBC source connector +#[derive(Clone, Deserialize, Serialize)] +pub struct JdbcSourceConfig { + /// JDBC connection URL (e.g., "jdbc:mysql://localhost:3306/mydb") + /// Can include credentials: "jdbc:mysql://localhost:3306/mydb?user=root&password=secret" + #[serde(serialize_with = "serde_secret::serialize_secret")] + pub jdbc_url: SecretString, + + /// JDBC driver class name (e.g., "com.mysql.cj.jdbc.Driver") + pub driver_class: String, + + /// Path to JDBC driver JAR file + pub driver_jar_path: String, + + /// Database username (optional if included in jdbc_url) + #[serde(default)] + pub username: Option, + + /// Database password (optional if included in jdbc_url) + #[serde(default, serialize_with = "serde_secret::serialize_optional_secret")] + pub password: Option, + + /// SQL query to execute for fetching data + /// Can use {last_offset} placeholder for incremental reads + pub query: String, + + /// Polling interval (e.g., "30s", "5m", "1h") + #[serde(with = "humantime_serde")] + pub poll_interval: Duration, + + /// Batch size - maximum rows to fetch per poll + #[serde(default = "default_batch_size")] + pub batch_size: u32, + + /// Tracking column for incremental reads (e.g., "id", "updated_at") + #[serde(default)] + pub tracking_column: Option, + + /// Initial offset value for the first poll + #[serde(default)] + pub initial_offset: Option, + + /// Source mode: "bulk" (full table scan) or "incremental" (track last offset) + #[serde(default = "default_mode")] + pub mode: Mode, + + /// Convert column names to snake_case + #[serde(default)] + pub snake_case_columns: bool, + + /// Include metadata in output (table name, operation type, timestamp) + #[serde(default = "default_true")] + pub include_metadata: bool, + + /// JVM options (e.g., ["-Xmx512m", "-Xms128m"]) + #[serde(default)] + pub jvm_options: Vec, + + /// Timeout for the per-poll `Connection.isValid` liveness check (default: + /// 30000). JDBC expresses this timeout in whole seconds, so the value is + /// converted to seconds and clamped to the 1..=30s range; it does not govern + /// connection establishment. + #[serde(default = "default_connection_timeout")] + pub connection_timeout_ms: u64, +} + +fn default_connection_timeout() -> u64 { + 30000 +} + +fn default_batch_size() -> u32 { + 1000 +} + +fn default_mode() -> Mode { + Mode::Incremental +} + +fn default_true() -> bool { + true +} + +impl std::fmt::Debug for JdbcSourceConfig { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("JdbcSourceConfig") + .field( + "jdbc_url", + &sanitize_jdbc_url(self.jdbc_url.expose_secret()), + ) + .field("driver_class", &self.driver_class) + .field("driver_jar_path", &self.driver_jar_path) + .field("username", &self.username) + .field("password", &self.password.as_ref().map(|_| "***")) + .field("query", &self.query) + .field("poll_interval", &self.poll_interval) + .field("batch_size", &self.batch_size) + .field("tracking_column", &self.tracking_column) + .field("initial_offset", &self.initial_offset) + .field("mode", &self.mode) + .field("snake_case_columns", &self.snake_case_columns) + .field("include_metadata", &self.include_metadata) + .field("connection_timeout_ms", &self.connection_timeout_ms) + .finish() + } +} + +/// Internal state tracking for the JDBC source +#[derive(Debug, Clone, Serialize, Deserialize)] +struct State { + /// Last tracked offset value (for incremental mode) + last_offset: Option, + + /// Total rows processed + processed_rows: u64, + + /// Last poll timestamp + last_poll_time: DateTime, +} + +impl Default for State { + fn default() -> Self { + Self { + last_offset: None, + processed_rows: 0, + last_poll_time: Utc::now(), + } + } +} + +/// Database record structure for output messages +#[derive(Debug, Serialize, Deserialize)] +pub struct DatabaseRecord { + pub table_name: Option, + pub operation_type: String, + pub timestamp: DateTime, + pub data: serde_json::Value, +} + +/// JDBC Source Connector +#[derive(Debug)] +pub struct JdbcSource { + id: u32, + config: JdbcSourceConfig, + jvm: Option>, + // Behind a Mutex so `poll()` (&self) can transparently re-establish a dead + // direct connection without `&mut self`. + connection: Mutex>, + state: Arc>, + // Scheduled start of the next poll, used to pace polls at a fixed cadence + // that does not drift with per-poll work time. `None` until the first poll. + next_poll_at: Mutex>, +} + +/// Sanitize JDBC URL by masking passwords for logging +fn sanitize_jdbc_url(url: &str) -> String { + // Pattern 1: user:password@host format (MySQL, PostgreSQL) + let url = RE_USER_PASS_AT.replace_all(url, "://$1:***@"); + + // Pattern 2: password=value format (PostgreSQL, SQL Server, H2) + let url = RE_PASSWORD_PARAM.replace_all(&url, "$1=***"); + + // Pattern 3: Oracle user/password@host format + let url = RE_ORACLE_PASS.replace_all(&url, "thin:$1/***@"); + + url.to_string() +} + +impl JdbcSource { + /// Create a new JDBC source connector + pub fn new(id: u32, config: JdbcSourceConfig, connector_state: Option) -> Self { + // Restore state from persistent storage if available + let state = connector_state + .and_then(|cs| cs.deserialize::(CONNECTOR_NAME, id)) + .unwrap_or_else(|| { + let mut default_state = State::default(); + // Use initial_offset from config if provided + if let Some(ref initial_offset) = config.initial_offset { + default_state.last_offset = Some(initial_offset.clone()); + } + default_state + }); + + Self { + id, + config, + jvm: None, + connection: Mutex::new(None), + state: Arc::new(Mutex::new(state)), + next_poll_at: Mutex::new(None), + } + } + + /// Obtain the process-wide JVM, creating it on first use. JNI permits only a + /// single JVM per OS process, so this is shared across all JDBC connector + /// instances (see [`get_or_create_jvm`]). + fn initialize_jvm(&mut self) -> Result<(), Error> { + info!("Initializing JVM for JDBC source connector [{}]", self.id); + let jvm = get_or_create_jvm(&self.config.driver_jar_path, &self.config.jvm_options)?; + self.jvm = Some(jvm); + Ok(()) + } + + /// Load the JDBC driver and open the database connection. + fn create_connection(&mut self) -> Result<(), Error> { + let jvm = self + .jvm + .as_ref() + .ok_or_else(|| Error::InitError("JVM not initialized".to_string()))?; + + let mut env = jvm + .attach_current_thread() + .map_err(|e| Error::InitError(format!("Failed to attach thread to JVM: {}", e)))?; + + info!("Loading JDBC driver: {}", self.config.driver_class); + + // Load driver using Class.forName() which triggers static initialization + info!( + "Loading driver class via Class.forName: {}", + self.config.driver_class + ); + + let class_class = jni_init!( + env, + env.find_class("java/lang/Class"), + "Failed to find Class" + ); + + let driver_class_name = jni_init!( + env, + env.new_string(&self.config.driver_class), + "Failed to create class name string" + ); + + // Call Class.forName(className) to load and initialize the driver + jni_init!( + env, + env.call_static_method( + class_class, + "forName", + "(Ljava/lang/String;)Ljava/lang/Class;", + &[JValue::Object(&driver_class_name.into())], + ), + format!("Failed to load driver class '{}'", self.config.driver_class) + ); + + info!("JDBC driver loaded and registered successfully"); + + info!( + "Creating direct JDBC connection to: {}", + sanitize_jdbc_url(self.config.jdbc_url.expose_secret()) + ); + let conn = self.create_direct_connection_internal(&mut env)?; + *lock_mutex(&self.connection, "connection")? = Some(conn); + + Ok(()) + } + + /// Create a direct JDBC connection via DriverManager, inside its own JNI + /// local-reference frame so the transient class/loader/URL locals do not + /// accumulate on the caller's frame. The returned handle is a `GlobalRef`, so + /// it survives the frame pop. + fn create_direct_connection_internal(&self, env: &mut JNIEnv) -> Result { + env.push_local_frame(16) + .map_err(|e| Error::InitError(format!("Failed to push local frame: {e}")))?; + let result = self.create_direct_connection_inner(env); + // SAFETY: `result` holds only a global reference (or an error); no JNI + // local reference escapes the frame. + let _ = unsafe { env.pop_local_frame(&JObject::null()) }; + result + } + + fn create_direct_connection_inner(&self, env: &mut JNIEnv) -> Result { + // Set the thread context class loader to help DriverManager find the driver + let current_thread_class = jni_init!( + env, + env.find_class("java/lang/Thread"), + "Failed to find Thread class" + ); + + let current_thread = jni_init!( + env, + env.call_static_method( + current_thread_class, + "currentThread", + "()Ljava/lang/Thread;", + &[], + ) + .and_then(|v| v.l()), + "Failed to get current thread" + ); + + // Get the class loader that loaded the driver + let driver_class = jni_init!( + env, + env.find_class(self.config.driver_class.replace('.', "/")), + "Failed to find driver class" + ); + + let driver_class_loader = jni_init!( + env, + env.call_method( + &driver_class, + "getClassLoader", + "()Ljava/lang/ClassLoader;", + &[], + ) + .and_then(|v| v.l()), + "Failed to get driver class loader" + ); + + // Set the context class loader + jni_init!( + env, + env.call_method( + ¤t_thread, + "setContextClassLoader", + "(Ljava/lang/ClassLoader;)V", + &[JValue::Object(&driver_class_loader)], + ), + "Failed to set context class loader" + ); + + info!( + "Set thread context class loader for driver: {}", + self.config.driver_class + ); + + // Get connection from DriverManager + let driver_manager = jni_init!( + env, + env.find_class("java/sql/DriverManager"), + "Failed to find DriverManager" + ); + + let jdbc_url = jni_init!( + env, + env.new_string(self.config.jdbc_url.expose_secret()), + "Failed to create JDBC URL string" + ); + + // If username/password are provided separately, use 3-arg getConnection + let connection_obj = if let (Some(username), Some(password)) = + (&self.config.username, &self.config.password) + { + info!("Using separate username/password authentication"); + let username_jstring = jni_init!( + env, + env.new_string(username), + "Failed to create username string" + ); + let password_jstring = jni_init!( + env, + env.new_string(password.expose_secret()), + "Failed to create password string" + ); + + jni_init!( + env, + env.call_static_method( + driver_manager, + "getConnection", + "(Ljava/lang/String;Ljava/lang/String;Ljava/lang/String;)Ljava/sql/Connection;", + &[ + JValue::Object(&jdbc_url.into()), + JValue::Object(&username_jstring.into()), + JValue::Object(&password_jstring.into()), + ], + ) + .and_then(|v| v.l()), + "Failed to create JDBC connection with credentials" + ) + } else { + info!("Using connection string with embedded credentials"); + jni_init!( + env, + env.call_static_method( + driver_manager, + "getConnection", + "(Ljava/lang/String;)Ljava/sql/Connection;", + &[JValue::Object(&jdbc_url.into())], + ) + .and_then(|v| v.l()), + "Failed to create JDBC connection from URL" + ) + }; + + let global_ref = env + .new_global_ref(connection_obj) + .map_err(|e| Error::InitError(format!("Failed to create global reference: {e}")))?; + + info!("Direct database connection established successfully"); + Ok(global_ref) + } + + /// Acquire the direct connection, transparently re-establishing it if it has + /// dropped since the last poll. + fn get_connection<'local>(&self, env: &mut JNIEnv<'local>) -> Result, Error> { + // Hold the connection lock across the whole check/close/create/store + // sequence so correctness does not depend on there being a single caller: + // a concurrent caller can no longer observe or replace the handle + // mid-reconnect. The lock is a std Mutex held only across synchronous JNI + // work (no .await), and neither `connection_is_valid` nor + // `create_direct_connection_internal` re-locks it, so this cannot deadlock. + let mut guard = lock_mutex(&self.connection, "connection")?; + + let needs_reconnect = match guard.as_ref() { + Some(conn) => !self.connection_is_valid(env, conn.as_obj()), + None => true, + }; + + if needs_reconnect { + info!("Direct JDBC connection is not valid; re-establishing"); + // Best-effort close of the old handle, then drop it before creating + // the replacement so a failed reconnect leaves no stale reference. + if let Some(old) = guard.as_ref() { + best_effort_close(env, old.as_obj()); + } + *guard = None; + *guard = Some(self.create_direct_connection_internal(env)?); + } + + let conn = guard + .as_ref() + .ok_or_else(|| Error::Connection("No connection available".to_string()))?; + let local_ref = env + .new_local_ref(conn.as_obj()) + .map_err(|e| Error::Connection(format!("Failed to create local ref: {e}")))?; + Ok(local_ref) + } + + /// Best-effort `Connection.isValid(timeout)` check. Returns false on any + /// JNI error so the caller re-establishes the connection. + fn connection_is_valid(&self, env: &mut JNIEnv, conn: &JObject) -> bool { + let timeout_secs = (self.config.connection_timeout_ms / 1000).clamp(1, 30) as i32; + match env + .call_method(conn, "isValid", "(I)Z", &[JValue::Int(timeout_secs)]) + .and_then(|v| v.z()) + { + Ok(valid) => valid, + Err(_) => { + // isValid may throw; clear so the reconnect path's next JNI call + // is not made with an exception pending. + clear_pending_exception(env); + false + } + } + } + + /// Execute query and fetch results. + /// + /// The mutex is held only briefly: once to read the current offset for + /// query building, and once after the JNI work to write the updated state. + fn execute_query(&self, env: &mut JNIEnv) -> Result, Error> { + let connection = self.get_connection(env)?; + + // Read current state snapshot (short lock) + let query = { + let state = lock_mutex(&self.state, "state")?; + self.build_query(&state) + }?; + // Logged at debug: the built query embeds the substituted offset value. + debug!("Executing query: {}", query); + + let (messages, row_count, max_offset) = + self.execute_statement_and_fetch_rows(env, &connection, &query)?; + + // Fail closed on bulk truncation. The bulk fetch probes with batch_size+1 + // rows, so seeing more than batch_size means the result set is larger than + // one batch and bulk mode (which has no pagination) would otherwise sync + // only an arbitrary subset. Erroring surfaces the misconfiguration instead + // of silently dropping rows. Full cross-database OFFSET pagination is a + // separate follow-up. + if self.config.mode == Mode::Bulk && row_count > self.config.batch_size as u64 { + return Err(Error::InvalidConfigValue(format!( + "bulk query returned more than batch_size ({}) rows; bulk mode does not paginate \ + and would sync only a truncated subset. Increase batch_size to cover the full \ + result set, or use incremental mode with an ordered tracking_column.", + self.config.batch_size + ))); + } + + // Update state with results (short lock) + { + let mut state = lock_mutex(&self.state, "state")?; + if let Some(offset) = max_offset { + state.last_offset = Some(offset); + } + state.processed_rows += row_count; + state.last_poll_time = Utc::now(); + info!( + "Fetched {} rows, total processed: {}", + row_count, state.processed_rows + ); + } + + Ok(messages) + } + + /// Prepare a JDBC statement, execute it, and read all result rows into messages. + fn execute_statement_and_fetch_rows( + &self, + env: &mut JNIEnv, + connection: &JObject, + query: &str, + ) -> Result<(Vec, u64, Option), Error> { + let query_jstring = jni!(env, env.new_string(query), "Failed to create query string"); + + let statement = match env + .call_method( + connection, + "prepareStatement", + "(Ljava/lang/String;)Ljava/sql/PreparedStatement;", + &[JValue::Object(&query_jstring.into())], + ) + .and_then(|v| v.l()) + { + Ok(s) => s, + Err(_) => return Err(classify_query_failure(env, "prepare statement")), + }; + + // Use setMaxRows for database-agnostic row limiting instead of SQL LIMIT clause. + // This works across all JDBC drivers (MySQL, Oracle, SQL Server, H2, etc.). + // In bulk mode fetch one extra row so the caller can detect an oversized + // (truncated) result set and fail closed rather than sync an arbitrary + // subset; incremental mode pages via the offset, so batch_size is the cap. + let max_rows = match self.config.mode { + Mode::Bulk => self.config.batch_size.saturating_add(1), + Mode::Incremental => self.config.batch_size, + }; + if let Err(err) = env.call_method( + &statement, + "setMaxRows", + "(I)V", + &[JValue::Int(max_rows.min(i32::MAX as u32) as i32)], + ) { + best_effort_close(env, &statement); + return Err(Error::Connection(format!("Failed to set max rows: {err}"))); + } + + let result_set = match env + .call_method(&statement, "executeQuery", "()Ljava/sql/ResultSet;", &[]) + .and_then(|v| v.l()) + { + Ok(rs) => rs, + Err(_) => { + // Read and classify the pending SQLException FIRST: this clears + // it, which then makes the statement close() safe (close() is a + // JNI call and must not run with an exception pending). + let error = classify_query_failure(env, "execute query"); + best_effort_close(env, &statement); + return Err(error); + } + }; + + // On any read error the callees clear the pending exception; close the + // statement here before propagating so it is not leaked. + let columns = match self.read_column_metadata(env, &result_set) { + Ok(columns) => columns, + Err(err) => { + best_effort_close(env, &statement); + return Err(err); + } + }; + let (messages, row_count, max_offset) = match self.read_rows(env, &result_set, &columns) { + Ok(result) => result, + Err(err) => { + best_effort_close(env, &statement); + return Err(err); + } + }; + + // Close statement (best-effort) + best_effort_close(env, &statement); + + Ok((messages, row_count, max_offset)) + } + + /// Read column names and types from the ResultSet metadata. + fn read_column_metadata( + &self, + env: &mut JNIEnv, + result_set: &JObject, + ) -> Result, Error> { + // Read all column metadata inside its own JNI local frame so the + // metadata object and per-column name references are reclaimed; a very + // wide table would otherwise accumulate one local ref per column on the + // outer frame for the whole poll. + env.push_local_frame(16) + .map_err(|e| Error::Connection(format!("Failed to push local frame: {}", e)))?; + let result = self.read_column_metadata_inner(env, result_set); + // SAFETY: the returned Vec is owned Rust data; no JNI reference escapes. + let _ = unsafe { env.pop_local_frame(&JObject::null()) }; + result + } + + fn read_column_metadata_inner( + &self, + env: &mut JNIEnv, + result_set: &JObject, + ) -> Result, Error> { + let metadata = jni!( + env, + env.call_method( + result_set, + "getMetaData", + "()Ljava/sql/ResultSetMetaData;", + &[], + ) + .and_then(|v| v.l()), + "Failed to get metadata" + ); + + let column_count = jni!( + env, + env.call_method(&metadata, "getColumnCount", "()I", &[]) + .and_then(|v| v.i()), + "Failed to get column count" + ); + + info!("Query returned {} columns", column_count); + + // Clamp a driver-supplied count before using it as an allocation size: a + // negative i32 would sign-extend to an enormous usize and abort on alloc. + let mut columns = Vec::with_capacity((column_count.max(0) as usize).min(8192)); + for i in 1..=column_count { + let col_name = self.get_column_name(env, &metadata, i)?; + let col_type = self.get_column_type(env, &metadata, i)?; + columns.push((col_name, col_type)); + } + + Ok(columns) + } + + /// Iterate over result set rows and convert each to a ProducedMessage. + fn read_rows( + &self, + env: &mut JNIEnv, + result_set: &JObject, + columns: &[(String, i32)], + ) -> Result<(Vec, u64, Option), Error> { + // setMaxRows caps the result set at batch_size, so that is the known + // upper bound; clamp the pre-allocation so an extreme batch_size cannot + // request an absurd allocation up front. + let mut messages = Vec::with_capacity((self.config.batch_size as usize).min(8192)); + let mut row_count: u64 = 0; + let mut max_offset: Option = None; + + loop { + let has_next = jni!( + env, + env.call_method(result_set, "next", "()Z", &[]) + .and_then(|v| v.z()), + "Failed to fetch next row" + ); + + if !has_next { + break; + } + + // Read each row inside its own JNI local-reference frame so the per + // -column local refs (getObject/getString/getBytes results) are + // reclaimed every iteration; otherwise a large result set would + // overflow the JNI local reference table and abort the JVM. + env.push_local_frame(32) + .map_err(|e| Error::Connection(format!("Failed to push local frame: {}", e)))?; + let row_result = self.read_single_row(env, result_set, columns); + // SAFETY: `read_single_row` returns only owned Rust data (a JSON map + // and an optional String); no JNI local reference escapes the frame. + let _ = unsafe { env.pop_local_frame(&JObject::null()) }; + let (row_data, offset) = row_result?; + + // Track the maximum tracking value across the batch rather than + // assuming the last row is the largest, so incremental mode is + // correct even if the query is not ordered by the tracking column. + if let Some(offset) = offset { + max_offset = Some(match max_offset { + Some(current) => larger_offset(current, offset), + None => offset, + }); + } + + let message = self.build_message(row_data)?; + messages.push(message); + row_count += 1; + } + + Ok((messages, row_count, max_offset)) + } + + /// Extract data from a single result set row, returning the row map and optional offset. + fn read_single_row( + &self, + env: &mut JNIEnv, + result_set: &JObject, + columns: &[(String, i32)], + ) -> Result<(serde_json::Map, Option), Error> { + let mut row_data = serde_json::Map::new(); + let mut offset = None; + + for (idx, (col_name, col_type)) in columns.iter().enumerate() { + let col_idx = (idx + 1) as i32; + let value = self.extract_column_value(env, result_set, col_idx, col_type)?; + + let final_col_name = if self.config.snake_case_columns { + to_snake_case(col_name) + } else { + col_name.clone() + }; + + // A snake_case conversion (or a query with duplicate labels) can map + // two source columns onto the same key; the later value silently wins. + // Warn so the loss is diagnosable rather than invisible. + if row_data.contains_key(&final_col_name) { + warn!( + "Column '{col_name}' maps to key '{final_col_name}', which already exists in the row; the earlier value is overwritten" + ); + } + + row_data.insert(final_col_name.clone(), value); + + // Track offset from the first column matching the tracking column + // (by driver label or normalized key, case-insensitively). Only the + // first match counts: a collapsed/duplicate label must not let a later + // column silently overwrite the offset with an unrelated value. + if offset.is_none() + && let Some(ref tracking_col) = self.config.tracking_column + && tracking_column_matches(tracking_col, col_name, &final_col_name) + { + let value = self.extract_offset_value(&row_data, &final_col_name); + offset = self.tracking_offset_or_error(value, tracking_col)?; + } + } + + Ok((row_data, offset)) + } + + /// Resolve a tracking-column value into an offset. In incremental mode a NULL + /// or empty value is a hard error: the row is emitted but the offset cannot + /// advance past NULL, so it would be re-read (and re-emitted) every poll. + fn tracking_offset_or_error( + &self, + value: Option, + tracking_col: &str, + ) -> Result, Error> { + match value { + Some(value) => Ok(Some(value)), + None if self.config.mode == Mode::Incremental => { + Err(Error::InvalidRecordValue(format!( + "tracking column '{tracking_col}' is NULL or empty in a returned row; incremental \ + mode cannot advance its offset past NULL values. Exclude them in the query, e.g. \ + add `AND {tracking_col} IS NOT NULL`." + ))) + } + None => Ok(None), + } + } + + /// Build a ProducedMessage from row data, optionally wrapping in DatabaseRecord metadata. + fn build_message( + &self, + row_data: serde_json::Map, + ) -> Result { + let now = Utc::now(); + let payload = if self.config.include_metadata { + let record = DatabaseRecord { + table_name: None, + operation_type: "SELECT".to_string(), + timestamp: now, + data: serde_json::Value::Object(row_data), + }; + serde_json::to_vec(&record) + .map_err(|e| Error::Serialization(format!("Failed to serialize record: {e}")))? + } else { + serde_json::to_vec(&serde_json::Value::Object(row_data)) + .map_err(|e| Error::Serialization(format!("Failed to serialize row data: {e}")))? + }; + + let now_ms = now.timestamp_millis() as u64; + Ok(ProducedMessage { + id: Some(Uuid::new_v4().as_u128()), + payload, + headers: None, + checksum: None, + timestamp: Some(now_ms), + origin_timestamp: Some(now_ms), + }) + } + + /// Build the query for this poll by substituting the `{tracking_column}` and + /// `{last_offset}` placeholders. Row limiting is handled via JDBC setMaxRows + /// rather than SQL LIMIT to ensure cross-database compatibility. + fn build_query(&self, state: &State) -> Result { + let mut query = self.config.query.clone(); + + if self.config.mode != Mode::Incremental { + return finalize_query(query); + } + + let offset = state + .last_offset + .as_deref() + .or(self.config.initial_offset.as_deref()); + + // Without an offset yet, drop the incremental predicate but keep the rest + // of the query (e.g. an ORDER BY) intact. + if offset.is_none() { + query = RE_INCREMENTAL_PREDICATE.replace(&query, "").into_owned(); + } + + // Substitute {tracking_column} wherever it still appears (a WHERE and/or + // an ORDER BY), validating it as a plain identifier to avoid injection. + if query.contains("{tracking_column}") { + let column = self.config.tracking_column.as_deref().ok_or_else(|| { + Error::InvalidConfigValue( + "query uses {tracking_column} but tracking_column is not set".to_string(), + ) + })?; + if !is_valid_identifier(column) { + return Err(Error::InvalidConfigValue(format!( + "tracking_column '{column}' is not a valid SQL identifier" + ))); + } + query = query.replace("{tracking_column}", column); + } + + // Substitute the offset value (quoted and escaped) when we have one. + if let Some(offset) = offset { + query = query.replace("{last_offset}", "e_sql_literal(offset)); + } + + finalize_query(query) + } + + /// Get column name from ResultSetMetaData + fn get_column_name( + &self, + env: &mut JNIEnv, + metadata: &JObject, + column_index: i32, + ) -> Result { + let col_name_obj = jni!( + env, + env.call_method( + metadata, + "getColumnName", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get column name" + ); + + let col_name: String = jni!( + env, + env.get_string(&JString::from(col_name_obj)), + "Failed to convert column name" + ) + .into(); + + Ok(col_name) + } + + /// Get column type from ResultSetMetaData + fn get_column_type( + &self, + env: &mut JNIEnv, + metadata: &JObject, + column_index: i32, + ) -> Result { + let col_type = jni!( + env, + env.call_method( + metadata, + "getColumnType", + "(I)I", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.i()), + "Failed to get column type" + ); + + Ok(col_type) + } + + /// Extract column value based on JDBC type + fn extract_column_value( + &self, + env: &mut JNIEnv, + result_set: &JObject, + column_index: i32, + sql_type: &i32, + ) -> Result { + use java::sql::Types; + + // Check if null first + let obj = jni!( + env, + env.call_method( + result_set, + "getObject", + "(I)Ljava/lang/Object;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get object" + ); + + if obj.is_null() { + return Ok(serde_json::Value::Null); + } + + match *sql_type { + Types::BIT | Types::BOOLEAN => { + let value = jni!( + env, + env.call_method( + result_set, + "getBoolean", + "(I)Z", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.z()), + "Failed to get boolean" + ); + Ok(serde_json::Value::Bool(value)) + } + Types::TINYINT | Types::SMALLINT | Types::INTEGER => { + let value = jni!( + env, + env.call_method(result_set, "getInt", "(I)I", &[JValue::Int(column_index)]) + .and_then(|v| v.i()), + "Failed to get int" + ); + Ok(serde_json::json!(value)) + } + Types::BIGINT => { + let value = jni!( + env, + env.call_method(result_set, "getLong", "(I)J", &[JValue::Int(column_index)]) + .and_then(|v| v.j()), + "Failed to get long" + ); + Ok(serde_json::json!(value)) + } + Types::FLOAT | Types::REAL => { + let value = jni!( + env, + env.call_method(result_set, "getFloat", "(I)F", &[JValue::Int(column_index)]) + .and_then(|v| v.f()), + "Failed to get float" + ); + Ok(serde_json::json!(value)) + } + Types::DOUBLE => { + let value = jni!( + env, + env.call_method( + result_set, + "getDouble", + "(I)D", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.d()), + "Failed to get double" + ); + Ok(serde_json::json!(value)) + } + // NUMERIC/DECIMAL can carry more precision than an f64 can represent + // (e.g. money/large decimals), so emit them as strings to avoid + // silent precision loss. + Types::NUMERIC | Types::DECIMAL => { + self.get_column_as_string(env, result_set, column_index) + } + // Binary columns are base64-encoded so arbitrary bytes survive the + // round-trip through JSON. + Types::BINARY | Types::VARBINARY | Types::LONGVARBINARY => { + let bytes_obj = jni!( + env, + env.call_method( + result_set, + "getBytes", + "(I)[B", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.l()), + "Failed to get bytes" + ); + if bytes_obj.is_null() { + return Ok(serde_json::Value::Null); + } + let buf = jni!( + env, + env.convert_byte_array(JByteArray::from(bytes_obj)), + "Failed to convert bytes" + ); + use base64::Engine; + Ok(serde_json::Value::String( + base64::engine::general_purpose::STANDARD.encode(&buf), + )) + } + Types::TIMESTAMP | Types::DATE | Types::TIME => { + let value = jni!( + env, + env.call_method( + result_set, + "getString", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get timestamp" + ); + let str_value: String = jni!( + env, + env.get_string(&JString::from(value)), + "Failed to convert timestamp" + ) + .into(); + Ok(serde_json::Value::String(str_value)) + } + // Default: getString for all other types (CHAR, VARCHAR, etc.) + _ => self.get_column_as_string(env, result_set, column_index), + } + } + + /// Read a column via `ResultSet.getString`, returning JSON `null` when the + /// value is SQL NULL. + fn get_column_as_string( + &self, + env: &mut JNIEnv, + result_set: &JObject, + column_index: i32, + ) -> Result { + let value = jni!( + env, + env.call_method( + result_set, + "getString", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get string" + ); + + if value.is_null() { + Ok(serde_json::Value::Null) + } else { + let str_value: String = jni!( + env, + env.get_string(&JString::from(value)), + "Failed to convert string" + ) + .into(); + Ok(serde_json::Value::String(str_value)) + } + } + + /// Extract the tracking-column value as a string offset, or `None` when it is + /// SQL NULL / empty / not comparable. In incremental mode a `None` here is + /// turned into a hard error by [`Self::tracking_offset_or_error`] (a NULL + /// tracking value cannot be watermarked), so the tracking column must be + /// NOT NULL; see the README. + fn extract_offset_value( + &self, + row_data: &serde_json::Map, + col_name: &str, + ) -> Option { + match row_data.get(col_name) { + Some(serde_json::Value::Number(n)) => Some(n.to_string()), + Some(serde_json::Value::String(s)) if !s.is_empty() => Some(s.clone()), + _ => None, + } + } + + /// Validate configuration before touching the JVM or the database, so bad + /// config surfaces immediately at `open()` with an actionable message rather + /// than as an opaque wrapped JVM error or only after the first poll sleep. + fn validate_config(&self) -> Result<(), Error> { + // The driver JAR must exist; a missing path otherwise surfaces as an + // opaque ClassNotFound wrapped deep inside JVM startup. + if !std::path::Path::new(&self.config.driver_jar_path).exists() { + return Err(Error::InvalidConfigValue(format!( + "driver_jar_path '{}' does not exist; set it to the path of the JDBC driver JAR", + self.config.driver_jar_path + ))); + } + + // batch_size drives JDBC setMaxRows. Zero means "no limit" to JDBC, which + // would defeat both the row cap and the bulk truncation probe, so require + // at least 1. Cap below i32::MAX so the bulk `batch_size + 1` probe still + // fits in the i32 setMaxRows takes and stays distinguishable from a full + // batch. + const MAX_BATCH_SIZE: u32 = i32::MAX as u32 - 1; + if self.config.batch_size == 0 || self.config.batch_size > MAX_BATCH_SIZE { + return Err(Error::InvalidConfigValue(format!( + "batch_size must be between 1 and {MAX_BATCH_SIZE}, got {}", + self.config.batch_size + ))); + } + + // Separate-credential auth requires both username and password; a + // half-set pair would silently fall through to URL-embedded credentials. + if self.config.username.is_some() != self.config.password.is_some() { + return Err(Error::InvalidConfigValue( + "username and password must both be set (for separate authentication) or both be \ + unset (to use credentials embedded in the JDBC URL)" + .to_string(), + )); + } + + // Incremental mode invariants. Without a tracking column the offset can + // never advance, so every poll re-reads the same rows. Without ordering + // by that column, setMaxRows returns an arbitrary subset, and advancing + // the offset to its max permanently skips the unread lower keys. + if self.config.mode == Mode::Incremental { + let Some(tracking_column) = self.config.tracking_column.as_deref() else { + return Err(Error::InvalidConfigValue( + "incremental mode requires tracking_column so the offset can advance; set \ + tracking_column, or use mode = \"bulk\"" + .to_string(), + )); + }; + if !query_orders_by_tracking_column(&self.config.query, tracking_column) { + return Err(Error::InvalidConfigValue(format!( + "incremental mode requires the query to order by the tracking column so each \ + batch is a contiguous ascending range; add `ORDER BY {tracking_column}` (or \ + `ORDER BY {{tracking_column}}`) to the query" + ))); + } + } + + // Dry-run the query build so an unresolved placeholder or an invalid + // tracking_column fails now instead of after the first poll interval. + let state = lock_mutex(&self.state, "state")?; + self.build_query(&state)?; + Ok(()) + } +} + +#[async_trait] +impl Source for JdbcSource { + async fn open(&mut self) -> Result<(), Error> { + info!("Opening JDBC source connector [{}]", self.id); + info!( + "Configuration: JDBC URL={}, Driver={}, Mode={:?}", + sanitize_jdbc_url(self.config.jdbc_url.expose_secret()), + self.config.driver_class, + self.config.mode + ); + + // Fail fast on bad config before starting the JVM or opening a connection. + self.validate_config()?; + + // Initialize JVM + self.initialize_jvm()?; + + // Create database connection + self.create_connection()?; + + info!("JDBC source connector [{}] opened successfully", self.id); + Ok(()) + } + + async fn poll(&self) -> Result { + // Pace polls on a fixed cadence measured from a scheduled start instant, + // so per-poll work time does not accumulate as drift and the first poll + // is not delayed by a full interval. The schedule is clamped forward to + // `now` whenever it has fallen behind (a long pause, a poll that overran + // the interval, or the runtime re-polling immediately after an error), + // so a lagging schedule can never collapse the sleep into a busy loop + // that hammers the database. + let scheduled = { + let mut next = lock_mutex(&self.next_poll_at, "next_poll_at")?; + let scheduled = next.map_or_else(Instant::now, |planned| planned.max(Instant::now())); + *next = Some(scheduled + self.config.poll_interval); + scheduled + }; + let now = Instant::now(); + if scheduled > now { + tokio::time::sleep(scheduled - now).await; + } + + // The JDBC/JNI fetch is synchronous, blocking work; run it via + // block_in_place so it does not monopolize a shared async-runtime worker + // while other connectors need to make progress. The connectors runtime is + // multi-threaded, which block_in_place requires. + let messages = tokio::task::block_in_place(|| -> Result, Error> { + let jvm = self + .jvm + .as_ref() + .ok_or_else(|| Error::InitError("JVM not initialized".to_string()))?; + let mut env = jvm + .attach_current_thread() + .map_err(|e| Error::InitError(format!("Failed to attach thread: {e}")))?; + // Defensive: clear any exception left pending by a prior failed poll + // on this thread before issuing JNI calls. + clear_pending_exception(&mut env); + // Bound this poll's local references (the connection local ref, the + // query string, and the statement/result-set handles) to a frame + // reclaimed when the poll returns. A tokio worker thread stays + // attached to the JVM across polls (attach_current_thread returns a + // no-detach nested guard once attached), so without this frame those + // per-poll locals accumulate on the thread's top-level frame and + // eventually overflow the JNI local reference table, aborting the JVM. + env.push_local_frame(16) + .map_err(|e| Error::Connection(format!("Failed to push local frame: {e}")))?; + let result = self.execute_query(&mut env); + // SAFETY: execute_query returns only owned Rust data (messages); no + // JNI local reference escapes the frame. + let _ = unsafe { env.pop_local_frame(&JObject::null()) }; + result + })?; + + // Persist state so offsets survive connector restarts + let connector_state = { + let state = lock_mutex(&self.state, "state")?; + ConnectorState::serialize(&*state, CONNECTOR_NAME, self.id) + }; + + Ok(ProducedMessages { + schema: Schema::Json, + messages, + state: connector_state, + }) + } + + async fn close(&mut self) -> Result<(), Error> { + info!("Closing JDBC source connector [{}]", self.id); + + if self.jvm.is_some() { + // Closing the JDBC connection is blocking JNI work; run it off the + // async worker like the poll path. + tokio::task::block_in_place(|| -> Result<(), Error> { + let Some(jvm) = self.jvm.as_ref() else { + return Ok(()); + }; + let Ok(mut env) = jvm.attach_current_thread() else { + return Ok(()); + }; + let mut guard = lock_mutex(&self.connection, "connection")?; + if let Some(connection) = guard.as_ref() { + best_effort_close(&mut env, connection.as_obj()); + info!("Database connection closed"); + } + *guard = None; + Ok(()) + })?; + } + + let state = lock_mutex(&self.state, "state")?; + info!( + "JDBC source connector [{}] closed. Total rows processed: {}", + self.id, state.processed_rows + ); + + Ok(()) + } +} + +/// Convert string to snake_case +fn to_snake_case(s: &str) -> String { + let mut result = String::new(); + let mut prev_is_upper = false; + + for (i, ch) in s.chars().enumerate() { + if ch.is_uppercase() { + if i > 0 && !prev_is_upper { + result.push('_'); + } + result.extend(ch.to_lowercase()); + prev_is_upper = true; + } else { + result.push(ch); + prev_is_upper = false; + } + } + + result +} + +/// Process-wide JVM. JNI allows only one `JavaVM` per OS process, so every JDBC +/// connector instance in this dynamic library shares this one. +static GLOBAL_JVM: Mutex>> = Mutex::new(None); + +/// Return the process JVM, creating it on first use within this dynamic +/// library. The first caller's `jvm_options`/classpath win; later callers (e.g. +/// a second JDBC connector of the same type) reuse the existing VM instead of +/// failing with `JNI_EEXIST`. +/// +/// Limitation: a JDBC *source* and a JDBC *sink* are separate dynamic libraries +/// and do not share this static, so configuring both in the *same* connectors +/// runtime process is not supported (the second to start cannot create a second +/// JVM). Run them in separate runtime processes. +fn get_or_create_jvm(driver_jar_path: &str, jvm_options: &[String]) -> Result, Error> { + let mut guard = lock_mutex(&GLOBAL_JVM, "jvm")?; + if let Some(jvm) = guard.as_ref() { + info!("Reusing existing process JVM"); + return Ok(jvm.clone()); + } + + let classpath_option = format!("-Djava.class.path={driver_jar_path}"); + let mut args_builder = jni::InitArgsBuilder::new() + .version(jni::JNIVersion::V8) + .option(&classpath_option); + for option in jvm_options { + args_builder = args_builder.option(option); + } + let jvm_args = args_builder + .build() + .map_err(|e| Error::InitError(format!("Failed to build JVM arguments: {e:?}")))?; + let jvm = JavaVM::new(jvm_args) + .map_err(|e| Error::InitError(format!("Failed to create JVM: {e:?}")))?; + + info!("JVM initialized successfully (classpath: {driver_jar_path})"); + let arc = Arc::new(jvm); + *guard = Some(arc.clone()); + Ok(arc) +} + +/// Quote a value as a SQL string literal for substituting the incremental +/// `{last_offset}` value, which originates from a (DB-controlled) tracking-column +/// value. Backslashes are escaped before single quotes are doubled: doubling +/// quotes alone is insufficient under MySQL's default `sql_mode`, where a +/// backslash is an escape character and a trailing `\` could otherwise consume +/// the closing quote and break out of the literal. Binding the offset as a +/// `PreparedStatement` parameter is the DB-agnostic long-term fix (tracked as a +/// follow-up); this keeps the string-substitution path safe in the meantime. +fn quote_sql_literal(value: &str) -> String { + let escaped = value.replace('\\', "\\\\").replace('\'', "''"); + format!("'{escaped}'") +} + +/// Reject a query that still contains an unresolved placeholder, so an invalid +/// statement is never sent to the driver. This guards misconfigurations such as +/// a bulk-mode query using `{last_offset}`, or an incremental query whose +/// predicate could not be auto-removed on the first (no-offset) poll. +fn finalize_query(query: String) -> Result { + if query.contains("{tracking_column}") || query.contains("{last_offset}") { + return Err(Error::InvalidConfigValue( + "query still contains an unresolved {tracking_column}/{last_offset} placeholder; \ + placeholders are only resolved in incremental mode, and require either a persisted \ + offset, an initial_offset, or the exact 'WHERE {tracking_column} > {last_offset}' form" + .to_string(), + )); + } + Ok(query) +} + +/// Return the larger of two offset values. Integer keys are compared as `i128` +/// first (exact across the full `BIGINT` range, avoiding the f64 precision loss +/// past 2^53 that would misorder large IDs and contradict the NUMERIC/DECIMAL +/// -as-string rationale elsewhere in this file); genuinely fractional values +/// fall back to f64; non-numeric values (ISO-8601 timestamps, text keys) compare +/// lexicographically. +/// +/// This assumes the tracking column is homogeneously typed and monotonic under +/// the database's own ordering. A column that mixes numeric-looking and +/// non-numeric strings, or a timestamp whose driver string form is not +/// lexicographically ordered, can make this Rust-side max disagree with the +/// database's `>` comparison and cause rows to be skipped or re-read. See the +/// tracking-column guidance in the README. +fn larger_offset(a: String, b: String) -> String { + let b_is_larger = match (a.parse::(), b.parse::()) { + (Ok(na), Ok(nb)) => nb > na, + _ => match (a.parse::(), b.parse::()) { + (Ok(na), Ok(nb)) => nb > na, + _ => b > a, + }, + }; + if b_is_larger { b } else { a } +} + +/// Check that an incremental query's result set is ordered ascending by the +/// tracking column, so `setMaxRows` returns a contiguous ascending prefix rather +/// than an arbitrary subset (which would let the advancing offset skip unread +/// lower keys). The tracking column must be the FIRST ordering term of the outer +/// `ORDER BY`, and must not be descending. +/// +/// This is a lexical check, not a SQL parser, so it errs strict: it inspects the +/// last `ORDER BY` in the text (the outer query's, not a subquery's), takes the +/// first ordering term, and requires it to be the `{tracking_column}` placeholder +/// or an identifier whose final path segment equals the tracking column +/// (case-insensitive). A trailing `DESC` is rejected, and a composite +/// `ORDER BY other, tracking` (tracking not primary) is rejected, because +/// truncation then yields a prefix ordered by `other`. +fn query_orders_by_tracking_column(query: &str, tracking_column: &str) -> bool { + let lower = query.to_lowercase(); + let Some(pos) = lower.rfind("order by") else { + return false; + }; + let after = lower[pos + "order by".len()..].trim_start(); + let first_term = after.split(',').next().unwrap_or("").trim(); + let mut tokens = first_term.split_whitespace(); + let key = tokens.next().unwrap_or(""); + // Any explicit descending direction breaks ascending offset advancement. + if tokens.any(|token| token == "desc") { + return false; + } + if key.starts_with("{tracking_column}") { + return true; + } + // Compare the final path segment (`t.updated_at` -> `updated_at`), keeping + // only leading identifier characters so trailing punctuation is ignored. + let key_ident: String = key + .rsplit('.') + .next() + .unwrap_or(key) + .chars() + .take_while(|c| c.is_ascii_alphanumeric() || *c == '_') + .collect(); + let tracking_lower = tracking_column.to_lowercase(); + let tracking_ident = tracking_lower.rsplit('.').next().unwrap_or(&tracking_lower); + !key_ident.is_empty() && key_ident == tracking_ident +} + +/// Whether a configured `tracking_column` name refers to this result column. +/// Compares case-insensitively against both the raw driver label and the +/// normalized (snake_cased) output key: drivers fold identifier case (e.g. +/// PostgreSQL lowercases unquoted names) and snake_case output would otherwise +/// never match the configured name, silently stalling the offset. +fn tracking_column_matches(tracking_column: &str, raw_name: &str, normalized_name: &str) -> bool { + tracking_column.eq_ignore_ascii_case(raw_name) + || tracking_column.eq_ignore_ascii_case(normalized_name) +} + +/// Validate that a string is a safe SQL identifier for interpolation: ASCII +/// letters, digits, underscore, and dot (for `table.column`), starting with a +/// letter or underscore. Prevents injection through the `{tracking_column}` +/// placeholder. +fn is_valid_identifier(name: &str) -> bool { + let mut chars = name.chars(); + match chars.next() { + Some(c) if c.is_ascii_alphabetic() || c == '_' => {} + _ => return false, + } + name.chars() + .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '.') +} + +/// Classify a JDBC `SQLState` class (first 2 chars) as transient vs permanent. +/// `08` connection, `40` rollback/serialization, `53` resources, `57` operator +/// intervention, `58` system error are transient; everything else (and an +/// unknown/absent state) is permanent. +fn is_transient_sql_state(sql_state: Option<&str>) -> bool { + // Use `get` rather than slicing: the SQLState comes from the driver and is + // not guaranteed ASCII, so `&s[..2]` could panic on a multi-byte boundary. + match sql_state.and_then(|s| s.get(..2)) { + Some(class) => matches!(class, "08" | "40" | "53" | "57" | "58"), + None => false, + } +} + +/// Inspect and CLEAR the pending Java exception after a failed query JNI call, +/// returning a classified `Error`: transient SQL states (connection/resource +/// classes) map to `Error::Connection`, permanent ones (syntax/constraint) to +/// `Error::InvalidRecordValue`. Clearing is required so the next JNI call on this +/// thread is not aborted. +/// +/// NOTE: the runtime does not yet branch on this distinction (there is no +/// per-variant backoff in the poll loop today), so the classification is +/// currently informational: it shapes the error variant and log message and is +/// kept ready for when SDK-side backoff lands. Do not claim differentiated +/// runtime backoff until that exists. +fn classify_query_failure(env: &mut JNIEnv, action: &str) -> Error { + let (sql_state, message) = take_pending_sql_exception(env); + let transient = is_transient_sql_state(sql_state.as_deref()); + let state = sql_state.as_deref().unwrap_or("?"); + let msg = format!("Failed to {action} (SQLState {state}): {message}"); + if transient { + Error::Connection(msg) + } else { + Error::InvalidRecordValue(msg) + } +} + +/// Take the pending Java exception (clearing it) and return its `SQLState` (if a +/// `java.sql.SQLException`) and message. +fn take_pending_sql_exception(env: &mut JNIEnv) -> (Option, String) { + let throwable = match env.exception_occurred() { + Ok(t) if !t.is_null() => t, + _ => return (None, "unknown error".to_string()), + }; + let _ = env.exception_clear(); + + let message = throwable_string_method(env, &throwable, "getMessage") + .unwrap_or_else(|| "unknown error".to_string()); + let sql_state = if env + .is_instance_of(&throwable, "java/sql/SQLException") + .unwrap_or(false) + { + throwable_string_method(env, &throwable, "getSQLState") + } else { + None + }; + (sql_state, message) +} + +/// Call a no-arg `String`-returning method on a throwable; None on JNI error/null. +fn throwable_string_method( + env: &mut JNIEnv, + throwable: &JThrowable, + method: &str, +) -> Option { + let obj = env + .call_method(throwable, method, "()Ljava/lang/String;", &[]) + .ok()? + .l() + .ok()?; + if obj.is_null() { + return None; + } + env.get_string(&JString::from(obj)).ok().map(|s| s.into()) +} + +/// JDBC SQL Types constants +mod java { + pub mod sql { + #[allow(dead_code)] + pub struct Types; + + #[allow(dead_code)] + impl Types { + pub const BIT: i32 = -7; + pub const TINYINT: i32 = -6; + pub const SMALLINT: i32 = 5; + pub const INTEGER: i32 = 4; + pub const BIGINT: i32 = -5; + pub const FLOAT: i32 = 6; + pub const REAL: i32 = 7; + pub const DOUBLE: i32 = 8; + pub const NUMERIC: i32 = 2; + pub const DECIMAL: i32 = 3; + pub const CHAR: i32 = 1; + pub const VARCHAR: i32 = 12; + pub const LONGVARCHAR: i32 = -1; + pub const DATE: i32 = 91; + pub const TIME: i32 = 92; + pub const TIMESTAMP: i32 = 93; + pub const BINARY: i32 = -2; + pub const VARBINARY: i32 = -3; + pub const LONGVARBINARY: i32 = -4; + pub const NULL: i32 = 0; + pub const BOOLEAN: i32 = 16; + } + } +} + +// Export the connector via SDK macro +source_connector!(JdbcSource); + +#[cfg(test)] +mod tests { + use super::*; + + /// A minimal, valid bulk-mode config for tests that need a `JdbcSourceConfig`. + fn base_config() -> JdbcSourceConfig { + JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + } + } + + /// Write a throwaway file to stand in for a driver JAR and return its path, + /// so `validate_config`'s existence check passes. + fn write_temp_jar(name: &str) -> String { + let path = std::env::temp_dir().join(name); + std::fs::write(&path, b"jar").expect("write temp jar"); + path.to_string_lossy().into_owned() + } + + #[test] + fn test_quote_sql_literal_escapes_backslash() { + // Backslash is doubled before quotes so a trailing backslash cannot + // consume the closing quote under MySQL's default sql_mode. + assert_eq!(quote_sql_literal(r"a\b"), r"'a\\b'"); + assert_eq!(quote_sql_literal(r"end\"), r"'end\\'"); + assert_eq!(quote_sql_literal(r"\'"), r"'\\'''"); + } + + #[test] + fn test_larger_offset_i128_precision_beyond_f64() { + // Two BIGINTs that differ only past 2^53 must still be ordered exactly; + // an f64 comparison would collapse them and misorder the offset. + let a = "9007199254740993".to_string(); // 2^53 + 1 + let b = "9007199254740992".to_string(); // 2^53 + assert_eq!(larger_offset(a.clone(), b.clone()), a); + assert_eq!(larger_offset(b, a.clone()), a); + } + + #[test] + fn test_build_query_removes_predicate_case_and_whitespace_insensitive() { + for query in [ + "select * from t where {tracking_column} > {last_offset} order by id", + "SELECT * FROM t WHERE {tracking_column} > {last_offset} ORDER BY id", + "SELECT * FROM t where {tracking_column}>{last_offset} ORDER BY id", + ] { + let mut config = base_config(); + config.query = query.to_string(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + let source = JdbcSource::new(1, config, None); + let state = State::default(); + let built = source.build_query(&state).expect("build query"); + assert!( + !built.contains("{last_offset}") && !built.contains("{tracking_column}"), + "predicate not removed for query variant: {query} -> {built}" + ); + assert!(built.to_lowercase().contains("order by id")); + } + } + + #[test] + fn test_validate_config_rejects_half_set_credentials() { + let jar = write_temp_jar("jdbc_validate_half_creds.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.username = Some("user".to_string()); + config.password = None; + let source = JdbcSource::new(1, config, None); + assert!(matches!( + source.validate_config(), + Err(Error::InvalidConfigValue(_)) + )); + } + + #[test] + fn test_validate_config_rejects_missing_driver_jar() { + let mut config = base_config(); + config.driver_jar_path = "/nonexistent/path/to/driver.jar".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source.validate_config().expect_err("missing jar must fail"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("driver_jar_path"))); + } + + #[test] + fn test_validate_config_dry_runs_query() { + let jar = write_temp_jar("jdbc_validate_dry_run.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + // Passes the incremental invariants (tracking_column set, ordered by it) + // but the non-canonical predicate leaves {last_offset} unresolved with no + // offset, so the dry-run build_query must reject it now. + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = "SELECT * FROM t WHERE x >= {last_offset} ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + assert!(matches!( + source.validate_config(), + Err(Error::InvalidConfigValue(_)) + )); + } + + #[test] + fn test_validate_config_accepts_valid_bulk() { + let jar = write_temp_jar("jdbc_validate_ok.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + let source = JdbcSource::new(1, config, None); + assert!(source.validate_config().is_ok()); + } + + #[test] + fn test_validate_config_accepts_valid_incremental() { + let jar = write_temp_jar("jdbc_validate_ok_incremental.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + // Canonical placeholder predicate so the cold-start (no offset) build + // auto-removes the WHERE; ordered by the tracking column. + config.query = + "SELECT id, name FROM t WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(); + let source = JdbcSource::new(1, config, None); + assert!(source.validate_config().is_ok()); + } + + #[test] + fn test_validate_config_incremental_requires_tracking_column() { + let jar = write_temp_jar("jdbc_validate_no_tracking.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = None; + config.query = "SELECT id FROM t ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must require tracking_column"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("tracking_column"))); + } + + #[test] + fn test_validate_config_incremental_requires_order_by_tracking_column() { + let jar = write_temp_jar("jdbc_validate_no_order.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + // No ORDER BY: an unordered incremental query can skip rows on truncation. + config.query = "SELECT id FROM t WHERE id > {last_offset}".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source.validate_config().expect_err("must require ORDER BY"); + assert!( + matches!(err, Error::InvalidConfigValue(msg) if msg.to_lowercase().contains("order by")) + ); + } + + #[test] + fn test_query_orders_by_tracking_column_accepts_valid() { + assert!(query_orders_by_tracking_column( + "SELECT * FROM t WHERE id > {last_offset} ORDER BY id", + "id" + )); + // Placeholder form, case/whitespace variance, explicit ASC. + assert!(query_orders_by_tracking_column( + "select * from t order by {tracking_column}", + "updated_at" + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY Updated_At ASC", + "updated_at" + )); + // Table-qualified column, and the outer ORDER BY after a subquery. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY t.updated_at", + "updated_at" + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM (SELECT * FROM t ORDER BY x) s ORDER BY id", + "id" + )); + } + + #[test] + fn test_query_orders_by_tracking_column_rejects_invalid() { + // No ORDER BY. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t WHERE id > 0", + "id" + )); + // Different column. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY name", + "id" + )); + // Descending breaks ascending offset advancement. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY updated_at DESC", + "updated_at" + )); + // Substring-only match must not pass (id is a substring of valid_flag/id_backup). + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY valid_flag", + "id" + )); + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY id_backup", + "id" + )); + // Tracking column not the primary (first) ordering term. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY name, id", + "id" + )); + } + + #[test] + fn test_validate_config_rejects_bad_batch_size() { + let jar = write_temp_jar("jdbc_validate_batch_size.jar"); + for bad in [0u32, i32::MAX as u32] { + let mut config = base_config(); + config.driver_jar_path = jar.clone(); + config.batch_size = bad; + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must reject invalid batch_size"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("batch_size"))); + } + } + + #[test] + fn test_tracking_column_matches_case_and_normalization() { + // Case-insensitive against the raw driver label (driver case-folding). + assert!(tracking_column_matches( + "OrderDate", + "orderdate", + "orderdate" + )); + // Matches the normalized (snake_cased) output key. + assert!(tracking_column_matches( + "order_date", + "OrderDate", + "order_date" + )); + // Plain lowercase match. + assert!(tracking_column_matches("id", "id", "id")); + // Genuinely different column does not match. + assert!(!tracking_column_matches("id", "name", "name")); + } + + #[test] + fn test_tracking_offset_or_error_rejects_null_in_incremental() { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + let source = JdbcSource::new(1, config, None); + // Non-null resolves; NULL/empty (None) is a hard error in incremental mode. + assert_eq!( + source + .tracking_offset_or_error(Some("42".to_string()), "id") + .unwrap(), + Some("42".to_string()) + ); + assert!(matches!( + source.tracking_offset_or_error(None, "id"), + Err(Error::InvalidRecordValue(_)) + )); + } + + #[test] + fn test_tracking_offset_or_error_allows_null_in_bulk() { + let source = JdbcSource::new(1, base_config(), None); // base_config is bulk + assert_eq!(source.tracking_offset_or_error(None, "id").unwrap(), None); + } + + #[test] + fn test_sanitize_jdbc_url_mysql_format() { + let url = "jdbc:mysql://root:SuperSecret123@localhost:3306/mydb"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:mysql://root:***@localhost:3306/mydb"); + assert!(!sanitized.contains("SuperSecret123")); + } + + #[test] + fn test_sanitize_jdbc_url_postgresql_query_params() { + let url = "jdbc:postgresql://localhost:5432/mydb?user=admin&password=P@ssw0rd&ssl=true"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!( + sanitized, + "jdbc:postgresql://localhost:5432/mydb?user=admin&password=***&ssl=true" + ); + assert!(!sanitized.contains("P@ssw0rd")); + } + + #[test] + fn test_sanitize_jdbc_url_oracle_format() { + let url = "jdbc:oracle:thin:system/oracle123@localhost:1521:XE"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:oracle:thin:system/***@localhost:1521:XE"); + assert!(!sanitized.contains("oracle123")); + } + + #[test] + fn test_sanitize_jdbc_url_sqlserver_format() { + let url = "jdbc:sqlserver://localhost:1433;user=sa;password=MySecretPass;database=Sales"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!( + sanitized, + "jdbc:sqlserver://localhost:1433;user=sa;password=***;database=Sales" + ); + assert!(!sanitized.contains("MySecretPass")); + } + + #[test] + fn test_sanitize_jdbc_url_h2_format() { + let url = "jdbc:h2:mem:testdb;USER=sa;PASSWORD=secret"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:h2:mem:testdb;USER=sa;PASSWORD=***"); + assert!(!sanitized.contains("secret")); + } + + #[test] + fn test_sanitize_jdbc_url_case_insensitive() { + let url1 = "jdbc:postgresql://localhost?password=secret"; + let url2 = "jdbc:postgresql://localhost?PASSWORD=secret"; + let url3 = "jdbc:postgresql://localhost?pwd=secret"; + let url4 = "jdbc:postgresql://localhost?PWD=secret"; + + for url in [url1, url2, url3, url4] { + let sanitized = sanitize_jdbc_url(url); + assert!(!sanitized.contains("secret"), "Failed for URL: {}", url); + assert!(sanitized.contains("***")); + } + } + + #[test] + fn test_sanitize_jdbc_url_no_password() { + let url = "jdbc:h2:mem:testdb"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, url); + } + + #[test] + fn test_sanitize_jdbc_url_multiple_passwords() { + let url = "jdbc:postgresql://localhost?password=secret1&pwd=secret2"; + let sanitized = sanitize_jdbc_url(url); + assert!(!sanitized.contains("secret1")); + assert!(!sanitized.contains("secret2")); + assert_eq!( + sanitized, + "jdbc:postgresql://localhost?password=***&pwd=***" + ); + } + + #[test] + fn test_build_query_incremental_with_offset() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM users WHERE id > {last_offset} ORDER BY id".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("id".to_string()), + initial_offset: Some("0".to_string()), + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + + // With initial offset (no last_offset yet) + let state = State { + last_offset: None, + processed_rows: 0, + last_poll_time: Utc::now(), + }; + let query = source.build_query(&state).expect("build query"); + assert_eq!(query, "SELECT * FROM users WHERE id > '0' ORDER BY id"); + + // With tracked offset + let state = State { + last_offset: Some("42".to_string()), + processed_rows: 42, + last_poll_time: Utc::now(), + }; + let query = source.build_query(&state).expect("build query"); + assert_eq!(query, "SELECT * FROM users WHERE id > '42' ORDER BY id"); + } + + #[test] + fn test_build_query_substitutes_tracking_column() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: + "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("updated_at".to_string()), + initial_offset: Some("2024-01-01".to_string()), + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + let state = State { + last_offset: Some("2024-06-15".to_string()), + processed_rows: 0, + last_poll_time: Utc::now(), + }; + let query = source.build_query(&state).expect("build query"); + assert_eq!( + query, + "SELECT * FROM orders WHERE updated_at > '2024-06-15' ORDER BY updated_at" + ); + } + + #[test] + fn test_build_query_no_offset_substitutes_tracking_column_in_order_by() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: + "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("updated_at".to_string()), + initial_offset: None, + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + // No last_offset and no initial_offset: the WHERE predicate is dropped, + // but the ORDER BY {tracking_column} must still be substituted. + let query = source + .build_query(&State { + last_offset: None, + processed_rows: 0, + last_poll_time: Utc::now(), + }) + .expect("build query"); + assert!(!query.contains("{tracking_column}"), "got: {query}"); + assert!(!query.contains("{last_offset}"), "got: {query}"); + assert!(query.contains("ORDER BY updated_at"), "got: {query}"); + } + + #[test] + fn test_build_query_rejects_injection_in_tracking_column() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM t WHERE {tracking_column} > {last_offset}".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("id; DROP TABLE t".to_string()), + initial_offset: Some("0".to_string()), + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + assert!(source.build_query(&State::default()).is_err()); + } + + #[test] + fn test_build_query_bulk_mode_no_limit_appended() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM products".to_string(), + poll_interval: Duration::from_secs(60), + batch_size: 5000, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: false, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + let state = State::default(); + let query = source.build_query(&state).expect("build query"); + // build_query should NOT append LIMIT; row limiting is done via setMaxRows + assert_eq!(query, "SELECT * FROM products"); + assert!(!query.to_uppercase().contains("LIMIT")); + } + + #[test] + fn test_state_restoration_from_connector_state() { + let original_state = State { + last_offset: Some("2024-06-15 12:00:00".to_string()), + processed_rows: 1500, + last_poll_time: Utc::now(), + }; + let connector_state = ConnectorState::serialize(&original_state, CONNECTOR_NAME, 1) + .expect("Failed to serialize state"); + + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM orders WHERE updated_at > {last_offset}".to_string(), + poll_interval: Duration::from_secs(30), + batch_size: 1000, + tracking_column: Some("updated_at".to_string()), + initial_offset: Some("2024-01-01 00:00:00".to_string()), + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, Some(connector_state)); + let state = source.state.lock().unwrap(); + assert_eq!(state.last_offset, Some("2024-06-15 12:00:00".to_string())); + assert_eq!(state.processed_rows, 1500); + } + + #[test] + fn test_quote_sql_literal_escapes_single_quotes() { + assert_eq!(quote_sql_literal("42"), "'42'"); + assert_eq!( + quote_sql_literal("2024-01-01 00:00:00"), + "'2024-01-01 00:00:00'" + ); + assert_eq!(quote_sql_literal("o'brien"), "'o''brien'"); + assert_eq!( + quote_sql_literal("x'; DROP TABLE t; --"), + "'x''; DROP TABLE t; --'" + ); + } + + #[test] + fn test_is_transient_sql_state() { + for s in [ + "08001", "08006", "40001", "40P01", "53300", "57P01", "58030", + ] { + assert!(is_transient_sql_state(Some(s)), "{s} should be transient"); + } + for s in ["22001", "23505", "42601", "42P01", "99999"] { + assert!(!is_transient_sql_state(Some(s)), "{s} should be permanent"); + } + assert!(!is_transient_sql_state(None)); + assert!(!is_transient_sql_state(Some(""))); + } + + #[test] + fn test_larger_offset_numeric_and_lexical() { + // Numeric comparison, not lexical: "100" > "9" numerically. + assert_eq!(larger_offset("9".into(), "100".into()), "100"); + assert_eq!(larger_offset("100".into(), "9".into()), "100"); + assert_eq!(larger_offset("10.5".into(), "9.9".into()), "10.5"); + // Non-numeric (timestamps / text) compare lexicographically. + assert_eq!( + larger_offset("2024-01-01".into(), "2024-06-15".into()), + "2024-06-15" + ); + } + + #[test] + fn test_is_valid_identifier() { + assert!(is_valid_identifier("id")); + assert!(is_valid_identifier("updated_at")); + assert!(is_valid_identifier("t.updated_at")); + assert!(!is_valid_identifier("id; DROP TABLE t")); + assert!(!is_valid_identifier("1col")); + assert!(!is_valid_identifier("col name")); + assert!(!is_valid_identifier("")); + } + + #[test] + fn test_to_snake_case() { + assert_eq!(to_snake_case("OrderDate"), "order_date"); + assert_eq!(to_snake_case("updatedAt"), "updated_at"); + assert_eq!(to_snake_case("ID"), "id"); // consecutive uppers stay together + assert_eq!(to_snake_case("already_snake"), "already_snake"); + assert_eq!(to_snake_case("simple"), "simple"); + } + + // ========================================================================= + // Config deserialization tests + // ========================================================================= + + #[test] + fn test_config_deserialization_minimal_toml() { + let toml_str = r#" + jdbc_url = "jdbc:h2:mem:test" + driver_class = "org.h2.Driver" + driver_jar_path = "/tmp/h2.jar" + query = "SELECT * FROM users" + poll_interval = "30s" + "#; + let config: JdbcSourceConfig = + toml::from_str(toml_str).expect("Failed to parse minimal TOML config"); + assert_eq!(config.driver_class, "org.h2.Driver"); + assert_eq!(config.query, "SELECT * FROM users"); + assert_eq!(config.poll_interval, Duration::from_secs(30)); + // Verify defaults are applied + assert_eq!(config.mode, Mode::Incremental); + assert_eq!(config.batch_size, 1000); + assert!(config.include_metadata); + assert!(!config.snake_case_columns); + assert_eq!(config.connection_timeout_ms, 30000); + assert!(config.username.is_none()); + assert!(config.password.is_none()); + assert!(config.tracking_column.is_none()); + assert!(config.initial_offset.is_none()); + assert!(config.jvm_options.is_empty()); + } + + #[test] + fn test_config_deserialization_full_toml() { + let toml_str = r#" + jdbc_url = "jdbc:mysql://localhost:3306/mydb" + driver_class = "com.mysql.cj.jdbc.Driver" + driver_jar_path = "/opt/drivers/mysql.jar" + username = "admin" + password = "s3cret" + query = "SELECT * FROM orders WHERE id > {last_offset} ORDER BY id" + poll_interval = "5m" + batch_size = 500 + tracking_column = "id" + initial_offset = "0" + mode = "incremental" + snake_case_columns = true + include_metadata = false + jvm_options = ["-Xmx512m", "-Xms128m"] + connection_timeout_ms = 60000 + "#; + let config: JdbcSourceConfig = + toml::from_str(toml_str).expect("Failed to parse full TOML config"); + assert_eq!(config.driver_class, "com.mysql.cj.jdbc.Driver"); + assert_eq!(config.username.as_deref(), Some("admin")); + assert!(config.password.is_some()); + assert_eq!(config.batch_size, 500); + assert_eq!(config.tracking_column.as_deref(), Some("id")); + assert_eq!(config.initial_offset.as_deref(), Some("0")); + assert_eq!(config.mode, Mode::Incremental); + assert!(config.snake_case_columns); + assert!(!config.include_metadata); + assert_eq!(config.jvm_options, vec!["-Xmx512m", "-Xms128m"]); + assert_eq!(config.connection_timeout_ms, 60000); + assert_eq!(config.poll_interval, Duration::from_secs(300)); + } + + #[test] + fn test_config_deserialization_bulk_mode() { + let toml_str = r#" + jdbc_url = "jdbc:h2:mem:test" + driver_class = "org.h2.Driver" + driver_jar_path = "/tmp/h2.jar" + query = "SELECT * FROM products" + poll_interval = "1h" + mode = "bulk" + "#; + let config: JdbcSourceConfig = + toml::from_str(toml_str).expect("Failed to parse bulk mode config"); + assert_eq!(config.mode, Mode::Bulk); + assert_eq!(config.poll_interval, Duration::from_secs(3600)); + } + + #[test] + fn test_config_deserialization_invalid_mode_fails() { + let toml_str = r#" + jdbc_url = "jdbc:h2:mem:test" + driver_class = "org.h2.Driver" + driver_jar_path = "/tmp/h2.jar" + query = "SELECT 1" + poll_interval = "1s" + mode = "invalid_mode" + "#; + let result = toml::from_str::(toml_str); + assert!( + result.is_err(), + "Expected error for invalid mode, but got: {:?}", + result + ); + } + + // ========================================================================= + // State restoration tests + // ========================================================================= + + #[test] + fn test_state_restoration_with_malformed_bytes_falls_back_to_default() { + let connector_state = ConnectorState(vec![0xFF, 0xFE, 0xFD, 0x00]); + + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, Some(connector_state)); + let state = source.state.lock().unwrap(); + // Should fall back to default state + assert!(state.last_offset.is_none()); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn test_state_restoration_with_empty_bytes_falls_back_to_default() { + let connector_state = ConnectorState(vec![]); + + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, Some(connector_state)); + let state = source.state.lock().unwrap(); + assert!(state.last_offset.is_none()); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn test_state_restoration_none_uses_initial_offset() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM orders WHERE id > {last_offset}".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("id".to_string()), + initial_offset: Some("100".to_string()), + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + let state = source.state.lock().unwrap(); + assert_eq!(state.last_offset, Some("100".to_string())); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn test_state_restoration_none_without_initial_offset() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM products".to_string(), + poll_interval: Duration::from_secs(60), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + let state = source.state.lock().unwrap(); + assert!(state.last_offset.is_none()); + assert_eq!(state.processed_rows, 0); + } + + // ========================================================================= + // extract_offset_value tests + // ========================================================================= + + #[test] + fn test_extract_offset_value_with_integer() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + + let mut row = serde_json::Map::new(); + row.insert("id".to_string(), serde_json::json!(42)); + assert_eq!( + source.extract_offset_value(&row, "id"), + Some("42".to_string()) + ); + } + + #[test] + fn test_extract_offset_value_with_string() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + + let mut row = serde_json::Map::new(); + row.insert( + "updated_at".to_string(), + serde_json::json!("2024-06-15 12:00:00"), + ); + assert_eq!( + source.extract_offset_value(&row, "updated_at"), + Some("2024-06-15 12:00:00".to_string()) + ); + } + + #[test] + fn test_extract_offset_value_with_float() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + + let mut row = serde_json::Map::new(); + row.insert("version".to_string(), serde_json::json!(3.5)); + assert_eq!( + source.extract_offset_value(&row, "version"), + Some("3.5".to_string()) + ); + } + + #[test] + fn test_extract_offset_value_with_null() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + + let mut row = serde_json::Map::new(); + row.insert("id".to_string(), serde_json::Value::Null); + // A SQL NULL tracking value must not become the persisted offset. + assert_eq!(source.extract_offset_value(&row, "id"), None); + } + + #[test] + fn test_extract_offset_value_missing_column() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + + let row = serde_json::Map::new(); + assert_eq!(source.extract_offset_value(&row, "nonexistent"), None); + } + + // ========================================================================= + // build_query edge case tests + // ========================================================================= + + #[test] + fn test_build_query_incremental_no_offset_no_initial_removes_where_clause() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM users WHERE {tracking_column} > {last_offset} ORDER BY id" + .to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("id".to_string()), + initial_offset: None, + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + let state = State { + last_offset: None, + processed_rows: 0, + last_poll_time: Utc::now(), + }; + let query = source.build_query(&state).expect("build query"); + // The WHERE clause placeholder should be removed + assert!( + !query.contains("{last_offset}"), + "Query should not contain unresolved placeholder: {}", + query + ); + } + + #[test] + fn test_build_query_rejects_unresolved_placeholder() { + // Incremental, no offset, and a non-canonical predicate that the + // auto-remove does not match: the unresolved {last_offset} must produce + // an error rather than being shipped to the driver as invalid SQL. + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM t WHERE id >= {last_offset}".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Incremental, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + assert!(source.build_query(&State::default()).is_err()); + } + + #[test] + fn test_build_query_bulk_mode_ignores_offset() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT * FROM users ORDER BY id".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: Some("id".to_string()), + initial_offset: Some("0".to_string()), + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + let source = JdbcSource::new(1, config, None); + let state = State { + last_offset: Some("42".to_string()), + processed_rows: 42, + last_poll_time: Utc::now(), + }; + // In bulk mode the query is used verbatim; the tracked offset is ignored. + let query = source.build_query(&state).expect("build query"); + assert_eq!(query, "SELECT * FROM users ORDER BY id"); + } + + // ========================================================================= + // Mode enum tests + // ========================================================================= + + #[test] + fn test_mode_serialization_roundtrip() { + let incremental = Mode::Incremental; + let serialized = serde_json::to_string(&incremental).unwrap(); + assert_eq!(serialized, r#""incremental""#); + let deserialized: Mode = serde_json::from_str(&serialized).unwrap(); + assert_eq!(deserialized, Mode::Incremental); + + let bulk = Mode::Bulk; + let serialized = serde_json::to_string(&bulk).unwrap(); + assert_eq!(serialized, r#""bulk""#); + let deserialized: Mode = serde_json::from_str(&serialized).unwrap(); + assert_eq!(deserialized, Mode::Bulk); + } + + #[test] + fn test_mode_deserialization_rejects_unknown() { + let result = serde_json::from_str::(r#""streaming""#); + assert!( + result.is_err(), + "Unknown mode 'streaming' should fail deserialization" + ); + } + + // ========================================================================= + // Debug impl tests (ensures secrets are not leaked) + // ========================================================================= + + #[test] + fn test_config_debug_does_not_leak_password() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:mysql://root:SuperSecret@localhost/db"), + driver_class: "com.mysql.cj.jdbc.Driver".to_string(), + driver_jar_path: "/tmp/mysql.jar".to_string(), + username: Some("admin".to_string()), + password: Some(SecretString::from("MyP@ssw0rd")), + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + + let debug_output = format!("{:?}", config); + assert!( + !debug_output.contains("SuperSecret"), + "Debug output should not contain JDBC URL password: {}", + debug_output + ); + assert!( + !debug_output.contains("MyP@ssw0rd"), + "Debug output should not contain password field: {}", + debug_output + ); + assert!( + debug_output.contains("***"), + "Debug output should contain masked password: {}", + debug_output + ); + } + + #[test] + fn test_config_debug_without_password() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Duration::from_secs(10), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + jvm_options: vec![], + connection_timeout_ms: 30000, + }; + + let debug_output = format!("{:?}", config); + // Should not panic and should contain the struct name + assert!(debug_output.contains("JdbcSourceConfig")); + } +} diff --git a/core/integration/tests/connectors/jdbc/config_postgres.toml b/core/integration/tests/connectors/jdbc/config_postgres.toml new file mode 100644 index 0000000000..19097f5988 --- /dev/null +++ b/core/integration/tests/connectors/jdbc/config_postgres.toml @@ -0,0 +1,22 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# JDBC Source Connector Runtime Configuration for Postgres + +[connectors] +config_type = "local" +config_dir = "tests/connectors/jdbc/connectors_config_postgres" diff --git a/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml b/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml new file mode 100644 index 0000000000..0f35c4c4c1 --- /dev/null +++ b/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml @@ -0,0 +1,49 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# JDBC Source Connector for PostgreSQL + +type = "source" +key = "jdbc_pg" +enabled = true +version = 0 +name = "JDBC PostgreSQL Source" +path = "../../target/debug/libiggy_connector_jdbc_source" + +[[streams]] +stream = "test_stream" +topic = "test_topic" +schema = "json" +partition_id = 1 + +[plugin_config] +# All required fields - values will be overridden by environment variables +# Use placeholder values that match the expected types +jdbc_url = "" +driver_class = "" +driver_jar_path = "" +username = "" +password = "" +query = "" +poll_interval = "1s" +batch_size = 100 +mode = "bulk" +# Used only by incremental-mode tests; ignored in bulk mode. +tracking_column = "id" +initial_offset = "0" +snake_case_columns = false +include_metadata = true diff --git a/core/integration/tests/connectors/jdbc/mod.rs b/core/integration/tests/connectors/jdbc/mod.rs new file mode 100644 index 0000000000..2d1214308a --- /dev/null +++ b/core/integration/tests/connectors/jdbc/mod.rs @@ -0,0 +1,21 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +// JDBC connector tests. +// Source PostgreSQL tests: test_with_postgres.rs +// Sink PostgreSQL tests: test_sink_with_postgres.rs +mod test_with_postgres; diff --git a/core/integration/tests/connectors/jdbc/test_with_postgres.rs b/core/integration/tests/connectors/jdbc/test_with_postgres.rs new file mode 100644 index 0000000000..6004bf80fe --- /dev/null +++ b/core/integration/tests/connectors/jdbc/test_with_postgres.rs @@ -0,0 +1,624 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use crate::connectors::{ConnectorsRuntime, IggySetup, setup_runtime}; +use serial_test::serial; +use sqlx::postgres::PgPoolOptions; +use std::collections::HashMap; +use std::time::Duration; +use testcontainers_modules::postgres::Postgres; +use testcontainers_modules::testcontainers::ContainerAsync; +use testcontainers_modules::testcontainers::runners::AsyncRunner; +use tokio::time::sleep; +use tracing::info; + +const POSTGRES_USER: &str = "postgres"; +const POSTGRES_PASSWORD: &str = "postgres"; +const POSTGRES_DB: &str = "postgres"; + +/// Maximum number of poll attempts before giving up +const POLL_ATTEMPTS: usize = 30; +/// Delay between poll attempts +const POLL_INTERVAL: Duration = Duration::from_millis(500); + +/// Setup Postgres container with test data +async fn setup_postgres_container() +-> Result<(ContainerAsync, String, String), Box> { + info!("Starting Postgres container for JDBC testing..."); + + let postgres = Postgres::default().start().await?; + + let host = postgres.get_host().await?; + let port = postgres.get_host_port_ipv4(5432).await?; + let jdbc_url: String = format!("jdbc:postgresql://{}:{}/{}", host, port, POSTGRES_DB); + + let postgres_jar: String = get_postgres_driver_jar().await?; + + info!("Postgres container started at {}:{}", host, port); + Ok((postgres, jdbc_url, postgres_jar)) +} + +/// Get PostgreSQL JDBC driver, downloading if necessary +async fn get_postgres_driver_jar() -> Result> { + let target_dir = std::env::var("CARGO_TARGET_DIR").unwrap_or_else(|_| "target".to_string()); + let jdbc_test_dir = format!("{}/test-jdbc-drivers", target_dir); + let jar_path = format!("{}/postgresql-42.7.1.jar", jdbc_test_dir); + + std::fs::create_dir_all(&jdbc_test_dir)?; + + if std::path::Path::new(&jar_path).exists() { + info!("PostgreSQL JDBC driver found at {}", jar_path); + let absolute_path = std::fs::canonicalize(&jar_path)? + .to_string_lossy() + .to_string(); + return Ok(absolute_path); + } + + info!("Downloading PostgreSQL JDBC driver..."); + let download_url = "https://jdbc.postgresql.org/download/postgresql-42.7.1.jar"; + + let response = reqwest::get(download_url).await?; + if !response.status().is_success() { + return Err(format!("Failed to download driver: HTTP {}", response.status()).into()); + } + + let bytes = response.bytes().await?; + std::fs::write(&jar_path, bytes)?; + + info!("PostgreSQL JDBC driver downloaded to {}", jar_path); + let absolute_path = std::fs::canonicalize(&jar_path)? + .to_string_lossy() + .to_string(); + Ok(absolute_path) +} + +/// Build the environment variables for a JDBC Postgres source connector. +fn build_jdbc_env( + jdbc_url: &str, + postgres_jar: &str, + query: &str, + mode: &str, + iggy_setup: &IggySetup, +) -> HashMap { + let mut envs = HashMap::new(); + + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_JDBC_URL".to_owned(), + jdbc_url.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_CLASS".to_owned(), + "org.postgresql.Driver".to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_JAR_PATH".to_owned(), + postgres_jar.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_USERNAME".to_owned(), + POSTGRES_USER.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_PASSWORD".to_owned(), + POSTGRES_PASSWORD.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_QUERY".to_owned(), + query.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_POLL_INTERVAL".to_owned(), + "1s".to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), + "100".to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_MODE".to_owned(), + mode.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_SNAKE_CASE_COLUMNS".to_owned(), + "false".to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INCLUDE_METADATA".to_owned(), + "true".to_owned(), + ); + + // Stream configuration + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_STREAM".to_owned(), + iggy_setup.stream.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_TOPIC".to_owned(), + iggy_setup.topic.to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_SCHEMA".to_owned(), + "json".to_owned(), + ); + + envs +} + +/// Poll messages from Iggy with retry logic, returning deserialized JSON values. +async fn poll_messages_with_retry( + client: &crate::connectors::ConnectorsIggyClient, + expected_count: usize, +) -> Vec { + let mut received: Vec = Vec::new(); + for attempt in 0..POLL_ATTEMPTS { + let polled_messages = client + .get_messages() + .await + .expect("Failed to poll messages"); + + for msg in &polled_messages.messages { + if let Ok(value) = serde_json::from_slice::(&msg.payload) { + received.push(value); + } + } + + if received.len() >= expected_count { + info!( + "Received {} messages after {} attempts", + received.len(), + attempt + 1 + ); + return received; + } + + sleep(POLL_INTERVAL).await; + } + + received +} + +/// Setup connector runtime with JDBC source for Postgres +async fn setup_jdbc_postgres_source( + jdbc_url: &str, + postgres_jar: &str, + query: &str, + mode: &str, +) -> Result< + (ConnectorsRuntime, crate::connectors::ConnectorsIggyClient), + Box, +> { + let iggy_setup = IggySetup::default(); + let envs = build_jdbc_env(jdbc_url, postgres_jar, query, mode, &iggy_setup); + + let mut runtime = setup_runtime(); + runtime + .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) + .await; + + let client = runtime.create_client().await; + Ok((runtime, client)) +} + +/// Test: basic bulk mode query produces messages with correct structure +#[tokio::test] +#[serial] +async fn bulk_query_produces_message_to_iggy() { + let (_postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {}", e); + return; + } + }; + + let query = "SELECT 1 as id, 'test' as name"; + let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") + .await + .expect("Failed to setup runtime"); + + info!("Waiting for JDBC connector to poll from Postgres..."); + let messages = poll_messages_with_retry(&client, 1).await; + + assert!( + !messages.is_empty(), + "Expected at least 1 message from JDBC Postgres source" + ); + + // Verify message structure: should have metadata wrapping (include_metadata=true) + let first = &messages[0]; + assert!( + first.get("data").is_some(), + "Expected 'data' field in message (include_metadata=true), got: {}", + first + ); + assert_eq!( + first.get("operation_type").and_then(|v| v.as_str()), + Some("SELECT"), + "Expected operation_type=SELECT" + ); + + // Verify the actual data content + let data = first.get("data").unwrap(); + assert_eq!( + data.get("id").and_then(|v| v.as_i64()), + Some(1), + "Expected id=1 in data" + ); + assert_eq!( + data.get("name").and_then(|v| v.as_str()), + Some("test"), + "Expected name='test' in data" + ); +} + +/// Test: bulk mode with multiple rows from an actual table +#[tokio::test] +#[serial] +async fn bulk_query_produces_multiple_rows_to_iggy() { + let (postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {}", e); + return; + } + }; + + // Use a multi-row SELECT to simulate table data without needing DDL + let query = r#" + SELECT * FROM (VALUES + (1, 'alice', true), + (2, 'bob', false), + (3, 'carol', true) + ) AS t(id, name, active) + "#; + + let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") + .await + .expect("Failed to setup runtime"); + + info!("Waiting for JDBC connector to poll multiple rows..."); + let messages = poll_messages_with_retry(&client, 3).await; + + assert!( + messages.len() >= 3, + "Expected at least 3 messages, got {}", + messages.len() + ); + + // Verify each row has the expected structure + for msg in &messages[..3] { + let data = msg.get("data").expect("Missing 'data' field"); + assert!(data.get("id").is_some(), "Missing 'id' column in row data"); + assert!( + data.get("name").is_some(), + "Missing 'name' column in row data" + ); + assert!( + data.get("active").is_some(), + "Missing 'active' column in row data" + ); + } + + // Verify specific values for the first row + let first_data = messages[0].get("data").unwrap(); + assert_eq!(first_data.get("id").and_then(|v| v.as_i64()), Some(1)); + assert_eq!( + first_data.get("name").and_then(|v| v.as_str()), + Some("alice") + ); + + // Keep container alive until assertions complete + drop(postgres_container); +} + +/// Test: message contains timestamp field when metadata is enabled +#[tokio::test] +#[serial] +async fn source_includes_metadata_fields_when_enabled() { + let (_postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {}", e); + return; + } + }; + + let query = "SELECT 42 as value"; + let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") + .await + .expect("Failed to setup runtime"); + + let messages = poll_messages_with_retry(&client, 1).await; + assert!(!messages.is_empty(), "Expected at least 1 message"); + + let msg = &messages[0]; + + // Verify all metadata fields are present + assert!( + msg.get("timestamp").is_some(), + "Missing 'timestamp' metadata field" + ); + assert!( + msg.get("operation_type").is_some(), + "Missing 'operation_type' metadata field" + ); + assert!(msg.get("data").is_some(), "Missing 'data' metadata field"); + + // table_name should be null for SELECT queries without a specific table + // (this is expected behavior for computed queries) + assert!( + msg.get("table_name").is_some(), + "Missing 'table_name' metadata field" + ); +} + +/// Derive a sqlx (`postgres://`) URL from the connector's JDBC URL so the test +/// can seed the table the source reads from. +fn pg_sqlx_url(jdbc_url: &str) -> String { + let host_and_db = jdbc_url + .strip_prefix("jdbc:postgresql://") + .unwrap_or(jdbc_url); + format!("postgres://{POSTGRES_USER}:{POSTGRES_PASSWORD}@{host_and_db}") +} + +/// Collect the `data.id` integer from each polled (metadata-wrapped) message. +fn collect_ids(messages: &[serde_json::Value]) -> Vec { + messages + .iter() + .filter_map(|m| { + m.get("data") + .and_then(|d| d.get("id")) + .and_then(|v| v.as_i64()) + }) + .collect() +} + +/// Test: incremental mode advances its tracking offset across polls; newly +/// inserted rows are delivered exactly once and previously read rows are not +/// re-delivered. +#[tokio::test] +#[serial] +async fn incremental_mode_advances_offset_across_polls() { + let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {e}"); + return; + } + }; + + // Seed a real table BEFORE the source starts polling. + let pool = PgPoolOptions::new() + .max_connections(2) + .connect(&pg_sqlx_url(&jdbc_url)) + .await + .expect("Failed to connect to Postgres for seeding"); + sqlx::query("CREATE TABLE inc_test (id INT PRIMARY KEY, name TEXT)") + .execute(&pool) + .await + .expect("Failed to create table"); + sqlx::query("INSERT INTO inc_test (id, name) VALUES (1, 'a'), (2, 'b'), (3, 'c')") + .execute(&pool) + .await + .expect("Failed to insert initial rows"); + + let query = "SELECT id, name FROM inc_test WHERE id > {last_offset} ORDER BY id"; + let (_runtime, client) = + setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "incremental") + .await + .expect("Failed to setup runtime"); + + // First batch: ids 1..3. + let first = poll_messages_with_retry(&client, 3).await; + let mut first_ids = collect_ids(&first); + first_ids.sort_unstable(); + assert_eq!(first_ids, vec![1, 2, 3], "Expected ids 1,2,3 on first poll"); + + // Insert more rows; only these (id > last_offset) should arrive next. + sqlx::query("INSERT INTO inc_test (id, name) VALUES (4, 'd'), (5, 'e')") + .execute(&pool) + .await + .expect("Failed to insert additional rows"); + + let second = poll_messages_with_retry(&client, 2).await; + let mut second_ids = collect_ids(&second); + second_ids.sort_unstable(); + assert_eq!( + second_ids, + vec![4, 5], + "Expected only the new ids 4,5 (offset must have advanced past 3), got {second_ids:?}" + ); +} + +/// Test: a single poll over many rows succeeds. This exercises the JNI +/// local-reference frame management in `read_rows`: a few-hundred-row result set +/// creates hundreds of per-column local references in one native call, which +/// would overflow the JNI local reference table (and abort the JVM) if each row +/// were not read inside its own local frame. +#[tokio::test] +#[serial] +async fn large_result_set_streams_without_crashing() { + let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {e}"); + return; + } + }; + + let pool = PgPoolOptions::new() + .max_connections(2) + .connect(&pg_sqlx_url(&jdbc_url)) + .await + .expect("Failed to connect to Postgres for seeding"); + sqlx::query("CREATE TABLE big_test (id INT PRIMARY KEY, name TEXT, val NUMERIC(12,2))") + .execute(&pool) + .await + .expect("Failed to create table"); + sqlx::query( + "INSERT INTO big_test (id, name, val) \ + SELECT g, 'row_' || g, (g * 1.5)::numeric(12,2) FROM generate_series(1, 300) g", + ) + .execute(&pool) + .await + .expect("Failed to insert rows"); + + // batch_size well above the row count so the whole table is read in a single + // poll (one read_rows call → hundreds of local refs). + let iggy_setup = IggySetup::default(); + let query = "SELECT id, name, val FROM big_test ORDER BY id"; + let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "bulk", &iggy_setup); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), + "5000".to_owned(), + ); + + let mut runtime = setup_runtime(); + runtime + .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) + .await; + let client = runtime.create_client().await; + + // The runtime would have crashed on the oversized poll without per-row local + // frames; receiving a healthy batch of well-formed messages proves it did not. + let messages = poll_messages_with_retry(&client, 150).await; + assert!( + messages.len() >= 150, + "Expected the source to stream a large result set without crashing; got {} messages", + messages.len() + ); + for msg in &messages[..150] { + let data = msg.get("data").expect("Missing 'data' field"); + assert!(data.get("id").and_then(|v| v.as_i64()).is_some()); + assert!(data.get("name").and_then(|v| v.as_str()).is_some()); + } +} + +/// Test: the source keeps polling and recovers after a query that raises a +/// SQLException on every poll. The query targets a table that does not exist +/// yet, so `executeQuery` throws each cycle; once the table is created the very +/// next poll must succeed and deliver its rows. +/// +/// This is the regression guard for exception clearing: a thrown Java exception +/// left pending would make the following JNI call (the statement `close()` in +/// the error path, or the next poll's `isValid`) run with an exception pending, +/// which the JNI spec forbids and which aborts the embedded JVM (the whole +/// runtime process). If that happened the runtime would die and never deliver +/// the post-recovery rows below. +#[tokio::test] +#[serial] +async fn source_recovers_after_repeated_query_errors() { + let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {e}"); + return; + } + }; + + // Start the source against a table that does not exist yet: every poll + // raises "relation does not exist" (SQLState 42P01). + let query = "SELECT id, name FROM recover_test ORDER BY id"; + let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") + .await + .expect("Failed to setup runtime"); + + // Let the source fail across several poll cycles (poll interval is 1s). + sleep(Duration::from_secs(4)).await; + + // Now create and seed the table; the next successful poll should deliver it. + let pool = PgPoolOptions::new() + .max_connections(2) + .connect(&pg_sqlx_url(&jdbc_url)) + .await + .expect("Failed to connect to Postgres for seeding"); + sqlx::query("CREATE TABLE recover_test (id INT PRIMARY KEY, name TEXT)") + .execute(&pool) + .await + .expect("Failed to create table"); + sqlx::query("INSERT INTO recover_test (id, name) VALUES (1, 'a'), (2, 'b')") + .execute(&pool) + .await + .expect("Failed to insert rows"); + + let messages = poll_messages_with_retry(&client, 2).await; + let mut ids = collect_ids(&messages); + ids.sort_unstable(); + assert_eq!( + ids, + vec![1, 2], + "Source must recover after repeated query failures and deliver ids 1,2, got {ids:?}" + ); +} + +/// Test: bulk mode fails closed when the result set is larger than batch_size, +/// rather than silently syncing a truncated subset. With batch_size below the +/// row count the source errors every poll and delivers nothing (in particular, +/// never a truncated partial set). +#[tokio::test] +#[serial] +async fn bulk_result_larger_than_batch_size_fails_closed() { + let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(e) => { + eprintln!("Skipping test: Failed to setup Postgres: {e}"); + return; + } + }; + + let pool = PgPoolOptions::new() + .max_connections(2) + .connect(&pg_sqlx_url(&jdbc_url)) + .await + .expect("Failed to connect to Postgres for seeding"); + sqlx::query("CREATE TABLE trunc_test (id INT PRIMARY KEY)") + .execute(&pool) + .await + .expect("Failed to create table"); + sqlx::query("INSERT INTO trunc_test (id) SELECT generate_series(1, 5)") + .execute(&pool) + .await + .expect("Failed to insert rows"); + + // Bulk mode with batch_size below the 5-row result set. + let iggy_setup = IggySetup::default(); + let query = "SELECT id FROM trunc_test ORDER BY id"; + let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "bulk", &iggy_setup); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), + "2".to_owned(), + ); + + let mut runtime = setup_runtime(); + runtime + .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) + .await; + let client = runtime.create_client().await; + + // Several poll cycles (poll interval is 1s). A fail-closed source delivers + // nothing, and in particular never the truncated 2-row subset. + sleep(Duration::from_secs(4)).await; + let polled = client + .get_messages() + .await + .expect("Failed to poll messages"); + assert!( + polled.messages.is_empty(), + "bulk truncation must fail closed and deliver nothing, got {} messages", + polled.messages.len() + ); +} diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index a1433160b9..fb1cc67afb 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -25,6 +25,7 @@ mod http; mod http_config_provider; mod iceberg; mod influxdb; +mod jdbc; mod mongodb; mod postgres; mod quickwit; @@ -35,8 +36,44 @@ mod s3; mod stdout; mod surrealdb; -use iggy_common::IggyTimestamp; +use iggy::prelude::{IggyClient, IggyMessage, Partitioning}; +use iggy_common::Client; +use iggy_common::{ + CompressionAlgorithm, IggyExpiry, IggyTimestamp, MaxTopicSize, MessageClient, PolledMessages, + StreamClient, TopicClient, +}; +use integration::harness::{ConnectorsRuntimeConfig, IpAddrKind, TestHarness, TestServerConfig}; use serde::{Deserialize, Serialize}; +use std::collections::HashMap; + +const DEFAULT_TEST_STREAM: &str = "test_stream"; +const DEFAULT_TEST_TOPIC: &str = "test_topic"; + +fn setup_runtime() -> ConnectorsRuntime { + ConnectorsRuntime { + harness: TestHarness::builder() + .server( + TestServerConfig::builder() + .ip_kind(IpAddrKind::V4) + .quic_enabled(false) + .http_enabled(false) + .websocket_enabled(false) + .extra_envs(HashMap::from([ + // The harness pre-reserves a fixed TCP port (see PortReserver), + // so the server binds a non-zero port. That relies on the server + // writing current_config.toml on bind regardless of how the port + // was chosen (see tcp_listener.rs); the harness reads that file to + // discover the bound address before it considers startup complete. + ("IGGY_TCP_ADDRESS".to_owned(), "127.0.0.1:0".to_owned()), + ])) + .build(), + ) + .build() + .unwrap(), + stream: "".to_owned(), + topic: "".to_owned(), + } +} const ONE_DAY_MICROS: u64 = 24 * 60 * 60 * 1_000_000; @@ -63,3 +100,154 @@ pub fn create_test_messages(count: usize) -> Vec { }) .collect() } + +#[derive(Debug)] +struct ConnectorsRuntime { + stream: String, + topic: String, + harness: TestHarness, +} + +#[derive(Debug)] +struct ConnectorsIggyClient { + stream: String, + topic: String, + client: IggyClient, +} + +impl ConnectorsIggyClient { + /// Send messages to the configured stream/topic (used by sink connector tests). + #[allow(dead_code)] + async fn send_messages( + &self, + messages: &mut [IggyMessage], + ) -> Result<(), iggy_common::IggyError> { + self.client + .send_messages( + &self.stream.clone().try_into().unwrap(), + &self.topic.clone().try_into().unwrap(), + &Partitioning::balanced(), + messages, + ) + .await + } + + async fn get_messages(&self) -> Result { + self.client + .poll_messages( + &self.stream.clone().try_into().unwrap(), + &self.topic.clone().try_into().unwrap(), + None, + &iggy_common::Consumer::new("test_consumer".try_into().unwrap()), + &iggy_common::PollingStrategy::next(), + 10, + true, + ) + .await + } +} + +#[derive(Debug)] +pub struct IggySetup { + pub stream: String, + pub topic: String, +} + +impl Default for IggySetup { + fn default() -> Self { + Self { + stream: DEFAULT_TEST_STREAM.to_owned(), + topic: DEFAULT_TEST_TOPIC.to_owned(), + } + } +} + +impl ConnectorsRuntime { + pub async fn init( + &mut self, + config_path: &str, + envs: Option>, + iggy_setup: IggySetup, + ) { + let config_path = format!("tests/connectors/{config_path}"); + let mut all_envs = HashMap::new(); + all_envs.insert( + "IGGY_CONNECTORS_CONFIG_PATH".to_owned(), + config_path.to_owned(), + ); + + if let Some(envs) = envs { + for (k, v) in envs { + all_envs.insert(k, v); + } + } + + // Start the iggy server + self.harness + .start() + .await + .expect("Failed to start test harness"); + + let client = self.create_iggy_client().await; + client + .create_stream(&iggy_setup.stream) + .await + .expect("Failed to create stream"); + let stream_id = iggy_setup + .stream + .clone() + .try_into() + .expect("Invalid stream name in Iggy setup"); + client + .create_topic( + &stream_id, + &iggy_setup.topic, + 1, + CompressionAlgorithm::None, + None, + IggyExpiry::ServerDefault, + MaxTopicSize::ServerDefault, + ) + .await + .expect("Failed to create topic"); + client.shutdown().await.expect("Failed to shutdown client"); + + let connectors_config = ConnectorsRuntimeConfig::builder() + .extra_envs(all_envs) + .build(); + + self.harness + .server_mut() + .set_connectors_runtime_config(connectors_config); + self.harness + .server_mut() + .start_dependents() + .await + .expect("Failed to start connectors runtime"); + + self.stream = iggy_setup.stream; + self.topic = iggy_setup.topic; + } + + pub async fn create_client(&self) -> ConnectorsIggyClient { + ConnectorsIggyClient { + stream: self.stream.clone(), + topic: self.topic.clone(), + client: self.create_iggy_client().await, + } + } + + async fn create_iggy_client(&self) -> IggyClient { + self.harness + .tcp_root_client() + .await + .expect("Failed to create root TCP client") + } + + pub fn connectors_api_address(&self) -> Option { + self.harness + .server() + .connectors_runtime() + .map(|cr| cr.http_address().to_string()) + } +} diff --git a/core/server/src/tcp/tcp_listener.rs b/core/server/src/tcp/tcp_listener.rs index 3f53e4e640..c805f28d10 100644 --- a/core/server/src/tcp/tcp_listener.rs +++ b/core/server/src/tcp/tcp_listener.rs @@ -73,11 +73,11 @@ pub async fn start( // Store bound address locally shard.tcp_bound_address.set(Some(actual_addr)); - if addr.port() == 0 { - // Notify config writer on shard 0 - let _ = shard.config_writer_notify.try_send(()); + // Always notify config writer so it can write the current_config.toml + let _ = shard.config_writer_notify.try_send(()); - // Broadcast to other shards for SO_REUSEPORT binding + if addr.port() == 0 { + // Broadcast to other shards for SO_REUSEPORT binding (only needed for dynamic ports) let event = ShardEvent::AddressBound { protocol: TransportProtocol::Tcp, address: actual_addr, From 6ecc1845033e5906b52ac6262a0ff3e0bd31c485 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Tue, 21 Jul 2026 02:46:07 +0530 Subject: [PATCH 02/20] refactor(connectors): align JDBC source with sibling conventions Aligns the JDBC source crate with the other source connectors: crate version/type and publish flag, and poll_interval parsing via the workspace humantime crate rather than a direct humantime-serde pin. Removes a redundant per-column JNI getObject probe (using wasNull after the typed getter instead) and caps the per-poll liveness check at five seconds so a dead connection cannot stall a shared worker. Drops the unused partition_id field from the example configs and docs. --- Cargo.lock | 14 +- .../connectors/jdbc_bulk_mode.toml | 1 - .../example_config/connectors/jdbc_h2.toml | 1 - .../example_config/connectors/jdbc_mysql.toml | 1 - .../connectors/jdbc_oracle.toml | 1 - .../connectors/jdbc_sqlserver.toml | 1 - .../connectors/test_jdbc_h2.toml | 1 - .../connectors/sources/jdbc_source/Cargo.toml | 11 +- core/connectors/sources/jdbc_source/README.md | 5 +- .../connectors/sources/jdbc_source/src/lib.rs | 191 +++++++++++++----- .../connectors_config_postgres/jdbc_pg.toml | 1 - 11 files changed, 145 insertions(+), 83 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index 31c3666e92..e5491f8778 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -6203,16 +6203,6 @@ version = "2.3.0" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "135b12329e5e3ce057a9f972339ea52bc954fe1e9358ef27f95e89716fbc5424" -[[package]] -name = "humantime-serde" -version = "1.1.1" -source = "registry+https://github.com/rust-lang/crates.io-index" -checksum = "57a3db5ea5923d99402c94e9feb261dc5ee9b4efa158b0315f788cf549cc200c" -dependencies = [ - "humantime", - "serde", -] - [[package]] name = "hwlocality" version = "1.0.0-alpha.12" @@ -7036,13 +7026,13 @@ dependencies = [ [[package]] name = "iggy_connector_jdbc_source" -version = "0.1.0" +version = "0.4.1-edge.1" dependencies = [ "async-trait", "base64", "chrono", "dashmap", - "humantime-serde", + "humantime", "iggy_common", "iggy_connector_sdk", "jni 0.21.1", diff --git a/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml index 25e2b458ae..a179b42a50 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml @@ -77,5 +77,4 @@ include_metadata = false [[streams]] stream = "warehouse" topic = "product_summary" -partition_id = 1 schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml index 82ee459ff5..4dd1f0cc6e 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml @@ -56,5 +56,4 @@ include_metadata = true [[streams]] stream = "test" topic = "users" -partition_id = 1 schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml b/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml index fd2ed5fc48..f246781611 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml @@ -81,5 +81,4 @@ jvm_options = ["-Xmx512m", "-Xms128m"] [[streams]] stream = "ecommerce" topic = "orders" -partition_id = 1 schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml index 04478bf4ab..92d5a1e99e 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml @@ -78,5 +78,4 @@ jvm_options = ["-Xmx512m", "-Xms256m"] [[streams]] stream = "crm" topic = "customers" -partition_id = 1 schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml b/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml index 9e28d08a03..895f4beaf7 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml @@ -62,5 +62,4 @@ connection_timeout_ms = 30000 [[streams]] stream = "sales" topic = "orders" -partition_id = 1 schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml index 06b65c21b2..30457d163f 100644 --- a/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml +++ b/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml @@ -41,5 +41,4 @@ include_metadata = true [[streams]] stream = "test" topic = "users" -partition_id = 1 schema = "json" diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml index c3eabd9e61..839b7e424a 100644 --- a/core/connectors/sources/jdbc_source/Cargo.toml +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -17,19 +17,20 @@ [package] name = "iggy_connector_jdbc_source" -version = "0.1.0" +version = "0.4.1-edge.1" edition = "2024" license = "Apache-2.0" keywords = ["iggy", "messaging", "streaming", "jdbc", "source"] categories = ["database"] description = "Generic JDBC source connector for Iggy - supports MySQL, Oracle, SQL Server, H2, and any JDBC-compliant database" readme = "README.md" +publish = false [package.metadata.cargo-machete] -ignored = ["dashmap", "humantime-serde"] +ignored = ["dashmap"] [lib] -crate-type = ["cdylib", "rlib"] +crate-type = ["cdylib", "lib"] [features] default = [] @@ -42,8 +43,8 @@ chrono = { workspace = true } # Required by source_connector! macro dashmap = { workspace = true } -# For parsing duration strings -humantime-serde = "1.1" +# For parsing duration strings (poll_interval) +humantime = { workspace = true } # Shared serde helpers (SecretString serialization) iggy_common = { workspace = true } diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index c2e4d979b4..65bd59edab 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -99,7 +99,6 @@ include_metadata = true [[streams]] stream = "ecommerce" topic = "orders" -partition_id = 1 ``` ### Bulk Mode Configuration @@ -188,12 +187,12 @@ topic = "orders" | `username` | string | No | - | Database username (optional if in jdbc_url) | | `password` | string | No | - | Database password (optional if in jdbc_url) | | `query` | string | Yes | - | SQL query to execute (supports `{last_offset}` and `{tracking_column}` placeholders) | -| `poll_interval` | duration | Yes | - | How often to poll (e.g., "30s", "5m", "1h") | +| `poll_interval` | string (duration) | No | 5s | How often to poll, as a humantime string (e.g., "30s", "5m", "1h") | | `batch_size` | u32 | No | 1000 | Maximum rows to fetch per poll | | `tracking_column` | string | Incremental | - | Column to track for incremental reads (required in incremental mode; the query must also `ORDER BY` it) | | `initial_offset` | string | No | - | Starting offset value for first poll | | `mode` | string | No | "incremental" | Sync mode: "incremental" or "bulk" (bulk works with ALL databases) | -| `connection_timeout_ms` | u64 | No | 30000 | Timeout (ms) for the per-poll connection liveness check | +| `connection_timeout_ms` | u64 | No | 5000 | Timeout (ms) for the per-poll `isValid` liveness check; converted to seconds and capped at 5s | | `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | | `snake_case_columns` | bool | No | false | Convert column names to snake_case | | `include_metadata` | bool | No | true | Include metadata (table, operation, timestamp) | diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index ea19ab0916..c99058a25b 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -107,6 +107,21 @@ static RE_INCREMENTAL_PREDICATE: std::sync::LazyLock = std::sync::LazyLoc const CONNECTOR_NAME: &str = "JDBC source"; +/// Poll interval used when `poll_interval` is unset or empty. +const DEFAULT_POLL_INTERVAL: Duration = Duration::from_secs(5); + +/// Parse the configured `poll_interval` humantime string into a `Duration`, +/// falling back to [`DEFAULT_POLL_INTERVAL`] when unset, empty, or unparseable. +/// `validate_config` separately rejects a set-but-unparseable value so a typo +/// surfaces at `open()` rather than silently defaulting. +fn parse_poll_interval(poll_interval: Option<&str>) -> Duration { + poll_interval + .map(str::trim) + .filter(|value| !value.is_empty()) + .and_then(|value| humantime::parse_duration(value).ok()) + .unwrap_or(DEFAULT_POLL_INTERVAL) +} + /// Source mode for the JDBC connector #[derive(Debug, Clone, Deserialize, Serialize, PartialEq)] #[serde(rename_all = "lowercase")] @@ -148,9 +163,10 @@ pub struct JdbcSourceConfig { /// Can use {last_offset} placeholder for incremental reads pub query: String, - /// Polling interval (e.g., "30s", "5m", "1h") - #[serde(with = "humantime_serde")] - pub poll_interval: Duration, + /// Polling interval as a humantime string (e.g., "30s", "5m", "1h"). Parsed + /// once at construction; defaults to 5s when unset. + #[serde(default)] + pub poll_interval: Option, /// Batch size - maximum rows to fetch per poll #[serde(default = "default_batch_size")] @@ -181,15 +197,16 @@ pub struct JdbcSourceConfig { pub jvm_options: Vec, /// Timeout for the per-poll `Connection.isValid` liveness check (default: - /// 30000). JDBC expresses this timeout in whole seconds, so the value is - /// converted to seconds and clamped to the 1..=30s range; it does not govern - /// connection establishment. + /// 5000). JDBC expresses this timeout in whole seconds, so the value is + /// converted to seconds and clamped to the 1..=5s range (the check runs on a + /// shared worker and must stay short); it does not govern connection + /// establishment. #[serde(default = "default_connection_timeout")] pub connection_timeout_ms: u64, } fn default_connection_timeout() -> u64 { - 30000 + 5000 } fn default_batch_size() -> u32 { @@ -270,6 +287,8 @@ pub struct JdbcSource { // direct connection without `&mut self`. connection: Mutex>, state: Arc>, + // Poll interval parsed once from `config.poll_interval` at construction. + poll_interval: Duration, // Scheduled start of the next poll, used to pace polls at a fixed cadence // that does not drift with per-poll work time. `None` until the first poll. next_poll_at: Mutex>, @@ -304,12 +323,14 @@ impl JdbcSource { default_state }); + let poll_interval = parse_poll_interval(config.poll_interval.as_deref()); Self { id, config, jvm: None, connection: Mutex::new(None), state: Arc::new(Mutex::new(state)), + poll_interval, next_poll_at: Mutex::new(None), } } @@ -555,7 +576,9 @@ impl JdbcSource { /// Best-effort `Connection.isValid(timeout)` check. Returns false on any /// JNI error so the caller re-establishes the connection. fn connection_is_valid(&self, env: &mut JNIEnv, conn: &JObject) -> bool { - let timeout_secs = (self.config.connection_timeout_ms / 1000).clamp(1, 30) as i32; + // Cap the liveness check short: it runs on a shared block_in_place worker, + // so a dead connection must not block it for tens of seconds. + let timeout_secs = (self.config.connection_timeout_ms / 1000).clamp(1, 5) as i32; match env .call_method(conn, "isValid", "(I)Z", &[JValue::Int(timeout_secs)]) .and_then(|v| v.z()) @@ -1006,6 +1029,27 @@ impl JdbcSource { Ok(col_type) } + /// Return `value`, or JSON `null` when the last primitive getter read a SQL + /// NULL (detected via `ResultSet.wasNull()`). + fn null_or( + &self, + env: &mut JNIEnv, + result_set: &JObject, + value: serde_json::Value, + ) -> Result { + let was_null = jni!( + env, + env.call_method(result_set, "wasNull", "()Z", &[]) + .and_then(|v| v.z()), + "Failed to check wasNull" + ); + Ok(if was_null { + serde_json::Value::Null + } else { + value + }) + } + /// Extract column value based on JDBC type fn extract_column_value( &self, @@ -1016,23 +1060,11 @@ impl JdbcSource { ) -> Result { use java::sql::Types; - // Check if null first - let obj = jni!( - env, - env.call_method( - result_set, - "getObject", - "(I)Ljava/lang/Object;", - &[JValue::Int(column_index)], - ) - .and_then(|v| v.l()), - "Failed to get object" - ); - - if obj.is_null() { - return Ok(serde_json::Value::Null); - } - + // Primitive getters (getInt/getBoolean/...) return 0/false for SQL NULL, + // so `null_or` consults ResultSet.wasNull() after the getter to tell an + // actual NULL from a zero value. Object getters (getString/getBytes) + // return a null reference for SQL NULL and are null-checked directly, so + // there is no separate getObject probe (one JNI call per column, not two). match *sql_type { Types::BIT | Types::BOOLEAN => { let value = jni!( @@ -1046,7 +1078,7 @@ impl JdbcSource { .and_then(|v| v.z()), "Failed to get boolean" ); - Ok(serde_json::Value::Bool(value)) + self.null_or(env, result_set, serde_json::Value::Bool(value)) } Types::TINYINT | Types::SMALLINT | Types::INTEGER => { let value = jni!( @@ -1055,7 +1087,7 @@ impl JdbcSource { .and_then(|v| v.i()), "Failed to get int" ); - Ok(serde_json::json!(value)) + self.null_or(env, result_set, serde_json::json!(value)) } Types::BIGINT => { let value = jni!( @@ -1064,7 +1096,7 @@ impl JdbcSource { .and_then(|v| v.j()), "Failed to get long" ); - Ok(serde_json::json!(value)) + self.null_or(env, result_set, serde_json::json!(value)) } Types::FLOAT | Types::REAL => { let value = jni!( @@ -1073,7 +1105,7 @@ impl JdbcSource { .and_then(|v| v.f()), "Failed to get float" ); - Ok(serde_json::json!(value)) + self.null_or(env, result_set, serde_json::json!(value)) } Types::DOUBLE => { let value = jni!( @@ -1087,7 +1119,7 @@ impl JdbcSource { .and_then(|v| v.d()), "Failed to get double" ); - Ok(serde_json::json!(value)) + self.null_or(env, result_set, serde_json::json!(value)) } // NUMERIC/DECIMAL can carry more precision than an f64 can represent // (e.g. money/large decimals), so emit them as strings to avoid @@ -1223,6 +1255,17 @@ impl JdbcSource { ))); } + // A set poll_interval must be a valid humantime string; an unparseable + // value would otherwise silently fall back to the default. + if let Some(value) = self.config.poll_interval.as_deref() + && !value.trim().is_empty() + && humantime::parse_duration(value.trim()).is_err() + { + return Err(Error::InvalidConfigValue(format!( + "poll_interval '{value}' is not a valid duration (e.g. \"30s\", \"5m\", \"1h\")" + ))); + } + // Separate-credential auth requires both username and password; a // half-set pair would silently fall through to URL-embedded credentials. if self.config.username.is_some() != self.config.password.is_some() { @@ -1297,7 +1340,7 @@ impl Source for JdbcSource { let scheduled = { let mut next = lock_mutex(&self.next_poll_at, "next_poll_at")?; let scheduled = next.map_or_else(Instant::now, |planned| planned.max(Instant::now())); - *next = Some(scheduled + self.config.poll_interval); + *next = Some(scheduled + self.poll_interval); scheduled }; let now = Instant::now(); @@ -1686,7 +1729,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -1706,6 +1749,32 @@ mod tests { path.to_string_lossy().into_owned() } + #[test] + fn test_parse_poll_interval() { + assert_eq!(parse_poll_interval(Some("30s")), Duration::from_secs(30)); + assert_eq!(parse_poll_interval(Some("5m")), Duration::from_secs(300)); + // Unset, empty, and unparseable all fall back to the default. + assert_eq!(parse_poll_interval(None), DEFAULT_POLL_INTERVAL); + assert_eq!(parse_poll_interval(Some(" ")), DEFAULT_POLL_INTERVAL); + assert_eq!( + parse_poll_interval(Some("not-a-duration")), + DEFAULT_POLL_INTERVAL + ); + } + + #[test] + fn test_validate_config_rejects_bad_poll_interval() { + let jar = write_temp_jar("jdbc_validate_poll_interval.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.poll_interval = Some("banana".to_string()); + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must reject bad poll_interval"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("poll_interval"))); + } + #[test] fn test_quote_sql_literal_escapes_backslash() { // Backslash is doubled before quotes so a trailing backslash cannot @@ -2051,7 +2120,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM users WHERE id > {last_offset} ORDER BY id".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: Some("0".to_string()), @@ -2093,7 +2162,7 @@ mod tests { query: "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" .to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("updated_at".to_string()), initial_offset: Some("2024-01-01".to_string()), @@ -2127,7 +2196,7 @@ mod tests { query: "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" .to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("updated_at".to_string()), initial_offset: None, @@ -2161,7 +2230,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM t WHERE {tracking_column} > {last_offset}".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("id; DROP TABLE t".to_string()), initial_offset: Some("0".to_string()), @@ -2184,7 +2253,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM products".to_string(), - poll_interval: Duration::from_secs(60), + poll_interval: Some("60s".to_string()), batch_size: 5000, tracking_column: None, initial_offset: None, @@ -2219,7 +2288,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM orders WHERE updated_at > {last_offset}".to_string(), - poll_interval: Duration::from_secs(30), + poll_interval: Some("30s".to_string()), batch_size: 1000, tracking_column: Some("updated_at".to_string()), initial_offset: Some("2024-01-01 00:00:00".to_string()), @@ -2313,13 +2382,17 @@ mod tests { toml::from_str(toml_str).expect("Failed to parse minimal TOML config"); assert_eq!(config.driver_class, "org.h2.Driver"); assert_eq!(config.query, "SELECT * FROM users"); - assert_eq!(config.poll_interval, Duration::from_secs(30)); + assert_eq!(config.poll_interval.as_deref(), Some("30s")); + assert_eq!( + parse_poll_interval(config.poll_interval.as_deref()), + Duration::from_secs(30) + ); // Verify defaults are applied assert_eq!(config.mode, Mode::Incremental); assert_eq!(config.batch_size, 1000); assert!(config.include_metadata); assert!(!config.snake_case_columns); - assert_eq!(config.connection_timeout_ms, 30000); + assert_eq!(config.connection_timeout_ms, 5000); assert!(config.username.is_none()); assert!(config.password.is_none()); assert!(config.tracking_column.is_none()); @@ -2359,7 +2432,10 @@ mod tests { assert!(!config.include_metadata); assert_eq!(config.jvm_options, vec!["-Xmx512m", "-Xms128m"]); assert_eq!(config.connection_timeout_ms, 60000); - assert_eq!(config.poll_interval, Duration::from_secs(300)); + assert_eq!( + parse_poll_interval(config.poll_interval.as_deref()), + Duration::from_secs(300) + ); } #[test] @@ -2375,7 +2451,10 @@ mod tests { let config: JdbcSourceConfig = toml::from_str(toml_str).expect("Failed to parse bulk mode config"); assert_eq!(config.mode, Mode::Bulk); - assert_eq!(config.poll_interval, Duration::from_secs(3600)); + assert_eq!( + parse_poll_interval(config.poll_interval.as_deref()), + Duration::from_secs(3600) + ); } #[test] @@ -2411,7 +2490,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2439,7 +2518,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2464,7 +2543,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM orders WHERE id > {last_offset}".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: Some("100".to_string()), @@ -2489,7 +2568,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM products".to_string(), - poll_interval: Duration::from_secs(60), + poll_interval: Some("60s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2518,7 +2597,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2547,7 +2626,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2579,7 +2658,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2608,7 +2687,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2635,7 +2714,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2665,7 +2744,7 @@ mod tests { password: None, query: "SELECT * FROM users WHERE {tracking_column} > {last_offset} ORDER BY id" .to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: None, @@ -2702,7 +2781,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM t WHERE id >= {last_offset}".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2725,7 +2804,7 @@ mod tests { username: None, password: None, query: "SELECT * FROM users ORDER BY id".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: Some("0".to_string()), @@ -2787,7 +2866,7 @@ mod tests { username: Some("admin".to_string()), password: Some(SecretString::from("MyP@ssw0rd")), query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, @@ -2825,7 +2904,7 @@ mod tests { username: None, password: None, query: "SELECT 1".to_string(), - poll_interval: Duration::from_secs(10), + poll_interval: Some("10s".to_string()), batch_size: 100, tracking_column: None, initial_offset: None, diff --git a/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml b/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml index 0f35c4c4c1..0e2e1bad76 100644 --- a/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml +++ b/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml @@ -28,7 +28,6 @@ path = "../../target/debug/libiggy_connector_jdbc_source" stream = "test_stream" topic = "test_topic" schema = "json" -partition_id = 1 [plugin_config] # All required fields - values will be overridden by environment variables From 0e8047987230460e56e6d5b75d8233412772f35a Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Thu, 23 Jul 2026 14:06:42 +0530 Subject: [PATCH 03/20] fix(connectors): harden JDBC source incremental cursor and null handling Derives the incremental cursor from the last ordered row rather than a Rust-side max, so the cursor always matches the database's own ORDER BY and degrades to re-reads rather than skips if the ordering is imperfect. Fails loudly when the tracking column is absent from the result set, and rejects an empty query, a blank tracking_column, or a blank initial_offset at open(). Routes date/time columns through the null-safe string path so a NULL value no longer fails the poll, and preserves companion WHERE conditions (such as an operator-added IS NOT NULL) when stripping the offset predicate at cold start. --- core/connectors/sources/jdbc_source/README.md | 35 ++- .../connectors/sources/jdbc_source/src/lib.rs | 236 ++++++++++++------ 2 files changed, 181 insertions(+), 90 deletions(-) diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index 65bd59edab..4bfbce3554 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -218,21 +218,36 @@ connector refuses to start otherwise): `ORDER BY {tracking_column}` or the column name (optionally table-qualified, e.g. `ORDER BY t.updated_at`). A composite order such as `ORDER BY other, id` (tracking column not first) or a `DESC` order is rejected at `open()`. + The `ORDER BY` check is a lexical heuristic that validates single-block + `SELECT`s; `UNION`, window-function, or CTE queries may not be validated + correctly, so verify the result ordering yourself for those. +The connector takes the tracking value of the **last row** of each ordered batch +as the next cursor, so the cursor always matches the database's own `ORDER BY`. The tracking column must also be: -- **Homogeneously typed and monotonic.** Offsets are compared as integers, then - floats, then lexicographically; a column mixing numeric-looking and - non-numeric strings can make the connector's comparison disagree with the - database's `>` and skip or re-read rows. Prefer an auto-increment ID or a - timestamp. +- **Unique / strictly increasing.** The next poll resumes with a strict + `> {last_offset}`, and each batch is capped by `setMaxRows`. If a batch ends in + the middle of a run of rows that share the same tracking value (common for a + non-unique column like a timestamp), the remaining same-value rows are skipped + on the next poll. Use a unique, strictly-increasing key (an auto-increment ID + is ideal). If you must track a non-unique column, ensure `batch_size` exceeds + the largest group of equal values so a tie never spans a batch boundary. + (Keyset pagination with a tie-break is a planned follow-up.) +- **Monotonic under the database's own ordering.** Because the cursor is the last + ordered row and is fed back as `WHERE {tracking_column} > ''`, the column + must increase monotonically under the same ordering the database applies to that + `>` (including its collation, for text). Prefer an auto-increment ID or a + timestamp; a case-insensitively-collated text key can order differently than its + bytes and skip or re-read rows. - **`NOT NULL`.** A NULL tracking value cannot be watermarked, so the connector errors the poll if it reads one and keeps erroring (making no progress) until the query is fixed. Exclude NULLs in the query, e.g. `AND {tracking_column} IS NOT NULL`. -- **Lexicographically ordered, for timestamps.** Timestamp columns are read via - the driver's string form; ensure that form is ordered (ISO-8601 is). A locale - format such as `MM/DD/YYYY` is not monotonic as text and will misorder. +- **Round-trippable string form, for timestamps.** Timestamp columns are read as + the driver's string form and substituted back into the next `WHERE`; ensure the + driver emits a form the database orders correctly and can parse back (ISO-8601 + is safe; a locale format such as `MM/DD/YYYY` is not). **Example:** @@ -324,7 +339,9 @@ JDBC SQL types are automatically mapped to JSON: guarantee is tracked as a follow-up. - **Connection recovery.** The connection is validated with `Connection.isValid` each poll and transparently re-established (closing the old handle) if it has - dropped. + dropped. The check runs on the shared `block_in_place` worker, so its timeout + (`connection_timeout_ms`) is intentionally converted to whole seconds and + capped at 5s: a dead connection must not block the worker for tens of seconds. - **`SQLState` classification is informational today.** Query failures are classified into transient vs permanent error variants, but the runtime does not yet apply differentiated backoff based on that distinction; it currently diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index c99058a25b..7bbfd6e3d2 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -97,11 +97,20 @@ static RE_PASSWORD_PARAM: std::sync::LazyLock = static RE_ORACLE_PASS: std::sync::LazyLock = std::sync::LazyLock::new(|| Regex::new(r"thin:([^/]+)/([^@]+)@").unwrap()); -/// Matches the canonical incremental predicate so it can be removed on the first -/// (no-offset) poll. Case- and whitespace-tolerant so `where {tracking_column}>{last_offset}` -/// and `WHERE {tracking_column} > {last_offset}` are all recognized, rather -/// than only one exact literal spelling. -static RE_INCREMENTAL_PREDICATE: std::sync::LazyLock = std::sync::LazyLock::new(|| { +/// Regexes that remove the incremental offset predicate `{tracking_column} > +/// {last_offset}` on the first (no-offset) poll while preserving any other +/// `WHERE` conditions. All are case- and whitespace-tolerant. They are applied in +/// order so an operator-added companion condition (e.g. +/// `AND {tracking_column} IS NOT NULL`) does not leave a dangling `WHERE ... AND` +/// or leading `AND ...` fragment. See [`strip_offset_predicate`]. +static RE_OFFSET_PREDICATE_AND_AFTER: std::sync::LazyLock = std::sync::LazyLock::new(|| { + Regex::new(r"(?i)\bWHERE\s+\{tracking_column\}\s*>\s*\{last_offset\}\s+AND\s+").unwrap() +}); +static RE_OFFSET_PREDICATE_AND_BEFORE: std::sync::LazyLock = + std::sync::LazyLock::new(|| { + Regex::new(r"(?i)\s+AND\s+\{tracking_column\}\s*>\s*\{last_offset\}").unwrap() + }); +static RE_OFFSET_PREDICATE_BARE: std::sync::LazyLock = std::sync::LazyLock::new(|| { Regex::new(r"(?i)\bWHERE\s+\{tracking_column\}\s*>\s*\{last_offset\}").unwrap() }); @@ -790,7 +799,7 @@ impl JdbcSource { // request an absurd allocation up front. let mut messages = Vec::with_capacity((self.config.batch_size as usize).min(8192)); let mut row_count: u64 = 0; - let mut max_offset: Option = None; + let mut last_offset: Option = None; loop { let has_next = jni!( @@ -816,14 +825,15 @@ impl JdbcSource { let _ = unsafe { env.pop_local_frame(&JObject::null()) }; let (row_data, offset) = row_result?; - // Track the maximum tracking value across the batch rather than - // assuming the last row is the largest, so incremental mode is - // correct even if the query is not ordered by the tracking column. + // Take the tracking value of the LAST row as the next offset. Rows + // arrive in ascending tracking order (validate_config enforces + // ORDER BY the tracking column ascending), so the last row is the + // high-water mark. Using the last row rather than a Rust-side max + // keeps the cursor consistent with the database's own ordering, and + // degrades to re-reads (safe, at-least-once) rather than skips if the + // ordering is ever imperfect. if let Some(offset) = offset { - max_offset = Some(match max_offset { - Some(current) => larger_offset(current, offset), - None => offset, - }); + last_offset = Some(offset); } let message = self.build_message(row_data)?; @@ -831,7 +841,20 @@ impl JdbcSource { row_count += 1; } - Ok((messages, row_count, max_offset)) + // In incremental mode a non-empty batch that never yielded a tracking + // value means the tracking column is absent from the result set: the + // offset could never advance, so the same batch would be re-read forever. + // Fail loudly instead of stalling silently. (A NULL tracking value in a + // present column is already rejected per-row by tracking_offset_or_error.) + if self.config.mode == Mode::Incremental && row_count > 0 && last_offset.is_none() { + return Err(Error::InvalidConfigValue(format!( + "tracking column '{}' is not present in the query result; incremental mode cannot \ + advance its offset. Include it in the SELECT list.", + self.config.tracking_column.as_deref().unwrap_or("") + ))); + } + + Ok((messages, row_count, last_offset)) } /// Extract data from a single result set row, returning the row map and optional offset. @@ -951,7 +974,7 @@ impl JdbcSource { // Without an offset yet, drop the incremental predicate but keep the rest // of the query (e.g. an ORDER BY) intact. if offset.is_none() { - query = RE_INCREMENTAL_PREDICATE.replace(&query, "").into_owned(); + query = strip_offset_predicate(&query); } // Substitute {tracking_column} wherever it still appears (a WHERE and/or @@ -1154,25 +1177,11 @@ impl JdbcSource { base64::engine::general_purpose::STANDARD.encode(&buf), )) } + // Date/time types are read via their driver string form. Route + // through the null-safe getString path so a NULL date/time yields + // JSON null instead of failing the whole poll on get_string(null). Types::TIMESTAMP | Types::DATE | Types::TIME => { - let value = jni!( - env, - env.call_method( - result_set, - "getString", - "(I)Ljava/lang/String;", - &[JValue::Int(column_index)], - ) - .and_then(|v| v.l()), - "Failed to get timestamp" - ); - let str_value: String = jni!( - env, - env.get_string(&JString::from(value)), - "Failed to convert timestamp" - ) - .into(); - Ok(serde_json::Value::String(str_value)) + self.get_column_as_string(env, result_set, column_index) } // Default: getString for all other types (CHAR, VARCHAR, etc.) _ => self.get_column_as_string(env, result_set, column_index), @@ -1266,6 +1275,25 @@ impl JdbcSource { ))); } + // The query must be non-empty; an empty query only fails later at + // prepareStatement with an opaque driver error. + if self.config.query.trim().is_empty() { + return Err(Error::InvalidConfigValue( + "query must not be empty".to_string(), + )); + } + + // A set initial_offset must be non-blank: an empty value would build + // `WHERE tracking > ''` (a type error or always-false on many databases) + // rather than the intended cold-start scan. + if let Some(initial_offset) = self.config.initial_offset.as_deref() + && initial_offset.trim().is_empty() + { + return Err(Error::InvalidConfigValue( + "initial_offset must not be empty; omit it to start from the beginning".to_string(), + )); + } + // Separate-credential auth requires both username and password; a // half-set pair would silently fall through to URL-embedded credentials. if self.config.username.is_some() != self.config.password.is_some() { @@ -1281,10 +1309,15 @@ impl JdbcSource { // by that column, setMaxRows returns an arbitrary subset, and advancing // the offset to its max permanently skips the unread lower keys. if self.config.mode == Mode::Incremental { - let Some(tracking_column) = self.config.tracking_column.as_deref() else { + let Some(tracking_column) = self + .config + .tracking_column + .as_deref() + .filter(|column| !column.trim().is_empty()) + else { return Err(Error::InvalidConfigValue( - "incremental mode requires tracking_column so the offset can advance; set \ - tracking_column, or use mode = \"bulk\"" + "incremental mode requires a non-empty tracking_column so the offset can \ + advance; set tracking_column, or use mode = \"bulk\"" .to_string(), )); }; @@ -1514,28 +1547,17 @@ fn finalize_query(query: String) -> Result { Ok(query) } -/// Return the larger of two offset values. Integer keys are compared as `i128` -/// first (exact across the full `BIGINT` range, avoiding the f64 precision loss -/// past 2^53 that would misorder large IDs and contradict the NUMERIC/DECIMAL -/// -as-string rationale elsewhere in this file); genuinely fractional values -/// fall back to f64; non-numeric values (ISO-8601 timestamps, text keys) compare -/// lexicographically. -/// -/// This assumes the tracking column is homogeneously typed and monotonic under -/// the database's own ordering. A column that mixes numeric-looking and -/// non-numeric strings, or a timestamp whose driver string form is not -/// lexicographically ordered, can make this Rust-side max disagree with the -/// database's `>` comparison and cause rows to be skipped or re-read. See the -/// tracking-column guidance in the README. -fn larger_offset(a: String, b: String) -> String { - let b_is_larger = match (a.parse::(), b.parse::()) { - (Ok(na), Ok(nb)) => nb > na, - _ => match (a.parse::(), b.parse::()) { - (Ok(na), Ok(nb)) => nb > na, - _ => b > a, - }, - }; - if b_is_larger { b } else { a } +/// Remove the incremental offset predicate `{tracking_column} > {last_offset}` +/// from a query for the cold-start (no-offset) poll, preserving any other `WHERE` +/// conditions. Handles the offset term followed by `AND ...`, preceded by +/// `... AND`, or standing alone, so a companion condition such as +/// `AND {tracking_column} IS NOT NULL` survives as a valid `WHERE`. +fn strip_offset_predicate(query: &str) -> String { + let query = RE_OFFSET_PREDICATE_AND_AFTER.replace_all(query, "WHERE "); + let query = RE_OFFSET_PREDICATE_AND_BEFORE.replace_all(&query, ""); + RE_OFFSET_PREDICATE_BARE + .replace_all(&query, "") + .into_owned() } /// Check that an incremental query's result set is ordered ascending by the @@ -1784,16 +1806,6 @@ mod tests { assert_eq!(quote_sql_literal(r"\'"), r"'\\'''"); } - #[test] - fn test_larger_offset_i128_precision_beyond_f64() { - // Two BIGINTs that differ only past 2^53 must still be ordered exactly; - // an f64 comparison would collapse them and misorder the offset. - let a = "9007199254740993".to_string(); // 2^53 + 1 - let b = "9007199254740992".to_string(); // 2^53 - assert_eq!(larger_offset(a.clone(), b.clone()), a); - assert_eq!(larger_offset(b, a.clone()), a); - } - #[test] fn test_build_query_removes_predicate_case_and_whitespace_insensitive() { for query in [ @@ -1816,6 +1828,48 @@ mod tests { } } + #[test] + fn test_strip_offset_predicate_preserves_other_conditions() { + // Offset term followed by a companion condition (the README-advised + // IS NOT NULL): the AND and the companion condition survive. + assert_eq!( + strip_offset_predicate( + "SELECT * FROM t WHERE {tracking_column} > {last_offset} AND {tracking_column} IS NOT NULL ORDER BY {tracking_column}" + ), + "SELECT * FROM t WHERE {tracking_column} IS NOT NULL ORDER BY {tracking_column}" + ); + // Offset term preceded by a condition. + assert_eq!( + strip_offset_predicate( + "SELECT * FROM t WHERE active = 1 AND {tracking_column} > {last_offset} ORDER BY id" + ), + "SELECT * FROM t WHERE active = 1 ORDER BY id" + ); + // Bare offset term (no other condition) drops the whole WHERE. + assert!( + !strip_offset_predicate( + "SELECT * FROM t WHERE {tracking_column} > {last_offset} ORDER BY id" + ) + .contains("{last_offset}") + ); + } + + #[test] + fn test_build_query_cold_start_compound_predicate_is_valid() { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = "SELECT * FROM t WHERE {tracking_column} > {last_offset} AND {tracking_column} IS NOT NULL ORDER BY {tracking_column}".to_string(); + let source = JdbcSource::new(1, config, None); + // Cold start: no persisted offset, no initial_offset. + let built = source.build_query(&State::default()).expect("build query"); + assert_eq!(built, "SELECT * FROM t WHERE id IS NOT NULL ORDER BY id"); + assert!(!built.contains("{last_offset}") && !built.contains("{tracking_column}")); + // No dangling AND / empty WHERE. + assert!(!built.to_uppercase().contains("WHERE AND")); + assert!(!built.to_uppercase().contains("AND ORDER")); + } + #[test] fn test_validate_config_rejects_half_set_credentials() { let jar = write_temp_jar("jdbc_validate_half_creds.jar"); @@ -1972,6 +2026,39 @@ mod tests { )); } + #[test] + fn test_validate_config_rejects_empty_query_and_blank_offsets() { + let jar = write_temp_jar("jdbc_validate_blanks.jar"); + // Empty query. + let mut config = base_config(); + config.driver_jar_path = jar.clone(); + config.query = " ".to_string(); + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(msg)) if msg.contains("query")) + ); + + // Blank initial_offset. + let mut config = base_config(); + config.driver_jar_path = jar.clone(); + config.initial_offset = Some(" ".to_string()); + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(msg)) if msg.contains("initial_offset")) + ); + + // Blank tracking_column in incremental mode. + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("".to_string()); + config.query = "SELECT id FROM t ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(msg)) if msg.contains("tracking_column")) + ); + } + #[test] fn test_validate_config_rejects_bad_batch_size() { let jar = write_temp_jar("jdbc_validate_batch_size.jar"); @@ -2332,19 +2419,6 @@ mod tests { assert!(!is_transient_sql_state(Some(""))); } - #[test] - fn test_larger_offset_numeric_and_lexical() { - // Numeric comparison, not lexical: "100" > "9" numerically. - assert_eq!(larger_offset("9".into(), "100".into()), "100"); - assert_eq!(larger_offset("100".into(), "9".into()), "100"); - assert_eq!(larger_offset("10.5".into(), "9.9".into()), "10.5"); - // Non-numeric (timestamps / text) compare lexicographically. - assert_eq!( - larger_offset("2024-01-01".into(), "2024-06-15".into()), - "2024-06-15" - ); - } - #[test] fn test_is_valid_identifier() { assert!(is_valid_identifier("id")); From dfc245761ada7ee4d038db0de94c73d4c55f8ddb Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Thu, 23 Jul 2026 14:11:08 +0530 Subject: [PATCH 04/20] docs(connectors): caveat non-unique timestamp tracking columns in JDBC examples The incremental examples used timestamp tracking columns (updated_at, OrderDate) without noting that a non-unique tracking value can skip rows when a same-value group is split across a batch boundary. Add that caveat to the affected examples and point to the unique/strictly-increasing requirement. --- core/connectors/sources/jdbc_source/README.md | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index 4bfbce3554..1537d7d7b1 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -90,6 +90,10 @@ password = "secret_password" query = "SELECT * FROM orders WHERE updated_at > {last_offset} ORDER BY updated_at ASC" poll_interval = "30s" batch_size = 1000 +# updated_at is a timestamp and may not be unique. A run of rows sharing one +# timestamp that is split across a batch boundary would skip the remainder (see +# "Unique / strictly increasing" below). Prefer a unique auto-increment key, or +# keep batch_size larger than any same-timestamp group. tracking_column = "updated_at" initial_offset = "2024-01-01 00:00:00" mode = "incremental" @@ -168,6 +172,9 @@ password = "YourPassword123" query = "SELECT * FROM Orders WHERE OrderDate > {last_offset} ORDER BY OrderDate" poll_interval = "15s" batch_size = 2000 +# OrderDate is a timestamp and may not be unique; see the uniqueness note on the +# MySQL example above. Prefer a unique key, or keep batch_size above any +# same-timestamp group. tracking_column = "OrderDate" initial_offset = "2024-01-01" mode = "incremental" @@ -519,11 +526,13 @@ query = "SELECT * FROM table WHERE {tracking_column} > {last_offset} ORDER BY {t - Efficient for large tables - Works with timestamps, IDs, or any orderable column -**Database Examples** (the query must order by the tracking column): +**Database Examples** (the query must order by the tracking column; a unique, +strictly-increasing key like an auto-increment ID is safest, see the tracking +column requirements above): -- MySQL: `WHERE updated_at > {last_offset} ORDER BY updated_at` +- MySQL: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; ensure `batch_size` exceeds any same-timestamp group, or track a unique id) - Oracle: `WHERE id > {last_offset} ORDER BY id` (use a monotonic key; `ROWNUM` is not a valid tracking column) -- SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` +- SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; same caveat as MySQL) - PostgreSQL: `WHERE id > {last_offset} ORDER BY id` ### Bulk Mode (Universal) From 6f1eedcfa8de5f0f92aa2a60fde03d11c2bb93bd Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Mon, 3 Aug 2026 19:29:34 +0530 Subject: [PATCH 05/20] fix(connectors): align JDBC ORDER BY validation with read-time matching Incremental-mode validation and row reading disagreed about what a tracking-column name refers to, so a config that reads rows correctly could be rejected at open(), and a config that skips rows could be accepted. Read-time tracking_column_matches compares against both the raw driver label and the snake_cased output key, but the ORDER BY check compared a literal lowercased string. With snake_case_columns on, a snake_cased tracking_column against a CamelCase ORDER BY column (order_date vs OrderDate) failed validation despite matching at runtime. The check now normalizes the ORDER BY key the same way and routes both paths through tracking_column_matches, and it strips a surrounding identifier quote so a case-preserving column such as PostgreSQL's ORDER BY "OrderDate" validates against its unquoted driver label. Validation also derived the outer ordering with rfind("order by"), which matched a window function's internal ORDER BY (ROW_NUMBER() OVER (ORDER BY id) with no outer clause). That orders values within the frame, not the emitted ResultSet, so setMaxRows still returns an arbitrary subset and the advancing cursor skips rows, reopening the skip-row bug at validate time. Ordering detection now scans at parenthesis depth zero, ignoring ORDER BY inside a subquery, CTE, or OVER (...), and skips single-quoted string literals so a parenthesis in a literal cannot perturb the depth. A query whose only ordering sits inside such a construct is rejected. Update the README ORDER BY caveat to match and drop a stray line-ending backslash in the bulk example query. --- .../connectors/jdbc_bulk_mode.toml | 2 +- core/connectors/sources/jdbc_source/README.md | 13 +- .../connectors/sources/jdbc_source/src/lib.rs | 230 +++++++++++++++--- 3 files changed, 211 insertions(+), 34 deletions(-) diff --git a/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml index a179b42a50..c939d2285c 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml @@ -44,7 +44,7 @@ driver_jar_path = "/opt/jdbc-drivers/postgresql-42.6.0.jar" # Can include JOINs, aggregations, complex WHERE clauses, etc. query = """ SELECT - p.product_id,\ + p.product_id, p.product_name, p.category, p.price, diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index 1537d7d7b1..ee2e6ada3d 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -225,9 +225,16 @@ connector refuses to start otherwise): `ORDER BY {tracking_column}` or the column name (optionally table-qualified, e.g. `ORDER BY t.updated_at`). A composite order such as `ORDER BY other, id` (tracking column not first) or a `DESC` order is rejected at `open()`. - The `ORDER BY` check is a lexical heuristic that validates single-block - `SELECT`s; `UNION`, window-function, or CTE queries may not be validated - correctly, so verify the result ordering yourself for those. + The `ORDER BY` check is a lexical heuristic, not a full SQL parser: it inspects + the last `ORDER BY` at parenthesis depth zero, so an `ORDER BY` inside a + subquery, CTE, or a window function's `OVER (...)` is ignored rather than + mistaken for the outer ordering (a query whose only ordering sits inside such a + construct is rejected, since it has no outer `ORDER BY`). A surrounding + identifier quote (`"OrderDate"`, `` `col` ``, `[col]`) and, when + `snake_case_columns` is set, snake_case folding are both accounted for, so the + same `tracking_column` that matches a row at read time also passes validation. + The check does not interpret `UNION`; for a multi-branch `UNION` verify the + result ordering yourself. The connector takes the tracking value of the **last row** of each ordered batch as the next cursor, so the cursor always matches the database's own `ORDER BY`. diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 7bbfd6e3d2..16e3b9e12c 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -1321,7 +1321,11 @@ impl JdbcSource { .to_string(), )); }; - if !query_orders_by_tracking_column(&self.config.query, tracking_column) { + if !query_orders_by_tracking_column( + &self.config.query, + tracking_column, + self.config.snake_case_columns, + ) { return Err(Error::InvalidConfigValue(format!( "incremental mode requires the query to order by the tracking column so each \ batch is a contiguous ascending range; add `ORDER BY {tracking_column}` (or \ @@ -1567,40 +1571,112 @@ fn strip_offset_predicate(query: &str) -> String { /// `ORDER BY`, and must not be descending. /// /// This is a lexical check, not a SQL parser, so it errs strict: it inspects the -/// last `ORDER BY` in the text (the outer query's, not a subquery's), takes the -/// first ordering term, and requires it to be the `{tracking_column}` placeholder -/// or an identifier whose final path segment equals the tracking column -/// (case-insensitive). A trailing `DESC` is rejected, and a composite -/// `ORDER BY other, tracking` (tracking not primary) is rejected, because -/// truncation then yields a prefix ordered by `other`. -fn query_orders_by_tracking_column(query: &str, tracking_column: &str) -> bool { - let lower = query.to_lowercase(); - let Some(pos) = lower.rfind("order by") else { +/// last `ORDER BY` at parenthesis depth zero (the outer query's clause, never one +/// inside a subquery or a window function's `OVER (...)`), takes the first +/// ordering term, and requires it to be the `{tracking_column}` placeholder or an +/// identifier whose final path segment equals the tracking column. Matching +/// mirrors the read-time `tracking_column_matches`: case-insensitive against the +/// raw identifier and, when `snake_case_columns` is set, against its snake_cased +/// form, so a CamelCase `ORDER BY OrderDate` with `tracking_column = "order_date"` +/// validates exactly as it matches when reading rows (otherwise validation would +/// reject a config that runs correctly). A trailing `DESC` is rejected, and a +/// composite `ORDER BY other, tracking` (tracking not primary) is rejected, +/// because truncation then yields a prefix ordered by `other`. +fn query_orders_by_tracking_column( + query: &str, + tracking_column: &str, + snake_case_columns: bool, +) -> bool { + let Some(order_by) = outer_order_by(query) else { return false; }; - let after = lower[pos + "order by".len()..].trim_start(); - let first_term = after.split(',').next().unwrap_or("").trim(); + let first_term = order_by.split(',').next().unwrap_or("").trim(); let mut tokens = first_term.split_whitespace(); let key = tokens.next().unwrap_or(""); // Any explicit descending direction breaks ascending offset advancement. - if tokens.any(|token| token == "desc") { + if tokens.any(|token| token.eq_ignore_ascii_case("desc")) { return false; } if key.starts_with("{tracking_column}") { return true; } - // Compare the final path segment (`t.updated_at` -> `updated_at`), keeping - // only leading identifier characters so trailing punctuation is ignored. - let key_ident: String = key + // Compare the final path segment (`t.updated_at` -> `updated_at`), first + // stripping a surrounding identifier quote (`"pg"`, `` `mysql` ``, `[mssql]`) + // then keeping only leading identifier characters so trailing punctuation is + // ignored. A case-preserving column such as PostgreSQL's `ORDER BY + // "OrderDate"` must validate against the same `tracking_column` that matches + // its unquoted driver label when reading rows. + let segment = key .rsplit('.') .next() .unwrap_or(key) + .trim_matches(|c| c == '"' || c == '`' || c == '[' || c == ']'); + let key_ident: String = segment .chars() .take_while(|c| c.is_ascii_alphanumeric() || *c == '_') .collect(); - let tracking_lower = tracking_column.to_lowercase(); - let tracking_ident = tracking_lower.rsplit('.').next().unwrap_or(&tracking_lower); - !key_ident.is_empty() && key_ident == tracking_ident + if key_ident.is_empty() { + return false; + } + // Normalize identically to the read-time path so validate-time cannot reject + // a config whose ORDER BY column would match a returned row at runtime. + let normalized = if snake_case_columns { + to_snake_case(&key_ident) + } else { + key_ident.clone() + }; + tracking_column_matches(tracking_column, &key_ident, &normalized) +} + +/// Return the text following the outer `ORDER BY` (the last `order by` token at +/// parenthesis depth zero), or `None` if the query has no top-level ordering. An +/// `ORDER BY` inside a subquery or a window function's `OVER (...)` sits at depth +/// greater than zero and is ignored, so a window's internal ordering (which +/// orders values within the frame, not the emitted ResultSet) is never mistaken +/// for the row-emission order. Single-quoted string literals are skipped so a +/// parenthesis or the words `order by` inside a literal cannot perturb the depth +/// or produce a spurious match. +fn outer_order_by(query: &str) -> Option<&str> { + const NEEDLE: &[u8] = b"order by"; + let bytes = query.as_bytes(); + // ASCII lowercasing is byte-length-preserving, so indices into `lower` map + // one-to-one onto `query`; this keeps the matched key in its original case. + let lower = query.to_ascii_lowercase(); + let lower_bytes = lower.as_bytes(); + let mut depth: u32 = 0; + let mut in_quote = false; + let mut last_end = None; + for i in 0..bytes.len() { + let byte = bytes[i]; + if in_quote { + if byte == b'\'' { + in_quote = false; + } + continue; + } + match byte { + b'\'' => in_quote = true, + b'(' => depth += 1, + b')' => depth = depth.saturating_sub(1), + _ if depth == 0 + && lower_bytes[i..].starts_with(NEEDLE) + && (i == 0 || !is_identifier_byte(bytes[i - 1])) => + { + let end = i + NEEDLE.len(); + if bytes.get(end).is_none_or(|next| !is_identifier_byte(*next)) { + last_end = Some(end); + } + } + _ => {} + } + } + last_end.map(|end| query[end..].trim_start()) +} + +/// Whether a byte is part of an unquoted SQL identifier (ASCII alphanumeric or +/// `_`), used to require whole-token boundaries around a matched keyword. +fn is_identifier_byte(byte: u8) -> bool { + byte.is_ascii_alphanumeric() || byte == b'_' } /// Whether a configured `tracking_column` name refers to this result column. @@ -1971,25 +2047,30 @@ mod tests { fn test_query_orders_by_tracking_column_accepts_valid() { assert!(query_orders_by_tracking_column( "SELECT * FROM t WHERE id > {last_offset} ORDER BY id", - "id" + "id", + false )); // Placeholder form, case/whitespace variance, explicit ASC. assert!(query_orders_by_tracking_column( "select * from t order by {tracking_column}", - "updated_at" + "updated_at", + false )); assert!(query_orders_by_tracking_column( "SELECT * FROM t ORDER BY Updated_At ASC", - "updated_at" + "updated_at", + false )); // Table-qualified column, and the outer ORDER BY after a subquery. assert!(query_orders_by_tracking_column( "SELECT * FROM t ORDER BY t.updated_at", - "updated_at" + "updated_at", + false )); assert!(query_orders_by_tracking_column( "SELECT * FROM (SELECT * FROM t ORDER BY x) s ORDER BY id", - "id" + "id", + false )); } @@ -1998,31 +2079,120 @@ mod tests { // No ORDER BY. assert!(!query_orders_by_tracking_column( "SELECT * FROM t WHERE id > 0", - "id" + "id", + false )); // Different column. assert!(!query_orders_by_tracking_column( "SELECT * FROM t ORDER BY name", - "id" + "id", + false )); // Descending breaks ascending offset advancement. assert!(!query_orders_by_tracking_column( "SELECT * FROM t ORDER BY updated_at DESC", - "updated_at" + "updated_at", + false )); // Substring-only match must not pass (id is a substring of valid_flag/id_backup). assert!(!query_orders_by_tracking_column( "SELECT * FROM t ORDER BY valid_flag", - "id" + "id", + false )); assert!(!query_orders_by_tracking_column( "SELECT * FROM t ORDER BY id_backup", - "id" + "id", + false )); // Tracking column not the primary (first) ordering term. assert!(!query_orders_by_tracking_column( "SELECT * FROM t ORDER BY name, id", - "id" + "id", + false + )); + } + + #[test] + fn test_query_orders_by_tracking_column_ignores_window_order_by() { + // A window function's internal ORDER BY orders values within the frame, + // not the emitted ResultSet, so it must not satisfy the outer-ordering + // requirement even though it is the only `order by` in the text. + assert!(!query_orders_by_tracking_column( + "SELECT id, ROW_NUMBER() OVER (ORDER BY id) rn FROM t", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT id, ROW_NUMBER() OVER (ORDER BY id) rn FROM t WHERE id > {last_offset}", + "id", + false + )); + // A window ORDER BY plus a genuine outer ORDER BY is accepted on the outer. + assert!(query_orders_by_tracking_column( + "SELECT id, ROW_NUMBER() OVER (ORDER BY x) rn FROM t ORDER BY id", + "id", + false + )); + // A parenthesis inside a string literal must not shift the depth and hide + // the real outer ORDER BY. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t WHERE note = 'a (b' ORDER BY id", + "id", + false + )); + } + + #[test] + fn test_query_orders_by_tracking_column_snake_case_matches_read_time() { + // snake_case_columns = true: a CamelCase ORDER BY column with a + // snake_cased tracking_column validates, mirroring tracking_column_matches + // so validate-time never rejects a config that reads rows correctly. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY OrderDate", + "order_date", + true + )); + // Without normalization enabled the same pair does not match (the driver + // label would not be snake_cased at read time either). + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY OrderDate", + "order_date", + false + )); + // Raw label match still works regardless of the normalization flag. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY OrderDate", + "orderdate", + true + )); + } + + #[test] + fn test_query_orders_by_tracking_column_strips_identifier_quotes() { + // PostgreSQL preserves case only for quoted identifiers, so a genuinely + // CamelCase column is ordered as `"OrderDate"`; the surrounding quotes + // must not defeat the match against its unquoted driver label. + assert!(query_orders_by_tracking_column( + r#"SELECT * FROM t ORDER BY "OrderDate""#, + "order_date", + true + )); + assert!(query_orders_by_tracking_column( + r#"SELECT * FROM t ORDER BY "updated_at""#, + "updated_at", + false + )); + // MySQL backtick and SQL Server bracket quoting. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY `updated_at`", + "updated_at", + false + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY [updated_at]", + "updated_at", + false )); } From d22fa4eaecd6fbeb95f8896d5c3a54e14ce710d1 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Tue, 4 Aug 2026 21:13:58 +0530 Subject: [PATCH 06/20] fix(connectors): harden JDBC source startup, types, and secret docs open() booted the JVM and ran DriverManager.getConnection synchronously and unbounded, so a single unreachable or slow-DNS database hung the whole connectors runtime at startup (sources open sequentially, and the FFI drives open() via block_on), taking every other source, sink, and the HTTP control API with it. Wrap the JVM init and connection in block_in_place like poll()/close(), and add login_timeout_ms (DriverManager.setLoginTimeout) so a stuck connect fails instead of blocking forever. BIGINT was emitted as a JSON number, silently rounding values above 2^53 in consumers that parse JSON numbers as f64. Emit it as a string, the same as NUMERIC/DECIMAL already do. The incremental cursor comment claimed it "degrades to re-reads rather than skips", which is false when a run of rows sharing one tracking value is split across a setMaxRows batch boundary: the remainder is skipped. Correct the comment and warn at validate time that the tracking column must be unique or batch_size must exceed the largest tie group. Hoist the jni dependency to the workspace so the pin is shared with the upcoming JDBC sink rather than duplicated. Document that SecretString redaction is not end-to-end: the runtime's generic configs/plugin control endpoint and trace-level config logging emit the raw plugin_config for every connector. --- Cargo.toml | 1 + .../connectors/sources/jdbc_source/Cargo.toml | 2 +- core/connectors/sources/jdbc_source/README.md | 12 ++- .../connectors/sources/jdbc_source/src/lib.rs | 100 ++++++++++++++++-- 4 files changed, 104 insertions(+), 11 deletions(-) diff --git a/Cargo.toml b/Cargo.toml index b24e95296d..da5a6d204c 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -207,6 +207,7 @@ iggy_common = { path = "core/common", version = "0.10.3-edge.3" } iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.3.1-edge.1" } indexmap = "2.14.0" integration = { path = "core/integration" } +jni = { version = "0.21", features = ["invocation"] } journal = { path = "core/journal" } js-sys = "0.3" jsonwebtoken = { version = "10.4.0", features = ["rust_crypto"] } diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml index 839b7e424a..686fcdb419 100644 --- a/core/connectors/sources/jdbc_source/Cargo.toml +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -53,7 +53,7 @@ iggy_common = { workspace = true } iggy_connector_sdk = { workspace = true } # JNI for Java interop with invocation support -jni = { version = "0.21", features = ["invocation"] } +jni = { workspace = true } # For sanitizing passwords in logs regex = { workspace = true } diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index ee2e6ada3d..139e3f6f4e 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -200,6 +200,7 @@ topic = "orders" | `initial_offset` | string | No | - | Starting offset value for first poll | | `mode` | string | No | "incremental" | Sync mode: "incremental" or "bulk" (bulk works with ALL databases) | | `connection_timeout_ms` | u64 | No | 5000 | Timeout (ms) for the per-poll `isValid` liveness check; converted to seconds and capped at 5s | +| `login_timeout_ms` | u64 | No | 30000 | Bound on establishing the connection (`DriverManager.setLoginTimeout`); rounded up to whole seconds. Stops an unreachable database from hanging startup | | `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | | `snake_case_columns` | bool | No | false | Convert column names to snake_case | | `include_metadata` | bool | No | true | Include metadata (table, operation, timestamp) | @@ -316,7 +317,7 @@ JDBC SQL types are automatically mapped to JSON: | ---------- | ----------- | ------- | | BIT, BOOLEAN | boolean | - | | TINYINT, SMALLINT, INTEGER | number | Integer | -| BIGINT | number | Long integer (values above 2^53 may lose precision in JSON consumers that parse numbers as f64) | +| BIGINT | string | Emitted as a string to preserve full 64-bit precision (many JSON consumers parse numbers as f64 and would lose precision above 2^53) | | FLOAT, REAL | number | Float | | DOUBLE | number | Double | | NUMERIC, DECIMAL | string | Emitted as a string to preserve arbitrary precision (e.g. money) | @@ -365,6 +366,15 @@ JDBC SQL types are automatically mapped to JSON: so the password is copied onto the JVM heap as an ordinary (non-zeroed) string for the lifetime of the connection. This is inherent to the JDBC surface and is an accepted risk. +- **`SecretString` redaction is not end-to-end.** `jdbc_url`/`password` are typed + as `SecretString`, and this connector's `Debug` output and startup logs redact + them. That redaction does **not** cover the connectors runtime's generic + config surface: the `GET /sources/{key}/configs/plugin` control-API endpoint + and the runtime's `trace`-level config logging emit the raw, untyped + `plugin_config` (credentials included), the same as every other connector. Do + not expose the control API to untrusted callers and do not run the runtime at + `trace` level in production. This is a known runtime-wide limitation shared by + all connectors, not something this connector can address on its own. ### Credential precedence diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 16e3b9e12c..890b33c789 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -212,12 +212,24 @@ pub struct JdbcSourceConfig { /// establishment. #[serde(default = "default_connection_timeout")] pub connection_timeout_ms: u64, + + /// Bound on establishing the JDBC connection (default: 30000). Applied via + /// `DriverManager.setLoginTimeout`, which is expressed in whole seconds, so + /// the value is rounded up to at least 1s. Prevents an unreachable or + /// slow-DNS database from hanging `open()` (and thus the whole connectors + /// runtime, which opens sources sequentially at startup) indefinitely. + #[serde(default = "default_login_timeout")] + pub login_timeout_ms: u64, } fn default_connection_timeout() -> u64 { 5000 } +fn default_login_timeout() -> u64 { + 30000 +} + fn default_batch_size() -> u32 { 1000 } @@ -250,6 +262,7 @@ impl std::fmt::Debug for JdbcSourceConfig { .field("snake_case_columns", &self.snake_case_columns) .field("include_metadata", &self.include_metadata) .field("connection_timeout_ms", &self.connection_timeout_ms) + .field("login_timeout_ms", &self.login_timeout_ms) .finish() } } @@ -492,6 +505,27 @@ impl JdbcSource { "Failed to create JDBC URL string" ); + // Bound connection establishment so an unreachable or slow-DNS database + // fails instead of hanging open() (which the runtime drives sequentially + // at startup) forever. setLoginTimeout is process-wide and in whole + // seconds, so round the configured milliseconds up to at least 1s. + let login_timeout_secs = self + .config + .login_timeout_ms + .div_ceil(1000) + .clamp(1, i32::MAX as u64) as i32; + jni_init!( + env, + env.call_static_method( + &driver_manager, + "setLoginTimeout", + "(I)V", + &[JValue::Int(login_timeout_secs)], + ) + .and_then(|v| v.v()), + "Failed to set JDBC login timeout" + ); + // If username/password are provided separately, use 3-arg getConnection let connection_obj = if let (Some(username), Some(password)) = (&self.config.username, &self.config.password) @@ -829,9 +863,16 @@ impl JdbcSource { // arrive in ascending tracking order (validate_config enforces // ORDER BY the tracking column ascending), so the last row is the // high-water mark. Using the last row rather than a Rust-side max - // keeps the cursor consistent with the database's own ordering, and - // degrades to re-reads (safe, at-least-once) rather than skips if the - // ordering is ever imperfect. + // keeps the cursor consistent with the database's own ordering. + // + // The next poll resumes with a strict `> last_offset`, so if this + // setMaxRows-capped batch ends in the middle of a run of rows sharing + // one tracking value (a non-unique column such as a timestamp), the + // remaining tied rows are skipped. The tracking column must therefore + // be unique / strictly increasing, or batch_size must exceed the + // largest group of equal values. See the README tracking-column + // requirements; keyset pagination with a tie-break is a planned + // follow-up. if let Some(offset) = offset { last_offset = Some(offset); } @@ -1112,6 +1153,9 @@ impl JdbcSource { ); self.null_or(env, result_set, serde_json::json!(value)) } + // BIGINT is emitted as a string (like NUMERIC/DECIMAL below) so a + // value above 2^53 is not silently rounded by a JSON consumer that + // parses numbers as f64. Types::BIGINT => { let value = jni!( env, @@ -1119,7 +1163,7 @@ impl JdbcSource { .and_then(|v| v.j()), "Failed to get long" ); - self.null_or(env, result_set, serde_json::json!(value)) + self.null_or(env, result_set, serde_json::json!(value.to_string())) } Types::FLOAT | Types::REAL => { let value = jni!( @@ -1332,6 +1376,19 @@ impl JdbcSource { `ORDER BY {{tracking_column}}`) to the query" ))); } + + // The cursor advances with a strict `> last_offset` per batch, so a + // group of rows sharing one tracking value that is split across a + // batch_size boundary loses its remainder. This cannot be detected + // without inspecting the data, so warn: the column should be unique / + // strictly increasing, or batch_size must exceed the largest tie group. + warn!( + "JDBC source [{}] incremental tracking_column '{tracking_column}': ensure it is \ + unique / strictly increasing, or that batch_size ({}) exceeds the largest group \ + of rows sharing one value. A tie split across a batch boundary skips the \ + remaining rows (common for non-unique timestamp columns).", + self.id, self.config.batch_size + ); } // Dry-run the query build so an unresolved placeholder or an invalid @@ -1356,11 +1413,15 @@ impl Source for JdbcSource { // Fail fast on bad config before starting the JVM or opening a connection. self.validate_config()?; - // Initialize JVM - self.initialize_jvm()?; - - // Create database connection - self.create_connection()?; + // JVM boot and DriverManager.getConnection are blocking JNI work. Run + // them via block_in_place (like poll()/close()) so they do not monopolize + // a shared async-runtime worker; the login timeout set in the connection + // path bounds how long a stuck connect can block. + tokio::task::block_in_place(|| -> Result<(), Error> { + self.initialize_jvm()?; + self.create_connection()?; + Ok(()) + })?; info!("JDBC source connector [{}] opened successfully", self.id); Ok(()) @@ -1836,6 +1897,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, } } @@ -2386,6 +2448,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); @@ -2428,6 +2491,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); let state = State { @@ -2462,6 +2526,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); // No last_offset and no initial_offset: the WHERE predicate is dropped, @@ -2496,6 +2561,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); assert!(source.build_query(&State::default()).is_err()); @@ -2519,6 +2585,7 @@ mod tests { include_metadata: false, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); let state = State::default(); @@ -2554,6 +2621,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, Some(connector_state)); let state = source.state.lock().unwrap(); @@ -2743,6 +2811,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, Some(connector_state)); let state = source.state.lock().unwrap(); @@ -2771,6 +2840,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, Some(connector_state)); let state = source.state.lock().unwrap(); @@ -2796,6 +2866,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); let state = source.state.lock().unwrap(); @@ -2821,6 +2892,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); let state = source.state.lock().unwrap(); @@ -2850,6 +2922,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); @@ -2879,6 +2952,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); @@ -2911,6 +2985,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); @@ -2940,6 +3015,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); @@ -2967,6 +3043,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); @@ -2997,6 +3074,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); let state = State { @@ -3034,6 +3112,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); assert!(source.build_query(&State::default()).is_err()); @@ -3057,6 +3136,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let source = JdbcSource::new(1, config, None); let state = State { @@ -3119,6 +3199,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let debug_output = format!("{:?}", config); @@ -3157,6 +3238,7 @@ mod tests { include_metadata: true, jvm_options: vec![], connection_timeout_ms: 30000, + login_timeout_ms: 30000, }; let debug_output = format!("{:?}", config); From 3e0dfa496f784c04633f2eda7e6361f9a2e2ed0d Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Tue, 4 Aug 2026 21:33:26 +0530 Subject: [PATCH 07/20] docs(connectors): correct JDBC source metadata and dedup overclaims A proactive claims-vs-code sweep found two README statements the code does not back. `table_name` is a reserved field hardcoded to null, never derived from the query, yet the metadata was advertised as including the table name; document that it is always null and that operation_type is always SELECT. "Prevents duplicate reads" contradicted the connector's own at-least-once guarantee; reword to "avoids re-reading rows below the tracked offset (at-least-once, not exactly-once)". --- core/connectors/sources/jdbc_source/README.md | 9 ++++++--- 1 file changed, 6 insertions(+), 3 deletions(-) diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index 139e3f6f4e..ab6b81ab88 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -13,7 +13,7 @@ This connector reads data from relational databases using JDBC (Java Database Co - **Bulk Mode**: Re-runs the query each poll for snapshots (capped at `batch_size` rows; see limitations) - **Type Mapping**: Automatic conversion of SQL types to JSON - **Configurable Polling**: Control how frequently data is fetched -- **State Management**: Automatically tracks offsets to prevent duplicate reads +- **State Management**: Tracks offsets so rows below the cursor are not re-read (at-least-once across restarts, not exactly-once) - **Flexible Queries**: Support for custom SQL queries with placeholders ## Supported Databases @@ -203,7 +203,7 @@ topic = "orders" | `login_timeout_ms` | u64 | No | 30000 | Bound on establishing the connection (`DriverManager.setLoginTimeout`); rounded up to whole seconds. Stops an unreachable database from hanging startup | | `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | | `snake_case_columns` | bool | No | false | Convert column names to snake_case | -| `include_metadata` | bool | No | true | Include metadata (table, operation, timestamp) | +| `include_metadata` | bool | No | true | Wrap each row with metadata (operation type, timestamp). `table_name` is a reserved field and is currently always null | ## Query Placeholders @@ -298,6 +298,9 @@ Each database row is converted to a JSON message: } ``` +`table_name` is always `null` today: it is a reserved field, not derived +from the query. `operation_type` is always `"SELECT"`. + ### Without Metadata ```json @@ -538,7 +541,7 @@ query = "SELECT * FROM table WHERE {tracking_column} > {last_offset} ORDER BY {t **Benefits:** -- Prevents duplicate reads +- Avoids re-reading rows below the tracked offset (at-least-once, not exactly-once) - Tracks offset automatically - Efficient for large tables - Works with timestamps, IDs, or any orderable column From c62990483b4b147a1047d9561a2bd8497e3198d8 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Wed, 5 Aug 2026 12:28:37 +0530 Subject: [PATCH 08/20] fix(connectors): address JDBC source review must-fixes Harden outer_order_by() to skip SQL line (--) and block (/* */) comments, so a commented-out ORDER BY can no longer satisfy the incremental ordering requirement and let an unordered query pass validation and skip rows at runtime. Complete JNI exception clearing in throwable_string_method(): a Java exception raised by getMessage()/getSQLState() or the string conversion is now cleared before returning, so the next JNI call on the thread is not aborted for running with an exception pending. Drop the Serialize derive (and the serialize_secret helpers) from JdbcSourceConfig: the runtime only deserializes it, and serializing would risk writing the jdbc_url/password SecretStrings in plaintext. This also removes the now-unused iggy_common dependency. Redacted logging still goes through the manual Debug impl. Read column names with ResultSetMetaData.getColumnLabel() instead of getColumnName() so a SELECT expr AS alias yields the alias the caller configured rather than the base-table column (or an empty string). Remove the unused connectors_api_address() test helper that failed integration clippy, and make the JDBC integration tests panic on a Postgres/testcontainers setup failure instead of returning early and reporting a false pass. --- Cargo.lock | 1 - .../connectors/sources/jdbc_source/Cargo.toml | 3 - .../connectors/sources/jdbc_source/src/lib.rs | 98 ++++++++++++++++--- .../connectors/jdbc/test_with_postgres.rs | 35 ++----- core/integration/tests/connectors/mod.rs | 7 -- 5 files changed, 89 insertions(+), 55 deletions(-) diff --git a/Cargo.lock b/Cargo.lock index bc41278d57..76a69133e2 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7072,7 +7072,6 @@ dependencies = [ "chrono", "dashmap", "humantime", - "iggy_common", "iggy_connector_sdk", "jni 0.21.1", "regex", diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml index 686fcdb419..425f6cb554 100644 --- a/core/connectors/sources/jdbc_source/Cargo.toml +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -46,9 +46,6 @@ dashmap = { workspace = true } # For parsing duration strings (poll_interval) humantime = { workspace = true } -# Shared serde helpers (SecretString serialization) -iggy_common = { workspace = true } - # Connector SDK iggy_connector_sdk = { workspace = true } diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 890b33c789..5e9a449cb5 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -17,7 +17,6 @@ use async_trait::async_trait; use chrono::{DateTime, Utc}; -use iggy_common::serde_secret; use iggy_connector_sdk::{ ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source_connector, }; @@ -146,12 +145,16 @@ pub enum Mode { Incremental, } -/// Configuration for JDBC source connector -#[derive(Clone, Deserialize, Serialize)] +/// Configuration for JDBC source connector. +/// +/// Deliberately does NOT derive `Serialize`: the runtime only ever deserializes +/// this from config, and serializing it would risk writing `jdbc_url`/`password` +/// (both `SecretString`) to logs or state in plaintext. Redacted logging goes +/// through the manual `Debug` impl below. +#[derive(Clone, Deserialize)] pub struct JdbcSourceConfig { /// JDBC connection URL (e.g., "jdbc:mysql://localhost:3306/mydb") /// Can include credentials: "jdbc:mysql://localhost:3306/mydb?user=root&password=secret" - #[serde(serialize_with = "serde_secret::serialize_secret")] pub jdbc_url: SecretString, /// JDBC driver class name (e.g., "com.mysql.cj.jdbc.Driver") @@ -165,7 +168,7 @@ pub struct JdbcSourceConfig { pub username: Option, /// Database password (optional if included in jdbc_url) - #[serde(default, serialize_with = "serde_secret::serialize_optional_secret")] + #[serde(default)] pub password: Option, /// SQL query to execute for fetching data @@ -813,7 +816,7 @@ impl JdbcSource { // negative i32 would sign-extend to an enormous usize and abort on alloc. let mut columns = Vec::with_capacity((column_count.max(0) as usize).min(8192)); for i in 1..=column_count { - let col_name = self.get_column_name(env, &metadata, i)?; + let col_name = self.get_column_label(env, &metadata, i)?; let col_type = self.get_column_type(env, &metadata, i)?; columns.push((col_name, col_type)); } @@ -1042,8 +1045,12 @@ impl JdbcSource { finalize_query(query) } - /// Get column name from ResultSetMetaData - fn get_column_name( + /// Get the column label from ResultSetMetaData. Uses `getColumnLabel` (not + /// `getColumnName`) so a `SELECT expr AS alias` yields the alias the caller + /// asked for; `getColumnName` returns the underlying base-table column (or + /// empty for computed columns), which would not match a configured + /// `tracking_column` alias and can be blank. + fn get_column_label( &self, env: &mut JNIEnv, metadata: &JObject, @@ -1053,12 +1060,12 @@ impl JdbcSource { env, env.call_method( metadata, - "getColumnName", + "getColumnLabel", "(I)Ljava/lang/String;", &[JValue::Int(column_index)], ) .and_then(|v| v.l()), - "Failed to get column name" + "Failed to get column label" ); let col_name: String = jni!( @@ -1707,12 +1714,33 @@ fn outer_order_by(query: &str) -> Option<&str> { let mut depth: u32 = 0; let mut in_quote = false; let mut last_end = None; - for i in 0..bytes.len() { + let mut i = 0; + while i < bytes.len() { let byte = bytes[i]; if in_quote { if byte == b'\'' { in_quote = false; } + i += 1; + continue; + } + // Skip SQL comments entirely: an `ORDER BY` (or a stray paren/quote) + // inside a comment must not be mistaken for the outer ordering, which + // would let a query with no real ordering pass validation and then skip + // rows at runtime. + if byte == b'-' && bytes.get(i + 1) == Some(&b'-') { + i += 2; + while i < bytes.len() && bytes[i] != b'\n' { + i += 1; + } + continue; + } + if byte == b'/' && bytes.get(i + 1) == Some(&b'*') { + i += 2; + while i < bytes.len() && !(bytes[i] == b'*' && bytes.get(i + 1) == Some(&b'/')) { + i += 1; + } + i += 2; // consume the closing `*/` continue; } match byte { @@ -1730,6 +1758,7 @@ fn outer_order_by(query: &str) -> Option<&str> { } _ => {} } + i += 1; } last_end.map(|end| query[end..].trim_start()) } @@ -1823,20 +1852,34 @@ fn take_pending_sql_exception(env: &mut JNIEnv) -> (Option, String) { } /// Call a no-arg `String`-returning method on a throwable; None on JNI error/null. +/// Any pending exception raised by the call itself is cleared before returning, +/// so a later JNI call on this thread is not aborted for running with an +/// exception pending (JNI forbids that). fn throwable_string_method( env: &mut JNIEnv, throwable: &JThrowable, method: &str, ) -> Option { - let obj = env + let obj = match env .call_method(throwable, method, "()Ljava/lang/String;", &[]) - .ok()? - .l() - .ok()?; + .and_then(|v| v.l()) + { + Ok(obj) => obj, + Err(_) => { + clear_pending_exception(env); + return None; + } + }; if obj.is_null() { return None; } - env.get_string(&JString::from(obj)).ok().map(|s| s.into()) + match env.get_string(&JString::from(obj)) { + Ok(s) => Some(s.into()), + Err(_) => { + clear_pending_exception(env); + None + } + } } /// JDBC SQL Types constants @@ -2203,6 +2246,29 @@ mod tests { "id", false )); + // An ORDER BY inside a SQL comment must NOT satisfy the requirement: the + // real query has no ordering, so accepting it would skip rows at runtime. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t -- ORDER BY id\n", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t /* ORDER BY id */", + "id", + false + )); + // A comment before the genuine outer ORDER BY must not hide it. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t /* pick a key */ ORDER BY id", + "id", + false + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM t -- note\n ORDER BY id", + "id", + false + )); } #[test] diff --git a/core/integration/tests/connectors/jdbc/test_with_postgres.rs b/core/integration/tests/connectors/jdbc/test_with_postgres.rs index 6004bf80fe..2e4c226f39 100644 --- a/core/integration/tests/connectors/jdbc/test_with_postgres.rs +++ b/core/integration/tests/connectors/jdbc/test_with_postgres.rs @@ -219,10 +219,7 @@ async fn setup_jdbc_postgres_source( async fn bulk_query_produces_message_to_iggy() { let (_postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {}", e); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; let query = "SELECT 1 as id, 'test' as name"; @@ -271,10 +268,7 @@ async fn bulk_query_produces_message_to_iggy() { async fn bulk_query_produces_multiple_rows_to_iggy() { let (postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {}", e); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; // Use a multi-row SELECT to simulate table data without needing DDL @@ -331,10 +325,7 @@ async fn bulk_query_produces_multiple_rows_to_iggy() { async fn source_includes_metadata_fields_when_enabled() { let (_postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {}", e); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; let query = "SELECT 42 as value"; @@ -395,10 +386,7 @@ fn collect_ids(messages: &[serde_json::Value]) -> Vec { async fn incremental_mode_advances_offset_across_polls() { let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {e}"); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; // Seed a real table BEFORE the source starts polling. @@ -454,10 +442,7 @@ async fn incremental_mode_advances_offset_across_polls() { async fn large_result_set_streams_without_crashing() { let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {e}"); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; let pool = PgPoolOptions::new() @@ -524,10 +509,7 @@ async fn large_result_set_streams_without_crashing() { async fn source_recovers_after_repeated_query_errors() { let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {e}"); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; // Start the source against a table that does not exist yet: every poll @@ -574,10 +556,7 @@ async fn source_recovers_after_repeated_query_errors() { async fn bulk_result_larger_than_batch_size_fails_closed() { let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { Ok(result) => result, - Err(e) => { - eprintln!("Skipping test: Failed to setup Postgres: {e}"); - return; - } + Err(e) => panic!("Failed to set up Postgres container: {e}"), }; let pool = PgPoolOptions::new() diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index 59f6126178..4859397730 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -244,11 +244,4 @@ impl ConnectorsRuntime { .await .expect("Failed to create root TCP client") } - - pub fn connectors_api_address(&self) -> Option { - self.harness - .server() - .connectors_runtime() - .map(|cr| cr.http_address().to_string()) - } } From 6187707a1131445632fb2c7498cda54dda4767ef Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Wed, 5 Aug 2026 17:35:24 +0530 Subject: [PATCH 09/20] fix(connectors): fix JDBC CI lint failures (typos, TOML format) Correct "unparseable" -> "unparsable" in poll_interval doc comments and code comments (flagged by the typos check), and reflow jdbc_oracle.toml to taplo's canonical format (a stray blank line). --- .../runtime/example_config/connectors/jdbc_oracle.toml | 1 - core/connectors/sources/jdbc_source/src/lib.rs | 8 ++++---- 2 files changed, 4 insertions(+), 5 deletions(-) diff --git a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml index 92d5a1e99e..1610ddf7b1 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml @@ -70,7 +70,6 @@ snake_case_columns = true # Include metadata wrapper in output messages include_metadata = true - # Custom JVM options (optional) jvm_options = ["-Xmx512m", "-Xms256m"] diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 5e9a449cb5..2d1165b8bb 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -119,8 +119,8 @@ const CONNECTOR_NAME: &str = "JDBC source"; const DEFAULT_POLL_INTERVAL: Duration = Duration::from_secs(5); /// Parse the configured `poll_interval` humantime string into a `Duration`, -/// falling back to [`DEFAULT_POLL_INTERVAL`] when unset, empty, or unparseable. -/// `validate_config` separately rejects a set-but-unparseable value so a typo +/// falling back to [`DEFAULT_POLL_INTERVAL`] when unset, empty, or unparsable. +/// `validate_config` separately rejects a set-but-unparsable value so a typo /// surfaces at `open()` rather than silently defaulting. fn parse_poll_interval(poll_interval: Option<&str>) -> Duration { poll_interval @@ -1315,7 +1315,7 @@ impl JdbcSource { ))); } - // A set poll_interval must be a valid humantime string; an unparseable + // A set poll_interval must be a valid humantime string; an unparsable // value would otherwise silently fall back to the default. if let Some(value) = self.config.poll_interval.as_deref() && !value.trim().is_empty() @@ -1956,7 +1956,7 @@ mod tests { fn test_parse_poll_interval() { assert_eq!(parse_poll_interval(Some("30s")), Duration::from_secs(30)); assert_eq!(parse_poll_interval(Some("5m")), Duration::from_secs(300)); - // Unset, empty, and unparseable all fall back to the default. + // Unset, empty, and unparsable all fall back to the default. assert_eq!(parse_poll_interval(None), DEFAULT_POLL_INTERVAL); assert_eq!(parse_poll_interval(Some(" ")), DEFAULT_POLL_INTERVAL); assert_eq!( From 850838fc2732e648b938f09eed816564644c3ab4 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Wed, 5 Aug 2026 17:44:45 +0530 Subject: [PATCH 10/20] fix(integration): integrity-check the JDBC driver download The JDBC Postgres tests downloaded the driver JAR at runtime and cached it without validating it. When the download returned a truncated body or an error page (seen in CI), the file was still cached and put on the JVM classpath; the JVM booted fine but Class.forName("org.postgresql.Driver") then failed, so the connector never initialized and every test expecting messages saw zero, while the bad jar was reused by every later test. Download from Maven Central, verify the bytes are a real JAR (ZIP magic plus a minimum size) before persisting, and write via a temp file and atomic rename so a partial or corrupt download can never be cached. A previously cached invalid jar now self-heals, and a genuinely bad download fails setup with a clear message instead of a later opaque ClassNotFoundException. --- .../connectors/jdbc/test_with_postgres.rs | 65 ++++++++++++++----- 1 file changed, 48 insertions(+), 17 deletions(-) diff --git a/core/integration/tests/connectors/jdbc/test_with_postgres.rs b/core/integration/tests/connectors/jdbc/test_with_postgres.rs index 2e4c226f39..fcf8cd20a0 100644 --- a/core/integration/tests/connectors/jdbc/test_with_postgres.rs +++ b/core/integration/tests/connectors/jdbc/test_with_postgres.rs @@ -52,38 +52,69 @@ async fn setup_postgres_container() Ok((postgres, jdbc_url, postgres_jar)) } -/// Get PostgreSQL JDBC driver, downloading if necessary +/// A file that starts with the ZIP local-file-header magic (`PK\x03\x04`) and is +/// at least this large is treated as a real driver JAR. A truncated download or +/// an HTML error page fails this check, so it is never cached or handed to the +/// JVM (where it would only surface later as a `ClassNotFoundException` on +/// `Class.forName`, with the bad file silently reused by every later test). +const MIN_DRIVER_JAR_BYTES: usize = 500_000; + +fn looks_like_jar(bytes: &[u8]) -> bool { + bytes.len() >= MIN_DRIVER_JAR_BYTES && bytes.starts_with(b"PK\x03\x04") +} + +/// Get the PostgreSQL JDBC driver, downloading and integrity-checking it if a +/// valid copy is not already cached. Downloads from Maven Central, verifies the +/// bytes are a real JAR before persisting, and writes via a temp file + atomic +/// rename so a partial or corrupt download can never be cached and reused. async fn get_postgres_driver_jar() -> Result> { let target_dir = std::env::var("CARGO_TARGET_DIR").unwrap_or_else(|_| "target".to_string()); - let jdbc_test_dir = format!("{}/test-jdbc-drivers", target_dir); - let jar_path = format!("{}/postgresql-42.7.1.jar", jdbc_test_dir); + let jdbc_test_dir = format!("{target_dir}/test-jdbc-drivers"); + let jar_path = format!("{jdbc_test_dir}/postgresql-42.7.1.jar"); std::fs::create_dir_all(&jdbc_test_dir)?; + // Reuse the cached jar only if it is actually a valid JAR; a previously + // cached bad download must self-heal rather than fail every run. if std::path::Path::new(&jar_path).exists() { - info!("PostgreSQL JDBC driver found at {}", jar_path); - let absolute_path = std::fs::canonicalize(&jar_path)? - .to_string_lossy() - .to_string(); - return Ok(absolute_path); + match std::fs::read(&jar_path) { + Ok(bytes) if looks_like_jar(&bytes) => { + info!("PostgreSQL JDBC driver found at {jar_path}"); + return Ok(std::fs::canonicalize(&jar_path)?.to_string_lossy().to_string()); + } + _ => { + info!("Cached JDBC driver at {jar_path} is invalid; re-downloading"); + let _ = std::fs::remove_file(&jar_path); + } + } } info!("Downloading PostgreSQL JDBC driver..."); - let download_url = "https://jdbc.postgresql.org/download/postgresql-42.7.1.jar"; - + let download_url = + "https://repo1.maven.org/maven2/org/postgresql/postgresql/42.7.1/postgresql-42.7.1.jar"; let response = reqwest::get(download_url).await?; if !response.status().is_success() { return Err(format!("Failed to download driver: HTTP {}", response.status()).into()); } - let bytes = response.bytes().await?; - std::fs::write(&jar_path, bytes)?; + if !looks_like_jar(&bytes) { + return Err(format!( + "Downloaded JDBC driver is not a valid JAR ({} bytes, magic {:02x?}); \ + the download endpoint may have returned an error page", + bytes.len(), + bytes.get(..4).unwrap_or(&[]) + ) + .into()); + } + + // Write to a unique temp file, then atomically rename, so a crash or a + // concurrent test never observes a half-written jar at `jar_path`. + let tmp_path = format!("{jar_path}.{}.tmp", std::process::id()); + std::fs::write(&tmp_path, &bytes)?; + std::fs::rename(&tmp_path, &jar_path)?; - info!("PostgreSQL JDBC driver downloaded to {}", jar_path); - let absolute_path = std::fs::canonicalize(&jar_path)? - .to_string_lossy() - .to_string(); - Ok(absolute_path) + info!("PostgreSQL JDBC driver downloaded to {jar_path}"); + Ok(std::fs::canonicalize(&jar_path)?.to_string_lossy().to_string()) } /// Build the environment variables for a JDBC Postgres source connector. From 487024c2beb28f8b1a5826487750cbf28105df33 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Wed, 5 Aug 2026 20:15:04 +0530 Subject: [PATCH 11/20] style(integration): rustfmt the JDBC driver-download helper --- .../tests/connectors/jdbc/test_with_postgres.rs | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/core/integration/tests/connectors/jdbc/test_with_postgres.rs b/core/integration/tests/connectors/jdbc/test_with_postgres.rs index fcf8cd20a0..ffb3701db2 100644 --- a/core/integration/tests/connectors/jdbc/test_with_postgres.rs +++ b/core/integration/tests/connectors/jdbc/test_with_postgres.rs @@ -80,7 +80,9 @@ async fn get_postgres_driver_jar() -> Result { info!("PostgreSQL JDBC driver found at {jar_path}"); - return Ok(std::fs::canonicalize(&jar_path)?.to_string_lossy().to_string()); + return Ok(std::fs::canonicalize(&jar_path)? + .to_string_lossy() + .to_string()); } _ => { info!("Cached JDBC driver at {jar_path} is invalid; re-downloading"); @@ -114,7 +116,9 @@ async fn get_postgres_driver_jar() -> Result Date: Tue, 25 Aug 2026 16:04:36 +0530 Subject: [PATCH 12/20] fix(connectors): gate the JDBC source cursor on delivery acknowledgement The incremental cursor advanced as soon as rows were fetched, so a batch whose send later failed was skipped permanently: the next poll queried past it, and its further-advanced offset then overwrote the checkpoint, removing any chance of replay. The connector could not fix this before, because the runtime handed a source its state only at open() with no per-batch delivery signal. Upstream now supplies that signal. Stage the advanced cursor in poll() and resolve it in on_batch_result: Ack (batch sent and checkpoint persisted) commits it, Nack discards it so the same range is re-polled. That makes at-least-once hold for an ordinary in-process send failure, not just across restarts, and follows the staging pattern documented for sources and implemented by random_source. An empty poll now checkpoints nothing rather than rewriting an unchanged cursor, which on the HTTP state backend costs a round-trip per interval. Also repair the shared connectors test harness for the create_topic and send_messages signature changes that came with master, and rename the JDBC integration test file to the sibling convention. --- core/connectors/sources/jdbc_source/README.md | 19 +- .../connectors/sources/jdbc_source/src/lib.rs | 251 ++++++++++++++---- .../{test_with_postgres.rs => jdbc_source.rs} | 0 core/integration/tests/connectors/jdbc/mod.rs | 6 +- core/integration/tests/connectors/mod.rs | 20 +- 5 files changed, 226 insertions(+), 70 deletions(-) rename core/integration/tests/connectors/jdbc/{test_with_postgres.rs => jdbc_source.rs} (100%) diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index ab6b81ab88..23fc69487e 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -13,7 +13,7 @@ This connector reads data from relational databases using JDBC (Java Database Co - **Bulk Mode**: Re-runs the query each poll for snapshots (capped at `batch_size` rows; see limitations) - **Type Mapping**: Automatic conversion of SQL types to JSON - **Configurable Polling**: Control how frequently data is fetched -- **State Management**: Tracks offsets so rows below the cursor are not re-read (at-least-once across restarts, not exactly-once) +- **State Management**: Tracks offsets so rows below the cursor are not re-read, committing the cursor only once the runtime acknowledges delivery (at-least-once, not exactly-once) - **Flexible Queries**: Support for custom SQL queries with placeholders ## Supported Databases @@ -347,14 +347,15 @@ JDBC SQL types are automatically mapped to JSON: than a batch, raise `batch_size` to cover the full result, or use incremental mode with an ordered `tracking_column`. (Full cross-database OFFSET pagination is a planned follow-up.) -- **Delivery semantics.** State (the incremental offset) is persisted by the - runtime only after a batch is successfully sent, so offsets are **at-least-once - across restarts**: a crash or restart never skips rows. One narrower gap - remains: a *transient in-process send failure without a restart* can skip the - batch it happened on, because the in-memory offset is not rolled back. This - matches the other offset-tracking source connectors and is a runtime-level - limitation (there is no per-poll delivery ack to the connector); a stronger - guarantee is tracked as a follow-up. +- **Delivery semantics: at-least-once.** The offset advanced by a poll is only + *staged*; it is committed after the runtime reports that the batch was both + sent and its checkpoint durably persisted (`SourceBatchResult::Ack`). If either + step fails (`Nack`), the staged offset is discarded and the next poll rebuilds + the same query from the committed offset, so the batch is **re-read rather than + skipped** - for a transient in-process send failure as well as for a crash or + restart. Rows can therefore be delivered more than once (message IDs are + random per poll, so downstream consumers must dedupe on a business key if they + need exactly-once); rows are never silently dropped. - **Connection recovery.** The connection is validated with `Connection.isValid` each poll and transparently re-established (closing the old handle) if it has dropped. The check runs on the shared `block_in_place` worker, so its timeout diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 2d1165b8bb..c2bbcee08b 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -18,7 +18,8 @@ use async_trait::async_trait; use chrono::{DateTime, Utc}; use iggy_connector_sdk::{ - ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source_connector, + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, + source::SourceBatchResult, source_connector, }; use jni::objects::{GlobalRef, JByteArray, JObject, JString, JThrowable, JValue}; use jni::{JNIEnv, JavaVM}; @@ -311,7 +312,13 @@ pub struct JdbcSource { // Behind a Mutex so `poll()` (&self) can transparently re-establish a dead // direct connection without `&mut self`. connection: Mutex>, + // The committed cursor: only ever advanced from `pending_state` once the + // runtime confirms the batch was both sent and its checkpoint persisted. state: Arc>, + // The cursor this in-flight batch would advance to, staged by `poll` and + // resolved by `on_batch_result`. The SDK keeps at most one batch in flight, + // so a single slot is sufficient. + pending_state: Mutex>, // Poll interval parsed once from `config.poll_interval` at construction. poll_interval: Duration, // Scheduled start of the next poll, used to pace polls at a fixed cadence @@ -355,6 +362,7 @@ impl JdbcSource { jvm: None, connection: Mutex::new(None), state: Arc::new(Mutex::new(state)), + pending_state: Mutex::new(None), poll_interval, next_poll_at: Mutex::new(None), } @@ -639,11 +647,20 @@ impl JdbcSource { } } - /// Execute query and fetch results. + /// Execute query and fetch results, returning the messages together with the + /// cursor this batch *would* advance to. /// - /// The mutex is held only briefly: once to read the current offset for - /// query building, and once after the JNI work to write the updated state. - fn execute_query(&self, env: &mut JNIEnv) -> Result, Error> { + /// The candidate cursor is deliberately not written to `self.state` here: the + /// caller stages it and commits it only once the runtime acknowledges that the + /// batch was both sent and its checkpoint persisted. Advancing the in-memory + /// cursor at fetch time would permanently skip a batch whose send later failed. + /// + /// The mutex is held only briefly: once to read the current offset for query + /// building, and once after the JNI work to snapshot the counters. + fn execute_query( + &self, + env: &mut JNIEnv, + ) -> Result<(Vec, Option), Error> { let connection = self.get_connection(env)?; // Read current state snapshot (short lock) @@ -672,21 +689,29 @@ impl JdbcSource { ))); } - // Update state with results (short lock) - { - let mut state = lock_mutex(&self.state, "state")?; - if let Some(offset) = max_offset { - state.last_offset = Some(offset); - } - state.processed_rows += row_count; - state.last_poll_time = Utc::now(); - info!( - "Fetched {} rows, total processed: {}", - row_count, state.processed_rows - ); + // An empty poll moves no cursor, so there is nothing to checkpoint. + // Returning no candidate state keeps an idle table from rewriting an + // unchanged checkpoint every interval, which on the HTTP state backend + // would be a network round-trip per poll. + if row_count == 0 { + return Ok((messages, None)); } - Ok(messages) + // Snapshot the cursor this batch would commit (short lock). + let candidate = { + let state = lock_mutex(&self.state, "state")?; + State { + last_offset: max_offset.or_else(|| state.last_offset.clone()), + processed_rows: state.processed_rows.saturating_add(row_count), + last_poll_time: Utc::now(), + } + }; + info!( + "Fetched {} rows, {} processed once this batch is acknowledged", + row_count, candidate.processed_rows + ); + + Ok((messages, Some(candidate))) } /// Prepare a JDBC statement, execute it, and read all result rows into messages. @@ -1457,37 +1482,49 @@ impl Source for JdbcSource { // block_in_place so it does not monopolize a shared async-runtime worker // while other connectors need to make progress. The connectors runtime is // multi-threaded, which block_in_place requires. - let messages = tokio::task::block_in_place(|| -> Result, Error> { - let jvm = self - .jvm - .as_ref() - .ok_or_else(|| Error::InitError("JVM not initialized".to_string()))?; - let mut env = jvm - .attach_current_thread() - .map_err(|e| Error::InitError(format!("Failed to attach thread: {e}")))?; - // Defensive: clear any exception left pending by a prior failed poll - // on this thread before issuing JNI calls. - clear_pending_exception(&mut env); - // Bound this poll's local references (the connection local ref, the - // query string, and the statement/result-set handles) to a frame - // reclaimed when the poll returns. A tokio worker thread stays - // attached to the JVM across polls (attach_current_thread returns a - // no-detach nested guard once attached), so without this frame those - // per-poll locals accumulate on the thread's top-level frame and - // eventually overflow the JNI local reference table, aborting the JVM. - env.push_local_frame(16) - .map_err(|e| Error::Connection(format!("Failed to push local frame: {e}")))?; - let result = self.execute_query(&mut env); - // SAFETY: execute_query returns only owned Rust data (messages); no - // JNI local reference escapes the frame. - let _ = unsafe { env.pop_local_frame(&JObject::null()) }; - result - })?; - - // Persist state so offsets survive connector restarts - let connector_state = { - let state = lock_mutex(&self.state, "state")?; - ConnectorState::serialize(&*state, CONNECTOR_NAME, self.id) + let (messages, candidate_state) = tokio::task::block_in_place( + || -> Result<(Vec, Option), Error> { + let jvm = self + .jvm + .as_ref() + .ok_or_else(|| Error::InitError("JVM not initialized".to_string()))?; + let mut env = jvm + .attach_current_thread() + .map_err(|e| Error::InitError(format!("Failed to attach thread: {e}")))?; + // Defensive: clear any exception left pending by a prior failed poll + // on this thread before issuing JNI calls. + clear_pending_exception(&mut env); + // Bound this poll's local references (the connection local ref, the + // query string, and the statement/result-set handles) to a frame + // reclaimed when the poll returns. A tokio worker thread stays + // attached to the JVM across polls (attach_current_thread returns a + // no-detach nested guard once attached), so without this frame those + // per-poll locals accumulate on the thread's top-level frame and + // eventually overflow the JNI local reference table, aborting the JVM. + env.push_local_frame(16) + .map_err(|e| Error::Connection(format!("Failed to push local frame: {e}")))?; + let result = self.execute_query(&mut env); + // SAFETY: execute_query returns only owned Rust data (messages and + // the candidate state); no JNI local reference escapes the frame. + let _ = unsafe { env.pop_local_frame(&JObject::null()) }; + result + }, + )?; + + // Stage the advanced cursor rather than committing it. The runtime saves + // the returned state only after the batch is sent, then reports the + // outcome to `on_batch_result`, which commits on Ack and discards on Nack + // so a failed send is re-polled instead of silently skipped. + let connector_state = match candidate_state { + Some(candidate) => { + let serialized = ConnectorState::serialize(&candidate, CONNECTOR_NAME, self.id) + .ok_or_else(|| { + Error::Serialization("failed to serialize JDBC source state".to_string()) + })?; + *lock_mutex(&self.pending_state, "pending_state")? = Some(candidate); + Some(serialized) + } + None => None, }; Ok(ProducedMessages { @@ -1497,6 +1534,34 @@ impl Source for JdbcSource { }) } + /// Resolve the cursor staged by the last `poll`. + /// + /// `Ack` means the runtime both sent the batch and durably persisted its + /// checkpoint, so the staged cursor becomes the committed one. `Nack` means + /// neither happened, so the staged cursor is dropped and the next poll rebuilds + /// the same query from the committed offset. That re-reads the batch (delivery + /// stays at-least-once) instead of skipping it, which is what advancing the + /// cursor at fetch time would have done. + async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + let candidate = lock_mutex(&self.pending_state, "pending_state")?.take(); + let Some(candidate) = candidate else { + return Ok(()); + }; + match result { + SourceBatchResult::Ack => { + *lock_mutex(&self.state, "state")? = candidate; + } + SourceBatchResult::Nack => { + warn!( + "JDBC source [{}] batch was not acknowledged; discarding the staged offset \ + {:?} and re-polling the same range", + self.id, candidate.last_offset + ); + } + } + Ok(()) + } + async fn close(&mut self) -> Result<(), Error> { info!("Closing JDBC source connector [{}]", self.id); @@ -3286,6 +3351,94 @@ mod tests { ); } + // ========================================================================= + // Batch-result (staged cursor) tests + // + // Named after the sibling `random_source` convention, which covers the same + // Ack/Nack contract. + // ========================================================================= + + /// An incremental source whose committed offset starts at `committed`. + fn incremental_source_at(committed: &str) -> JdbcSource { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = + "SELECT id FROM t WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(); + let source = JdbcSource::new(1, config, None); + source.state.lock().expect("state lock").last_offset = Some(committed.to_string()); + source + } + + /// Stage the cursor a fetched batch would advance to, as `poll` does. + fn stage_candidate(source: &JdbcSource, offset: &str, processed_rows: u64) { + *source.pending_state.lock().expect("pending lock") = Some(State { + last_offset: Some(offset.to_string()), + processed_rows, + last_poll_time: Utc::now(), + }); + } + + fn block_on(future: F) -> F::Output { + tokio::runtime::Builder::new_current_thread() + .build() + .expect("build test runtime") + .block_on(future) + } + + #[test] + fn given_ack_when_batch_is_staged_should_commit_candidate_offset() { + let source = incremental_source_at("10"); + stage_candidate(&source, "42", 32); + + block_on(source.on_batch_result(SourceBatchResult::Ack)).expect("ACK should apply"); + + let state = source.state.lock().expect("state lock"); + assert_eq!(state.last_offset, Some("42".to_string())); + assert_eq!(state.processed_rows, 32); + assert!(source.pending_state.lock().expect("pending lock").is_none()); + } + + #[test] + fn given_nack_when_batch_is_staged_should_keep_committed_offset() { + let source = incremental_source_at("10"); + stage_candidate(&source, "42", 32); + + block_on(source.on_batch_result(SourceBatchResult::Nack)).expect("NACK should apply"); + + let state = source.state.lock().expect("state lock"); + // The undelivered batch must not advance the cursor, and the staged value + // must be discarded rather than lingering for a later batch to commit. + assert_eq!(state.last_offset, Some("10".to_string())); + assert_eq!(state.processed_rows, 0); + assert!(source.pending_state.lock().expect("pending lock").is_none()); + } + + #[test] + fn given_nack_when_next_poll_builds_query_should_re_read_the_same_range() { + let source = incremental_source_at("10"); + stage_candidate(&source, "42", 32); + block_on(source.on_batch_result(SourceBatchResult::Nack)).expect("NACK should apply"); + + // The regression this guards: advancing the cursor at fetch time would + // rebuild the next query from '42' and permanently skip rows 11..=42. + let state = source.state.lock().expect("state lock"); + let query = source.build_query(&state).expect("build query"); + assert_eq!(query, "SELECT id FROM t WHERE id > '10' ORDER BY id"); + } + + #[test] + fn given_ack_when_nothing_is_staged_should_leave_committed_state_unchanged() { + let source = incremental_source_at("10"); + // An empty poll stages nothing, so an ACK for it must not disturb the cursor. + block_on(source.on_batch_result(SourceBatchResult::Ack)).expect("ACK should apply"); + + let state = source.state.lock().expect("state lock"); + assert_eq!(state.last_offset, Some("10".to_string())); + assert_eq!(state.processed_rows, 0); + } + #[test] fn test_config_debug_without_password() { let config = JdbcSourceConfig { diff --git a/core/integration/tests/connectors/jdbc/test_with_postgres.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs similarity index 100% rename from core/integration/tests/connectors/jdbc/test_with_postgres.rs rename to core/integration/tests/connectors/jdbc/jdbc_source.rs diff --git a/core/integration/tests/connectors/jdbc/mod.rs b/core/integration/tests/connectors/jdbc/mod.rs index 2d1214308a..0c6129cce8 100644 --- a/core/integration/tests/connectors/jdbc/mod.rs +++ b/core/integration/tests/connectors/jdbc/mod.rs @@ -15,7 +15,5 @@ // specific language governing permissions and limitations // under the License. -// JDBC connector tests. -// Source PostgreSQL tests: test_with_postgres.rs -// Sink PostgreSQL tests: test_sink_with_postgres.rs -mod test_with_postgres; +// JDBC connector tests, exercised against PostgreSQL over the JDBC driver. +mod jdbc_source; diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index 4859397730..7011fc302b 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -41,7 +41,7 @@ use iggy::prelude::{IggyClient, IggyMessage, Partitioning}; use iggy_common::Client; use iggy_common::{ CompressionAlgorithm, IggyExpiry, IggyTimestamp, MaxTopicSize, MessageClient, PolledMessages, - StreamClient, TopicClient, + StreamClient, TopicClient, TopicCreateOptions, }; use integration::harness::{ConnectorsRuntimeConfig, IpAddrKind, TestHarness, TestServerConfig}; use serde::{Deserialize, Serialize}; @@ -63,8 +63,9 @@ fn setup_runtime() -> ConnectorsRuntime { // The harness pre-reserves a fixed TCP port (see PortReserver), // so the server binds a non-zero port. That relies on the server // writing current_config.toml on bind regardless of how the port - // was chosen (see tcp_listener.rs); the harness reads that file to - // discover the bound address before it considers startup complete. + // was chosen (see config_writer::write_current_config); the harness + // reads that file to discover the bound address before it considers + // startup complete. ("IGGY_TCP_ADDRESS".to_owned(), "127.0.0.1:0".to_owned()), ])) .build(), @@ -131,6 +132,7 @@ impl ConnectorsIggyClient { messages, ) .await + .map(|_| ()) } async fn get_messages(&self) -> Result { @@ -203,11 +205,13 @@ impl ConnectorsRuntime { .create_topic( &stream_id, &iggy_setup.topic, - 1, - CompressionAlgorithm::None, - None, - IggyExpiry::ServerDefault, - MaxTopicSize::ServerDefault, + &TopicCreateOptions { + partitions_count: Some(1), + compression_algorithm: Some(CompressionAlgorithm::None), + message_expiry: Some(IggyExpiry::ServerDefault), + max_topic_size: Some(MaxTopicSize::ServerDefault), + ..TopicCreateOptions::default() + }, ) .await .expect("Failed to create topic"); From 349cf86ad5433017c2e267bd81b6da0ff2a3c9e7 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Tue, 25 Aug 2026 19:35:48 +0530 Subject: [PATCH 13/20] fix(connectors): log plugin open failures across the FFI boundary Only a status code crosses the FFI boundary when a plugin is opened, and both containers discarded the error behind `result.is_ok()`. The runtime could then report no more than "plugin initialization failed", so an operator saw a connector silently skipped at startup with nothing to say why, and the plugin's own diagnosis was thrown away. A JDBC source whose driver JAR could not be read cost a full CI investigation for exactly this reason: the connector had already produced "Failed to load driver class 'org.postgresql.Driver'" and it never reached a log. Log the cause before collapsing it to a code, for sinks and sources alike. The signature is unchanged, so pre-built plugins keep working. --- core/connectors/sdk/src/sink.rs | 12 +++++++++++- core/connectors/sdk/src/source.rs | 12 +++++++++++- 2 files changed, 22 insertions(+), 2 deletions(-) diff --git a/core/connectors/sdk/src/sink.rs b/core/connectors/sdk/src/sink.rs index 332f73c4e0..a37dbc9263 100644 --- a/core/connectors/sdk/src/sink.rs +++ b/core/connectors/sdk/src/sink.rs @@ -89,7 +89,17 @@ impl SinkContainer { let result = runtime.block_on(sink.open()); self.id = id; self.sink = Some(sink); - if result.is_ok() { 0 } else { 1 } + // Only a status code crosses the FFI boundary, so log the cause here + // or it is lost: the runtime can then report no more than "plugin + // initialization failed", leaving an operator with a skipped + // connector and nothing to explain why. + match result { + Ok(()) => 0, + Err(error) => { + error!("Failed to open sink connector with ID: {id}. {error}"); + 1 + } + } } } diff --git a/core/connectors/sdk/src/source.rs b/core/connectors/sdk/src/source.rs index 532d2643f9..82f29ab68e 100644 --- a/core/connectors/sdk/src/source.rs +++ b/core/connectors/sdk/src/source.rs @@ -177,7 +177,17 @@ impl SourceContainer { let result = runtime.block_on(source.open()); self.id = id; self.source = Some(Arc::new(source)); - if result.is_ok() { 0 } else { 1 } + // Only a status code crosses the FFI boundary, so log the cause here + // or it is lost: the runtime can then report no more than "plugin + // initialization failed", leaving an operator with a skipped + // connector and nothing to explain why. + match result { + Ok(()) => 0, + Err(error) => { + error!("Failed to open source connector with ID: {id}. {error}"); + 1 + } + } } } From e251533ad805c6ec1b3e6b64f06bf420c9f29164 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Tue, 25 Aug 2026 19:36:01 +0530 Subject: [PATCH 14/20] fix(integration): reject a truncated JDBC driver JAR The driver-JAR integrity check accepted any file of at least 500 KB starting with the ZIP local-file-header magic. A download truncated past that size still satisfies both conditions while its central directory is gone, so it is not readable as an archive at all. The check's own comment claimed such a download would fail it. That file was then cached and reused by every later JDBC test in the job, and the JVM could not load the driver class from it. `Class.forName` threw, the connector's `open()` failed, and the runtime skipped the source, so tests failed having received no messages rather than pointing at the JAR. One bad download made this deterministic across all four nextest retries and took out every JDBC test sharing the runner. Verify instead that the bytes open as an archive containing `org/postgresql/Driver.class`, which is the property the JVM needs; a cached file that fails is already deleted and re-downloaded. Also stop comparing the recovery assertion against the raw message list. Bulk mode re-runs its query every poll interval, so it re-delivers the same rows indefinitely and two batches arriving in one client poll turned a healthy run into an id mismatch. Match on the distinct ids seen, and report the received count so "nothing was delivered" is distinguishable from "delivered without the expected data.id". --- .../tests/connectors/jdbc/jdbc_source.rs | 105 ++++++++++++++---- 1 file changed, 86 insertions(+), 19 deletions(-) diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs index ffb3701db2..99c183d89c 100644 --- a/core/integration/tests/connectors/jdbc/jdbc_source.rs +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -18,13 +18,15 @@ use crate::connectors::{ConnectorsRuntime, IggySetup, setup_runtime}; use serial_test::serial; use sqlx::postgres::PgPoolOptions; -use std::collections::HashMap; -use std::time::Duration; +use std::collections::{BTreeSet, HashMap}; +use std::io::Cursor; +use std::time::{Duration, Instant}; use testcontainers_modules::postgres::Postgres; use testcontainers_modules::testcontainers::ContainerAsync; use testcontainers_modules::testcontainers::runners::AsyncRunner; use tokio::time::sleep; use tracing::info; +use zip::ZipArchive; const POSTGRES_USER: &str = "postgres"; const POSTGRES_PASSWORD: &str = "postgres"; @@ -52,21 +54,34 @@ async fn setup_postgres_container() Ok((postgres, jdbc_url, postgres_jar)) } -/// A file that starts with the ZIP local-file-header magic (`PK\x03\x04`) and is -/// at least this large is treated as a real driver JAR. A truncated download or -/// an HTML error page fails this check, so it is never cached or handed to the -/// JVM (where it would only surface later as a `ClassNotFoundException` on -/// `Class.forName`, with the bad file silently reused by every later test). -const MIN_DRIVER_JAR_BYTES: usize = 500_000; +/// The class the connector hands to `Class.forName`. Checking that the archive +/// actually contains it is what makes the integrity check below meaningful. +const DRIVER_CLASS_ENTRY: &str = "org/postgresql/Driver.class"; +/// Whether `bytes` is a driver JAR the JVM can actually load from: a readable +/// ZIP archive that contains the driver class. +/// +/// Magic bytes plus a minimum size are not enough. A download truncated past +/// that size still starts with the ZIP local-file-header magic while its central +/// directory is gone, so it passes a size check yet cannot be read as an +/// archive. It is then cached and reused by every later JDBC test in the job, +/// and the only symptom is a `ClassNotFoundException` on `Class.forName` that +/// fails the connector's `open()`, so the runtime skips the source and every +/// JDBC test fails having received no messages at all. Opening the archive and +/// looking the entry up tests the property the JVM needs, and a cached file that +/// fails it is deleted and re-downloaded rather than reused. fn looks_like_jar(bytes: &[u8]) -> bool { - bytes.len() >= MIN_DRIVER_JAR_BYTES && bytes.starts_with(b"PK\x03\x04") + let Ok(mut archive) = ZipArchive::new(Cursor::new(bytes)) else { + return false; + }; + archive.by_name(DRIVER_CLASS_ENTRY).is_ok() } /// Get the PostgreSQL JDBC driver, downloading and integrity-checking it if a /// valid copy is not already cached. Downloads from Maven Central, verifies the -/// bytes are a real JAR before persisting, and writes via a temp file + atomic -/// rename so a partial or corrupt download can never be cached and reused. +/// bytes open as an archive holding the driver class before persisting, and +/// writes via a temp file + atomic rename so a partial or corrupt download can +/// never be cached and reused. async fn get_postgres_driver_jar() -> Result> { let target_dir = std::env::var("CARGO_TARGET_DIR").unwrap_or_else(|_| "target".to_string()); let jdbc_test_dir = format!("{target_dir}/test-jdbc-drivers"); @@ -101,10 +116,9 @@ async fn get_postgres_driver_jar() -> Result Vec { .collect() } +/// Poll until every id in `expected` has been seen, or `timeout` elapses. +/// Returns the distinct ids seen, ascending, plus the raw number of messages +/// received. +/// +/// Bulk mode re-runs its query on every poll interval, so it keeps re-delivering +/// the same rows for as long as the source runs. Matching on the distinct ids +/// seen, rather than on a fixed-length prefix of the received messages, keeps a +/// second re-delivered batch from failing an otherwise healthy run. The returned +/// count separates "the source delivered nothing" from "it delivered messages +/// that did not carry the expected `data.id`", which a bare id list cannot +/// express: `collect_ids` drops unmatched messages silently. +async fn poll_until_ids_seen( + client: &crate::connectors::ConnectorsIggyClient, + expected: &[i64], + timeout: Duration, +) -> (Vec, usize) { + let deadline = Instant::now() + timeout; + let mut seen: BTreeSet = BTreeSet::new(); + let mut received = 0usize; + + loop { + let polled_messages = client + .get_messages() + .await + .expect("Failed to poll messages"); + + for msg in &polled_messages.messages { + received += 1; + if let Ok(value) = serde_json::from_slice::(&msg.payload) + && let Some(id) = value + .get("data") + .and_then(|data| data.get("id")) + .and_then(|id| id.as_i64()) + { + seen.insert(id); + } + } + + if expected.iter().all(|id| seen.contains(id)) { + info!("Saw expected ids {expected:?} in {received} received messages"); + break; + } + + if Instant::now() >= deadline { + break; + } + + sleep(POLL_INTERVAL).await; + } + + (seen.into_iter().collect(), received) +} + /// Test: incremental mode advances its tracking offset across polls; newly /// inserted rows are delivered exactly once and previously read rows are not /// re-delivered. @@ -572,13 +639,13 @@ async fn source_recovers_after_repeated_query_errors() { .await .expect("Failed to insert rows"); - let messages = poll_messages_with_retry(&client, 2).await; - let mut ids = collect_ids(&messages); - ids.sort_unstable(); + let (ids, received) = + poll_until_ids_seen(&client, &[1, 2], POLL_INTERVAL * POLL_ATTEMPTS as u32).await; assert_eq!( ids, vec![1, 2], - "Source must recover after repeated query failures and deliver ids 1,2, got {ids:?}" + "Source must recover after repeated query failures and deliver ids 1,2; \ + got ids {ids:?} from {received} received message(s)" ); } From f36cc8c29ffbe6df87195f9498df9766632f113e Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Wed, 26 Aug 2026 00:57:02 +0530 Subject: [PATCH 15/20] fix(integration): bound JDBC source polls by deadline, not attempts The shared connectors test client polled in batches of 10, and the JDBC collect-until-N loop gave up after 30 attempts. Those multiply into a ceiling: the loop could never observe more than 300 messages however many the source delivered. large_result_set_streams_without_crashing waited for 150 of that 300 and locally spent ~16 attempts draining 10 at a time, so a slower runner exhausted the budget before the count was reached and the test failed on every retry rather than flaking. The five tests that pass are all bulk mode, which re-runs its query each interval and so retries for free; the two that failed were the only ones whose expectations could not absorb a slow runner. Let the caller choose the batch count, request more per poll than any test waits for, and bound the wait with a wall-clock deadline so runner speed costs time instead of correctness. Assert the incremental cursor on the distinct ids seen rather than on a fixed-length prefix of the received messages. The claim under test is unchanged, that the offset advanced past 3, but a re-delivered batch no longer fails it, which matters because the connector's delivery contract is at-least-once. Both panics now report the ids seen and the raw message count, so a future failure distinguishes a stalled source from a replay. --- .../tests/connectors/jdbc/jdbc_source.rs | 84 +++++++++---------- core/integration/tests/connectors/mod.rs | 11 ++- 2 files changed, 50 insertions(+), 45 deletions(-) diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs index 99c183d89c..207ad0fd35 100644 --- a/core/integration/tests/connectors/jdbc/jdbc_source.rs +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -32,10 +32,18 @@ const POSTGRES_USER: &str = "postgres"; const POSTGRES_PASSWORD: &str = "postgres"; const POSTGRES_DB: &str = "postgres"; -/// Maximum number of poll attempts before giving up -const POLL_ATTEMPTS: usize = 30; +/// How long to wait for the source to deliver what a test expects. +/// +/// A deadline, not an attempt count: an attempt count doubles as a cap on the +/// total a collect-until-N loop can ever drain (attempts x batch size), so a +/// test asking for more than a fraction of that cap fails on a slow runner even +/// though the source delivered everything. +const POLL_TIMEOUT: Duration = Duration::from_secs(30); /// Delay between poll attempts const POLL_INTERVAL: Duration = Duration::from_millis(500); +/// Messages requested per poll, kept well above any single expectation below so +/// the deadline alone bounds how long the source may take. +const POLL_BATCH: u32 = 500; /// Setup Postgres container with test data async fn setup_postgres_container() @@ -207,15 +215,18 @@ fn build_jdbc_env( envs } -/// Poll messages from Iggy with retry logic, returning deserialized JSON values. +/// Poll until at least `expected_count` messages have been collected or +/// `POLL_TIMEOUT` elapses, returning them deserialized. async fn poll_messages_with_retry( client: &crate::connectors::ConnectorsIggyClient, expected_count: usize, ) -> Vec { + let deadline = Instant::now() + POLL_TIMEOUT; let mut received: Vec = Vec::new(); - for attempt in 0..POLL_ATTEMPTS { + + loop { let polled_messages = client - .get_messages() + .get_messages(POLL_BATCH) .await .expect("Failed to poll messages"); @@ -226,18 +237,16 @@ async fn poll_messages_with_retry( } if received.len() >= expected_count { - info!( - "Received {} messages after {} attempts", - received.len(), - attempt + 1 - ); + info!("Received {} messages", received.len()); + return received; + } + + if Instant::now() >= deadline { return received; } sleep(POLL_INTERVAL).await; } - - received } /// Setup connector runtime with JDBC source for Postgres @@ -415,29 +424,17 @@ fn pg_sqlx_url(jdbc_url: &str) -> String { format!("postgres://{POSTGRES_USER}:{POSTGRES_PASSWORD}@{host_and_db}") } -/// Collect the `data.id` integer from each polled (metadata-wrapped) message. -fn collect_ids(messages: &[serde_json::Value]) -> Vec { - messages - .iter() - .filter_map(|m| { - m.get("data") - .and_then(|d| d.get("id")) - .and_then(|v| v.as_i64()) - }) - .collect() -} - /// Poll until every id in `expected` has been seen, or `timeout` elapses. /// Returns the distinct ids seen, ascending, plus the raw number of messages /// received. /// -/// Bulk mode re-runs its query on every poll interval, so it keeps re-delivering -/// the same rows for as long as the source runs. Matching on the distinct ids -/// seen, rather than on a fixed-length prefix of the received messages, keeps a -/// second re-delivered batch from failing an otherwise healthy run. The returned -/// count separates "the source delivered nothing" from "it delivered messages -/// that did not carry the expected `data.id`", which a bare id list cannot -/// express: `collect_ids` drops unmatched messages silently. +/// A source can deliver a row more than once: bulk mode re-runs its query every +/// poll interval, and delivery is at-least-once, so a nacked batch is re-read. +/// Matching on the distinct ids seen, rather than on a fixed-length prefix of the +/// received messages, keeps a re-delivered batch from failing an otherwise +/// healthy run. The returned count separates "the source delivered nothing" from +/// "it delivered messages that did not carry the expected `data.id`", which a +/// bare id list cannot express: unmatched messages are dropped silently. async fn poll_until_ids_seen( client: &crate::connectors::ConnectorsIggyClient, expected: &[i64], @@ -449,7 +446,7 @@ async fn poll_until_ids_seen( loop { let polled_messages = client - .get_messages() + .get_messages(POLL_BATCH) .await .expect("Failed to poll messages"); @@ -513,10 +510,13 @@ async fn incremental_mode_advances_offset_across_polls() { .expect("Failed to setup runtime"); // First batch: ids 1..3. - let first = poll_messages_with_retry(&client, 3).await; - let mut first_ids = collect_ids(&first); - first_ids.sort_unstable(); - assert_eq!(first_ids, vec![1, 2, 3], "Expected ids 1,2,3 on first poll"); + let (first_ids, first_received) = poll_until_ids_seen(&client, &[1, 2, 3], POLL_TIMEOUT).await; + assert_eq!( + first_ids, + vec![1, 2, 3], + "Expected ids 1,2,3 on the first poll; got ids {first_ids:?} from {first_received} \ + received message(s)" + ); // Insert more rows; only these (id > last_offset) should arrive next. sqlx::query("INSERT INTO inc_test (id, name) VALUES (4, 'd'), (5, 'e')") @@ -524,13 +524,12 @@ async fn incremental_mode_advances_offset_across_polls() { .await .expect("Failed to insert additional rows"); - let second = poll_messages_with_retry(&client, 2).await; - let mut second_ids = collect_ids(&second); - second_ids.sort_unstable(); + let (second_ids, second_received) = poll_until_ids_seen(&client, &[4, 5], POLL_TIMEOUT).await; assert_eq!( second_ids, vec![4, 5], - "Expected only the new ids 4,5 (offset must have advanced past 3), got {second_ids:?}" + "Expected only the new ids 4,5 (offset must have advanced past 3); got ids \ + {second_ids:?} from {second_received} received message(s)" ); } @@ -639,8 +638,7 @@ async fn source_recovers_after_repeated_query_errors() { .await .expect("Failed to insert rows"); - let (ids, received) = - poll_until_ids_seen(&client, &[1, 2], POLL_INTERVAL * POLL_ATTEMPTS as u32).await; + let (ids, received) = poll_until_ids_seen(&client, &[1, 2], POLL_TIMEOUT).await; assert_eq!( ids, vec![1, 2], @@ -694,7 +692,7 @@ async fn bulk_result_larger_than_batch_size_fails_closed() { // nothing, and in particular never the truncated 2-row subset. sleep(Duration::from_secs(4)).await; let polled = client - .get_messages() + .get_messages(POLL_BATCH) .await .expect("Failed to poll messages"); assert!( diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index 7011fc302b..9dfcd2759f 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -135,7 +135,14 @@ impl ConnectorsIggyClient { .map(|_| ()) } - async fn get_messages(&self) -> Result { + /// Poll up to `count` messages from the configured stream/topic. + /// + /// `count` is the caller's, because it bounds how much a collect-until-N loop + /// can drain per attempt: a batch smaller than what the caller waits for turns + /// the loop's attempt budget into a cap on the total it can ever see, so the + /// test fails on a slow runner even though the source delivered everything. + /// Sibling source tests poll in batches of 100. + async fn get_messages(&self, count: u32) -> Result { self.client .poll_messages( &self.stream.clone().try_into().unwrap(), @@ -143,7 +150,7 @@ impl ConnectorsIggyClient { None, &iggy_common::Consumer::new("test_consumer".try_into().unwrap()), &iggy_common::PollingStrategy::next(), - 10, + count, true, ) .await From ca8cd540bb6cc5c8720c3d1439fc86f24bce19a1 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Wed, 23 Sep 2026 19:21:25 +0530 Subject: [PATCH 16/20] fix(connectors): harden JDBC JNI frame cleanup --- .../connectors/jdbc_oracle.toml | 2 +- core/connectors/sources/jdbc_source/README.md | 20 +-- .../connectors/sources/jdbc_source/src/lib.rs | 128 +++++++++++++----- 3 files changed, 106 insertions(+), 44 deletions(-) diff --git a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml index 1610ddf7b1..9dfae77c54 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml @@ -45,7 +45,7 @@ username = "system" password = "oracle" # SQL query to execute -# Oracle example with ROWNUM or use a numeric/timestamp column +# Use a stable numeric or timestamp cursor; Oracle ROWNUM is not a valid tracking column. query = "SELECT * FROM CUSTOMERS WHERE ID > {last_offset} ORDER BY ID" # How often to poll the database diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index 23fc69487e..ee28ec5c3c 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -347,15 +347,17 @@ JDBC SQL types are automatically mapped to JSON: than a batch, raise `batch_size` to cover the full result, or use incremental mode with an ordered `tracking_column`. (Full cross-database OFFSET pagination is a planned follow-up.) -- **Delivery semantics: at-least-once.** The offset advanced by a poll is only - *staged*; it is committed after the runtime reports that the batch was both - sent and its checkpoint durably persisted (`SourceBatchResult::Ack`). If either - step fails (`Nack`), the staged offset is discarded and the next poll rebuilds - the same query from the committed offset, so the batch is **re-read rather than - skipped** - for a transient in-process send failure as well as for a crash or - restart. Rows can therefore be delivered more than once (message IDs are - random per poll, so downstream consumers must dedupe on a business key if they - need exactly-once); rows are never silently dropped. +- **Fetched-batch delivery is at-least-once.** The offset advanced by a poll is + only *staged*; it is committed after the runtime reports that the batch was + both sent and its checkpoint durably persisted (`SourceBatchResult::Ack`). If + either step fails (`Nack`), the staged offset is discarded and the next poll + rebuilds the same query from the committed offset, so the batch is **re-read + rather than skipped** - for a transient in-process send failure as well as for + a crash or restart. Rows can therefore be delivered more than once (message + IDs are random per poll, so downstream consumers must dedupe on a business key + if they need exactly-once); send or checkpoint failures do not silently drop + an already-fetched batch. The separate tracking-column uniqueness requirement + above still applies while fetching rows from the database. - **Connection recovery.** The connection is validated with `Connection.isValid` each poll and transparently re-established (closing the old handle) if it has dropped. The check runs on the shared `block_in_place` worker, so its timeout diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index c2bbcee08b..3778f517e2 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -41,6 +41,33 @@ fn clear_pending_exception(env: &mut JNIEnv) { let _ = env.exception_clear(); } +/// Reclaim a JNI local frame without masking the operation error that caused the +/// frame to unwind. JNI can leave an exception pending when `PopLocalFrame` +/// fails, so clear before and after the pop attempt. When both the operation and +/// frame cleanup fail, preserve the operation error because it identifies the +/// original failure. +fn finish_local_frame( + env: &mut JNIEnv, + operation_result: Result, + context: &str, + map_error: impl FnOnce(String) -> Error, +) -> Result { + clear_pending_exception(env); + // SAFETY: callers pass a null result because no local reference escapes the + // frame; successful operations return owned Rust data or a `GlobalRef`. + let pop_result = unsafe { env.pop_local_frame(&JObject::null()) }; + match pop_result { + Ok(_) => operation_result, + Err(err) => { + clear_pending_exception(env); + match operation_result { + Ok(_) => Err(map_error(format!("{context}: {err}"))), + Err(operation_error) => Err(operation_error), + } + } + } +} + /// Best-effort `close()` on a JDBC handle used in error/cleanup paths. Clears /// any pending exception first (`close()` is a `CallVoidMethod`, which JNI /// forbids while an exception is pending) and again afterwards in case the @@ -438,13 +465,18 @@ impl JdbcSource { /// accumulate on the caller's frame. The returned handle is a `GlobalRef`, so /// it survives the frame pop. fn create_direct_connection_internal(&self, env: &mut JNIEnv) -> Result { - env.push_local_frame(16) - .map_err(|e| Error::InitError(format!("Failed to push local frame: {e}")))?; + jni_init!( + env, + env.push_local_frame(16), + "Failed to push connection local frame" + ); let result = self.create_direct_connection_inner(env); - // SAFETY: `result` holds only a global reference (or an error); no JNI - // local reference escapes the frame. - let _ = unsafe { env.pop_local_frame(&JObject::null()) }; - result + finish_local_frame( + env, + result, + "Failed to pop connection local frame", + Error::InitError, + ) } fn create_direct_connection_inner(&self, env: &mut JNIEnv) -> Result { @@ -583,9 +615,11 @@ impl JdbcSource { ) }; - let global_ref = env - .new_global_ref(connection_obj) - .map_err(|e| Error::InitError(format!("Failed to create global reference: {e}")))?; + let global_ref = jni_init!( + env, + env.new_global_ref(connection_obj), + "Failed to create global reference" + ); info!("Direct database connection established successfully"); Ok(global_ref) @@ -621,9 +655,11 @@ impl JdbcSource { let conn = guard .as_ref() .ok_or_else(|| Error::Connection("No connection available".to_string()))?; - let local_ref = env - .new_local_ref(conn.as_obj()) - .map_err(|e| Error::Connection(format!("Failed to create local ref: {e}")))?; + let local_ref = jni!( + env, + env.new_local_ref(conn.as_obj()), + "Failed to create connection local reference" + ); Ok(local_ref) } @@ -803,12 +839,18 @@ impl JdbcSource { // metadata object and per-column name references are reclaimed; a very // wide table would otherwise accumulate one local ref per column on the // outer frame for the whole poll. - env.push_local_frame(16) - .map_err(|e| Error::Connection(format!("Failed to push local frame: {}", e)))?; + jni!( + env, + env.push_local_frame(16), + "Failed to push metadata local frame" + ); let result = self.read_column_metadata_inner(env, result_set); - // SAFETY: the returned Vec is owned Rust data; no JNI reference escapes. - let _ = unsafe { env.pop_local_frame(&JObject::null()) }; - result + finish_local_frame( + env, + result, + "Failed to pop metadata local frame", + Error::Connection, + ) } fn read_column_metadata_inner( @@ -879,13 +921,18 @@ impl JdbcSource { // -column local refs (getObject/getString/getBytes results) are // reclaimed every iteration; otherwise a large result set would // overflow the JNI local reference table and abort the JVM. - env.push_local_frame(32) - .map_err(|e| Error::Connection(format!("Failed to push local frame: {}", e)))?; + jni!( + env, + env.push_local_frame(32), + "Failed to push row local frame" + ); let row_result = self.read_single_row(env, result_set, columns); - // SAFETY: `read_single_row` returns only owned Rust data (a JSON map - // and an optional String); no JNI local reference escapes the frame. - let _ = unsafe { env.pop_local_frame(&JObject::null()) }; - let (row_data, offset) = row_result?; + let (row_data, offset) = finish_local_frame( + env, + row_result, + "Failed to pop row local frame", + Error::Connection, + )?; // Take the tracking value of the LAST row as the next offset. Rows // arrive in ascending tracking order (validate_config enforces @@ -1501,13 +1548,18 @@ impl Source for JdbcSource { // no-detach nested guard once attached), so without this frame those // per-poll locals accumulate on the thread's top-level frame and // eventually overflow the JNI local reference table, aborting the JVM. - env.push_local_frame(16) - .map_err(|e| Error::Connection(format!("Failed to push local frame: {e}")))?; + jni!( + env, + env.push_local_frame(16), + "Failed to push poll local frame" + ); let result = self.execute_query(&mut env); - // SAFETY: execute_query returns only owned Rust data (messages and - // the candidate state); no JNI local reference escapes the frame. - let _ = unsafe { env.pop_local_frame(&JObject::null()) }; - result + finish_local_frame( + &mut env, + result, + "Failed to pop poll local frame", + Error::Connection, + ) }, )?; @@ -1899,16 +1951,24 @@ fn classify_query_failure(env: &mut JNIEnv, action: &str) -> Error { fn take_pending_sql_exception(env: &mut JNIEnv) -> (Option, String) { let throwable = match env.exception_occurred() { Ok(t) if !t.is_null() => t, - _ => return (None, "unknown error".to_string()), + Ok(_) => return (None, "unknown error".to_string()), + Err(_) => { + clear_pending_exception(env); + return (None, "unknown error".to_string()); + } }; let _ = env.exception_clear(); let message = throwable_string_method(env, &throwable, "getMessage") .unwrap_or_else(|| "unknown error".to_string()); - let sql_state = if env - .is_instance_of(&throwable, "java/sql/SQLException") - .unwrap_or(false) - { + let is_sql_exception = match env.is_instance_of(&throwable, "java/sql/SQLException") { + Ok(is_sql_exception) => is_sql_exception, + Err(_) => { + clear_pending_exception(env); + false + } + }; + let sql_state = if is_sql_exception { throwable_string_method(env, &throwable, "getSQLState") } else { None From 3cc03f62fb9d6a1e9d0981868e607d6543be61c5 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Thu, 24 Sep 2026 09:47:03 +0530 Subject: [PATCH 17/20] fix(connectors): load JDBC drivers with system class loader --- .../connectors/sources/jdbc_source/src/lib.rs | 124 ++++++++++-------- .../tests/connectors/jdbc/jdbc_source.rs | 21 +++ 2 files changed, 91 insertions(+), 54 deletions(-) diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 3778f517e2..2681b7ca4d 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -41,6 +41,23 @@ fn clear_pending_exception(env: &mut JNIEnv) { let _ = env.exception_clear(); } +/// Take and clear a pending Java exception, returning its class and message via +/// `Throwable.toString()`. JNI's Rust error only reports that Java threw; the +/// throwable carries the diagnostic that distinguishes a missing driver from a +/// linkage or initialization failure. +fn take_pending_java_exception(env: &mut JNIEnv) -> Option { + let throwable = match env.exception_occurred() { + Ok(throwable) if !throwable.is_null() => throwable, + Ok(_) => return None, + Err(_) => { + clear_pending_exception(env); + return None; + } + }; + clear_pending_exception(env); + throwable_string_method(env, &throwable, "toString") +} + /// Reclaim a JNI local frame without masking the operation error that caused the /// frame to unwind. JNI can leave an exception pending when `PopLocalFrame` /// fails, so clear before and after the pop attempt. When both the operation and @@ -86,8 +103,11 @@ macro_rules! jni { match $call { Ok(value) => value, Err(err) => { - clear_pending_exception(&mut *$env); - return Err(Error::Connection(format!("{}: {err}", $ctx))); + let java_exception = take_pending_java_exception(&mut *$env); + let detail = java_exception + .map(|exception| format!("{err}: {exception}")) + .unwrap_or_else(|| err.to_string()); + return Err(Error::Connection(format!("{}: {detail}", $ctx))); } } }; @@ -99,8 +119,11 @@ macro_rules! jni_init { match $call { Ok(value) => value, Err(err) => { - clear_pending_exception(&mut *$env); - return Err(Error::InitError(format!("{}: {err}", $ctx))); + let java_exception = take_pending_java_exception(&mut *$env); + let detail = java_exception + .map(|exception| format!("{err}: {exception}")) + .unwrap_or_else(|| err.to_string()); + return Err(Error::InitError(format!("{}: {detail}", $ctx))); } } }; @@ -416,40 +439,6 @@ impl JdbcSource { .attach_current_thread() .map_err(|e| Error::InitError(format!("Failed to attach thread to JVM: {}", e)))?; - info!("Loading JDBC driver: {}", self.config.driver_class); - - // Load driver using Class.forName() which triggers static initialization - info!( - "Loading driver class via Class.forName: {}", - self.config.driver_class - ); - - let class_class = jni_init!( - env, - env.find_class("java/lang/Class"), - "Failed to find Class" - ); - - let driver_class_name = jni_init!( - env, - env.new_string(&self.config.driver_class), - "Failed to create class name string" - ); - - // Call Class.forName(className) to load and initialize the driver - jni_init!( - env, - env.call_static_method( - class_class, - "forName", - "(Ljava/lang/String;)Ljava/lang/Class;", - &[JValue::Object(&driver_class_name.into())], - ), - format!("Failed to load driver class '{}'", self.config.driver_class) - ); - - info!("JDBC driver loaded and registered successfully"); - info!( "Creating direct JDBC connection to: {}", sanitize_jdbc_url(self.config.jdbc_url.expose_secret()) @@ -467,7 +456,7 @@ impl JdbcSource { fn create_direct_connection_internal(&self, env: &mut JNIEnv) -> Result { jni_init!( env, - env.push_local_frame(16), + env.push_local_frame(24), "Failed to push connection local frame" ); let result = self.create_direct_connection_inner(env); @@ -480,7 +469,9 @@ impl JdbcSource { } fn create_direct_connection_inner(&self, env: &mut JNIEnv) -> Result { - // Set the thread context class loader to help DriverManager find the driver + // Use the system loader explicitly for both driver initialization and the + // thread context. This keeps DriverManager's view of the driver aligned + // with the JVM classpath even when invoked from an attached native thread. let current_thread_class = jni_init!( env, env.find_class("java/lang/Thread"), @@ -499,42 +490,67 @@ impl JdbcSource { "Failed to get current thread" ); - // Get the class loader that loaded the driver - let driver_class = jni_init!( + let class_loader_class = jni_init!( env, - env.find_class(self.config.driver_class.replace('.', "/")), - "Failed to find driver class" + env.find_class("java/lang/ClassLoader"), + "Failed to find ClassLoader" ); - let driver_class_loader = jni_init!( + let system_class_loader = jni_init!( env, - env.call_method( - &driver_class, - "getClassLoader", + env.call_static_method( + class_loader_class, + "getSystemClassLoader", "()Ljava/lang/ClassLoader;", &[], ) .and_then(|v| v.l()), - "Failed to get driver class loader" + "Failed to get system class loader" ); - // Set the context class loader jni_init!( env, env.call_method( ¤t_thread, "setContextClassLoader", "(Ljava/lang/ClassLoader;)V", - &[JValue::Object(&driver_class_loader)], + &[JValue::Object(&system_class_loader)], ), "Failed to set context class loader" ); info!( - "Set thread context class loader for driver: {}", + "Loading JDBC driver via the system class loader: {}", self.config.driver_class ); + let class_class = jni_init!( + env, + env.find_class("java/lang/Class"), + "Failed to find Class" + ); + let driver_class_name = jni_init!( + env, + env.new_string(&self.config.driver_class), + "Failed to create class name string" + ); + jni_init!( + env, + env.call_static_method( + class_class, + "forName", + "(Ljava/lang/String;ZLjava/lang/ClassLoader;)Ljava/lang/Class;", + &[ + JValue::Object(&driver_class_name.into()), + JValue::Bool(1), + JValue::Object(&system_class_loader), + ], + ), + format!("Failed to load driver class '{}'", self.config.driver_class) + ); + + info!("JDBC driver loaded and registered successfully"); + // Get connection from DriverManager let driver_manager = jni_init!( env, @@ -1950,14 +1966,14 @@ fn classify_query_failure(env: &mut JNIEnv, action: &str) -> Error { /// `java.sql.SQLException`) and message. fn take_pending_sql_exception(env: &mut JNIEnv) -> (Option, String) { let throwable = match env.exception_occurred() { - Ok(t) if !t.is_null() => t, + Ok(throwable) if !throwable.is_null() => throwable, Ok(_) => return (None, "unknown error".to_string()), Err(_) => { clear_pending_exception(env); return (None, "unknown error".to_string()); } }; - let _ = env.exception_clear(); + clear_pending_exception(env); let message = throwable_string_method(env, &throwable, "getMessage") .unwrap_or_else(|| "unknown error".to_string()); diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs index 207ad0fd35..dfccc76485 100644 --- a/core/integration/tests/connectors/jdbc/jdbc_source.rs +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -688,6 +688,27 @@ async fn bulk_result_larger_than_batch_size_fails_closed() { .await; let client = runtime.create_client().await; + // Prove the connector opened before checking the fail-closed behavior. A + // startup failure also produces no messages and would otherwise make this + // negative-path test pass for the wrong reason. + let api_url = runtime + .harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let sources: serde_json::Value = reqwest::get(format!("{api_url}/sources")) + .await + .expect("Failed to query source status") + .error_for_status() + .expect("Source status endpoint returned an error") + .json() + .await + .expect("Failed to deserialize source status"); + assert_eq!( + sources[0]["status"], "running", + "JDBC source must be running before exercising bulk fail-closed behavior: {sources}" + ); + // Several poll cycles (poll interval is 1s). A fail-closed source delivers // nothing, and in particular never the truncated 2-row subset. sleep(Duration::from_secs(4)).await; From 81ef6d2aaf390272a8e7e33fc2c2d1f99ab6e916 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Tue, 29 Sep 2026 16:53:51 +0530 Subject: [PATCH 18/20] fix(connectors): address JDBC source review findings --- .../example_config/connectors/jdbc_h2.toml | 5 +- core/connectors/sdk/src/sink.rs | 10 +- core/connectors/sdk/src/source.rs | 11 +- .../connectors/sources/jdbc_source/Cargo.toml | 3 + core/connectors/sources/jdbc_source/README.md | 55 ++- .../connectors/sources/jdbc_source/src/lib.rs | 451 ++++++++++++------ .../tests/connectors/jdbc/jdbc_source.rs | 89 +++- core/integration/tests/connectors/mod.rs | 19 +- 8 files changed, 441 insertions(+), 202 deletions(-) diff --git a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml index 4dd1f0cc6e..a603909cbe 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml @@ -27,8 +27,9 @@ path = "target/release/libiggy_connector_jdbc_source" plugin_config_format = "toml" [plugin_config] -# H2 connection URL (in-memory database) -jdbc_url = "jdbc:h2:mem:testdb;DB_CLOSE_DELAY=-1" +# H2 connection URL (in-memory database). INIT creates and seeds the table used +# by the query so this example runs without a separate setup step. +jdbc_url = "jdbc:h2:mem:testdb;DB_CLOSE_DELAY=-1;INIT=CREATE TABLE IF NOT EXISTS users (id INT PRIMARY KEY AUTO_INCREMENT, name VARCHAR(100), email VARCHAR(100), created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)\\;INSERT INTO users (name, email) SELECT 'User ' || x, 'user' || x || '@test.com' FROM SYSTEM_RANGE(1, 100) WHERE NOT EXISTS (SELECT 1 FROM users)" # H2 JDBC driver driver_class = "org.h2.Driver" diff --git a/core/connectors/sdk/src/sink.rs b/core/connectors/sdk/src/sink.rs index 3edcf1bc92..66ace46d2e 100644 --- a/core/connectors/sdk/src/sink.rs +++ b/core/connectors/sdk/src/sink.rs @@ -91,14 +91,12 @@ impl SinkContainer { let result = runtime.block_on(sink.open()); self.id = id; self.sink = Some(sink); - // Only a status code crosses the FFI boundary, so log the cause here - // or it is lost: the runtime can then report no more than "plugin - // initialization failed", leaving an operator with a skipped - // connector and nothing to explain why. match result { Ok(()) => 0, - Err(error) => { - error!("Failed to open sink connector with ID: {id}. {error}"); + Err(_) => { + // Connector errors may contain secrets from external clients. + // Only the status is safe to log at this generic boundary. + error!("Failed to open sink connector with ID: {id}"); 1 } } diff --git a/core/connectors/sdk/src/source.rs b/core/connectors/sdk/src/source.rs index 072c0fe770..0a9f64aced 100644 --- a/core/connectors/sdk/src/source.rs +++ b/core/connectors/sdk/src/source.rs @@ -187,14 +187,13 @@ impl SourceContainer { let result = runtime.block_on(source.open()); self.id = id; self.source = Some(Arc::new(source)); - // Only a status code crosses the FFI boundary, so log the cause here - // or it is lost: the runtime can then report no more than "plugin - // initialization failed", leaving an operator with a skipped - // connector and nothing to explain why. match result { Ok(()) => 0, - Err(error) => { - error!("Failed to open source connector with ID: {id}. {error}"); + Err(_) => { + // Connector errors may contain secrets from external clients + // (for example a JDBC URL echoed by a driver). Only the status + // is safe to log at this generic boundary. + error!("Failed to open source connector with ID: {id}"); 1 } } diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml index 425f6cb554..38d823459a 100644 --- a/core/connectors/sources/jdbc_source/Cargo.toml +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -63,3 +63,6 @@ uuid = { workspace = true, features = ["v4"] } [dev-dependencies] toml = { workspace = true } + +[lints] +workspace = true diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index ee28ec5c3c..f2ae1eb368 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -90,10 +90,10 @@ password = "secret_password" query = "SELECT * FROM orders WHERE updated_at > {last_offset} ORDER BY updated_at ASC" poll_interval = "30s" batch_size = 1000 -# updated_at is a timestamp and may not be unique. A run of rows sharing one -# timestamp that is split across a batch boundary would skip the remainder (see -# "Unique / strictly increasing" below). Prefer a unique auto-increment key, or -# keep batch_size larger than any same-timestamp group. +# updated_at is a timestamp and may not be unique. If equal timestamps cross a +# batch boundary, the poll fails closed (see "Unique / strictly increasing" +# below). Prefer a unique auto-increment key, or keep batch_size larger than any +# same-timestamp group. tracking_column = "updated_at" initial_offset = "2024-01-01 00:00:00" mode = "incremental" @@ -193,8 +193,8 @@ topic = "orders" | `driver_jar_path` | string | Yes | - | Path to the JDBC driver JAR (checked to exist at startup; passed to the embedded JVM as `-Djava.class.path`) | | `username` | string | No | - | Database username (optional if in jdbc_url) | | `password` | string | No | - | Database password (optional if in jdbc_url) | -| `query` | string | Yes | - | SQL query to execute (supports `{last_offset}` and `{tracking_column}` placeholders) | -| `poll_interval` | string (duration) | No | 5s | How often to poll, as a humantime string (e.g., "30s", "5m", "1h") | +| `query` | string | Yes | - | SQL query to execute. Incremental mode requires `{last_offset}`; `{tracking_column}` is also supported | +| `poll_interval` | string (duration) | No | 5s | Positive polling interval as a humantime string (e.g., "30s", "5m", "1h"); zero is rejected | | `batch_size` | u32 | No | 1000 | Maximum rows to fetch per poll | | `tracking_column` | string | Incremental | - | Column to track for incremental reads (required in incremental mode; the query must also `ORDER BY` it) | | `initial_offset` | string | No | - | Starting offset value for first poll | @@ -217,6 +217,12 @@ connector refuses to start otherwise): - **`tracking_column` is required.** Without it the offset can never advance and every poll re-reads the same rows. +- **The query must contain `{last_offset}`.** Without it each poll would execute + the same query and re-read the first batch while the stored cursor had no + effect. +- **Exactly one result column must match `tracking_column`.** The connector + rejects a missing match and duplicate matching labels. Use explicit, unique + aliases when a query joins tables that expose the same column name. - **The query must order by the tracking column, ascending, as the first `ORDER BY` term.** Row limiting uses `setMaxRows`, so an unordered (or otherwise-ordered) query returns an arbitrary subset; advancing the offset to @@ -242,13 +248,13 @@ as the next cursor, so the cursor always matches the database's own `ORDER BY`. The tracking column must also be: - **Unique / strictly increasing.** The next poll resumes with a strict - `> {last_offset}`, and each batch is capped by `setMaxRows`. If a batch ends in - the middle of a run of rows that share the same tracking value (common for a - non-unique column like a timestamp), the remaining same-value rows are skipped - on the next poll. Use a unique, strictly-increasing key (an auto-increment ID - is ideal). If you must track a non-unique column, ensure `batch_size` exceeds - the largest group of equal values so a tie never spans a batch boundary. - (Keyset pagination with a tie-break is a planned follow-up.) + `> {last_offset}`. The connector probes one row past `batch_size`; if that row + has the same tracking value as the last row in the batch, the entire poll + fails before emitting messages or advancing the checkpoint. This prevents + silent skips, but the source cannot progress until `batch_size` exceeds that + equal-value group or the query uses a unique, strictly-increasing key. An + auto-increment ID is ideal. Keyset pagination with a tie-break is a planned + follow-up. - **Monotonic under the database's own ordering.** Because the cursor is the last ordered row and is fed back as `WHERE {tracking_column} > ''`, the column must increase monotonically under the same ordering the database applies to that @@ -270,8 +276,9 @@ The tracking column must also be: -- Configuration tracking_column = "id" query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" +initial_offset = "0" --- First poll (no offset yet) +-- First poll (initial_offset configured as 0) SELECT * FROM users WHERE id > '0' ORDER BY id -- After processing rows up to id=100 @@ -333,9 +340,11 @@ JDBC SQL types are automatically mapped to JSON: - **Embedded JVM, one per process.** JNI permits a single `JavaVM` per OS process. All JDBC *source* instances in the connectors runtime share one JVM - (the first instance's `jvm_options`/classpath win). A JDBC source and a JDBC - sink are separate shared libraries and **cannot both create a JVM in the same - runtime process** — run them in separate connectors-runtime processes. + and must configure the same `driver_jar_path` and `jvm_options`; a later source + with different values is rejected instead of silently using the first + source's classpath. A JDBC source and a JDBC sink are separate shared libraries + and **cannot both create a JVM in the same runtime process**. Run them in + separate connectors-runtime processes. - **Blocking I/O.** JDBC calls go through JNI and are synchronous. The fetch in `poll()` (and the close in `close()`) runs under `tokio::task::block_in_place` so it does not monopolize a shared async-runtime worker, but the work is still @@ -347,6 +356,10 @@ JDBC SQL types are automatically mapped to JSON: than a batch, raise `batch_size` to cover the full result, or use incremental mode with an ordered `tracking_column`. (Full cross-database OFFSET pagination is a planned follow-up.) +- **Incremental boundaries fail closed on tied tracking values.** The connector + fetches one probe row beyond `batch_size`. If the probe and last in-batch row + share a tracking value, no messages are emitted and no cursor is staged. Raise + `batch_size` above the tie group or use a unique tracking column. - **Fetched-batch delivery is at-least-once.** The offset advanced by a poll is only *staged*; it is committed after the runtime reports that the batch was both sent and its checkpoint durably persisted (`SourceBatchResult::Ack`). If @@ -540,6 +553,7 @@ jdbc_url = "jdbc:h2:file:/data/mydb;USER=sa;PASSWORD=sa" mode = "incremental" tracking_column = "updated_at" # or "id", "created_at", etc. query = "SELECT * FROM table WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" +initial_offset = "2024-01-01 00:00:00" # value appropriate for the column type ``` **Benefits:** @@ -547,15 +561,16 @@ query = "SELECT * FROM table WHERE {tracking_column} > {last_offset} ORDER BY {t - Avoids re-reading rows below the tracked offset (at-least-once, not exactly-once) - Tracks offset automatically - Efficient for large tables -- Works with timestamps, IDs, or any orderable column +- Works with timestamps, IDs, or other orderable, non-null columns; a unique + strictly increasing value avoids fail-closed tie boundaries **Database Examples** (the query must order by the tracking column; a unique, strictly-increasing key like an auto-increment ID is safest, see the tracking column requirements above): -- MySQL: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; ensure `batch_size` exceeds any same-timestamp group, or track a unique id) +- MySQL: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; if `batch_size` splits a same-timestamp group, the poll fails closed until the batch is enlarged or a unique id is tracked) - Oracle: `WHERE id > {last_offset} ORDER BY id` (use a monotonic key; `ROWNUM` is not a valid tracking column) -- SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; same caveat as MySQL) +- SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; same fail-closed boundary behavior as MySQL) - PostgreSQL: `WHERE id > {last_offset} ORDER BY id` ### Bulk Mode (Universal) diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index 2681b7ca4d..aefc958a79 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -26,9 +26,10 @@ use jni::{JNIEnv, JavaVM}; use regex::Regex; use secrecy::{ExposeSecret, SecretString}; use serde::{Deserialize, Serialize}; +use std::path::{Path, PathBuf}; use std::sync::{Arc, Mutex, MutexGuard}; use std::time::{Duration, Instant}; -use tracing::{debug, info, warn}; +use tracing::{debug, error, info, warn}; use uuid::Uuid; /// Clear any pending Java exception on the current thread. The JNI spec forbids @@ -141,11 +142,11 @@ fn lock_mutex<'a, T>(mutex: &'a Mutex, what: &str) -> Result = - std::sync::LazyLock::new(|| Regex::new(r"://([^:]+):([^@?;/]+)@").unwrap()); + std::sync::LazyLock::new(|| Regex::new(r"://([^:/?#]+):([^/?#]*)@").unwrap()); static RE_PASSWORD_PARAM: std::sync::LazyLock = std::sync::LazyLock::new(|| Regex::new(r"(?i)(password|pwd|pass)=([^;&\s]+)").unwrap()); static RE_ORACLE_PASS: std::sync::LazyLock = - std::sync::LazyLock::new(|| Regex::new(r"thin:([^/]+)/([^@]+)@").unwrap()); + std::sync::LazyLock::new(|| Regex::new(r"thin:([^/]+)/([^?#]*)@").unwrap()); /// Regexes that remove the incremental offset predicate `{tracking_column} > /// {last_offset}` on the first (no-offset) poll while preserving any other @@ -322,26 +323,20 @@ impl std::fmt::Debug for JdbcSourceConfig { } /// Internal state tracking for the JDBC source -#[derive(Debug, Clone, Serialize, Deserialize)] +#[derive(Debug, Clone, Default, Serialize, Deserialize)] struct State { /// Last tracked offset value (for incremental mode) last_offset: Option, /// Total rows processed processed_rows: u64, - - /// Last poll timestamp - last_poll_time: DateTime, } -impl Default for State { - fn default() -> Self { - Self { - last_offset: None, - processed_rows: 0, - last_poll_time: Utc::now(), - } - } +#[derive(Debug)] +struct ColumnMetadata { + output_name: String, + sql_type: i32, + is_tracking: bool, } /// Database record structure for output messages @@ -364,7 +359,7 @@ pub struct JdbcSource { connection: Mutex>, // The committed cursor: only ever advanced from `pending_state` once the // runtime confirms the batch was both sent and its checkpoint persisted. - state: Arc>, + state: Mutex, // The cursor this in-flight batch would advance to, staged by `poll` and // resolved by `on_batch_result`. The SDK keeps at most one batch in flight, // so a single slot is sufficient. @@ -390,6 +385,16 @@ fn sanitize_jdbc_url(url: &str) -> String { url.to_string() } +fn sanitize_jdbc_error(error: Error, jdbc_url: &str) -> Error { + let sanitized_url = sanitize_jdbc_url(jdbc_url); + let sanitize = |message: String| message.replace(jdbc_url, &sanitized_url); + match error { + Error::InitError(message) => Error::InitError(sanitize(message)), + Error::Connection(message) => Error::Connection(sanitize(message)), + error => error, + } +} + impl JdbcSource { /// Create a new JDBC source connector pub fn new(id: u32, config: JdbcSourceConfig, connector_state: Option) -> Self { @@ -411,7 +416,7 @@ impl JdbcSource { config, jvm: None, connection: Mutex::new(None), - state: Arc::new(Mutex::new(state)), + state: Mutex::new(state), pending_state: Mutex::new(None), poll_interval, next_poll_at: Mutex::new(None), @@ -459,7 +464,9 @@ impl JdbcSource { env.push_local_frame(24), "Failed to push connection local frame" ); - let result = self.create_direct_connection_inner(env); + let result = self + .create_direct_connection_inner(env) + .map_err(|error| sanitize_jdbc_error(error, self.config.jdbc_url.expose_secret())); finish_local_frame( env, result, @@ -755,7 +762,6 @@ impl JdbcSource { State { last_offset: max_offset.or_else(|| state.last_offset.clone()), processed_rows: state.processed_rows.saturating_add(row_count), - last_poll_time: Utc::now(), } }; info!( @@ -788,15 +794,11 @@ impl JdbcSource { Err(_) => return Err(classify_query_failure(env, "prepare statement")), }; - // Use setMaxRows for database-agnostic row limiting instead of SQL LIMIT clause. - // This works across all JDBC drivers (MySQL, Oracle, SQL Server, H2, etc.). - // In bulk mode fetch one extra row so the caller can detect an oversized - // (truncated) result set and fail closed rather than sync an arbitrary - // subset; incremental mode pages via the offset, so batch_size is the cap. - let max_rows = match self.config.mode { - Mode::Bulk => self.config.batch_size.saturating_add(1), - Mode::Incremental => self.config.batch_size, - }; + // Use setMaxRows for database-agnostic row limiting instead of SQL LIMIT. + // Both modes fetch one probe row beyond the batch. Bulk mode uses it to + // detect unsupported pagination; incremental mode uses it to detect an + // equal tracking value split across the boundary before any row is emitted. + let max_rows = self.config.batch_size.saturating_add(1); if let Err(err) = env.call_method( &statement, "setMaxRows", @@ -850,7 +852,7 @@ impl JdbcSource { &self, env: &mut JNIEnv, result_set: &JObject, - ) -> Result, Error> { + ) -> Result, Error> { // Read all column metadata inside its own JNI local frame so the // metadata object and per-column name references are reclaimed; a very // wide table would otherwise accumulate one local ref per column on the @@ -873,7 +875,7 @@ impl JdbcSource { &self, env: &mut JNIEnv, result_set: &JObject, - ) -> Result, Error> { + ) -> Result, Error> { let metadata = jni!( env, env.call_method( @@ -897,11 +899,66 @@ impl JdbcSource { // Clamp a driver-supplied count before using it as an allocation size: a // negative i32 would sign-extend to an enormous usize and abort on alloc. - let mut columns = Vec::with_capacity((column_count.max(0) as usize).min(8192)); + let mut raw_columns = Vec::with_capacity((column_count.max(0) as usize).min(8192)); for i in 1..=column_count { - let col_name = self.get_column_label(env, &metadata, i)?; - let col_type = self.get_column_type(env, &metadata, i)?; - columns.push((col_name, col_type)); + let source_name = self.get_column_label(env, &metadata, i)?; + let sql_type = self.get_column_type(env, &metadata, i)?; + raw_columns.push((source_name, sql_type)); + } + + self.prepare_column_metadata(raw_columns) + } + + fn prepare_column_metadata( + &self, + raw_columns: Vec<(String, i32)>, + ) -> Result, Error> { + let mut columns = Vec::with_capacity(raw_columns.len()); + let mut output_names = std::collections::HashSet::new(); + let mut tracking_matches = 0usize; + for (source_name, sql_type) in raw_columns { + let output_name = if self.config.snake_case_columns { + to_snake_case(&source_name) + } else { + source_name.clone() + }; + if !output_names.insert(output_name.clone()) { + warn!( + "Column '{source_name}' maps to output key '{output_name}', which is already in the result; later values overwrite earlier ones" + ); + } + let is_tracking = self + .config + .tracking_column + .as_deref() + .is_some_and(|tracking| { + tracking_column_matches(tracking, &source_name, &output_name) + }); + if is_tracking { + tracking_matches += 1; + } + columns.push(ColumnMetadata { + output_name, + sql_type, + is_tracking, + }); + } + + if self.config.mode == Mode::Incremental { + let tracking_column = self.config.tracking_column.as_deref().unwrap_or(""); + match tracking_matches { + 1 => {} + 0 => { + return Err(Error::InvalidConfigValue(format!( + "tracking column '{tracking_column}' is not present in the query result; include it exactly once in the SELECT list" + ))); + } + count => { + return Err(Error::InvalidConfigValue(format!( + "tracking column '{tracking_column}' matches {count} columns in the query result; use unique column aliases so it matches exactly once" + ))); + } + } } Ok(columns) @@ -912,11 +969,11 @@ impl JdbcSource { &self, env: &mut JNIEnv, result_set: &JObject, - columns: &[(String, i32)], + columns: &[ColumnMetadata], ) -> Result<(Vec, u64, Option), Error> { - // setMaxRows caps the result set at batch_size, so that is the known - // upper bound; clamp the pre-allocation so an extreme batch_size cannot - // request an absurd allocation up front. + // The emitted batch is capped at batch_size; the possible extra row is a + // probe and is never retained. Clamp the pre-allocation so an extreme + // batch_size cannot request an absurd allocation up front. let mut messages = Vec::with_capacity((self.config.batch_size as usize).min(8192)); let mut row_count: u64 = 0; let mut last_offset: Option = None; @@ -950,20 +1007,26 @@ impl JdbcSource { Error::Connection, )?; + // The extra row is a boundary probe, never part of the emitted batch. + // If it shares the last in-batch tracking value, advancing with strict + // `>` would skip it and any following ties. Fail the complete poll so + // no messages or checkpoint can escape from an unsafe page. + if self.config.mode == Mode::Incremental && row_count == self.config.batch_size as u64 { + if offset == last_offset { + return Err(Error::InvalidConfigValue(format!( + "incremental batch boundary splits rows with tracking value '{}'; increase batch_size or use a unique, strictly increasing tracking_column", + offset.as_deref().unwrap_or("") + ))); + } + break; + } + // Take the tracking value of the LAST row as the next offset. Rows // arrive in ascending tracking order (validate_config enforces // ORDER BY the tracking column ascending), so the last row is the // high-water mark. Using the last row rather than a Rust-side max // keeps the cursor consistent with the database's own ordering. // - // The next poll resumes with a strict `> last_offset`, so if this - // setMaxRows-capped batch ends in the middle of a run of rows sharing - // one tracking value (a non-unique column such as a timestamp), the - // remaining tied rows are skipped. The tracking column must therefore - // be unique / strictly increasing, or batch_size must exceed the - // largest group of equal values. See the README tracking-column - // requirements; keyset pagination with a tie-break is a planned - // follow-up. if let Some(offset) = offset { last_offset = Some(offset); } @@ -973,19 +1036,6 @@ impl JdbcSource { row_count += 1; } - // In incremental mode a non-empty batch that never yielded a tracking - // value means the tracking column is absent from the result set: the - // offset could never advance, so the same batch would be re-read forever. - // Fail loudly instead of stalling silently. (A NULL tracking value in a - // present column is already rejected per-row by tracking_offset_or_error.) - if self.config.mode == Mode::Incremental && row_count > 0 && last_offset.is_none() { - return Err(Error::InvalidConfigValue(format!( - "tracking column '{}' is not present in the query result; incremental mode cannot \ - advance its offset. Include it in the SELECT list.", - self.config.tracking_column.as_deref().unwrap_or("") - ))); - } - Ok((messages, row_count, last_offset)) } @@ -994,42 +1044,21 @@ impl JdbcSource { &self, env: &mut JNIEnv, result_set: &JObject, - columns: &[(String, i32)], + columns: &[ColumnMetadata], ) -> Result<(serde_json::Map, Option), Error> { let mut row_data = serde_json::Map::new(); let mut offset = None; - for (idx, (col_name, col_type)) in columns.iter().enumerate() { + for (idx, column) in columns.iter().enumerate() { let col_idx = (idx + 1) as i32; - let value = self.extract_column_value(env, result_set, col_idx, col_type)?; - - let final_col_name = if self.config.snake_case_columns { - to_snake_case(col_name) - } else { - col_name.clone() - }; + let value = self.extract_column_value(env, result_set, col_idx, &column.sql_type)?; + row_data.insert(column.output_name.clone(), value); - // A snake_case conversion (or a query with duplicate labels) can map - // two source columns onto the same key; the later value silently wins. - // Warn so the loss is diagnosable rather than invisible. - if row_data.contains_key(&final_col_name) { - warn!( - "Column '{col_name}' maps to key '{final_col_name}', which already exists in the row; the earlier value is overwritten" - ); - } - - row_data.insert(final_col_name.clone(), value); - - // Track offset from the first column matching the tracking column - // (by driver label or normalized key, case-insensitively). Only the - // first match counts: a collapsed/duplicate label must not let a later - // column silently overwrite the offset with an unrelated value. - if offset.is_none() - && let Some(ref tracking_col) = self.config.tracking_column - && tracking_column_matches(tracking_col, col_name, &final_col_name) + if column.is_tracking + && let Some(ref tracking_column) = self.config.tracking_column { - let value = self.extract_offset_value(&row_data, &final_col_name); - offset = self.tracking_offset_or_error(value, tracking_col)?; + let value = self.extract_offset_value(&row_data, &column.output_name); + offset = self.tracking_offset_or_error(value, tracking_column)?; } } @@ -1037,8 +1066,8 @@ impl JdbcSource { } /// Resolve a tracking-column value into an offset. In incremental mode a NULL - /// or empty value is a hard error: the row is emitted but the offset cannot - /// advance past NULL, so it would be re-read (and re-emitted) every poll. + /// or empty value is a hard error: the batch cannot be safely emitted because + /// its checkpoint could not advance past that row. fn tracking_offset_or_error( &self, value: Option, @@ -1403,15 +1432,25 @@ impl JdbcSource { ))); } - // A set poll_interval must be a valid humantime string; an unparsable - // value would otherwise silently fall back to the default. - if let Some(value) = self.config.poll_interval.as_deref() - && !value.trim().is_empty() - && humantime::parse_duration(value.trim()).is_err() + // A set poll_interval must be valid and positive. Zero would make every + // successful poll immediately eligible to run again and hammer the DB. + if let Some(value) = self + .config + .poll_interval + .as_deref() + .map(str::trim) + .filter(|value| !value.is_empty()) { - return Err(Error::InvalidConfigValue(format!( - "poll_interval '{value}' is not a valid duration (e.g. \"30s\", \"5m\", \"1h\")" - ))); + let duration = humantime::parse_duration(value).map_err(|_| { + Error::InvalidConfigValue(format!( + "poll_interval '{value}' is not a valid duration (e.g. \"30s\", \"5m\", \"1h\")" + )) + })?; + if duration.is_zero() { + return Err(Error::InvalidConfigValue( + "poll_interval must be greater than zero".to_string(), + )); + } } // The query must be non-empty; an empty query only fails later at @@ -1460,6 +1499,12 @@ impl JdbcSource { .to_string(), )); }; + if !self.config.query.contains("{last_offset}") { + return Err(Error::InvalidConfigValue( + "incremental mode requires the query to contain {last_offset}; without it each poll would re-read the same first batch" + .to_string(), + )); + } if !query_orders_by_tracking_column( &self.config.query, tracking_column, @@ -1471,19 +1516,6 @@ impl JdbcSource { `ORDER BY {{tracking_column}}`) to the query" ))); } - - // The cursor advances with a strict `> last_offset` per batch, so a - // group of rows sharing one tracking value that is split across a - // batch_size boundary loses its remainder. This cannot be detected - // without inspecting the data, so warn: the column should be unique / - // strictly increasing, or batch_size must exceed the largest tie group. - warn!( - "JDBC source [{}] incremental tracking_column '{tracking_column}': ensure it is \ - unique / strictly increasing, or that batch_size ({}) exceeds the largest group \ - of rows sharing one value. A tie split across a batch boundary skips the \ - remaining rows (common for non-unique timestamp columns).", - self.id, self.config.batch_size - ); } // Dry-run the query build so an unresolved placeholder or an invalid @@ -1505,18 +1537,29 @@ impl Source for JdbcSource { self.config.mode ); - // Fail fast on bad config before starting the JVM or opening a connection. - self.validate_config()?; - - // JVM boot and DriverManager.getConnection are blocking JNI work. Run - // them via block_in_place (like poll()/close()) so they do not monopolize - // a shared async-runtime worker; the login timeout set in the connection - // path bounds how long a stuck connect can block. - tokio::task::block_in_place(|| -> Result<(), Error> { - self.initialize_jvm()?; - self.create_connection()?; - Ok(()) - })?; + let result = (|| -> Result<(), Error> { + // Fail fast on bad config before starting the JVM or opening a connection. + self.validate_config()?; + + // JVM boot and DriverManager.getConnection are blocking JNI work. Run + // them via block_in_place (like poll()/close()) so they do not monopolize + // a shared async-runtime worker; the login timeout set in the connection + // path bounds how long a stuck connect can block. + tokio::task::block_in_place(|| -> Result<(), Error> { + self.initialize_jvm()?; + self.create_connection()?; + Ok(()) + }) + })(); + + if let Err(open_error) = result { + let open_error = sanitize_jdbc_error(open_error, self.config.jdbc_url.expose_secret()); + error!( + "Failed to open JDBC source connector [{}]: {open_error}", + self.id + ); + return Err(open_error); + } info!("JDBC source connector [{}] opened successfully", self.id); Ok(()) @@ -1684,27 +1727,64 @@ fn to_snake_case(s: &str) -> String { result } +struct SharedJvm { + vm: Arc, + configuration: JvmConfiguration, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct JvmConfiguration { + driver_jar_path: PathBuf, + options: Vec, +} + +impl JvmConfiguration { + fn new(driver_jar_path: &str, options: &[String]) -> Result { + let driver_jar_path = Path::new(driver_jar_path).canonicalize().map_err(|error| { + Error::InitError(format!( + "Failed to canonicalize driver_jar_path '{driver_jar_path}': {error}" + )) + })?; + Ok(Self { + driver_jar_path, + options: options.to_vec(), + }) + } + + fn ensure_compatible_with(&self, requested: &Self) -> Result<(), Error> { + if self == requested { + return Ok(()); + } + Err(Error::InvalidConfigValue( + "all JDBC source instances in one runtime process must use the same driver_jar_path and jvm_options because they share one JVM" + .to_string(), + )) + } +} + /// Process-wide JVM. JNI allows only one `JavaVM` per OS process, so every JDBC /// connector instance in this dynamic library shares this one. -static GLOBAL_JVM: Mutex>> = Mutex::new(None); +static GLOBAL_JVM: Mutex> = Mutex::new(None); /// Return the process JVM, creating it on first use within this dynamic -/// library. The first caller's `jvm_options`/classpath win; later callers (e.g. -/// a second JDBC connector of the same type) reuse the existing VM instead of -/// failing with `JNI_EEXIST`. +/// library. Later callers reuse it only when their classpath and options match; +/// otherwise startup fails instead of silently running with the wrong driver. /// /// Limitation: a JDBC *source* and a JDBC *sink* are separate dynamic libraries /// and do not share this static, so configuring both in the *same* connectors /// runtime process is not supported (the second to start cannot create a second /// JVM). Run them in separate runtime processes. fn get_or_create_jvm(driver_jar_path: &str, jvm_options: &[String]) -> Result, Error> { + let requested = JvmConfiguration::new(driver_jar_path, jvm_options)?; let mut guard = lock_mutex(&GLOBAL_JVM, "jvm")?; - if let Some(jvm) = guard.as_ref() { + if let Some(shared) = guard.as_ref() { + shared.configuration.ensure_compatible_with(&requested)?; info!("Reusing existing process JVM"); - return Ok(jvm.clone()); + return Ok(shared.vm.clone()); } - let classpath_option = format!("-Djava.class.path={driver_jar_path}"); + let canonical_jar_path = requested.driver_jar_path.to_string_lossy(); + let classpath_option = format!("-Djava.class.path={canonical_jar_path}"); let mut args_builder = jni::InitArgsBuilder::new() .version(jni::JNIVersion::V8) .option(&classpath_option); @@ -1717,9 +1797,12 @@ fn get_or_create_jvm(driver_jar_path: &str, jvm_options: &[String]) -> Result '0' ORDER BY id"); @@ -2672,7 +2838,6 @@ mod tests { let state = State { last_offset: Some("42".to_string()), processed_rows: 42, - last_poll_time: Utc::now(), }; let query = source.build_query(&state).expect("build query"); assert_eq!(query, "SELECT * FROM users WHERE id > '42' ORDER BY id"); @@ -2704,7 +2869,6 @@ mod tests { let state = State { last_offset: Some("2024-06-15".to_string()), processed_rows: 0, - last_poll_time: Utc::now(), }; let query = source.build_query(&state).expect("build query"); assert_eq!( @@ -2742,7 +2906,6 @@ mod tests { .build_query(&State { last_offset: None, processed_rows: 0, - last_poll_time: Utc::now(), }) .expect("build query"); assert!(!query.contains("{tracking_column}"), "got: {query}"); @@ -2807,7 +2970,6 @@ mod tests { let original_state = State { last_offset: Some("2024-06-15 12:00:00".to_string()), processed_rows: 1500, - last_poll_time: Utc::now(), }; let connector_state = ConnectorState::serialize(&original_state, CONNECTOR_NAME, 1) .expect("Failed to serialize state"); @@ -3287,7 +3449,6 @@ mod tests { let state = State { last_offset: None, processed_rows: 0, - last_poll_time: Utc::now(), }; let query = source.build_query(&state).expect("build query"); // The WHERE clause placeholder should be removed @@ -3349,7 +3510,6 @@ mod tests { let state = State { last_offset: Some("42".to_string()), processed_rows: 42, - last_poll_time: Utc::now(), }; // In bulk mode the query is used verbatim; the tracked offset is ignored. let query = source.build_query(&state).expect("build query"); @@ -3452,7 +3612,6 @@ mod tests { *source.pending_state.lock().expect("pending lock") = Some(State { last_offset: Some(offset.to_string()), processed_rows, - last_poll_time: Utc::now(), }); } diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs index dfccc76485..8aeae968ae 100644 --- a/core/integration/tests/connectors/jdbc/jdbc_source.rs +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -504,10 +504,17 @@ async fn incremental_mode_advances_offset_across_polls() { .expect("Failed to insert initial rows"); let query = "SELECT id, name FROM inc_test WHERE id > {last_offset} ORDER BY id"; - let (_runtime, client) = - setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "incremental") - .await - .expect("Failed to setup runtime"); + let iggy_setup = IggySetup::default(); + let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "incremental", &iggy_setup); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), + "2".to_owned(), + ); + let mut runtime = setup_runtime(); + runtime + .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) + .await; + let client = runtime.create_client().await; // First batch: ids 1..3. let (first_ids, first_received) = poll_until_ids_seen(&client, &[1, 2, 3], POLL_TIMEOUT).await; @@ -722,3 +729,77 @@ async fn bulk_result_larger_than_batch_size_fails_closed() { polled.messages.len() ); } + +/// Test: incremental mode must not advance past rows tied at a batch boundary. +/// The source probes one row beyond batch_size and fails the complete poll when +/// that row shares the last in-batch tracking value, so no partial page is sent. +#[tokio::test] +#[serial] +async fn incremental_tie_at_batch_boundary_fails_closed() { + let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(error) => panic!("Failed to set up Postgres container: {error}"), + }; + + let pool = PgPoolOptions::new() + .max_connections(2) + .connect(&pg_sqlx_url(&jdbc_url)) + .await + .expect("Failed to connect to Postgres for seeding"); + sqlx::query("CREATE TABLE tie_test (id INT PRIMARY KEY, position INT NOT NULL)") + .execute(&pool) + .await + .expect("Failed to create table"); + sqlx::query("INSERT INTO tie_test (id, position) VALUES (1, 1), (2, 2), (3, 2), (4, 3)") + .execute(&pool) + .await + .expect("Failed to insert rows"); + + let iggy_setup = IggySetup::default(); + let query = + "SELECT id, position FROM tie_test WHERE position > {last_offset} ORDER BY position, id"; + let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "incremental", &iggy_setup); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), + "2".to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_TRACKING_COLUMN".to_owned(), + "position".to_owned(), + ); + + let mut runtime = setup_runtime(); + runtime + .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) + .await; + let client = runtime.create_client().await; + + let api_url = runtime + .harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let sources: serde_json::Value = reqwest::get(format!("{api_url}/sources")) + .await + .expect("Failed to query source status") + .error_for_status() + .expect("Source status endpoint returned an error") + .json() + .await + .expect("Failed to deserialize source status"); + assert_eq!( + sources[0]["status"], "running", + "JDBC source must be running before exercising incremental fail-closed behavior: {sources}" + ); + + sleep(Duration::from_secs(4)).await; + let polled = client + .get_messages(POLL_BATCH) + .await + .expect("Failed to poll messages"); + assert!( + polled.messages.is_empty(), + "a tied incremental page must fail before delivering a partial batch, got {} messages", + polled.messages.len() + ); +} diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index 2a915eb1e0..f73f63ca67 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -39,7 +39,7 @@ mod s3; mod stdout; mod surrealdb; -use iggy::prelude::{IggyClient, IggyMessage, Partitioning}; +use iggy::prelude::IggyClient; use iggy_common::Client; use iggy_common::{ CompressionAlgorithm, Durability, IggyExpiry, IggyTimestamp, MaxTopicSize, MessageClient, @@ -120,23 +120,6 @@ struct ConnectorsIggyClient { } impl ConnectorsIggyClient { - /// Send messages to the configured stream/topic (used by sink connector tests). - #[allow(dead_code)] - async fn send_messages( - &self, - messages: &mut [IggyMessage], - ) -> Result<(), iggy_common::IggyError> { - self.client - .send_messages( - &self.stream.clone().try_into().unwrap(), - &self.topic.clone().try_into().unwrap(), - &Partitioning::balanced(), - messages, - ) - .await - .map(|_| ()) - } - /// Poll up to `count` messages from the configured stream/topic. /// /// `count` is the caller's, because it bounds how much a collect-until-N loop From 0a248c30a2464cde5825cb0ad65dc77ae822cd55 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Fri, 2 Oct 2026 15:45:28 +0530 Subject: [PATCH 19/20] fix(connectors): address latest JDBC review findings --- Cargo.lock | 3 +- .../example_config/connectors/jdbc_h2.toml | 1 + .../connectors/test_jdbc_h2.toml | 44 - .../connectors/sources/jdbc_source/Cargo.toml | 3 +- core/connectors/sources/jdbc_source/README.md | 32 +- .../connectors/sources/jdbc_source/src/lib.rs | 1232 ++++++++--------- .../tests/connectors/jdbc/jdbc_source.rs | 59 + 7 files changed, 639 insertions(+), 735 deletions(-) delete mode 100644 core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml diff --git a/Cargo.lock b/Cargo.lock index 40f462a45a..090e06ed8a 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7343,7 +7343,7 @@ dependencies = [ [[package]] name = "iggy_connector_jdbc_source" -version = "0.4.1-edge.1" +version = "0.5.0" dependencies = [ "async-trait", "base64 0.23.1", @@ -7359,7 +7359,6 @@ dependencies = [ "tokio", "toml 1.1.6+spec-1.1.0", "tracing", - "uuid", ] [[package]] diff --git a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml index a603909cbe..d1576dcfd2 100644 --- a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml +++ b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml @@ -53,6 +53,7 @@ initial_offset = "0" mode = "incremental" snake_case_columns = false include_metadata = true +verbose_logging = false [[streams]] stream = "test" diff --git a/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml deleted file mode 100644 index 30457d163f..0000000000 --- a/core/connectors/runtime/example_config/connectors/test_jdbc_h2.toml +++ /dev/null @@ -1,44 +0,0 @@ -# Licensed to the Apache Software Foundation (ASF) under one -# or more contributor license agreements. See the NOTICE file -# distributed with this work for additional information -# regarding copyright ownership. The ASF licenses this file -# to you under the Apache License, Version 2.0 (the -# "License"); you may not use this file except in compliance -# with the License. You may obtain a copy of the License at -# -# http://www.apache.org/licenses/LICENSE-2.0 -# -# Unless required by applicable law or agreed to in writing, -# software distributed under the License is distributed on an -# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY -# KIND, either express or implied. See the License for the -# specific language governing permissions and limitations -# under the License. - -type = "source" -key = "jdbc_h2_test" -enabled = true -version = 0 -name = "JDBC H2 Test Source" -path = "target/release/libiggy_connector_jdbc_source" -plugin_config_format = "toml" - -[plugin_config] -jdbc_url = "jdbc:h2:mem:testdb;DB_CLOSE_DELAY=-1;INIT=CREATE TABLE IF NOT EXISTS users (id INT PRIMARY KEY AUTO_INCREMENT, name VARCHAR(100), email VARCHAR(100), created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP);INSERT INTO users (name, email) SELECT 'User ' || x, 'user' || x || '@test.com' FROM SYSTEM_RANGE(1, 100) WHERE NOT EXISTS (SELECT 1 FROM users);" -driver_class = "org.h2.Driver" -driver_jar_path = "/tmp/jdbc-drivers/h2-2.2.224.jar" -username = "sa" -password = "" -query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" -poll_interval = "5s" -batch_size = 10 -tracking_column = "id" -initial_offset = "0" -mode = "incremental" -snake_case_columns = true -include_metadata = true - -[[streams]] -stream = "test" -topic = "users" -schema = "json" diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml index 38d823459a..b5ab85409f 100644 --- a/core/connectors/sources/jdbc_source/Cargo.toml +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -17,7 +17,7 @@ [package] name = "iggy_connector_jdbc_source" -version = "0.4.1-edge.1" +version = "0.5.0" edition = "2024" license = "Apache-2.0" keywords = ["iggy", "messaging", "streaming", "jdbc", "source"] @@ -59,7 +59,6 @@ serde = { workspace = true, features = ["derive"] } serde_json = { workspace = true } tokio = { workspace = true, features = ["full"] } tracing = { workspace = true } -uuid = { workspace = true, features = ["v4"] } [dev-dependencies] toml = { workspace = true } diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index f2ae1eb368..5090a49e63 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -99,6 +99,7 @@ initial_offset = "2024-01-01 00:00:00" mode = "incremental" snake_case_columns = true include_metadata = true +verbose_logging = false [[streams]] stream = "ecommerce" @@ -204,12 +205,13 @@ topic = "orders" | `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | | `snake_case_columns` | bool | No | false | Convert column names to snake_case | | `include_metadata` | bool | No | true | Wrap each row with metadata (operation type, timestamp). `table_name` is a reserved field and is currently always null | +| `verbose_logging` | bool | No | false | Log per-poll row and column counts at info instead of debug | ## Query Placeholders The `query` parameter supports placeholders for dynamic queries: -- `{last_offset}`: Replaced with the last tracked offset value, wrapped in quotes and escaped +- `{last_offset}`: Replaced with a JDBC `PreparedStatement` parameter and bound using the parameter's reported SQL type - `{tracking_column}`: Replaced with the configured `tracking_column` (validated as a plain SQL identifier) Incremental mode is validated at `open()` and enforces the following (the @@ -256,7 +258,8 @@ The tracking column must also be: auto-increment ID is ideal. Keyset pagination with a tie-break is a planned follow-up. - **Monotonic under the database's own ordering.** Because the cursor is the last - ordered row and is fed back as `WHERE {tracking_column} > ''`, the column + ordered row and is fed back as a bound parameter in `WHERE {tracking_column} > + ?`, the column must increase monotonically under the same ordering the database applies to that `>` (including its collation, for text). Prefer an auto-increment ID or a timestamp; a case-insensitively-collated text key can order differently than its @@ -266,9 +269,9 @@ The tracking column must also be: the query is fixed. Exclude NULLs in the query, e.g. `AND {tracking_column} IS NOT NULL`. - **Round-trippable string form, for timestamps.** Timestamp columns are read as - the driver's string form and substituted back into the next `WHERE`; ensure the - driver emits a form the database orders correctly and can parse back (ISO-8601 - is safe; a locale format such as `MM/DD/YYYY` is not). + the driver's string form and rebound through JDBC using the parameter's SQL + type; ensure the driver emits a form it can convert back (ISO-8601 is safe; a + locale format such as `MM/DD/YYYY` may not be). **Example:** @@ -278,11 +281,11 @@ tracking_column = "id" query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" initial_offset = "0" --- First poll (initial_offset configured as 0) -SELECT * FROM users WHERE id > '0' ORDER BY id +-- Prepared SQL for the first poll; JDBC binds parameter 1 to 0 as the type of id +SELECT * FROM users WHERE id > ? ORDER BY id --- After processing rows up to id=100 -SELECT * FROM users WHERE id > '100' ORDER BY id +-- After processing rows up to id=100, the SQL stays the same and parameter 1 is 100 +SELECT * FROM users WHERE id > ? ORDER BY id ``` ## Output Format @@ -367,10 +370,13 @@ JDBC SQL types are automatically mapped to JSON: rebuilds the same query from the committed offset, so the batch is **re-read rather than skipped** - for a transient in-process send failure as well as for a crash or restart. Rows can therefore be delivered more than once (message - IDs are random per poll, so downstream consumers must dedupe on a business key - if they need exactly-once); send or checkpoint failures do not silently drop - an already-fetched batch. The separate tracking-column uniqueness requirement - above still applies while fetching rows from the database. + IDs are assigned by the producer and are not stable across a replay, so + downstream consumers must dedupe on a business key if they need exactly-once); + send or checkpoint failures do not silently drop an already-fetched batch. + Five consecutive Nacks stop the source; an operator must restart it after the + underlying send or checkpoint failure is fixed. The separate tracking-column + uniqueness requirement above still applies while fetching rows from the + database. - **Connection recovery.** The connection is validated with `Connection.isValid` each poll and transparently re-established (closing the old handle) if it has dropped. The check runs on the shared `block_in_place` worker, so its timeout diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index aefc958a79..c7f8d7def8 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -16,11 +16,13 @@ // under the License. use async_trait::async_trait; +use base64::Engine; use chrono::{DateTime, Utc}; use iggy_connector_sdk::{ ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, source::SourceBatchResult, source_connector, }; +use java::sql::Types; use jni::objects::{GlobalRef, JByteArray, JObject, JString, JThrowable, JValue}; use jni::{JNIEnv, JavaVM}; use regex::Regex; @@ -30,7 +32,6 @@ use std::path::{Path, PathBuf}; use std::sync::{Arc, Mutex, MutexGuard}; use std::time::{Duration, Instant}; use tracing::{debug, error, info, warn}; -use uuid::Uuid; /// Clear any pending Java exception on the current thread. The JNI spec forbids /// making most calls while an exception is pending; doing so aborts the whole @@ -101,22 +102,9 @@ fn best_effort_close(env: &mut JNIEnv, handle: &JObject) { /// being left pending for the next JNI call on this thread. macro_rules! jni { ($env:expr, $call:expr, $ctx:expr) => { - match $call { - Ok(value) => value, - Err(err) => { - let java_exception = take_pending_java_exception(&mut *$env); - let detail = java_exception - .map(|exception| format!("{err}: {exception}")) - .unwrap_or_else(|| err.to_string()); - return Err(Error::Connection(format!("{}: {detail}", $ctx))); - } - } + jni!($env, $call, $ctx, Error::Connection) }; -} - -/// Like [`jni!`] but returns `Error::InitError`, for the connection-setup path. -macro_rules! jni_init { - ($env:expr, $call:expr, $ctx:expr) => { + ($env:expr, $call:expr, $ctx:expr, $map_error:expr) => { match $call { Ok(value) => value, Err(err) => { @@ -124,7 +112,7 @@ macro_rules! jni_init { let detail = java_exception .map(|exception| format!("{err}: {exception}")) .unwrap_or_else(|| err.to_string()); - return Err(Error::InitError(format!("{}: {detail}", $ctx))); + return Err($map_error(format!("{}: {detail}", $ctx))); } } }; @@ -256,6 +244,10 @@ pub struct JdbcSourceConfig { #[serde(default = "default_true")] pub include_metadata: bool, + /// Log per-poll row and column counts at info instead of debug. + #[serde(default)] + pub verbose_logging: Option, + /// JVM options (e.g., ["-Xmx512m", "-Xms128m"]) #[serde(default)] pub jvm_options: Vec, @@ -316,6 +308,7 @@ impl std::fmt::Debug for JdbcSourceConfig { .field("mode", &self.mode) .field("snake_case_columns", &self.snake_case_columns) .field("include_metadata", &self.include_metadata) + .field("verbose_logging", &self.verbose_logging) .field("connection_timeout_ms", &self.connection_timeout_ms) .field("login_timeout_ms", &self.login_timeout_ms) .finish() @@ -339,6 +332,13 @@ struct ColumnMetadata { is_tracking: bool, } +#[derive(Debug, PartialEq, Eq)] +struct PreparedQuery { + sql: String, + offset: Option, + offset_parameter_count: usize, +} + /// Database record structure for output messages #[derive(Debug, Serialize, Deserialize)] pub struct DatabaseRecord { @@ -366,6 +366,8 @@ pub struct JdbcSource { pending_state: Mutex>, // Poll interval parsed once from `config.poll_interval` at construction. poll_interval: Duration, + // Whether per-poll details should be promoted from debug to info. + verbose: bool, // Scheduled start of the next poll, used to pace polls at a fixed cadence // that does not drift with per-poll work time. `None` until the first poll. next_poll_at: Mutex>, @@ -411,6 +413,7 @@ impl JdbcSource { }); let poll_interval = parse_poll_interval(config.poll_interval.as_deref()); + let verbose = config.verbose_logging.unwrap_or(false); Self { id, config, @@ -419,6 +422,7 @@ impl JdbcSource { state: Mutex::new(state), pending_state: Mutex::new(None), poll_interval, + verbose, next_poll_at: Mutex::new(None), } } @@ -459,10 +463,11 @@ impl JdbcSource { /// accumulate on the caller's frame. The returned handle is a `GlobalRef`, so /// it survives the frame pop. fn create_direct_connection_internal(&self, env: &mut JNIEnv) -> Result { - jni_init!( + jni!( env, env.push_local_frame(24), - "Failed to push connection local frame" + "Failed to push connection local frame", + Error::InitError ); let result = self .create_direct_connection_inner(env) @@ -479,13 +484,14 @@ impl JdbcSource { // Use the system loader explicitly for both driver initialization and the // thread context. This keeps DriverManager's view of the driver aligned // with the JVM classpath even when invoked from an attached native thread. - let current_thread_class = jni_init!( + let current_thread_class = jni!( env, env.find_class("java/lang/Thread"), - "Failed to find Thread class" + "Failed to find Thread class", + Error::InitError ); - let current_thread = jni_init!( + let current_thread = jni!( env, env.call_static_method( current_thread_class, @@ -494,16 +500,18 @@ impl JdbcSource { &[], ) .and_then(|v| v.l()), - "Failed to get current thread" + "Failed to get current thread", + Error::InitError ); - let class_loader_class = jni_init!( + let class_loader_class = jni!( env, env.find_class("java/lang/ClassLoader"), - "Failed to find ClassLoader" + "Failed to find ClassLoader", + Error::InitError ); - let system_class_loader = jni_init!( + let system_class_loader = jni!( env, env.call_static_method( class_loader_class, @@ -512,10 +520,11 @@ impl JdbcSource { &[], ) .and_then(|v| v.l()), - "Failed to get system class loader" + "Failed to get system class loader", + Error::InitError ); - jni_init!( + jni!( env, env.call_method( ¤t_thread, @@ -523,7 +532,8 @@ impl JdbcSource { "(Ljava/lang/ClassLoader;)V", &[JValue::Object(&system_class_loader)], ), - "Failed to set context class loader" + "Failed to set context class loader", + Error::InitError ); info!( @@ -531,17 +541,19 @@ impl JdbcSource { self.config.driver_class ); - let class_class = jni_init!( + let class_class = jni!( env, env.find_class("java/lang/Class"), - "Failed to find Class" + "Failed to find Class", + Error::InitError ); - let driver_class_name = jni_init!( + let driver_class_name = jni!( env, env.new_string(&self.config.driver_class), - "Failed to create class name string" + "Failed to create class name string", + Error::InitError ); - jni_init!( + jni!( env, env.call_static_method( class_class, @@ -553,22 +565,25 @@ impl JdbcSource { JValue::Object(&system_class_loader), ], ), - format!("Failed to load driver class '{}'", self.config.driver_class) + format!("Failed to load driver class '{}'", self.config.driver_class), + Error::InitError ); info!("JDBC driver loaded and registered successfully"); // Get connection from DriverManager - let driver_manager = jni_init!( + let driver_manager = jni!( env, env.find_class("java/sql/DriverManager"), - "Failed to find DriverManager" + "Failed to find DriverManager", + Error::InitError ); - let jdbc_url = jni_init!( + let jdbc_url = jni!( env, env.new_string(self.config.jdbc_url.expose_secret()), - "Failed to create JDBC URL string" + "Failed to create JDBC URL string", + Error::InitError ); // Bound connection establishment so an unreachable or slow-DNS database @@ -580,7 +595,7 @@ impl JdbcSource { .login_timeout_ms .div_ceil(1000) .clamp(1, i32::MAX as u64) as i32; - jni_init!( + jni!( env, env.call_static_method( &driver_manager, @@ -589,7 +604,8 @@ impl JdbcSource { &[JValue::Int(login_timeout_secs)], ) .and_then(|v| v.v()), - "Failed to set JDBC login timeout" + "Failed to set JDBC login timeout", + Error::InitError ); // If username/password are provided separately, use 3-arg getConnection @@ -597,18 +613,20 @@ impl JdbcSource { (&self.config.username, &self.config.password) { info!("Using separate username/password authentication"); - let username_jstring = jni_init!( + let username_jstring = jni!( env, env.new_string(username), - "Failed to create username string" + "Failed to create username string", + Error::InitError ); - let password_jstring = jni_init!( + let password_jstring = jni!( env, env.new_string(password.expose_secret()), - "Failed to create password string" + "Failed to create password string", + Error::InitError ); - jni_init!( + jni!( env, env.call_static_method( driver_manager, @@ -621,11 +639,12 @@ impl JdbcSource { ], ) .and_then(|v| v.l()), - "Failed to create JDBC connection with credentials" + "Failed to create JDBC connection with credentials", + Error::InitError ) } else { info!("Using connection string with embedded credentials"); - jni_init!( + jni!( env, env.call_static_method( driver_manager, @@ -634,14 +653,16 @@ impl JdbcSource { &[JValue::Object(&jdbc_url.into())], ) .and_then(|v| v.l()), - "Failed to create JDBC connection from URL" + "Failed to create JDBC connection from URL", + Error::InitError ) }; - let global_ref = jni_init!( + let global_ref = jni!( env, env.new_global_ref(connection_obj), - "Failed to create global reference" + "Failed to create global reference", + Error::InitError ); info!("Direct database connection established successfully"); @@ -728,7 +749,7 @@ impl JdbcSource { self.build_query(&state) }?; // Logged at debug: the built query embeds the substituted offset value. - debug!("Executing query: {}", query); + debug!("Executing query: {}", query.sql); let (messages, row_count, max_offset) = self.execute_statement_and_fetch_rows(env, &connection, &query)?; @@ -764,10 +785,17 @@ impl JdbcSource { processed_rows: state.processed_rows.saturating_add(row_count), } }; - info!( - "Fetched {} rows, {} processed once this batch is acknowledged", - row_count, candidate.processed_rows - ); + if self.verbose { + info!( + "JDBC source connector [{}] fetched {} rows, {} processed once this batch is acknowledged", + self.id, row_count, candidate.processed_rows + ); + } else { + debug!( + "JDBC source connector [{}] fetched {} rows, {} processed once this batch is acknowledged", + self.id, row_count, candidate.processed_rows + ); + } Ok((messages, Some(candidate))) } @@ -777,9 +805,13 @@ impl JdbcSource { &self, env: &mut JNIEnv, connection: &JObject, - query: &str, + query: &PreparedQuery, ) -> Result<(Vec, u64, Option), Error> { - let query_jstring = jni!(env, env.new_string(query), "Failed to create query string"); + let query_jstring = jni!( + env, + env.new_string(&query.sql), + "Failed to create query string" + ); let statement = match env .call_method( @@ -794,6 +826,11 @@ impl JdbcSource { Err(_) => return Err(classify_query_failure(env, "prepare statement")), }; + if let Err(error) = bind_offset_parameters(env, &statement, query) { + best_effort_close(env, &statement); + return Err(error); + } + // Use setMaxRows for database-agnostic row limiting instead of SQL LIMIT. // Both modes fetch one probe row beyond the batch. Bulk mode uses it to // detect unsupported pagination; incremental mode uses it to detect an @@ -895,14 +932,24 @@ impl JdbcSource { "Failed to get column count" ); - info!("Query returned {} columns", column_count); + if self.verbose { + info!( + "JDBC source connector [{}] query returned {} columns", + self.id, column_count + ); + } else { + debug!( + "JDBC source connector [{}] query returned {} columns", + self.id, column_count + ); + } // Clamp a driver-supplied count before using it as an allocation size: a // negative i32 would sign-extend to an enormous usize and abort on alloc. let mut raw_columns = Vec::with_capacity((column_count.max(0) as usize).min(8192)); for i in 1..=column_count { - let source_name = self.get_column_label(env, &metadata, i)?; - let sql_type = self.get_column_type(env, &metadata, i)?; + let source_name = get_column_label(env, &metadata, i)?; + let sql_type = get_column_type(env, &metadata, i)?; raw_columns.push((source_name, sql_type)); } @@ -1051,13 +1098,13 @@ impl JdbcSource { for (idx, column) in columns.iter().enumerate() { let col_idx = (idx + 1) as i32; - let value = self.extract_column_value(env, result_set, col_idx, &column.sql_type)?; + let value = extract_column_value(env, result_set, col_idx, &column.sql_type)?; row_data.insert(column.output_name.clone(), value); if column.is_tracking && let Some(ref tracking_column) = self.config.tracking_column { - let value = self.extract_offset_value(&row_data, &column.output_name); + let value = extract_offset_value(&row_data, &column.output_name); offset = self.tracking_offset_or_error(value, tracking_column)?; } } @@ -1108,7 +1155,7 @@ impl JdbcSource { let now_ms = now.timestamp_millis() as u64; Ok(ProducedMessage { - id: Some(Uuid::new_v4().as_u128()), + id: None, payload, headers: None, checksum: None, @@ -1117,14 +1164,18 @@ impl JdbcSource { }) } - /// Build the query for this poll by substituting the `{tracking_column}` and - /// `{last_offset}` placeholders. Row limiting is handled via JDBC setMaxRows - /// rather than SQL LIMIT to ensure cross-database compatibility. - fn build_query(&self, state: &State) -> Result { + /// Build the query for this poll by substituting `{tracking_column}` and + /// converting each `{last_offset}` into a bound JDBC parameter. Row limiting + /// is handled via JDBC setMaxRows rather than SQL LIMIT. + fn build_query(&self, state: &State) -> Result { let mut query = self.config.query.clone(); if self.config.mode != Mode::Incremental { - return finalize_query(query); + return Ok(PreparedQuery { + sql: finalize_query(query)?, + offset: None, + offset_parameter_count: 0, + }); } let offset = state @@ -1154,258 +1205,249 @@ impl JdbcSource { query = query.replace("{tracking_column}", column); } - // Substitute the offset value (quoted and escaped) when we have one. - if let Some(offset) = offset { - query = query.replace("{last_offset}", "e_sql_literal(offset)); + let offset_parameter_count = query.matches("{last_offset}").count(); + if offset.is_some() { + query = query.replace("{last_offset}", "?"); } - finalize_query(query) + Ok(PreparedQuery { + sql: finalize_query(query)?, + offset: offset.map(str::to_owned), + offset_parameter_count, + }) } +} - /// Get the column label from ResultSetMetaData. Uses `getColumnLabel` (not - /// `getColumnName`) so a `SELECT expr AS alias` yields the alias the caller - /// asked for; `getColumnName` returns the underlying base-table column (or - /// empty for computed columns), which would not match a configured - /// `tracking_column` alias and can be blank. - fn get_column_label( - &self, - env: &mut JNIEnv, - metadata: &JObject, - column_index: i32, - ) -> Result { - let col_name_obj = jni!( - env, - env.call_method( - metadata, - "getColumnLabel", - "(I)Ljava/lang/String;", - &[JValue::Int(column_index)], - ) - .and_then(|v| v.l()), - "Failed to get column label" - ); - - let col_name: String = jni!( - env, - env.get_string(&JString::from(col_name_obj)), - "Failed to convert column name" +/// Get the column label from ResultSetMetaData. Uses `getColumnLabel` (not +/// `getColumnName`) so a `SELECT expr AS alias` yields the alias the caller +/// asked for; `getColumnName` returns the underlying base-table column (or +/// empty for computed columns), which would not match a configured +/// `tracking_column` alias and can be blank. +fn get_column_label( + env: &mut JNIEnv, + metadata: &JObject, + column_index: i32, +) -> Result { + let col_name_obj = jni!( + env, + env.call_method( + metadata, + "getColumnLabel", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], ) - .into(); + .and_then(|v| v.l()), + "Failed to get column label" + ); + + let col_name: String = jni!( + env, + env.get_string(&JString::from(col_name_obj)), + "Failed to convert column name" + ) + .into(); + + Ok(col_name) +} - Ok(col_name) - } +/// Get column type from ResultSetMetaData. +fn get_column_type(env: &mut JNIEnv, metadata: &JObject, column_index: i32) -> Result { + let col_type = jni!( + env, + env.call_method( + metadata, + "getColumnType", + "(I)I", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.i()), + "Failed to get column type" + ); - /// Get column type from ResultSetMetaData - fn get_column_type( - &self, - env: &mut JNIEnv, - metadata: &JObject, - column_index: i32, - ) -> Result { - let col_type = jni!( - env, - env.call_method( - metadata, - "getColumnType", - "(I)I", - &[JValue::Int(column_index)], - ) - .and_then(|v| v.i()), - "Failed to get column type" - ); + Ok(col_type) +} - Ok(col_type) - } +/// Return `value`, or JSON `null` when the last primitive getter read a SQL +/// NULL (detected via `ResultSet.wasNull()`). +fn null_or( + env: &mut JNIEnv, + result_set: &JObject, + value: serde_json::Value, +) -> Result { + let was_null = jni!( + env, + env.call_method(result_set, "wasNull", "()Z", &[]) + .and_then(|v| v.z()), + "Failed to check wasNull" + ); + Ok(if was_null { + serde_json::Value::Null + } else { + value + }) +} - /// Return `value`, or JSON `null` when the last primitive getter read a SQL - /// NULL (detected via `ResultSet.wasNull()`). - fn null_or( - &self, - env: &mut JNIEnv, - result_set: &JObject, - value: serde_json::Value, - ) -> Result { - let was_null = jni!( - env, - env.call_method(result_set, "wasNull", "()Z", &[]) +/// Extract column value based on JDBC type. +fn extract_column_value( + env: &mut JNIEnv, + result_set: &JObject, + column_index: i32, + sql_type: &i32, +) -> Result { + // Primitive getters (getInt/getBoolean/...) return 0/false for SQL NULL, + // so `null_or` consults ResultSet.wasNull() after the getter to tell an + // actual NULL from a zero value. Object getters (getString/getBytes) + // return a null reference for SQL NULL and are null-checked directly, so + // there is no separate getObject probe (one JNI call per column, not two). + match *sql_type { + Types::BIT | Types::BOOLEAN => { + let value = jni!( + env, + env.call_method( + result_set, + "getBoolean", + "(I)Z", + &[JValue::Int(column_index)] + ) .and_then(|v| v.z()), - "Failed to check wasNull" - ); - Ok(if was_null { - serde_json::Value::Null - } else { - value - }) - } - - /// Extract column value based on JDBC type - fn extract_column_value( - &self, - env: &mut JNIEnv, - result_set: &JObject, - column_index: i32, - sql_type: &i32, - ) -> Result { - use java::sql::Types; - - // Primitive getters (getInt/getBoolean/...) return 0/false for SQL NULL, - // so `null_or` consults ResultSet.wasNull() after the getter to tell an - // actual NULL from a zero value. Object getters (getString/getBytes) - // return a null reference for SQL NULL and are null-checked directly, so - // there is no separate getObject probe (one JNI call per column, not two). - match *sql_type { - Types::BIT | Types::BOOLEAN => { - let value = jni!( - env, - env.call_method( - result_set, - "getBoolean", - "(I)Z", - &[JValue::Int(column_index)] - ) - .and_then(|v| v.z()), - "Failed to get boolean" - ); - self.null_or(env, result_set, serde_json::Value::Bool(value)) - } - Types::TINYINT | Types::SMALLINT | Types::INTEGER => { - let value = jni!( - env, - env.call_method(result_set, "getInt", "(I)I", &[JValue::Int(column_index)]) - .and_then(|v| v.i()), - "Failed to get int" - ); - self.null_or(env, result_set, serde_json::json!(value)) - } - // BIGINT is emitted as a string (like NUMERIC/DECIMAL below) so a - // value above 2^53 is not silently rounded by a JSON consumer that - // parses numbers as f64. - Types::BIGINT => { - let value = jni!( - env, - env.call_method(result_set, "getLong", "(I)J", &[JValue::Int(column_index)]) - .and_then(|v| v.j()), - "Failed to get long" - ); - self.null_or(env, result_set, serde_json::json!(value.to_string())) - } - Types::FLOAT | Types::REAL => { - let value = jni!( - env, - env.call_method(result_set, "getFloat", "(I)F", &[JValue::Int(column_index)]) - .and_then(|v| v.f()), - "Failed to get float" - ); - self.null_or(env, result_set, serde_json::json!(value)) - } - Types::DOUBLE => { - let value = jni!( - env, - env.call_method( - result_set, - "getDouble", - "(I)D", - &[JValue::Int(column_index)] - ) - .and_then(|v| v.d()), - "Failed to get double" - ); - self.null_or(env, result_set, serde_json::json!(value)) - } - // NUMERIC/DECIMAL can carry more precision than an f64 can represent - // (e.g. money/large decimals), so emit them as strings to avoid - // silent precision loss. - Types::NUMERIC | Types::DECIMAL => { - self.get_column_as_string(env, result_set, column_index) - } - // Binary columns are base64-encoded so arbitrary bytes survive the - // round-trip through JSON. - Types::BINARY | Types::VARBINARY | Types::LONGVARBINARY => { - let bytes_obj = jni!( - env, - env.call_method( - result_set, - "getBytes", - "(I)[B", - &[JValue::Int(column_index)] - ) - .and_then(|v| v.l()), - "Failed to get bytes" - ); - if bytes_obj.is_null() { - return Ok(serde_json::Value::Null); - } - let buf = jni!( - env, - env.convert_byte_array(JByteArray::from(bytes_obj)), - "Failed to convert bytes" - ); - use base64::Engine; - Ok(serde_json::Value::String( - base64::engine::general_purpose::STANDARD.encode(&buf), - )) - } - // Date/time types are read via their driver string form. Route - // through the null-safe getString path so a NULL date/time yields - // JSON null instead of failing the whole poll on get_string(null). - Types::TIMESTAMP | Types::DATE | Types::TIME => { - self.get_column_as_string(env, result_set, column_index) + "Failed to get boolean" + ); + null_or(env, result_set, serde_json::Value::Bool(value)) + } + Types::TINYINT | Types::SMALLINT | Types::INTEGER => { + let value = jni!( + env, + env.call_method(result_set, "getInt", "(I)I", &[JValue::Int(column_index)]) + .and_then(|v| v.i()), + "Failed to get int" + ); + null_or(env, result_set, serde_json::json!(value)) + } + // BIGINT is emitted as a string (like NUMERIC/DECIMAL below) so a + // value above 2^53 is not silently rounded by a JSON consumer that + // parses numbers as f64. + Types::BIGINT => { + let value = jni!( + env, + env.call_method(result_set, "getLong", "(I)J", &[JValue::Int(column_index)]) + .and_then(|v| v.j()), + "Failed to get long" + ); + null_or(env, result_set, serde_json::json!(value.to_string())) + } + Types::FLOAT | Types::REAL => { + let value = jni!( + env, + env.call_method(result_set, "getFloat", "(I)F", &[JValue::Int(column_index)]) + .and_then(|v| v.f()), + "Failed to get float" + ); + null_or(env, result_set, serde_json::json!(value)) + } + Types::DOUBLE => { + let value = jni!( + env, + env.call_method( + result_set, + "getDouble", + "(I)D", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.d()), + "Failed to get double" + ); + null_or(env, result_set, serde_json::json!(value)) + } + // NUMERIC/DECIMAL can carry more precision than an f64 can represent + // (e.g. money/large decimals), so emit them as strings to avoid + // silent precision loss. + Types::NUMERIC | Types::DECIMAL => get_column_as_string(env, result_set, column_index), + // Binary columns are base64-encoded so arbitrary bytes survive the + // round-trip through JSON. + Types::BINARY | Types::VARBINARY | Types::LONGVARBINARY => { + let bytes_obj = jni!( + env, + env.call_method( + result_set, + "getBytes", + "(I)[B", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.l()), + "Failed to get bytes" + ); + if bytes_obj.is_null() { + return Ok(serde_json::Value::Null); } - // Default: getString for all other types (CHAR, VARCHAR, etc.) - _ => self.get_column_as_string(env, result_set, column_index), + let buf = jni!( + env, + env.convert_byte_array(JByteArray::from(bytes_obj)), + "Failed to convert bytes" + ); + Ok(serde_json::Value::String( + base64::engine::general_purpose::STANDARD.encode(&buf), + )) + } + // Date/time types are read via their driver string form. Route + // through the null-safe getString path so a NULL date/time yields + // JSON null instead of failing the whole poll on get_string(null). + Types::TIMESTAMP | Types::DATE | Types::TIME => { + get_column_as_string(env, result_set, column_index) } + // Default: getString for all other types (CHAR, VARCHAR, etc.) + _ => get_column_as_string(env, result_set, column_index), } +} - /// Read a column via `ResultSet.getString`, returning JSON `null` when the - /// value is SQL NULL. - fn get_column_as_string( - &self, - env: &mut JNIEnv, - result_set: &JObject, - column_index: i32, - ) -> Result { - let value = jni!( - env, - env.call_method( - result_set, - "getString", - "(I)Ljava/lang/String;", - &[JValue::Int(column_index)], - ) - .and_then(|v| v.l()), - "Failed to get string" - ); +/// Read a column via `ResultSet.getString`, returning JSON `null` when the +/// value is SQL NULL. +fn get_column_as_string( + env: &mut JNIEnv, + result_set: &JObject, + column_index: i32, +) -> Result { + let value = jni!( + env, + env.call_method( + result_set, + "getString", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get string" + ); - if value.is_null() { - Ok(serde_json::Value::Null) - } else { - let str_value: String = jni!( - env, - env.get_string(&JString::from(value)), - "Failed to convert string" - ) - .into(); - Ok(serde_json::Value::String(str_value)) - } + if value.is_null() { + Ok(serde_json::Value::Null) + } else { + let str_value: String = jni!( + env, + env.get_string(&JString::from(value)), + "Failed to convert string" + ) + .into(); + Ok(serde_json::Value::String(str_value)) } +} - /// Extract the tracking-column value as a string offset, or `None` when it is - /// SQL NULL / empty / not comparable. In incremental mode a `None` here is - /// turned into a hard error by [`Self::tracking_offset_or_error`] (a NULL - /// tracking value cannot be watermarked), so the tracking column must be - /// NOT NULL; see the README. - fn extract_offset_value( - &self, - row_data: &serde_json::Map, - col_name: &str, - ) -> Option { - match row_data.get(col_name) { - Some(serde_json::Value::Number(n)) => Some(n.to_string()), - Some(serde_json::Value::String(s)) if !s.is_empty() => Some(s.clone()), - _ => None, - } +/// Extract the tracking-column value as a string offset, or `None` when it is +/// SQL NULL / empty / not comparable. In incremental mode a `None` here is +/// turned into a hard error by [`JdbcSource::tracking_offset_or_error`] (a NULL +/// tracking value cannot be watermarked), so the tracking column must be +/// NOT NULL; see the README. +fn extract_offset_value( + row_data: &serde_json::Map, + col_name: &str, +) -> Option { + match row_data.get(col_name) { + Some(serde_json::Value::Number(n)) => Some(n.to_string()), + Some(serde_json::Value::String(s)) if !s.is_empty() => Some(s.clone()), + _ => None, } +} +impl JdbcSource { /// Validate configuration before touching the JVM or the database, so bad /// config surfaces immediately at `open()` with an actionable message rather /// than as an opaque wrapped JVM error or only after the first poll sleep. @@ -1676,13 +1718,10 @@ impl Source for JdbcSource { async fn close(&mut self) -> Result<(), Error> { info!("Closing JDBC source connector [{}]", self.id); - if self.jvm.is_some() { + if let Some(jvm) = self.jvm.as_ref() { // Closing the JDBC connection is blocking JNI work; run it off the // async worker like the poll path. tokio::task::block_in_place(|| -> Result<(), Error> { - let Some(jvm) = self.jvm.as_ref() else { - return Ok(()); - }; let Ok(mut env) = jvm.attach_current_thread() else { return Ok(()); }; @@ -1806,17 +1845,82 @@ fn get_or_create_jvm(driver_jar_path: &str, jvm_options: &[String]) -> Result String { - let escaped = value.replace('\\', "\\\\").replace('\'', "''"); - format!("'{escaped}'") +/// Bind each incremental offset placeholder using the SQL type the driver +/// reports for that parameter. Passing the explicit target type lets JDBC +/// convert the persisted string representation back to an integer, timestamp, +/// decimal, or text value without embedding dialect-specific literals in SQL. +fn bind_offset_parameters( + env: &mut JNIEnv, + statement: &JObject, + query: &PreparedQuery, +) -> Result<(), Error> { + if query.offset_parameter_count == 0 { + return Ok(()); + } + let offset = query.offset.as_deref().ok_or(Error::InvalidState)?; + let expected_count = i32::try_from(query.offset_parameter_count).map_err(|_| { + Error::InvalidConfigValue("query contains too many {last_offset} placeholders".to_string()) + })?; + + let metadata = jni!( + env, + env.call_method( + statement, + "getParameterMetaData", + "()Ljava/sql/ParameterMetaData;", + &[], + ) + .and_then(|value| value.l()), + "Failed to get JDBC parameter metadata" + ); + let actual_count = jni!( + env, + env.call_method(&metadata, "getParameterCount", "()I", &[]) + .and_then(|value| value.i()), + "Failed to get JDBC parameter count" + ); + if actual_count != expected_count { + return Err(Error::InvalidConfigValue(format!( + "query contains {actual_count} JDBC parameters but {expected_count} came from \ + {{last_offset}}; raw '?' parameters are not supported" + ))); + } + + let offset_string = jni!( + env, + env.new_string(offset), + "Failed to create offset parameter string" + ); + let offset_object = JObject::from(offset_string); + for parameter_index in 1..=expected_count { + let sql_type = jni!( + env, + env.call_method( + &metadata, + "getParameterType", + "(I)I", + &[JValue::Int(parameter_index)], + ) + .and_then(|value| value.i()), + "Failed to get JDBC parameter type" + ); + if env + .call_method( + statement, + "setObject", + "(ILjava/lang/Object;I)V", + &[ + JValue::Int(parameter_index), + JValue::Object(&offset_object), + JValue::Int(sql_type), + ], + ) + .is_err() + { + return Err(classify_query_failure(env, "bind offset parameter")); + } + } + Ok(()) } /// Reject a query that still contains an unresolved placeholder, so an invalid @@ -2156,6 +2260,7 @@ mod tests { mode: Mode::Bulk, snake_case_columns: false, include_metadata: true, + verbose_logging: None, jvm_options: vec![], connection_timeout_ms: 30000, login_timeout_ms: 30000, @@ -2209,15 +2314,6 @@ mod tests { assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("greater than zero"))); } - #[test] - fn test_quote_sql_literal_escapes_backslash() { - // Backslash is doubled before quotes so a trailing backslash cannot - // consume the closing quote under MySQL's default sql_mode. - assert_eq!(quote_sql_literal(r"a\b"), r"'a\\b'"); - assert_eq!(quote_sql_literal(r"end\"), r"'end\\'"); - assert_eq!(quote_sql_literal(r"\'"), r"'\\'''"); - } - #[test] fn test_build_query_removes_predicate_case_and_whitespace_insensitive() { for query in [ @@ -2233,10 +2329,11 @@ mod tests { let state = State::default(); let built = source.build_query(&state).expect("build query"); assert!( - !built.contains("{last_offset}") && !built.contains("{tracking_column}"), - "predicate not removed for query variant: {query} -> {built}" + !built.sql.contains("{last_offset}") && !built.sql.contains("{tracking_column}"), + "predicate not removed for query variant: {query} -> {}", + built.sql ); - assert!(built.to_lowercase().contains("order by id")); + assert!(built.sql.to_lowercase().contains("order by id")); } } @@ -2275,11 +2372,14 @@ mod tests { let source = JdbcSource::new(1, config, None); // Cold start: no persisted offset, no initial_offset. let built = source.build_query(&State::default()).expect("build query"); - assert_eq!(built, "SELECT * FROM t WHERE id IS NOT NULL ORDER BY id"); - assert!(!built.contains("{last_offset}") && !built.contains("{tracking_column}")); + assert_eq!( + built.sql, + "SELECT * FROM t WHERE id IS NOT NULL ORDER BY id" + ); + assert!(!built.sql.contains("{last_offset}") && !built.sql.contains("{tracking_column}")); // No dangling AND / empty WHERE. - assert!(!built.to_uppercase().contains("WHERE AND")); - assert!(!built.to_uppercase().contains("AND ORDER")); + assert!(!built.sql.to_uppercase().contains("WHERE AND")); + assert!(!built.sql.to_uppercase().contains("AND ORDER")); } #[test] @@ -2807,22 +2907,11 @@ mod tests { #[test] fn test_build_query_incremental_with_offset() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM users WHERE id > {last_offset} ORDER BY id".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: Some("0".to_string()), mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); @@ -2832,7 +2921,9 @@ mod tests { processed_rows: 0, }; let query = source.build_query(&state).expect("build query"); - assert_eq!(query, "SELECT * FROM users WHERE id > '0' ORDER BY id"); + assert_eq!(query.sql, "SELECT * FROM users WHERE id > ? ORDER BY id"); + assert_eq!(query.offset.as_deref(), Some("0")); + assert_eq!(query.offset_parameter_count, 1); // With tracked offset let state = State { @@ -2840,30 +2931,45 @@ mod tests { processed_rows: 42, }; let query = source.build_query(&state).expect("build query"); - assert_eq!(query, "SELECT * FROM users WHERE id > '42' ORDER BY id"); + assert_eq!(query.sql, "SELECT * FROM users WHERE id > ? ORDER BY id"); + assert_eq!(query.offset.as_deref(), Some("42")); + assert_eq!(query.offset_parameter_count, 1); + } + + #[test] + fn given_repeated_offset_placeholder_when_query_is_built_should_bind_each_parameter() { + let config = JdbcSourceConfig { + query: "SELECT * FROM users WHERE id > {last_offset} OR parent_id > {last_offset} ORDER BY id" + .to_string(), + tracking_column: Some("id".to_string()), + initial_offset: Some(r"a\b".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + + let query = source + .build_query(&State::default()) + .expect("build prepared query"); + + assert_eq!( + query.sql, + "SELECT * FROM users WHERE id > ? OR parent_id > ? ORDER BY id" + ); + assert_eq!(query.offset.as_deref(), Some(r"a\b")); + assert_eq!(query.offset_parameter_count, 2); } #[test] fn test_build_query_substitutes_tracking_column() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" .to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("updated_at".to_string()), initial_offset: Some("2024-01-01".to_string()), mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); let state = State { @@ -2872,32 +2978,21 @@ mod tests { }; let query = source.build_query(&state).expect("build query"); assert_eq!( - query, - "SELECT * FROM orders WHERE updated_at > '2024-06-15' ORDER BY updated_at" + query.sql, + "SELECT * FROM orders WHERE updated_at > ? ORDER BY updated_at" ); + assert_eq!(query.offset.as_deref(), Some("2024-06-15")); } #[test] fn test_build_query_no_offset_substitutes_tracking_column_in_order_by() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" .to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("updated_at".to_string()), - initial_offset: None, mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); // No last_offset and no initial_offset: the WHERE predicate is dropped, @@ -2908,30 +3003,28 @@ mod tests { processed_rows: 0, }) .expect("build query"); - assert!(!query.contains("{tracking_column}"), "got: {query}"); - assert!(!query.contains("{last_offset}"), "got: {query}"); - assert!(query.contains("ORDER BY updated_at"), "got: {query}"); + assert!( + !query.sql.contains("{tracking_column}"), + "got: {}", + query.sql + ); + assert!(!query.sql.contains("{last_offset}"), "got: {}", query.sql); + assert!( + query.sql.contains("ORDER BY updated_at"), + "got: {}", + query.sql + ); + assert!(query.offset.is_none()); } #[test] fn test_build_query_rejects_injection_in_tracking_column() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM t WHERE {tracking_column} > {last_offset}".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("id; DROP TABLE t".to_string()), initial_offset: Some("0".to_string()), mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); assert!(source.build_query(&State::default()).is_err()); @@ -2940,29 +3033,18 @@ mod tests { #[test] fn test_build_query_bulk_mode_no_limit_appended() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM products".to_string(), poll_interval: Some("60s".to_string()), batch_size: 5000, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, include_metadata: false, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); let state = State::default(); let query = source.build_query(&state).expect("build query"); // build_query should NOT append LIMIT; row limiting is done via setMaxRows - assert_eq!(query, "SELECT * FROM products"); - assert!(!query.to_uppercase().contains("LIMIT")); + assert_eq!(query.sql, "SELECT * FROM products"); + assert!(!query.sql.to_uppercase().contains("LIMIT")); } #[test] @@ -2975,22 +3057,13 @@ mod tests { .expect("Failed to serialize state"); let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM orders WHERE updated_at > {last_offset}".to_string(), poll_interval: Some("30s".to_string()), batch_size: 1000, tracking_column: Some("updated_at".to_string()), initial_offset: Some("2024-01-01 00:00:00".to_string()), mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, Some(connector_state)); let state = source.state.lock().unwrap(); @@ -2998,20 +3071,6 @@ mod tests { assert_eq!(state.processed_rows, 1500); } - #[test] - fn test_quote_sql_literal_escapes_single_quotes() { - assert_eq!(quote_sql_literal("42"), "'42'"); - assert_eq!( - quote_sql_literal("2024-01-01 00:00:00"), - "'2024-01-01 00:00:00'" - ); - assert_eq!(quote_sql_literal("o'brien"), "'o''brien'"); - assert_eq!( - quote_sql_literal("x'; DROP TABLE t; --"), - "'x''; DROP TABLE t; --'" - ); - } - #[test] fn test_is_transient_sql_state() { for s in [ @@ -3073,6 +3132,7 @@ mod tests { assert_eq!(config.batch_size, 1000); assert!(config.include_metadata); assert!(!config.snake_case_columns); + assert_eq!(config.verbose_logging, None); assert_eq!(config.connection_timeout_ms, 5000); assert!(config.username.is_none()); assert!(config.password.is_none()); @@ -3097,6 +3157,7 @@ mod tests { mode = "incremental" snake_case_columns = true include_metadata = false + verbose_logging = true jvm_options = ["-Xmx512m", "-Xms128m"] connection_timeout_ms = 60000 "#; @@ -3111,6 +3172,7 @@ mod tests { assert_eq!(config.mode, Mode::Incremental); assert!(config.snake_case_columns); assert!(!config.include_metadata); + assert_eq!(config.verbose_logging, Some(true)); assert_eq!(config.jvm_options, vec!["-Xmx512m", "-Xms128m"]); assert_eq!(config.connection_timeout_ms, 60000); assert_eq!( @@ -3156,6 +3218,56 @@ mod tests { ); } + #[test] + fn given_shipped_examples_should_parse_each_plugin_config() { + let examples = [ + ( + "bulk", + include_str!("../../../runtime/example_config/connectors/jdbc_bulk_mode.toml"), + ), + ( + "h2", + include_str!("../../../runtime/example_config/connectors/jdbc_h2.toml"), + ), + ( + "mysql", + include_str!("../../../runtime/example_config/connectors/jdbc_mysql.toml"), + ), + ( + "oracle", + include_str!("../../../runtime/example_config/connectors/jdbc_oracle.toml"), + ), + ( + "sqlserver", + include_str!("../../../runtime/example_config/connectors/jdbc_sqlserver.toml"), + ), + ]; + + for (name, example) in examples { + let document: toml::Value = toml::from_str(example).unwrap_or_else(|error| { + panic!("shipped {name} example must be valid TOML: {error}") + }); + let plugin_config = document + .get("plugin_config") + .cloned() + .unwrap_or_else(|| panic!("shipped {name} example must contain plugin_config")); + let config: JdbcSourceConfig = plugin_config.try_into().unwrap_or_else(|error| { + panic!("shipped {name} plugin_config must deserialize: {error}") + }); + + if name == "h2" { + assert_eq!(config.mode, Mode::Incremental); + assert!( + config + .jdbc_url + .expose_secret() + .contains(r"\;INSERT INTO users"), + "H2 INIT statements must be separated with an escaped semicolon" + ); + } + } + } + // ========================================================================= // State restoration tests // ========================================================================= @@ -3164,25 +3276,7 @@ mod tests { fn test_state_restoration_with_malformed_bytes_falls_back_to_default() { let connector_state = ConnectorState(vec![0xFF, 0xFE, 0xFD, 0x00]); - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, Some(connector_state)); + let source = JdbcSource::new(1, base_config(), Some(connector_state)); let state = source.state.lock().unwrap(); // Should fall back to default state assert!(state.last_offset.is_none()); @@ -3193,25 +3287,7 @@ mod tests { fn test_state_restoration_with_empty_bytes_falls_back_to_default() { let connector_state = ConnectorState(vec![]); - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, Some(connector_state)); + let source = JdbcSource::new(1, base_config(), Some(connector_state)); let state = source.state.lock().unwrap(); assert!(state.last_offset.is_none()); assert_eq!(state.processed_rows, 0); @@ -3220,22 +3296,11 @@ mod tests { #[test] fn test_state_restoration_none_uses_initial_offset() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM orders WHERE id > {last_offset}".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: Some("100".to_string()), mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); let state = source.state.lock().unwrap(); @@ -3246,22 +3311,9 @@ mod tests { #[test] fn test_state_restoration_none_without_initial_offset() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM products".to_string(), poll_interval: Some("60s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); let state = source.state.lock().unwrap(); @@ -3275,149 +3327,46 @@ mod tests { #[test] fn test_extract_offset_value_with_integer() { - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, None); - let mut row = serde_json::Map::new(); row.insert("id".to_string(), serde_json::json!(42)); - assert_eq!( - source.extract_offset_value(&row, "id"), - Some("42".to_string()) - ); + assert_eq!(extract_offset_value(&row, "id"), Some("42".to_string())); } #[test] fn test_extract_offset_value_with_string() { - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, None); - let mut row = serde_json::Map::new(); row.insert( "updated_at".to_string(), serde_json::json!("2024-06-15 12:00:00"), ); assert_eq!( - source.extract_offset_value(&row, "updated_at"), + extract_offset_value(&row, "updated_at"), Some("2024-06-15 12:00:00".to_string()) ); } #[test] fn test_extract_offset_value_with_float() { - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, None); - let mut row = serde_json::Map::new(); row.insert("version".to_string(), serde_json::json!(3.5)); assert_eq!( - source.extract_offset_value(&row, "version"), + extract_offset_value(&row, "version"), Some("3.5".to_string()) ); } #[test] fn test_extract_offset_value_with_null() { - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, None); - let mut row = serde_json::Map::new(); row.insert("id".to_string(), serde_json::Value::Null); // A SQL NULL tracking value must not become the persisted offset. - assert_eq!(source.extract_offset_value(&row, "id"), None); + assert_eq!(extract_offset_value(&row, "id"), None); } #[test] fn test_extract_offset_value_missing_column() { - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - let source = JdbcSource::new(1, config, None); - let row = serde_json::Map::new(); - assert_eq!(source.extract_offset_value(&row, "nonexistent"), None); + assert_eq!(extract_offset_value(&row, "nonexistent"), None); } // ========================================================================= @@ -3427,23 +3376,11 @@ mod tests { #[test] fn test_build_query_incremental_no_offset_no_initial_removes_where_clause() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM users WHERE {tracking_column} > {last_offset} ORDER BY id" .to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("id".to_string()), - initial_offset: None, mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); let state = State { @@ -3453,9 +3390,9 @@ mod tests { let query = source.build_query(&state).expect("build query"); // The WHERE clause placeholder should be removed assert!( - !query.contains("{last_offset}"), + !query.sql.contains("{last_offset}"), "Query should not contain unresolved placeholder: {}", - query + query.sql ); } @@ -3465,22 +3402,9 @@ mod tests { // auto-remove does not match: the unresolved {last_offset} must produce // an error rather than being shipped to the driver as invalid SQL. let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM t WHERE id >= {last_offset}".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, mode: Mode::Incremental, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); assert!(source.build_query(&State::default()).is_err()); @@ -3489,22 +3413,10 @@ mod tests { #[test] fn test_build_query_bulk_mode_ignores_offset() { let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, query: "SELECT * FROM users ORDER BY id".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, tracking_column: Some("id".to_string()), initial_offset: Some("0".to_string()), - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let source = JdbcSource::new(1, config, None); let state = State { @@ -3513,7 +3425,7 @@ mod tests { }; // In bulk mode the query is used verbatim; the tracked offset is ignored. let query = source.build_query(&state).expect("build query"); - assert_eq!(query, "SELECT * FROM users ORDER BY id"); + assert_eq!(query.sql, "SELECT * FROM users ORDER BY id"); } // ========================================================================= @@ -3556,17 +3468,7 @@ mod tests { driver_jar_path: "/tmp/mysql.jar".to_string(), username: Some("admin".to_string()), password: Some(SecretString::from("MyP@ssw0rd")), - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, + ..base_config() }; let debug_output = format!("{:?}", config); @@ -3660,7 +3562,8 @@ mod tests { // rebuild the next query from '42' and permanently skip rows 11..=42. let state = source.state.lock().expect("state lock"); let query = source.build_query(&state).expect("build query"); - assert_eq!(query, "SELECT id FROM t WHERE id > '10' ORDER BY id"); + assert_eq!(query.sql, "SELECT id FROM t WHERE id > ? ORDER BY id"); + assert_eq!(query.offset.as_deref(), Some("10")); } #[test] @@ -3676,26 +3579,7 @@ mod tests { #[test] fn test_config_debug_without_password() { - let config = JdbcSourceConfig { - jdbc_url: SecretString::from("jdbc:h2:mem:test"), - driver_class: "org.h2.Driver".to_string(), - driver_jar_path: "/tmp/h2.jar".to_string(), - username: None, - password: None, - query: "SELECT 1".to_string(), - poll_interval: Some("10s".to_string()), - batch_size: 100, - tracking_column: None, - initial_offset: None, - mode: Mode::Bulk, - snake_case_columns: false, - include_metadata: true, - jvm_options: vec![], - connection_timeout_ms: 30000, - login_timeout_ms: 30000, - }; - - let debug_output = format!("{:?}", config); + let debug_output = format!("{:?}", base_config()); // Should not panic and should contain the struct name assert!(debug_output.contains("JdbcSourceConfig")); } diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs index 8aeae968ae..b028f774cc 100644 --- a/core/integration/tests/connectors/jdbc/jdbc_source.rs +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -540,6 +540,65 @@ async fn incremental_mode_advances_offset_across_polls() { ); } +/// Text cursors are bound as JDBC parameters. A backslash must reach PostgreSQL +/// unchanged instead of being doubled for MySQL literal syntax, which would +/// move the comparison boundary and re-read the row at the saved cursor. +#[tokio::test] +#[serial] +async fn incremental_text_offset_with_backslash_preserves_cursor_boundary() { + let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { + Ok(result) => result, + Err(error) => panic!("Failed to set up Postgres container: {error}"), + }; + + let pool = PgPoolOptions::new() + .max_connections(2) + .connect(&pg_sqlx_url(&jdbc_url)) + .await + .expect("Failed to connect to Postgres for seeding"); + sqlx::query( + "CREATE TABLE text_cursor_test (id INT PRIMARY KEY, cursor_value TEXT UNIQUE NOT NULL)", + ) + .execute(&pool) + .await + .expect("Failed to create table"); + sqlx::query("INSERT INTO text_cursor_test (id, cursor_value) VALUES ($1, $2), ($3, $4)") + .bind(1_i32) + .bind(r"a\b") + .bind(2_i32) + .bind("z") + .execute(&pool) + .await + .expect("Failed to insert text cursor rows"); + + let iggy_setup = IggySetup::default(); + let query = "SELECT id, cursor_value FROM text_cursor_test \ + WHERE cursor_value > {last_offset} ORDER BY cursor_value"; + let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "incremental", &iggy_setup); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_TRACKING_COLUMN".to_owned(), + "cursor_value".to_owned(), + ); + envs.insert( + "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INITIAL_OFFSET".to_owned(), + r"a\b".to_owned(), + ); + + let mut runtime = setup_runtime(); + runtime + .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) + .await; + let client = runtime.create_client().await; + + let (ids, received) = poll_until_ids_seen(&client, &[2], POLL_TIMEOUT).await; + assert_eq!( + ids, + vec![2], + "Expected only the row after the exact backslash cursor; got ids {ids:?} from \ + {received} received message(s)" + ); +} + /// Test: a single poll over many rows succeeds. This exercises the JNI /// local-reference frame management in `read_rows`: a few-hundred-row result set /// creates hundreds of per-column local references in one native call, which From 9c706ee30858e3ec2b7085a2948a89bcfd813417 Mon Sep 17 00:00:00 2001 From: shbhmrzd Date: Sat, 3 Oct 2026 01:24:00 +0530 Subject: [PATCH 20/20] fix(connectors): harden JDBC source review follow-ups --- Cargo.lock | 2 + Cargo.toml | 1 + .../connectors/sources/jdbc_source/Cargo.toml | 3 +- core/connectors/sources/jdbc_source/README.md | 34 +- .../sources/jdbc_source/config.toml | 3 + .../connectors/sources/jdbc_source/src/lib.rs | 438 ++++--- core/integration/Cargo.toml | 1 + .../tests/connectors/fixtures/jdbc.rs | 358 ++++++ .../tests/connectors/fixtures/mod.rs | 6 + .../connectors/fixtures/postgres/container.rs | 6 +- .../tests/connectors/fixtures/postgres/mod.rs | 1 + .../tests/connectors/jdbc/jdbc_source.rs | 1029 +++++------------ core/integration/tests/connectors/mod.rs | 177 +-- 13 files changed, 968 insertions(+), 1091 deletions(-) create mode 100644 core/integration/tests/connectors/fixtures/jdbc.rs diff --git a/Cargo.lock b/Cargo.lock index 090e06ed8a..cbd7549936 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -7353,6 +7353,7 @@ dependencies = [ "iggy_connector_sdk", "jni 0.21.1", "regex", + "rmp-serde", "secrecy", "serde", "serde_json", @@ -7851,6 +7852,7 @@ dependencies = [ "serde", "serde_json", "serial_test", + "sha2 0.10.9", "sqlparser 0.63.0", "sqlx", "sysinfo 0.39.6", diff --git a/Cargo.toml b/Cargo.toml index 170419b544..13285f3b8e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -310,6 +310,7 @@ serde_yaml_ng = "0.10.0" serial_test = "4.0.1" server = { path = "core/server", default-features = false } server_common = { path = "core/server_common" } +sha2 = "0.10.9" shard = { path = "core/shard" } simd-json = { version = "0.18.1", features = ["serde_impl"] } smallvec = "1.16" diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml index b5ab85409f..1b2016dfa0 100644 --- a/core/connectors/sources/jdbc_source/Cargo.toml +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -22,7 +22,7 @@ edition = "2024" license = "Apache-2.0" keywords = ["iggy", "messaging", "streaming", "jdbc", "source"] categories = ["database"] -description = "Generic JDBC source connector for Iggy - supports MySQL, Oracle, SQL Server, H2, and any JDBC-compliant database" +description = "JDBC source connector for Iggy with PostgreSQL integration coverage" readme = "README.md" publish = false @@ -61,6 +61,7 @@ tokio = { workspace = true, features = ["full"] } tracing = { workspace = true } [dev-dependencies] +rmp-serde = { workspace = true } toml = { workspace = true } [lints] diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md index 5090a49e63..599e91cb8f 100644 --- a/core/connectors/sources/jdbc_source/README.md +++ b/core/connectors/sources/jdbc_source/README.md @@ -1,6 +1,6 @@ # JDBC Source Connector -A generic JDBC source connector for Iggy that supports any JDBC-compliant database including MySQL, PostgreSQL, Oracle, SQL Server, H2, Derby, and more. +A JDBC source connector for Iggy designed to work with JDBC-compliant relational databases. PostgreSQL is covered by the runtime integration suite; the other driver examples below are configuration guides and are not yet exercised in CI. ## Overview @@ -8,7 +8,7 @@ This connector reads data from relational databases using JDBC (Java Database Co ## Features -- **Universal Database Support**: Works with any database that has a JDBC driver +- **JDBC Driver Support**: Uses a supplied JDBC driver without database-specific connector code - **Incremental Sync**: Track changes using timestamps or auto-increment IDs - **Bulk Mode**: Re-runs the query each poll for snapshots (capped at `batch_size` rows; see limitations) - **Type Mapping**: Automatic conversion of SQL types to JSON @@ -18,10 +18,13 @@ This connector reads data from relational databases using JDBC (Java Database Co ## Supported Databases -**ALL JDBC-compliant databases are supported for both bulk and incremental modes:** +PostgreSQL bulk and incremental modes are covered by end-to-end tests with a +real database and the PostgreSQL JDBC driver. The connector is designed around +standard JDBC APIs, so the following databases are expected to work with a +compatible driver, but they are not currently part of the integration test +matrix: - MySQL / MariaDB -- PostgreSQL - Oracle Database - Microsoft SQL Server - H2 Database @@ -33,9 +36,11 @@ This connector reads data from relational databases using JDBC (Java Database Co - Snowflake - Amazon Redshift - Google BigQuery -- Any other JDBC-compliant database +- Other JDBC-compliant relational databases -**Key Point:** The JDBC connector provides a **single, universal implementation** that works with all these databases. You don't need separate connectors for MySQL, Oracle, etc. Just swap the JDBC driver JAR and connection string! +Driver behavior and SQL syntax vary. Validate the query, type mappings, timeout +behavior, and cursor semantics against the exact driver version before using an +untested database in production. ## Prerequisites @@ -199,9 +204,10 @@ topic = "orders" | `batch_size` | u32 | No | 1000 | Maximum rows to fetch per poll | | `tracking_column` | string | Incremental | - | Column to track for incremental reads (required in incremental mode; the query must also `ORDER BY` it) | | `initial_offset` | string | No | - | Starting offset value for first poll | -| `mode` | string | No | "incremental" | Sync mode: "incremental" or "bulk" (bulk works with ALL databases) | +| `mode` | string | No | "incremental" | Sync mode: "incremental" or "bulk" | | `connection_timeout_ms` | u64 | No | 5000 | Timeout (ms) for the per-poll `isValid` liveness check; converted to seconds and capped at 5s | | `login_timeout_ms` | u64 | No | 30000 | Bound on establishing the connection (`DriverManager.setLoginTimeout`); rounded up to whole seconds. Stops an unreachable database from hanging startup | +| `query_timeout_ms` | u64 | No | 30000 | Positive bound on each query execution (`Statement.setQueryTimeout`); rounded up to whole seconds | | `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | | `snake_case_columns` | bool | No | false | Convert column names to snake_case | | `include_metadata` | bool | No | true | Wrap each row with metadata (operation type, timestamp). `table_name` is a reserved field and is currently always null | @@ -382,6 +388,10 @@ JDBC SQL types are automatically mapped to JSON: dropped. The check runs on the shared `block_in_place` worker, so its timeout (`connection_timeout_ms`) is intentionally converted to whole seconds and capped at 5s: a dead connection must not block the worker for tens of seconds. +- **Bounded query execution.** Every prepared statement receives + `query_timeout_ms` through `Statement.setQueryTimeout`. JDBC drivers implement + cancellation differently, so verify timeout behavior for drivers outside the + PostgreSQL integration matrix. - **`SQLState` classification is informational today.** Query failures are classified into transient vs permanent error variants, but the runtime does not yet apply differentiated backoff based on that distinction; it currently @@ -551,9 +561,10 @@ jdbc_url = "jdbc:h2:file:/data/mydb;USER=sa;PASSWORD=sa" ## Mode Comparison -### Incremental Mode (Universal) +### Incremental Mode -**Works with ALL databases** - requires only a tracking column: +Uses standard JDBC prepared statements and requires an orderable tracking +column. PostgreSQL is integration-tested; validate this mode with other drivers. ```toml mode = "incremental" @@ -579,9 +590,10 @@ column requirements above): - SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; same fail-closed boundary behavior as MySQL) - PostgreSQL: `WHERE id > {last_offset} ORDER BY id` -### Bulk Mode (Universal) +### Bulk Mode -**Works with ALL databases** - no special requirements: +Uses standard JDBC result-set APIs and requires no tracking column. PostgreSQL +is integration-tested; validate query and type behavior with other drivers. ```toml mode = "bulk" diff --git a/core/connectors/sources/jdbc_source/config.toml b/core/connectors/sources/jdbc_source/config.toml index 8cb3f06441..8bb339ac7d 100644 --- a/core/connectors/sources/jdbc_source/config.toml +++ b/core/connectors/sources/jdbc_source/config.toml @@ -48,3 +48,6 @@ initial_offset = "0" mode = "incremental" snake_case_columns = false include_metadata = true +connection_timeout_ms = 5000 +login_timeout_ms = 30000 +query_timeout_ms = 30000 diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs index c7f8d7def8..1a1c7b7564 100644 --- a/core/connectors/sources/jdbc_source/src/lib.rs +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -30,7 +30,7 @@ use secrecy::{ExposeSecret, SecretString}; use serde::{Deserialize, Serialize}; use std::path::{Path, PathBuf}; use std::sync::{Arc, Mutex, MutexGuard}; -use std::time::{Duration, Instant}; +use std::time::Duration; use tracing::{debug, error, info, warn}; /// Clear any pending Java exception on the current thread. The JNI spec forbids @@ -267,6 +267,12 @@ pub struct JdbcSourceConfig { /// runtime, which opens sources sequentially at startup) indefinitely. #[serde(default = "default_login_timeout")] pub login_timeout_ms: u64, + + /// Bound on each query execution (default: 30000). Applied through JDBC's + /// `Statement.setQueryTimeout`, which uses whole seconds, so the configured + /// value is rounded up to at least one second. + #[serde(default = "default_query_timeout")] + pub query_timeout_ms: u64, } fn default_connection_timeout() -> u64 { @@ -277,6 +283,10 @@ fn default_login_timeout() -> u64 { 30000 } +fn default_query_timeout() -> u64 { + 30000 +} + fn default_batch_size() -> u32 { 1000 } @@ -311,6 +321,7 @@ impl std::fmt::Debug for JdbcSourceConfig { .field("verbose_logging", &self.verbose_logging) .field("connection_timeout_ms", &self.connection_timeout_ms) .field("login_timeout_ms", &self.login_timeout_ms) + .field("query_timeout_ms", &self.query_timeout_ms) .finish() } } @@ -368,9 +379,6 @@ pub struct JdbcSource { poll_interval: Duration, // Whether per-poll details should be promoted from debug to info. verbose: bool, - // Scheduled start of the next poll, used to pace polls at a fixed cadence - // that does not drift with per-poll work time. `None` until the first poll. - next_poll_at: Mutex>, } /// Sanitize JDBC URL by masking passwords for logging @@ -423,7 +431,6 @@ impl JdbcSource { pending_state: Mutex::new(None), poll_interval, verbose, - next_poll_at: Mutex::new(None), } } @@ -432,7 +439,11 @@ impl JdbcSource { /// instances (see [`get_or_create_jvm`]). fn initialize_jvm(&mut self) -> Result<(), Error> { info!("Initializing JVM for JDBC source connector [{}]", self.id); - let jvm = get_or_create_jvm(&self.config.driver_jar_path, &self.config.jvm_options)?; + let jvm = get_or_create_jvm( + &self.config.driver_jar_path, + &self.config.jvm_options, + self.id, + )?; self.jvm = Some(jvm); Ok(()) } @@ -449,7 +460,8 @@ impl JdbcSource { .map_err(|e| Error::InitError(format!("Failed to attach thread to JVM: {}", e)))?; info!( - "Creating direct JDBC connection to: {}", + "{CONNECTOR_NAME} connector [{}] creating direct connection to: {}", + self.id, sanitize_jdbc_url(self.config.jdbc_url.expose_secret()) ); let conn = self.create_direct_connection_internal(&mut env)?; @@ -537,8 +549,8 @@ impl JdbcSource { ); info!( - "Loading JDBC driver via the system class loader: {}", - self.config.driver_class + "{CONNECTOR_NAME} connector [{}] loading driver via the system class loader: {}", + self.id, self.config.driver_class ); let class_class = jni!( @@ -569,7 +581,10 @@ impl JdbcSource { Error::InitError ); - info!("JDBC driver loaded and registered successfully"); + info!( + "{CONNECTOR_NAME} connector [{}] loaded and registered the JDBC driver", + self.id + ); // Get connection from DriverManager let driver_manager = jni!( @@ -586,6 +601,14 @@ impl JdbcSource { Error::InitError ); + // DriverManager's login timeout is process-global. Keep setting it and + // opening the corresponding connection in one critical section so two + // connector instances cannot apply each other's timeout. + let _driver_manager_guard = lock_mutex( + &DRIVER_MANAGER_CONNECT_LOCK, + "JDBC DriverManager connection", + )?; + // Bound connection establishment so an unreachable or slow-DNS database // fails instead of hanging open() (which the runtime drives sequentially // at startup) forever. setLoginTimeout is process-wide and in whole @@ -612,7 +635,10 @@ impl JdbcSource { let connection_obj = if let (Some(username), Some(password)) = (&self.config.username, &self.config.password) { - info!("Using separate username/password authentication"); + info!( + "{CONNECTOR_NAME} connector [{}] using separate username/password authentication", + self.id + ); let username_jstring = jni!( env, env.new_string(username), @@ -643,7 +669,10 @@ impl JdbcSource { Error::InitError ) } else { - info!("Using connection string with embedded credentials"); + info!( + "{CONNECTOR_NAME} connector [{}] using credentials from the connection string", + self.id + ); jni!( env, env.call_static_method( @@ -665,7 +694,10 @@ impl JdbcSource { Error::InitError ); - info!("Direct database connection established successfully"); + info!( + "{CONNECTOR_NAME} connector [{}] established the database connection", + self.id + ); Ok(global_ref) } @@ -686,7 +718,10 @@ impl JdbcSource { }; if needs_reconnect { - info!("Direct JDBC connection is not valid; re-establishing"); + info!( + "{CONNECTOR_NAME} connector [{}] connection is invalid; re-establishing", + self.id + ); // Best-effort close of the old handle, then drop it before creating // the replacement so a failed reconnect leaves no stale reference. if let Some(old) = guard.as_ref() { @@ -749,7 +784,10 @@ impl JdbcSource { self.build_query(&state) }?; // Logged at debug: the built query embeds the substituted offset value. - debug!("Executing query: {}", query.sql); + debug!( + "{CONNECTOR_NAME} connector [{}] executing query: {}", + self.id, query.sql + ); let (messages, row_count, max_offset) = self.execute_statement_and_fetch_rows(env, &connection, &query)?; @@ -846,6 +884,24 @@ impl JdbcSource { return Err(Error::Connection(format!("Failed to set max rows: {err}"))); } + let query_timeout_secs = self + .config + .query_timeout_ms + .div_ceil(1000) + .clamp(1, i32::MAX as u64) as i32; + if let Err(err) = env.call_method( + &statement, + "setQueryTimeout", + "(I)V", + &[JValue::Int(query_timeout_secs)], + ) { + clear_pending_exception(env); + best_effort_close(env, &statement); + return Err(Error::Connection(format!( + "Failed to set query timeout: {err}" + ))); + } + let result_set = match env .call_method(&statement, "executeQuery", "()Ljava/sql/ResultSet;", &[]) .and_then(|v| v.l()) @@ -970,9 +1026,9 @@ impl JdbcSource { source_name.clone() }; if !output_names.insert(output_name.clone()) { - warn!( - "Column '{source_name}' maps to output key '{output_name}', which is already in the result; later values overwrite earlier ones" - ); + return Err(Error::InvalidConfigValue(format!( + "query result contains duplicate output key '{output_name}' after mapping column '{source_name}'; use unique column aliases" + ))); } let is_tracking = self .config @@ -1153,14 +1209,13 @@ impl JdbcSource { .map_err(|e| Error::Serialization(format!("Failed to serialize row data: {e}")))? }; - let now_ms = now.timestamp_millis() as u64; Ok(ProducedMessage { id: None, payload, headers: None, checksum: None, - timestamp: Some(now_ms), - origin_timestamp: Some(now_ms), + timestamp: None, + origin_timestamp: None, }) } @@ -1495,6 +1550,12 @@ impl JdbcSource { } } + if self.config.query_timeout_ms == 0 { + return Err(Error::InvalidConfigValue( + "query_timeout_ms must be greater than zero".to_string(), + )); + } + // The query must be non-empty; an empty query only fails later at // prepareStatement with an opaque driver error. if self.config.query.trim().is_empty() { @@ -1573,7 +1634,8 @@ impl Source for JdbcSource { async fn open(&mut self) -> Result<(), Error> { info!("Opening JDBC source connector [{}]", self.id); info!( - "Configuration: JDBC URL={}, Driver={}, Mode={:?}", + "{CONNECTOR_NAME} connector [{}] configuration: JDBC URL={}, Driver={}, Mode={:?}", + self.id, sanitize_jdbc_url(self.config.jdbc_url.expose_secret()), self.config.driver_class, self.config.mode @@ -1608,23 +1670,10 @@ impl Source for JdbcSource { } async fn poll(&self) -> Result { - // Pace polls on a fixed cadence measured from a scheduled start instant, - // so per-poll work time does not accumulate as drift and the first poll - // is not delayed by a full interval. The schedule is clamped forward to - // `now` whenever it has fallen behind (a long pause, a poll that overran - // the interval, or the runtime re-polling immediately after an error), - // so a lagging schedule can never collapse the sleep into a busy loop - // that hammers the database. - let scheduled = { - let mut next = lock_mutex(&self.next_poll_at, "next_poll_at")?; - let scheduled = next.map_or_else(Instant::now, |planned| planned.max(Instant::now())); - *next = Some(scheduled + self.poll_interval); - scheduled - }; - let now = Instant::now(); - if scheduled > now { - tokio::time::sleep(scheduled - now).await; - } + // Wait before every fetch. This keeps startup from hammering the + // database and guarantees an error followed by an immediate runtime + // retry still observes the configured poll interval. + tokio::time::sleep(self.poll_interval).await; // The JDBC/JNI fetch is synchronous, blocking work; run it via // block_in_place so it does not monopolize a shared async-runtime worker @@ -1728,7 +1777,10 @@ impl Source for JdbcSource { let mut guard = lock_mutex(&self.connection, "connection")?; if let Some(connection) = guard.as_ref() { best_effort_close(&mut env, connection.as_obj()); - info!("Database connection closed"); + info!( + "{CONNECTOR_NAME} connector [{}] database connection closed", + self.id + ); } *guard = None; Ok(()) @@ -1805,6 +1857,10 @@ impl JvmConfiguration { /// connector instance in this dynamic library shares this one. static GLOBAL_JVM: Mutex> = Mutex::new(None); +/// Serializes the process-global `DriverManager.setLoginTimeout` setting with +/// the connection attempt it governs. +static DRIVER_MANAGER_CONNECT_LOCK: Mutex<()> = Mutex::new(()); + /// Return the process JVM, creating it on first use within this dynamic /// library. Later callers reuse it only when their classpath and options match; /// otherwise startup fails instead of silently running with the wrong driver. @@ -1813,12 +1869,16 @@ static GLOBAL_JVM: Mutex> = Mutex::new(None); /// and do not share this static, so configuring both in the *same* connectors /// runtime process is not supported (the second to start cannot create a second /// JVM). Run them in separate runtime processes. -fn get_or_create_jvm(driver_jar_path: &str, jvm_options: &[String]) -> Result, Error> { +fn get_or_create_jvm( + driver_jar_path: &str, + jvm_options: &[String], + connector_id: u32, +) -> Result, Error> { let requested = JvmConfiguration::new(driver_jar_path, jvm_options)?; let mut guard = lock_mutex(&GLOBAL_JVM, "jvm")?; if let Some(shared) = guard.as_ref() { shared.configuration.ensure_compatible_with(&requested)?; - info!("Reusing existing process JVM"); + info!("{CONNECTOR_NAME} connector [{connector_id}] reusing existing process JVM"); return Ok(shared.vm.clone()); } @@ -1836,7 +1896,9 @@ fn get_or_create_jvm(driver_jar_path: &str, jvm_options: &[String]) -> Result>(); + if !modifiers.is_empty() + && !matches!(modifiers.as_slice(), [direction] if direction == "asc") + && !matches!( + modifiers.as_slice(), + [nulls, position] + if nulls == "nulls" && matches!(position.as_str(), "first" | "last") + ) + && !matches!( + modifiers.as_slice(), + [direction, nulls, position] + if direction == "asc" + && nulls == "nulls" + && matches!(position.as_str(), "first" | "last") + ) + { return false; } - if key.starts_with("{tracking_column}") { + if key == "{tracking_column}" { return true; } - // Compare the final path segment (`t.updated_at` -> `updated_at`), first - // stripping a surrounding identifier quote (`"pg"`, `` `mysql` ``, `[mssql]`) - // then keeping only leading identifier characters so trailing punctuation is - // ignored. A case-preserving column such as PostgreSQL's `ORDER BY - // "OrderDate"` must validate against the same `tracking_column` that matches - // its unquoted driver label when reading rows. - let segment = key - .rsplit('.') - .next() - .unwrap_or(key) - .trim_matches(|c| c == '"' || c == '`' || c == '[' || c == ']'); - let key_ident: String = segment - .chars() - .take_while(|c| c.is_ascii_alphanumeric() || *c == '_') - .collect(); - if key_ident.is_empty() { + // Compare the final path segment (`t.updated_at` -> `updated_at`) after + // validating every segment as a complete plain or quoted identifier. A + // case-preserving column such as PostgreSQL's `ORDER BY "OrderDate"` must + // validate against the same `tracking_column` that matches its unquoted + // driver label when reading rows. + let Some(key_ident) = exact_identifier_final_segment(key) else { return false; - } + }; // Normalize identically to the read-time path so validate-time cannot reject // a config whose ORDER BY column would match a returned row at runtime. let normalized = if snake_case_columns { - to_snake_case(&key_ident) + to_snake_case(key_ident) } else { - key_ident.clone() + key_ident.to_string() }; - tracking_column_matches(tracking_column, &key_ident, &normalized) + tracking_column_matches(tracking_column, key_ident, &normalized) +} + +/// Return the final segment of a plain or qualified SQL identifier. Expressions, +/// function calls, casts, and trailing operators are rejected so validation +/// cannot mistake `ORDER BY id % 10` for ordering by the cursor itself. +fn exact_identifier_final_segment(identifier: &str) -> Option<&str> { + let mut final_segment = None; + for segment in identifier.split('.') { + let unquoted = if segment.starts_with('"') && segment.ends_with('"') + || segment.starts_with('`') && segment.ends_with('`') + || segment.starts_with('[') && segment.ends_with(']') + { + &segment[1..segment.len().checked_sub(1)?] + } else { + segment + }; + if unquoted.is_empty() + || !unquoted.chars().enumerate().all(|(index, character)| { + character == '_' + || character == '$' + || character.is_ascii_alphanumeric() + && (index > 0 || !character.is_ascii_digit()) + }) + { + return None; + } + final_segment = Some(unquoted); + } + final_segment } /// Return the text following the outer `ORDER BY` (the last `order by` token at @@ -2264,6 +2367,7 @@ mod tests { jvm_options: vec![], connection_timeout_ms: 30000, login_timeout_ms: 30000, + query_timeout_ms: 30000, } } @@ -2276,7 +2380,7 @@ mod tests { } #[test] - fn test_parse_poll_interval() { + fn given_poll_interval_should_parse_or_default() { assert_eq!(parse_poll_interval(Some("30s")), Duration::from_secs(30)); assert_eq!(parse_poll_interval(Some("5m")), Duration::from_secs(300)); // Unset, empty, and unparsable all fall back to the default. @@ -2289,7 +2393,7 @@ mod tests { } #[test] - fn test_validate_config_rejects_bad_poll_interval() { + fn given_invalid_poll_interval_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_poll_interval.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2302,7 +2406,7 @@ mod tests { } #[test] - fn test_validate_config_rejects_zero_poll_interval() { + fn given_zero_poll_interval_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_zero_poll_interval.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2315,7 +2419,19 @@ mod tests { } #[test] - fn test_build_query_removes_predicate_case_and_whitespace_insensitive() { + fn given_zero_query_timeout_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_query_timeout.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.query_timeout_ms = 0; + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(message)) if message.contains("query_timeout_ms")) + ); + } + + #[test] + fn given_cold_start_should_remove_offset_predicate() { for query in [ "select * from t where {tracking_column} > {last_offset} order by id", "SELECT * FROM t WHERE {tracking_column} > {last_offset} ORDER BY id", @@ -2338,7 +2454,7 @@ mod tests { } #[test] - fn test_strip_offset_predicate_preserves_other_conditions() { + fn given_compound_predicate_should_preserve_other_conditions() { // Offset term followed by a companion condition (the README-advised // IS NOT NULL): the AND and the companion condition survive. assert_eq!( @@ -2364,7 +2480,7 @@ mod tests { } #[test] - fn test_build_query_cold_start_compound_predicate_is_valid() { + fn given_cold_start_compound_predicate_should_build_valid_query() { let mut config = base_config(); config.mode = Mode::Incremental; config.tracking_column = Some("id".to_string()); @@ -2383,7 +2499,7 @@ mod tests { } #[test] - fn test_validate_config_rejects_half_set_credentials() { + fn given_half_set_credentials_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_half_creds.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2397,7 +2513,7 @@ mod tests { } #[test] - fn test_validate_config_rejects_missing_driver_jar() { + fn given_missing_driver_jar_should_be_rejected() { let mut config = base_config(); config.driver_jar_path = "/nonexistent/path/to/driver.jar".to_string(); let source = JdbcSource::new(1, config, None); @@ -2406,7 +2522,7 @@ mod tests { } #[test] - fn test_validate_config_dry_runs_query() { + fn given_invalid_built_query_should_fail_validation() { let jar = write_temp_jar("jdbc_validate_dry_run.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2424,7 +2540,7 @@ mod tests { } #[test] - fn test_validate_config_accepts_valid_bulk() { + fn given_valid_bulk_config_should_pass_validation() { let jar = write_temp_jar("jdbc_validate_ok.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2433,7 +2549,7 @@ mod tests { } #[test] - fn test_validate_config_accepts_valid_incremental() { + fn given_valid_incremental_config_should_pass_validation() { let jar = write_temp_jar("jdbc_validate_ok_incremental.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2449,7 +2565,7 @@ mod tests { } #[test] - fn test_validate_config_incremental_requires_last_offset_placeholder() { + fn given_incremental_query_without_offset_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_last_offset.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2464,7 +2580,7 @@ mod tests { } #[test] - fn test_incremental_metadata_rejects_duplicate_tracking_labels() { + fn given_duplicate_tracking_labels_should_be_rejected() { let mut config = base_config(); config.mode = Mode::Incremental; config.tracking_column = Some("id".to_string()); @@ -2476,7 +2592,18 @@ mod tests { } #[test] - fn test_incremental_metadata_requires_tracking_column_in_result() { + fn given_duplicate_bulk_labels_should_be_rejected() { + let source = JdbcSource::new(1, base_config(), None); + let error = source + .prepare_column_metadata(vec![("id".to_string(), 4), ("id".to_string(), 4)]) + .expect_err("duplicate output keys must be rejected in every mode"); + assert!( + matches!(error, Error::InvalidConfigValue(message) if message.contains("duplicate output key")) + ); + } + + #[test] + fn given_missing_tracking_result_column_should_be_rejected() { let mut config = base_config(); config.mode = Mode::Incremental; config.tracking_column = Some("id".to_string()); @@ -2488,7 +2615,7 @@ mod tests { } #[test] - fn test_validate_config_incremental_requires_tracking_column() { + fn given_incremental_config_without_tracking_column_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_no_tracking.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2503,7 +2630,7 @@ mod tests { } #[test] - fn test_validate_config_incremental_requires_order_by_tracking_column() { + fn given_unordered_incremental_query_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_no_order.jar"); let mut config = base_config(); config.driver_jar_path = jar; @@ -2519,7 +2646,7 @@ mod tests { } #[test] - fn test_query_orders_by_tracking_column_accepts_valid() { + fn given_valid_tracking_order_should_be_accepted() { assert!(query_orders_by_tracking_column( "SELECT * FROM t WHERE id > {last_offset} ORDER BY id", "id", @@ -2550,7 +2677,7 @@ mod tests { } #[test] - fn test_query_orders_by_tracking_column_rejects_invalid() { + fn given_invalid_tracking_order_should_be_rejected() { // No ORDER BY. assert!(!query_orders_by_tracking_column( "SELECT * FROM t WHERE id > 0", @@ -2586,10 +2713,22 @@ mod tests { "id", false )); + // An expression beginning with the tracking column does not produce a + // contiguous cursor-ordered prefix. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY id % 10, id", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY lower(id)", + "id", + false + )); } #[test] - fn test_query_orders_by_tracking_column_ignores_window_order_by() { + fn given_window_order_by_should_not_satisfy_result_ordering() { // A window function's internal ORDER BY orders values within the frame, // not the emitted ResultSet, so it must not satisfy the outer-ordering // requirement even though it is the only `order by` in the text. @@ -2642,7 +2781,7 @@ mod tests { } #[test] - fn test_query_orders_by_tracking_column_snake_case_matches_read_time() { + fn given_snake_case_tracking_order_should_match_read_time() { // snake_case_columns = true: a CamelCase ORDER BY column with a // snake_cased tracking_column validates, mirroring tracking_column_matches // so validate-time never rejects a config that reads rows correctly. @@ -2667,7 +2806,7 @@ mod tests { } #[test] - fn test_query_orders_by_tracking_column_strips_identifier_quotes() { + fn given_quoted_tracking_identifier_should_match() { // PostgreSQL preserves case only for quoted identifiers, so a genuinely // CamelCase column is ordered as `"OrderDate"`; the surrounding quotes // must not defeat the match against its unquoted driver label. @@ -2695,7 +2834,7 @@ mod tests { } #[test] - fn test_validate_config_rejects_empty_query_and_blank_offsets() { + fn given_empty_query_or_offset_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_blanks.jar"); // Empty query. let mut config = base_config(); @@ -2728,7 +2867,7 @@ mod tests { } #[test] - fn test_validate_config_rejects_bad_batch_size() { + fn given_invalid_batch_size_should_be_rejected() { let jar = write_temp_jar("jdbc_validate_batch_size.jar"); for bad in [0u32, i32::MAX as u32] { let mut config = base_config(); @@ -2743,7 +2882,7 @@ mod tests { } #[test] - fn test_tracking_column_matches_case_and_normalization() { + fn given_tracking_column_should_match_case_and_normalization() { // Case-insensitive against the raw driver label (driver case-folding). assert!(tracking_column_matches( "OrderDate", @@ -2763,7 +2902,7 @@ mod tests { } #[test] - fn test_tracking_offset_or_error_rejects_null_in_incremental() { + fn given_null_incremental_cursor_should_be_rejected() { let mut config = base_config(); config.mode = Mode::Incremental; config.tracking_column = Some("id".to_string()); @@ -2782,13 +2921,13 @@ mod tests { } #[test] - fn test_tracking_offset_or_error_allows_null_in_bulk() { + fn given_null_bulk_cursor_should_be_allowed() { let source = JdbcSource::new(1, base_config(), None); // base_config is bulk assert_eq!(source.tracking_offset_or_error(None, "id").unwrap(), None); } #[test] - fn test_sanitize_jdbc_url_mysql_format() { + fn given_mysql_url_should_redact_password() { let url = "jdbc:mysql://root:SuperSecret123@localhost:3306/mydb"; let sanitized = sanitize_jdbc_url(url); assert_eq!(sanitized, "jdbc:mysql://root:***@localhost:3306/mydb"); @@ -2796,7 +2935,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_masks_password_through_last_at_sign() { + fn given_password_with_at_sign_should_be_fully_redacted() { let url = "jdbc:mysql://root:p@ss@localhost:3306/mydb"; let sanitized = sanitize_jdbc_url(url); assert_eq!(sanitized, "jdbc:mysql://root:***@localhost:3306/mydb"); @@ -2804,7 +2943,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_error_masks_echoed_url() { + fn given_echoed_jdbc_url_should_be_redacted_from_error() { let url = "jdbc:mysql://root:p@ss@localhost:3306/mydb"; let error = Error::InitError(format!("No suitable driver found for {url}")); let sanitized = sanitize_jdbc_error(error, url).to_string(); @@ -2816,7 +2955,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_postgresql_query_params() { + fn given_postgres_url_should_redact_password() { let url = "jdbc:postgresql://localhost:5432/mydb?user=admin&password=P@ssw0rd&ssl=true"; let sanitized = sanitize_jdbc_url(url); assert_eq!( @@ -2827,7 +2966,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_oracle_format() { + fn given_oracle_url_should_redact_password() { let url = "jdbc:oracle:thin:system/oracle123@localhost:1521:XE"; let sanitized = sanitize_jdbc_url(url); assert_eq!(sanitized, "jdbc:oracle:thin:system/***@localhost:1521:XE"); @@ -2835,7 +2974,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_sqlserver_format() { + fn given_sql_server_url_should_redact_password() { let url = "jdbc:sqlserver://localhost:1433;user=sa;password=MySecretPass;database=Sales"; let sanitized = sanitize_jdbc_url(url); assert_eq!( @@ -2846,7 +2985,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_h2_format() { + fn given_h2_url_should_redact_password() { let url = "jdbc:h2:mem:testdb;USER=sa;PASSWORD=secret"; let sanitized = sanitize_jdbc_url(url); assert_eq!(sanitized, "jdbc:h2:mem:testdb;USER=sa;PASSWORD=***"); @@ -2854,7 +2993,7 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_case_insensitive() { + fn given_mixed_case_password_key_should_be_redacted() { let url1 = "jdbc:postgresql://localhost?password=secret"; let url2 = "jdbc:postgresql://localhost?PASSWORD=secret"; let url3 = "jdbc:postgresql://localhost?pwd=secret"; @@ -2868,14 +3007,14 @@ mod tests { } #[test] - fn test_sanitize_jdbc_url_no_password() { + fn given_url_without_password_should_remain_unchanged() { let url = "jdbc:h2:mem:testdb"; let sanitized = sanitize_jdbc_url(url); assert_eq!(sanitized, url); } #[test] - fn test_sanitize_jdbc_url_multiple_passwords() { + fn given_multiple_passwords_should_all_be_redacted() { let url = "jdbc:postgresql://localhost?password=secret1&pwd=secret2"; let sanitized = sanitize_jdbc_url(url); assert!(!sanitized.contains("secret1")); @@ -2887,7 +3026,7 @@ mod tests { } #[test] - fn test_jvm_configuration_rejects_different_jar_or_options() { + fn given_different_jvm_config_should_be_rejected() { let first_jar = write_temp_jar("jdbc_jvm_first.jar"); let second_jar = write_temp_jar("jdbc_jvm_second.jar"); let first = JvmConfiguration::new(&first_jar, &["-Xmx128m".to_string()]) @@ -2905,7 +3044,7 @@ mod tests { } #[test] - fn test_build_query_incremental_with_offset() { + fn given_incremental_offset_should_be_bound() { let config = JdbcSourceConfig { query: "SELECT * FROM users WHERE id > {last_offset} ORDER BY id".to_string(), tracking_column: Some("id".to_string()), @@ -2961,7 +3100,7 @@ mod tests { } #[test] - fn test_build_query_substitutes_tracking_column() { + fn given_tracking_placeholder_should_be_substituted() { let config = JdbcSourceConfig { query: "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" @@ -2985,7 +3124,7 @@ mod tests { } #[test] - fn test_build_query_no_offset_substitutes_tracking_column_in_order_by() { + fn given_no_offset_should_still_substitute_ordering_column() { let config = JdbcSourceConfig { query: "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" @@ -3018,7 +3157,7 @@ mod tests { } #[test] - fn test_build_query_rejects_injection_in_tracking_column() { + fn given_unsafe_tracking_identifier_should_be_rejected() { let config = JdbcSourceConfig { query: "SELECT * FROM t WHERE {tracking_column} > {last_offset}".to_string(), tracking_column: Some("id; DROP TABLE t".to_string()), @@ -3031,7 +3170,7 @@ mod tests { } #[test] - fn test_build_query_bulk_mode_no_limit_appended() { + fn given_bulk_query_should_not_append_limit() { let config = JdbcSourceConfig { query: "SELECT * FROM products".to_string(), poll_interval: Some("60s".to_string()), @@ -3048,7 +3187,7 @@ mod tests { } #[test] - fn test_state_restoration_from_connector_state() { + fn given_persisted_state_should_restore_cursor_and_count() { let original_state = State { last_offset: Some("2024-06-15 12:00:00".to_string()), processed_rows: 1500, @@ -3072,7 +3211,34 @@ mod tests { } #[test] - fn test_is_transient_sql_state() { + fn state_should_be_serializable_and_deserializable() { + let original = State { + last_offset: Some("2026-10-02T10:15:00Z".to_string()), + processed_rows: 42, + }; + let bytes = rmp_serde::to_vec(&original).expect("state should serialize"); + let restored: State = rmp_serde::from_slice(&bytes).expect("state should deserialize"); + assert_eq!(restored.last_offset, original.last_offset); + assert_eq!(restored.processed_rows, original.processed_rows); + } + + #[test] + fn given_built_message_should_leave_broker_timestamps_unset() { + let source = JdbcSource::new(1, base_config(), None); + let mut row = serde_json::Map::new(); + row.insert("id".to_string(), serde_json::json!(1)); + + let message = source.build_message(row).expect("message should build"); + + assert!(message.timestamp.is_none()); + assert!(message.origin_timestamp.is_none()); + let payload: DatabaseRecord = + serde_json::from_slice(&message.payload).expect("metadata payload should deserialize"); + assert_eq!(payload.data["id"], 1); + } + + #[test] + fn given_sql_state_should_be_classified() { for s in [ "08001", "08006", "40001", "40P01", "53300", "57P01", "58030", ] { @@ -3086,7 +3252,7 @@ mod tests { } #[test] - fn test_is_valid_identifier() { + fn given_sql_identifier_should_be_validated() { assert!(is_valid_identifier("id")); assert!(is_valid_identifier("updated_at")); assert!(is_valid_identifier("t.updated_at")); @@ -3097,7 +3263,7 @@ mod tests { } #[test] - fn test_to_snake_case() { + fn given_column_name_should_convert_to_snake_case() { assert_eq!(to_snake_case("OrderDate"), "order_date"); assert_eq!(to_snake_case("updatedAt"), "updated_at"); assert_eq!(to_snake_case("ID"), "id"); // consecutive uppers stay together @@ -3110,7 +3276,7 @@ mod tests { // ========================================================================= #[test] - fn test_config_deserialization_minimal_toml() { + fn given_minimal_toml_should_apply_defaults() { let toml_str = r#" jdbc_url = "jdbc:h2:mem:test" driver_class = "org.h2.Driver" @@ -3134,6 +3300,8 @@ mod tests { assert!(!config.snake_case_columns); assert_eq!(config.verbose_logging, None); assert_eq!(config.connection_timeout_ms, 5000); + assert_eq!(config.login_timeout_ms, 30000); + assert_eq!(config.query_timeout_ms, 30000); assert!(config.username.is_none()); assert!(config.password.is_none()); assert!(config.tracking_column.is_none()); @@ -3142,7 +3310,7 @@ mod tests { } #[test] - fn test_config_deserialization_full_toml() { + fn given_full_toml_should_deserialize_all_fields() { let toml_str = r#" jdbc_url = "jdbc:mysql://localhost:3306/mydb" driver_class = "com.mysql.cj.jdbc.Driver" @@ -3160,6 +3328,8 @@ mod tests { verbose_logging = true jvm_options = ["-Xmx512m", "-Xms128m"] connection_timeout_ms = 60000 + login_timeout_ms = 45000 + query_timeout_ms = 120000 "#; let config: JdbcSourceConfig = toml::from_str(toml_str).expect("Failed to parse full TOML config"); @@ -3175,6 +3345,8 @@ mod tests { assert_eq!(config.verbose_logging, Some(true)); assert_eq!(config.jvm_options, vec!["-Xmx512m", "-Xms128m"]); assert_eq!(config.connection_timeout_ms, 60000); + assert_eq!(config.login_timeout_ms, 45000); + assert_eq!(config.query_timeout_ms, 120000); assert_eq!( parse_poll_interval(config.poll_interval.as_deref()), Duration::from_secs(300) @@ -3182,7 +3354,7 @@ mod tests { } #[test] - fn test_config_deserialization_bulk_mode() { + fn given_bulk_toml_should_deserialize_mode() { let toml_str = r#" jdbc_url = "jdbc:h2:mem:test" driver_class = "org.h2.Driver" @@ -3201,7 +3373,7 @@ mod tests { } #[test] - fn test_config_deserialization_invalid_mode_fails() { + fn given_invalid_mode_should_fail_deserialization() { let toml_str = r#" jdbc_url = "jdbc:h2:mem:test" driver_class = "org.h2.Driver" @@ -3273,7 +3445,7 @@ mod tests { // ========================================================================= #[test] - fn test_state_restoration_with_malformed_bytes_falls_back_to_default() { + fn given_malformed_state_should_start_fresh() { let connector_state = ConnectorState(vec![0xFF, 0xFE, 0xFD, 0x00]); let source = JdbcSource::new(1, base_config(), Some(connector_state)); @@ -3284,7 +3456,7 @@ mod tests { } #[test] - fn test_state_restoration_with_empty_bytes_falls_back_to_default() { + fn given_empty_state_should_start_fresh() { let connector_state = ConnectorState(vec![]); let source = JdbcSource::new(1, base_config(), Some(connector_state)); @@ -3294,7 +3466,7 @@ mod tests { } #[test] - fn test_state_restoration_none_uses_initial_offset() { + fn given_no_state_should_use_initial_offset() { let config = JdbcSourceConfig { query: "SELECT * FROM orders WHERE id > {last_offset}".to_string(), tracking_column: Some("id".to_string()), @@ -3309,7 +3481,7 @@ mod tests { } #[test] - fn test_state_restoration_none_without_initial_offset() { + fn given_no_state_or_initial_offset_should_start_fresh() { let config = JdbcSourceConfig { query: "SELECT * FROM products".to_string(), poll_interval: Some("60s".to_string()), @@ -3326,14 +3498,14 @@ mod tests { // ========================================================================= #[test] - fn test_extract_offset_value_with_integer() { + fn given_integer_cursor_should_extract_as_string() { let mut row = serde_json::Map::new(); row.insert("id".to_string(), serde_json::json!(42)); assert_eq!(extract_offset_value(&row, "id"), Some("42".to_string())); } #[test] - fn test_extract_offset_value_with_string() { + fn given_string_cursor_should_extract_unchanged() { let mut row = serde_json::Map::new(); row.insert( "updated_at".to_string(), @@ -3346,7 +3518,7 @@ mod tests { } #[test] - fn test_extract_offset_value_with_float() { + fn given_float_cursor_should_extract_as_string() { let mut row = serde_json::Map::new(); row.insert("version".to_string(), serde_json::json!(3.5)); assert_eq!( @@ -3356,7 +3528,7 @@ mod tests { } #[test] - fn test_extract_offset_value_with_null() { + fn given_null_cursor_should_not_extract() { let mut row = serde_json::Map::new(); row.insert("id".to_string(), serde_json::Value::Null); // A SQL NULL tracking value must not become the persisted offset. @@ -3364,7 +3536,7 @@ mod tests { } #[test] - fn test_extract_offset_value_missing_column() { + fn given_missing_cursor_column_should_not_extract() { let row = serde_json::Map::new(); assert_eq!(extract_offset_value(&row, "nonexistent"), None); } @@ -3374,7 +3546,7 @@ mod tests { // ========================================================================= #[test] - fn test_build_query_incremental_no_offset_no_initial_removes_where_clause() { + fn given_no_incremental_offset_should_remove_where_clause() { let config = JdbcSourceConfig { query: "SELECT * FROM users WHERE {tracking_column} > {last_offset} ORDER BY id" .to_string(), @@ -3397,7 +3569,7 @@ mod tests { } #[test] - fn test_build_query_rejects_unresolved_placeholder() { + fn given_unresolved_placeholder_should_be_rejected() { // Incremental, no offset, and a non-canonical predicate that the // auto-remove does not match: the unresolved {last_offset} must produce // an error rather than being shipped to the driver as invalid SQL. @@ -3411,7 +3583,7 @@ mod tests { } #[test] - fn test_build_query_bulk_mode_ignores_offset() { + fn given_bulk_mode_should_ignore_offset() { let config = JdbcSourceConfig { query: "SELECT * FROM users ORDER BY id".to_string(), tracking_column: Some("id".to_string()), @@ -3433,7 +3605,7 @@ mod tests { // ========================================================================= #[test] - fn test_mode_serialization_roundtrip() { + fn given_mode_should_round_trip_through_json() { let incremental = Mode::Incremental; let serialized = serde_json::to_string(&incremental).unwrap(); assert_eq!(serialized, r#""incremental""#); @@ -3448,7 +3620,7 @@ mod tests { } #[test] - fn test_mode_deserialization_rejects_unknown() { + fn given_unknown_mode_should_fail_deserialization() { let result = serde_json::from_str::(r#""streaming""#); assert!( result.is_err(), @@ -3461,7 +3633,7 @@ mod tests { // ========================================================================= #[test] - fn test_config_debug_does_not_leak_password() { + fn given_config_debug_should_not_leak_password() { let config = JdbcSourceConfig { jdbc_url: SecretString::from("jdbc:mysql://root:SuperSecret@localhost/db"), driver_class: "com.mysql.cj.jdbc.Driver".to_string(), @@ -3578,7 +3750,7 @@ mod tests { } #[test] - fn test_config_debug_without_password() { + fn given_config_without_password_should_debug_safely() { let debug_output = format!("{:?}", base_config()); // Should not panic and should contain the struct name assert!(debug_output.contains("JdbcSourceConfig")); diff --git a/core/integration/Cargo.toml b/core/integration/Cargo.toml index ba6ff18a41..d25e6167c8 100644 --- a/core/integration/Cargo.toml +++ b/core/integration/Cargo.toml @@ -96,6 +96,7 @@ secrecy = { workspace = true } serde = { workspace = true } serde_json = { workspace = true } serial_test = { workspace = true } +sha2 = { workspace = true } sqlparser = { workspace = true } sqlx = { workspace = true } sysinfo = { workspace = true } diff --git a/core/integration/tests/connectors/fixtures/jdbc.rs b/core/integration/tests/connectors/fixtures/jdbc.rs new file mode 100644 index 0000000000..ad4f1ee0fc --- /dev/null +++ b/core/integration/tests/connectors/fixtures/jdbc.rs @@ -0,0 +1,358 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use super::postgres::PostgresContainer; +use async_trait::async_trait; +use integration::harness::{TestBinaryError, TestFixture, seeds}; +use sha2::{Digest, Sha256}; +use sqlx::{Pool, Postgres}; +use std::collections::HashMap; +use std::io::Cursor; +use zip::ZipArchive; + +const DRIVER_CLASS_ENTRY: &str = "org/postgresql/Driver.class"; +const POSTGRES_DRIVER_SHA256: &str = + "49bba9c3200d4f64ae73903d56ce1bd09c74517dfe31acb44745506b4fcede53"; + +const ENV_JDBC_URL: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_JDBC_URL"; +const ENV_DRIVER_CLASS: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_CLASS"; +const ENV_DRIVER_JAR_PATH: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_JAR_PATH"; +const ENV_USERNAME: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_USERNAME"; +const ENV_PASSWORD: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_PASSWORD"; +const ENV_QUERY: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_QUERY"; +const ENV_POLL_INTERVAL: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_POLL_INTERVAL"; +const ENV_BATCH_SIZE: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE"; +const ENV_MODE: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_MODE"; +const ENV_TRACKING_COLUMN: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_TRACKING_COLUMN"; +const ENV_INITIAL_OFFSET: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INITIAL_OFFSET"; +const ENV_INCLUDE_METADATA: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INCLUDE_METADATA"; +const ENV_QUERY_TIMEOUT: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_QUERY_TIMEOUT_MS"; +const ENV_STREAM: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_STREAM"; +const ENV_TOPIC: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_TOPIC"; +const ENV_SCHEMA: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_SCHEMA"; + +#[derive(Clone, Copy)] +enum Scenario { + Empty, + Incremental, + TextCursor, + LargeResult, + BulkOverflow, + TieBoundary, +} + +struct JdbcPostgresFixture { + container: PostgresContainer, + jdbc_url: String, + driver_jar_path: String, +} + +impl JdbcPostgresFixture { + async fn setup(scenario: Scenario) -> Result { + let container = PostgresContainer::start().await?; + let host_and_port = container + .connection_string() + .rsplit('@') + .next() + .ok_or_else(|| fixture_error("PostgreSQL connection string has no host"))?; + let fixture = Self { + jdbc_url: format!("jdbc:postgresql://{host_and_port}/postgres"), + driver_jar_path: postgres_driver_jar().await?, + container, + }; + fixture.seed_database(scenario).await?; + Ok(fixture) + } + + async fn create_pool(&self) -> Result, TestBinaryError> { + self.container.create_pool().await + } + + async fn seed_database(&self, scenario: Scenario) -> Result<(), TestBinaryError> { + let pool = self.create_pool().await?; + let statements: &[&str] = match scenario { + Scenario::Empty => &[], + Scenario::Incremental => &[ + "CREATE TABLE inc_test (id INT PRIMARY KEY, name TEXT)", + "INSERT INTO inc_test (id, name) VALUES (1, 'a'), (2, 'b'), (3, 'c')", + ], + Scenario::TextCursor => &[ + "CREATE TABLE text_cursor_test (id INT PRIMARY KEY, cursor_value TEXT UNIQUE NOT NULL)", + r"INSERT INTO text_cursor_test (id, cursor_value) VALUES (1, E'a\\b'), (2, 'z')", + ], + Scenario::LargeResult => &[ + "CREATE TABLE big_test (id INT PRIMARY KEY, name TEXT, val NUMERIC(12,2))", + "INSERT INTO big_test (id, name, val) SELECT g, 'row_' || g, (g * 1.5)::numeric(12,2) FROM generate_series(1, 300) g", + ], + Scenario::BulkOverflow => &[ + "CREATE TABLE trunc_test (id INT PRIMARY KEY)", + "INSERT INTO trunc_test (id) SELECT generate_series(1, 5)", + ], + Scenario::TieBoundary => &[ + "CREATE TABLE tie_test (id INT PRIMARY KEY, position INT NOT NULL)", + "INSERT INTO tie_test (id, position) VALUES (1, 1), (2, 2), (3, 2), (4, 3)", + ], + }; + for statement in statements { + sqlx::query(*statement) + .execute(&pool) + .await + .map_err(|error| fixture_error(format!("failed to seed PostgreSQL: {error}")))?; + } + pool.close().await; + Ok(()) + } + + fn envs( + &self, + query: &str, + mode: &str, + batch_size: u32, + tracking_column: &str, + initial_offset: Option<&str>, + ) -> HashMap { + let mut envs = HashMap::from([ + (ENV_JDBC_URL.to_string(), self.jdbc_url.clone()), + ( + ENV_DRIVER_CLASS.to_string(), + "org.postgresql.Driver".to_string(), + ), + ( + ENV_DRIVER_JAR_PATH.to_string(), + self.driver_jar_path.clone(), + ), + (ENV_USERNAME.to_string(), "postgres".to_string()), + (ENV_PASSWORD.to_string(), "postgres".to_string()), + (ENV_QUERY.to_string(), query.to_string()), + (ENV_POLL_INTERVAL.to_string(), "100ms".to_string()), + (ENV_BATCH_SIZE.to_string(), batch_size.to_string()), + (ENV_MODE.to_string(), mode.to_string()), + (ENV_TRACKING_COLUMN.to_string(), tracking_column.to_string()), + (ENV_INCLUDE_METADATA.to_string(), "true".to_string()), + (ENV_STREAM.to_string(), seeds::names::STREAM.to_string()), + (ENV_TOPIC.to_string(), seeds::names::TOPIC.to_string()), + (ENV_SCHEMA.to_string(), "json".to_string()), + ]); + if let Some(initial_offset) = initial_offset { + envs.insert(ENV_INITIAL_OFFSET.to_string(), initial_offset.to_string()); + } + envs + } +} + +macro_rules! jdbc_fixture { + ($name:ident, $scenario:expr, $query:expr, $mode:expr, $batch_size:expr, $tracking:expr, $offset:expr) => { + pub struct $name { + base: JdbcPostgresFixture, + } + + #[async_trait] + impl TestFixture for $name { + async fn setup() -> Result { + Ok(Self { + base: JdbcPostgresFixture::setup($scenario).await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + self.base + .envs($query, $mode, $batch_size, $tracking, $offset) + } + } + }; +} + +jdbc_fixture!( + JdbcBulkFixture, + Scenario::Empty, + "SELECT 1 AS id, 'test' AS name", + "bulk", + 100, + "id", + None +); + +impl JdbcIncrementalFixture { + pub async fn create_pool(&self) -> Result, TestBinaryError> { + self.base.create_pool().await + } +} + +impl JdbcRecoveryFixture { + pub async fn create_pool(&self) -> Result, TestBinaryError> { + self.base.create_pool().await + } +} +jdbc_fixture!( + JdbcBulkRowsFixture, + Scenario::Empty, + "SELECT * FROM (VALUES (1, 'alice', true), (2, 'bob', false), (3, 'carol', true)) AS t(id, name, active)", + "bulk", + 100, + "id", + None +); +jdbc_fixture!( + JdbcMetadataFixture, + Scenario::Empty, + "SELECT 42 AS value", + "bulk", + 100, + "id", + None +); +jdbc_fixture!( + JdbcIncrementalFixture, + Scenario::Incremental, + "SELECT id, name FROM inc_test WHERE id > {last_offset} ORDER BY id", + "incremental", + 2, + "id", + None +); +jdbc_fixture!( + JdbcTextCursorFixture, + Scenario::TextCursor, + "SELECT id, cursor_value FROM text_cursor_test WHERE cursor_value > {last_offset} ORDER BY cursor_value", + "incremental", + 100, + "cursor_value", + Some(r"a\b") +); +jdbc_fixture!( + JdbcLargeResultFixture, + Scenario::LargeResult, + "SELECT id, name, val FROM big_test ORDER BY id", + "bulk", + 5000, + "id", + None +); +jdbc_fixture!( + JdbcRecoveryFixture, + Scenario::Empty, + "SELECT id, name FROM recover_test ORDER BY id", + "bulk", + 100, + "id", + None +); +jdbc_fixture!( + JdbcBulkOverflowFixture, + Scenario::BulkOverflow, + "SELECT id FROM trunc_test ORDER BY id", + "bulk", + 2, + "id", + None +); +jdbc_fixture!( + JdbcTieBoundaryFixture, + Scenario::TieBoundary, + "SELECT id, position FROM tie_test WHERE position > {last_offset} ORDER BY position, id", + "incremental", + 2, + "position", + None +); + +pub struct JdbcQueryTimeoutFixture { + base: JdbcPostgresFixture, +} + +#[async_trait] +impl TestFixture for JdbcQueryTimeoutFixture { + async fn setup() -> Result { + Ok(Self { + base: JdbcPostgresFixture::setup(Scenario::Empty).await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self + .base + .envs("SELECT 1 AS id FROM pg_sleep(2)", "bulk", 100, "id", None); + envs.insert(ENV_QUERY_TIMEOUT.to_string(), "1000".to_string()); + envs + } +} + +async fn postgres_driver_jar() -> Result { + let target_dir = std::env::var("CARGO_TARGET_DIR").unwrap_or_else(|_| "target".to_string()); + let jdbc_test_dir = format!("{target_dir}/test-jdbc-drivers"); + let jar_path = format!("{jdbc_test_dir}/postgresql-42.7.1.jar"); + std::fs::create_dir_all(&jdbc_test_dir) + .map_err(|error| fixture_error(format!("failed to create driver cache: {error}")))?; + + if std::path::Path::new(&jar_path).exists() { + match std::fs::read(&jar_path) { + Ok(bytes) if is_expected_driver_jar(&bytes) => return canonical_path(&jar_path), + _ => { + std::fs::remove_file(&jar_path).map_err(|error| { + fixture_error(format!("failed to remove invalid cached driver: {error}")) + })?; + } + } + } + + let response = reqwest::get( + "https://repo1.maven.org/maven2/org/postgresql/postgresql/42.7.1/postgresql-42.7.1.jar", + ) + .await + .map_err(|error| fixture_error(format!("failed to download JDBC driver: {error}")))? + .error_for_status() + .map_err(|error| fixture_error(format!("JDBC driver download failed: {error}")))?; + let bytes = response + .bytes() + .await + .map_err(|error| fixture_error(format!("failed to read JDBC driver: {error}")))?; + if !is_expected_driver_jar(&bytes) { + return Err(fixture_error(format!( + "downloaded JDBC driver failed SHA-256 or archive validation for {DRIVER_CLASS_ENTRY}" + ))); + } + + let temporary_path = format!("{jar_path}.{}.tmp", std::process::id()); + std::fs::write(&temporary_path, &bytes) + .map_err(|error| fixture_error(format!("failed to cache JDBC driver: {error}")))?; + std::fs::rename(&temporary_path, &jar_path) + .map_err(|error| fixture_error(format!("failed to publish JDBC driver cache: {error}")))?; + canonical_path(&jar_path) +} + +fn is_expected_driver_jar(bytes: &[u8]) -> bool { + if format!("{:x}", Sha256::digest(bytes)) != POSTGRES_DRIVER_SHA256 { + return false; + } + let Ok(mut archive) = ZipArchive::new(Cursor::new(bytes)) else { + return false; + }; + archive.by_name(DRIVER_CLASS_ENTRY).is_ok() +} + +fn canonical_path(path: &str) -> Result { + std::fs::canonicalize(path) + .map(|path| path.to_string_lossy().into_owned()) + .map_err(|error| fixture_error(format!("failed to resolve JDBC driver path: {error}"))) +} + +fn fixture_error(message: impl Into) -> TestBinaryError { + TestBinaryError::FixtureSetup { + fixture_type: "JDBC PostgreSQL".to_string(), + message: message.into(), + } +} diff --git a/core/integration/tests/connectors/fixtures/mod.rs b/core/integration/tests/connectors/fixtures/mod.rs index cddee53ab1..cdb2bb5a3e 100644 --- a/core/integration/tests/connectors/fixtures/mod.rs +++ b/core/integration/tests/connectors/fixtures/mod.rs @@ -25,6 +25,7 @@ mod floci; mod http; mod iceberg; mod influxdb; +mod jdbc; mod meilisearch; mod mongodb; mod postgres; @@ -75,6 +76,11 @@ pub use influxdb::{ InfluxDbSinkNoMetadataFixture, InfluxDbSinkNsPrecisionFixture, InfluxDbSinkTextFixture, InfluxDbSourceFixture, InfluxDbSourceRawFixture, InfluxDbSourceTextFixture, }; +pub use jdbc::{ + JdbcBulkFixture, JdbcBulkOverflowFixture, JdbcBulkRowsFixture, JdbcIncrementalFixture, + JdbcLargeResultFixture, JdbcMetadataFixture, JdbcQueryTimeoutFixture, JdbcRecoveryFixture, + JdbcTextCursorFixture, JdbcTieBoundaryFixture, +}; pub use meilisearch::{MeilisearchOps, MeilisearchSinkFixture, TEST_INDEX}; pub use mongodb::{ MongoDbOps, MongoDbSinkAutoCreateFixture, MongoDbSinkBatchFixture, MongoDbSinkFailpointFixture, diff --git a/core/integration/tests/connectors/fixtures/postgres/container.rs b/core/integration/tests/connectors/fixtures/postgres/container.rs index 508607f0c6..68ac156f7d 100644 --- a/core/integration/tests/connectors/fixtures/postgres/container.rs +++ b/core/integration/tests/connectors/fixtures/postgres/container.rs @@ -128,7 +128,7 @@ pub struct PostgresContainer { } impl PostgresContainer { - pub(super) async fn start() -> Result { + pub(crate) async fn start() -> Result { Self::start_with_image(postgres::Postgres::default().into()).await } @@ -180,4 +180,8 @@ impl PostgresContainer { message: format!("Failed to connect: {e}"), }) } + + pub(crate) fn connection_string(&self) -> &str { + &self.connection_string + } } diff --git a/core/integration/tests/connectors/fixtures/postgres/mod.rs b/core/integration/tests/connectors/fixtures/postgres/mod.rs index 7f186d406b..0d8b138f39 100644 --- a/core/integration/tests/connectors/fixtures/postgres/mod.rs +++ b/core/integration/tests/connectors/fixtures/postgres/mod.rs @@ -21,6 +21,7 @@ mod sink; mod source; pub use cdc::{PostgresSourceCdcFixture, PostgresSourceCdcSlowPollFixture}; +pub(crate) use container::PostgresContainer; pub use container::{PostgresOps, PostgresSourceOps}; pub use sink::{ POSTGRES_LARGE_BATCH_SIZE, PostgresSinkByteaFixture, PostgresSinkFixture, diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs index b028f774cc..8af10f1da8 100644 --- a/core/integration/tests/connectors/jdbc/jdbc_source.rs +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -15,850 +15,341 @@ // specific language governing permissions and limitations // under the License. -use crate::connectors::{ConnectorsRuntime, IggySetup, setup_runtime}; -use serial_test::serial; -use sqlx::postgres::PgPoolOptions; -use std::collections::{BTreeSet, HashMap}; -use std::io::Cursor; +use crate::connectors::fixtures::{ + JdbcBulkFixture, JdbcBulkOverflowFixture, JdbcBulkRowsFixture, JdbcIncrementalFixture, + JdbcLargeResultFixture, JdbcMetadataFixture, JdbcQueryTimeoutFixture, JdbcRecoveryFixture, + JdbcTextCursorFixture, JdbcTieBoundaryFixture, +}; +use iggy::prelude::IggyClient; +use iggy_common::{Consumer, Identifier, MessageClient, PollingStrategy}; +use integration::harness::{TestHarness, seeds}; +use integration::iggy_harness; +use std::collections::BTreeSet; use std::time::{Duration, Instant}; -use testcontainers_modules::postgres::Postgres; -use testcontainers_modules::testcontainers::ContainerAsync; -use testcontainers_modules::testcontainers::runners::AsyncRunner; use tokio::time::sleep; -use tracing::info; -use zip::ZipArchive; -const POSTGRES_USER: &str = "postgres"; -const POSTGRES_PASSWORD: &str = "postgres"; -const POSTGRES_DB: &str = "postgres"; - -/// How long to wait for the source to deliver what a test expects. -/// -/// A deadline, not an attempt count: an attempt count doubles as a cap on the -/// total a collect-until-N loop can ever drain (attempts x batch size), so a -/// test asking for more than a fraction of that cap fails on a slow runner even -/// though the source delivered everything. const POLL_TIMEOUT: Duration = Duration::from_secs(30); -/// Delay between poll attempts -const POLL_INTERVAL: Duration = Duration::from_millis(500); -/// Messages requested per poll, kept well above any single expectation below so -/// the deadline alone bounds how long the source may take. +const POLL_INTERVAL: Duration = Duration::from_millis(100); const POLL_BATCH: u32 = 500; -/// Setup Postgres container with test data -async fn setup_postgres_container() --> Result<(ContainerAsync, String, String), Box> { - info!("Starting Postgres container for JDBC testing..."); - - let postgres = Postgres::default().start().await?; - - let host = postgres.get_host().await?; - let port = postgres.get_host_port_ipv4(5432).await?; - let jdbc_url: String = format!("jdbc:postgresql://{}:{}/{}", host, port, POSTGRES_DB); - - let postgres_jar: String = get_postgres_driver_jar().await?; - - info!("Postgres container started at {}:{}", host, port); - Ok((postgres, jdbc_url, postgres_jar)) -} - -/// The class the connector hands to `Class.forName`. Checking that the archive -/// actually contains it is what makes the integrity check below meaningful. -const DRIVER_CLASS_ENTRY: &str = "org/postgresql/Driver.class"; - -/// Whether `bytes` is a driver JAR the JVM can actually load from: a readable -/// ZIP archive that contains the driver class. -/// -/// Magic bytes plus a minimum size are not enough. A download truncated past -/// that size still starts with the ZIP local-file-header magic while its central -/// directory is gone, so it passes a size check yet cannot be read as an -/// archive. It is then cached and reused by every later JDBC test in the job, -/// and the only symptom is a `ClassNotFoundException` on `Class.forName` that -/// fails the connector's `open()`, so the runtime skips the source and every -/// JDBC test fails having received no messages at all. Opening the archive and -/// looking the entry up tests the property the JVM needs, and a cached file that -/// fails it is deleted and re-downloaded rather than reused. -fn looks_like_jar(bytes: &[u8]) -> bool { - let Ok(mut archive) = ZipArchive::new(Cursor::new(bytes)) else { - return false; - }; - archive.by_name(DRIVER_CLASS_ENTRY).is_ok() -} - -/// Get the PostgreSQL JDBC driver, downloading and integrity-checking it if a -/// valid copy is not already cached. Downloads from Maven Central, verifies the -/// bytes open as an archive holding the driver class before persisting, and -/// writes via a temp file + atomic rename so a partial or corrupt download can -/// never be cached and reused. -async fn get_postgres_driver_jar() -> Result> { - let target_dir = std::env::var("CARGO_TARGET_DIR").unwrap_or_else(|_| "target".to_string()); - let jdbc_test_dir = format!("{target_dir}/test-jdbc-drivers"); - let jar_path = format!("{jdbc_test_dir}/postgresql-42.7.1.jar"); - - std::fs::create_dir_all(&jdbc_test_dir)?; - - // Reuse the cached jar only if it is actually a valid JAR; a previously - // cached bad download must self-heal rather than fail every run. - if std::path::Path::new(&jar_path).exists() { - match std::fs::read(&jar_path) { - Ok(bytes) if looks_like_jar(&bytes) => { - info!("PostgreSQL JDBC driver found at {jar_path}"); - return Ok(std::fs::canonicalize(&jar_path)? - .to_string_lossy() - .to_string()); - } - _ => { - info!("Cached JDBC driver at {jar_path} is invalid; re-downloading"); - let _ = std::fs::remove_file(&jar_path); - } - } - } - - info!("Downloading PostgreSQL JDBC driver..."); - let download_url = - "https://repo1.maven.org/maven2/org/postgresql/postgresql/42.7.1/postgresql-42.7.1.jar"; - let response = reqwest::get(download_url).await?; - if !response.status().is_success() { - return Err(format!("Failed to download driver: HTTP {}", response.status()).into()); - } - let bytes = response.bytes().await?; - if !looks_like_jar(&bytes) { - return Err(format!( - "Downloaded JDBC driver is not a readable JAR containing {DRIVER_CLASS_ENTRY} \ - ({} bytes); the download was truncated or the endpoint returned an error page", - bytes.len() - ) - .into()); - } - - // Write to a unique temp file, then atomically rename, so a crash or a - // concurrent test never observes a half-written jar at `jar_path`. - let tmp_path = format!("{jar_path}.{}.tmp", std::process::id()); - std::fs::write(&tmp_path, &bytes)?; - std::fs::rename(&tmp_path, &jar_path)?; - - info!("PostgreSQL JDBC driver downloaded to {jar_path}"); - Ok(std::fs::canonicalize(&jar_path)? - .to_string_lossy() - .to_string()) -} +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn bulk_query_produces_message_to_iggy(harness: &TestHarness, _fixture: JdbcBulkFixture) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_bulk", 1).await; + assert!(!messages.is_empty(), "expected a JDBC message"); -/// Build the environment variables for a JDBC Postgres source connector. -fn build_jdbc_env( - jdbc_url: &str, - postgres_jar: &str, - query: &str, - mode: &str, - iggy_setup: &IggySetup, -) -> HashMap { - let mut envs = HashMap::new(); - - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_JDBC_URL".to_owned(), - jdbc_url.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_CLASS".to_owned(), - "org.postgresql.Driver".to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_JAR_PATH".to_owned(), - postgres_jar.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_USERNAME".to_owned(), - POSTGRES_USER.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_PASSWORD".to_owned(), - POSTGRES_PASSWORD.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_QUERY".to_owned(), - query.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_POLL_INTERVAL".to_owned(), - "1s".to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), - "100".to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_MODE".to_owned(), - mode.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_SNAKE_CASE_COLUMNS".to_owned(), - "false".to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INCLUDE_METADATA".to_owned(), - "true".to_owned(), - ); - - // Stream configuration - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_STREAM".to_owned(), - iggy_setup.stream.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_TOPIC".to_owned(), - iggy_setup.topic.to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_SCHEMA".to_owned(), - "json".to_owned(), - ); - - envs -} - -/// Poll until at least `expected_count` messages have been collected or -/// `POLL_TIMEOUT` elapses, returning them deserialized. -async fn poll_messages_with_retry( - client: &crate::connectors::ConnectorsIggyClient, - expected_count: usize, -) -> Vec { - let deadline = Instant::now() + POLL_TIMEOUT; - let mut received: Vec = Vec::new(); - - loop { - let polled_messages = client - .get_messages(POLL_BATCH) - .await - .expect("Failed to poll messages"); - - for msg in &polled_messages.messages { - if let Ok(value) = serde_json::from_slice::(&msg.payload) { - received.push(value); - } - } - - if received.len() >= expected_count { - info!("Received {} messages", received.len()); - return received; - } - - if Instant::now() >= deadline { - return received; - } - - sleep(POLL_INTERVAL).await; - } -} - -/// Setup connector runtime with JDBC source for Postgres -async fn setup_jdbc_postgres_source( - jdbc_url: &str, - postgres_jar: &str, - query: &str, - mode: &str, -) -> Result< - (ConnectorsRuntime, crate::connectors::ConnectorsIggyClient), - Box, -> { - let iggy_setup = IggySetup::default(); - let envs = build_jdbc_env(jdbc_url, postgres_jar, query, mode, &iggy_setup); - - let mut runtime = setup_runtime(); - runtime - .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) - .await; - - let client = runtime.create_client().await; - Ok((runtime, client)) -} - -/// Test: basic bulk mode query produces messages with correct structure -#[tokio::test] -#[serial] -async fn bulk_query_produces_message_to_iggy() { - let (_postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - let query = "SELECT 1 as id, 'test' as name"; - let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") - .await - .expect("Failed to setup runtime"); - - info!("Waiting for JDBC connector to poll from Postgres..."); - let messages = poll_messages_with_retry(&client, 1).await; - - assert!( - !messages.is_empty(), - "Expected at least 1 message from JDBC Postgres source" - ); - - // Verify message structure: should have metadata wrapping (include_metadata=true) let first = &messages[0]; - assert!( - first.get("data").is_some(), - "Expected 'data' field in message (include_metadata=true), got: {}", - first - ); assert_eq!( - first.get("operation_type").and_then(|v| v.as_str()), - Some("SELECT"), - "Expected operation_type=SELECT" + first.get("operation_type").and_then(|value| value.as_str()), + Some("SELECT") ); - - // Verify the actual data content - let data = first.get("data").unwrap(); + let data = first.get("data").expect("metadata should contain data"); + assert_eq!(data.get("id").and_then(|value| value.as_i64()), Some(1)); assert_eq!( - data.get("id").and_then(|v| v.as_i64()), - Some(1), - "Expected id=1 in data" - ); - assert_eq!( - data.get("name").and_then(|v| v.as_str()), - Some("test"), - "Expected name='test' in data" + data.get("name").and_then(|value| value.as_str()), + Some("test") ); } -/// Test: bulk mode with multiple rows from an actual table -#[tokio::test] -#[serial] -async fn bulk_query_produces_multiple_rows_to_iggy() { - let (postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - // Use a multi-row SELECT to simulate table data without needing DDL - let query = r#" - SELECT * FROM (VALUES - (1, 'alice', true), - (2, 'bob', false), - (3, 'carol', true) - ) AS t(id, name, active) - "#; - - let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") - .await - .expect("Failed to setup runtime"); - - info!("Waiting for JDBC connector to poll multiple rows..."); - let messages = poll_messages_with_retry(&client, 3).await; - - assert!( - messages.len() >= 3, - "Expected at least 3 messages, got {}", - messages.len() - ); - - // Verify each row has the expected structure - for msg in &messages[..3] { - let data = msg.get("data").expect("Missing 'data' field"); - assert!(data.get("id").is_some(), "Missing 'id' column in row data"); - assert!( - data.get("name").is_some(), - "Missing 'name' column in row data" - ); - assert!( - data.get("active").is_some(), - "Missing 'active' column in row data" - ); +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn bulk_query_produces_multiple_rows_to_iggy( + harness: &TestHarness, + _fixture: JdbcBulkRowsFixture, +) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_bulk_rows", 3).await; + assert!(messages.len() >= 3, "expected three JDBC messages"); + + for message in &messages[..3] { + let data = message.get("data").expect("metadata should contain data"); + assert!(data.get("id").is_some()); + assert!(data.get("name").is_some()); + assert!(data.get("active").is_some()); } - - // Verify specific values for the first row - let first_data = messages[0].get("data").unwrap(); - assert_eq!(first_data.get("id").and_then(|v| v.as_i64()), Some(1)); + let first = messages[0].get("data").unwrap(); + assert_eq!(first.get("id").and_then(|value| value.as_i64()), Some(1)); assert_eq!( - first_data.get("name").and_then(|v| v.as_str()), + first.get("name").and_then(|value| value.as_str()), Some("alice") ); - - // Keep container alive until assertions complete - drop(postgres_container); } -/// Test: message contains timestamp field when metadata is enabled -#[tokio::test] -#[serial] -async fn source_includes_metadata_fields_when_enabled() { - let (_postgres_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - let query = "SELECT 42 as value"; - let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") - .await - .expect("Failed to setup runtime"); - - let messages = poll_messages_with_retry(&client, 1).await; - assert!(!messages.is_empty(), "Expected at least 1 message"); - - let msg = &messages[0]; - - // Verify all metadata fields are present - assert!( - msg.get("timestamp").is_some(), - "Missing 'timestamp' metadata field" - ); - assert!( - msg.get("operation_type").is_some(), - "Missing 'operation_type' metadata field" - ); - assert!(msg.get("data").is_some(), "Missing 'data' metadata field"); - - // table_name should be null for SELECT queries without a specific table - // (this is expected behavior for computed queries) - assert!( - msg.get("table_name").is_some(), - "Missing 'table_name' metadata field" - ); -} - -/// Derive a sqlx (`postgres://`) URL from the connector's JDBC URL so the test -/// can seed the table the source reads from. -fn pg_sqlx_url(jdbc_url: &str) -> String { - let host_and_db = jdbc_url - .strip_prefix("jdbc:postgresql://") - .unwrap_or(jdbc_url); - format!("postgres://{POSTGRES_USER}:{POSTGRES_PASSWORD}@{host_and_db}") +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn source_includes_metadata_fields_when_enabled( + harness: &TestHarness, + _fixture: JdbcMetadataFixture, +) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_metadata", 1).await; + let message = messages.first().expect("expected a JDBC message"); + assert!(message.get("timestamp").is_some()); + assert!(message.get("operation_type").is_some()); + assert!(message.get("data").is_some()); + assert!(message.get("table_name").is_some()); } -/// Poll until every id in `expected` has been seen, or `timeout` elapses. -/// Returns the distinct ids seen, ascending, plus the raw number of messages -/// received. -/// -/// A source can deliver a row more than once: bulk mode re-runs its query every -/// poll interval, and delivery is at-least-once, so a nacked batch is re-read. -/// Matching on the distinct ids seen, rather than on a fixed-length prefix of the -/// received messages, keeps a re-delivered batch from failing an otherwise -/// healthy run. The returned count separates "the source delivered nothing" from -/// "it delivered messages that did not carry the expected `data.id`", which a -/// bare id list cannot express: unmatched messages are dropped silently. -async fn poll_until_ids_seen( - client: &crate::connectors::ConnectorsIggyClient, - expected: &[i64], - timeout: Duration, -) -> (Vec, usize) { - let deadline = Instant::now() + timeout; - let mut seen: BTreeSet = BTreeSet::new(); - let mut received = 0usize; - - loop { - let polled_messages = client - .get_messages(POLL_BATCH) - .await - .expect("Failed to poll messages"); - - for msg in &polled_messages.messages { - received += 1; - if let Ok(value) = serde_json::from_slice::(&msg.payload) - && let Some(id) = value - .get("data") - .and_then(|data| data.get("id")) - .and_then(|id| id.as_i64()) - { - seen.insert(id); - } - } - - if expected.iter().all(|id| seen.contains(id)) { - info!("Saw expected ids {expected:?} in {received} received messages"); - break; - } - - if Instant::now() >= deadline { - break; - } - - sleep(POLL_INTERVAL).await; - } - - (seen.into_iter().collect(), received) -} - -/// Test: incremental mode advances its tracking offset across polls; newly -/// inserted rows are delivered exactly once and previously read rows are not -/// re-delivered. -#[tokio::test] -#[serial] -async fn incremental_mode_advances_offset_across_polls() { - let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - // Seed a real table BEFORE the source starts polling. - let pool = PgPoolOptions::new() - .max_connections(2) - .connect(&pg_sqlx_url(&jdbc_url)) - .await - .expect("Failed to connect to Postgres for seeding"); - sqlx::query("CREATE TABLE inc_test (id INT PRIMARY KEY, name TEXT)") - .execute(&pool) - .await - .expect("Failed to create table"); - sqlx::query("INSERT INTO inc_test (id, name) VALUES (1, 'a'), (2, 'b'), (3, 'c')") - .execute(&pool) - .await - .expect("Failed to insert initial rows"); - - let query = "SELECT id, name FROM inc_test WHERE id > {last_offset} ORDER BY id"; - let iggy_setup = IggySetup::default(); - let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "incremental", &iggy_setup); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), - "2".to_owned(), - ); - let mut runtime = setup_runtime(); - runtime - .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) - .await; - let client = runtime.create_client().await; - - // First batch: ids 1..3. - let (first_ids, first_received) = poll_until_ids_seen(&client, &[1, 2, 3], POLL_TIMEOUT).await; +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn incremental_mode_advances_offset_across_polls( + harness: &TestHarness, + fixture: JdbcIncrementalFixture, +) { + let client = harness.root_client().await.expect("root client"); + let consumer = consumer("jdbc_incremental"); + let (first_ids, first_received) = + poll_until_ids_seen(&client, &consumer, &[1, 2, 3], POLL_TIMEOUT).await; assert_eq!( first_ids, vec![1, 2, 3], - "Expected ids 1,2,3 on the first poll; got ids {first_ids:?} from {first_received} \ - received message(s)" + "expected ids 1,2,3, got {first_ids:?} from {first_received} messages" ); - // Insert more rows; only these (id > last_offset) should arrive next. + let pool = fixture.create_pool().await.expect("PostgreSQL pool"); sqlx::query("INSERT INTO inc_test (id, name) VALUES (4, 'd'), (5, 'e')") .execute(&pool) .await - .expect("Failed to insert additional rows"); + .expect("insert additional rows"); + pool.close().await; - let (second_ids, second_received) = poll_until_ids_seen(&client, &[4, 5], POLL_TIMEOUT).await; + let (second_ids, second_received) = + poll_until_ids_seen(&client, &consumer, &[4, 5], POLL_TIMEOUT).await; assert_eq!( second_ids, vec![4, 5], - "Expected only the new ids 4,5 (offset must have advanced past 3); got ids \ - {second_ids:?} from {second_received} received message(s)" + "expected only ids 4,5, got {second_ids:?} from {second_received} messages" ); } -/// Text cursors are bound as JDBC parameters. A backslash must reach PostgreSQL -/// unchanged instead of being doubled for MySQL literal syntax, which would -/// move the comparison boundary and re-read the row at the saved cursor. -#[tokio::test] -#[serial] -async fn incremental_text_offset_with_backslash_preserves_cursor_boundary() { - let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(error) => panic!("Failed to set up Postgres container: {error}"), - }; - - let pool = PgPoolOptions::new() - .max_connections(2) - .connect(&pg_sqlx_url(&jdbc_url)) - .await - .expect("Failed to connect to Postgres for seeding"); - sqlx::query( - "CREATE TABLE text_cursor_test (id INT PRIMARY KEY, cursor_value TEXT UNIQUE NOT NULL)", - ) - .execute(&pool) - .await - .expect("Failed to create table"); - sqlx::query("INSERT INTO text_cursor_test (id, cursor_value) VALUES ($1, $2), ($3, $4)") - .bind(1_i32) - .bind(r"a\b") - .bind(2_i32) - .bind("z") - .execute(&pool) - .await - .expect("Failed to insert text cursor rows"); - - let iggy_setup = IggySetup::default(); - let query = "SELECT id, cursor_value FROM text_cursor_test \ - WHERE cursor_value > {last_offset} ORDER BY cursor_value"; - let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "incremental", &iggy_setup); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_TRACKING_COLUMN".to_owned(), - "cursor_value".to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INITIAL_OFFSET".to_owned(), - r"a\b".to_owned(), - ); - - let mut runtime = setup_runtime(); - runtime - .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) - .await; - let client = runtime.create_client().await; - - let (ids, received) = poll_until_ids_seen(&client, &[2], POLL_TIMEOUT).await; +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn incremental_text_offset_with_backslash_preserves_cursor_boundary( + harness: &TestHarness, + _fixture: JdbcTextCursorFixture, +) { + let client = harness.root_client().await.expect("root client"); + let (ids, received) = + poll_until_ids_seen(&client, &consumer("jdbc_text_cursor"), &[2], POLL_TIMEOUT).await; assert_eq!( ids, vec![2], - "Expected only the row after the exact backslash cursor; got ids {ids:?} from \ - {received} received message(s)" + "expected only the row after the exact cursor, got {ids:?} from {received} messages" ); } -/// Test: a single poll over many rows succeeds. This exercises the JNI -/// local-reference frame management in `read_rows`: a few-hundred-row result set -/// creates hundreds of per-column local references in one native call, which -/// would overflow the JNI local reference table (and abort the JVM) if each row -/// were not read inside its own local frame. -#[tokio::test] -#[serial] -async fn large_result_set_streams_without_crashing() { - let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - let pool = PgPoolOptions::new() - .max_connections(2) - .connect(&pg_sqlx_url(&jdbc_url)) - .await - .expect("Failed to connect to Postgres for seeding"); - sqlx::query("CREATE TABLE big_test (id INT PRIMARY KEY, name TEXT, val NUMERIC(12,2))") - .execute(&pool) - .await - .expect("Failed to create table"); - sqlx::query( - "INSERT INTO big_test (id, name, val) \ - SELECT g, 'row_' || g, (g * 1.5)::numeric(12,2) FROM generate_series(1, 300) g", - ) - .execute(&pool) - .await - .expect("Failed to insert rows"); - - // batch_size well above the row count so the whole table is read in a single - // poll (one read_rows call → hundreds of local refs). - let iggy_setup = IggySetup::default(); - let query = "SELECT id, name, val FROM big_test ORDER BY id"; - let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "bulk", &iggy_setup); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), - "5000".to_owned(), - ); - - let mut runtime = setup_runtime(); - runtime - .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) - .await; - let client = runtime.create_client().await; - - // The runtime would have crashed on the oversized poll without per-row local - // frames; receiving a healthy batch of well-formed messages proves it did not. - let messages = poll_messages_with_retry(&client, 150).await; - assert!( - messages.len() >= 150, - "Expected the source to stream a large result set without crashing; got {} messages", - messages.len() - ); - for msg in &messages[..150] { - let data = msg.get("data").expect("Missing 'data' field"); - assert!(data.get("id").and_then(|v| v.as_i64()).is_some()); - assert!(data.get("name").and_then(|v| v.as_str()).is_some()); +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn large_result_set_streams_without_crashing( + harness: &TestHarness, + _fixture: JdbcLargeResultFixture, +) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_large_result", 150).await; + assert!(messages.len() >= 150, "expected at least 150 messages"); + for message in &messages[..150] { + let data = message.get("data").expect("metadata should contain data"); + assert!(data.get("id").and_then(|value| value.as_i64()).is_some()); + assert!(data.get("name").and_then(|value| value.as_str()).is_some()); } } -/// Test: the source keeps polling and recovers after a query that raises a -/// SQLException on every poll. The query targets a table that does not exist -/// yet, so `executeQuery` throws each cycle; once the table is created the very -/// next poll must succeed and deliver its rows. -/// -/// This is the regression guard for exception clearing: a thrown Java exception -/// left pending would make the following JNI call (the statement `close()` in -/// the error path, or the next poll's `isValid`) run with an exception pending, -/// which the JNI spec forbids and which aborts the embedded JVM (the whole -/// runtime process). If that happened the runtime would die and never deliver -/// the post-recovery rows below. -#[tokio::test] -#[serial] -async fn source_recovers_after_repeated_query_errors() { - let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - // Start the source against a table that does not exist yet: every poll - // raises "relation does not exist" (SQLState 42P01). - let query = "SELECT id, name FROM recover_test ORDER BY id"; - let (_runtime, client) = setup_jdbc_postgres_source(&jdbc_url, &postgres_jar, query, "bulk") - .await - .expect("Failed to setup runtime"); - - // Let the source fail across several poll cycles (poll interval is 1s). - sleep(Duration::from_secs(4)).await; - - // Now create and seed the table; the next successful poll should deliver it. - let pool = PgPoolOptions::new() - .max_connections(2) - .connect(&pg_sqlx_url(&jdbc_url)) - .await - .expect("Failed to connect to Postgres for seeding"); +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn source_recovers_after_repeated_query_errors( + harness: &TestHarness, + fixture: JdbcRecoveryFixture, +) { + assert_source_running(harness).await; + sleep(Duration::from_millis(400)).await; + + let pool = fixture.create_pool().await.expect("PostgreSQL pool"); sqlx::query("CREATE TABLE recover_test (id INT PRIMARY KEY, name TEXT)") .execute(&pool) .await - .expect("Failed to create table"); + .expect("create recovery table"); sqlx::query("INSERT INTO recover_test (id, name) VALUES (1, 'a'), (2, 'b')") .execute(&pool) .await - .expect("Failed to insert rows"); + .expect("insert recovery rows"); + pool.close().await; - let (ids, received) = poll_until_ids_seen(&client, &[1, 2], POLL_TIMEOUT).await; + let client = harness.root_client().await.expect("root client"); + let (ids, received) = + poll_until_ids_seen(&client, &consumer("jdbc_recovery"), &[1, 2], POLL_TIMEOUT).await; assert_eq!( ids, vec![1, 2], - "Source must recover after repeated query failures and deliver ids 1,2; \ - got ids {ids:?} from {received} received message(s)" + "source did not recover: got {ids:?} from {received} messages" ); } -/// Test: bulk mode fails closed when the result set is larger than batch_size, -/// rather than silently syncing a truncated subset. With batch_size below the -/// row count the source errors every poll and delivers nothing (in particular, -/// never a truncated partial set). -#[tokio::test] -#[serial] -async fn bulk_result_larger_than_batch_size_fails_closed() { - let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(e) => panic!("Failed to set up Postgres container: {e}"), - }; - - let pool = PgPoolOptions::new() - .max_connections(2) - .connect(&pg_sqlx_url(&jdbc_url)) - .await - .expect("Failed to connect to Postgres for seeding"); - sqlx::query("CREATE TABLE trunc_test (id INT PRIMARY KEY)") - .execute(&pool) - .await - .expect("Failed to create table"); - sqlx::query("INSERT INTO trunc_test (id) SELECT generate_series(1, 5)") - .execute(&pool) - .await - .expect("Failed to insert rows"); +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn bulk_result_larger_than_batch_size_fails_closed( + harness: &TestHarness, + _fixture: JdbcBulkOverflowFixture, +) { + assert_source_running(harness).await; + assert_no_messages_for(harness, "jdbc_bulk_overflow", Duration::from_millis(600)).await; +} - // Bulk mode with batch_size below the 5-row result set. - let iggy_setup = IggySetup::default(); - let query = "SELECT id FROM trunc_test ORDER BY id"; - let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "bulk", &iggy_setup); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), - "2".to_owned(), - ); +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn incremental_tie_at_batch_boundary_fails_closed( + harness: &TestHarness, + _fixture: JdbcTieBoundaryFixture, +) { + assert_source_running(harness).await; + assert_no_messages_for(harness, "jdbc_tie_boundary", Duration::from_millis(600)).await; +} - let mut runtime = setup_runtime(); - runtime - .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) - .await; - let client = runtime.create_client().await; +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn query_timeout_cancels_slow_statement( + harness: &TestHarness, + _fixture: JdbcQueryTimeoutFixture, +) { + assert_source_running(harness).await; + // Without Statement.setQueryTimeout the two-second query emits a row in + // this window. A one-second timeout must cancel every retry before emission. + assert_no_messages_for(harness, "jdbc_query_timeout", Duration::from_secs(3)).await; +} - // Prove the connector opened before checking the fail-closed behavior. A - // startup failure also produces no messages and would otherwise make this - // negative-path test pass for the wrong reason. - let api_url = runtime - .harness - .connectors_runtime() - .expect("connectors runtime") - .http_url(); - let sources: serde_json::Value = reqwest::get(format!("{api_url}/sources")) - .await - .expect("Failed to query source status") - .error_for_status() - .expect("Source status endpoint returned an error") - .json() - .await - .expect("Failed to deserialize source status"); - assert_eq!( - sources[0]["status"], "running", - "JDBC source must be running before exercising bulk fail-closed behavior: {sources}" - ); +async fn poll_json_messages( + client: &IggyClient, + consumer_name: &str, + expected_count: usize, +) -> Vec { + let deadline = Instant::now() + POLL_TIMEOUT; + let consumer = consumer(consumer_name); + let mut received = Vec::new(); + loop { + let polled = poll(client, &consumer).await; + received.extend( + polled + .messages + .iter() + .filter_map(|message| serde_json::from_slice(&message.payload).ok()), + ); + if received.len() >= expected_count || Instant::now() >= deadline { + return received; + } + sleep(POLL_INTERVAL).await; + } +} - // Several poll cycles (poll interval is 1s). A fail-closed source delivers - // nothing, and in particular never the truncated 2-row subset. - sleep(Duration::from_secs(4)).await; - let polled = client - .get_messages(POLL_BATCH) - .await - .expect("Failed to poll messages"); - assert!( - polled.messages.is_empty(), - "bulk truncation must fail closed and deliver nothing, got {} messages", - polled.messages.len() - ); +async fn poll_until_ids_seen( + client: &IggyClient, + consumer: &Consumer, + expected: &[i64], + timeout: Duration, +) -> (Vec, usize) { + let deadline = Instant::now() + timeout; + let mut seen = BTreeSet::new(); + let mut received = 0; + loop { + let polled = poll(client, consumer).await; + for message in &polled.messages { + received += 1; + if let Ok(value) = serde_json::from_slice::(&message.payload) + && let Some(id) = value + .get("data") + .and_then(|data| data.get("id")) + .and_then(|id| id.as_i64()) + { + seen.insert(id); + } + } + if expected.iter().all(|id| seen.contains(id)) || Instant::now() >= deadline { + return (seen.into_iter().collect(), received); + } + sleep(POLL_INTERVAL).await; + } } -/// Test: incremental mode must not advance past rows tied at a batch boundary. -/// The source probes one row beyond batch_size and fails the complete poll when -/// that row shares the last in-batch tracking value, so no partial page is sent. -#[tokio::test] -#[serial] -async fn incremental_tie_at_batch_boundary_fails_closed() { - let (_container, jdbc_url, postgres_jar) = match setup_postgres_container().await { - Ok(result) => result, - Err(error) => panic!("Failed to set up Postgres container: {error}"), - }; +async fn assert_no_messages_for(harness: &TestHarness, consumer_name: &str, duration: Duration) { + let client = harness.root_client().await.expect("root client"); + let consumer = consumer(consumer_name); + let deadline = Instant::now() + duration; + loop { + let polled = poll(&client, &consumer).await; + assert!( + polled.messages.is_empty(), + "expected the source to fail closed, got {} messages", + polled.messages.len() + ); + if Instant::now() >= deadline { + return; + } + sleep(POLL_INTERVAL).await; + } +} - let pool = PgPoolOptions::new() - .max_connections(2) - .connect(&pg_sqlx_url(&jdbc_url)) - .await - .expect("Failed to connect to Postgres for seeding"); - sqlx::query("CREATE TABLE tie_test (id INT PRIMARY KEY, position INT NOT NULL)") - .execute(&pool) - .await - .expect("Failed to create table"); - sqlx::query("INSERT INTO tie_test (id, position) VALUES (1, 1), (2, 2), (3, 2), (4, 3)") - .execute(&pool) +async fn poll(client: &IggyClient, consumer: &Consumer) -> iggy_common::PolledMessages { + let stream: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic: Identifier = seeds::names::TOPIC.try_into().unwrap(); + client + .poll_messages( + &stream, + &topic, + None, + consumer, + &PollingStrategy::next(), + POLL_BATCH, + true, + ) .await - .expect("Failed to insert rows"); - - let iggy_setup = IggySetup::default(); - let query = - "SELECT id, position FROM tie_test WHERE position > {last_offset} ORDER BY position, id"; - let mut envs = build_jdbc_env(&jdbc_url, &postgres_jar, query, "incremental", &iggy_setup); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE".to_owned(), - "2".to_owned(), - ); - envs.insert( - "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_TRACKING_COLUMN".to_owned(), - "position".to_owned(), - ); + .expect("poll JDBC messages") +} - let mut runtime = setup_runtime(); - runtime - .init("jdbc/config_postgres.toml", Some(envs), iggy_setup) - .await; - let client = runtime.create_client().await; +fn consumer(name: &str) -> Consumer { + Consumer::new(name.try_into().expect("valid consumer name")) +} - let api_url = runtime - .harness +async fn assert_source_running(harness: &TestHarness) { + let api_url = harness .connectors_runtime() .expect("connectors runtime") .http_url(); let sources: serde_json::Value = reqwest::get(format!("{api_url}/sources")) .await - .expect("Failed to query source status") + .expect("query source status") .error_for_status() - .expect("Source status endpoint returned an error") + .expect("source status response") .json() .await - .expect("Failed to deserialize source status"); - assert_eq!( - sources[0]["status"], "running", - "JDBC source must be running before exercising incremental fail-closed behavior: {sources}" - ); - - sleep(Duration::from_secs(4)).await; - let polled = client - .get_messages(POLL_BATCH) - .await - .expect("Failed to poll messages"); - assert!( - polled.messages.is_empty(), - "a tied incremental page must fail before delivering a partial batch, got {} messages", - polled.messages.len() - ); + .expect("deserialize source status"); + assert_eq!(sources[0]["status"], "running", "source status: {sources}"); } diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index f73f63ca67..7975015065 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -39,45 +39,8 @@ mod s3; mod stdout; mod surrealdb; -use iggy::prelude::IggyClient; -use iggy_common::Client; -use iggy_common::{ - CompressionAlgorithm, Durability, IggyExpiry, IggyTimestamp, MaxTopicSize, MessageClient, - PolledMessages, StreamClient, TopicClient, TopicCreateOptions, -}; -use integration::harness::{ConnectorsRuntimeConfig, IpAddrKind, TestHarness, TestServerConfig}; +use iggy_common::IggyTimestamp; use serde::{Deserialize, Serialize}; -use std::collections::HashMap; - -const DEFAULT_TEST_STREAM: &str = "test_stream"; -const DEFAULT_TEST_TOPIC: &str = "test_topic"; - -fn setup_runtime() -> ConnectorsRuntime { - ConnectorsRuntime { - harness: TestHarness::builder() - .server( - TestServerConfig::builder() - .ip_kind(IpAddrKind::V4) - .quic_enabled(false) - .http_enabled(false) - .websocket_enabled(false) - .extra_envs(HashMap::from([ - // The harness pre-reserves a fixed TCP port (see PortReserver), - // so the server binds a non-zero port. That relies on the server - // writing current_config.toml on bind regardless of how the port - // was chosen (see config_writer::write_current_config); the harness - // reads that file to discover the bound address before it considers - // startup complete. - ("IGGY_TCP_ADDRESS".to_owned(), "127.0.0.1:0".to_owned()), - ])) - .build(), - ) - .build() - .unwrap(), - stream: "".to_owned(), - topic: "".to_owned(), - } -} const ONE_DAY_MICROS: u64 = 24 * 60 * 60 * 1_000_000; @@ -104,141 +67,3 @@ pub fn create_test_messages(count: usize) -> Vec { }) .collect() } - -#[derive(Debug)] -struct ConnectorsRuntime { - stream: String, - topic: String, - harness: TestHarness, -} - -#[derive(Debug)] -struct ConnectorsIggyClient { - stream: String, - topic: String, - client: IggyClient, -} - -impl ConnectorsIggyClient { - /// Poll up to `count` messages from the configured stream/topic. - /// - /// `count` is the caller's, because it bounds how much a collect-until-N loop - /// can drain per attempt: a batch smaller than what the caller waits for turns - /// the loop's attempt budget into a cap on the total it can ever see, so the - /// test fails on a slow runner even though the source delivered everything. - /// Sibling source tests poll in batches of 100. - async fn get_messages(&self, count: u32) -> Result { - self.client - .poll_messages( - &self.stream.clone().try_into().unwrap(), - &self.topic.clone().try_into().unwrap(), - None, - &iggy_common::Consumer::new("test_consumer".try_into().unwrap()), - &iggy_common::PollingStrategy::next(), - count, - true, - ) - .await - } -} - -#[derive(Debug)] -pub struct IggySetup { - pub stream: String, - pub topic: String, -} - -impl Default for IggySetup { - fn default() -> Self { - Self { - stream: DEFAULT_TEST_STREAM.to_owned(), - topic: DEFAULT_TEST_TOPIC.to_owned(), - } - } -} - -impl ConnectorsRuntime { - pub async fn init( - &mut self, - config_path: &str, - envs: Option>, - iggy_setup: IggySetup, - ) { - let config_path = format!("tests/connectors/{config_path}"); - let mut all_envs = HashMap::new(); - all_envs.insert( - "IGGY_CONNECTORS_CONFIG_PATH".to_owned(), - config_path.to_owned(), - ); - - if let Some(envs) = envs { - for (k, v) in envs { - all_envs.insert(k, v); - } - } - - // Start the iggy server - self.harness - .start() - .await - .expect("Failed to start test harness"); - - let client = self.create_iggy_client().await; - client - .create_stream(&iggy_setup.stream) - .await - .expect("Failed to create stream"); - let stream_id = iggy_setup - .stream - .clone() - .try_into() - .expect("Invalid stream name in Iggy setup"); - client - .create_topic( - &stream_id, - &iggy_setup.topic, - &TopicCreateOptions { - partitions_count: Some(1), - compression_algorithm: Some(CompressionAlgorithm::None), - message_expiry: Some(IggyExpiry::ServerDefault), - max_topic_size: Some(MaxTopicSize::ServerDefault), - durability: Durability::Persisted, - ..TopicCreateOptions::default() - }, - ) - .await - .expect("Failed to create topic"); - client.shutdown().await.expect("Failed to shutdown client"); - - let connectors_config = ConnectorsRuntimeConfig::builder() - .extra_envs(all_envs) - .build(); - - self.harness - .server_mut() - .set_connectors_runtime_config(connectors_config); - self.harness - .server_mut() - .start_dependents() - .await - .expect("Failed to start connectors runtime"); - - self.stream = iggy_setup.stream; - self.topic = iggy_setup.topic; - } - - pub async fn create_client(&self) -> ConnectorsIggyClient { - ConnectorsIggyClient { - stream: self.stream.clone(), - topic: self.topic.clone(), - client: self.create_iggy_client().await, - } - } - - async fn create_iggy_client(&self) -> IggyClient { - self.harness - .tcp_root_client() - .await - .expect("Failed to create root TCP client") - } -}