diff --git a/.config/nextest.toml b/.config/nextest.toml index c8428a4e49..a1e45cc8c4 100644 --- a/.config/nextest.toml +++ b/.config/nextest.toml @@ -39,6 +39,18 @@ max-threads = 1 filter = 'package(integration) and test(/connectors::elasticsearch::/)' test-group = "elasticsearch" +# JDBC tests each start their own iggy-server plus a Postgres testcontainer and +# an embedded JVM, and they share a JDBC driver JAR downloaded to a single path +# on first run. nextest runs each test in its own process, so the in-source +# `#[serial]` guard does not serialize them here; `max-threads = 1` does, which +# avoids resource contention and a race to download the same driver JAR. +[test-groups.jdbc] +max-threads = 1 + +[[profile.default.overrides]] +filter = 'package(integration) and test(/connectors::jdbc::/)' +test-group = "jdbc" + # Each test in these binaries spawns its own real iggy-server subprocess (isolated data dir, port # from PortGuard's own flock, so no two ever share a port). #[serial] on them is a dead letter # under nextest - nextest runs every test as its own process, and #[serial]'s mutex only excludes diff --git a/.github/workflows/_build_rust_artifacts.yml b/.github/workflows/_build_rust_artifacts.yml index 6b4d568f8b..176e8e255c 100644 --- a/.github/workflows/_build_rust_artifacts.yml +++ b/.github/workflows/_build_rust_artifacts.yml @@ -46,7 +46,7 @@ on: connector_plugins: type: string required: false - default: "iggy_connector_elasticsearch_sink,iggy_connector_elasticsearch_source,iggy_connector_iceberg_sink,iggy_connector_postgres_sink,iggy_connector_postgres_source,iggy_connector_quickwit_sink,iggy_connector_random_source,iggy_connector_s3_sink,iggy_connector_stdout_sink,iggy_connector_surrealdb_sink, iggy_connector_rabbitmq_sink" + default: "iggy_connector_elasticsearch_sink,iggy_connector_elasticsearch_source,iggy_connector_iceberg_sink,iggy_connector_jdbc_source,iggy_connector_postgres_sink,iggy_connector_postgres_source,iggy_connector_quickwit_sink,iggy_connector_rabbitmq_sink,iggy_connector_random_source,iggy_connector_s3_sink,iggy_connector_stdout_sink,iggy_connector_surrealdb_sink" description: "Comma-separated list of connector plugin crates to build as shared libraries" outputs: artifact_name: diff --git a/.github/workflows/edge-release.yml b/.github/workflows/edge-release.yml index d68b36fcda..3b58e07afb 100644 --- a/.github/workflows/edge-release.yml +++ b/.github/workflows/edge-release.yml @@ -98,6 +98,7 @@ jobs: - `iggy_connector_elasticsearch_sink` - `iggy_connector_elasticsearch_source` - `iggy_connector_iceberg_sink` + - `iggy_connector_jdbc_source` - `iggy_connector_postgres_sink` - `iggy_connector_postgres_source` - `iggy_connector_quickwit_sink` diff --git a/Cargo.lock b/Cargo.lock index d0b272a665..cbd7549936 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2689,6 +2689,12 @@ dependencies = [ "shlex 2.0.1", ] +[[package]] +name = "cesu8" +version = "1.1.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6d43a04d8753f35258c91f8ec639f792891f748a1edbd759cf1dcea3382ad83c" + [[package]] name = "cexpr" version = "0.6.0" @@ -2806,7 +2812,7 @@ checksum = "157a8ba7b480713b56f4c09fd13fc3e0a22a5dfab8097ba61cbc5feef950788a" dependencies = [ "glob", "libc", - "libloading", + "libloading 0.8.9", ] [[package]] @@ -6189,7 +6195,7 @@ dependencies = [ "hickory-proto", "idna", "ipnet", - "jni", + "jni 0.22.4", "rand 0.10.2", "thiserror 2.0.20", "tinyvec", @@ -6207,7 +6213,7 @@ dependencies = [ "data-encoding", "idna", "ipnet", - "jni", + "jni 0.22.4", "once_cell", "prefix-trie", "rand 0.10.2", @@ -6230,7 +6236,7 @@ dependencies = [ "hickory-proto", "ipconfig", "ipnet", - "jni", + "jni 0.22.4", "moka", "ndk-context", "once_cell", @@ -7335,6 +7341,27 @@ dependencies = [ "uuid", ] +[[package]] +name = "iggy_connector_jdbc_source" +version = "0.5.0" +dependencies = [ + "async-trait", + "base64 0.23.1", + "chrono", + "dashmap", + "humantime", + "iggy_connector_sdk", + "jni 0.21.1", + "regex", + "rmp-serde", + "secrecy", + "serde", + "serde_json", + "tokio", + "toml 1.1.6+spec-1.1.0", + "tracing", +] + [[package]] name = "iggy_connector_meilisearch_sink" version = "0.5.0" @@ -7825,6 +7852,7 @@ dependencies = [ "serde", "serde_json", "serial_test", + "sha2 0.10.9", "sqlparser 0.63.0", "sqlx", "sysinfo 0.39.6", @@ -7948,6 +7976,15 @@ version = "1.0.18" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "8f42a60cbdf9a97f5d2305f08a87dc4e09308d1276d28c869c684d7777685682" +[[package]] +name = "java-locator" +version = "0.1.9" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "09c46c1fe465c59b1474e665e85e1256c3893dd00927b8d55f63b09044c1e64f" +dependencies = [ + "glob", +] + [[package]] name = "jiff" version = "0.2.37" @@ -8004,6 +8041,24 @@ dependencies = [ "jiff-tzdb", ] +[[package]] +name = "jni" +version = "0.21.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "1a87aa2bb7d2af34197c04845522473242e1aa17c12f4935d5856491a7fb8c97" +dependencies = [ + "cesu8", + "cfg-if", + "combine", + "java-locator", + "jni-sys 0.3.1", + "libloading 0.7.4", + "log", + "thiserror 1.0.69", + "walkdir", + "windows-sys 0.45.0", +] + [[package]] name = "jni" version = "0.22.4" @@ -8013,7 +8068,7 @@ dependencies = [ "cfg-if", "combine", "jni-macros", - "jni-sys", + "jni-sys 0.4.1", "log", "simd_cesu8", "thiserror 2.0.20", @@ -8034,6 +8089,15 @@ dependencies = [ "syn 2.0.119", ] +[[package]] +name = "jni-sys" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "41a652e1f9b6e0275df1f15b32661cf0d4b78d4d87ddec5e0c3c20f097433258" +dependencies = [ + "jni-sys 0.4.1", +] + [[package]] name = "jni-sys" version = "0.4.1" @@ -8401,6 +8465,16 @@ dependencies = [ "pkg-config", ] +[[package]] +name = "libloading" +version = "0.7.4" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b67380fd3b2fbe7527a606e18729d21c6f3951633d0500574c4dc22d2d638b9f" +dependencies = [ + "cfg-if", + "winapi", +] + [[package]] name = "libloading" version = "0.8.9" @@ -12172,7 +12246,7 @@ checksum = "26d1e2536ce4f35f4846aa13bff16bd0ff40157cdb14cc056c7b14ba41233ba0" dependencies = [ "core-foundation 0.10.1", "core-foundation-sys", - "jni", + "jni 0.22.4", "log", "once_cell", "rustls", @@ -15635,6 +15709,15 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "windows-sys" +version = "0.45.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "75283be5efb2831d37ea142365f009c02ec203cd29a3ebecbc093d52315b66d0" +dependencies = [ + "windows-targets 0.42.2", +] + [[package]] name = "windows-sys" version = "0.48.0" @@ -15680,6 +15763,21 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "windows-targets" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8e5180c00cd44c9b1c88adb3693291f1cd93605ded80c250a75d472756b4d071" +dependencies = [ + "windows_aarch64_gnullvm 0.42.2", + "windows_aarch64_msvc 0.42.2", + "windows_i686_gnu 0.42.2", + "windows_i686_msvc 0.42.2", + "windows_x86_64_gnu 0.42.2", + "windows_x86_64_gnullvm 0.42.2", + "windows_x86_64_msvc 0.42.2", +] + [[package]] name = "windows-targets" version = "0.48.5" @@ -15746,6 +15844,12 @@ dependencies = [ "windows-link 0.2.1", ] +[[package]] +name = "windows_aarch64_gnullvm" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "597a5118570b68bc08d8d59125332c54f1ba9d9adeedeef5b99b02ba2b0698f8" + [[package]] name = "windows_aarch64_gnullvm" version = "0.48.5" @@ -15764,6 +15868,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "a9d8416fa8b42f5c947f8482c43e7d89e73a173cead56d044f6a56104a6d1b53" +[[package]] +name = "windows_aarch64_msvc" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e08e8864a60f06ef0d0ff4ba04124db8b0fb3be5776a5cd47641e942e58c4d43" + [[package]] name = "windows_aarch64_msvc" version = "0.48.5" @@ -15782,6 +15892,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "b9d782e804c2f632e395708e99a94275910eb9100b2114651e04744e9b125006" +[[package]] +name = "windows_i686_gnu" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "c61d927d8da41da96a81f029489353e68739737d3beca43145c8afec9a31a84f" + [[package]] name = "windows_i686_gnu" version = "0.48.5" @@ -15812,6 +15928,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "fa7359d10048f68ab8b09fa71c3daccfb0e9b559aed648a8f95469c27057180c" +[[package]] +name = "windows_i686_msvc" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "44d840b6ec649f480a41c8d80f9c65108b92d89345dd94027bfe06ac444d1060" + [[package]] name = "windows_i686_msvc" version = "0.48.5" @@ -15830,6 +15952,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "1e7ac75179f18232fe9c285163565a57ef8d3c89254a30685b57d83a38d326c2" +[[package]] +name = "windows_x86_64_gnu" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8de912b8b8feb55c064867cf047dda097f92d51efad5b491dfb98f6bbb70cb36" + [[package]] name = "windows_x86_64_gnu" version = "0.48.5" @@ -15848,6 +15976,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "9c3842cdd74a865a8066ab39c8a7a473c0778a3f29370b5fd6b4b9aa7df4a499" +[[package]] +name = "windows_x86_64_gnullvm" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "26d41b46a36d453748aedef1486d5c7a85db22e56aff34643984ea85514e94a3" + [[package]] name = "windows_x86_64_gnullvm" version = "0.48.5" @@ -15866,6 +16000,12 @@ version = "0.53.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "0ffa179e2d07eee8ad8f57493436566c7cc30ac536a3379fdf008f47f6bb7ae1" +[[package]] +name = "windows_x86_64_msvc" +version = "0.42.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9aec5da331524158c6d1a4ac0ab1541149c0b9505fde06423b02f5ef0106b9f0" + [[package]] name = "windows_x86_64_msvc" version = "0.48.5" diff --git a/Cargo.toml b/Cargo.toml index 7956ce5587..13285f3b8e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -52,6 +52,7 @@ members = [ "core/connectors/sources/elasticsearch_source", "core/connectors/sources/http_source", "core/connectors/sources/influxdb_source", + "core/connectors/sources/jdbc_source", "core/connectors/sources/postgres_source", "core/connectors/sources/random_source", "core/consensus", @@ -213,6 +214,7 @@ iggy_connector_sdk = { path = "core/connectors/sdk", version = "0.5.0" } indexmap = "2.14.2" integration = { path = "core/integration" } ipnet = "2.12.2" +jni = { version = "0.21", features = ["invocation"] } journal = { path = "core/journal" } js-sys = "0.3" jsonwebtoken = { version = "11.0.0", features = ["rust_crypto"] } @@ -308,6 +310,7 @@ serde_yaml_ng = "0.10.0" serial_test = "4.0.1" server = { path = "core/server", default-features = false } server_common = { path = "core/server_common" } +sha2 = "0.10.9" shard = { path = "core/shard" } simd-json = { version = "0.18.1", features = ["serde_impl"] } smallvec = "1.16" diff --git a/core/connectors/README.md b/core/connectors/README.md index b334009278..c03320ff1e 100644 --- a/core/connectors/README.md +++ b/core/connectors/README.md @@ -122,6 +122,7 @@ Please refer to the **[Source documentation](https://github.com/apache/iggy/tree ### Available Sources - **Elasticsearch Source** - polls documents from Elasticsearch indices +- **JDBC Source** - reads rows from any JDBC-compliant database (PostgreSQL, MySQL, Oracle, SQL Server, H2) via an embedded JVM; bulk and incremental modes - **PostgreSQL Source** - reads rows from PostgreSQL tables with multiple consumption strategies (delete after read, mark as processed, timestamp tracking) - **Random Source** - generates random test messages (useful for testing/development) diff --git a/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml new file mode 100644 index 0000000000..c939d2285c --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_bulk_mode.toml @@ -0,0 +1,80 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration - BULK MODE +# Bulk mode works with ALL JDBC databases without any special requirements +# No tracking column needed - just executes your query and fetches results + +type = "source" +key = "jdbc_bulk_example" +enabled = true +version = 0 +name = "JDBC Bulk Mode Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# This example uses PostgreSQL, but bulk mode works identically with: +# MySQL, Oracle, SQL Server, H2, Derby, DB2, etc. +jdbc_url = "jdbc:postgresql://localhost:5432/warehouse" + +# Database credentials can be in URL or separate +# jdbc_url = "jdbc:postgresql://localhost:5432/warehouse?user=myuser&password=mypass" +username = "warehouse_user" +password = "secret" + +driver_class = "org.postgresql.Driver" +driver_jar_path = "/opt/jdbc-drivers/postgresql-42.6.0.jar" + +# Bulk mode: Any valid SELECT query +# Can include JOINs, aggregations, complex WHERE clauses, etc. +query = """ +SELECT + p.product_id, + p.product_name, + p.category, + p.price, + COUNT(o.order_id) as total_orders, + SUM(o.quantity) as total_quantity +FROM products p +LEFT JOIN orders o ON p.product_id = o.product_id +GROUP BY p.product_id, p.product_name, p.category, p.price +""" + +# Poll once per hour for daily snapshots +poll_interval = "1h" + +# Large batch size for full table scans +batch_size = 10000 + +# BULK MODE - no tracking column needed! +mode = "bulk" + +# Bulk mode benefits: +# - No tracking column required +# - Works with any SELECT query +# - Supports complex queries (JOINs, aggregations, window functions) +# - Perfect for periodic snapshots +# - Universal compatibility with all databases + +snake_case_columns = true +include_metadata = false + +[[streams]] +stream = "warehouse" +topic = "product_summary" +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_h2.toml b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml new file mode 100644 index 0000000000..d1576dcfd2 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_h2.toml @@ -0,0 +1,61 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for H2 Database +# H2 is useful for testing and development as it's an embedded Java database + +type = "source" +key = "jdbc_h2_example" +enabled = true +version = 0 +name = "JDBC H2 Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# H2 connection URL (in-memory database). INIT creates and seeds the table used +# by the query so this example runs without a separate setup step. +jdbc_url = "jdbc:h2:mem:testdb;DB_CLOSE_DELAY=-1;INIT=CREATE TABLE IF NOT EXISTS users (id INT PRIMARY KEY AUTO_INCREMENT, name VARCHAR(100), email VARCHAR(100), created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP)\\;INSERT INTO users (name, email) SELECT 'User ' || x, 'user' || x || '@test.com' FROM SYSTEM_RANGE(1, 100) WHERE NOT EXISTS (SELECT 1 FROM users)" + +# H2 JDBC driver +driver_class = "org.h2.Driver" + +# Path to H2 driver JAR +# Download from: https://repo1.maven.org/maven2/com/h2database/h2/2.2.224/h2-2.2.224.jar +# Note: Update this path to match where you downloaded the JAR file +driver_jar_path = "/tmp/jdbc-drivers/h2-2.2.224.jar" + +# H2 credentials (default) +username = "sa" +password = "" + +# Simple query for testing +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" + +poll_interval = "10s" +batch_size = 100 +tracking_column = "id" +initial_offset = "0" +mode = "incremental" +snake_case_columns = false +include_metadata = true +verbose_logging = false + +[[streams]] +stream = "test" +topic = "users" +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml b/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml new file mode 100644 index 0000000000..f246781611 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_mysql.toml @@ -0,0 +1,84 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for MySQL +# This file demonstrates how to configure the JDBC source connector +# to read data from a MySQL database and publish to Iggy streams. + +type = "source" +key = "jdbc_mysql_example" +enabled = true +version = 0 +name = "JDBC MySQL Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# JDBC connection URL +# Option 1: Separate credentials (recommended) +jdbc_url = "jdbc:mysql://localhost:3306/ecommerce?useSSL=false&serverTimezone=UTC" + +# Option 2: Embedded credentials in URL (alternative) +# jdbc_url = "jdbc:mysql://iggy_user:iggy_password@localhost:3306/ecommerce?useSSL=false&serverTimezone=UTC" + +# JDBC driver class name +driver_class = "com.mysql.cj.jdbc.Driver" + +# Path to JDBC driver JAR file +# Download from: https://repo1.maven.org/maven2/com/mysql/mysql-connector-j/8.0.33/mysql-connector-j-8.0.33.jar +driver_jar_path = "/opt/jdbc-drivers/mysql-connector-j-8.0.33.jar" + +# Database credentials (optional if included in jdbc_url) +username = "iggy_user" +password = "iggy_password" + +# SQL query to execute +# Use {last_offset} placeholder for incremental reads +query = "SELECT * FROM orders WHERE updated_at > {last_offset} ORDER BY updated_at ASC" + +# How often to poll the database +poll_interval = "30s" + +# Maximum number of rows to fetch per poll +batch_size = 1000 + +# Column to track for incremental reads (must be in query result) +tracking_column = "updated_at" + +# Initial offset value for the first poll +initial_offset = "2024-01-01 00:00:00" + +# Source mode: "incremental" or "bulk" +# Note: Both modes work with ALL JDBC databases (MySQL, Oracle, PostgreSQL, etc.) +# - incremental: Tracks last offset, avoids duplicate reads +# - bulk: Full table scan, no offset tracking +mode = "incremental" + +# Convert column names to snake_case (e.g., OrderDate -> order_date) +snake_case_columns = true + +# Include metadata wrapper in output messages +include_metadata = true + +# Custom JVM options (optional) +jvm_options = ["-Xmx512m", "-Xms128m"] + +# Target Iggy stream and topic +[[streams]] +stream = "ecommerce" +topic = "orders" +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml new file mode 100644 index 0000000000..9dfae77c54 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_oracle.toml @@ -0,0 +1,80 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for Oracle Database + +type = "source" +key = "jdbc_oracle_example" +enabled = true +version = 0 +name = "JDBC Oracle Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# JDBC connection URL +# Option 1: Separate credentials +jdbc_url = "jdbc:oracle:thin:@localhost:1521:XE" + +# Option 2: Embedded credentials in URL (Oracle uses / separator) +# jdbc_url = "jdbc:oracle:thin:system/oracle@localhost:1521:XE" + +# JDBC driver class name +driver_class = "oracle.jdbc.OracleDriver" + +# Path to JDBC driver JAR file +# Download from: https://www.oracle.com/database/technologies/appdev/jdbc-downloads.html +driver_jar_path = "/opt/jdbc-drivers/ojdbc11.jar" + +# Database credentials (optional if included in jdbc_url) +username = "system" +password = "oracle" + +# SQL query to execute +# Use a stable numeric or timestamp cursor; Oracle ROWNUM is not a valid tracking column. +query = "SELECT * FROM CUSTOMERS WHERE ID > {last_offset} ORDER BY ID" + +# How often to poll the database +poll_interval = "1m" + +# Maximum number of rows to fetch per poll +batch_size = 500 + +# Column to track for incremental reads (must be in query result) +tracking_column = "ID" + +# Initial offset value for the first poll +initial_offset = "0" + +# Source mode: "incremental" or "bulk" +# Works with ALL JDBC databases universally +mode = "incremental" + +# Convert column names to snake_case (e.g., OrderDate -> order_date) +snake_case_columns = true + +# Include metadata wrapper in output messages +include_metadata = true + +# Custom JVM options (optional) +jvm_options = ["-Xmx512m", "-Xms256m"] + +# Target Iggy stream and topic +[[streams]] +stream = "crm" +topic = "customers" +schema = "json" diff --git a/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml b/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml new file mode 100644 index 0000000000..895f4beaf7 --- /dev/null +++ b/core/connectors/runtime/example_config/connectors/jdbc_sqlserver.toml @@ -0,0 +1,65 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# Example JDBC Source Connector Configuration for Microsoft SQL Server + +type = "source" +key = "jdbc_sqlserver_example" +enabled = true +version = 0 +name = "JDBC SQL Server Source" +path = "target/release/libiggy_connector_jdbc_source" +plugin_config_format = "toml" + +[plugin_config] +# JDBC connection URL +# Option 1: Separate credentials +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=Sales;encrypt=false" + +# Option 2: Embedded credentials in URL +# jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=Sales;user=sa;password=YourPassword123;encrypt=false" + +# JDBC driver class name +driver_class = "com.microsoft.sqlserver.jdbc.SQLServerDriver" + +# Path to JDBC driver JAR file +# Download from: https://repo1.maven.org/maven2/com/microsoft/sqlserver/mssql-jdbc/ +driver_jar_path = "/opt/jdbc-drivers/mssql-jdbc-12.4.1.jre11.jar" + +# Database credentials (optional if included in jdbc_url) +username = "sa" +password = "YourPassword123" + +# SQL query to execute +query = "SELECT * FROM Orders WHERE OrderDate > {last_offset} ORDER BY OrderDate" + +poll_interval = "15s" +batch_size = 2000 +tracking_column = "OrderDate" +initial_offset = "2024-01-01" +mode = "incremental" + +# Convert SQL Server naming to snake_case +snake_case_columns = true +include_metadata = true + +connection_timeout_ms = 30000 + +[[streams]] +stream = "sales" +topic = "orders" +schema = "json" diff --git a/core/connectors/sdk/src/sink.rs b/core/connectors/sdk/src/sink.rs index 9891d22436..66ace46d2e 100644 --- a/core/connectors/sdk/src/sink.rs +++ b/core/connectors/sdk/src/sink.rs @@ -91,7 +91,15 @@ impl SinkContainer { let result = runtime.block_on(sink.open()); self.id = id; self.sink = Some(sink); - if result.is_ok() { 0 } else { 1 } + match result { + Ok(()) => 0, + Err(_) => { + // Connector errors may contain secrets from external clients. + // Only the status is safe to log at this generic boundary. + error!("Failed to open sink connector with ID: {id}"); + 1 + } + } } } diff --git a/core/connectors/sdk/src/source.rs b/core/connectors/sdk/src/source.rs index 09d18bf673..0a9f64aced 100644 --- a/core/connectors/sdk/src/source.rs +++ b/core/connectors/sdk/src/source.rs @@ -187,7 +187,16 @@ impl SourceContainer { let result = runtime.block_on(source.open()); self.id = id; self.source = Some(Arc::new(source)); - if result.is_ok() { 0 } else { 1 } + match result { + Ok(()) => 0, + Err(_) => { + // Connector errors may contain secrets from external clients + // (for example a JDBC URL echoed by a driver). Only the status + // is safe to log at this generic boundary. + error!("Failed to open source connector with ID: {id}"); + 1 + } + } } } diff --git a/core/connectors/sources/README.md b/core/connectors/sources/README.md index e63f8cb17f..45ec10b556 100644 --- a/core/connectors/sources/README.md +++ b/core/connectors/sources/README.md @@ -11,6 +11,7 @@ Source connectors are responsible for ingesting data from external sources into | **elasticsearch_source** | Polls documents from Elasticsearch indices with timestamp-based tracking | | **http_source** | Webhook gateway: an embedded HTTP server shared by every instance, with per-endpoint bearer/HMAC auth and a management API for endpoints registered at runtime | | **influxdb_source** | Polls InfluxDB with cursor-based timestamp tracking; supports V2 (Flux, annotated CSV) and V3 (SQL, JSONL) | +| **jdbc_source** | Reads rows from any JDBC-compliant database (PostgreSQL, MySQL, Oracle, SQL Server, H2) via an embedded JVM; bulk and incremental modes | | **postgres_source** | Reads rows from PostgreSQL tables with multiple strategies: delete after read, mark as processed, or timestamp tracking | | **random_source** | Generates random test messages (useful for testing and development) | diff --git a/core/connectors/sources/jdbc_source/Cargo.toml b/core/connectors/sources/jdbc_source/Cargo.toml new file mode 100644 index 0000000000..1b2016dfa0 --- /dev/null +++ b/core/connectors/sources/jdbc_source/Cargo.toml @@ -0,0 +1,68 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +[package] +name = "iggy_connector_jdbc_source" +version = "0.5.0" +edition = "2024" +license = "Apache-2.0" +keywords = ["iggy", "messaging", "streaming", "jdbc", "source"] +categories = ["database"] +description = "JDBC source connector for Iggy with PostgreSQL integration coverage" +readme = "README.md" +publish = false + +[package.metadata.cargo-machete] +ignored = ["dashmap"] + +[lib] +crate-type = ["cdylib", "lib"] + +[features] +default = [] + +[dependencies] +async-trait = { workspace = true } +base64 = { workspace = true } +chrono = { workspace = true } + +# Required by source_connector! macro +dashmap = { workspace = true } + +# For parsing duration strings (poll_interval) +humantime = { workspace = true } + +# Connector SDK +iggy_connector_sdk = { workspace = true } + +# JNI for Java interop with invocation support +jni = { workspace = true } + +# For sanitizing passwords in logs +regex = { workspace = true } +secrecy = { workspace = true } +serde = { workspace = true, features = ["derive"] } +serde_json = { workspace = true } +tokio = { workspace = true, features = ["full"] } +tracing = { workspace = true } + +[dev-dependencies] +rmp-serde = { workspace = true } +toml = { workspace = true } + +[lints] +workspace = true diff --git a/core/connectors/sources/jdbc_source/README.md b/core/connectors/sources/jdbc_source/README.md new file mode 100644 index 0000000000..599e91cb8f --- /dev/null +++ b/core/connectors/sources/jdbc_source/README.md @@ -0,0 +1,620 @@ +# JDBC Source Connector + +A JDBC source connector for Iggy designed to work with JDBC-compliant relational databases. PostgreSQL is covered by the runtime integration suite; the other driver examples below are configuration guides and are not yet exercised in CI. + +## Overview + +This connector reads data from relational databases using JDBC (Java Database Connectivity) and publishes it as messages to Iggy streams. It supports both bulk and incremental data synchronization modes. + +## Features + +- **JDBC Driver Support**: Uses a supplied JDBC driver without database-specific connector code +- **Incremental Sync**: Track changes using timestamps or auto-increment IDs +- **Bulk Mode**: Re-runs the query each poll for snapshots (capped at `batch_size` rows; see limitations) +- **Type Mapping**: Automatic conversion of SQL types to JSON +- **Configurable Polling**: Control how frequently data is fetched +- **State Management**: Tracks offsets so rows below the cursor are not re-read, committing the cursor only once the runtime acknowledges delivery (at-least-once, not exactly-once) +- **Flexible Queries**: Support for custom SQL queries with placeholders + +## Supported Databases + +PostgreSQL bulk and incremental modes are covered by end-to-end tests with a +real database and the PostgreSQL JDBC driver. The connector is designed around +standard JDBC APIs, so the following databases are expected to work with a +compatible driver, but they are not currently part of the integration test +matrix: + +- MySQL / MariaDB +- Oracle Database +- Microsoft SQL Server +- H2 Database +- Apache Derby +- IBM DB2 +- SQLite (via JDBC) +- SAP HANA +- Teradata +- Snowflake +- Amazon Redshift +- Google BigQuery +- Other JDBC-compliant relational databases + +Driver behavior and SQL syntax vary. Validate the query, type mappings, timeout +behavior, and cursor semantics against the exact driver version before using an +untested database in production. + +## Prerequisites + +1. **Java Runtime Environment (JRE)**: JRE 8 or later must be installed +2. **JDBC Driver**: Download the appropriate JDBC driver JAR for your database + +### Downloading JDBC Drivers + +**MySQL:** + +```bash +wget https://repo1.maven.org/maven2/com/mysql/mysql-connector-j/8.0.33/mysql-connector-j-8.0.33.jar +``` + +**PostgreSQL:** + +```bash +wget https://jdbc.postgresql.org/download/postgresql-42.6.0.jar +``` + +**Oracle:** + +- Download from [Oracle JDBC Driver Downloads](https://www.oracle.com/database/technologies/appdev/jdbc-downloads.html) + +**SQL Server:** + +```bash +wget https://repo1.maven.org/maven2/com/microsoft/sqlserver/mssql-jdbc/12.4.1.jre11/mssql-jdbc-12.4.1.jre11.jar +``` + +**H2:** + +```bash +wget https://repo1.maven.org/maven2/com/h2database/h2/2.2.224/h2-2.2.224.jar +``` + +## Configuration + +### Basic Configuration (Incremental Sync) + +```toml +type = "source" +key = "jdbc_mysql_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:mysql://localhost:3306/ecommerce" +driver_class = "com.mysql.cj.jdbc.Driver" +driver_jar_path = "/opt/jdbc-drivers/mysql-connector-j-8.0.33.jar" +username = "iggy_user" +password = "secret_password" +query = "SELECT * FROM orders WHERE updated_at > {last_offset} ORDER BY updated_at ASC" +poll_interval = "30s" +batch_size = 1000 +# updated_at is a timestamp and may not be unique. If equal timestamps cross a +# batch boundary, the poll fails closed (see "Unique / strictly increasing" +# below). Prefer a unique auto-increment key, or keep batch_size larger than any +# same-timestamp group. +tracking_column = "updated_at" +initial_offset = "2024-01-01 00:00:00" +mode = "incremental" +snake_case_columns = true +include_metadata = true +verbose_logging = false + +[[streams]] +stream = "ecommerce" +topic = "orders" +``` + +### Bulk Mode Configuration + +```toml +type = "source" +key = "jdbc_bulk_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:postgresql://localhost:5432/warehouse" +driver_class = "org.postgresql.Driver" +driver_jar_path = "/opt/jdbc-drivers/postgresql-42.6.0.jar" +username = "warehouse_user" +password = "secret" +query = "SELECT * FROM product_catalog" +poll_interval = "1h" +batch_size = 5000 +mode = "bulk" +snake_case_columns = false +include_metadata = true + +[[streams]] +stream = "warehouse" +topic = "products" +``` + +### Oracle Database Example + +```toml +type = "source" +key = "jdbc_oracle_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:oracle:thin:@localhost:1521:XE" +driver_class = "oracle.jdbc.OracleDriver" +driver_jar_path = "/opt/jdbc-drivers/ojdbc11.jar" +username = "system" +password = "oracle" +query = "SELECT * FROM CUSTOMERS WHERE ID > {last_offset} ORDER BY ID" +poll_interval = "1m" +batch_size = 500 +tracking_column = "ID" +initial_offset = "0" +mode = "incremental" +jvm_options = ["-Xmx256m", "-Xms128m"] + +[[streams]] +stream = "crm" +topic = "customers" +``` + +### SQL Server Example + +```toml +type = "source" +key = "jdbc_sqlserver_source" +enabled = true + +[plugin_config] +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=Sales;encrypt=false" +driver_class = "com.microsoft.sqlserver.jdbc.SQLServerDriver" +driver_jar_path = "/opt/jdbc-drivers/mssql-jdbc-12.4.1.jre11.jar" +username = "sa" +password = "YourPassword123" +query = "SELECT * FROM Orders WHERE OrderDate > {last_offset} ORDER BY OrderDate" +poll_interval = "15s" +batch_size = 2000 +# OrderDate is a timestamp and may not be unique; see the uniqueness note on the +# MySQL example above. Prefer a unique key, or keep batch_size above any +# same-timestamp group. +tracking_column = "OrderDate" +initial_offset = "2024-01-01" +mode = "incremental" + +[[streams]] +stream = "sales" +topic = "orders" +``` + +## Configuration Parameters + +| Parameter | Type | Required | Default | Description | +| ----------- | ------ | ---------- | --------- | ------------- | +| `jdbc_url` | string | Yes | - | JDBC connection URL (can include credentials) | +| `driver_class` | string | Yes | - | JDBC driver class name | +| `driver_jar_path` | string | Yes | - | Path to the JDBC driver JAR (checked to exist at startup; passed to the embedded JVM as `-Djava.class.path`) | +| `username` | string | No | - | Database username (optional if in jdbc_url) | +| `password` | string | No | - | Database password (optional if in jdbc_url) | +| `query` | string | Yes | - | SQL query to execute. Incremental mode requires `{last_offset}`; `{tracking_column}` is also supported | +| `poll_interval` | string (duration) | No | 5s | Positive polling interval as a humantime string (e.g., "30s", "5m", "1h"); zero is rejected | +| `batch_size` | u32 | No | 1000 | Maximum rows to fetch per poll | +| `tracking_column` | string | Incremental | - | Column to track for incremental reads (required in incremental mode; the query must also `ORDER BY` it) | +| `initial_offset` | string | No | - | Starting offset value for first poll | +| `mode` | string | No | "incremental" | Sync mode: "incremental" or "bulk" | +| `connection_timeout_ms` | u64 | No | 5000 | Timeout (ms) for the per-poll `isValid` liveness check; converted to seconds and capped at 5s | +| `login_timeout_ms` | u64 | No | 30000 | Bound on establishing the connection (`DriverManager.setLoginTimeout`); rounded up to whole seconds. Stops an unreachable database from hanging startup | +| `query_timeout_ms` | u64 | No | 30000 | Positive bound on each query execution (`Statement.setQueryTimeout`); rounded up to whole seconds | +| `jvm_options` | array | No | [] | Custom JVM options (e.g., ["-Xmx1g"]) | +| `snake_case_columns` | bool | No | false | Convert column names to snake_case | +| `include_metadata` | bool | No | true | Wrap each row with metadata (operation type, timestamp). `table_name` is a reserved field and is currently always null | +| `verbose_logging` | bool | No | false | Log per-poll row and column counts at info instead of debug | + +## Query Placeholders + +The `query` parameter supports placeholders for dynamic queries: + +- `{last_offset}`: Replaced with a JDBC `PreparedStatement` parameter and bound using the parameter's reported SQL type +- `{tracking_column}`: Replaced with the configured `tracking_column` (validated as a plain SQL identifier) + +Incremental mode is validated at `open()` and enforces the following (the +connector refuses to start otherwise): + +- **`tracking_column` is required.** Without it the offset can never advance and + every poll re-reads the same rows. +- **The query must contain `{last_offset}`.** Without it each poll would execute + the same query and re-read the first batch while the stored cursor had no + effect. +- **Exactly one result column must match `tracking_column`.** The connector + rejects a missing match and duplicate matching labels. Use explicit, unique + aliases when a query joins tables that expose the same column name. +- **The query must order by the tracking column, ascending, as the first + `ORDER BY` term.** Row limiting uses `setMaxRows`, so an unordered (or + otherwise-ordered) query returns an arbitrary subset; advancing the offset to + that subset's max would permanently skip the unread lower keys. The validator + therefore requires the tracking column to be the **first** ordering term of the + outer `ORDER BY` and rejects a descending (`DESC`) direction. Write either + `ORDER BY {tracking_column}` or the column name (optionally table-qualified, + e.g. `ORDER BY t.updated_at`). A composite order such as `ORDER BY other, id` + (tracking column not first) or a `DESC` order is rejected at `open()`. + The `ORDER BY` check is a lexical heuristic, not a full SQL parser: it inspects + the last `ORDER BY` at parenthesis depth zero, so an `ORDER BY` inside a + subquery, CTE, or a window function's `OVER (...)` is ignored rather than + mistaken for the outer ordering (a query whose only ordering sits inside such a + construct is rejected, since it has no outer `ORDER BY`). A surrounding + identifier quote (`"OrderDate"`, `` `col` ``, `[col]`) and, when + `snake_case_columns` is set, snake_case folding are both accounted for, so the + same `tracking_column` that matches a row at read time also passes validation. + The check does not interpret `UNION`; for a multi-branch `UNION` verify the + result ordering yourself. + +The connector takes the tracking value of the **last row** of each ordered batch +as the next cursor, so the cursor always matches the database's own `ORDER BY`. +The tracking column must also be: + +- **Unique / strictly increasing.** The next poll resumes with a strict + `> {last_offset}`. The connector probes one row past `batch_size`; if that row + has the same tracking value as the last row in the batch, the entire poll + fails before emitting messages or advancing the checkpoint. This prevents + silent skips, but the source cannot progress until `batch_size` exceeds that + equal-value group or the query uses a unique, strictly-increasing key. An + auto-increment ID is ideal. Keyset pagination with a tie-break is a planned + follow-up. +- **Monotonic under the database's own ordering.** Because the cursor is the last + ordered row and is fed back as a bound parameter in `WHERE {tracking_column} > + ?`, the column + must increase monotonically under the same ordering the database applies to that + `>` (including its collation, for text). Prefer an auto-increment ID or a + timestamp; a case-insensitively-collated text key can order differently than its + bytes and skip or re-read rows. +- **`NOT NULL`.** A NULL tracking value cannot be watermarked, so the connector + errors the poll if it reads one and keeps erroring (making no progress) until + the query is fixed. Exclude NULLs in the query, e.g. `AND {tracking_column} IS + NOT NULL`. +- **Round-trippable string form, for timestamps.** Timestamp columns are read as + the driver's string form and rebound through JDBC using the parameter's SQL + type; ensure the driver emits a form it can convert back (ISO-8601 is safe; a + locale format such as `MM/DD/YYYY` may not be). + +**Example:** + +```sql +-- Configuration +tracking_column = "id" +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" +initial_offset = "0" + +-- Prepared SQL for the first poll; JDBC binds parameter 1 to 0 as the type of id +SELECT * FROM users WHERE id > ? ORDER BY id + +-- After processing rows up to id=100, the SQL stays the same and parameter 1 is 100 +SELECT * FROM users WHERE id > ? ORDER BY id +``` + +## Output Format + +Each database row is converted to a JSON message: + +### With Metadata (default) + +```json +{ + "table_name": null, + "operation_type": "SELECT", + "timestamp": "2024-01-09T10:30:00Z", + "data": { + "id": 123, + "name": "John Doe", + "email": "john@example.com", + "created_at": "2024-01-08T15:20:00" + } +} +``` + +`table_name` is always `null` today: it is a reserved field, not derived +from the query. `operation_type` is always `"SELECT"`. + +### Without Metadata + +```json +{ + "id": 123, + "name": "John Doe", + "email": "john@example.com", + "created_at": "2024-01-08T15:20:00" +} +``` + +## Type Mapping + +JDBC SQL types are automatically mapped to JSON: + +| SQL Type | JSON Type | Notes | +| ---------- | ----------- | ------- | +| BIT, BOOLEAN | boolean | - | +| TINYINT, SMALLINT, INTEGER | number | Integer | +| BIGINT | string | Emitted as a string to preserve full 64-bit precision (many JSON consumers parse numbers as f64 and would lose precision above 2^53) | +| FLOAT, REAL | number | Float | +| DOUBLE | number | Double | +| NUMERIC, DECIMAL | string | Emitted as a string to preserve arbitrary precision (e.g. money) | +| CHAR, VARCHAR, TEXT | string | - | +| DATE, TIME, TIMESTAMP | string | Driver string form | +| BINARY, VARBINARY, LONGVARBINARY | string | Base64 encoded | +| NULL | null | - | + +## Runtime notes & limitations + +- **Embedded JVM, one per process.** JNI permits a single `JavaVM` per OS + process. All JDBC *source* instances in the connectors runtime share one JVM + and must configure the same `driver_jar_path` and `jvm_options`; a later source + with different values is rejected instead of silently using the first + source's classpath. A JDBC source and a JDBC sink are separate shared libraries + and **cannot both create a JVM in the same runtime process**. Run them in + separate connectors-runtime processes. +- **Blocking I/O.** JDBC calls go through JNI and are synchronous. The fetch in + `poll()` (and the close in `close()`) runs under `tokio::task::block_in_place` + so it does not monopolize a shared async-runtime worker, but the work is still + blocking; size the runtime and `poll_interval`/`batch_size` accordingly. +- **Bulk mode has no pagination beyond `batch_size`, and fails closed.** Row + limiting uses JDBC `setMaxRows`. In bulk mode the connector probes with + `batch_size + 1` rows and, if the result set is larger than `batch_size`, + **errors the poll instead of syncing a truncated subset**. For tables larger + than a batch, raise `batch_size` to cover the full result, or use incremental + mode with an ordered `tracking_column`. (Full cross-database OFFSET pagination + is a planned follow-up.) +- **Incremental boundaries fail closed on tied tracking values.** The connector + fetches one probe row beyond `batch_size`. If the probe and last in-batch row + share a tracking value, no messages are emitted and no cursor is staged. Raise + `batch_size` above the tie group or use a unique tracking column. +- **Fetched-batch delivery is at-least-once.** The offset advanced by a poll is + only *staged*; it is committed after the runtime reports that the batch was + both sent and its checkpoint durably persisted (`SourceBatchResult::Ack`). If + either step fails (`Nack`), the staged offset is discarded and the next poll + rebuilds the same query from the committed offset, so the batch is **re-read + rather than skipped** - for a transient in-process send failure as well as for + a crash or restart. Rows can therefore be delivered more than once (message + IDs are assigned by the producer and are not stable across a replay, so + downstream consumers must dedupe on a business key if they need exactly-once); + send or checkpoint failures do not silently drop an already-fetched batch. + Five consecutive Nacks stop the source; an operator must restart it after the + underlying send or checkpoint failure is fixed. The separate tracking-column + uniqueness requirement above still applies while fetching rows from the + database. +- **Connection recovery.** The connection is validated with `Connection.isValid` + each poll and transparently re-established (closing the old handle) if it has + dropped. The check runs on the shared `block_in_place` worker, so its timeout + (`connection_timeout_ms`) is intentionally converted to whole seconds and + capped at 5s: a dead connection must not block the worker for tens of seconds. +- **Bounded query execution.** Every prepared statement receives + `query_timeout_ms` through `Statement.setQueryTimeout`. JDBC drivers implement + cancellation differently, so verify timeout behavior for drivers outside the + PostgreSQL integration matrix. +- **`SQLState` classification is informational today.** Query failures are + classified into transient vs permanent error variants, but the runtime does + not yet apply differentiated backoff based on that distinction; it currently + shapes the error variant and the log message only. +- **Credentials reach the JVM heap unzeroed.** `password` is held as a + `SecretString` on the Rust side, but the JDBC API takes a `java.lang.String`, + so the password is copied onto the JVM heap as an ordinary (non-zeroed) string + for the lifetime of the connection. This is inherent to the JDBC surface and + is an accepted risk. +- **`SecretString` redaction is not end-to-end.** `jdbc_url`/`password` are typed + as `SecretString`, and this connector's `Debug` output and startup logs redact + them. That redaction does **not** cover the connectors runtime's generic + config surface: the `GET /sources/{key}/configs/plugin` control-API endpoint + and the runtime's `trace`-level config logging emit the raw, untyped + `plugin_config` (credentials included), the same as every other connector. Do + not expose the control API to untrusted callers and do not run the runtime at + `trace` level in production. This is a known runtime-wide limitation shared by + all connectors, not something this connector can address on its own. + +### Credential precedence + +Provide credentials **either** via `username` + `password` **or** embedded in the +`jdbc_url`, not both: + +- `username` and `password` must be **both set** (separate-credential auth) or + **both unset** (URL-embedded credentials). A half-set pair is rejected at + `open()`. +- When both `username`/`password` and URL-embedded credentials are present, the + driver decides precedence (typically the explicit `getConnection(url, user, + pass)` arguments win). Avoid the ambiguity by using only one method. + +## Troubleshooting + +### Connection Failures + +**Error**: "Failed to create JDBC connection" + +**Solution**: + +- Verify JDBC URL format for your database +- Check username/password +- Ensure database server is accessible +- Verify firewall rules + +### Driver Not Found + +**Error**: "Failed to find driver class" + +**Solution**: + +- Verify `driver_jar_path` points to correct JAR file +- Check `driver_class` name matches your JDBC driver +- Ensure JAR file has read permissions + +### JVM Issues + +**Error**: "Failed to create JVM" + +**Solution**: + +- Ensure Java is installed: `java -version` +- Increase JVM memory: + + ```toml + jvm_options = ["-Xmx1g", "-Xms512m"] + ``` + +### No Data Being Fetched + +**Check**: + +- Verify query returns results when run directly in database +- Check `initial_offset` value +- Review connector logs for errors +- Ensure `tracking_column` exists in query result + +## Performance Tuning + +### Optimize Batch Size + +```toml +# Small batches for low latency +batch_size = 100 +poll_interval = "5s" + +# Large batches for throughput +batch_size = 10000 +poll_interval = "1m" +``` + +### JVM Memory Tuning + +```toml +jvm_options = [ + "-Xmx1g", # Maximum heap size + "-Xms512m", # Initial heap size + "-XX:+UseG1GC" # Use G1 garbage collector +] +``` + +### Query Optimization + +- Add indexes on tracking columns +- Use efficient WHERE clauses +- Avoid SELECT * in production (specify columns) +- Consider database-specific optimizations + +## Connection String Formats + +### MySQL + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:mysql://localhost:3306/mydb" +username = "user" +password = "pass" + +# Option 2: Embedded in URL +jdbc_url = "jdbc:mysql://user:pass@localhost:3306/mydb" +``` + +### PostgreSQL + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:postgresql://localhost:5432/mydb" +username = "user" +password = "pass" + +# Option 2: Embedded in URL +jdbc_url = "jdbc:postgresql://localhost:5432/mydb?user=myuser&password=mypass" +``` + +### Oracle + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:oracle:thin:@localhost:1521:XE" +username = "system" +password = "oracle" + +# Option 2: Embedded in URL (Oracle uses @ for host) +jdbc_url = "jdbc:oracle:thin:system/oracle@localhost:1521:XE" +``` + +### SQL Server + +```toml +# Option 1: Separate credentials +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=mydb" +username = "sa" +password = "YourPassword123" + +# Option 2: Embedded in URL +jdbc_url = "jdbc:sqlserver://localhost:1433;databaseName=mydb;user=sa;password=YourPassword123" +``` + +### H2 (In-Memory) + +```toml +# No credentials needed for in-memory +jdbc_url = "jdbc:h2:mem:testdb" + +# Or with file-based +jdbc_url = "jdbc:h2:file:/data/mydb;USER=sa;PASSWORD=sa" +``` + +## Mode Comparison + +### Incremental Mode + +Uses standard JDBC prepared statements and requires an orderable tracking +column. PostgreSQL is integration-tested; validate this mode with other drivers. + +```toml +mode = "incremental" +tracking_column = "updated_at" # or "id", "created_at", etc. +query = "SELECT * FROM table WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" +initial_offset = "2024-01-01 00:00:00" # value appropriate for the column type +``` + +**Benefits:** + +- Avoids re-reading rows below the tracked offset (at-least-once, not exactly-once) +- Tracks offset automatically +- Efficient for large tables +- Works with timestamps, IDs, or other orderable, non-null columns; a unique + strictly increasing value avoids fail-closed tie boundaries + +**Database Examples** (the query must order by the tracking column; a unique, +strictly-increasing key like an auto-increment ID is safest, see the tracking +column requirements above): + +- MySQL: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; if `batch_size` splits a same-timestamp group, the poll fails closed until the batch is enlarged or a unique id is tracked) +- Oracle: `WHERE id > {last_offset} ORDER BY id` (use a monotonic key; `ROWNUM` is not a valid tracking column) +- SQL Server: `WHERE updated_at > {last_offset} ORDER BY updated_at` (timestamp; same fail-closed boundary behavior as MySQL) +- PostgreSQL: `WHERE id > {last_offset} ORDER BY id` + +### Bulk Mode + +Uses standard JDBC result-set APIs and requires no tracking column. PostgreSQL +is integration-tested; validate query and type behavior with other drivers. + +```toml +mode = "bulk" +query = "SELECT * FROM table" # Any valid SELECT query +``` + +**Benefits:** + +- No tracking column needed +- Works with any SELECT query +- Good for snapshots +- Supports complex queries with JOINs, aggregations, etc. + +**Limitation:** the result set is capped at `batch_size` rows (`setMaxRows`) with +no pagination beyond it. Rather than sync a truncated subset, bulk mode **fails +closed**: if the result set is larger than `batch_size` the poll errors. Raise +`batch_size` to cover the full table, or use incremental mode, for large tables. + +**Use Cases:** + +- Initial data load +- Periodic full snapshots (that fit within `batch_size`) +- Complex analytical queries +- Tables without tracking columns diff --git a/core/connectors/sources/jdbc_source/config.toml b/core/connectors/sources/jdbc_source/config.toml new file mode 100644 index 0000000000..8bb339ac7d --- /dev/null +++ b/core/connectors/sources/jdbc_source/config.toml @@ -0,0 +1,53 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +type = "source" +key = "jdbc" +enabled = true +version = 0 +name = "JDBC source" +path = "../../target/release/libiggy_connector_jdbc_source" +verbose = false + +[[streams]] +stream = "user_events" +topic = "users" +schema = "json" +batch_length = 100 + +[plugin_config] +jdbc_url = "jdbc:postgresql://localhost:5432/database" +driver_class = "org.postgresql.Driver" +driver_jar_path = "/tmp/jdbc-drivers/postgresql-42.7.1.jar" +# Credential precedence: provide credentials EITHER via username + password here +# OR embedded in jdbc_url, not both. username and password must be both set or +# both unset (a half-set pair is rejected at startup). If both this pair and +# URL-embedded credentials are present, the driver decides which wins; avoid the +# ambiguity by using only one method. +username = "postgres" +password = "postgres" +query = "SELECT * FROM users WHERE id > {last_offset} ORDER BY id" +poll_interval = "1s" +batch_size = 1000 +tracking_column = "id" +initial_offset = "0" +mode = "incremental" +snake_case_columns = false +include_metadata = true +connection_timeout_ms = 5000 +login_timeout_ms = 30000 +query_timeout_ms = 30000 diff --git a/core/connectors/sources/jdbc_source/src/lib.rs b/core/connectors/sources/jdbc_source/src/lib.rs new file mode 100644 index 0000000000..1a1c7b7564 --- /dev/null +++ b/core/connectors/sources/jdbc_source/src/lib.rs @@ -0,0 +1,3758 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use async_trait::async_trait; +use base64::Engine; +use chrono::{DateTime, Utc}; +use iggy_connector_sdk::{ + ConnectorState, Error, ProducedMessage, ProducedMessages, Schema, Source, + source::SourceBatchResult, source_connector, +}; +use java::sql::Types; +use jni::objects::{GlobalRef, JByteArray, JObject, JString, JThrowable, JValue}; +use jni::{JNIEnv, JavaVM}; +use regex::Regex; +use secrecy::{ExposeSecret, SecretString}; +use serde::{Deserialize, Serialize}; +use std::path::{Path, PathBuf}; +use std::sync::{Arc, Mutex, MutexGuard}; +use std::time::Duration; +use tracing::{debug, error, info, warn}; + +/// Clear any pending Java exception on the current thread. The JNI spec forbids +/// making most calls while an exception is pending; doing so aborts the whole +/// embedded JVM (the entire connectors-runtime process). Every fallible JNI +/// call in this connector clears on error before returning so a thrown Java +/// exception is never left pending for the next call on this thread. No-op when +/// nothing is pending. +fn clear_pending_exception(env: &mut JNIEnv) { + let _ = env.exception_clear(); +} + +/// Take and clear a pending Java exception, returning its class and message via +/// `Throwable.toString()`. JNI's Rust error only reports that Java threw; the +/// throwable carries the diagnostic that distinguishes a missing driver from a +/// linkage or initialization failure. +fn take_pending_java_exception(env: &mut JNIEnv) -> Option { + let throwable = match env.exception_occurred() { + Ok(throwable) if !throwable.is_null() => throwable, + Ok(_) => return None, + Err(_) => { + clear_pending_exception(env); + return None; + } + }; + clear_pending_exception(env); + throwable_string_method(env, &throwable, "toString") +} + +/// Reclaim a JNI local frame without masking the operation error that caused the +/// frame to unwind. JNI can leave an exception pending when `PopLocalFrame` +/// fails, so clear before and after the pop attempt. When both the operation and +/// frame cleanup fail, preserve the operation error because it identifies the +/// original failure. +fn finish_local_frame( + env: &mut JNIEnv, + operation_result: Result, + context: &str, + map_error: impl FnOnce(String) -> Error, +) -> Result { + clear_pending_exception(env); + // SAFETY: callers pass a null result because no local reference escapes the + // frame; successful operations return owned Rust data or a `GlobalRef`. + let pop_result = unsafe { env.pop_local_frame(&JObject::null()) }; + match pop_result { + Ok(_) => operation_result, + Err(err) => { + clear_pending_exception(env); + match operation_result { + Ok(_) => Err(map_error(format!("{context}: {err}"))), + Err(operation_error) => Err(operation_error), + } + } + } +} + +/// Best-effort `close()` on a JDBC handle used in error/cleanup paths. Clears +/// any pending exception first (`close()` is a `CallVoidMethod`, which JNI +/// forbids while an exception is pending) and again afterwards in case the +/// close itself throws. +fn best_effort_close(env: &mut JNIEnv, handle: &JObject) { + clear_pending_exception(env); + let _ = env.call_method(handle, "close", "()V", &[]); + clear_pending_exception(env); +} + +/// Evaluate a fallible JNI call; on error clear any pending Java exception and +/// return `Error::Connection` with context. Keeps a thrown Java exception from +/// being left pending for the next JNI call on this thread. +macro_rules! jni { + ($env:expr, $call:expr, $ctx:expr) => { + jni!($env, $call, $ctx, Error::Connection) + }; + ($env:expr, $call:expr, $ctx:expr, $map_error:expr) => { + match $call { + Ok(value) => value, + Err(err) => { + let java_exception = take_pending_java_exception(&mut *$env); + let detail = java_exception + .map(|exception| format!("{err}: {exception}")) + .unwrap_or_else(|| err.to_string()); + return Err($map_error(format!("{}: {detail}", $ctx))); + } + } + }; +} + +/// Lock a mutex, mapping a poisoned lock to a returned `Error` instead of +/// panicking. A thread that panicked while holding the lock then surfaces as a +/// logged, recoverable failure rather than a permanent panic loop on every +/// subsequent poll. +fn lock_mutex<'a, T>(mutex: &'a Mutex, what: &str) -> Result, Error> { + mutex + .lock() + .map_err(|_| Error::Connection(format!("{what} mutex poisoned"))) +} + +/// Cached compiled regex patterns for password sanitization +static RE_USER_PASS_AT: std::sync::LazyLock = + std::sync::LazyLock::new(|| Regex::new(r"://([^:/?#]+):([^/?#]*)@").unwrap()); +static RE_PASSWORD_PARAM: std::sync::LazyLock = + std::sync::LazyLock::new(|| Regex::new(r"(?i)(password|pwd|pass)=([^;&\s]+)").unwrap()); +static RE_ORACLE_PASS: std::sync::LazyLock = + std::sync::LazyLock::new(|| Regex::new(r"thin:([^/]+)/([^?#]*)@").unwrap()); + +/// Regexes that remove the incremental offset predicate `{tracking_column} > +/// {last_offset}` on the first (no-offset) poll while preserving any other +/// `WHERE` conditions. All are case- and whitespace-tolerant. They are applied in +/// order so an operator-added companion condition (e.g. +/// `AND {tracking_column} IS NOT NULL`) does not leave a dangling `WHERE ... AND` +/// or leading `AND ...` fragment. See [`strip_offset_predicate`]. +static RE_OFFSET_PREDICATE_AND_AFTER: std::sync::LazyLock = std::sync::LazyLock::new(|| { + Regex::new(r"(?i)\bWHERE\s+\{tracking_column\}\s*>\s*\{last_offset\}\s+AND\s+").unwrap() +}); +static RE_OFFSET_PREDICATE_AND_BEFORE: std::sync::LazyLock = + std::sync::LazyLock::new(|| { + Regex::new(r"(?i)\s+AND\s+\{tracking_column\}\s*>\s*\{last_offset\}").unwrap() + }); +static RE_OFFSET_PREDICATE_BARE: std::sync::LazyLock = std::sync::LazyLock::new(|| { + Regex::new(r"(?i)\bWHERE\s+\{tracking_column\}\s*>\s*\{last_offset\}").unwrap() +}); + +const CONNECTOR_NAME: &str = "JDBC source"; + +/// Poll interval used when `poll_interval` is unset or empty. +const DEFAULT_POLL_INTERVAL: Duration = Duration::from_secs(5); + +/// Parse the configured `poll_interval` humantime string into a `Duration`, +/// falling back to [`DEFAULT_POLL_INTERVAL`] when unset, empty, or unparsable. +/// `validate_config` separately rejects a set-but-unparsable value so a typo +/// surfaces at `open()` rather than silently defaulting. +fn parse_poll_interval(poll_interval: Option<&str>) -> Duration { + poll_interval + .map(str::trim) + .filter(|value| !value.is_empty()) + .and_then(|value| humantime::parse_duration(value).ok()) + .unwrap_or(DEFAULT_POLL_INTERVAL) +} + +/// Source mode for the JDBC connector +#[derive(Debug, Clone, Deserialize, Serialize, PartialEq)] +#[serde(rename_all = "lowercase")] +pub enum Mode { + /// Re-run the query on every poll, capped at `batch_size` rows via JDBC + /// `setMaxRows`. There is no pagination beyond that cap, so rather than sync a + /// truncated subset, bulk mode fails closed: the poll probes with + /// `batch_size + 1` rows and errors if the result set is larger than + /// `batch_size`. Raise `batch_size` to cover the full table, or use + /// incremental mode, for tables larger than a batch. + Bulk, + /// Track the last offset and fetch only rows beyond it on each poll. + Incremental, +} + +/// Configuration for JDBC source connector. +/// +/// Deliberately does NOT derive `Serialize`: the runtime only ever deserializes +/// this from config, and serializing it would risk writing `jdbc_url`/`password` +/// (both `SecretString`) to logs or state in plaintext. Redacted logging goes +/// through the manual `Debug` impl below. +#[derive(Clone, Deserialize)] +pub struct JdbcSourceConfig { + /// JDBC connection URL (e.g., "jdbc:mysql://localhost:3306/mydb") + /// Can include credentials: "jdbc:mysql://localhost:3306/mydb?user=root&password=secret" + pub jdbc_url: SecretString, + + /// JDBC driver class name (e.g., "com.mysql.cj.jdbc.Driver") + pub driver_class: String, + + /// Path to JDBC driver JAR file + pub driver_jar_path: String, + + /// Database username (optional if included in jdbc_url) + #[serde(default)] + pub username: Option, + + /// Database password (optional if included in jdbc_url) + #[serde(default)] + pub password: Option, + + /// SQL query to execute for fetching data + /// Can use {last_offset} placeholder for incremental reads + pub query: String, + + /// Polling interval as a humantime string (e.g., "30s", "5m", "1h"). Parsed + /// once at construction; defaults to 5s when unset. + #[serde(default)] + pub poll_interval: Option, + + /// Batch size - maximum rows to fetch per poll + #[serde(default = "default_batch_size")] + pub batch_size: u32, + + /// Tracking column for incremental reads (e.g., "id", "updated_at") + #[serde(default)] + pub tracking_column: Option, + + /// Initial offset value for the first poll + #[serde(default)] + pub initial_offset: Option, + + /// Source mode: "bulk" (full table scan) or "incremental" (track last offset) + #[serde(default = "default_mode")] + pub mode: Mode, + + /// Convert column names to snake_case + #[serde(default)] + pub snake_case_columns: bool, + + /// Include metadata in output (table name, operation type, timestamp) + #[serde(default = "default_true")] + pub include_metadata: bool, + + /// Log per-poll row and column counts at info instead of debug. + #[serde(default)] + pub verbose_logging: Option, + + /// JVM options (e.g., ["-Xmx512m", "-Xms128m"]) + #[serde(default)] + pub jvm_options: Vec, + + /// Timeout for the per-poll `Connection.isValid` liveness check (default: + /// 5000). JDBC expresses this timeout in whole seconds, so the value is + /// converted to seconds and clamped to the 1..=5s range (the check runs on a + /// shared worker and must stay short); it does not govern connection + /// establishment. + #[serde(default = "default_connection_timeout")] + pub connection_timeout_ms: u64, + + /// Bound on establishing the JDBC connection (default: 30000). Applied via + /// `DriverManager.setLoginTimeout`, which is expressed in whole seconds, so + /// the value is rounded up to at least 1s. Prevents an unreachable or + /// slow-DNS database from hanging `open()` (and thus the whole connectors + /// runtime, which opens sources sequentially at startup) indefinitely. + #[serde(default = "default_login_timeout")] + pub login_timeout_ms: u64, + + /// Bound on each query execution (default: 30000). Applied through JDBC's + /// `Statement.setQueryTimeout`, which uses whole seconds, so the configured + /// value is rounded up to at least one second. + #[serde(default = "default_query_timeout")] + pub query_timeout_ms: u64, +} + +fn default_connection_timeout() -> u64 { + 5000 +} + +fn default_login_timeout() -> u64 { + 30000 +} + +fn default_query_timeout() -> u64 { + 30000 +} + +fn default_batch_size() -> u32 { + 1000 +} + +fn default_mode() -> Mode { + Mode::Incremental +} + +fn default_true() -> bool { + true +} + +impl std::fmt::Debug for JdbcSourceConfig { + fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result { + f.debug_struct("JdbcSourceConfig") + .field( + "jdbc_url", + &sanitize_jdbc_url(self.jdbc_url.expose_secret()), + ) + .field("driver_class", &self.driver_class) + .field("driver_jar_path", &self.driver_jar_path) + .field("username", &self.username) + .field("password", &self.password.as_ref().map(|_| "***")) + .field("query", &self.query) + .field("poll_interval", &self.poll_interval) + .field("batch_size", &self.batch_size) + .field("tracking_column", &self.tracking_column) + .field("initial_offset", &self.initial_offset) + .field("mode", &self.mode) + .field("snake_case_columns", &self.snake_case_columns) + .field("include_metadata", &self.include_metadata) + .field("verbose_logging", &self.verbose_logging) + .field("connection_timeout_ms", &self.connection_timeout_ms) + .field("login_timeout_ms", &self.login_timeout_ms) + .field("query_timeout_ms", &self.query_timeout_ms) + .finish() + } +} + +/// Internal state tracking for the JDBC source +#[derive(Debug, Clone, Default, Serialize, Deserialize)] +struct State { + /// Last tracked offset value (for incremental mode) + last_offset: Option, + + /// Total rows processed + processed_rows: u64, +} + +#[derive(Debug)] +struct ColumnMetadata { + output_name: String, + sql_type: i32, + is_tracking: bool, +} + +#[derive(Debug, PartialEq, Eq)] +struct PreparedQuery { + sql: String, + offset: Option, + offset_parameter_count: usize, +} + +/// Database record structure for output messages +#[derive(Debug, Serialize, Deserialize)] +pub struct DatabaseRecord { + pub table_name: Option, + pub operation_type: String, + pub timestamp: DateTime, + pub data: serde_json::Value, +} + +/// JDBC Source Connector +#[derive(Debug)] +pub struct JdbcSource { + id: u32, + config: JdbcSourceConfig, + jvm: Option>, + // Behind a Mutex so `poll()` (&self) can transparently re-establish a dead + // direct connection without `&mut self`. + connection: Mutex>, + // The committed cursor: only ever advanced from `pending_state` once the + // runtime confirms the batch was both sent and its checkpoint persisted. + state: Mutex, + // The cursor this in-flight batch would advance to, staged by `poll` and + // resolved by `on_batch_result`. The SDK keeps at most one batch in flight, + // so a single slot is sufficient. + pending_state: Mutex>, + // Poll interval parsed once from `config.poll_interval` at construction. + poll_interval: Duration, + // Whether per-poll details should be promoted from debug to info. + verbose: bool, +} + +/// Sanitize JDBC URL by masking passwords for logging +fn sanitize_jdbc_url(url: &str) -> String { + // Pattern 1: user:password@host format (MySQL, PostgreSQL) + let url = RE_USER_PASS_AT.replace_all(url, "://$1:***@"); + + // Pattern 2: password=value format (PostgreSQL, SQL Server, H2) + let url = RE_PASSWORD_PARAM.replace_all(&url, "$1=***"); + + // Pattern 3: Oracle user/password@host format + let url = RE_ORACLE_PASS.replace_all(&url, "thin:$1/***@"); + + url.to_string() +} + +fn sanitize_jdbc_error(error: Error, jdbc_url: &str) -> Error { + let sanitized_url = sanitize_jdbc_url(jdbc_url); + let sanitize = |message: String| message.replace(jdbc_url, &sanitized_url); + match error { + Error::InitError(message) => Error::InitError(sanitize(message)), + Error::Connection(message) => Error::Connection(sanitize(message)), + error => error, + } +} + +impl JdbcSource { + /// Create a new JDBC source connector + pub fn new(id: u32, config: JdbcSourceConfig, connector_state: Option) -> Self { + // Restore state from persistent storage if available + let state = connector_state + .and_then(|cs| cs.deserialize::(CONNECTOR_NAME, id)) + .unwrap_or_else(|| { + let mut default_state = State::default(); + // Use initial_offset from config if provided + if let Some(ref initial_offset) = config.initial_offset { + default_state.last_offset = Some(initial_offset.clone()); + } + default_state + }); + + let poll_interval = parse_poll_interval(config.poll_interval.as_deref()); + let verbose = config.verbose_logging.unwrap_or(false); + Self { + id, + config, + jvm: None, + connection: Mutex::new(None), + state: Mutex::new(state), + pending_state: Mutex::new(None), + poll_interval, + verbose, + } + } + + /// Obtain the process-wide JVM, creating it on first use. JNI permits only a + /// single JVM per OS process, so this is shared across all JDBC connector + /// instances (see [`get_or_create_jvm`]). + fn initialize_jvm(&mut self) -> Result<(), Error> { + info!("Initializing JVM for JDBC source connector [{}]", self.id); + let jvm = get_or_create_jvm( + &self.config.driver_jar_path, + &self.config.jvm_options, + self.id, + )?; + self.jvm = Some(jvm); + Ok(()) + } + + /// Load the JDBC driver and open the database connection. + fn create_connection(&mut self) -> Result<(), Error> { + let jvm = self + .jvm + .as_ref() + .ok_or_else(|| Error::InitError("JVM not initialized".to_string()))?; + + let mut env = jvm + .attach_current_thread() + .map_err(|e| Error::InitError(format!("Failed to attach thread to JVM: {}", e)))?; + + info!( + "{CONNECTOR_NAME} connector [{}] creating direct connection to: {}", + self.id, + sanitize_jdbc_url(self.config.jdbc_url.expose_secret()) + ); + let conn = self.create_direct_connection_internal(&mut env)?; + *lock_mutex(&self.connection, "connection")? = Some(conn); + + Ok(()) + } + + /// Create a direct JDBC connection via DriverManager, inside its own JNI + /// local-reference frame so the transient class/loader/URL locals do not + /// accumulate on the caller's frame. The returned handle is a `GlobalRef`, so + /// it survives the frame pop. + fn create_direct_connection_internal(&self, env: &mut JNIEnv) -> Result { + jni!( + env, + env.push_local_frame(24), + "Failed to push connection local frame", + Error::InitError + ); + let result = self + .create_direct_connection_inner(env) + .map_err(|error| sanitize_jdbc_error(error, self.config.jdbc_url.expose_secret())); + finish_local_frame( + env, + result, + "Failed to pop connection local frame", + Error::InitError, + ) + } + + fn create_direct_connection_inner(&self, env: &mut JNIEnv) -> Result { + // Use the system loader explicitly for both driver initialization and the + // thread context. This keeps DriverManager's view of the driver aligned + // with the JVM classpath even when invoked from an attached native thread. + let current_thread_class = jni!( + env, + env.find_class("java/lang/Thread"), + "Failed to find Thread class", + Error::InitError + ); + + let current_thread = jni!( + env, + env.call_static_method( + current_thread_class, + "currentThread", + "()Ljava/lang/Thread;", + &[], + ) + .and_then(|v| v.l()), + "Failed to get current thread", + Error::InitError + ); + + let class_loader_class = jni!( + env, + env.find_class("java/lang/ClassLoader"), + "Failed to find ClassLoader", + Error::InitError + ); + + let system_class_loader = jni!( + env, + env.call_static_method( + class_loader_class, + "getSystemClassLoader", + "()Ljava/lang/ClassLoader;", + &[], + ) + .and_then(|v| v.l()), + "Failed to get system class loader", + Error::InitError + ); + + jni!( + env, + env.call_method( + ¤t_thread, + "setContextClassLoader", + "(Ljava/lang/ClassLoader;)V", + &[JValue::Object(&system_class_loader)], + ), + "Failed to set context class loader", + Error::InitError + ); + + info!( + "{CONNECTOR_NAME} connector [{}] loading driver via the system class loader: {}", + self.id, self.config.driver_class + ); + + let class_class = jni!( + env, + env.find_class("java/lang/Class"), + "Failed to find Class", + Error::InitError + ); + let driver_class_name = jni!( + env, + env.new_string(&self.config.driver_class), + "Failed to create class name string", + Error::InitError + ); + jni!( + env, + env.call_static_method( + class_class, + "forName", + "(Ljava/lang/String;ZLjava/lang/ClassLoader;)Ljava/lang/Class;", + &[ + JValue::Object(&driver_class_name.into()), + JValue::Bool(1), + JValue::Object(&system_class_loader), + ], + ), + format!("Failed to load driver class '{}'", self.config.driver_class), + Error::InitError + ); + + info!( + "{CONNECTOR_NAME} connector [{}] loaded and registered the JDBC driver", + self.id + ); + + // Get connection from DriverManager + let driver_manager = jni!( + env, + env.find_class("java/sql/DriverManager"), + "Failed to find DriverManager", + Error::InitError + ); + + let jdbc_url = jni!( + env, + env.new_string(self.config.jdbc_url.expose_secret()), + "Failed to create JDBC URL string", + Error::InitError + ); + + // DriverManager's login timeout is process-global. Keep setting it and + // opening the corresponding connection in one critical section so two + // connector instances cannot apply each other's timeout. + let _driver_manager_guard = lock_mutex( + &DRIVER_MANAGER_CONNECT_LOCK, + "JDBC DriverManager connection", + )?; + + // Bound connection establishment so an unreachable or slow-DNS database + // fails instead of hanging open() (which the runtime drives sequentially + // at startup) forever. setLoginTimeout is process-wide and in whole + // seconds, so round the configured milliseconds up to at least 1s. + let login_timeout_secs = self + .config + .login_timeout_ms + .div_ceil(1000) + .clamp(1, i32::MAX as u64) as i32; + jni!( + env, + env.call_static_method( + &driver_manager, + "setLoginTimeout", + "(I)V", + &[JValue::Int(login_timeout_secs)], + ) + .and_then(|v| v.v()), + "Failed to set JDBC login timeout", + Error::InitError + ); + + // If username/password are provided separately, use 3-arg getConnection + let connection_obj = if let (Some(username), Some(password)) = + (&self.config.username, &self.config.password) + { + info!( + "{CONNECTOR_NAME} connector [{}] using separate username/password authentication", + self.id + ); + let username_jstring = jni!( + env, + env.new_string(username), + "Failed to create username string", + Error::InitError + ); + let password_jstring = jni!( + env, + env.new_string(password.expose_secret()), + "Failed to create password string", + Error::InitError + ); + + jni!( + env, + env.call_static_method( + driver_manager, + "getConnection", + "(Ljava/lang/String;Ljava/lang/String;Ljava/lang/String;)Ljava/sql/Connection;", + &[ + JValue::Object(&jdbc_url.into()), + JValue::Object(&username_jstring.into()), + JValue::Object(&password_jstring.into()), + ], + ) + .and_then(|v| v.l()), + "Failed to create JDBC connection with credentials", + Error::InitError + ) + } else { + info!( + "{CONNECTOR_NAME} connector [{}] using credentials from the connection string", + self.id + ); + jni!( + env, + env.call_static_method( + driver_manager, + "getConnection", + "(Ljava/lang/String;)Ljava/sql/Connection;", + &[JValue::Object(&jdbc_url.into())], + ) + .and_then(|v| v.l()), + "Failed to create JDBC connection from URL", + Error::InitError + ) + }; + + let global_ref = jni!( + env, + env.new_global_ref(connection_obj), + "Failed to create global reference", + Error::InitError + ); + + info!( + "{CONNECTOR_NAME} connector [{}] established the database connection", + self.id + ); + Ok(global_ref) + } + + /// Acquire the direct connection, transparently re-establishing it if it has + /// dropped since the last poll. + fn get_connection<'local>(&self, env: &mut JNIEnv<'local>) -> Result, Error> { + // Hold the connection lock across the whole check/close/create/store + // sequence so correctness does not depend on there being a single caller: + // a concurrent caller can no longer observe or replace the handle + // mid-reconnect. The lock is a std Mutex held only across synchronous JNI + // work (no .await), and neither `connection_is_valid` nor + // `create_direct_connection_internal` re-locks it, so this cannot deadlock. + let mut guard = lock_mutex(&self.connection, "connection")?; + + let needs_reconnect = match guard.as_ref() { + Some(conn) => !self.connection_is_valid(env, conn.as_obj()), + None => true, + }; + + if needs_reconnect { + info!( + "{CONNECTOR_NAME} connector [{}] connection is invalid; re-establishing", + self.id + ); + // Best-effort close of the old handle, then drop it before creating + // the replacement so a failed reconnect leaves no stale reference. + if let Some(old) = guard.as_ref() { + best_effort_close(env, old.as_obj()); + } + *guard = None; + *guard = Some(self.create_direct_connection_internal(env)?); + } + + let conn = guard + .as_ref() + .ok_or_else(|| Error::Connection("No connection available".to_string()))?; + let local_ref = jni!( + env, + env.new_local_ref(conn.as_obj()), + "Failed to create connection local reference" + ); + Ok(local_ref) + } + + /// Best-effort `Connection.isValid(timeout)` check. Returns false on any + /// JNI error so the caller re-establishes the connection. + fn connection_is_valid(&self, env: &mut JNIEnv, conn: &JObject) -> bool { + // Cap the liveness check short: it runs on a shared block_in_place worker, + // so a dead connection must not block it for tens of seconds. + let timeout_secs = (self.config.connection_timeout_ms / 1000).clamp(1, 5) as i32; + match env + .call_method(conn, "isValid", "(I)Z", &[JValue::Int(timeout_secs)]) + .and_then(|v| v.z()) + { + Ok(valid) => valid, + Err(_) => { + // isValid may throw; clear so the reconnect path's next JNI call + // is not made with an exception pending. + clear_pending_exception(env); + false + } + } + } + + /// Execute query and fetch results, returning the messages together with the + /// cursor this batch *would* advance to. + /// + /// The candidate cursor is deliberately not written to `self.state` here: the + /// caller stages it and commits it only once the runtime acknowledges that the + /// batch was both sent and its checkpoint persisted. Advancing the in-memory + /// cursor at fetch time would permanently skip a batch whose send later failed. + /// + /// The mutex is held only briefly: once to read the current offset for query + /// building, and once after the JNI work to snapshot the counters. + fn execute_query( + &self, + env: &mut JNIEnv, + ) -> Result<(Vec, Option), Error> { + let connection = self.get_connection(env)?; + + // Read current state snapshot (short lock) + let query = { + let state = lock_mutex(&self.state, "state")?; + self.build_query(&state) + }?; + // Logged at debug: the built query embeds the substituted offset value. + debug!( + "{CONNECTOR_NAME} connector [{}] executing query: {}", + self.id, query.sql + ); + + let (messages, row_count, max_offset) = + self.execute_statement_and_fetch_rows(env, &connection, &query)?; + + // Fail closed on bulk truncation. The bulk fetch probes with batch_size+1 + // rows, so seeing more than batch_size means the result set is larger than + // one batch and bulk mode (which has no pagination) would otherwise sync + // only an arbitrary subset. Erroring surfaces the misconfiguration instead + // of silently dropping rows. Full cross-database OFFSET pagination is a + // separate follow-up. + if self.config.mode == Mode::Bulk && row_count > self.config.batch_size as u64 { + return Err(Error::InvalidConfigValue(format!( + "bulk query returned more than batch_size ({}) rows; bulk mode does not paginate \ + and would sync only a truncated subset. Increase batch_size to cover the full \ + result set, or use incremental mode with an ordered tracking_column.", + self.config.batch_size + ))); + } + + // An empty poll moves no cursor, so there is nothing to checkpoint. + // Returning no candidate state keeps an idle table from rewriting an + // unchanged checkpoint every interval, which on the HTTP state backend + // would be a network round-trip per poll. + if row_count == 0 { + return Ok((messages, None)); + } + + // Snapshot the cursor this batch would commit (short lock). + let candidate = { + let state = lock_mutex(&self.state, "state")?; + State { + last_offset: max_offset.or_else(|| state.last_offset.clone()), + processed_rows: state.processed_rows.saturating_add(row_count), + } + }; + if self.verbose { + info!( + "JDBC source connector [{}] fetched {} rows, {} processed once this batch is acknowledged", + self.id, row_count, candidate.processed_rows + ); + } else { + debug!( + "JDBC source connector [{}] fetched {} rows, {} processed once this batch is acknowledged", + self.id, row_count, candidate.processed_rows + ); + } + + Ok((messages, Some(candidate))) + } + + /// Prepare a JDBC statement, execute it, and read all result rows into messages. + fn execute_statement_and_fetch_rows( + &self, + env: &mut JNIEnv, + connection: &JObject, + query: &PreparedQuery, + ) -> Result<(Vec, u64, Option), Error> { + let query_jstring = jni!( + env, + env.new_string(&query.sql), + "Failed to create query string" + ); + + let statement = match env + .call_method( + connection, + "prepareStatement", + "(Ljava/lang/String;)Ljava/sql/PreparedStatement;", + &[JValue::Object(&query_jstring.into())], + ) + .and_then(|v| v.l()) + { + Ok(s) => s, + Err(_) => return Err(classify_query_failure(env, "prepare statement")), + }; + + if let Err(error) = bind_offset_parameters(env, &statement, query) { + best_effort_close(env, &statement); + return Err(error); + } + + // Use setMaxRows for database-agnostic row limiting instead of SQL LIMIT. + // Both modes fetch one probe row beyond the batch. Bulk mode uses it to + // detect unsupported pagination; incremental mode uses it to detect an + // equal tracking value split across the boundary before any row is emitted. + let max_rows = self.config.batch_size.saturating_add(1); + if let Err(err) = env.call_method( + &statement, + "setMaxRows", + "(I)V", + &[JValue::Int(max_rows.min(i32::MAX as u32) as i32)], + ) { + best_effort_close(env, &statement); + return Err(Error::Connection(format!("Failed to set max rows: {err}"))); + } + + let query_timeout_secs = self + .config + .query_timeout_ms + .div_ceil(1000) + .clamp(1, i32::MAX as u64) as i32; + if let Err(err) = env.call_method( + &statement, + "setQueryTimeout", + "(I)V", + &[JValue::Int(query_timeout_secs)], + ) { + clear_pending_exception(env); + best_effort_close(env, &statement); + return Err(Error::Connection(format!( + "Failed to set query timeout: {err}" + ))); + } + + let result_set = match env + .call_method(&statement, "executeQuery", "()Ljava/sql/ResultSet;", &[]) + .and_then(|v| v.l()) + { + Ok(rs) => rs, + Err(_) => { + // Read and classify the pending SQLException FIRST: this clears + // it, which then makes the statement close() safe (close() is a + // JNI call and must not run with an exception pending). + let error = classify_query_failure(env, "execute query"); + best_effort_close(env, &statement); + return Err(error); + } + }; + + // On any read error the callees clear the pending exception; close the + // statement here before propagating so it is not leaked. + let columns = match self.read_column_metadata(env, &result_set) { + Ok(columns) => columns, + Err(err) => { + best_effort_close(env, &statement); + return Err(err); + } + }; + let (messages, row_count, max_offset) = match self.read_rows(env, &result_set, &columns) { + Ok(result) => result, + Err(err) => { + best_effort_close(env, &statement); + return Err(err); + } + }; + + // Close statement (best-effort) + best_effort_close(env, &statement); + + Ok((messages, row_count, max_offset)) + } + + /// Read column names and types from the ResultSet metadata. + fn read_column_metadata( + &self, + env: &mut JNIEnv, + result_set: &JObject, + ) -> Result, Error> { + // Read all column metadata inside its own JNI local frame so the + // metadata object and per-column name references are reclaimed; a very + // wide table would otherwise accumulate one local ref per column on the + // outer frame for the whole poll. + jni!( + env, + env.push_local_frame(16), + "Failed to push metadata local frame" + ); + let result = self.read_column_metadata_inner(env, result_set); + finish_local_frame( + env, + result, + "Failed to pop metadata local frame", + Error::Connection, + ) + } + + fn read_column_metadata_inner( + &self, + env: &mut JNIEnv, + result_set: &JObject, + ) -> Result, Error> { + let metadata = jni!( + env, + env.call_method( + result_set, + "getMetaData", + "()Ljava/sql/ResultSetMetaData;", + &[], + ) + .and_then(|v| v.l()), + "Failed to get metadata" + ); + + let column_count = jni!( + env, + env.call_method(&metadata, "getColumnCount", "()I", &[]) + .and_then(|v| v.i()), + "Failed to get column count" + ); + + if self.verbose { + info!( + "JDBC source connector [{}] query returned {} columns", + self.id, column_count + ); + } else { + debug!( + "JDBC source connector [{}] query returned {} columns", + self.id, column_count + ); + } + + // Clamp a driver-supplied count before using it as an allocation size: a + // negative i32 would sign-extend to an enormous usize and abort on alloc. + let mut raw_columns = Vec::with_capacity((column_count.max(0) as usize).min(8192)); + for i in 1..=column_count { + let source_name = get_column_label(env, &metadata, i)?; + let sql_type = get_column_type(env, &metadata, i)?; + raw_columns.push((source_name, sql_type)); + } + + self.prepare_column_metadata(raw_columns) + } + + fn prepare_column_metadata( + &self, + raw_columns: Vec<(String, i32)>, + ) -> Result, Error> { + let mut columns = Vec::with_capacity(raw_columns.len()); + let mut output_names = std::collections::HashSet::new(); + let mut tracking_matches = 0usize; + for (source_name, sql_type) in raw_columns { + let output_name = if self.config.snake_case_columns { + to_snake_case(&source_name) + } else { + source_name.clone() + }; + if !output_names.insert(output_name.clone()) { + return Err(Error::InvalidConfigValue(format!( + "query result contains duplicate output key '{output_name}' after mapping column '{source_name}'; use unique column aliases" + ))); + } + let is_tracking = self + .config + .tracking_column + .as_deref() + .is_some_and(|tracking| { + tracking_column_matches(tracking, &source_name, &output_name) + }); + if is_tracking { + tracking_matches += 1; + } + columns.push(ColumnMetadata { + output_name, + sql_type, + is_tracking, + }); + } + + if self.config.mode == Mode::Incremental { + let tracking_column = self.config.tracking_column.as_deref().unwrap_or(""); + match tracking_matches { + 1 => {} + 0 => { + return Err(Error::InvalidConfigValue(format!( + "tracking column '{tracking_column}' is not present in the query result; include it exactly once in the SELECT list" + ))); + } + count => { + return Err(Error::InvalidConfigValue(format!( + "tracking column '{tracking_column}' matches {count} columns in the query result; use unique column aliases so it matches exactly once" + ))); + } + } + } + + Ok(columns) + } + + /// Iterate over result set rows and convert each to a ProducedMessage. + fn read_rows( + &self, + env: &mut JNIEnv, + result_set: &JObject, + columns: &[ColumnMetadata], + ) -> Result<(Vec, u64, Option), Error> { + // The emitted batch is capped at batch_size; the possible extra row is a + // probe and is never retained. Clamp the pre-allocation so an extreme + // batch_size cannot request an absurd allocation up front. + let mut messages = Vec::with_capacity((self.config.batch_size as usize).min(8192)); + let mut row_count: u64 = 0; + let mut last_offset: Option = None; + + loop { + let has_next = jni!( + env, + env.call_method(result_set, "next", "()Z", &[]) + .and_then(|v| v.z()), + "Failed to fetch next row" + ); + + if !has_next { + break; + } + + // Read each row inside its own JNI local-reference frame so the per + // -column local refs (getObject/getString/getBytes results) are + // reclaimed every iteration; otherwise a large result set would + // overflow the JNI local reference table and abort the JVM. + jni!( + env, + env.push_local_frame(32), + "Failed to push row local frame" + ); + let row_result = self.read_single_row(env, result_set, columns); + let (row_data, offset) = finish_local_frame( + env, + row_result, + "Failed to pop row local frame", + Error::Connection, + )?; + + // The extra row is a boundary probe, never part of the emitted batch. + // If it shares the last in-batch tracking value, advancing with strict + // `>` would skip it and any following ties. Fail the complete poll so + // no messages or checkpoint can escape from an unsafe page. + if self.config.mode == Mode::Incremental && row_count == self.config.batch_size as u64 { + if offset == last_offset { + return Err(Error::InvalidConfigValue(format!( + "incremental batch boundary splits rows with tracking value '{}'; increase batch_size or use a unique, strictly increasing tracking_column", + offset.as_deref().unwrap_or("") + ))); + } + break; + } + + // Take the tracking value of the LAST row as the next offset. Rows + // arrive in ascending tracking order (validate_config enforces + // ORDER BY the tracking column ascending), so the last row is the + // high-water mark. Using the last row rather than a Rust-side max + // keeps the cursor consistent with the database's own ordering. + // + if let Some(offset) = offset { + last_offset = Some(offset); + } + + let message = self.build_message(row_data)?; + messages.push(message); + row_count += 1; + } + + Ok((messages, row_count, last_offset)) + } + + /// Extract data from a single result set row, returning the row map and optional offset. + fn read_single_row( + &self, + env: &mut JNIEnv, + result_set: &JObject, + columns: &[ColumnMetadata], + ) -> Result<(serde_json::Map, Option), Error> { + let mut row_data = serde_json::Map::new(); + let mut offset = None; + + for (idx, column) in columns.iter().enumerate() { + let col_idx = (idx + 1) as i32; + let value = extract_column_value(env, result_set, col_idx, &column.sql_type)?; + row_data.insert(column.output_name.clone(), value); + + if column.is_tracking + && let Some(ref tracking_column) = self.config.tracking_column + { + let value = extract_offset_value(&row_data, &column.output_name); + offset = self.tracking_offset_or_error(value, tracking_column)?; + } + } + + Ok((row_data, offset)) + } + + /// Resolve a tracking-column value into an offset. In incremental mode a NULL + /// or empty value is a hard error: the batch cannot be safely emitted because + /// its checkpoint could not advance past that row. + fn tracking_offset_or_error( + &self, + value: Option, + tracking_col: &str, + ) -> Result, Error> { + match value { + Some(value) => Ok(Some(value)), + None if self.config.mode == Mode::Incremental => { + Err(Error::InvalidRecordValue(format!( + "tracking column '{tracking_col}' is NULL or empty in a returned row; incremental \ + mode cannot advance its offset past NULL values. Exclude them in the query, e.g. \ + add `AND {tracking_col} IS NOT NULL`." + ))) + } + None => Ok(None), + } + } + + /// Build a ProducedMessage from row data, optionally wrapping in DatabaseRecord metadata. + fn build_message( + &self, + row_data: serde_json::Map, + ) -> Result { + let now = Utc::now(); + let payload = if self.config.include_metadata { + let record = DatabaseRecord { + table_name: None, + operation_type: "SELECT".to_string(), + timestamp: now, + data: serde_json::Value::Object(row_data), + }; + serde_json::to_vec(&record) + .map_err(|e| Error::Serialization(format!("Failed to serialize record: {e}")))? + } else { + serde_json::to_vec(&serde_json::Value::Object(row_data)) + .map_err(|e| Error::Serialization(format!("Failed to serialize row data: {e}")))? + }; + + Ok(ProducedMessage { + id: None, + payload, + headers: None, + checksum: None, + timestamp: None, + origin_timestamp: None, + }) + } + + /// Build the query for this poll by substituting `{tracking_column}` and + /// converting each `{last_offset}` into a bound JDBC parameter. Row limiting + /// is handled via JDBC setMaxRows rather than SQL LIMIT. + fn build_query(&self, state: &State) -> Result { + let mut query = self.config.query.clone(); + + if self.config.mode != Mode::Incremental { + return Ok(PreparedQuery { + sql: finalize_query(query)?, + offset: None, + offset_parameter_count: 0, + }); + } + + let offset = state + .last_offset + .as_deref() + .or(self.config.initial_offset.as_deref()); + + // Without an offset yet, drop the incremental predicate but keep the rest + // of the query (e.g. an ORDER BY) intact. + if offset.is_none() { + query = strip_offset_predicate(&query); + } + + // Substitute {tracking_column} wherever it still appears (a WHERE and/or + // an ORDER BY), validating it as a plain identifier to avoid injection. + if query.contains("{tracking_column}") { + let column = self.config.tracking_column.as_deref().ok_or_else(|| { + Error::InvalidConfigValue( + "query uses {tracking_column} but tracking_column is not set".to_string(), + ) + })?; + if !is_valid_identifier(column) { + return Err(Error::InvalidConfigValue(format!( + "tracking_column '{column}' is not a valid SQL identifier" + ))); + } + query = query.replace("{tracking_column}", column); + } + + let offset_parameter_count = query.matches("{last_offset}").count(); + if offset.is_some() { + query = query.replace("{last_offset}", "?"); + } + + Ok(PreparedQuery { + sql: finalize_query(query)?, + offset: offset.map(str::to_owned), + offset_parameter_count, + }) + } +} + +/// Get the column label from ResultSetMetaData. Uses `getColumnLabel` (not +/// `getColumnName`) so a `SELECT expr AS alias` yields the alias the caller +/// asked for; `getColumnName` returns the underlying base-table column (or +/// empty for computed columns), which would not match a configured +/// `tracking_column` alias and can be blank. +fn get_column_label( + env: &mut JNIEnv, + metadata: &JObject, + column_index: i32, +) -> Result { + let col_name_obj = jni!( + env, + env.call_method( + metadata, + "getColumnLabel", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get column label" + ); + + let col_name: String = jni!( + env, + env.get_string(&JString::from(col_name_obj)), + "Failed to convert column name" + ) + .into(); + + Ok(col_name) +} + +/// Get column type from ResultSetMetaData. +fn get_column_type(env: &mut JNIEnv, metadata: &JObject, column_index: i32) -> Result { + let col_type = jni!( + env, + env.call_method( + metadata, + "getColumnType", + "(I)I", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.i()), + "Failed to get column type" + ); + + Ok(col_type) +} + +/// Return `value`, or JSON `null` when the last primitive getter read a SQL +/// NULL (detected via `ResultSet.wasNull()`). +fn null_or( + env: &mut JNIEnv, + result_set: &JObject, + value: serde_json::Value, +) -> Result { + let was_null = jni!( + env, + env.call_method(result_set, "wasNull", "()Z", &[]) + .and_then(|v| v.z()), + "Failed to check wasNull" + ); + Ok(if was_null { + serde_json::Value::Null + } else { + value + }) +} + +/// Extract column value based on JDBC type. +fn extract_column_value( + env: &mut JNIEnv, + result_set: &JObject, + column_index: i32, + sql_type: &i32, +) -> Result { + // Primitive getters (getInt/getBoolean/...) return 0/false for SQL NULL, + // so `null_or` consults ResultSet.wasNull() after the getter to tell an + // actual NULL from a zero value. Object getters (getString/getBytes) + // return a null reference for SQL NULL and are null-checked directly, so + // there is no separate getObject probe (one JNI call per column, not two). + match *sql_type { + Types::BIT | Types::BOOLEAN => { + let value = jni!( + env, + env.call_method( + result_set, + "getBoolean", + "(I)Z", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.z()), + "Failed to get boolean" + ); + null_or(env, result_set, serde_json::Value::Bool(value)) + } + Types::TINYINT | Types::SMALLINT | Types::INTEGER => { + let value = jni!( + env, + env.call_method(result_set, "getInt", "(I)I", &[JValue::Int(column_index)]) + .and_then(|v| v.i()), + "Failed to get int" + ); + null_or(env, result_set, serde_json::json!(value)) + } + // BIGINT is emitted as a string (like NUMERIC/DECIMAL below) so a + // value above 2^53 is not silently rounded by a JSON consumer that + // parses numbers as f64. + Types::BIGINT => { + let value = jni!( + env, + env.call_method(result_set, "getLong", "(I)J", &[JValue::Int(column_index)]) + .and_then(|v| v.j()), + "Failed to get long" + ); + null_or(env, result_set, serde_json::json!(value.to_string())) + } + Types::FLOAT | Types::REAL => { + let value = jni!( + env, + env.call_method(result_set, "getFloat", "(I)F", &[JValue::Int(column_index)]) + .and_then(|v| v.f()), + "Failed to get float" + ); + null_or(env, result_set, serde_json::json!(value)) + } + Types::DOUBLE => { + let value = jni!( + env, + env.call_method( + result_set, + "getDouble", + "(I)D", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.d()), + "Failed to get double" + ); + null_or(env, result_set, serde_json::json!(value)) + } + // NUMERIC/DECIMAL can carry more precision than an f64 can represent + // (e.g. money/large decimals), so emit them as strings to avoid + // silent precision loss. + Types::NUMERIC | Types::DECIMAL => get_column_as_string(env, result_set, column_index), + // Binary columns are base64-encoded so arbitrary bytes survive the + // round-trip through JSON. + Types::BINARY | Types::VARBINARY | Types::LONGVARBINARY => { + let bytes_obj = jni!( + env, + env.call_method( + result_set, + "getBytes", + "(I)[B", + &[JValue::Int(column_index)] + ) + .and_then(|v| v.l()), + "Failed to get bytes" + ); + if bytes_obj.is_null() { + return Ok(serde_json::Value::Null); + } + let buf = jni!( + env, + env.convert_byte_array(JByteArray::from(bytes_obj)), + "Failed to convert bytes" + ); + Ok(serde_json::Value::String( + base64::engine::general_purpose::STANDARD.encode(&buf), + )) + } + // Date/time types are read via their driver string form. Route + // through the null-safe getString path so a NULL date/time yields + // JSON null instead of failing the whole poll on get_string(null). + Types::TIMESTAMP | Types::DATE | Types::TIME => { + get_column_as_string(env, result_set, column_index) + } + // Default: getString for all other types (CHAR, VARCHAR, etc.) + _ => get_column_as_string(env, result_set, column_index), + } +} + +/// Read a column via `ResultSet.getString`, returning JSON `null` when the +/// value is SQL NULL. +fn get_column_as_string( + env: &mut JNIEnv, + result_set: &JObject, + column_index: i32, +) -> Result { + let value = jni!( + env, + env.call_method( + result_set, + "getString", + "(I)Ljava/lang/String;", + &[JValue::Int(column_index)], + ) + .and_then(|v| v.l()), + "Failed to get string" + ); + + if value.is_null() { + Ok(serde_json::Value::Null) + } else { + let str_value: String = jni!( + env, + env.get_string(&JString::from(value)), + "Failed to convert string" + ) + .into(); + Ok(serde_json::Value::String(str_value)) + } +} + +/// Extract the tracking-column value as a string offset, or `None` when it is +/// SQL NULL / empty / not comparable. In incremental mode a `None` here is +/// turned into a hard error by [`JdbcSource::tracking_offset_or_error`] (a NULL +/// tracking value cannot be watermarked), so the tracking column must be +/// NOT NULL; see the README. +fn extract_offset_value( + row_data: &serde_json::Map, + col_name: &str, +) -> Option { + match row_data.get(col_name) { + Some(serde_json::Value::Number(n)) => Some(n.to_string()), + Some(serde_json::Value::String(s)) if !s.is_empty() => Some(s.clone()), + _ => None, + } +} + +impl JdbcSource { + /// Validate configuration before touching the JVM or the database, so bad + /// config surfaces immediately at `open()` with an actionable message rather + /// than as an opaque wrapped JVM error or only after the first poll sleep. + fn validate_config(&self) -> Result<(), Error> { + // The driver JAR must exist; a missing path otherwise surfaces as an + // opaque ClassNotFound wrapped deep inside JVM startup. + if !std::path::Path::new(&self.config.driver_jar_path).exists() { + return Err(Error::InvalidConfigValue(format!( + "driver_jar_path '{}' does not exist; set it to the path of the JDBC driver JAR", + self.config.driver_jar_path + ))); + } + + // batch_size drives JDBC setMaxRows. Zero means "no limit" to JDBC, which + // would defeat both the row cap and the bulk truncation probe, so require + // at least 1. Cap below i32::MAX so the bulk `batch_size + 1` probe still + // fits in the i32 setMaxRows takes and stays distinguishable from a full + // batch. + const MAX_BATCH_SIZE: u32 = i32::MAX as u32 - 1; + if self.config.batch_size == 0 || self.config.batch_size > MAX_BATCH_SIZE { + return Err(Error::InvalidConfigValue(format!( + "batch_size must be between 1 and {MAX_BATCH_SIZE}, got {}", + self.config.batch_size + ))); + } + + // A set poll_interval must be valid and positive. Zero would make every + // successful poll immediately eligible to run again and hammer the DB. + if let Some(value) = self + .config + .poll_interval + .as_deref() + .map(str::trim) + .filter(|value| !value.is_empty()) + { + let duration = humantime::parse_duration(value).map_err(|_| { + Error::InvalidConfigValue(format!( + "poll_interval '{value}' is not a valid duration (e.g. \"30s\", \"5m\", \"1h\")" + )) + })?; + if duration.is_zero() { + return Err(Error::InvalidConfigValue( + "poll_interval must be greater than zero".to_string(), + )); + } + } + + if self.config.query_timeout_ms == 0 { + return Err(Error::InvalidConfigValue( + "query_timeout_ms must be greater than zero".to_string(), + )); + } + + // The query must be non-empty; an empty query only fails later at + // prepareStatement with an opaque driver error. + if self.config.query.trim().is_empty() { + return Err(Error::InvalidConfigValue( + "query must not be empty".to_string(), + )); + } + + // A set initial_offset must be non-blank: an empty value would build + // `WHERE tracking > ''` (a type error or always-false on many databases) + // rather than the intended cold-start scan. + if let Some(initial_offset) = self.config.initial_offset.as_deref() + && initial_offset.trim().is_empty() + { + return Err(Error::InvalidConfigValue( + "initial_offset must not be empty; omit it to start from the beginning".to_string(), + )); + } + + // Separate-credential auth requires both username and password; a + // half-set pair would silently fall through to URL-embedded credentials. + if self.config.username.is_some() != self.config.password.is_some() { + return Err(Error::InvalidConfigValue( + "username and password must both be set (for separate authentication) or both be \ + unset (to use credentials embedded in the JDBC URL)" + .to_string(), + )); + } + + // Incremental mode invariants. Without a tracking column the offset can + // never advance, so every poll re-reads the same rows. Without ordering + // by that column, setMaxRows returns an arbitrary subset, and advancing + // the offset to its max permanently skips the unread lower keys. + if self.config.mode == Mode::Incremental { + let Some(tracking_column) = self + .config + .tracking_column + .as_deref() + .filter(|column| !column.trim().is_empty()) + else { + return Err(Error::InvalidConfigValue( + "incremental mode requires a non-empty tracking_column so the offset can \ + advance; set tracking_column, or use mode = \"bulk\"" + .to_string(), + )); + }; + if !self.config.query.contains("{last_offset}") { + return Err(Error::InvalidConfigValue( + "incremental mode requires the query to contain {last_offset}; without it each poll would re-read the same first batch" + .to_string(), + )); + } + if !query_orders_by_tracking_column( + &self.config.query, + tracking_column, + self.config.snake_case_columns, + ) { + return Err(Error::InvalidConfigValue(format!( + "incremental mode requires the query to order by the tracking column so each \ + batch is a contiguous ascending range; add `ORDER BY {tracking_column}` (or \ + `ORDER BY {{tracking_column}}`) to the query" + ))); + } + } + + // Dry-run the query build so an unresolved placeholder or an invalid + // tracking_column fails now instead of after the first poll interval. + let state = lock_mutex(&self.state, "state")?; + self.build_query(&state)?; + Ok(()) + } +} + +#[async_trait] +impl Source for JdbcSource { + async fn open(&mut self) -> Result<(), Error> { + info!("Opening JDBC source connector [{}]", self.id); + info!( + "{CONNECTOR_NAME} connector [{}] configuration: JDBC URL={}, Driver={}, Mode={:?}", + self.id, + sanitize_jdbc_url(self.config.jdbc_url.expose_secret()), + self.config.driver_class, + self.config.mode + ); + + let result = (|| -> Result<(), Error> { + // Fail fast on bad config before starting the JVM or opening a connection. + self.validate_config()?; + + // JVM boot and DriverManager.getConnection are blocking JNI work. Run + // them via block_in_place (like poll()/close()) so they do not monopolize + // a shared async-runtime worker; the login timeout set in the connection + // path bounds how long a stuck connect can block. + tokio::task::block_in_place(|| -> Result<(), Error> { + self.initialize_jvm()?; + self.create_connection()?; + Ok(()) + }) + })(); + + if let Err(open_error) = result { + let open_error = sanitize_jdbc_error(open_error, self.config.jdbc_url.expose_secret()); + error!( + "Failed to open JDBC source connector [{}]: {open_error}", + self.id + ); + return Err(open_error); + } + + info!("JDBC source connector [{}] opened successfully", self.id); + Ok(()) + } + + async fn poll(&self) -> Result { + // Wait before every fetch. This keeps startup from hammering the + // database and guarantees an error followed by an immediate runtime + // retry still observes the configured poll interval. + tokio::time::sleep(self.poll_interval).await; + + // The JDBC/JNI fetch is synchronous, blocking work; run it via + // block_in_place so it does not monopolize a shared async-runtime worker + // while other connectors need to make progress. The connectors runtime is + // multi-threaded, which block_in_place requires. + let (messages, candidate_state) = tokio::task::block_in_place( + || -> Result<(Vec, Option), Error> { + let jvm = self + .jvm + .as_ref() + .ok_or_else(|| Error::InitError("JVM not initialized".to_string()))?; + let mut env = jvm + .attach_current_thread() + .map_err(|e| Error::InitError(format!("Failed to attach thread: {e}")))?; + // Defensive: clear any exception left pending by a prior failed poll + // on this thread before issuing JNI calls. + clear_pending_exception(&mut env); + // Bound this poll's local references (the connection local ref, the + // query string, and the statement/result-set handles) to a frame + // reclaimed when the poll returns. A tokio worker thread stays + // attached to the JVM across polls (attach_current_thread returns a + // no-detach nested guard once attached), so without this frame those + // per-poll locals accumulate on the thread's top-level frame and + // eventually overflow the JNI local reference table, aborting the JVM. + jni!( + env, + env.push_local_frame(16), + "Failed to push poll local frame" + ); + let result = self.execute_query(&mut env); + finish_local_frame( + &mut env, + result, + "Failed to pop poll local frame", + Error::Connection, + ) + }, + )?; + + // Stage the advanced cursor rather than committing it. The runtime saves + // the returned state only after the batch is sent, then reports the + // outcome to `on_batch_result`, which commits on Ack and discards on Nack + // so a failed send is re-polled instead of silently skipped. + let connector_state = match candidate_state { + Some(candidate) => { + let serialized = ConnectorState::serialize(&candidate, CONNECTOR_NAME, self.id) + .ok_or_else(|| { + Error::Serialization("failed to serialize JDBC source state".to_string()) + })?; + *lock_mutex(&self.pending_state, "pending_state")? = Some(candidate); + Some(serialized) + } + None => None, + }; + + Ok(ProducedMessages { + schema: Schema::Json, + messages, + state: connector_state, + }) + } + + /// Resolve the cursor staged by the last `poll`. + /// + /// `Ack` means the runtime both sent the batch and durably persisted its + /// checkpoint, so the staged cursor becomes the committed one. `Nack` means + /// neither happened, so the staged cursor is dropped and the next poll rebuilds + /// the same query from the committed offset. That re-reads the batch (delivery + /// stays at-least-once) instead of skipping it, which is what advancing the + /// cursor at fetch time would have done. + async fn on_batch_result(&self, result: SourceBatchResult) -> Result<(), Error> { + let candidate = lock_mutex(&self.pending_state, "pending_state")?.take(); + let Some(candidate) = candidate else { + return Ok(()); + }; + match result { + SourceBatchResult::Ack => { + *lock_mutex(&self.state, "state")? = candidate; + } + SourceBatchResult::Nack => { + warn!( + "JDBC source [{}] batch was not acknowledged; discarding the staged offset \ + {:?} and re-polling the same range", + self.id, candidate.last_offset + ); + } + } + Ok(()) + } + + async fn close(&mut self) -> Result<(), Error> { + info!("Closing JDBC source connector [{}]", self.id); + + if let Some(jvm) = self.jvm.as_ref() { + // Closing the JDBC connection is blocking JNI work; run it off the + // async worker like the poll path. + tokio::task::block_in_place(|| -> Result<(), Error> { + let Ok(mut env) = jvm.attach_current_thread() else { + return Ok(()); + }; + let mut guard = lock_mutex(&self.connection, "connection")?; + if let Some(connection) = guard.as_ref() { + best_effort_close(&mut env, connection.as_obj()); + info!( + "{CONNECTOR_NAME} connector [{}] database connection closed", + self.id + ); + } + *guard = None; + Ok(()) + })?; + } + + let state = lock_mutex(&self.state, "state")?; + info!( + "JDBC source connector [{}] closed. Total rows processed: {}", + self.id, state.processed_rows + ); + + Ok(()) + } +} + +/// Convert string to snake_case +fn to_snake_case(s: &str) -> String { + let mut result = String::new(); + let mut prev_is_upper = false; + + for (i, ch) in s.chars().enumerate() { + if ch.is_uppercase() { + if i > 0 && !prev_is_upper { + result.push('_'); + } + result.extend(ch.to_lowercase()); + prev_is_upper = true; + } else { + result.push(ch); + prev_is_upper = false; + } + } + + result +} + +struct SharedJvm { + vm: Arc, + configuration: JvmConfiguration, +} + +#[derive(Debug, Clone, PartialEq, Eq)] +struct JvmConfiguration { + driver_jar_path: PathBuf, + options: Vec, +} + +impl JvmConfiguration { + fn new(driver_jar_path: &str, options: &[String]) -> Result { + let driver_jar_path = Path::new(driver_jar_path).canonicalize().map_err(|error| { + Error::InitError(format!( + "Failed to canonicalize driver_jar_path '{driver_jar_path}': {error}" + )) + })?; + Ok(Self { + driver_jar_path, + options: options.to_vec(), + }) + } + + fn ensure_compatible_with(&self, requested: &Self) -> Result<(), Error> { + if self == requested { + return Ok(()); + } + Err(Error::InvalidConfigValue( + "all JDBC source instances in one runtime process must use the same driver_jar_path and jvm_options because they share one JVM" + .to_string(), + )) + } +} + +/// Process-wide JVM. JNI allows only one `JavaVM` per OS process, so every JDBC +/// connector instance in this dynamic library shares this one. +static GLOBAL_JVM: Mutex> = Mutex::new(None); + +/// Serializes the process-global `DriverManager.setLoginTimeout` setting with +/// the connection attempt it governs. +static DRIVER_MANAGER_CONNECT_LOCK: Mutex<()> = Mutex::new(()); + +/// Return the process JVM, creating it on first use within this dynamic +/// library. Later callers reuse it only when their classpath and options match; +/// otherwise startup fails instead of silently running with the wrong driver. +/// +/// Limitation: a JDBC *source* and a JDBC *sink* are separate dynamic libraries +/// and do not share this static, so configuring both in the *same* connectors +/// runtime process is not supported (the second to start cannot create a second +/// JVM). Run them in separate runtime processes. +fn get_or_create_jvm( + driver_jar_path: &str, + jvm_options: &[String], + connector_id: u32, +) -> Result, Error> { + let requested = JvmConfiguration::new(driver_jar_path, jvm_options)?; + let mut guard = lock_mutex(&GLOBAL_JVM, "jvm")?; + if let Some(shared) = guard.as_ref() { + shared.configuration.ensure_compatible_with(&requested)?; + info!("{CONNECTOR_NAME} connector [{connector_id}] reusing existing process JVM"); + return Ok(shared.vm.clone()); + } + + let canonical_jar_path = requested.driver_jar_path.to_string_lossy(); + let classpath_option = format!("-Djava.class.path={canonical_jar_path}"); + let mut args_builder = jni::InitArgsBuilder::new() + .version(jni::JNIVersion::V8) + .option(&classpath_option); + for option in jvm_options { + args_builder = args_builder.option(option); + } + let jvm_args = args_builder + .build() + .map_err(|e| Error::InitError(format!("Failed to build JVM arguments: {e:?}")))?; + let jvm = JavaVM::new(jvm_args) + .map_err(|e| Error::InitError(format!("Failed to create JVM: {e:?}")))?; + + info!( + "{CONNECTOR_NAME} connector [{connector_id}] initialized JVM successfully (classpath: {canonical_jar_path})" + ); + let arc = Arc::new(jvm); + *guard = Some(SharedJvm { + vm: arc.clone(), + configuration: requested, + }); + Ok(arc) +} + +/// Bind each incremental offset placeholder using the SQL type the driver +/// reports for that parameter. Passing the explicit target type lets JDBC +/// convert the persisted string representation back to an integer, timestamp, +/// decimal, or text value without embedding dialect-specific literals in SQL. +fn bind_offset_parameters( + env: &mut JNIEnv, + statement: &JObject, + query: &PreparedQuery, +) -> Result<(), Error> { + if query.offset_parameter_count == 0 { + return Ok(()); + } + let offset = query.offset.as_deref().ok_or(Error::InvalidState)?; + let expected_count = i32::try_from(query.offset_parameter_count).map_err(|_| { + Error::InvalidConfigValue("query contains too many {last_offset} placeholders".to_string()) + })?; + + let metadata = jni!( + env, + env.call_method( + statement, + "getParameterMetaData", + "()Ljava/sql/ParameterMetaData;", + &[], + ) + .and_then(|value| value.l()), + "Failed to get JDBC parameter metadata" + ); + let actual_count = jni!( + env, + env.call_method(&metadata, "getParameterCount", "()I", &[]) + .and_then(|value| value.i()), + "Failed to get JDBC parameter count" + ); + if actual_count != expected_count { + return Err(Error::InvalidConfigValue(format!( + "query contains {actual_count} JDBC parameters but {expected_count} came from \ + {{last_offset}}; raw '?' parameters are not supported" + ))); + } + + let offset_string = jni!( + env, + env.new_string(offset), + "Failed to create offset parameter string" + ); + let offset_object = JObject::from(offset_string); + for parameter_index in 1..=expected_count { + let sql_type = jni!( + env, + env.call_method( + &metadata, + "getParameterType", + "(I)I", + &[JValue::Int(parameter_index)], + ) + .and_then(|value| value.i()), + "Failed to get JDBC parameter type" + ); + if env + .call_method( + statement, + "setObject", + "(ILjava/lang/Object;I)V", + &[ + JValue::Int(parameter_index), + JValue::Object(&offset_object), + JValue::Int(sql_type), + ], + ) + .is_err() + { + return Err(classify_query_failure(env, "bind offset parameter")); + } + } + Ok(()) +} + +/// Reject a query that still contains an unresolved placeholder, so an invalid +/// statement is never sent to the driver. This guards misconfigurations such as +/// a bulk-mode query using `{last_offset}`, or an incremental query whose +/// predicate could not be auto-removed on the first (no-offset) poll. +fn finalize_query(query: String) -> Result { + if query.contains("{tracking_column}") || query.contains("{last_offset}") { + return Err(Error::InvalidConfigValue( + "query still contains an unresolved {tracking_column}/{last_offset} placeholder; \ + placeholders are only resolved in incremental mode, and require either a persisted \ + offset, an initial_offset, or the exact 'WHERE {tracking_column} > {last_offset}' form" + .to_string(), + )); + } + Ok(query) +} + +/// Remove the incremental offset predicate `{tracking_column} > {last_offset}` +/// from a query for the cold-start (no-offset) poll, preserving any other `WHERE` +/// conditions. Handles the offset term followed by `AND ...`, preceded by +/// `... AND`, or standing alone, so a companion condition such as +/// `AND {tracking_column} IS NOT NULL` survives as a valid `WHERE`. +fn strip_offset_predicate(query: &str) -> String { + let query = RE_OFFSET_PREDICATE_AND_AFTER.replace_all(query, "WHERE "); + let query = RE_OFFSET_PREDICATE_AND_BEFORE.replace_all(&query, ""); + RE_OFFSET_PREDICATE_BARE + .replace_all(&query, "") + .into_owned() +} + +/// Check that an incremental query's result set is ordered ascending by the +/// tracking column, so `setMaxRows` returns a contiguous ascending prefix rather +/// than an arbitrary subset (which would let the advancing offset skip unread +/// lower keys). The tracking column must be the FIRST ordering term of the outer +/// `ORDER BY`, and must not be descending. +/// +/// This is a lexical check, not a SQL parser, so it errs strict: it inspects the +/// last `ORDER BY` at parenthesis depth zero (the outer query's clause, never one +/// inside a subquery or a window function's `OVER (...)`), takes the first +/// ordering term, and requires it to be the `{tracking_column}` placeholder or an +/// identifier whose final path segment equals the tracking column. Matching +/// mirrors the read-time `tracking_column_matches`: case-insensitive against the +/// raw identifier and, when `snake_case_columns` is set, against its snake_cased +/// form, so a CamelCase `ORDER BY OrderDate` with `tracking_column = "order_date"` +/// validates exactly as it matches when reading rows (otherwise validation would +/// reject a config that runs correctly). A trailing `DESC` is rejected, and a +/// composite `ORDER BY other, tracking` (tracking not primary) is rejected, +/// because truncation then yields a prefix ordered by `other`. +fn query_orders_by_tracking_column( + query: &str, + tracking_column: &str, + snake_case_columns: bool, +) -> bool { + let Some(order_by) = outer_order_by(query) else { + return false; + }; + let first_term = order_by + .split(',') + .next() + .unwrap_or("") + .trim() + .trim_end_matches(';') + .trim(); + let mut tokens = first_term.split_whitespace(); + let key = tokens.next().unwrap_or(""); + let modifiers = tokens + .map(|token| token.to_ascii_lowercase()) + .collect::>(); + if !modifiers.is_empty() + && !matches!(modifiers.as_slice(), [direction] if direction == "asc") + && !matches!( + modifiers.as_slice(), + [nulls, position] + if nulls == "nulls" && matches!(position.as_str(), "first" | "last") + ) + && !matches!( + modifiers.as_slice(), + [direction, nulls, position] + if direction == "asc" + && nulls == "nulls" + && matches!(position.as_str(), "first" | "last") + ) + { + return false; + } + if key == "{tracking_column}" { + return true; + } + // Compare the final path segment (`t.updated_at` -> `updated_at`) after + // validating every segment as a complete plain or quoted identifier. A + // case-preserving column such as PostgreSQL's `ORDER BY "OrderDate"` must + // validate against the same `tracking_column` that matches its unquoted + // driver label when reading rows. + let Some(key_ident) = exact_identifier_final_segment(key) else { + return false; + }; + // Normalize identically to the read-time path so validate-time cannot reject + // a config whose ORDER BY column would match a returned row at runtime. + let normalized = if snake_case_columns { + to_snake_case(key_ident) + } else { + key_ident.to_string() + }; + tracking_column_matches(tracking_column, key_ident, &normalized) +} + +/// Return the final segment of a plain or qualified SQL identifier. Expressions, +/// function calls, casts, and trailing operators are rejected so validation +/// cannot mistake `ORDER BY id % 10` for ordering by the cursor itself. +fn exact_identifier_final_segment(identifier: &str) -> Option<&str> { + let mut final_segment = None; + for segment in identifier.split('.') { + let unquoted = if segment.starts_with('"') && segment.ends_with('"') + || segment.starts_with('`') && segment.ends_with('`') + || segment.starts_with('[') && segment.ends_with(']') + { + &segment[1..segment.len().checked_sub(1)?] + } else { + segment + }; + if unquoted.is_empty() + || !unquoted.chars().enumerate().all(|(index, character)| { + character == '_' + || character == '$' + || character.is_ascii_alphanumeric() + && (index > 0 || !character.is_ascii_digit()) + }) + { + return None; + } + final_segment = Some(unquoted); + } + final_segment +} + +/// Return the text following the outer `ORDER BY` (the last `order by` token at +/// parenthesis depth zero), or `None` if the query has no top-level ordering. An +/// `ORDER BY` inside a subquery or a window function's `OVER (...)` sits at depth +/// greater than zero and is ignored, so a window's internal ordering (which +/// orders values within the frame, not the emitted ResultSet) is never mistaken +/// for the row-emission order. Single-quoted string literals are skipped so a +/// parenthesis or the words `order by` inside a literal cannot perturb the depth +/// or produce a spurious match. +fn outer_order_by(query: &str) -> Option<&str> { + const NEEDLE: &[u8] = b"order by"; + let bytes = query.as_bytes(); + // ASCII lowercasing is byte-length-preserving, so indices into `lower` map + // one-to-one onto `query`; this keeps the matched key in its original case. + let lower = query.to_ascii_lowercase(); + let lower_bytes = lower.as_bytes(); + let mut depth: u32 = 0; + let mut in_quote = false; + let mut last_end = None; + let mut i = 0; + while i < bytes.len() { + let byte = bytes[i]; + if in_quote { + if byte == b'\'' { + in_quote = false; + } + i += 1; + continue; + } + // Skip SQL comments entirely: an `ORDER BY` (or a stray paren/quote) + // inside a comment must not be mistaken for the outer ordering, which + // would let a query with no real ordering pass validation and then skip + // rows at runtime. + if byte == b'-' && bytes.get(i + 1) == Some(&b'-') { + i += 2; + while i < bytes.len() && bytes[i] != b'\n' { + i += 1; + } + continue; + } + if byte == b'/' && bytes.get(i + 1) == Some(&b'*') { + i += 2; + while i < bytes.len() && !(bytes[i] == b'*' && bytes.get(i + 1) == Some(&b'/')) { + i += 1; + } + i += 2; // consume the closing `*/` + continue; + } + match byte { + b'\'' => in_quote = true, + b'(' => depth += 1, + b')' => depth = depth.saturating_sub(1), + _ if depth == 0 + && lower_bytes[i..].starts_with(NEEDLE) + && (i == 0 || !is_identifier_byte(bytes[i - 1])) => + { + let end = i + NEEDLE.len(); + if bytes.get(end).is_none_or(|next| !is_identifier_byte(*next)) { + last_end = Some(end); + } + } + _ => {} + } + i += 1; + } + last_end.map(|end| query[end..].trim_start()) +} + +/// Whether a byte is part of an unquoted SQL identifier (ASCII alphanumeric or +/// `_`), used to require whole-token boundaries around a matched keyword. +fn is_identifier_byte(byte: u8) -> bool { + byte.is_ascii_alphanumeric() || byte == b'_' +} + +/// Whether a configured `tracking_column` name refers to this result column. +/// Compares case-insensitively against both the raw driver label and the +/// normalized (snake_cased) output key: drivers fold identifier case (e.g. +/// PostgreSQL lowercases unquoted names) and snake_case output would otherwise +/// never match the configured name, silently stalling the offset. +fn tracking_column_matches(tracking_column: &str, raw_name: &str, normalized_name: &str) -> bool { + tracking_column.eq_ignore_ascii_case(raw_name) + || tracking_column.eq_ignore_ascii_case(normalized_name) +} + +/// Validate that a string is a safe SQL identifier for interpolation: ASCII +/// letters, digits, underscore, and dot (for `table.column`), starting with a +/// letter or underscore. Prevents injection through the `{tracking_column}` +/// placeholder. +fn is_valid_identifier(name: &str) -> bool { + let mut chars = name.chars(); + match chars.next() { + Some(c) if c.is_ascii_alphabetic() || c == '_' => {} + _ => return false, + } + name.chars() + .all(|c| c.is_ascii_alphanumeric() || c == '_' || c == '.') +} + +/// Classify a JDBC `SQLState` class (first 2 chars) as transient vs permanent. +/// `08` connection, `40` rollback/serialization, `53` resources, `57` operator +/// intervention, `58` system error are transient; everything else (and an +/// unknown/absent state) is permanent. +fn is_transient_sql_state(sql_state: Option<&str>) -> bool { + // Use `get` rather than slicing: the SQLState comes from the driver and is + // not guaranteed ASCII, so `&s[..2]` could panic on a multi-byte boundary. + match sql_state.and_then(|s| s.get(..2)) { + Some(class) => matches!(class, "08" | "40" | "53" | "57" | "58"), + None => false, + } +} + +/// Inspect and CLEAR the pending Java exception after a failed query JNI call, +/// returning a classified `Error`: transient SQL states (connection/resource +/// classes) map to `Error::Connection`, permanent ones (syntax/constraint) to +/// `Error::InvalidRecordValue`. Clearing is required so the next JNI call on this +/// thread is not aborted. +/// +/// NOTE: the runtime does not yet branch on this distinction (there is no +/// per-variant backoff in the poll loop today), so the classification is +/// currently informational: it shapes the error variant and log message and is +/// kept ready for when SDK-side backoff lands. Do not claim differentiated +/// runtime backoff until that exists. +fn classify_query_failure(env: &mut JNIEnv, action: &str) -> Error { + let (sql_state, message) = take_pending_sql_exception(env); + let transient = is_transient_sql_state(sql_state.as_deref()); + let state = sql_state.as_deref().unwrap_or("?"); + let msg = format!("Failed to {action} (SQLState {state}): {message}"); + if transient { + Error::Connection(msg) + } else { + Error::InvalidRecordValue(msg) + } +} + +/// Take the pending Java exception (clearing it) and return its `SQLState` (if a +/// `java.sql.SQLException`) and message. +fn take_pending_sql_exception(env: &mut JNIEnv) -> (Option, String) { + let throwable = match env.exception_occurred() { + Ok(throwable) if !throwable.is_null() => throwable, + Ok(_) => return (None, "unknown error".to_string()), + Err(_) => { + clear_pending_exception(env); + return (None, "unknown error".to_string()); + } + }; + clear_pending_exception(env); + + let message = throwable_string_method(env, &throwable, "getMessage") + .unwrap_or_else(|| "unknown error".to_string()); + let is_sql_exception = match env.is_instance_of(&throwable, "java/sql/SQLException") { + Ok(is_sql_exception) => is_sql_exception, + Err(_) => { + clear_pending_exception(env); + false + } + }; + let sql_state = if is_sql_exception { + throwable_string_method(env, &throwable, "getSQLState") + } else { + None + }; + (sql_state, message) +} + +/// Call a no-arg `String`-returning method on a throwable; None on JNI error/null. +/// Any pending exception raised by the call itself is cleared before returning, +/// so a later JNI call on this thread is not aborted for running with an +/// exception pending (JNI forbids that). +fn throwable_string_method( + env: &mut JNIEnv, + throwable: &JThrowable, + method: &str, +) -> Option { + let obj = match env + .call_method(throwable, method, "()Ljava/lang/String;", &[]) + .and_then(|v| v.l()) + { + Ok(obj) => obj, + Err(_) => { + clear_pending_exception(env); + return None; + } + }; + if obj.is_null() { + return None; + } + match env.get_string(&JString::from(obj)) { + Ok(s) => Some(s.into()), + Err(_) => { + clear_pending_exception(env); + None + } + } +} + +/// JDBC SQL Types constants +mod java { + pub mod sql { + pub struct Types; + + impl Types { + pub const BIT: i32 = -7; + pub const TINYINT: i32 = -6; + pub const SMALLINT: i32 = 5; + pub const INTEGER: i32 = 4; + pub const BIGINT: i32 = -5; + pub const FLOAT: i32 = 6; + pub const REAL: i32 = 7; + pub const DOUBLE: i32 = 8; + pub const NUMERIC: i32 = 2; + pub const DECIMAL: i32 = 3; + pub const DATE: i32 = 91; + pub const TIME: i32 = 92; + pub const TIMESTAMP: i32 = 93; + pub const BINARY: i32 = -2; + pub const VARBINARY: i32 = -3; + pub const LONGVARBINARY: i32 = -4; + pub const BOOLEAN: i32 = 16; + } + } +} + +// Export the connector via SDK macro +source_connector!(JdbcSource); + +#[cfg(test)] +mod tests { + use super::*; + + /// A minimal, valid bulk-mode config for tests that need a `JdbcSourceConfig`. + fn base_config() -> JdbcSourceConfig { + JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:h2:mem:test"), + driver_class: "org.h2.Driver".to_string(), + driver_jar_path: "/tmp/h2.jar".to_string(), + username: None, + password: None, + query: "SELECT 1".to_string(), + poll_interval: Some("10s".to_string()), + batch_size: 100, + tracking_column: None, + initial_offset: None, + mode: Mode::Bulk, + snake_case_columns: false, + include_metadata: true, + verbose_logging: None, + jvm_options: vec![], + connection_timeout_ms: 30000, + login_timeout_ms: 30000, + query_timeout_ms: 30000, + } + } + + /// Write a throwaway file to stand in for a driver JAR and return its path, + /// so `validate_config`'s existence check passes. + fn write_temp_jar(name: &str) -> String { + let path = std::env::temp_dir().join(name); + std::fs::write(&path, b"jar").expect("write temp jar"); + path.to_string_lossy().into_owned() + } + + #[test] + fn given_poll_interval_should_parse_or_default() { + assert_eq!(parse_poll_interval(Some("30s")), Duration::from_secs(30)); + assert_eq!(parse_poll_interval(Some("5m")), Duration::from_secs(300)); + // Unset, empty, and unparsable all fall back to the default. + assert_eq!(parse_poll_interval(None), DEFAULT_POLL_INTERVAL); + assert_eq!(parse_poll_interval(Some(" ")), DEFAULT_POLL_INTERVAL); + assert_eq!( + parse_poll_interval(Some("not-a-duration")), + DEFAULT_POLL_INTERVAL + ); + } + + #[test] + fn given_invalid_poll_interval_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_poll_interval.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.poll_interval = Some("banana".to_string()); + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must reject bad poll_interval"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("poll_interval"))); + } + + #[test] + fn given_zero_poll_interval_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_zero_poll_interval.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.poll_interval = Some("0s".to_string()); + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must reject a zero poll interval"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("greater than zero"))); + } + + #[test] + fn given_zero_query_timeout_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_query_timeout.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.query_timeout_ms = 0; + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(message)) if message.contains("query_timeout_ms")) + ); + } + + #[test] + fn given_cold_start_should_remove_offset_predicate() { + for query in [ + "select * from t where {tracking_column} > {last_offset} order by id", + "SELECT * FROM t WHERE {tracking_column} > {last_offset} ORDER BY id", + "SELECT * FROM t where {tracking_column}>{last_offset} ORDER BY id", + ] { + let mut config = base_config(); + config.query = query.to_string(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + let source = JdbcSource::new(1, config, None); + let state = State::default(); + let built = source.build_query(&state).expect("build query"); + assert!( + !built.sql.contains("{last_offset}") && !built.sql.contains("{tracking_column}"), + "predicate not removed for query variant: {query} -> {}", + built.sql + ); + assert!(built.sql.to_lowercase().contains("order by id")); + } + } + + #[test] + fn given_compound_predicate_should_preserve_other_conditions() { + // Offset term followed by a companion condition (the README-advised + // IS NOT NULL): the AND and the companion condition survive. + assert_eq!( + strip_offset_predicate( + "SELECT * FROM t WHERE {tracking_column} > {last_offset} AND {tracking_column} IS NOT NULL ORDER BY {tracking_column}" + ), + "SELECT * FROM t WHERE {tracking_column} IS NOT NULL ORDER BY {tracking_column}" + ); + // Offset term preceded by a condition. + assert_eq!( + strip_offset_predicate( + "SELECT * FROM t WHERE active = 1 AND {tracking_column} > {last_offset} ORDER BY id" + ), + "SELECT * FROM t WHERE active = 1 ORDER BY id" + ); + // Bare offset term (no other condition) drops the whole WHERE. + assert!( + !strip_offset_predicate( + "SELECT * FROM t WHERE {tracking_column} > {last_offset} ORDER BY id" + ) + .contains("{last_offset}") + ); + } + + #[test] + fn given_cold_start_compound_predicate_should_build_valid_query() { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = "SELECT * FROM t WHERE {tracking_column} > {last_offset} AND {tracking_column} IS NOT NULL ORDER BY {tracking_column}".to_string(); + let source = JdbcSource::new(1, config, None); + // Cold start: no persisted offset, no initial_offset. + let built = source.build_query(&State::default()).expect("build query"); + assert_eq!( + built.sql, + "SELECT * FROM t WHERE id IS NOT NULL ORDER BY id" + ); + assert!(!built.sql.contains("{last_offset}") && !built.sql.contains("{tracking_column}")); + // No dangling AND / empty WHERE. + assert!(!built.sql.to_uppercase().contains("WHERE AND")); + assert!(!built.sql.to_uppercase().contains("AND ORDER")); + } + + #[test] + fn given_half_set_credentials_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_half_creds.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.username = Some("user".to_string()); + config.password = None; + let source = JdbcSource::new(1, config, None); + assert!(matches!( + source.validate_config(), + Err(Error::InvalidConfigValue(_)) + )); + } + + #[test] + fn given_missing_driver_jar_should_be_rejected() { + let mut config = base_config(); + config.driver_jar_path = "/nonexistent/path/to/driver.jar".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source.validate_config().expect_err("missing jar must fail"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("driver_jar_path"))); + } + + #[test] + fn given_invalid_built_query_should_fail_validation() { + let jar = write_temp_jar("jdbc_validate_dry_run.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + // Passes the incremental invariants (tracking_column set, ordered by it) + // but the non-canonical predicate leaves {last_offset} unresolved with no + // offset, so the dry-run build_query must reject it now. + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = "SELECT * FROM t WHERE x >= {last_offset} ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + assert!(matches!( + source.validate_config(), + Err(Error::InvalidConfigValue(_)) + )); + } + + #[test] + fn given_valid_bulk_config_should_pass_validation() { + let jar = write_temp_jar("jdbc_validate_ok.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + let source = JdbcSource::new(1, config, None); + assert!(source.validate_config().is_ok()); + } + + #[test] + fn given_valid_incremental_config_should_pass_validation() { + let jar = write_temp_jar("jdbc_validate_ok_incremental.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + // Canonical placeholder predicate so the cold-start (no offset) build + // auto-removes the WHERE; ordered by the tracking column. + config.query = + "SELECT id, name FROM t WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(); + let source = JdbcSource::new(1, config, None); + assert!(source.validate_config().is_ok()); + } + + #[test] + fn given_incremental_query_without_offset_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_last_offset.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = "SELECT id FROM t ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must require {last_offset}"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("{last_offset}"))); + } + + #[test] + fn given_duplicate_tracking_labels_should_be_rejected() { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + let source = JdbcSource::new(1, config, None); + let err = source + .prepare_column_metadata(vec![("id".to_string(), 4), ("ID".to_string(), 4)]) + .expect_err("duplicate tracking labels must be rejected"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("matches 2 columns"))); + } + + #[test] + fn given_duplicate_bulk_labels_should_be_rejected() { + let source = JdbcSource::new(1, base_config(), None); + let error = source + .prepare_column_metadata(vec![("id".to_string(), 4), ("id".to_string(), 4)]) + .expect_err("duplicate output keys must be rejected in every mode"); + assert!( + matches!(error, Error::InvalidConfigValue(message) if message.contains("duplicate output key")) + ); + } + + #[test] + fn given_missing_tracking_result_column_should_be_rejected() { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + let source = JdbcSource::new(1, config, None); + let err = source + .prepare_column_metadata(vec![("name".to_string(), 12)]) + .expect_err("missing tracking label must be rejected"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("not present"))); + } + + #[test] + fn given_incremental_config_without_tracking_column_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_no_tracking.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = None; + config.query = "SELECT id FROM t ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must require tracking_column"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("tracking_column"))); + } + + #[test] + fn given_unordered_incremental_query_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_no_order.jar"); + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + // No ORDER BY: an unordered incremental query can skip rows on truncation. + config.query = "SELECT id FROM t WHERE id > {last_offset}".to_string(); + let source = JdbcSource::new(1, config, None); + let err = source.validate_config().expect_err("must require ORDER BY"); + assert!( + matches!(err, Error::InvalidConfigValue(msg) if msg.to_lowercase().contains("order by")) + ); + } + + #[test] + fn given_valid_tracking_order_should_be_accepted() { + assert!(query_orders_by_tracking_column( + "SELECT * FROM t WHERE id > {last_offset} ORDER BY id", + "id", + false + )); + // Placeholder form, case/whitespace variance, explicit ASC. + assert!(query_orders_by_tracking_column( + "select * from t order by {tracking_column}", + "updated_at", + false + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY Updated_At ASC", + "updated_at", + false + )); + // Table-qualified column, and the outer ORDER BY after a subquery. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY t.updated_at", + "updated_at", + false + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM (SELECT * FROM t ORDER BY x) s ORDER BY id", + "id", + false + )); + } + + #[test] + fn given_invalid_tracking_order_should_be_rejected() { + // No ORDER BY. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t WHERE id > 0", + "id", + false + )); + // Different column. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY name", + "id", + false + )); + // Descending breaks ascending offset advancement. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY updated_at DESC", + "updated_at", + false + )); + // Substring-only match must not pass (id is a substring of valid_flag/id_backup). + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY valid_flag", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY id_backup", + "id", + false + )); + // Tracking column not the primary (first) ordering term. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY name, id", + "id", + false + )); + // An expression beginning with the tracking column does not produce a + // contiguous cursor-ordered prefix. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY id % 10, id", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY lower(id)", + "id", + false + )); + } + + #[test] + fn given_window_order_by_should_not_satisfy_result_ordering() { + // A window function's internal ORDER BY orders values within the frame, + // not the emitted ResultSet, so it must not satisfy the outer-ordering + // requirement even though it is the only `order by` in the text. + assert!(!query_orders_by_tracking_column( + "SELECT id, ROW_NUMBER() OVER (ORDER BY id) rn FROM t", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT id, ROW_NUMBER() OVER (ORDER BY id) rn FROM t WHERE id > {last_offset}", + "id", + false + )); + // A window ORDER BY plus a genuine outer ORDER BY is accepted on the outer. + assert!(query_orders_by_tracking_column( + "SELECT id, ROW_NUMBER() OVER (ORDER BY x) rn FROM t ORDER BY id", + "id", + false + )); + // A parenthesis inside a string literal must not shift the depth and hide + // the real outer ORDER BY. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t WHERE note = 'a (b' ORDER BY id", + "id", + false + )); + // An ORDER BY inside a SQL comment must NOT satisfy the requirement: the + // real query has no ordering, so accepting it would skip rows at runtime. + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t -- ORDER BY id\n", + "id", + false + )); + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t /* ORDER BY id */", + "id", + false + )); + // A comment before the genuine outer ORDER BY must not hide it. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t /* pick a key */ ORDER BY id", + "id", + false + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM t -- note\n ORDER BY id", + "id", + false + )); + } + + #[test] + fn given_snake_case_tracking_order_should_match_read_time() { + // snake_case_columns = true: a CamelCase ORDER BY column with a + // snake_cased tracking_column validates, mirroring tracking_column_matches + // so validate-time never rejects a config that reads rows correctly. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY OrderDate", + "order_date", + true + )); + // Without normalization enabled the same pair does not match (the driver + // label would not be snake_cased at read time either). + assert!(!query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY OrderDate", + "order_date", + false + )); + // Raw label match still works regardless of the normalization flag. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY OrderDate", + "orderdate", + true + )); + } + + #[test] + fn given_quoted_tracking_identifier_should_match() { + // PostgreSQL preserves case only for quoted identifiers, so a genuinely + // CamelCase column is ordered as `"OrderDate"`; the surrounding quotes + // must not defeat the match against its unquoted driver label. + assert!(query_orders_by_tracking_column( + r#"SELECT * FROM t ORDER BY "OrderDate""#, + "order_date", + true + )); + assert!(query_orders_by_tracking_column( + r#"SELECT * FROM t ORDER BY "updated_at""#, + "updated_at", + false + )); + // MySQL backtick and SQL Server bracket quoting. + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY `updated_at`", + "updated_at", + false + )); + assert!(query_orders_by_tracking_column( + "SELECT * FROM t ORDER BY [updated_at]", + "updated_at", + false + )); + } + + #[test] + fn given_empty_query_or_offset_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_blanks.jar"); + // Empty query. + let mut config = base_config(); + config.driver_jar_path = jar.clone(); + config.query = " ".to_string(); + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(msg)) if msg.contains("query")) + ); + + // Blank initial_offset. + let mut config = base_config(); + config.driver_jar_path = jar.clone(); + config.initial_offset = Some(" ".to_string()); + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(msg)) if msg.contains("initial_offset")) + ); + + // Blank tracking_column in incremental mode. + let mut config = base_config(); + config.driver_jar_path = jar; + config.mode = Mode::Incremental; + config.tracking_column = Some("".to_string()); + config.query = "SELECT id FROM t ORDER BY id".to_string(); + let source = JdbcSource::new(1, config, None); + assert!( + matches!(source.validate_config(), Err(Error::InvalidConfigValue(msg)) if msg.contains("tracking_column")) + ); + } + + #[test] + fn given_invalid_batch_size_should_be_rejected() { + let jar = write_temp_jar("jdbc_validate_batch_size.jar"); + for bad in [0u32, i32::MAX as u32] { + let mut config = base_config(); + config.driver_jar_path = jar.clone(); + config.batch_size = bad; + let source = JdbcSource::new(1, config, None); + let err = source + .validate_config() + .expect_err("must reject invalid batch_size"); + assert!(matches!(err, Error::InvalidConfigValue(msg) if msg.contains("batch_size"))); + } + } + + #[test] + fn given_tracking_column_should_match_case_and_normalization() { + // Case-insensitive against the raw driver label (driver case-folding). + assert!(tracking_column_matches( + "OrderDate", + "orderdate", + "orderdate" + )); + // Matches the normalized (snake_cased) output key. + assert!(tracking_column_matches( + "order_date", + "OrderDate", + "order_date" + )); + // Plain lowercase match. + assert!(tracking_column_matches("id", "id", "id")); + // Genuinely different column does not match. + assert!(!tracking_column_matches("id", "name", "name")); + } + + #[test] + fn given_null_incremental_cursor_should_be_rejected() { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + let source = JdbcSource::new(1, config, None); + // Non-null resolves; NULL/empty (None) is a hard error in incremental mode. + assert_eq!( + source + .tracking_offset_or_error(Some("42".to_string()), "id") + .unwrap(), + Some("42".to_string()) + ); + assert!(matches!( + source.tracking_offset_or_error(None, "id"), + Err(Error::InvalidRecordValue(_)) + )); + } + + #[test] + fn given_null_bulk_cursor_should_be_allowed() { + let source = JdbcSource::new(1, base_config(), None); // base_config is bulk + assert_eq!(source.tracking_offset_or_error(None, "id").unwrap(), None); + } + + #[test] + fn given_mysql_url_should_redact_password() { + let url = "jdbc:mysql://root:SuperSecret123@localhost:3306/mydb"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:mysql://root:***@localhost:3306/mydb"); + assert!(!sanitized.contains("SuperSecret123")); + } + + #[test] + fn given_password_with_at_sign_should_be_fully_redacted() { + let url = "jdbc:mysql://root:p@ss@localhost:3306/mydb"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:mysql://root:***@localhost:3306/mydb"); + assert!(!sanitized.contains("p@ss")); + } + + #[test] + fn given_echoed_jdbc_url_should_be_redacted_from_error() { + let url = "jdbc:mysql://root:p@ss@localhost:3306/mydb"; + let error = Error::InitError(format!("No suitable driver found for {url}")); + let sanitized = sanitize_jdbc_error(error, url).to_string(); + assert_eq!( + sanitized, + "Init error: No suitable driver found for jdbc:mysql://root:***@localhost:3306/mydb" + ); + assert!(!sanitized.contains("p@ss")); + } + + #[test] + fn given_postgres_url_should_redact_password() { + let url = "jdbc:postgresql://localhost:5432/mydb?user=admin&password=P@ssw0rd&ssl=true"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!( + sanitized, + "jdbc:postgresql://localhost:5432/mydb?user=admin&password=***&ssl=true" + ); + assert!(!sanitized.contains("P@ssw0rd")); + } + + #[test] + fn given_oracle_url_should_redact_password() { + let url = "jdbc:oracle:thin:system/oracle123@localhost:1521:XE"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:oracle:thin:system/***@localhost:1521:XE"); + assert!(!sanitized.contains("oracle123")); + } + + #[test] + fn given_sql_server_url_should_redact_password() { + let url = "jdbc:sqlserver://localhost:1433;user=sa;password=MySecretPass;database=Sales"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!( + sanitized, + "jdbc:sqlserver://localhost:1433;user=sa;password=***;database=Sales" + ); + assert!(!sanitized.contains("MySecretPass")); + } + + #[test] + fn given_h2_url_should_redact_password() { + let url = "jdbc:h2:mem:testdb;USER=sa;PASSWORD=secret"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, "jdbc:h2:mem:testdb;USER=sa;PASSWORD=***"); + assert!(!sanitized.contains("secret")); + } + + #[test] + fn given_mixed_case_password_key_should_be_redacted() { + let url1 = "jdbc:postgresql://localhost?password=secret"; + let url2 = "jdbc:postgresql://localhost?PASSWORD=secret"; + let url3 = "jdbc:postgresql://localhost?pwd=secret"; + let url4 = "jdbc:postgresql://localhost?PWD=secret"; + + for url in [url1, url2, url3, url4] { + let sanitized = sanitize_jdbc_url(url); + assert!(!sanitized.contains("secret"), "Failed for URL: {}", url); + assert!(sanitized.contains("***")); + } + } + + #[test] + fn given_url_without_password_should_remain_unchanged() { + let url = "jdbc:h2:mem:testdb"; + let sanitized = sanitize_jdbc_url(url); + assert_eq!(sanitized, url); + } + + #[test] + fn given_multiple_passwords_should_all_be_redacted() { + let url = "jdbc:postgresql://localhost?password=secret1&pwd=secret2"; + let sanitized = sanitize_jdbc_url(url); + assert!(!sanitized.contains("secret1")); + assert!(!sanitized.contains("secret2")); + assert_eq!( + sanitized, + "jdbc:postgresql://localhost?password=***&pwd=***" + ); + } + + #[test] + fn given_different_jvm_config_should_be_rejected() { + let first_jar = write_temp_jar("jdbc_jvm_first.jar"); + let second_jar = write_temp_jar("jdbc_jvm_second.jar"); + let first = JvmConfiguration::new(&first_jar, &["-Xmx128m".to_string()]) + .expect("first JVM configuration"); + let same = JvmConfiguration::new(&first_jar, &["-Xmx128m".to_string()]) + .expect("same JVM configuration"); + let different_jar = JvmConfiguration::new(&second_jar, &["-Xmx128m".to_string()]) + .expect("different jar configuration"); + let different_options = JvmConfiguration::new(&first_jar, &["-Xmx256m".to_string()]) + .expect("different options configuration"); + + assert!(first.ensure_compatible_with(&same).is_ok()); + assert!(first.ensure_compatible_with(&different_jar).is_err()); + assert!(first.ensure_compatible_with(&different_options).is_err()); + } + + #[test] + fn given_incremental_offset_should_be_bound() { + let config = JdbcSourceConfig { + query: "SELECT * FROM users WHERE id > {last_offset} ORDER BY id".to_string(), + tracking_column: Some("id".to_string()), + initial_offset: Some("0".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + + // With initial offset (no last_offset yet) + let state = State { + last_offset: None, + processed_rows: 0, + }; + let query = source.build_query(&state).expect("build query"); + assert_eq!(query.sql, "SELECT * FROM users WHERE id > ? ORDER BY id"); + assert_eq!(query.offset.as_deref(), Some("0")); + assert_eq!(query.offset_parameter_count, 1); + + // With tracked offset + let state = State { + last_offset: Some("42".to_string()), + processed_rows: 42, + }; + let query = source.build_query(&state).expect("build query"); + assert_eq!(query.sql, "SELECT * FROM users WHERE id > ? ORDER BY id"); + assert_eq!(query.offset.as_deref(), Some("42")); + assert_eq!(query.offset_parameter_count, 1); + } + + #[test] + fn given_repeated_offset_placeholder_when_query_is_built_should_bind_each_parameter() { + let config = JdbcSourceConfig { + query: "SELECT * FROM users WHERE id > {last_offset} OR parent_id > {last_offset} ORDER BY id" + .to_string(), + tracking_column: Some("id".to_string()), + initial_offset: Some(r"a\b".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + + let query = source + .build_query(&State::default()) + .expect("build prepared query"); + + assert_eq!( + query.sql, + "SELECT * FROM users WHERE id > ? OR parent_id > ? ORDER BY id" + ); + assert_eq!(query.offset.as_deref(), Some(r"a\b")); + assert_eq!(query.offset_parameter_count, 2); + } + + #[test] + fn given_tracking_placeholder_should_be_substituted() { + let config = JdbcSourceConfig { + query: + "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(), + tracking_column: Some("updated_at".to_string()), + initial_offset: Some("2024-01-01".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + let state = State { + last_offset: Some("2024-06-15".to_string()), + processed_rows: 0, + }; + let query = source.build_query(&state).expect("build query"); + assert_eq!( + query.sql, + "SELECT * FROM orders WHERE updated_at > ? ORDER BY updated_at" + ); + assert_eq!(query.offset.as_deref(), Some("2024-06-15")); + } + + #[test] + fn given_no_offset_should_still_substitute_ordering_column() { + let config = JdbcSourceConfig { + query: + "SELECT * FROM orders WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(), + tracking_column: Some("updated_at".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + // No last_offset and no initial_offset: the WHERE predicate is dropped, + // but the ORDER BY {tracking_column} must still be substituted. + let query = source + .build_query(&State { + last_offset: None, + processed_rows: 0, + }) + .expect("build query"); + assert!( + !query.sql.contains("{tracking_column}"), + "got: {}", + query.sql + ); + assert!(!query.sql.contains("{last_offset}"), "got: {}", query.sql); + assert!( + query.sql.contains("ORDER BY updated_at"), + "got: {}", + query.sql + ); + assert!(query.offset.is_none()); + } + + #[test] + fn given_unsafe_tracking_identifier_should_be_rejected() { + let config = JdbcSourceConfig { + query: "SELECT * FROM t WHERE {tracking_column} > {last_offset}".to_string(), + tracking_column: Some("id; DROP TABLE t".to_string()), + initial_offset: Some("0".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + assert!(source.build_query(&State::default()).is_err()); + } + + #[test] + fn given_bulk_query_should_not_append_limit() { + let config = JdbcSourceConfig { + query: "SELECT * FROM products".to_string(), + poll_interval: Some("60s".to_string()), + batch_size: 5000, + include_metadata: false, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + let state = State::default(); + let query = source.build_query(&state).expect("build query"); + // build_query should NOT append LIMIT; row limiting is done via setMaxRows + assert_eq!(query.sql, "SELECT * FROM products"); + assert!(!query.sql.to_uppercase().contains("LIMIT")); + } + + #[test] + fn given_persisted_state_should_restore_cursor_and_count() { + let original_state = State { + last_offset: Some("2024-06-15 12:00:00".to_string()), + processed_rows: 1500, + }; + let connector_state = ConnectorState::serialize(&original_state, CONNECTOR_NAME, 1) + .expect("Failed to serialize state"); + + let config = JdbcSourceConfig { + query: "SELECT * FROM orders WHERE updated_at > {last_offset}".to_string(), + poll_interval: Some("30s".to_string()), + batch_size: 1000, + tracking_column: Some("updated_at".to_string()), + initial_offset: Some("2024-01-01 00:00:00".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, Some(connector_state)); + let state = source.state.lock().unwrap(); + assert_eq!(state.last_offset, Some("2024-06-15 12:00:00".to_string())); + assert_eq!(state.processed_rows, 1500); + } + + #[test] + fn state_should_be_serializable_and_deserializable() { + let original = State { + last_offset: Some("2026-10-02T10:15:00Z".to_string()), + processed_rows: 42, + }; + let bytes = rmp_serde::to_vec(&original).expect("state should serialize"); + let restored: State = rmp_serde::from_slice(&bytes).expect("state should deserialize"); + assert_eq!(restored.last_offset, original.last_offset); + assert_eq!(restored.processed_rows, original.processed_rows); + } + + #[test] + fn given_built_message_should_leave_broker_timestamps_unset() { + let source = JdbcSource::new(1, base_config(), None); + let mut row = serde_json::Map::new(); + row.insert("id".to_string(), serde_json::json!(1)); + + let message = source.build_message(row).expect("message should build"); + + assert!(message.timestamp.is_none()); + assert!(message.origin_timestamp.is_none()); + let payload: DatabaseRecord = + serde_json::from_slice(&message.payload).expect("metadata payload should deserialize"); + assert_eq!(payload.data["id"], 1); + } + + #[test] + fn given_sql_state_should_be_classified() { + for s in [ + "08001", "08006", "40001", "40P01", "53300", "57P01", "58030", + ] { + assert!(is_transient_sql_state(Some(s)), "{s} should be transient"); + } + for s in ["22001", "23505", "42601", "42P01", "99999"] { + assert!(!is_transient_sql_state(Some(s)), "{s} should be permanent"); + } + assert!(!is_transient_sql_state(None)); + assert!(!is_transient_sql_state(Some(""))); + } + + #[test] + fn given_sql_identifier_should_be_validated() { + assert!(is_valid_identifier("id")); + assert!(is_valid_identifier("updated_at")); + assert!(is_valid_identifier("t.updated_at")); + assert!(!is_valid_identifier("id; DROP TABLE t")); + assert!(!is_valid_identifier("1col")); + assert!(!is_valid_identifier("col name")); + assert!(!is_valid_identifier("")); + } + + #[test] + fn given_column_name_should_convert_to_snake_case() { + assert_eq!(to_snake_case("OrderDate"), "order_date"); + assert_eq!(to_snake_case("updatedAt"), "updated_at"); + assert_eq!(to_snake_case("ID"), "id"); // consecutive uppers stay together + assert_eq!(to_snake_case("already_snake"), "already_snake"); + assert_eq!(to_snake_case("simple"), "simple"); + } + + // ========================================================================= + // Config deserialization tests + // ========================================================================= + + #[test] + fn given_minimal_toml_should_apply_defaults() { + let toml_str = r#" + jdbc_url = "jdbc:h2:mem:test" + driver_class = "org.h2.Driver" + driver_jar_path = "/tmp/h2.jar" + query = "SELECT * FROM users" + poll_interval = "30s" + "#; + let config: JdbcSourceConfig = + toml::from_str(toml_str).expect("Failed to parse minimal TOML config"); + assert_eq!(config.driver_class, "org.h2.Driver"); + assert_eq!(config.query, "SELECT * FROM users"); + assert_eq!(config.poll_interval.as_deref(), Some("30s")); + assert_eq!( + parse_poll_interval(config.poll_interval.as_deref()), + Duration::from_secs(30) + ); + // Verify defaults are applied + assert_eq!(config.mode, Mode::Incremental); + assert_eq!(config.batch_size, 1000); + assert!(config.include_metadata); + assert!(!config.snake_case_columns); + assert_eq!(config.verbose_logging, None); + assert_eq!(config.connection_timeout_ms, 5000); + assert_eq!(config.login_timeout_ms, 30000); + assert_eq!(config.query_timeout_ms, 30000); + assert!(config.username.is_none()); + assert!(config.password.is_none()); + assert!(config.tracking_column.is_none()); + assert!(config.initial_offset.is_none()); + assert!(config.jvm_options.is_empty()); + } + + #[test] + fn given_full_toml_should_deserialize_all_fields() { + let toml_str = r#" + jdbc_url = "jdbc:mysql://localhost:3306/mydb" + driver_class = "com.mysql.cj.jdbc.Driver" + driver_jar_path = "/opt/drivers/mysql.jar" + username = "admin" + password = "s3cret" + query = "SELECT * FROM orders WHERE id > {last_offset} ORDER BY id" + poll_interval = "5m" + batch_size = 500 + tracking_column = "id" + initial_offset = "0" + mode = "incremental" + snake_case_columns = true + include_metadata = false + verbose_logging = true + jvm_options = ["-Xmx512m", "-Xms128m"] + connection_timeout_ms = 60000 + login_timeout_ms = 45000 + query_timeout_ms = 120000 + "#; + let config: JdbcSourceConfig = + toml::from_str(toml_str).expect("Failed to parse full TOML config"); + assert_eq!(config.driver_class, "com.mysql.cj.jdbc.Driver"); + assert_eq!(config.username.as_deref(), Some("admin")); + assert!(config.password.is_some()); + assert_eq!(config.batch_size, 500); + assert_eq!(config.tracking_column.as_deref(), Some("id")); + assert_eq!(config.initial_offset.as_deref(), Some("0")); + assert_eq!(config.mode, Mode::Incremental); + assert!(config.snake_case_columns); + assert!(!config.include_metadata); + assert_eq!(config.verbose_logging, Some(true)); + assert_eq!(config.jvm_options, vec!["-Xmx512m", "-Xms128m"]); + assert_eq!(config.connection_timeout_ms, 60000); + assert_eq!(config.login_timeout_ms, 45000); + assert_eq!(config.query_timeout_ms, 120000); + assert_eq!( + parse_poll_interval(config.poll_interval.as_deref()), + Duration::from_secs(300) + ); + } + + #[test] + fn given_bulk_toml_should_deserialize_mode() { + let toml_str = r#" + jdbc_url = "jdbc:h2:mem:test" + driver_class = "org.h2.Driver" + driver_jar_path = "/tmp/h2.jar" + query = "SELECT * FROM products" + poll_interval = "1h" + mode = "bulk" + "#; + let config: JdbcSourceConfig = + toml::from_str(toml_str).expect("Failed to parse bulk mode config"); + assert_eq!(config.mode, Mode::Bulk); + assert_eq!( + parse_poll_interval(config.poll_interval.as_deref()), + Duration::from_secs(3600) + ); + } + + #[test] + fn given_invalid_mode_should_fail_deserialization() { + let toml_str = r#" + jdbc_url = "jdbc:h2:mem:test" + driver_class = "org.h2.Driver" + driver_jar_path = "/tmp/h2.jar" + query = "SELECT 1" + poll_interval = "1s" + mode = "invalid_mode" + "#; + let result = toml::from_str::(toml_str); + assert!( + result.is_err(), + "Expected error for invalid mode, but got: {:?}", + result + ); + } + + #[test] + fn given_shipped_examples_should_parse_each_plugin_config() { + let examples = [ + ( + "bulk", + include_str!("../../../runtime/example_config/connectors/jdbc_bulk_mode.toml"), + ), + ( + "h2", + include_str!("../../../runtime/example_config/connectors/jdbc_h2.toml"), + ), + ( + "mysql", + include_str!("../../../runtime/example_config/connectors/jdbc_mysql.toml"), + ), + ( + "oracle", + include_str!("../../../runtime/example_config/connectors/jdbc_oracle.toml"), + ), + ( + "sqlserver", + include_str!("../../../runtime/example_config/connectors/jdbc_sqlserver.toml"), + ), + ]; + + for (name, example) in examples { + let document: toml::Value = toml::from_str(example).unwrap_or_else(|error| { + panic!("shipped {name} example must be valid TOML: {error}") + }); + let plugin_config = document + .get("plugin_config") + .cloned() + .unwrap_or_else(|| panic!("shipped {name} example must contain plugin_config")); + let config: JdbcSourceConfig = plugin_config.try_into().unwrap_or_else(|error| { + panic!("shipped {name} plugin_config must deserialize: {error}") + }); + + if name == "h2" { + assert_eq!(config.mode, Mode::Incremental); + assert!( + config + .jdbc_url + .expose_secret() + .contains(r"\;INSERT INTO users"), + "H2 INIT statements must be separated with an escaped semicolon" + ); + } + } + } + + // ========================================================================= + // State restoration tests + // ========================================================================= + + #[test] + fn given_malformed_state_should_start_fresh() { + let connector_state = ConnectorState(vec![0xFF, 0xFE, 0xFD, 0x00]); + + let source = JdbcSource::new(1, base_config(), Some(connector_state)); + let state = source.state.lock().unwrap(); + // Should fall back to default state + assert!(state.last_offset.is_none()); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn given_empty_state_should_start_fresh() { + let connector_state = ConnectorState(vec![]); + + let source = JdbcSource::new(1, base_config(), Some(connector_state)); + let state = source.state.lock().unwrap(); + assert!(state.last_offset.is_none()); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn given_no_state_should_use_initial_offset() { + let config = JdbcSourceConfig { + query: "SELECT * FROM orders WHERE id > {last_offset}".to_string(), + tracking_column: Some("id".to_string()), + initial_offset: Some("100".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + let state = source.state.lock().unwrap(); + assert_eq!(state.last_offset, Some("100".to_string())); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn given_no_state_or_initial_offset_should_start_fresh() { + let config = JdbcSourceConfig { + query: "SELECT * FROM products".to_string(), + poll_interval: Some("60s".to_string()), + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + let state = source.state.lock().unwrap(); + assert!(state.last_offset.is_none()); + assert_eq!(state.processed_rows, 0); + } + + // ========================================================================= + // extract_offset_value tests + // ========================================================================= + + #[test] + fn given_integer_cursor_should_extract_as_string() { + let mut row = serde_json::Map::new(); + row.insert("id".to_string(), serde_json::json!(42)); + assert_eq!(extract_offset_value(&row, "id"), Some("42".to_string())); + } + + #[test] + fn given_string_cursor_should_extract_unchanged() { + let mut row = serde_json::Map::new(); + row.insert( + "updated_at".to_string(), + serde_json::json!("2024-06-15 12:00:00"), + ); + assert_eq!( + extract_offset_value(&row, "updated_at"), + Some("2024-06-15 12:00:00".to_string()) + ); + } + + #[test] + fn given_float_cursor_should_extract_as_string() { + let mut row = serde_json::Map::new(); + row.insert("version".to_string(), serde_json::json!(3.5)); + assert_eq!( + extract_offset_value(&row, "version"), + Some("3.5".to_string()) + ); + } + + #[test] + fn given_null_cursor_should_not_extract() { + let mut row = serde_json::Map::new(); + row.insert("id".to_string(), serde_json::Value::Null); + // A SQL NULL tracking value must not become the persisted offset. + assert_eq!(extract_offset_value(&row, "id"), None); + } + + #[test] + fn given_missing_cursor_column_should_not_extract() { + let row = serde_json::Map::new(); + assert_eq!(extract_offset_value(&row, "nonexistent"), None); + } + + // ========================================================================= + // build_query edge case tests + // ========================================================================= + + #[test] + fn given_no_incremental_offset_should_remove_where_clause() { + let config = JdbcSourceConfig { + query: "SELECT * FROM users WHERE {tracking_column} > {last_offset} ORDER BY id" + .to_string(), + tracking_column: Some("id".to_string()), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + let state = State { + last_offset: None, + processed_rows: 0, + }; + let query = source.build_query(&state).expect("build query"); + // The WHERE clause placeholder should be removed + assert!( + !query.sql.contains("{last_offset}"), + "Query should not contain unresolved placeholder: {}", + query.sql + ); + } + + #[test] + fn given_unresolved_placeholder_should_be_rejected() { + // Incremental, no offset, and a non-canonical predicate that the + // auto-remove does not match: the unresolved {last_offset} must produce + // an error rather than being shipped to the driver as invalid SQL. + let config = JdbcSourceConfig { + query: "SELECT * FROM t WHERE id >= {last_offset}".to_string(), + mode: Mode::Incremental, + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + assert!(source.build_query(&State::default()).is_err()); + } + + #[test] + fn given_bulk_mode_should_ignore_offset() { + let config = JdbcSourceConfig { + query: "SELECT * FROM users ORDER BY id".to_string(), + tracking_column: Some("id".to_string()), + initial_offset: Some("0".to_string()), + ..base_config() + }; + let source = JdbcSource::new(1, config, None); + let state = State { + last_offset: Some("42".to_string()), + processed_rows: 42, + }; + // In bulk mode the query is used verbatim; the tracked offset is ignored. + let query = source.build_query(&state).expect("build query"); + assert_eq!(query.sql, "SELECT * FROM users ORDER BY id"); + } + + // ========================================================================= + // Mode enum tests + // ========================================================================= + + #[test] + fn given_mode_should_round_trip_through_json() { + let incremental = Mode::Incremental; + let serialized = serde_json::to_string(&incremental).unwrap(); + assert_eq!(serialized, r#""incremental""#); + let deserialized: Mode = serde_json::from_str(&serialized).unwrap(); + assert_eq!(deserialized, Mode::Incremental); + + let bulk = Mode::Bulk; + let serialized = serde_json::to_string(&bulk).unwrap(); + assert_eq!(serialized, r#""bulk""#); + let deserialized: Mode = serde_json::from_str(&serialized).unwrap(); + assert_eq!(deserialized, Mode::Bulk); + } + + #[test] + fn given_unknown_mode_should_fail_deserialization() { + let result = serde_json::from_str::(r#""streaming""#); + assert!( + result.is_err(), + "Unknown mode 'streaming' should fail deserialization" + ); + } + + // ========================================================================= + // Debug impl tests (ensures secrets are not leaked) + // ========================================================================= + + #[test] + fn given_config_debug_should_not_leak_password() { + let config = JdbcSourceConfig { + jdbc_url: SecretString::from("jdbc:mysql://root:SuperSecret@localhost/db"), + driver_class: "com.mysql.cj.jdbc.Driver".to_string(), + driver_jar_path: "/tmp/mysql.jar".to_string(), + username: Some("admin".to_string()), + password: Some(SecretString::from("MyP@ssw0rd")), + ..base_config() + }; + + let debug_output = format!("{:?}", config); + assert!( + !debug_output.contains("SuperSecret"), + "Debug output should not contain JDBC URL password: {}", + debug_output + ); + assert!( + !debug_output.contains("MyP@ssw0rd"), + "Debug output should not contain password field: {}", + debug_output + ); + assert!( + debug_output.contains("***"), + "Debug output should contain masked password: {}", + debug_output + ); + } + + // ========================================================================= + // Batch-result (staged cursor) tests + // + // Named after the sibling `random_source` convention, which covers the same + // Ack/Nack contract. + // ========================================================================= + + /// An incremental source whose committed offset starts at `committed`. + fn incremental_source_at(committed: &str) -> JdbcSource { + let mut config = base_config(); + config.mode = Mode::Incremental; + config.tracking_column = Some("id".to_string()); + config.query = + "SELECT id FROM t WHERE {tracking_column} > {last_offset} ORDER BY {tracking_column}" + .to_string(); + let source = JdbcSource::new(1, config, None); + source.state.lock().expect("state lock").last_offset = Some(committed.to_string()); + source + } + + /// Stage the cursor a fetched batch would advance to, as `poll` does. + fn stage_candidate(source: &JdbcSource, offset: &str, processed_rows: u64) { + *source.pending_state.lock().expect("pending lock") = Some(State { + last_offset: Some(offset.to_string()), + processed_rows, + }); + } + + fn block_on(future: F) -> F::Output { + tokio::runtime::Builder::new_current_thread() + .build() + .expect("build test runtime") + .block_on(future) + } + + #[test] + fn given_ack_when_batch_is_staged_should_commit_candidate_offset() { + let source = incremental_source_at("10"); + stage_candidate(&source, "42", 32); + + block_on(source.on_batch_result(SourceBatchResult::Ack)).expect("ACK should apply"); + + let state = source.state.lock().expect("state lock"); + assert_eq!(state.last_offset, Some("42".to_string())); + assert_eq!(state.processed_rows, 32); + assert!(source.pending_state.lock().expect("pending lock").is_none()); + } + + #[test] + fn given_nack_when_batch_is_staged_should_keep_committed_offset() { + let source = incremental_source_at("10"); + stage_candidate(&source, "42", 32); + + block_on(source.on_batch_result(SourceBatchResult::Nack)).expect("NACK should apply"); + + let state = source.state.lock().expect("state lock"); + // The undelivered batch must not advance the cursor, and the staged value + // must be discarded rather than lingering for a later batch to commit. + assert_eq!(state.last_offset, Some("10".to_string())); + assert_eq!(state.processed_rows, 0); + assert!(source.pending_state.lock().expect("pending lock").is_none()); + } + + #[test] + fn given_nack_when_next_poll_builds_query_should_re_read_the_same_range() { + let source = incremental_source_at("10"); + stage_candidate(&source, "42", 32); + block_on(source.on_batch_result(SourceBatchResult::Nack)).expect("NACK should apply"); + + // The regression this guards: advancing the cursor at fetch time would + // rebuild the next query from '42' and permanently skip rows 11..=42. + let state = source.state.lock().expect("state lock"); + let query = source.build_query(&state).expect("build query"); + assert_eq!(query.sql, "SELECT id FROM t WHERE id > ? ORDER BY id"); + assert_eq!(query.offset.as_deref(), Some("10")); + } + + #[test] + fn given_ack_when_nothing_is_staged_should_leave_committed_state_unchanged() { + let source = incremental_source_at("10"); + // An empty poll stages nothing, so an ACK for it must not disturb the cursor. + block_on(source.on_batch_result(SourceBatchResult::Ack)).expect("ACK should apply"); + + let state = source.state.lock().expect("state lock"); + assert_eq!(state.last_offset, Some("10".to_string())); + assert_eq!(state.processed_rows, 0); + } + + #[test] + fn given_config_without_password_should_debug_safely() { + let debug_output = format!("{:?}", base_config()); + // Should not panic and should contain the struct name + assert!(debug_output.contains("JdbcSourceConfig")); + } +} diff --git a/core/integration/Cargo.toml b/core/integration/Cargo.toml index ba6ff18a41..d25e6167c8 100644 --- a/core/integration/Cargo.toml +++ b/core/integration/Cargo.toml @@ -96,6 +96,7 @@ secrecy = { workspace = true } serde = { workspace = true } serde_json = { workspace = true } serial_test = { workspace = true } +sha2 = { workspace = true } sqlparser = { workspace = true } sqlx = { workspace = true } sysinfo = { workspace = true } diff --git a/core/integration/tests/connectors/fixtures/jdbc.rs b/core/integration/tests/connectors/fixtures/jdbc.rs new file mode 100644 index 0000000000..ad4f1ee0fc --- /dev/null +++ b/core/integration/tests/connectors/fixtures/jdbc.rs @@ -0,0 +1,358 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use super::postgres::PostgresContainer; +use async_trait::async_trait; +use integration::harness::{TestBinaryError, TestFixture, seeds}; +use sha2::{Digest, Sha256}; +use sqlx::{Pool, Postgres}; +use std::collections::HashMap; +use std::io::Cursor; +use zip::ZipArchive; + +const DRIVER_CLASS_ENTRY: &str = "org/postgresql/Driver.class"; +const POSTGRES_DRIVER_SHA256: &str = + "49bba9c3200d4f64ae73903d56ce1bd09c74517dfe31acb44745506b4fcede53"; + +const ENV_JDBC_URL: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_JDBC_URL"; +const ENV_DRIVER_CLASS: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_CLASS"; +const ENV_DRIVER_JAR_PATH: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_DRIVER_JAR_PATH"; +const ENV_USERNAME: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_USERNAME"; +const ENV_PASSWORD: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_PASSWORD"; +const ENV_QUERY: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_QUERY"; +const ENV_POLL_INTERVAL: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_POLL_INTERVAL"; +const ENV_BATCH_SIZE: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_BATCH_SIZE"; +const ENV_MODE: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_MODE"; +const ENV_TRACKING_COLUMN: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_TRACKING_COLUMN"; +const ENV_INITIAL_OFFSET: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INITIAL_OFFSET"; +const ENV_INCLUDE_METADATA: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_INCLUDE_METADATA"; +const ENV_QUERY_TIMEOUT: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_PLUGIN_CONFIG_QUERY_TIMEOUT_MS"; +const ENV_STREAM: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_STREAM"; +const ENV_TOPIC: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_TOPIC"; +const ENV_SCHEMA: &str = "IGGY_CONNECTORS_SOURCE_JDBC_PG_STREAMS_0_SCHEMA"; + +#[derive(Clone, Copy)] +enum Scenario { + Empty, + Incremental, + TextCursor, + LargeResult, + BulkOverflow, + TieBoundary, +} + +struct JdbcPostgresFixture { + container: PostgresContainer, + jdbc_url: String, + driver_jar_path: String, +} + +impl JdbcPostgresFixture { + async fn setup(scenario: Scenario) -> Result { + let container = PostgresContainer::start().await?; + let host_and_port = container + .connection_string() + .rsplit('@') + .next() + .ok_or_else(|| fixture_error("PostgreSQL connection string has no host"))?; + let fixture = Self { + jdbc_url: format!("jdbc:postgresql://{host_and_port}/postgres"), + driver_jar_path: postgres_driver_jar().await?, + container, + }; + fixture.seed_database(scenario).await?; + Ok(fixture) + } + + async fn create_pool(&self) -> Result, TestBinaryError> { + self.container.create_pool().await + } + + async fn seed_database(&self, scenario: Scenario) -> Result<(), TestBinaryError> { + let pool = self.create_pool().await?; + let statements: &[&str] = match scenario { + Scenario::Empty => &[], + Scenario::Incremental => &[ + "CREATE TABLE inc_test (id INT PRIMARY KEY, name TEXT)", + "INSERT INTO inc_test (id, name) VALUES (1, 'a'), (2, 'b'), (3, 'c')", + ], + Scenario::TextCursor => &[ + "CREATE TABLE text_cursor_test (id INT PRIMARY KEY, cursor_value TEXT UNIQUE NOT NULL)", + r"INSERT INTO text_cursor_test (id, cursor_value) VALUES (1, E'a\\b'), (2, 'z')", + ], + Scenario::LargeResult => &[ + "CREATE TABLE big_test (id INT PRIMARY KEY, name TEXT, val NUMERIC(12,2))", + "INSERT INTO big_test (id, name, val) SELECT g, 'row_' || g, (g * 1.5)::numeric(12,2) FROM generate_series(1, 300) g", + ], + Scenario::BulkOverflow => &[ + "CREATE TABLE trunc_test (id INT PRIMARY KEY)", + "INSERT INTO trunc_test (id) SELECT generate_series(1, 5)", + ], + Scenario::TieBoundary => &[ + "CREATE TABLE tie_test (id INT PRIMARY KEY, position INT NOT NULL)", + "INSERT INTO tie_test (id, position) VALUES (1, 1), (2, 2), (3, 2), (4, 3)", + ], + }; + for statement in statements { + sqlx::query(*statement) + .execute(&pool) + .await + .map_err(|error| fixture_error(format!("failed to seed PostgreSQL: {error}")))?; + } + pool.close().await; + Ok(()) + } + + fn envs( + &self, + query: &str, + mode: &str, + batch_size: u32, + tracking_column: &str, + initial_offset: Option<&str>, + ) -> HashMap { + let mut envs = HashMap::from([ + (ENV_JDBC_URL.to_string(), self.jdbc_url.clone()), + ( + ENV_DRIVER_CLASS.to_string(), + "org.postgresql.Driver".to_string(), + ), + ( + ENV_DRIVER_JAR_PATH.to_string(), + self.driver_jar_path.clone(), + ), + (ENV_USERNAME.to_string(), "postgres".to_string()), + (ENV_PASSWORD.to_string(), "postgres".to_string()), + (ENV_QUERY.to_string(), query.to_string()), + (ENV_POLL_INTERVAL.to_string(), "100ms".to_string()), + (ENV_BATCH_SIZE.to_string(), batch_size.to_string()), + (ENV_MODE.to_string(), mode.to_string()), + (ENV_TRACKING_COLUMN.to_string(), tracking_column.to_string()), + (ENV_INCLUDE_METADATA.to_string(), "true".to_string()), + (ENV_STREAM.to_string(), seeds::names::STREAM.to_string()), + (ENV_TOPIC.to_string(), seeds::names::TOPIC.to_string()), + (ENV_SCHEMA.to_string(), "json".to_string()), + ]); + if let Some(initial_offset) = initial_offset { + envs.insert(ENV_INITIAL_OFFSET.to_string(), initial_offset.to_string()); + } + envs + } +} + +macro_rules! jdbc_fixture { + ($name:ident, $scenario:expr, $query:expr, $mode:expr, $batch_size:expr, $tracking:expr, $offset:expr) => { + pub struct $name { + base: JdbcPostgresFixture, + } + + #[async_trait] + impl TestFixture for $name { + async fn setup() -> Result { + Ok(Self { + base: JdbcPostgresFixture::setup($scenario).await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + self.base + .envs($query, $mode, $batch_size, $tracking, $offset) + } + } + }; +} + +jdbc_fixture!( + JdbcBulkFixture, + Scenario::Empty, + "SELECT 1 AS id, 'test' AS name", + "bulk", + 100, + "id", + None +); + +impl JdbcIncrementalFixture { + pub async fn create_pool(&self) -> Result, TestBinaryError> { + self.base.create_pool().await + } +} + +impl JdbcRecoveryFixture { + pub async fn create_pool(&self) -> Result, TestBinaryError> { + self.base.create_pool().await + } +} +jdbc_fixture!( + JdbcBulkRowsFixture, + Scenario::Empty, + "SELECT * FROM (VALUES (1, 'alice', true), (2, 'bob', false), (3, 'carol', true)) AS t(id, name, active)", + "bulk", + 100, + "id", + None +); +jdbc_fixture!( + JdbcMetadataFixture, + Scenario::Empty, + "SELECT 42 AS value", + "bulk", + 100, + "id", + None +); +jdbc_fixture!( + JdbcIncrementalFixture, + Scenario::Incremental, + "SELECT id, name FROM inc_test WHERE id > {last_offset} ORDER BY id", + "incremental", + 2, + "id", + None +); +jdbc_fixture!( + JdbcTextCursorFixture, + Scenario::TextCursor, + "SELECT id, cursor_value FROM text_cursor_test WHERE cursor_value > {last_offset} ORDER BY cursor_value", + "incremental", + 100, + "cursor_value", + Some(r"a\b") +); +jdbc_fixture!( + JdbcLargeResultFixture, + Scenario::LargeResult, + "SELECT id, name, val FROM big_test ORDER BY id", + "bulk", + 5000, + "id", + None +); +jdbc_fixture!( + JdbcRecoveryFixture, + Scenario::Empty, + "SELECT id, name FROM recover_test ORDER BY id", + "bulk", + 100, + "id", + None +); +jdbc_fixture!( + JdbcBulkOverflowFixture, + Scenario::BulkOverflow, + "SELECT id FROM trunc_test ORDER BY id", + "bulk", + 2, + "id", + None +); +jdbc_fixture!( + JdbcTieBoundaryFixture, + Scenario::TieBoundary, + "SELECT id, position FROM tie_test WHERE position > {last_offset} ORDER BY position, id", + "incremental", + 2, + "position", + None +); + +pub struct JdbcQueryTimeoutFixture { + base: JdbcPostgresFixture, +} + +#[async_trait] +impl TestFixture for JdbcQueryTimeoutFixture { + async fn setup() -> Result { + Ok(Self { + base: JdbcPostgresFixture::setup(Scenario::Empty).await?, + }) + } + + fn connectors_runtime_envs(&self) -> HashMap { + let mut envs = self + .base + .envs("SELECT 1 AS id FROM pg_sleep(2)", "bulk", 100, "id", None); + envs.insert(ENV_QUERY_TIMEOUT.to_string(), "1000".to_string()); + envs + } +} + +async fn postgres_driver_jar() -> Result { + let target_dir = std::env::var("CARGO_TARGET_DIR").unwrap_or_else(|_| "target".to_string()); + let jdbc_test_dir = format!("{target_dir}/test-jdbc-drivers"); + let jar_path = format!("{jdbc_test_dir}/postgresql-42.7.1.jar"); + std::fs::create_dir_all(&jdbc_test_dir) + .map_err(|error| fixture_error(format!("failed to create driver cache: {error}")))?; + + if std::path::Path::new(&jar_path).exists() { + match std::fs::read(&jar_path) { + Ok(bytes) if is_expected_driver_jar(&bytes) => return canonical_path(&jar_path), + _ => { + std::fs::remove_file(&jar_path).map_err(|error| { + fixture_error(format!("failed to remove invalid cached driver: {error}")) + })?; + } + } + } + + let response = reqwest::get( + "https://repo1.maven.org/maven2/org/postgresql/postgresql/42.7.1/postgresql-42.7.1.jar", + ) + .await + .map_err(|error| fixture_error(format!("failed to download JDBC driver: {error}")))? + .error_for_status() + .map_err(|error| fixture_error(format!("JDBC driver download failed: {error}")))?; + let bytes = response + .bytes() + .await + .map_err(|error| fixture_error(format!("failed to read JDBC driver: {error}")))?; + if !is_expected_driver_jar(&bytes) { + return Err(fixture_error(format!( + "downloaded JDBC driver failed SHA-256 or archive validation for {DRIVER_CLASS_ENTRY}" + ))); + } + + let temporary_path = format!("{jar_path}.{}.tmp", std::process::id()); + std::fs::write(&temporary_path, &bytes) + .map_err(|error| fixture_error(format!("failed to cache JDBC driver: {error}")))?; + std::fs::rename(&temporary_path, &jar_path) + .map_err(|error| fixture_error(format!("failed to publish JDBC driver cache: {error}")))?; + canonical_path(&jar_path) +} + +fn is_expected_driver_jar(bytes: &[u8]) -> bool { + if format!("{:x}", Sha256::digest(bytes)) != POSTGRES_DRIVER_SHA256 { + return false; + } + let Ok(mut archive) = ZipArchive::new(Cursor::new(bytes)) else { + return false; + }; + archive.by_name(DRIVER_CLASS_ENTRY).is_ok() +} + +fn canonical_path(path: &str) -> Result { + std::fs::canonicalize(path) + .map(|path| path.to_string_lossy().into_owned()) + .map_err(|error| fixture_error(format!("failed to resolve JDBC driver path: {error}"))) +} + +fn fixture_error(message: impl Into) -> TestBinaryError { + TestBinaryError::FixtureSetup { + fixture_type: "JDBC PostgreSQL".to_string(), + message: message.into(), + } +} diff --git a/core/integration/tests/connectors/fixtures/mod.rs b/core/integration/tests/connectors/fixtures/mod.rs index cddee53ab1..cdb2bb5a3e 100644 --- a/core/integration/tests/connectors/fixtures/mod.rs +++ b/core/integration/tests/connectors/fixtures/mod.rs @@ -25,6 +25,7 @@ mod floci; mod http; mod iceberg; mod influxdb; +mod jdbc; mod meilisearch; mod mongodb; mod postgres; @@ -75,6 +76,11 @@ pub use influxdb::{ InfluxDbSinkNoMetadataFixture, InfluxDbSinkNsPrecisionFixture, InfluxDbSinkTextFixture, InfluxDbSourceFixture, InfluxDbSourceRawFixture, InfluxDbSourceTextFixture, }; +pub use jdbc::{ + JdbcBulkFixture, JdbcBulkOverflowFixture, JdbcBulkRowsFixture, JdbcIncrementalFixture, + JdbcLargeResultFixture, JdbcMetadataFixture, JdbcQueryTimeoutFixture, JdbcRecoveryFixture, + JdbcTextCursorFixture, JdbcTieBoundaryFixture, +}; pub use meilisearch::{MeilisearchOps, MeilisearchSinkFixture, TEST_INDEX}; pub use mongodb::{ MongoDbOps, MongoDbSinkAutoCreateFixture, MongoDbSinkBatchFixture, MongoDbSinkFailpointFixture, diff --git a/core/integration/tests/connectors/fixtures/postgres/container.rs b/core/integration/tests/connectors/fixtures/postgres/container.rs index 508607f0c6..68ac156f7d 100644 --- a/core/integration/tests/connectors/fixtures/postgres/container.rs +++ b/core/integration/tests/connectors/fixtures/postgres/container.rs @@ -128,7 +128,7 @@ pub struct PostgresContainer { } impl PostgresContainer { - pub(super) async fn start() -> Result { + pub(crate) async fn start() -> Result { Self::start_with_image(postgres::Postgres::default().into()).await } @@ -180,4 +180,8 @@ impl PostgresContainer { message: format!("Failed to connect: {e}"), }) } + + pub(crate) fn connection_string(&self) -> &str { + &self.connection_string + } } diff --git a/core/integration/tests/connectors/fixtures/postgres/mod.rs b/core/integration/tests/connectors/fixtures/postgres/mod.rs index 7f186d406b..0d8b138f39 100644 --- a/core/integration/tests/connectors/fixtures/postgres/mod.rs +++ b/core/integration/tests/connectors/fixtures/postgres/mod.rs @@ -21,6 +21,7 @@ mod sink; mod source; pub use cdc::{PostgresSourceCdcFixture, PostgresSourceCdcSlowPollFixture}; +pub(crate) use container::PostgresContainer; pub use container::{PostgresOps, PostgresSourceOps}; pub use sink::{ POSTGRES_LARGE_BATCH_SIZE, PostgresSinkByteaFixture, PostgresSinkFixture, diff --git a/core/integration/tests/connectors/jdbc/config_postgres.toml b/core/integration/tests/connectors/jdbc/config_postgres.toml new file mode 100644 index 0000000000..19097f5988 --- /dev/null +++ b/core/integration/tests/connectors/jdbc/config_postgres.toml @@ -0,0 +1,22 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# JDBC Source Connector Runtime Configuration for Postgres + +[connectors] +config_type = "local" +config_dir = "tests/connectors/jdbc/connectors_config_postgres" diff --git a/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml b/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml new file mode 100644 index 0000000000..0e2e1bad76 --- /dev/null +++ b/core/integration/tests/connectors/jdbc/connectors_config_postgres/jdbc_pg.toml @@ -0,0 +1,48 @@ +# Licensed to the Apache Software Foundation (ASF) under one +# or more contributor license agreements. See the NOTICE file +# distributed with this work for additional information +# regarding copyright ownership. The ASF licenses this file +# to you under the Apache License, Version 2.0 (the +# "License"); you may not use this file except in compliance +# with the License. You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, +# software distributed under the License is distributed on an +# "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +# KIND, either express or implied. See the License for the +# specific language governing permissions and limitations +# under the License. + +# JDBC Source Connector for PostgreSQL + +type = "source" +key = "jdbc_pg" +enabled = true +version = 0 +name = "JDBC PostgreSQL Source" +path = "../../target/debug/libiggy_connector_jdbc_source" + +[[streams]] +stream = "test_stream" +topic = "test_topic" +schema = "json" + +[plugin_config] +# All required fields - values will be overridden by environment variables +# Use placeholder values that match the expected types +jdbc_url = "" +driver_class = "" +driver_jar_path = "" +username = "" +password = "" +query = "" +poll_interval = "1s" +batch_size = 100 +mode = "bulk" +# Used only by incremental-mode tests; ignored in bulk mode. +tracking_column = "id" +initial_offset = "0" +snake_case_columns = false +include_metadata = true diff --git a/core/integration/tests/connectors/jdbc/jdbc_source.rs b/core/integration/tests/connectors/jdbc/jdbc_source.rs new file mode 100644 index 0000000000..8af10f1da8 --- /dev/null +++ b/core/integration/tests/connectors/jdbc/jdbc_source.rs @@ -0,0 +1,355 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +use crate::connectors::fixtures::{ + JdbcBulkFixture, JdbcBulkOverflowFixture, JdbcBulkRowsFixture, JdbcIncrementalFixture, + JdbcLargeResultFixture, JdbcMetadataFixture, JdbcQueryTimeoutFixture, JdbcRecoveryFixture, + JdbcTextCursorFixture, JdbcTieBoundaryFixture, +}; +use iggy::prelude::IggyClient; +use iggy_common::{Consumer, Identifier, MessageClient, PollingStrategy}; +use integration::harness::{TestHarness, seeds}; +use integration::iggy_harness; +use std::collections::BTreeSet; +use std::time::{Duration, Instant}; +use tokio::time::sleep; + +const POLL_TIMEOUT: Duration = Duration::from_secs(30); +const POLL_INTERVAL: Duration = Duration::from_millis(100); +const POLL_BATCH: u32 = 500; + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn bulk_query_produces_message_to_iggy(harness: &TestHarness, _fixture: JdbcBulkFixture) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_bulk", 1).await; + assert!(!messages.is_empty(), "expected a JDBC message"); + + let first = &messages[0]; + assert_eq!( + first.get("operation_type").and_then(|value| value.as_str()), + Some("SELECT") + ); + let data = first.get("data").expect("metadata should contain data"); + assert_eq!(data.get("id").and_then(|value| value.as_i64()), Some(1)); + assert_eq!( + data.get("name").and_then(|value| value.as_str()), + Some("test") + ); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn bulk_query_produces_multiple_rows_to_iggy( + harness: &TestHarness, + _fixture: JdbcBulkRowsFixture, +) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_bulk_rows", 3).await; + assert!(messages.len() >= 3, "expected three JDBC messages"); + + for message in &messages[..3] { + let data = message.get("data").expect("metadata should contain data"); + assert!(data.get("id").is_some()); + assert!(data.get("name").is_some()); + assert!(data.get("active").is_some()); + } + let first = messages[0].get("data").unwrap(); + assert_eq!(first.get("id").and_then(|value| value.as_i64()), Some(1)); + assert_eq!( + first.get("name").and_then(|value| value.as_str()), + Some("alice") + ); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn source_includes_metadata_fields_when_enabled( + harness: &TestHarness, + _fixture: JdbcMetadataFixture, +) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_metadata", 1).await; + let message = messages.first().expect("expected a JDBC message"); + assert!(message.get("timestamp").is_some()); + assert!(message.get("operation_type").is_some()); + assert!(message.get("data").is_some()); + assert!(message.get("table_name").is_some()); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn incremental_mode_advances_offset_across_polls( + harness: &TestHarness, + fixture: JdbcIncrementalFixture, +) { + let client = harness.root_client().await.expect("root client"); + let consumer = consumer("jdbc_incremental"); + let (first_ids, first_received) = + poll_until_ids_seen(&client, &consumer, &[1, 2, 3], POLL_TIMEOUT).await; + assert_eq!( + first_ids, + vec![1, 2, 3], + "expected ids 1,2,3, got {first_ids:?} from {first_received} messages" + ); + + let pool = fixture.create_pool().await.expect("PostgreSQL pool"); + sqlx::query("INSERT INTO inc_test (id, name) VALUES (4, 'd'), (5, 'e')") + .execute(&pool) + .await + .expect("insert additional rows"); + pool.close().await; + + let (second_ids, second_received) = + poll_until_ids_seen(&client, &consumer, &[4, 5], POLL_TIMEOUT).await; + assert_eq!( + second_ids, + vec![4, 5], + "expected only ids 4,5, got {second_ids:?} from {second_received} messages" + ); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn incremental_text_offset_with_backslash_preserves_cursor_boundary( + harness: &TestHarness, + _fixture: JdbcTextCursorFixture, +) { + let client = harness.root_client().await.expect("root client"); + let (ids, received) = + poll_until_ids_seen(&client, &consumer("jdbc_text_cursor"), &[2], POLL_TIMEOUT).await; + assert_eq!( + ids, + vec![2], + "expected only the row after the exact cursor, got {ids:?} from {received} messages" + ); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn large_result_set_streams_without_crashing( + harness: &TestHarness, + _fixture: JdbcLargeResultFixture, +) { + let client = harness.root_client().await.expect("root client"); + let messages = poll_json_messages(&client, "jdbc_large_result", 150).await; + assert!(messages.len() >= 150, "expected at least 150 messages"); + for message in &messages[..150] { + let data = message.get("data").expect("metadata should contain data"); + assert!(data.get("id").and_then(|value| value.as_i64()).is_some()); + assert!(data.get("name").and_then(|value| value.as_str()).is_some()); + } +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn source_recovers_after_repeated_query_errors( + harness: &TestHarness, + fixture: JdbcRecoveryFixture, +) { + assert_source_running(harness).await; + sleep(Duration::from_millis(400)).await; + + let pool = fixture.create_pool().await.expect("PostgreSQL pool"); + sqlx::query("CREATE TABLE recover_test (id INT PRIMARY KEY, name TEXT)") + .execute(&pool) + .await + .expect("create recovery table"); + sqlx::query("INSERT INTO recover_test (id, name) VALUES (1, 'a'), (2, 'b')") + .execute(&pool) + .await + .expect("insert recovery rows"); + pool.close().await; + + let client = harness.root_client().await.expect("root client"); + let (ids, received) = + poll_until_ids_seen(&client, &consumer("jdbc_recovery"), &[1, 2], POLL_TIMEOUT).await; + assert_eq!( + ids, + vec![1, 2], + "source did not recover: got {ids:?} from {received} messages" + ); +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn bulk_result_larger_than_batch_size_fails_closed( + harness: &TestHarness, + _fixture: JdbcBulkOverflowFixture, +) { + assert_source_running(harness).await; + assert_no_messages_for(harness, "jdbc_bulk_overflow", Duration::from_millis(600)).await; +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn incremental_tie_at_batch_boundary_fails_closed( + harness: &TestHarness, + _fixture: JdbcTieBoundaryFixture, +) { + assert_source_running(harness).await; + assert_no_messages_for(harness, "jdbc_tie_boundary", Duration::from_millis(600)).await; +} + +#[iggy_harness( + cluster_nodes = 1, + server(connectors_runtime(config_path = "tests/connectors/jdbc/config_postgres.toml")), + seed = seeds::connector_stream +)] +async fn query_timeout_cancels_slow_statement( + harness: &TestHarness, + _fixture: JdbcQueryTimeoutFixture, +) { + assert_source_running(harness).await; + // Without Statement.setQueryTimeout the two-second query emits a row in + // this window. A one-second timeout must cancel every retry before emission. + assert_no_messages_for(harness, "jdbc_query_timeout", Duration::from_secs(3)).await; +} + +async fn poll_json_messages( + client: &IggyClient, + consumer_name: &str, + expected_count: usize, +) -> Vec { + let deadline = Instant::now() + POLL_TIMEOUT; + let consumer = consumer(consumer_name); + let mut received = Vec::new(); + loop { + let polled = poll(client, &consumer).await; + received.extend( + polled + .messages + .iter() + .filter_map(|message| serde_json::from_slice(&message.payload).ok()), + ); + if received.len() >= expected_count || Instant::now() >= deadline { + return received; + } + sleep(POLL_INTERVAL).await; + } +} + +async fn poll_until_ids_seen( + client: &IggyClient, + consumer: &Consumer, + expected: &[i64], + timeout: Duration, +) -> (Vec, usize) { + let deadline = Instant::now() + timeout; + let mut seen = BTreeSet::new(); + let mut received = 0; + loop { + let polled = poll(client, consumer).await; + for message in &polled.messages { + received += 1; + if let Ok(value) = serde_json::from_slice::(&message.payload) + && let Some(id) = value + .get("data") + .and_then(|data| data.get("id")) + .and_then(|id| id.as_i64()) + { + seen.insert(id); + } + } + if expected.iter().all(|id| seen.contains(id)) || Instant::now() >= deadline { + return (seen.into_iter().collect(), received); + } + sleep(POLL_INTERVAL).await; + } +} + +async fn assert_no_messages_for(harness: &TestHarness, consumer_name: &str, duration: Duration) { + let client = harness.root_client().await.expect("root client"); + let consumer = consumer(consumer_name); + let deadline = Instant::now() + duration; + loop { + let polled = poll(&client, &consumer).await; + assert!( + polled.messages.is_empty(), + "expected the source to fail closed, got {} messages", + polled.messages.len() + ); + if Instant::now() >= deadline { + return; + } + sleep(POLL_INTERVAL).await; + } +} + +async fn poll(client: &IggyClient, consumer: &Consumer) -> iggy_common::PolledMessages { + let stream: Identifier = seeds::names::STREAM.try_into().unwrap(); + let topic: Identifier = seeds::names::TOPIC.try_into().unwrap(); + client + .poll_messages( + &stream, + &topic, + None, + consumer, + &PollingStrategy::next(), + POLL_BATCH, + true, + ) + .await + .expect("poll JDBC messages") +} + +fn consumer(name: &str) -> Consumer { + Consumer::new(name.try_into().expect("valid consumer name")) +} + +async fn assert_source_running(harness: &TestHarness) { + let api_url = harness + .connectors_runtime() + .expect("connectors runtime") + .http_url(); + let sources: serde_json::Value = reqwest::get(format!("{api_url}/sources")) + .await + .expect("query source status") + .error_for_status() + .expect("source status response") + .json() + .await + .expect("deserialize source status"); + assert_eq!(sources[0]["status"], "running", "source status: {sources}"); +} diff --git a/core/integration/tests/connectors/jdbc/mod.rs b/core/integration/tests/connectors/jdbc/mod.rs new file mode 100644 index 0000000000..0c6129cce8 --- /dev/null +++ b/core/integration/tests/connectors/jdbc/mod.rs @@ -0,0 +1,19 @@ +// Licensed to the Apache Software Foundation (ASF) under one +// or more contributor license agreements. See the NOTICE file +// distributed with this work for additional information +// regarding copyright ownership. The ASF licenses this file +// to you under the Apache License, Version 2.0 (the +// "License"); you may not use this file except in compliance +// with the License. You may obtain a copy of the License at +// +// http://www.apache.org/licenses/LICENSE-2.0 +// +// Unless required by applicable law or agreed to in writing, +// software distributed under the License is distributed on an +// "AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY +// KIND, either express or implied. See the License for the +// specific language governing permissions and limitations +// under the License. + +// JDBC connector tests, exercised against PostgreSQL over the JDBC driver. +mod jdbc_source; diff --git a/core/integration/tests/connectors/mod.rs b/core/integration/tests/connectors/mod.rs index 54dd1437b9..7975015065 100644 --- a/core/integration/tests/connectors/mod.rs +++ b/core/integration/tests/connectors/mod.rs @@ -25,6 +25,7 @@ mod http; mod http_config_provider; mod iceberg; mod influxdb; +mod jdbc; mod meilisearch; mod mongodb; mod postgres;