diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index 152b39e..9fe8b12 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -91,6 +91,56 @@ jobs: - run: python3 tests/databases.py mariadb cockroachdb clickhouse trino - if: always() run: docker compose -f tests/compose.yaml down -v + jdbc-engines: + runs-on: ubuntu-24.04 + timeout-minutes: 90 + steps: + - uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 + - uses: dtolnay/rust-toolchain@2c7215f132e9ebf062739d9130488b56d53c060c + with: + toolchain: 1.95.0 + - uses: Swatinem/rust-cache@e18b497796c12c097a38f9edb9d0641fb99eee32 + - uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 + with: + node-version: "22" + cache: npm + cache-dependency-path: ui/package-lock.json + - uses: actions/setup-java@c1e323688fd81a25caa38c78aa6df2d33d3e20d9 + with: + distribution: temurin + java-version: "17" + cache: maven + cache-dependency-path: java/jdbc/pom.xml + - run: npm --prefix ui ci + - run: npm --prefix ui run build + - run: cargo build --workspace --locked + - run: mvn -B -f java/jdbc/pom.xml verify + # Presto and Hive start in seconds; each fixture stages its driver with jdbc-fixture.sh. + - run: docker compose -f tests/compose.yaml up -d --wait --wait-timeout 300 presto hive + - run: bash scripts/jdbc-fixture.sh presto + - run: bash scripts/jdbc-fixture.sh hive + - run: python3 tests/databases.py presto hive + - if: always() + run: docker compose -f tests/compose.yaml down -v + # Kylin loads a sample cube on first boot, so it needs a longer budget on its own. + - run: docker compose -f tests/compose.yaml up -d --wait --wait-timeout 900 kylin + - run: bash scripts/jdbc-fixture.sh kylin + - run: python3 tests/databases.py kylin + - if: always() + run: docker compose -f tests/compose.yaml down -v + # Db2 configures an instance and restarts it before it accepts the first connection. + - run: docker compose -f tests/compose.yaml up -d --wait --wait-timeout 900 db2 + - run: bash scripts/jdbc-fixture.sh db2 + - run: python3 tests/databases.py db2 + # The release path, where the CLI resolves the JRE and the JDBC runner from the published + # manifest and loads the driver the user installed, is checked by scripts/release-driver-check.sh + # against a release that already carries the engine; the fixtures above run it from the + # development directory instead. + - if: always() + run: docker compose -f tests/compose.yaml down -v + # XuguDB and Informix need credentials or host entries their images do not publish, and SUNDB + # and GBase 8s need a licensed installation and a vendor driver. tests/databases.py documents + # the environment variable each one reads, and the README shows how to run them locally. wire-databases: runs-on: ubuntu-24.04 timeout-minutes: 40 diff --git a/README.md b/README.md index a76d80f..e44f2ed 100644 --- a/README.md +++ b/README.md @@ -1,6 +1,6 @@ # SQLX -Connect to MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng, KingbaseES, Redis and MongoDB, or open a local SQLite, DuckDB or H2 file, from your terminal or from your agent. Connections are saved encrypted, one invocation runs one or more statements or commands, and the results come back complete and structured. +Connect to MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, Presto, StarRocks, Apache Doris, TDengine, Dameng, KingbaseES, Apache Kylin, XuguDB, IBM Db2, IBM Informix, SUNDB, GBase 8s, Redis and MongoDB, or open a local SQLite, DuckDB or H2 file, from your terminal or from your agent. Connections are saved encrypted, one invocation runs one or more statements or commands, and the results come back complete and structured. ## Quick start @@ -170,6 +170,14 @@ sqlx sql execute --datasource dev --command "SELECT current_database()" --comman | SQLite | `sqlite`, `sqlite3` | native worker | `--path ` (or `--database`) opens or creates a local file; no host, port or credentials; `--property mode=ro` and `--property busy_timeout=` | | DuckDB | `duckdb` | native worker | `--path ` opens or creates a local file, `:memory:` keeps one for the invocation; `--property read_only=true` and `--property threads=` | | H2 | `h2` | JDBC worker | `--path ` opens a local file, or `--host` and `--port` (9092 by default) reach a TCP server | +| Presto | `presto`, `prestodb` | JDBC worker | port 8080 by default; `--database [.]` is required; `--username` is required and a password is only sent over TLS | +| Hive | `hive` | JDBC worker | port 10000 by default; `--database` selects the Hive database, `default` when unset | +| Apache Kylin | `kylin` | JDBC worker | port 7070 by default; `--database` carries the Kylin project, and the default account is `ADMIN`/`KYLIN` | +| XuguDB | `xugu`, `xugudb` | JDBC worker | port 5138 by default; `SYSTEM` is the system database | +| IBM Db2 | `db2`, `ibmdb2` | JDBC worker | port 50000 by default; needs a driver you provide, see [Drivers you provide](#drivers-you-provide) | +| IBM Informix | `informix`, `ifx` | JDBC worker | port 9088 by default; `--service ` is required and the driver must be provided | +| SUNDB | `sundb` | JDBC worker | port 22581 by default; runs the Goldilocks engine and needs the driver from the vendor image | +| GBase 8s | `gbase8s`, `gbasedbt` | JDBC worker | port 9088 by default; `--service ` is required and the driver must be provided | `--id` and `--datasource` accept a stable datasource UUID or its unique name. @@ -264,7 +272,8 @@ Each `connection` is the same object `--connection-stdin` accepts. `merge` (the | Remove a saved connection | `sqlx datasource remove --id dev` | | Test connectivity | `sqlx datasource test --id dev` | | Execute SQL | `sqlx sql execute --datasource dev --command "SELECT 1" --command "SELECT 2"` | -| Download workers, the JDBC runtime and the UI ahead of time | `sqlx prefetch mysql ui` (`mariadb`, `tidb`, `greatsql`, `oceanbase`, `starrocks`, `doris`, `postgres`, `cockroachdb`, `yugabytedb`, `opengauss`, `oracle`, `sqlserver`, `clickhouse`, `trino`, `tdengine`, `dameng`, `kingbase`, `redis`, `mongodb`, `sqlite`, `duckdb`, `h2`, `skill` or `all`) | +| Download workers, the JDBC runtime and the UI ahead of time | `sqlx prefetch mysql ui` (`mariadb`, `tidb`, `greatsql`, `oceanbase`, `starrocks`, `doris`, `postgres`, `cockroachdb`, `yugabytedb`, `opengauss`, `oracle`, `sqlserver`, `clickhouse`, `trino`, `presto`, `hive`, `kylin`, `xugu`, `db2`, `informix`, `sundb`, `gbase8s`, `tdengine`, `dameng`, `kingbase`, `redis`, `mongodb`, `sqlite`, `duckdb`, `h2`, `skill` or `all`; the four engines whose driver you provide fetch the shared runtime only) | +| Install a driver the release cannot ship | `sqlx driver add --type db2 --jar `, `sqlx driver list`, `sqlx driver remove --type db2` | | Execute and open a result page | `sqlx sql execute --datasource dev --command "SELECT 1" --view` | | Read a stored result | `sqlx results list`, `sqlx results rows --id --offset 100 --limit 50` | | Show or change settings | `sqlx setting list`, `sqlx setting set preview-rows 20`, `sqlx setting set results-dir ~/sqlx-results` | @@ -324,13 +333,13 @@ To build your own interface, see the [UI plugin guide](docs/ui-plugins.md), the User data lives in `~/.sqlx/`; use `--data-dir` or `SQLX_DATA_DIR` for another location. Settings are stored in `~/.sqlx/settings.json` and managed with `sqlx setting list|get|set|unset`; `SQLX_PREVIEW_ROWS`, `SQLX_RESULTS_DIR`, `SQLX_RESULTS_RETENTION_HOURS` and `SQLX_RESULT_MODE` override the file for one environment, and a command line flag overrides both. Stored results live in the result directory described above, keep 24 hours by default and stay under 1 GiB in total; older results are removed before the next command runs, and page results from `--view` are managed by the local service. Saved connections use AES-256-GCM with an independently generated local key: back up the key together with the encrypted data, because losing the key prevents decryption. Device identity is generated locally and this version uploads no device information. -The main executable contains no database drivers; each database's worker is downloaded on first use. MySQL, MariaDB, TiDB, GreatSQL, OceanBase, StarRocks and Apache Doris share the MySQL worker, Redis, MongoDB, SQLite and DuckDB run in their own native workers, PostgreSQL, CockroachDB and YugabyteDB share the PostgreSQL worker, and Oracle, SQL Server, ClickHouse, Trino, TDengine, openGauss, Dameng, KingbaseES and H2 use the JDBC worker (the [database table](#create-a-connection) lists which worker serves which database). Downloaded resources come from the fixed release manifest of the running CLI version and are verified before use; `--manifest ` selects another manifest or a local test server. +The main executable contains no database drivers; each database's worker is downloaded on first use. MySQL, MariaDB, TiDB, GreatSQL, OceanBase, StarRocks and Apache Doris share the MySQL worker, Redis, MongoDB, SQLite and DuckDB run in their own native workers, PostgreSQL, CockroachDB and YugabyteDB share the PostgreSQL worker, and Oracle, SQL Server, ClickHouse, Trino, Presto, TDengine, openGauss, Dameng, KingbaseES, H2, Hive, Apache Kylin, XuguDB, IBM Db2, IBM Informix, SUNDB and GBase 8s use the JDBC worker (the [database table](#create-a-connection) lists which worker serves which database). Downloaded resources come from the fixed release manifest of the running CLI version and are verified before use; `--manifest ` selects another manifest or a local test server. Downloads happen on first use and are cached afterwards. Each one prints `Downloading …` with speed and estimated time, and a final `Downloaded … in 12.3s (390 KB/s)` line on stderr; the progress line is refreshed only when stderr is a terminal, so piped JSON stays clean. An interrupted transfer is retried up to three times, and rerunning a failed command reuses every component that is already installed. To avoid waiting inside the first query or page: ```sh sqlx prefetch mysql ui # MySQL worker and the local browser UI -sqlx prefetch all # adds the PostgreSQL, CockroachDB, YugabyteDB, openGauss, MariaDB, TiDB, GreatSQL, OceanBase, StarRocks, Doris, Oracle, SQL Server, ClickHouse, Trino, TDengine, Dameng, KingbaseES, Redis, MongoDB, SQLite, DuckDB and H2 components, the JDBC runtime and the JRE +sqlx prefetch all # adds the PostgreSQL, CockroachDB, YugabyteDB, openGauss, MariaDB, TiDB, GreatSQL, OceanBase, StarRocks, Doris, Oracle, SQL Server, ClickHouse, Trino, Presto, Hive, Kylin, XuguDB, TDengine, Dameng, KingbaseES, Redis, MongoDB, SQLite, DuckDB and H2 components, the JDBC runtime and the JRE ``` The [database references](skills/sqlx/references/) explain each SQL operation's purpose, parameters, result and official documentation link. @@ -438,10 +447,60 @@ docker compose -f tests/compose.yaml down -v # Dameng and KingbaseES have no public image; point the fixtures at a local instance # and export SQLX_TEST_DAMENG_PASSWORD or SQLX_TEST_KINGBASE_PASSWORD when they differ python3 tests/databases.py dameng kingbase +docker compose -f tests/compose.yaml up -d --wait presto hive +bash scripts/jdbc-fixture.sh presto +bash scripts/jdbc-fixture.sh hive +python3 tests/databases.py presto hive +docker compose -f tests/compose.yaml down -v +# Kylin, XuguDB, Db2 and Informix take turns because each needs several gigabytes +docker compose -f tests/compose.yaml up -d --wait kylin +bash scripts/jdbc-fixture.sh kylin +python3 tests/databases.py kylin +docker compose -f tests/compose.yaml down -v +# GBase 8s takes the vendor driver and its own instance; SUNDB needs a licensed installation +# and XuguDB a trial image whose password is known. tests/databases.py lists every variable. +export SQLX_TEST_GBASE8S_DRIVER=~/gbasedbt-jdbc.jar +python3 tests/databases.py gbase8s ``` For local native workers, set `SQLX_WORKER_DIR` to the absolute `target/debug` directory. For JDBC development, that directory also contains `sqlx-jdbc.jar` and `ojdbc.jar` or `mssql-jdbc.jar`; `SQLX_JAVA_BIN` can select Java 17 explicitly. These overrides are for development, not prerequisites for release users. The fixture scripts use dedicated test containers and test-only credentials. +## Drivers you provide + +Most JDBC drivers ship with the release and are downloaded on first use. Vendors that do not allow +their driver to be redistributed are not packaged: IBM Db2, IBM Informix, SUNDB and GBase 8s connect +with a driver you install once from the vendor. + +```sh +sqlx driver add --type db2 --jar ~/Downloads/jcc-12.1.0.0.jar +sqlx driver list +sqlx driver remove --type db2 +``` + +`sqlx driver add` copies the jar into `/drivers//`, checks that it really carries the +engine's driver class, and every later command loads it from there. A jar you provide also wins over a +released component, which is how a newer vendor driver is used before the release catches up. Running a +statement for one of these engines without a driver explains the exact command to run: + +```sh +sqlx sql execute --datasource --command "SELECT 1 FROM SYSIBM.SYSDUMMY1" +# SQLX does not redistribute the db2 driver; run `sqlx driver add --type db2 --jar ` with the vendor driver jar first +``` + +`scripts/release-driver-check.sh ` runs the same +path a release user takes — install the jar, then execute with no development overrides, so the CLI +resolves the JRE and the JDBC runner from the published manifest. It needs a release whose JDBC runner +already knows the engine, so run it against a published version rather than from a feature branch. + +Where the drivers come from: + +| Engine | File to provide | Driver class | +| --- | --- | --- | +| IBM Db2 | `jcc-.jar` from IBM or Maven Central (`com.ibm.db2:jcc`) | `com.ibm.db2.jcc.DB2Driver` | +| IBM Informix | the Informix JDBC driver, 4.50 line (`com.ibm.informix:jdbc`); the 15.x line fails against an Informix 14.10 server | `com.informix.jdbc.IfxDriver` | +| SUNDB | `goldilocks8.jar` from the vendor image at `/goldilocks_home/lib/` | `sunje.goldilocks.jdbc.GoldilocksDriver` | +| GBase 8s | the vendor's `ifxjdbc.jar`; a wrapper jar that contains it must be unpacked first | `com.gbasedbt.jdbc.IfxDriver` | + ## Releases and documentation - Packages and checksums: [GitHub Releases](https://github.com/OtterMind/sqlx/releases) diff --git a/crates/cli/src/drivers.rs b/crates/cli/src/drivers.rs new file mode 100644 index 0000000..89c280c --- /dev/null +++ b/crates/cli/src/drivers.rs @@ -0,0 +1,124 @@ +//! JDBC drivers the user provides for engines SQLX cannot redistribute. +//! +//! A bundled engine downloads its driver from the release. Some vendors do not allow their driver to +//! be redistributed, so nothing for those engines is published: the user copies the vendor jar into +//! the engine's driver directory and every later command loads it exactly like a bundled one. +use anyhow::{bail, Context, Result}; +use std::{ + fs, + path::{Path, PathBuf}, +}; + +/// Directory holding the jars a user provided for one engine component. +/// +/// It sits next to the released components (`/drivers////`), and +/// a jar here always wins over a released one. +pub fn directory(root: &Path, component: &str) -> PathBuf { + root.join("drivers").join(component) +} + +/// The jars a user provided for one engine, sorted; an absent directory reports none. +pub fn provided(root: &Path, component: &str) -> Result> { + let directory = directory(root, component); + if !directory.is_dir() { + return Ok(Vec::new()); + } + let mut jars: Vec = fs::read_dir(&directory) + .with_context(|| format!("cannot read {}", directory.display()))? + .filter_map(|entry| entry.ok().map(|entry| entry.path())) + .filter(|path| path.is_file() && is_jar(path)) + .collect(); + jars.sort(); + Ok(jars) +} + +/// Whether a file is a jar by its name; the extension decides for the driver directory and for the +/// files a user passes, so a jar named with another extension is rejected instead of stored and +/// silently ignored later. +fn is_jar(path: &Path) -> bool { + path.extension() + .is_some_and(|extension| extension.eq_ignore_ascii_case("jar")) +} + +/// Whether a jar carries the driver class the JDBC worker loads. +fn carries(jar: &Path, class: &str) -> Result { + let file = fs::File::open(jar) + .with_context(|| format!("cannot read the driver jar {}", jar.display()))?; + let mut archive = zip::ZipArchive::new(file) + .with_context(|| format!("{} is not a jar archive", jar.display()))?; + let entry = format!("{}.class", class.replace('.', "/")); + let carried = archive.by_name(&entry).is_ok(); + Ok(carried) +} + +/// Copy the given jars into the engine directory, after checking the driver is among them. +/// +/// One of the jars must carry the driver class, and the rest are stored as its dependencies; a set +/// that carries no driver class is rejected instead of being stored, because it would fail later, +/// inside the worker, with a far less useful message. +pub fn install(root: &Path, component: &str, class: &str, jars: &[PathBuf]) -> Result> { + if jars.is_empty() { + bail!("at least one --jar is required"); + } + let directory = directory(root, component); + let mut carries_driver = false; + for jar in jars { + if !jar.is_file() { + bail!("the driver jar {} does not exist", jar.display()); + } + if !is_jar(jar) { + bail!( + "{} is not named as a jar; give it a .jar name so the worker loads it", + jar.display() + ); + } + carries_driver |= carries(jar, class)?; + } + if !carries_driver { + bail!( + "none of the given jars contains {class}; pass the vendor's JDBC driver jar, and its \ + dependencies alongside it when the driver needs them. GBase 8s ships a wrapper jar: \ + unpack its inner ifxjdbc.jar and pass that file" + ); + } + fs::create_dir_all(&directory) + .with_context(|| format!("cannot create {}", directory.display()))?; + let mut stored = Vec::new(); + for jar in jars { + let name = jar + .file_name() + .context("the driver jar has no file name")? + .to_string_lossy() + .into_owned(); + let target = directory.join(&name); + if jar.canonicalize().ok().as_deref() != target.canonicalize().ok().as_deref() { + fs::copy(jar, &target) + .with_context(|| format!("cannot copy the driver into {}", target.display()))?; + } + stored.push(name); + } + stored.sort(); + Ok(stored) +} + +/// Remove every jar a user provided for one engine; the released components stay untouched. +pub fn remove(root: &Path, component: &str) -> Result> { + let mut removed = Vec::new(); + for jar in provided(root, component)? { + if let Some(name) = jar.file_name() { + removed.push(name.to_string_lossy().into_owned()); + } + fs::remove_file(&jar).with_context(|| format!("cannot remove {}", jar.display()))?; + } + removed.sort(); + Ok(removed) +} + +/// Whether the released driver component is installed. +/// +/// Driver components are platform-independent, so the components manager installs them under `any`. +pub fn released(root: &Path, component: &str) -> bool { + fs::read_dir(root.join("drivers").join(component).join("any")) + .map(|entries| entries.flatten().any(|entry| entry.path().is_dir())) + .unwrap_or(false) +} diff --git a/crates/cli/src/execution.rs b/crates/cli/src/execution.rs index b930237..535b8b2 100644 --- a/crates/cli/src/execution.rs +++ b/crates/cli/src/execution.rs @@ -1,6 +1,6 @@ use crate::{ components::{platform, Components}, - output, settings, + drivers, output, settings, storage::Datasource, }; use anyhow::{bail, Context, Result}; @@ -21,7 +21,7 @@ pub struct PreparedExecution { } /// Wire-compatible engines reuse the native workers. -fn native_worker(kind: Database) -> Option<&'static str> { +pub fn native_worker(kind: Database) -> Option<&'static str> { match kind { Database::Mysql | Database::Mariadb @@ -38,51 +38,109 @@ fn native_worker(kind: Database) -> Option<&'static str> { _ => None, } } -struct JdbcDriver { - component: &'static str, - jars: &'static [&'static str], +/// One engine's JDBC driver: the release component, its jar names, the driver class and whether +/// SQLX may ship it at all. +pub struct JdbcDriver { + pub component: &'static str, + pub jars: &'static [&'static str], + /// Driver class the worker loads, and the entry a user-provided jar must carry. + pub class: &'static str, + /// A bundled driver arrives with the release; the rest must be provided by the user. + pub bundled: bool, +} + +/// Whether the engine runs through the JDBC worker at all. +pub fn uses_jdbc_driver(kind: Database) -> bool { + native_worker(kind).is_none() } -fn jdbc_driver(kind: Database) -> Result { +pub fn jdbc_driver(kind: Database) -> Result { + let driver = |component, jars, class| JdbcDriver { + component, + jars, + class, + bundled: true, + }; Ok(match kind { - Database::Oracle => JdbcDriver { - component: "oracle", - jars: &["ojdbc.jar"], - }, - Database::Sqlserver => JdbcDriver { - component: "sqlserver", - jars: &["mssql-jdbc.jar"], - }, - Database::Clickhouse => JdbcDriver { - component: "clickhouse", - jars: &["clickhouse-jdbc.jar", "slf4j-api.jar", "slf4j-nop.jar"], - }, - Database::Trino => JdbcDriver { - component: "trino", - jars: &["trino-jdbc.jar"], - }, - Database::Opengauss => JdbcDriver { - component: "opengauss", - jars: &["opengauss-jdbc.jar"], - }, - Database::Dameng => JdbcDriver { - component: "dameng", - jars: &["dm-jdbc.jar"], - }, - Database::Kingbase => JdbcDriver { - component: "kingbase", - jars: &["kingbase8-jdbc.jar"], - }, - Database::Tdengine => JdbcDriver { - component: "tdengine", - jars: &["taos-jdbcdriver.jar", "slf4j-nop.jar"], - }, - Database::H2 => JdbcDriver { - component: "h2", - jars: &["h2.jar"], - }, + Database::Oracle => driver("oracle", &["ojdbc.jar"], "oracle.jdbc.OracleDriver"), + Database::Sqlserver => driver( + "sqlserver", + &["mssql-jdbc.jar"], + "com.microsoft.sqlserver.jdbc.SQLServerDriver", + ), + Database::Clickhouse => driver( + "clickhouse", + &["clickhouse-jdbc.jar", "slf4j-api.jar", "slf4j-nop.jar"], + "com.clickhouse.jdbc.ClickHouseDriver", + ), + Database::Trino => driver("trino", &["trino-jdbc.jar"], "io.trino.jdbc.TrinoDriver"), + Database::Opengauss => driver("opengauss", &["opengauss-jdbc.jar"], "org.opengauss.Driver"), + Database::Dameng => driver("dameng", &["dm-jdbc.jar"], "dm.jdbc.driver.DmDriver"), + Database::Kingbase => driver("kingbase", &["kingbase8-jdbc.jar"], "com.kingbase8.Driver"), + Database::Tdengine => driver( + "tdengine", + &["taos-jdbcdriver.jar", "slf4j-nop.jar"], + "com.taosdata.jdbc.ws.WebSocketDriver", + ), + Database::H2 => driver("h2", &["h2.jar"], "org.h2.Driver"), + Database::Presto => driver( + "presto", + &["presto-jdbc.jar"], + "com.facebook.presto.jdbc.PrestoDriver", + ), + Database::Hive => driver( + "hive", + &["hive-jdbc.jar", "slf4j-nop.jar"], + "org.apache.hive.jdbc.HiveDriver", + ), + // Kylin's driver uses JAXB, which the JDK dropped in Java 11, and an slf4j 1.7 binding. + Database::Kylin => driver( + "kylin", + &[ + "kylin-jdbc.jar", + "jakarta.xml.bind-api.jar", + "jaxb-runtime.jar", + "istack-commons-runtime.jar", + "jakarta.activation-api.jar", + "txw2.jar", + "slf4j-nop.jar", + ], + "org.apache.kylin.jdbc.Driver", + ), + Database::Xugu => driver("xugu", &["xugu-jdbc.jar"], "com.xugu.cloudjdbc.Driver"), + // The vendors below do not allow redistribution, so SQLX publishes no driver for them and + // the user drops the vendor jar into `sqlx driver add`. + Database::Db2 => provided("db2", &["db2-jcc.jar"], "com.ibm.db2.jcc.DB2Driver"), + Database::Informix => provided( + "informix", + &["informix-jdbc.jar"], + "com.informix.jdbc.IfxDriver", + ), + Database::Sundb => provided( + "sundb", + &["goldilocks8.jar"], + "sunje.goldilocks.jdbc.GoldilocksDriver", + ), + Database::Gbase8s => provided( + "gbase8s", + &["gbasedbt-jdbc.jar"], + "com.gbasedbt.jdbc.IfxDriver", + ), other => bail!("{other:?} is not a JDBC database"), }) } +/// A driver the release does not carry, because its vendor does not allow redistribution. +fn provided( + component: &'static str, + jars: &'static [&'static str], + class: &'static str, +) -> JdbcDriver { + JdbcDriver { + component, + jars, + class, + bundled: false, + } +} pub fn prepare( root: PathBuf, manifest: String, @@ -114,42 +172,76 @@ pub fn prepare( "-jar".to_owned(), dir.join("sqlx-jdbc.jar").to_string_lossy().into_owned(), ]); + // A driver installed with `sqlx driver add` is loaded here too, so a development run and a + // release run use the same classpath. + let mut jars = drivers::provided(&root, driver.component)?; for jar in driver.jars { let path = dir .join(jar) .canonicalize() .with_context(|| format!("development JDBC driver {jar} is missing"))?; + if jars + .iter() + .any(|provided| provided.file_name() == path.file_name()) + { + continue; + } + jars.push(path); + } + if jars.is_empty() { + bail!("no JDBC driver is available for {}", kind.name()); + } + for jar in jars { request .driver_jars - .push(path.to_string_lossy().into_owned()); + .push(jar.canonicalize()?.to_string_lossy().into_owned()); } PathBuf::from(std::env::var_os("SQLX_JAVA_BIN").unwrap_or_else(|| "java".into())) } } else { let manager = Components::new(root, manifest); - let m = manager.manifest(false)?; let platform = platform()?; if let Some(name) = native { + let m = manager.manifest(false)?; manager.ensure(name, &platform, manager.asset(&m, name, &platform)?)? } else { let driver = jdbc_driver(kind)?; + // A driver the user provides is resolved before anything is downloaded: an engine whose + // vendor allows no redistribution must not fetch a JRE only to report that the driver is + // missing. + let provided = drivers::provided(&manager.root, driver.component)?; + if provided.is_empty() && !driver.bundled { + bail!( + "SQLX does not redistribute the {} driver; run `sqlx driver add --type {} --jar ` with the vendor driver jar first", + driver.component, + kind.name() + ); + } + let m = manager.manifest(false)?; let java = manager.ensure("java", &platform, manager.asset(&m, "java", &platform)?)?; let runner = manager.ensure("jdbc", "any", manager.asset(&m, "jdbc", "any")?)?; - let entry = manager.ensure( - driver.component, - "any", - manager.asset(&m, driver.component, "any")?, - )?; args.extend(["-jar".into(), runner.to_string_lossy().into_owned()]); - // A JDBC component can ship more than the driver itself, such as a logging API. - let directory = entry.parent().context("JDBC component has no directory")?; - let mut jars: Vec = fs::read_dir(directory)? - .filter_map(|item| item.ok().map(|item| item.path())) - .filter(|path| path.extension().is_some_and(|extension| extension == "jar")) - .collect(); - jars.sort(); - if jars.is_empty() { - bail!("JDBC component {} contains no jar", driver.component); + // A provided jar is loaded before the released one it replaces, and the rest of the + // component keeps supplying what the driver needs, such as a logging API or JAXB. An + // engine the release does not carry loads only what the user provided. + let mut jars = provided; + if driver.bundled { + let entry = manager.ensure( + driver.component, + "any", + manager.asset(&m, driver.component, "any")?, + )?; + let directory = entry.parent().context("JDBC component has no directory")?; + let mut released: Vec = fs::read_dir(directory)? + .filter_map(|item| item.ok().map(|item| item.path())) + .filter(|path| path.extension().is_some_and(|extension| extension == "jar")) + .filter(|path| !jars.iter().any(|jar| jar.file_name() == path.file_name())) + .collect(); + released.sort(); + jars.extend(released); + if jars.is_empty() { + bail!("JDBC component {} contains no jar", driver.component); + } } for jar in jars { request @@ -159,19 +251,10 @@ pub fn prepare( java } }; - request.driver_class = match kind { - Database::Oracle => "oracle.jdbc.OracleDriver", - Database::Sqlserver => "com.microsoft.sqlserver.jdbc.SQLServerDriver", - Database::Clickhouse => "com.clickhouse.jdbc.ClickHouseDriver", - Database::Trino => "io.trino.jdbc.TrinoDriver", - Database::Tdengine => "com.taosdata.jdbc.ws.WebSocketDriver", - Database::Opengauss => "org.opengauss.Driver", - Database::Dameng => "dm.jdbc.driver.DmDriver", - Database::Kingbase => "com.kingbase8.Driver", - Database::H2 => "org.h2.Driver", - _ => "", - } - .into(); + request.driver_class = match native { + Some(_) => String::new(), + None => jdbc_driver(kind)?.class.to_owned(), + }; Ok(PreparedExecution { binary, args, @@ -373,8 +456,15 @@ impl PreparedExecution { let mut event: Event = match serde_json::from_str(&line) { Ok(e) => e, Err(_) => { - stream_error = Some("worker emitted an invalid protocol event".into()); - break; + // A driver may print to the stream the worker owns -- Kylin's Avatica client does + // -- and that line is not part of the protocol. Report it where diagnostics go and + // keep reading: the events that matter are still well formed, and a missing + // completion stays an error below. + eprintln!( + "sqlx: ignoring unexpected worker output: {}", + sqlx_protocol::redact(&line, &request.connection) + ); + continue; } }; let validation: Result<()> = (|| { diff --git a/crates/cli/src/import.rs b/crates/cli/src/import.rs index 3f9ec46..890331b 100644 --- a/crates/cli/src/import.rs +++ b/crates/cli/src/import.rs @@ -289,7 +289,7 @@ mod tests { {"name":"6f1d0f0e-8c58-4a5f-9d34-6a5a4f3b2c11","connection":connection()}, item("twin"), item("twin"), - {"name":"hive","connection":{"database_type":"hive","host":"h","port":10000}}, + {"name":"unknown","connection":{"database_type":"snowflake","host":"h","port":10000}}, {"name":"nohost","connection":{"database_type":"postgresql","port":5432}}, ])); let (_temp, store) = temp_store(); @@ -304,7 +304,7 @@ mod tests { "invalid_name".to_string() ), ("twin".to_string(), "duplicate_name".to_string()), - ("hive".to_string(), "invalid_connection".to_string()), + ("unknown".to_string(), "invalid_connection".to_string()), ("nohost".to_string(), "invalid_connection".to_string()), ] ); @@ -319,7 +319,7 @@ mod tests { let (_temp, store) = temp_store(); let input = document(json!([ item("ok"), - {"name":"hive","connection":{"database_type":"hive","host":"h","port":1}} + {"name":"unknown","connection":{"database_type":"snowflake","host":"h","port":1}} ])); let error = run(&store, &input, false, true).unwrap_err().to_string(); assert!(error.contains("cannot be imported"), "{error}"); diff --git a/crates/cli/src/lib.rs b/crates/cli/src/lib.rs index 9ef8049..787ad3c 100644 --- a/crates/cli/src/lib.rs +++ b/crates/cli/src/lib.rs @@ -1,4 +1,5 @@ pub mod components; +pub mod drivers; pub mod execution; pub mod mcp; pub mod output; diff --git a/crates/cli/src/main.rs b/crates/cli/src/main.rs index c0af00c..6c82073 100644 --- a/crates/cli/src/main.rs +++ b/crates/cli/src/main.rs @@ -1,6 +1,6 @@ mod import; mod skill; -use sqlx_core::{components, execution, mcp, plugins, prefetch, storage, ui, updates}; +use sqlx_core::{components, drivers, execution, mcp, plugins, prefetch, storage, ui, updates}; use anyhow::{anyhow, bail, Context, Result}; use clap::{Args, Parser, Subcommand}; @@ -44,12 +44,12 @@ enum Commands { Mcp, /// Download database workers, the JDBC runtime and the browser UI before they are needed. Prefetch { - /// Components to download: mysql, mariadb, tidb, greatsql, oceanbase, starrocks, doris, postgres, cockroachdb, yugabytedb, opengauss, oracle, sqlserver, clickhouse, trino, tdengine, dameng, kingbase, redis, mongodb, sqlite, duckdb, h2, ui, skill or all. + /// Components to download: mysql, mariadb, tidb, greatsql, oceanbase, starrocks, doris, postgres, cockroachdb, yugabytedb, opengauss, oracle, sqlserver, clickhouse, trino, tdengine, dameng, kingbase, redis, mongodb, sqlite, duckdb, h2, presto, hive, kylin, xugu, db2, informix, sundb, gbase8s, ui, skill or all. #[arg( value_name = "COMPONENT", required = true, num_args = 1.., - value_parser = ["mysql", "mariadb", "tidb", "greatsql", "oceanbase", "starrocks", "doris", "postgres", "cockroachdb", "yugabytedb", "opengauss", "oracle", "sqlserver", "clickhouse", "trino", "tdengine", "dameng", "kingbase", "redis", "mongodb", "sqlite", "duckdb", "h2", "ui", "skill", "all"] + value_parser = ["mysql", "mariadb", "tidb", "greatsql", "oceanbase", "starrocks", "doris", "postgres", "cockroachdb", "yugabytedb", "opengauss", "oracle", "sqlserver", "clickhouse", "trino", "tdengine", "dameng", "kingbase", "redis", "mongodb", "sqlite", "duckdb", "h2", "presto", "hive", "kylin", "xugu", "db2", "informix", "sundb", "gbase8s", "ui", "skill", "all"] )] components: Vec, }, @@ -85,6 +85,34 @@ enum Commands { #[command(subcommand)] command: Option, }, + /// Provide or inspect the JDBC driver of an engine SQLX does not ship one for. + Driver { + #[command(subcommand)] + command: DriverCommand, + }, +} +#[derive(Subcommand)] +enum DriverCommand { + /// Copy a vendor driver jar into the SQLX driver directory for one engine. + Add { + /// Database type the driver belongs to, such as db2. + #[arg(long = "type", value_name = "DATABASE_TYPE")] + database_type: String, + /// Vendor driver jar; repeat when the driver needs companion jars. + #[arg(long = "jar", value_name = "PATH", required = true, num_args = 1..)] + jars: Vec, + }, + /// Show where each engine's driver comes from. + List { + /// Only report this database type. + #[arg(long = "type", value_name = "DATABASE_TYPE")] + database_type: Option, + }, + /// Remove the jars provided for one engine. + Remove { + #[arg(long = "type", value_name = "DATABASE_TYPE")] + database_type: String, + }, } #[derive(Subcommand)] enum UpdateCommand { @@ -611,6 +639,9 @@ fn run(cli: Cli) -> Result { }; print(value); } + Commands::Driver { command } => { + return driver_command(&root, command); + } Commands::Ui { command } => match command { Some(UiCommand::Plugin { command }) => { { @@ -815,6 +846,79 @@ fn setting_command(root: &Path, command: SettingCommand) -> Result { } Ok(true) } +/// Manage the JDBC driver of an engine whose vendor does not allow redistribution. +fn driver_command(root: &Path, command: DriverCommand) -> Result { + let describe = |kind: Database| -> Result { + let driver = execution::jdbc_driver(kind)?; + let provided = drivers::provided(root, driver.component)?; + let released = drivers::released(root, driver.component); + let source = if !provided.is_empty() { + "provided" + } else if driver.bundled { + "release" + } else { + "missing" + }; + Ok(json!({ + "type": kind.name(), + "component": driver.component, + "driver_class": driver.class, + "bundled": driver.bundled, + "installed": released, + "source": source, + "directory": drivers::directory(root, driver.component).to_string_lossy(), + "jars": provided + .iter() + .filter_map(|jar| jar.file_name().map(|name| name.to_string_lossy().into_owned())) + .collect::>(), + })) + }; + match command { + DriverCommand::Add { + database_type, + jars, + } => { + let kind = database_type + .parse::() + .map_err(|error| anyhow!(error))?; + let driver = execution::jdbc_driver(kind)?; + let stored = drivers::install(root, driver.component, driver.class, &jars)?; + print(json!({ + "type": kind.name(), + "component": driver.component, + "directory": drivers::directory(root, driver.component).to_string_lossy(), + "jars": stored, + "next": format!("sqlx datasource add --type {} ... then run a statement", kind.name()), + })); + } + DriverCommand::List { database_type } => { + let reported = match database_type { + Some(name) => vec![describe( + name.parse::().map_err(|error| anyhow!(error))?, + )?], + None => Database::ALL + .iter() + .filter(|kind| execution::uses_jdbc_driver(**kind)) + .map(|kind| describe(*kind)) + .collect::>>()?, + }; + print(json!({"drivers": reported})); + } + DriverCommand::Remove { database_type } => { + let kind = database_type + .parse::() + .map_err(|error| anyhow!(error))?; + let driver = execution::jdbc_driver(kind)?; + let removed = drivers::remove(root, driver.component)?; + print(json!({ + "type": kind.name(), + "component": driver.component, + "removed": removed, + })); + } + } + Ok(true) +} fn print_line(value: &Value) { if let Err(error) = writeln!(io::stdout(), "{value}") { if error.kind() == io::ErrorKind::BrokenPipe { @@ -911,6 +1015,14 @@ impl ConnectionArgs { // A local engine has no port, and H2 uses its file mode until a host is given. Database::Sqlite | Database::Duckdb => 0, Database::H2 => 9092, + Database::Presto => 8080, + Database::Hive => 10000, + Database::Kylin => 7070, + Database::Xugu => 5138, + Database::Db2 => 50000, + Database::Informix => 9088, + Database::Sundb => 22581, + Database::Gbase8s => 9088, }, database: String::new(), service: String::new(), @@ -987,3 +1099,31 @@ impl ConnectionArgs { Ok(c) } } + +#[cfg(test)] +mod tests { + use super::*; + /// Every component the prefetch command documents must be accepted by the parser: the help text, + /// the MCP tool schema and the clap list are three copies of the same set. + #[test] + fn prefetch_accepts_every_documented_component() { + use clap::CommandFactory; + for component in sqlx_core::prefetch::CHOICES { + assert!( + Cli::try_parse_from(["sqlx", "prefetch", component]).is_ok(), + "prefetch {component} is documented but rejected by the parser" + ); + } + let help = Cli::command() + .find_subcommand_mut("prefetch") + .expect("the prefetch subcommand exists") + .render_long_help() + .to_string(); + for component in sqlx_core::prefetch::CHOICES { + assert!( + help.contains(component), + "the help text does not name {component}" + ); + } + } +} diff --git a/crates/cli/src/prefetch.rs b/crates/cli/src/prefetch.rs index 252fe2f..697f756 100644 --- a/crates/cli/src/prefetch.rs +++ b/crates/cli/src/prefetch.rs @@ -5,7 +5,7 @@ use serde_json::{json, Value}; use std::{path::Path, time::Instant}; /// Components accepted on the command line, in the order `all` downloads them. -pub(crate) const CHOICES: [&str; 26] = [ +pub const CHOICES: &[&str] = &[ "mysql", "mariadb", "tidb", @@ -29,6 +29,14 @@ pub(crate) const CHOICES: [&str; 26] = [ "sqlite", "duckdb", "h2", + "presto", + "hive", + "kylin", + "xugu", + "db2", + "informix", + "sundb", + "gbase8s", "ui", "skill", "all", @@ -47,11 +55,17 @@ fn expand(name: &str, platform: &str) -> Result> { "cockroachdb" | "yugabytedb" => vec![("postgres".to_owned(), platform.to_owned())], // The JDBC databases need the shared runner and the pinned JRE as well. "oracle" | "sqlserver" | "clickhouse" | "trino" | "tdengine" | "opengauss" | "dameng" - | "kingbase" | "h2" => vec![ + | "kingbase" | "h2" | "presto" | "hive" | "kylin" | "xugu" => vec![ ("java".to_owned(), platform.to_owned()), ("jdbc".to_owned(), "any".to_owned()), (name.to_owned(), "any".to_owned()), ], + // These engines connect through the JDBC worker too, but their vendor does not allow the + // driver to be redistributed, so prefetch only prepares the shared runtime. + "db2" | "informix" | "sundb" | "gbase8s" => vec![ + ("java".to_owned(), platform.to_owned()), + ("jdbc".to_owned(), "any".to_owned()), + ], "ui" => vec![ ("ui".to_owned(), platform.to_owned()), ("ui-default".to_owned(), "any".to_owned()), @@ -85,6 +99,10 @@ fn expand(name: &str, platform: &str) -> Result> { "sqlite", "duckdb", "h2", + "presto", + "hive", + "kylin", + "xugu", ] { all.extend(expand(target, platform)?); } @@ -230,6 +248,16 @@ mod tests { ] { assert!(all.iter().any(|(entry, _)| entry == name), "missing {name}"); } + // Every documented component other than `all` must be reachable through `all`, so adding an + // engine to the choices cannot leave it out of the batch download. + for choice in CHOICES.iter().filter(|choice| **choice != "all") { + for entry in expand(choice, "macos-arm64").unwrap() { + assert!( + all.contains(&entry), + "{choice} expands to {entry:?}, which `all` does not download" + ); + } + } let mut unique = all.clone(); unique.sort(); unique.dedup(); diff --git a/crates/cli/tests/cli.rs b/crates/cli/tests/cli.rs index 68d6f1b..b3746a4 100644 --- a/crates/cli/tests/cli.rs +++ b/crates/cli/tests/cli.rs @@ -130,7 +130,7 @@ fn import_merges_documents_reports_skips_and_keeps_credentials_secret() { let root = temp.path().join("data"); let document = json!({"version":1,"mode":"merge","datasources":[ {"name":"imported","connection":connection()}, - {"name":"hive","connection":{"database_type":"hive","host":"h","port":10000}}, + {"name":"unknown","connection":{"database_type":"snowflake","host":"h","port":10000}}, {"name":"","connection":connection()} ]}); let out = call(&root, &["datasource", "import", "--stdin"], Some(&document)); @@ -143,7 +143,7 @@ fn import_merges_documents_reports_skips_and_keeps_credentials_secret() { assert_eq!(value["success"], true); assert_eq!(value["data"]["added"], 1); assert_eq!(value["data"]["updated"], 0); - assert_eq!(value["data"]["skipped"][0]["name"], "hive"); + assert_eq!(value["data"]["skipped"][0]["name"], "unknown"); assert_eq!(value["data"]["skipped"][0]["reason"], "invalid_connection"); assert_eq!(value["data"]["skipped"][1]["reason"], "invalid_name"); // Credentials stay out of the report and off the disk in clear text. @@ -254,3 +254,275 @@ fn documented_import_example_stays_valid() { assert!(std::path::Path::new(path).is_absolute(), "{path}"); assert!(path.ends_with("data/app.db"), "{path}"); } + +/// A driver the release cannot redistribute has to be provided, checked and removed through the CLI. +#[test] +fn provided_drivers_are_validated_installed_and_removed() { + let temp = tempfile::tempdir().unwrap(); + let root = temp.path().join("data"); + let jar = temp.path().join("jcc-12.1.0.0.jar"); + write_jar(&jar, &["com/ibm/db2/jcc/DB2Driver.class"]); + + // One of the jars must carry the driver class the worker loads, not just any archive. + let wrong = temp.path().join("wrong.jar"); + write_jar(&wrong, &["com/example/Other.class"]); + let rejected = call( + &root, + &[ + "driver", + "add", + "--type", + "db2", + "--jar", + wrong.to_str().unwrap(), + ], + None, + ); + assert!(!rejected.status.success()); + assert!( + String::from_utf8_lossy(&rejected.stdout).contains("contains com.ibm.db2.jcc.DB2Driver"), + "{}", + String::from_utf8_lossy(&rejected.stdout) + ); + + let added = call( + &root, + &[ + "driver", + "add", + "--type", + "db2", + "--jar", + jar.to_str().unwrap(), + ], + None, + ); + assert!( + added.status.success(), + "{}", + String::from_utf8_lossy(&added.stderr) + ); + let value: Value = serde_json::from_slice(&added.stdout).unwrap(); + assert_eq!(value["data"]["component"], "db2"); + assert_eq!(value["data"]["jars"][0], "jcc-12.1.0.0.jar"); + assert!(root.join("drivers/db2/jcc-12.1.0.0.jar").is_file()); + + // A provided driver is reported as the source, and an engine served by a native worker is not. + let listed = call(&root, &["driver", "list", "--type", "db2"], None); + let value: Value = serde_json::from_slice(&listed.stdout).unwrap(); + assert_eq!(value["data"]["drivers"][0]["source"], "provided"); + assert_eq!(value["data"]["drivers"][0]["bundled"], false); + assert_eq!( + value["data"]["drivers"][0]["driver_class"], + "com.ibm.db2.jcc.DB2Driver" + ); + let every = call(&root, &["driver", "list"], None); + let value: Value = serde_json::from_slice(&every.stdout).unwrap(); + let engines: Vec<&str> = value["data"]["drivers"] + .as_array() + .unwrap() + .iter() + .map(|driver| driver["type"].as_str().unwrap()) + .collect(); + assert!( + engines.contains(&"presto") && engines.contains(&"gbase8s"), + "{engines:?}" + ); + assert!( + !engines.contains(&"mysql") && !engines.contains(&"redis"), + "{engines:?}" + ); + + assert!( + !root.join("drivers/db2/wrong.jar").exists(), + "a rejected jar must not be installed" + ); + + // A file that is not named as a jar would be stored and then never loaded, so it is refused. + let misnamed = temp.path().join("driver.jar.bak"); + write_jar(&misnamed, &["com/ibm/db2/jcc/DB2Driver.class"]); + let refused = call( + &root, + &[ + "driver", + "add", + "--type", + "db2", + "--jar", + misnamed.to_str().unwrap(), + ], + None, + ); + assert!(!refused.status.success()); + assert!( + String::from_utf8_lossy(&refused.stdout).contains("is not named as a jar"), + "{}", + String::from_utf8_lossy(&refused.stdout) + ); + + // Driver components are platform independent, so an installed one lives under `any`. + std::fs::create_dir_all(root.join("drivers/oracle/any/1.0.0")).unwrap(); + let installed = call(&root, &["driver", "list", "--type", "oracle"], None); + let value: Value = serde_json::from_slice(&installed.stdout).unwrap(); + assert_eq!(value["data"]["drivers"][0]["installed"], true, "{value}"); + assert_eq!(value["data"]["drivers"][0]["source"], "release", "{value}"); + + let removed = call(&root, &["driver", "remove", "--type", "db2"], None); + let value: Value = serde_json::from_slice(&removed.stdout).unwrap(); + assert_eq!(value["data"]["removed"][0], "jcc-12.1.0.0.jar"); + assert!(!root.join("drivers/db2/jcc-12.1.0.0.jar").exists()); + + // A driver that needs dependencies is accepted with them, as long as the driver is among them. + let helper = temp.path().join("helper.jar"); + write_jar(&helper, &["com/example/Helper.class"]); + let with_helper = call( + &root, + &[ + "driver", + "add", + "--type", + "db2", + "--jar", + jar.to_str().unwrap(), + "--jar", + helper.to_str().unwrap(), + ], + None, + ); + assert!( + with_helper.status.success(), + "{}", + String::from_utf8_lossy(&with_helper.stdout) + ); + assert!(root.join("drivers/db2/helper.jar").is_file()); +} +/// An engine served by the JDBC worker that has no driver yet explains how to provide one. +#[test] +fn an_engine_without_a_driver_explains_how_to_provide_it() { + let temp = tempfile::tempdir().unwrap(); + let root = temp.path().join("data"); + // A connection whose driver is missing is rejected before anything is downloaded, so the message + // names the exact command instead of failing inside a worker. + let connection = json!({"database_type":"db2","host":"127.0.0.1","port":50000,"database":"sample","username":"","password":"","tls":"disable"}); + let added = call( + &root, + &[ + "datasource", + "add", + "--name", + "warehouse", + "--connection-stdin", + ], + Some(&connection), + ); + assert!( + added.status.success(), + "{}", + String::from_utf8_lossy(&added.stderr) + ); + let out = call( + &root, + &[ + "sql", + "execute", + "--datasource", + "warehouse", + "--command", + "SELECT 1", + ], + None, + ); + assert!(!out.status.success()); + let message = String::from_utf8_lossy(&out.stdout); + assert!(message.contains("sqlx driver add --type db2"), "{message}"); + assert!( + !root.join("drivers").exists(), + "nothing was downloaded for a missing driver" + ); +} +/// A minimal zip that carries the listed entries, so the driver checks can be exercised offline. +fn write_jar(path: &std::path::Path, entries: &[&str]) { + use std::io::Write as _; + let file = std::fs::File::create(path).unwrap(); + let mut zip = zip::ZipWriter::new(file); + let options = zip::write::SimpleFileOptions::default(); + for entry in entries { + zip.start_file(*entry, options).unwrap(); + zip.write_all(&[0u8; 8]).unwrap(); + } + zip.finish().unwrap(); +} + +/// A worker that prints an unrelated line must not break the protocol: Kylin's Avatica client does. +#[cfg(unix)] +#[test] +fn unrelated_worker_output_is_ignored() { + use std::os::unix::fs::PermissionsExt; + let temp = tempfile::tempdir().unwrap(); + let root = temp.path().join("data"); + let worker_dir = temp.path().join("workers"); + std::fs::create_dir_all(&worker_dir).unwrap(); + let worker = worker_dir.join("sqlx-driver-mysql"); + std::fs::write( + &worker, + concat!( + "#!/bin/sh\n", + // The request arrives on stdin; a worker that never reads it makes the writer see EPIPE, + // which is exactly what the CLI reports with "worker exited without a valid completion". + "cat > /dev/null\n", + "printf '%s\\n' '{\"event\":\"ready\",\"protocol_version\":1}'\n", + "printf '%s\\n' 'Avatica: connection established'\n", + "printf '%s\\n' '{\"event\":\"connected\"}'\n", + "printf '%s\\n' '{\"event\":\"statement_start\",\"index\":0}'\n", + "printf '%s\\n' '{\"event\":\"columns\",\"index\":0,\"result\":0,\"columns\":[{\"name\":\"ok\",\"database_type\":\"INTEGER\",\"encoding\":\"string\"}]}'\n", + "printf '%s\\n' '{\"event\":\"row\",\"index\":0,\"result\":0,\"values\":[\"1\"]}'\n", + "printf '%s\\n' '{\"event\":\"result_end\",\"index\":0,\"result\":0,\"rows\":\"1\",\"affected_rows\":null}'\n", + "printf '%s\\n' '{\"event\":\"statement_end\",\"index\":0}'\n", + "printf '%s\\n' '{\"event\":\"complete\",\"success\":true}'\n", + ), + ) + .unwrap(); + std::fs::set_permissions(&worker, std::fs::Permissions::from_mode(0o755)).unwrap(); + let connection = json!({"database_type":"mysql","host":"127.0.0.1","port":3306,"database":"fixture","username":"u","password":"p","tls":"disable"}); + let workers = worker_dir.to_str().unwrap(); + let added = call( + &root, + &[ + "--worker-dir", + workers, + "datasource", + "add", + "--name", + "chatty", + "--connection-stdin", + ], + Some(&connection), + ); + assert!( + added.status.success(), + "{}", + String::from_utf8_lossy(&added.stderr) + ); + let out = call( + &root, + &[ + "--worker-dir", + workers, + "sql", + "execute", + "--datasource", + "chatty", + "--command", + "SELECT 1 AS ok", + ], + None, + ); + let value: Value = serde_json::from_slice(&out.stdout).unwrap(); + assert_eq!(value["success"], true, "{value}"); + assert_eq!(value["results"][0]["rows"][0][0], "1", "{value}"); + assert!( + String::from_utf8_lossy(&out.stderr).contains("Avatica: connection established"), + "{}", + String::from_utf8_lossy(&out.stderr) + ); +} diff --git a/crates/protocol/src/lib.rs b/crates/protocol/src/lib.rs index 04ae8e3..847e872 100644 --- a/crates/protocol/src/lib.rs +++ b/crates/protocol/src/lib.rs @@ -35,6 +35,14 @@ pub enum Database { Sqlite, Duckdb, H2, + Presto, + Hive, + Kylin, + Xugu, + Db2, + Informix, + Sundb, + Gbase8s, } impl Database { @@ -42,6 +50,79 @@ impl Database { pub fn is_file_based(self) -> bool { matches!(self, Database::Sqlite | Database::Duckdb) } + + /// Every engine, in the order the documentation lists them. + pub const ALL: [Database; 31] = [ + Database::Mysql, + Database::Mariadb, + Database::Tidb, + Database::Greatsql, + Database::Oceanbase, + Database::Postgresql, + Database::Cockroachdb, + Database::Yugabytedb, + Database::Opengauss, + Database::Oracle, + Database::Sqlserver, + Database::Clickhouse, + Database::Trino, + Database::Starrocks, + Database::Doris, + Database::Tdengine, + Database::Dameng, + Database::Kingbase, + Database::Redis, + Database::Mongodb, + Database::Sqlite, + Database::Duckdb, + Database::H2, + Database::Presto, + Database::Hive, + Database::Kylin, + Database::Xugu, + Database::Db2, + Database::Informix, + Database::Sundb, + Database::Gbase8s, + ]; + + /// The name every other component uses for this engine: `--type`, driver components and + /// prefetch arguments all spell it the same way. + pub fn name(self) -> &'static str { + match self { + Database::Mysql => "mysql", + Database::Mariadb => "mariadb", + Database::Tidb => "tidb", + Database::Greatsql => "greatsql", + Database::Oceanbase => "oceanbase", + Database::Postgresql => "postgresql", + Database::Cockroachdb => "cockroachdb", + Database::Yugabytedb => "yugabytedb", + Database::Opengauss => "opengauss", + Database::Oracle => "oracle", + Database::Sqlserver => "sqlserver", + Database::Clickhouse => "clickhouse", + Database::Trino => "trino", + Database::Starrocks => "starrocks", + Database::Doris => "doris", + Database::Tdengine => "tdengine", + Database::Dameng => "dameng", + Database::Kingbase => "kingbase", + Database::Redis => "redis", + Database::Mongodb => "mongodb", + Database::Sqlite => "sqlite", + Database::Duckdb => "duckdb", + Database::H2 => "h2", + Database::Presto => "presto", + Database::Hive => "hive", + Database::Kylin => "kylin", + Database::Xugu => "xugu", + Database::Db2 => "db2", + Database::Informix => "informix", + Database::Sundb => "sundb", + Database::Gbase8s => "gbase8s", + } + } } impl std::str::FromStr for Database { @@ -71,8 +152,16 @@ impl std::str::FromStr for Database { "sqlite" | "sqlite3" => Ok(Self::Sqlite), "duckdb" => Ok(Self::Duckdb), "h2" => Ok(Self::H2), + "presto" | "prestodb" => Ok(Self::Presto), + "hive" => Ok(Self::Hive), + "kylin" => Ok(Self::Kylin), + "xugu" | "xugudb" => Ok(Self::Xugu), + "db2" | "ibmdb2" => Ok(Self::Db2), + "informix" | "ifx" => Ok(Self::Informix), + "sundb" => Ok(Self::Sundb), + "gbase8s" | "gbasedbt" => Ok(Self::Gbase8s), _ => Err( - "expected mysql, mariadb, tidb, greatsql, oceanbase, postgresql, cockroachdb, yugabytedb, opengauss, oracle, sqlserver, clickhouse, trino, starrocks, doris, tdengine, dameng, kingbase, redis, mongodb, sqlite, duckdb, or h2" + "expected mysql, mariadb, tidb, greatsql, oceanbase, postgresql, cockroachdb, yugabytedb, opengauss, oracle, sqlserver, clickhouse, trino, starrocks, doris, tdengine, dameng, kingbase, redis, mongodb, sqlite, duckdb, h2, presto, hive, kylin, xugu, db2, informix, sundb, or gbase8s" .into(), ), } @@ -335,6 +424,23 @@ mod tests { properties: std::collections::BTreeMap::new(), } } + #[test] + fn every_engine_name_round_trips() { + let mut names = std::collections::BTreeSet::new(); + for kind in Database::ALL { + let name = kind.name(); + assert!(names.insert(name), "{name} is declared twice"); + assert_eq!( + name.parse::(), + Ok(kind), + "{name} does not parse back" + ); + let serialized = serde_json::to_value(kind).expect("an engine serializes"); + assert_eq!(serialized, serde_json::Value::String(name.into())); + } + assert_eq!(names.len(), Database::ALL.len()); + } + #[test] fn redaction_keeps_hostnames_and_words() { let connection = connection("app", "oracle"); diff --git a/docs/design.md b/docs/design.md index 6a0b7ef..e5698cd 100644 --- a/docs/design.md +++ b/docs/design.md @@ -67,7 +67,7 @@ The table below lists the capabilities planned for the first release and the pro | Help | Show command and argument documentation | `sqlx --help`, `sqlx --help` | | Version | Show the main program version so users and the Skill can judge compatibility | `sqlx --version` | -Datasource and SQL responses are structured; help text stays human-readable. The release manifest controls the exact driver and runtime versions, and the first release adds no driver command that makes users manage versions themselves. +Datasource and SQL responses are structured; help text stays human-readable. The release manifest controls the exact driver and runtime versions, so no command asks a user to pick a version. `sqlx driver add` exists for the engines whose vendor does not allow redistribution: it installs a driver the user obtained from that vendor, checks that the jar really carries the engine's driver class, and loads it before the released one, so a user can override any engine's driver without changing the version the release installs. The first release excludes SQL file input, cross-call sessions, parameters for transaction mode and failure strategy, result pagination or writing the full result to a file, automatic main-program updates, and device information reporting and daily-active statistics services. Device identity is generated locally first and the related networked features come later; explicit Skill installation and update are part of the first release. @@ -89,8 +89,12 @@ Native components and the JRE are published per operating system and CPU archite |---|---|---| | MySQL | Rust native execution component | Extract the `mysql_async` connection and execution logic from Chat2DB-Rust first and publish it with the component build | | PostgreSQL | Rust native execution component | Extract the `tokio-postgres` connection and execution logic from Chat2DB-Rust first and publish it with the component build | +| MariaDB, TiDB, GreatSQL, OceanBase, StarRocks, Apache Doris | Rust native execution component | Speak the MySQL protocol, so they reuse the MySQL component | +| CockroachDB, YugabyteDB | Rust native execution component | Speak the PostgreSQL protocol, so they reuse the PostgreSQL component | +| Redis, MongoDB, SQLite, DuckDB | Rust native execution component | Their own component; SQLite and DuckDB open a local file | | Oracle | Official JDBC Thin driver | Runs in the private Java runtime, with no separate Oracle client installation | -| SQL Server | Official Microsoft JDBC driver | Reuse the official driver and the shared JDBC runner | +| SQL Server, ClickHouse, Trino, Presto, TDengine, openGauss, Dameng, KingbaseES, H2, Hive, Apache Kylin, XuguDB | Official or vendor JDBC driver | Reuse the official driver and the shared JDBC runner | +| IBM Db2, IBM Informix, SUNDB, GBase 8s | Vendor JDBC driver the user provides | The vendor does not allow redistribution, so `sqlx driver add` installs the jar the user obtained and the JDBC runner loads it | The last two are designed for username and password connections first; whether connection methods such as integrated authentication bring extra local dependencies needs separate verification and cannot inherit the delivery conclusions of a pure-Java connection method directly. @@ -99,8 +103,9 @@ flowchart LR A[Agent + Skill] --> C[Rust CLI] C --> N[Native execution component] C --> J[Private JRE + JDBC runner] - N --> M[MySQL / PostgreSQL] - J --> O[Oracle / SQL Server] + N --> M[MySQL / PostgreSQL / Redis / MongoDB / SQLite / DuckDB] + J --> O[Oracle / SQL Server / ClickHouse / Trino / Hive / Kylin / XuguDB ...] + V[Vendor driver jar] -.sqlx driver add.-> J R[GitHub Releases] -.download on demand.-> N R -.download on demand.-> J ``` diff --git a/integrations/claude/plugins/sqlx/.claude-plugin/plugin.json b/integrations/claude/plugins/sqlx/.claude-plugin/plugin.json index f0447f5..650c180 100644 --- a/integrations/claude/plugins/sqlx/.claude-plugin/plugin.json +++ b/integrations/claude/plugins/sqlx/.claude-plugin/plugin.json @@ -1,6 +1,6 @@ { "name": "sqlx", - "description": "List saved OtterMind SQLX datasources, test connections, run SQL against MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng and KingbaseES, run Redis commands and MongoDB command documents, and open local result pages. Credentials stay in the SQLX data directory. Requires the sqlx CLI (0.1.16+) through the plugin launcher.", + "description": "List saved OtterMind SQLX datasources, test connections, run SQL against MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng, KingbaseES, Presto, Hive, Apache Kylin, XuguDB, Db2, Informix, SUNDB and GBase 8s, run Redis commands and MongoDB command documents, and open local result pages. Credentials stay in the SQLX data directory. Requires the sqlx CLI (0.1.16+) through the plugin launcher.", "version": "0.1.16", "author": { "name": "OtterMind" @@ -31,6 +31,14 @@ "kingbase", "redis", "mongodb", + "presto", + "hive", + "kylin", + "xugu", + "db2", + "informix", + "sundb", + "gbase8s", "mcp" ] } diff --git a/integrations/codex/plugins/sqlx/.codex-plugin/plugin.json b/integrations/codex/plugins/sqlx/.codex-plugin/plugin.json index d472742..6a30836 100644 --- a/integrations/codex/plugins/sqlx/.codex-plugin/plugin.json +++ b/integrations/codex/plugins/sqlx/.codex-plugin/plugin.json @@ -1,7 +1,7 @@ { "name": "sqlx", "version": "0.1.16", - "description": "List saved OtterMind SQLX datasources, test connections, run SQL against MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng and KingbaseES, run Redis commands and MongoDB command documents, and open local result pages. Credentials stay in the SQLX data directory. Requires the sqlx CLI (0.1.16+) through the plugin launcher.", + "description": "List saved OtterMind SQLX datasources, test connections, run SQL against MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng, KingbaseES, Presto, Hive, Apache Kylin, XuguDB, Db2, Informix, SUNDB and GBase 8s, run Redis commands and MongoDB command documents, and open local result pages. Credentials stay in the SQLX data directory. Requires the sqlx CLI (0.1.16+) through the plugin launcher.", "author": { "name": "OtterMind" }, @@ -31,6 +31,14 @@ "kingbase", "redis", "mongodb", + "presto", + "hive", + "kylin", + "xugu", + "db2", + "informix", + "sundb", + "gbase8s", "mcp" ], "mcpServers": "./.mcp.json", diff --git a/java/jdbc/src/main/java/ai/ottermind/sqlx/JdbcWorker.java b/java/jdbc/src/main/java/ai/ottermind/sqlx/JdbcWorker.java index 7c182d8..0eb9e8d 100644 --- a/java/jdbc/src/main/java/ai/ottermind/sqlx/JdbcWorker.java +++ b/java/jdbc/src/main/java/ai/ottermind/sqlx/JdbcWorker.java @@ -30,9 +30,14 @@ private static void quietDriverLogging() { LogManager.getLogManager().reset(); Logger.getLogger("").setLevel(Level.OFF); } - public static void main(String[] args) { + public static void main(String[] args) throws java.io.IOException { quietDriverLogging(); - JdbcWorker worker = new JdbcWorker(System.out); + // The protocol owns standard output, so the worker keeps its own handle on it and sends + // anything a driver prints there to standard error instead: Kylin's Avatica client does that + // while answering a query, and a stray line would corrupt the event stream. + PrintStream protocol = new PrintStream(new java.io.FileOutputStream(java.io.FileDescriptor.out), true, "UTF-8"); + System.setOut(new PrintStream(new java.io.FileOutputStream(java.io.FileDescriptor.err), true, "UTF-8")); + JdbcWorker worker = new JdbcWorker(protocol); try { worker.emit("ready", "protocol_version", 1); JsonNode request = JSON.readTree(System.in); @@ -75,6 +80,8 @@ boolean run(JsonNode request) throws IOException { if (!Files.isRegularFile(p)) throw new IllegalArgumentException("driver JAR is missing"); jars[i] = p.toUri().toURL(); } + // The failure is reported from inside this scope: reading a driver's own error message can + // load a class from the driver jar, which a closed loader refuses. try (URLClassLoader loader = new URLClassLoader(jars, ClassLoader.getPlatformClassLoader())) { Class type = Class.forName(request.path("driver_class").asText(), true, loader); if (type.getClassLoader() != loader || !Driver.class.isAssignableFrom(type)) throw new IllegalArgumentException("invalid driver class"); @@ -94,67 +101,112 @@ boolean run(JsonNode request) throws IOException { properties.setProperty("trustServerCertificate", "false"); properties.setProperty("loginTimeout", "15"); } + // A failure is reported inside this scope: a driver renders its message lazily and + // loads one of its own classes while doing so, which a closed loader refuses. Thread.currentThread().setContextClassLoader(loader); - try (Connection connection = driver.connect(url(config), properties)) { - if (connection == null) throw new SQLException("driver rejected connection URL"); - connection.setAutoCommit(true); - if (!connection.isValid(15)) throw new SQLException("connection validation failed"); - emit("connected"); - if (request.path("action").asText().equals("execute")) { - if (!sql.isArray() || sql.isEmpty()) throw new IllegalArgumentException("SQL statements are required"); - for (int i=0; i throw new IllegalArgumentException("JDBC worker supports oracle, sqlserver, clickhouse, trino, tdengine, opengauss, dameng, kingbase and h2"); + case "presto" -> { + // PrestoDB addresses a catalog and an optional schema; --database carries catalog[.schema]. + yield "jdbc:presto://" + authority + "/" + c.path("database").asText().replace('.', '/') + + (c.path("tls").asText().equals("disable") ? "" : "?SSL=true"); + } + case "hive" -> "jdbc:hive2://" + authority + "/" + c.path("database").asText() + + (c.path("tls").asText().equals("disable") ? "" : ";ssl=true"); + case "kylin" -> "jdbc:kylin://" + authority + "/" + c.path("database").asText(); + case "xugu" -> "jdbc:xugu://" + authority + "/" + c.path("database").asText(); + case "db2" -> "jdbc:db2://" + authority + "/" + c.path("database").asText() + + (c.path("tls").asText().equals("disable") ? "" : ":sslConnection=true;"); + // SUNDB runs the Goldilocks engine and keeps its URL scheme. + case "sundb" -> "jdbc:goldilocks://" + authority + "/" + c.path("database").asText(); + // Informix and its GBase 8s derivative name a server instance in the URL; each driver + // spells that parameter its own way. + case "informix" -> "jdbc:informix-sqli://" + authority + "/" + c.path("database").asText() + + informixParameters(c, "INFORMIXSERVER"); + // The GBase 8s driver fails inside its own parser without a client locale, so every URL + // carries one; --property DB_LOCALE or CLIENT_LOCALE replaces the default. + case "gbase8s" -> "jdbc:gbasedbt-sqli://" + authority + "/" + c.path("database").asText() + + gbaseParameters(c); + default -> throw new IllegalArgumentException("JDBC worker supports oracle, sqlserver, clickhouse, trino, tdengine, opengauss, dameng, kingbase, h2, presto, hive, kylin, xugu, db2, informix, sundb and gbase8s"); }; } + /** The parameter block a GBase 8s URL appends: the server instance and the client locales. */ + static String gbaseParameters(JsonNode c) { + String instance = informixParameters(c, "GBASEDBTSERVER"); + StringBuilder parameters = new StringBuilder(instance.isEmpty() ? ":" : instance); + parameters.append("DB_LOCALE=").append(locale(c, "DB_LOCALE")).append(';'); + parameters.append("CLIENT_LOCALE=").append(locale(c, "CLIENT_LOCALE")).append(';'); + return parameters.toString(); + } + /** A caller-supplied locale, or the locale the GBase 8s driver expects by default. */ + static String locale(JsonNode c, String key) { + JsonNode properties = c.path("properties"); + for (String name : List.of(key, key.toLowerCase(java.util.Locale.ROOT))) { + String value = properties.path(name).asText(); + if (!value.isEmpty()) return value; + } + return "en_US.819"; + } + /** The parameter block an Informix-derived URL appends after the database name. */ + static String informixParameters(JsonNode c, String serverKey) { + String service = c.path("service").asText(); + boolean tls = !c.path("tls").asText().equals("disable"); + if (service.isEmpty() && !tls) return ""; + StringBuilder parameters = new StringBuilder(":"); + if (!service.isEmpty()) parameters.append(serverKey).append('=').append(service).append(';'); + if (tls) parameters.append("sslConnection=true;"); + return parameters.toString(); + } /** Some drivers, such as the TDengine RESTful driver, only implement the JDBC 1 update count. */ static long updateCount(Statement st) throws SQLException { try { diff --git a/java/jdbc/src/test/java/ai/ottermind/sqlx/JdbcWorkerTest.java b/java/jdbc/src/test/java/ai/ottermind/sqlx/JdbcWorkerTest.java index 1046194..7d54518 100644 --- a/java/jdbc/src/test/java/ai/ottermind/sqlx/JdbcWorkerTest.java +++ b/java/jdbc/src/test/java/ai/ottermind/sqlx/JdbcWorkerTest.java @@ -28,5 +28,34 @@ class JdbcWorkerTest { ObjectMapper json=new ObjectMapper(); var oracle=json.readTree("{\"database_type\":\"oracle\",\"host\":\"localhost\",\"port\":1521,\"service\":\"FREEPDB1\",\"tls\":\"disable\"}"); assertEquals("jdbc:oracle:thin:@tcp://localhost:1521/FREEPDB1",JdbcWorker.url(oracle)); + var hive=json.readTree("{\"database_type\":\"hive\",\"host\":\"localhost\",\"port\":10000,\"database\":\"default\",\"tls\":\"disable\"}"); + assertEquals("jdbc:hive2://localhost:10000/default",JdbcWorker.url(hive)); + var presto=json.readTree("{\"database_type\":\"presto\",\"host\":\"localhost\",\"port\":8080,\"database\":\"tpch.tiny\",\"tls\":\"disable\"}"); + assertEquals("jdbc:presto://localhost:8080/tpch/tiny",JdbcWorker.url(presto)); + var kylin=json.readTree("{\"database_type\":\"kylin\",\"host\":\"localhost\",\"port\":7070,\"database\":\"learn_kylin\",\"tls\":\"disable\"}"); + assertEquals("jdbc:kylin://localhost:7070/learn_kylin",JdbcWorker.url(kylin)); + var xugu=json.readTree("{\"database_type\":\"xugu\",\"host\":\"localhost\",\"port\":5138,\"database\":\"SYSTEM\",\"tls\":\"disable\"}"); + assertEquals("jdbc:xugu://localhost:5138/SYSTEM",JdbcWorker.url(xugu)); + var db2=json.readTree("{\"database_type\":\"db2\",\"host\":\"localhost\",\"port\":50000,\"database\":\"testdb\",\"tls\":\"disable\"}"); + assertEquals("jdbc:db2://localhost:50000/testdb",JdbcWorker.url(db2)); + var sundb=json.readTree("{\"database_type\":\"sundb\",\"host\":\"localhost\",\"port\":22581,\"database\":\"sundb\",\"tls\":\"disable\"}"); + assertEquals("jdbc:goldilocks://localhost:22581/sundb",JdbcWorker.url(sundb)); + // Informix and GBase 8s refuse a URL without a server instance and take it from --service. + var informix=json.readTree("{\"database_type\":\"informix\",\"host\":\"localhost\",\"port\":9088,\"database\":\"sysmaster\",\"service\":\"informix\",\"tls\":\"disable\"}"); + assertEquals("jdbc:informix-sqli://localhost:9088/sysmaster:INFORMIXSERVER=informix;",JdbcWorker.url(informix)); + var gbase=json.readTree("{\"database_type\":\"gbase8s\",\"host\":\"localhost\",\"port\":9088,\"database\":\"sysmaster\",\"service\":\"gbase01\",\"tls\":\"disable\"}"); + assertEquals("jdbc:gbasedbt-sqli://localhost:9088/sysmaster:GBASEDBTSERVER=gbase01;DB_LOCALE=en_US.819;CLIENT_LOCALE=en_US.819;",JdbcWorker.url(gbase)); + var gbaseLocale=json.readTree("{\"database_type\":\"gbase8s\",\"host\":\"localhost\",\"port\":9088,\"database\":\"sysmaster\",\"service\":\"gbase01\",\"tls\":\"disable\",\"properties\":{\"DB_LOCALE\":\"zh_CN.GB18030-2000\"}}"); + assertEquals("jdbc:gbasedbt-sqli://localhost:9088/sysmaster:GBASEDBTSERVER=gbase01;DB_LOCALE=zh_CN.GB18030-2000;CLIENT_LOCALE=en_US.819;",JdbcWorker.url(gbaseLocale)); + var informixTls=json.readTree("{\"database_type\":\"informix\",\"host\":\"localhost\",\"port\":9088,\"database\":\"sysmaster\",\"tls\":\"verify-full\"}"); + assertEquals("jdbc:informix-sqli://localhost:9088/sysmaster:sslConnection=true;",JdbcWorker.url(informixTls)); + } + @Test void anEngineWithoutAUrlArmIsRejected() throws Exception { + ObjectMapper json=new ObjectMapper(); + var unknown=json.readTree("{\"database_type\":\"mongodb\",\"host\":\"localhost\",\"port\":27017,\"tls\":\"disable\"}"); + // The message names every engine the worker can reach, so a new arm is never silently missing. + IllegalArgumentException failure=assertThrows(IllegalArgumentException.class,()->JdbcWorker.url(unknown)); + assertTrue(failure.getMessage().contains("db2"),failure.getMessage()); + assertTrue(failure.getMessage().contains("gbase8s"),failure.getMessage()); } } diff --git a/scripts/jdbc-fixture.sh b/scripts/jdbc-fixture.sh index 85811c5..58d475f 100644 --- a/scripts/jdbc-fixture.sh +++ b/scripts/jdbc-fixture.sh @@ -41,5 +41,52 @@ case "$kind" in curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/com/taosdata/jdbc/taos-jdbcdriver/3.6.3/taos-jdbcdriver-3.6.3-dist.jar' -o target/debug/taos-jdbcdriver.jar curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/slf4j/slf4j-nop/2.0.16/slf4j-nop-2.0.16.jar' -o target/debug/slf4j-nop.jar ;; + # Presto, Hive, Kylin and XuguDB run from tests/compose.yaml, so these cases stage their drivers. + presto) + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/com/facebook/presto/presto-jdbc/0.293/presto-jdbc-0.293.jar' -o target/debug/presto-jdbc.jar + ;; + hive) + # The standalone jar is the only self-contained Hive driver; its slf4j 1.7 API needs a binding. + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/apache/hive/hive-jdbc/4.0.1/hive-jdbc-4.0.1-standalone.jar' -o target/debug/hive-jdbc.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/slf4j/slf4j-nop/1.7.36/slf4j-nop-1.7.36.jar' -o target/debug/slf4j-nop.jar + ;; + kylin) + # The driver needs JAXB, which the JDK dropped in Java 11, and an slf4j 1.7 binding. + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/apache/kylin/kylin-jdbc/5.0.3/kylin-jdbc-5.0.3.jar' -o target/debug/kylin-jdbc.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/jakarta/xml/bind/jakarta.xml.bind-api/2.3.3/jakarta.xml.bind-api-2.3.3.jar' -o target/debug/jakarta.xml.bind-api.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/glassfish/jaxb/jaxb-runtime/2.3.9/jaxb-runtime-2.3.9.jar' -o target/debug/jaxb-runtime.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/com/sun/istack/istack-commons-runtime/4.1.2/istack-commons-runtime-4.1.2.jar' -o target/debug/istack-commons-runtime.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/jakarta/activation/jakarta.activation-api/1.2.2/jakarta.activation-api-1.2.2.jar' -o target/debug/jakarta.activation-api.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/glassfish/jaxb/txw2/2.3.9/txw2-2.3.9.jar' -o target/debug/txw2.jar + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/org/slf4j/slf4j-nop/1.7.36/slf4j-nop-1.7.36.jar' -o target/debug/slf4j-nop.jar + ;; + xugu) + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/com/xugudb/xugu-jdbc/12.3.4/xugu-jdbc-12.3.4.jar' -o target/debug/xugu-jdbc.jar + ;; + # The vendors below do not allow redistribution, so their driver has to come from the vendor. + # Db2 and Informix publish theirs on Maven Central, which is enough for a local fixture. + db2) + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/com/ibm/db2/jcc/12.1.0.0/jcc-12.1.0.0.jar' -o target/debug/db2-jcc.jar + ;; + informix) + # The 15.x driver fails inside its own ASF layer against an Informix 14.10 server (a null + # pointer, surfaced as "An unexpected error occurred"); the 4.50 line connects. + curl --fail --location --retry 3 'https://repo.maven.apache.org/maven2/com/ibm/informix/jdbc/4.50.14/jdbc-4.50.14.jar' -o target/debug/informix-jdbc.jar + ;; + sundb) + # SUNDB runs the Goldilocks engine; the driver lives in the vendor image. + docker cp "$(docker create sundb/sundb_standlone:01):/goldilocks_home/lib/goldilocks8.jar" target/debug/goldilocks8.jar + ;; + gbase8s) + # GBase 8s is commercial, so its driver comes from an installed instance rather than a registry. + # The vendor ships a wrapper jar whose inner ifxjdbc.jar is the file the JVM can load. + : "${SQLX_TEST_GBASE8S_DRIVER:?set SQLX_TEST_GBASE8S_DRIVER to the vendor jdbc jar or wrapper jar}" + cp "$SQLX_TEST_GBASE8S_DRIVER" target/debug/gbasedbt-provided.jar + if unzip -l target/debug/gbasedbt-provided.jar ifxjdbc.jar >/dev/null 2>&1; then + (cd target/debug && unzip -o -q gbasedbt-provided.jar ifxjdbc.jar && mv ifxjdbc.jar gbasedbt-jdbc.jar) + else + mv target/debug/gbasedbt-provided.jar target/debug/gbasedbt-jdbc.jar + fi + ;; *) echo 'Unsupported fixture' >&2; exit 1 ;; esac diff --git a/scripts/package-shared.py b/scripts/package-shared.py index 48d22e7..a465168 100644 --- a/scripts/package-shared.py +++ b/scripts/package-shared.py @@ -25,6 +25,31 @@ 'dameng':dict(files=[('https://repo.maven.apache.org/maven2/com/dameng/DmJdbcDriver18/8.1.3.140/DmJdbcDriver18-8.1.3.140.jar','dm-jdbc.jar')],entry='dm-jdbc.jar',license_url='https://repo1.maven.org/maven2/com/dameng/DmJdbcDriver18/8.1.3.140/DmJdbcDriver18-8.1.3.140.pom'), 'kingbase':dict(files=[('https://repo.maven.apache.org/maven2/cn/com/kingbase/kingbase8/9.0.1.jre7/kingbase8-9.0.1.jre7.jar','kingbase8-jdbc.jar')],entry='kingbase8-jdbc.jar',license_url='https://repo1.maven.org/maven2/cn/com/kingbase/kingbase8/9.0.1.jre7/kingbase8-9.0.1.jre7.pom'), 'opengauss':dict(files=[('https://repo.maven.apache.org/maven2/org/opengauss/opengauss-jdbc/6.0.0-b041-og/opengauss-jdbc-6.0.0-b041-og.jar','opengauss-jdbc.jar')],entry='opengauss-jdbc.jar',license_url='https://raw.githubusercontent.com/opengauss-mirror/openGauss-connector-jdbc/master/LICENSE'), + 'presto':dict(files=[('https://repo.maven.apache.org/maven2/com/facebook/presto/presto-jdbc/0.293/presto-jdbc-0.293.jar','presto-jdbc.jar')],entry='presto-jdbc.jar',license_url='https://raw.githubusercontent.com/prestodb/presto/master/LICENSE'), + # Hive needs its standalone jar: the plain artifact pulls the whole Hadoop dependency tree. + # 4.0.1 is the last release whose driver still targets Java 8; 4.2.x is compiled for Java 21 and + # cannot load in the pinned JRE 17. + 'hive':dict(files=[('https://repo.maven.apache.org/maven2/org/apache/hive/hive-jdbc/4.0.1/hive-jdbc-4.0.1-standalone.jar','hive-jdbc.jar'), + ('https://repo.maven.apache.org/maven2/org/slf4j/slf4j-nop/1.7.36/slf4j-nop-1.7.36.jar','slf4j-nop.jar')], + entry='hive-jdbc.jar',license_url='https://raw.githubusercontent.com/apache/hive/rel/release-4.0.1/LICENSE', + # The slf4j 1.7 binding carries no license file inside the jar, so its license is + # fetched from the project instead. + extra_licenses={'LICENSE-slf4j.txt':('https://www.slf4j.org/license.html',None)}), + # Kylin's driver uses JAXB, which the JDK dropped in Java 11, and an slf4j 1.7 binding. + 'kylin':dict(files=[('https://repo.maven.apache.org/maven2/org/apache/kylin/kylin-jdbc/5.0.3/kylin-jdbc-5.0.3.jar','kylin-jdbc.jar'), + ('https://repo.maven.apache.org/maven2/jakarta/xml/bind/jakarta.xml.bind-api/2.3.3/jakarta.xml.bind-api-2.3.3.jar','jakarta.xml.bind-api.jar'), + ('https://repo.maven.apache.org/maven2/org/glassfish/jaxb/jaxb-runtime/2.3.9/jaxb-runtime-2.3.9.jar','jaxb-runtime.jar'), + ('https://repo.maven.apache.org/maven2/com/sun/istack/istack-commons-runtime/4.1.2/istack-commons-runtime-4.1.2.jar','istack-commons-runtime.jar'), + ('https://repo.maven.apache.org/maven2/jakarta/activation/jakarta.activation-api/1.2.2/jakarta.activation-api-1.2.2.jar','jakarta.activation-api.jar'), + ('https://repo.maven.apache.org/maven2/org/glassfish/jaxb/txw2/2.3.9/txw2-2.3.9.jar','txw2.jar'), + ('https://repo.maven.apache.org/maven2/org/slf4j/slf4j-nop/1.7.36/slf4j-nop-1.7.36.jar','slf4j-nop.jar')], + entry='kylin-jdbc.jar',license_url='https://raw.githubusercontent.com/apache/kylin/master/LICENSE', + # The JAXB runtime ships its own Eclipse Distribution License, so it is read from the + # jar instead of a documentation URL. + extra_licenses={'LICENSE-jaxb.txt':('jaxb-runtime.jar','META-INF/LICENSE.md'), + 'NOTICE-jaxb.txt':('jaxb-runtime.jar','META-INF/NOTICE.md'), + 'LICENSE-slf4j.txt':('https://www.slf4j.org/license.html',None)}), + 'xugu':dict(files=[('https://repo.maven.apache.org/maven2/com/xugudb/xugu-jdbc/12.3.4/xugu-jdbc-12.3.4.jar','xugu-jdbc.jar')],entry='xugu-jdbc.jar',license_url='https://www.apache.org/licenses/LICENSE-2.0.txt'), # The TDengine RESTful driver ships as one bundled jar (its own dependencies included) and # needs an slf4j binding, because the bundle carries the slf4j API without a provider. 'tdengine':dict(files=[('https://repo.maven.apache.org/maven2/com/taosdata/jdbc/taos-jdbcdriver/3.6.3/taos-jdbcdriver-3.6.3-dist.jar','taos-jdbcdriver.jar'), @@ -65,13 +90,24 @@ def add(name,entries,entrypoint,version=None): entries={};sources=[] for url,filename in spec['files']: entries[filename]=get(url);sources.append(url) + # Every driver runs in the pinned JRE 17, so a class compiled for a newer Java would only + # fail on a user's machine. Check each driver entry point here, where the fix is a version + # bump: a bundled helper that loads eagerly would break the engine just the same. + with zipfile.ZipFile(io.BytesIO(entries[spec['entry']])) as jar: + too_new=[(entry,int.from_bytes(jar.read(entry)[6:8],'big')) + for entry in jar.namelist() if entry.endswith('Driver.class') and '$' not in entry] + too_new=[(entry,major) for entry,major in too_new if major>61] + if too_new: + listed=', '.join(f'{entry} (Java {major-44})' for entry,major in too_new) + raise ValueError(f'{name} ships a driver class newer than the pinned JRE 17: {listed}') if 'license_from_jar' in spec: # Preserve the license shipped with this exact driver; the HTML page blocks automated downloads. with zipfile.ZipFile(io.BytesIO(entries[spec['entry']])) as jar:license_text=jar.read(spec['license_from_jar']) else:license_text=get(spec['license_url']) entries['LICENSE.txt']=license_text for target,(archive,member) in spec.get('extra_licenses',{}).items(): - with zipfile.ZipFile(io.BytesIO(entries[archive])) as jar:entries[target]=jar.read(member) + # A URL source is fetched directly; a jar member is read from the driver already loaded. + entries[target]=get(archive) if member is None else zipfile.ZipFile(io.BytesIO(entries[archive])).read(member) entries['SOURCE.txt']=('\n'.join(sources+[spec['license_url']])+'\n').encode() add(name,entries,spec['entry']) for platform,os_name,arch in [('macos-arm64','mac','aarch64'),('macos-x64','mac','x64'),('windows-x64','windows','x64'),('linux-arm64','linux','aarch64'),('linux-x64','linux','x64')]: diff --git a/scripts/release-driver-check.sh b/scripts/release-driver-check.sh new file mode 100755 index 0000000..69d272d --- /dev/null +++ b/scripts/release-driver-check.sh @@ -0,0 +1,34 @@ +#!/usr/bin/env bash +# Exercise the path a release user takes for an engine whose driver is not redistributed: +# install the jar the vendor provides, then run a statement with no development overrides, so the CLI +# resolves the JRE and the JDBC runner from the release manifest and loads the provided driver. +# +# usage: scripts/release-driver-check.sh +set -euo pipefail +engine="${1:?engine required, for example db2}" +jar="${2:?path to the vendor driver jar required}" +data_dir="${3:?data directory required}" +connection="${4:?connection JSON required}" +statement="${5:?one statement required}" + +sqlx_bin="${SQLX_BIN:-target/debug/sqlx}" +mkdir -p "${data_dir}" + +# A declared component is not needed for these engines, so the check would fail if the CLI expected +# one; the jar has to be found through the driver directory alone. +"${sqlx_bin}" --data-dir "${data_dir}" driver add --type "${engine}" --jar "${jar}" >/dev/null + +"${sqlx_bin}" --data-dir "${data_dir}" datasource add --name release-check --connection-stdin <<<"${connection}" >/dev/null + +if ! result="$("${sqlx_bin}" --data-dir "${data_dir}" sql execute --datasource release-check --command "${statement}")"; then + echo "the release path failed for ${engine}: ${result}" >&2 + exit 1 +fi +case "${result}" in + *'"success":true'*) ;; + *) + echo "the release path failed for ${engine}: ${result}" >&2 + exit 1 + ;; +esac +echo "release path: ${engine} installed a provided driver, resolved the runtime and answered a query" diff --git a/skills/sqlx/SKILL.md b/skills/sqlx/SKILL.md index 4a45be4..7fa81ea 100644 --- a/skills/sqlx/SKILL.md +++ b/skills/sqlx/SKILL.md @@ -1,6 +1,6 @@ --- name: sqlx -description: Manage encrypted database connections and execute SQL with the OtterMind SQLX CLI. Use for MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng, KingbaseES, Redis, MongoDB, SQLite, DuckDB, and H2 connection checks, queries, DDL, and schema inspection. This skill targets OtterMind/sqlx, not the Rust SQLx migration CLI. +description: Manage encrypted database connections and execute SQL with the OtterMind SQLX CLI. Use for MySQL, MariaDB, TiDB, GreatSQL, OceanBase, PostgreSQL, CockroachDB, YugabyteDB, openGauss, Oracle, SQL Server, ClickHouse, Trino, StarRocks, Apache Doris, TDengine, Dameng, KingbaseES, Redis, MongoDB, SQLite, DuckDB, H2, Presto, Hive, Kylin, XuguDB, Db2, Informix, SUNDB, and GBase 8s connection checks, queries, DDL, and schema inspection. This skill targets OtterMind/sqlx, not the Rust SQLx migration CLI. metadata: cli-compat: ">=0.1.15, <0.2.0" --- @@ -31,7 +31,7 @@ sqlx sql execute --datasource --command "SELECT 1" --command "SELECT 2" Load the recipe of the connected datasource, and only that one. Each recipe links the vendor's own documentation for its operations; match the connected server version. -MySQL `references/mysql.md` · MariaDB `references/mariadb.md` · TiDB `references/tidb.md` · GreatSQL `references/greatsql.md` · OceanBase `references/oceanbase.md` · PostgreSQL `references/postgresql.md` · CockroachDB `references/cockroachdb.md` · YugabyteDB `references/yugabytedb.md` · openGauss `references/opengauss.md` · Oracle `references/oracle.md` · SQL Server `references/sqlserver.md` · ClickHouse `references/clickhouse.md` · Trino `references/trino.md` · StarRocks `references/starrocks.md` · Apache Doris `references/doris.md` · TDengine `references/tdengine.md` · Dameng `references/dameng.md` · KingbaseES `references/kingbase.md` · Redis `references/redis.md` · MongoDB `references/mongodb.md` · SQLite `references/sqlite.md` · DuckDB `references/duckdb.md` · H2 `references/h2.md` +MySQL `references/mysql.md` · MariaDB `references/mariadb.md` · TiDB `references/tidb.md` · GreatSQL `references/greatsql.md` · OceanBase `references/oceanbase.md` · PostgreSQL `references/postgresql.md` · CockroachDB `references/cockroachdb.md` · YugabyteDB `references/yugabytedb.md` · openGauss `references/opengauss.md` · Oracle `references/oracle.md` · SQL Server `references/sqlserver.md` · ClickHouse `references/clickhouse.md` · Trino `references/trino.md` · StarRocks `references/starrocks.md` · Apache Doris `references/doris.md` · TDengine `references/tdengine.md` · Dameng `references/dameng.md` · KingbaseES `references/kingbase.md` · Redis `references/redis.md` · MongoDB `references/mongodb.md` · SQLite `references/sqlite.md` · DuckDB `references/duckdb.md` · H2 `references/h2.md` · Presto `references/presto.md` · Hive `references/hive.md` · Apache Kylin `references/kylin.md` · XuguDB `references/xugu.md` · IBM Db2 `references/db2.md` · IBM Informix `references/informix.md` · SUNDB `references/sundb.md` · GBase 8s `references/gbase8s.md` TiDB, GreatSQL, OceanBase, StarRocks and Apache Doris reuse the MySQL worker, and YugabyteDB reuses the PostgreSQL worker; ClickHouse needs its HTTP port, Trino needs `--database [.]`, and the remaining engines, including TDengine, openGauss, Dameng and KingbaseES, connect through the JDBC worker. Redis and MongoDB are not SQL: each `--command` is one Redis command or one MongoDB command document, and `references/approval.md` lists the write commands for both. SQLite and DuckDB run in their own native workers and open a local file instead of a server, and H2 uses the JDBC worker as an embedded file or over its TCP server. diff --git a/skills/sqlx/references/cli.md b/skills/sqlx/references/cli.md index f198023..9681b50 100644 --- a/skills/sqlx/references/cli.md +++ b/skills/sqlx/references/cli.md @@ -4,6 +4,8 @@ The repository is https://github.com/OtterMind/sqlx. The supported CLI range for `sqlx setting list` shows the settings in effect, and `sqlx setting set ` changes the preview size, the result directory or the result retention; see [results](results.md). +A few engines -- IBM Db2, IBM Informix, SUNDB and GBase 8s -- connect with a driver their vendor does not allow SQLX to redistribute. `sqlx driver list` shows where each engine's driver comes from, and `sqlx driver add --type --jar ` installs the jar the user obtained from the vendor after checking it carries that engine's driver class; `sqlx driver remove --type ` takes it back. Running a statement for one of those engines without a driver fails before anything is downloaded and names the exact command to run, so relay that command to the user instead of guessing a URL. + ## Install ```sh diff --git a/skills/sqlx/references/db2.md b/skills/sqlx/references/db2.md new file mode 100644 index 0000000..2a23c3b --- /dev/null +++ b/skills/sqlx/references/db2.md @@ -0,0 +1,101 @@ +# IBM Db2 operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `SCHEMA_NAME`, `TABLE_NAME`, and `COLUMN_NAME` with actual identifiers. Db2 folds unquoted identifiers to upper case, so an unquoted name is stored in upper case and is only reachable in that spelling; double quotes preserve the exact case and let a keyword stand as a name. + +`--type db2` runs on the JDBC worker and builds `jdbc:db2://host:port/database`, where `--database` is the database name and the default port is 50000. A username and a password are required. IBM's driver is not bundled: run `sqlx driver add --type db2 --jar ` once with the driver from IBM, and `sqlx driver list` to see where each engine's driver comes from. Schemas are objects inside the connected database, not separate databases. The documentation root is https://www.ibm.com/docs/en/db2. + +## 1. Identify the current connection + +**Purpose:** Check the database, the authorization ID, and the current schema before operating on a target. No placeholders need replacement. + +```sql +SELECT CURRENT SERVER AS database_name, + CURRENT USER AS authorization_id, + CURRENT SCHEMA AS schema_name +FROM SYSIBM.SYSDUMMY1; +``` + +**Result:** One context row. `SYSIBM.SYSDUMMY1` is the one-row dummy table a statement selects from when it has no real table. `CURRENT SCHEMA` is what unqualified names resolve against, and it is not the same as the authorization ID. + +**Official documentation:** [SELECT statement](https://www.ibm.com/docs/en/db2/11.5?topic=statements-select) + +## 2. List schemas + +**Purpose:** Discover the schemas that hold tables the account can see. No placeholders need replacement. + +```sql +SELECT DISTINCT TABSCHEMA AS schema_name FROM SYSCAT.TABLES ORDER BY TABSCHEMA; +``` + +**Result:** One schema per row. `SYSCAT` holds the catalog views and is readable by any account with access to the database; the system schemas such as `SYSIBM` and `SYSCAT` appear alongside the user schemas. + +**Official documentation:** [SYSCAT.TABLES](https://www.ibm.com/docs/en/db2/11.5?topic=views-syscattables) + +## 3. List tables and views + +**Purpose:** Find the table to inspect or query. Replace `SCHEMA_NAME`. + +```sql +SELECT TABSCHEMA, TABNAME, TYPE FROM SYSCAT.TABLES +WHERE TABSCHEMA = 'SCHEMA_NAME' AND TYPE IN ('T', 'V') +ORDER BY TABNAME; +``` + +**Result:** One table or view per row, with `TYPE` `T` for a table and `V` for a view. Restrict `TABSCHEMA` rather than scanning the whole catalog, which on a large database is slow. + +**Official documentation:** [SYSCAT.TABLES](https://www.ibm.com/docs/en/db2/11.5?topic=views-syscattables) + +## 4. Inspect a table's columns + +**Purpose:** Read column names, types, lengths, nullability, and defaults. Replace `SCHEMA_NAME` and `TABLE_NAME`. + +```sql +SELECT COLNO, COLNAME, TYPENAME, LENGTH, SCALE, NULLS, KEYSEQ, DEFAULT +FROM SYSCAT.COLUMNS +WHERE TABSCHEMA = 'SCHEMA_NAME' AND TABNAME = 'TABLE_NAME' +ORDER BY COLNO; +``` + +**Result:** One row per column. `COLNO` starts at 0, `NULLS` is `Y` or `N`, `KEYSEQ` is the position inside the primary key, and `DEFAULT` is a CLOB, so a long default may need a cast to read comfortably. + +**Official documentation:** [SYSCAT.COLUMNS](https://www.ibm.com/docs/en/db2/11.5?topic=views-syscatcolumns) + +## 5. Read row counts and statistics + +**Purpose:** Get a table's row count and when its statistics were collected, without scanning it. Replace `SCHEMA_NAME` and `TABLE_NAME`. + +```sql +SELECT TABSCHEMA, TABNAME, CARD, STATS_TIME FROM SYSCAT.TABLES +WHERE TABSCHEMA = 'SCHEMA_NAME' AND TABNAME = 'TABLE_NAME'; +``` + +**Result:** One row. `CARD` is the row count recorded by `RUNSTATS` and is `-1` when statistics were never collected, so it must not be reported as the real row count without checking it. `STATS_TIME` is when they were last collected. + +**Official documentation:** [SYSCAT.TABLES](https://www.ibm.com/docs/en/db2/11.5?topic=views-syscattables) + +## 6. Reconstruct a table's definition + +**Purpose:** Describe a table when a `SHOW CREATE TABLE`-style statement does not exist here. Replace `SCHEMA_NAME` and `TABLE_NAME`. + +```sql +SELECT COLNAME, TYPENAME, LENGTH, SCALE, NULLS, DEFAULT FROM SYSCAT.COLUMNS +WHERE TABSCHEMA = 'SCHEMA_NAME' AND TABNAME = 'TABLE_NAME' ORDER BY COLNO; +``` + +**Result:** The column list a CREATE TABLE would repeat, but not the DDL text: Db2 has no in-SQL statement that returns the table's CREATE. The `db2look` tool shipped with the server generates the DDL, and constraint, index, and partition details come from other catalog views such as `SYSCAT.KEYCOLUSE`, `SYSCAT.INDEXES`, and `SYSCAT.DATAPARTITIONS`. + +**Official documentation:** [SYSCAT.COLUMNS](https://www.ibm.com/docs/en/db2/11.5?topic=views-syscatcolumns) · [SYSCAT.TABLES](https://www.ibm.com/docs/en/db2/11.5?topic=views-syscattables) + +## 7. Preview up to 100 rows + +**Purpose:** Inspect a small sample while controlling the agent's output volume. Replace `SCHEMA_NAME` and `TABLE_NAME`. + +```sql +SELECT * FROM SCHEMA_NAME.TABLE_NAME FETCH FIRST 100 ROWS ONLY; +``` + +**Result:** At most 100 rows. Db2 has no `LIMIT`; `FETCH FIRST n ROWS ONLY` is the spelling it accepts, and `ORDER BY` before it is what makes the sample repeatable. + +**Official documentation:** [SELECT statement](https://www.ibm.com/docs/en/db2/11.5?topic=statements-select) diff --git a/skills/sqlx/references/downloads.md b/skills/sqlx/references/downloads.md index a22b576..368ecff 100644 --- a/skills/sqlx/references/downloads.md +++ b/skills/sqlx/references/downloads.md @@ -9,4 +9,4 @@ The CLI does not embed database drivers, the JDBC runtime or the browser UI. The Each download prints `Downloading …`, a progress line with speed and estimated time, and a final `Downloaded … in 12.3s (390 KB/s)` line on stderr. An interrupted transfer is retried up to three times. Tell the user that a first command can wait for a download instead of reporting it as a hang, and re-run the same command after a failure: components that are already installed are reused. -When the network is slow, prefetch ahead of time with `sqlx prefetch `. The accepted names are `mysql`, `mariadb`, `tidb`, `starrocks`, `doris`, `postgres`, `cockroachdb`, `yugabytedb`, `oracle`, `sqlserver`, `clickhouse`, `trino`, `ui`, `skill` and `all`; a name that shares a protocol fetches the same worker, and `all` includes the JDBC runtime and the JRE. +Run `sqlx prefetch `; the accepted names are the engine names `sqlx prefetch --help` lists, plus `ui`, `skill` and `all`. An engine whose vendor does not allow redistribution, such as `db2`, prefetches the shared Java runtime and the JDBC runner; its driver comes from `sqlx driver add` instead. diff --git a/skills/sqlx/references/gbase8s.md b/skills/sqlx/references/gbase8s.md new file mode 100644 index 0000000..dea82ba --- /dev/null +++ b/skills/sqlx/references/gbase8s.md @@ -0,0 +1,86 @@ +# GBase 8s operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `table_name`, `column_name`, and `owner_name` with actual identifiers. Unquoted identifiers are case-insensitive and are stored in lower case, so `Orders` reaches the table stored as `orders`; a double-quoted delimited identifier keeps its exact case and must then always be quoted. + +`--type gbase8s` runs on the JDBC worker and builds `jdbc:gbasedbt-sqli://host:port/database:GBASEDBTSERVER=;`, where `--database` is the database name and the default port is 9088. The server instance name comes from `--service `; without it the driver answers `GBASEDBTSERVER has to be specified` and the connection never opens. The driver is not bundled: run `sqlx driver add --type gbase8s --jar `, and `sqlx driver list` to see where each engine's driver comes from. The vendor ships a wrapper jar that contains the real `ifxjdbc.jar`, and the file to provide is that inner jar, whose driver class is `com.gbasedbt.jdbc.IfxDriver`. A username and a password are required. GBase 8s is Informix-compatible, so it uses the same `systables` and `syscolumns` catalogs and the same `SELECT ... FROM systables WHERE tabid = 1` idiom. Two limits come from the driver rather than the server: it reads a `VARCHAR` column back empty, so store varying text in `LVARCHAR`, and it does not implement the JDBC connection and multi-result calls SQLX probes with, which the worker tolerates. The documentation root is https://www.gbase.cn/, and the product documentation set is at https://docs.gbasedbt.com/gbase8s/. + +## 1. Identify the connection + +**Purpose:** Check the server version and the database the connection opened. No placeholders need replacement. + +```sql +SELECT DBINFO('version', 'full') AS engine_version FROM systables WHERE tabid = 1; +SELECT DBINFO('dbname') AS database_name FROM systables WHERE tabid = 1; +``` + +**Result:** One row each. `systables` with `tabid = 1` is the single row an expression-only select uses, because this engine has no `DUAL`; the row exists in every database, so the statement also proves the database was reached. + +**Official documentation:** [SYSTABLES](https://docs.gbasedbt.com/gbase8s/sqr/ids_sqr_072.html) + +## 2. List tables, views and sequences + +**Purpose:** Discover the objects defined in the connected database. No placeholders need replacement. + +```sql +SELECT tabname, owner, tabtype FROM systables WHERE tabtype IN ('T', 'V') ORDER BY tabname; +``` + +**Result:** One object per row, with `T` for a table, `V` for a view, `Q` for a sequence, and `E` for an external table. The system catalog tables appear with the user tables; their `tabid` is below 100, while a user object's `tabid` starts at 100. + +**Official documentation:** [SYSTABLES](https://docs.gbasedbt.com/gbase8s/sqr/ids_sqr_072.html) + +## 3. Inspect a table's columns + +**Purpose:** Read column names, type codes, and lengths for one table. Replace `table_name`. + +```sql +SELECT c.colno, c.colname, c.coltype, c.collength +FROM syscolumns c, systables t +WHERE c.tabid = t.tabid AND t.tabname = 'table_name' +ORDER BY c.colno; +``` + +**Result:** One row per column, in the order the columns were defined. `coltype` is a numeric code, and it is incremented by 256 when the column is `NOT NULL`, so a code such as 262 means a `SERIAL` column that cannot be null. + +**Official documentation:** [SYSCOLUMNS](https://docs.gbasedbt.com/gbase8s/sqr/ids_sqr_025.html) + +## 4. Read row counts and statistics + +**Purpose:** Get an approximate row count without scanning the table. Replace `table_name`. + +```sql +SELECT tabname, nrows, npused, ustlowts FROM systables WHERE tabname = 'table_name'; +``` + +**Result:** One row. `nrows` and `npused` are estimates that `UPDATE STATISTICS` maintains, and `ustlowts` is when they were last recorded; a table whose statistics were never collected reports stale values that must not be presented as an exact count. + +**Official documentation:** [SYSTABLES](https://docs.gbasedbt.com/gbase8s/sqr/ids_sqr_072.html) · [Updating catalog statistics](https://docs.gbasedbt.com/gbase8s/sqr/ids_sqr_013.html) + +## 5. Reconstruct a table's definition + +**Purpose:** Describe a table when no `SHOW CREATE TABLE` exists here. Replace `table_name`. + +```sql +SELECT c.colname, c.coltype, c.collength, t.ncols +FROM syscolumns c, systables t +WHERE c.tabid = t.tabid AND t.tabname = 'table_name' +ORDER BY c.colno; +``` + +**Result:** The catalog's view of the table, not DDL text. GBase 8s generates DDL through the `dbschema` utility rather than through SQL, so a definition that includes constraints, indexes, and storage clauses comes from that utility or from the other `sys*` catalogs. + +**Official documentation:** [Using the system catalog](https://docs.gbasedbt.com/gbase8s/sqr/ids_sqr_011.html) + +## 6. Read a bounded page with FIRST + +**Purpose:** Read a page of rows without loading a whole table. Replace `table_name` and `column_name`. + +```sql +SELECT FIRST 100 column_name FROM table_name ORDER BY column_name; +``` + +**Result:** Up to 100 rows. `FIRST` and `SKIP` are this engine's row-limiting options and belong inside the SELECT statement; `SKIP 100 FIRST 100` reads the second page. There is no `LIMIT` here, so a query copied from another engine usually has to be rewritten. + +**Official documentation:** [FIRST option](https://docs.gbasedbt.com/gbase8s/sqs/ids_sqs_0984.html) · [FIRST and SKIP as column names](https://docs.gbasedbt.com/gbase8s/sqs/ids_sqs_0986.html) diff --git a/skills/sqlx/references/hive.md b/skills/sqlx/references/hive.md new file mode 100644 index 0000000..e402340 --- /dev/null +++ b/skills/sqlx/references/hive.md @@ -0,0 +1,96 @@ +# Hive operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `database_name`, `table_name`, and `column_name` with actual identifiers. Hive is case-insensitive for unquoted identifiers and stores them in lower case, so `Orders` and `orders` reach the same table; use backticks when an identifier needs exact case, spaces, or a keyword. String literals use single quotes. + +`--type hive` runs on the JDBC worker with the driver `org.apache.hive.jdbc.HiveDriver`, and the SQLX component ships the standalone Hive 4.0.1 driver. The driver builds `jdbc:hive2://host:port/database`, so `--database` names the Hive database, often `default`, and the default port is 10000. Users are not authenticated by default: `--username` is whatever the server was configured to expect, often `hive`. Official links target the Apache Hive language manual; the documentation root is https://hive.apache.org/. + +## 1. Identify the current connection + +**Purpose:** Check the Hive version, the current database, and the session user before operating on a target. No placeholders need replacement. + +```sql +SELECT version() AS engine_version, + current_database() AS database_name, + current_user() AS session_user; +``` + +**Result:** One context row. `current_user()` reports the user name HiveServer2 received, which on an unauthenticated server is whatever `--username` sent. + +**Official documentation:** [Operators and UDFs](https://hive.apache.org/docs/latest/language/languagemanual-udf/) · [Commands](https://hive.apache.org/docs/latest/language/languagemanual-commands/) + +## 2. List databases + +**Purpose:** Discover the databases the metastore exposes. No placeholders need replacement. + +```sql +SHOW DATABASES; +``` + +**Result:** One database name per row. Metadata is served by the Hive metastore, so this statement is cheap and does not start a cluster job. + +**Official documentation:** [DDL statements](https://hive.apache.org/docs/latest/language/languagemanual-ddl/) + +## 3. List tables and views + +**Purpose:** Discover queryable objects in one database. Replace `database_name`. + +```sql +SHOW TABLES IN database_name; +``` + +**Result:** One table or view name per row. Table metadata can also be read through `information_schema`, but not every Hive version enables it. + +**Official documentation:** [DDL statements](https://hive.apache.org/docs/latest/language/languagemanual-ddl/) + +## 4. Inspect columns and storage + +**Purpose:** Read column names and types, then the location, input format, and SerDe behind them. Replace `database_name` and `table_name`. + +```sql +DESCRIBE database_name.table_name; +DESCRIBE FORMATTED database_name.table_name; +``` + +**Result:** The first statement returns one row per column in declaration order, including partition columns at the end. The second adds `Location`, `InputFormat`, `SerDe`, and table properties, which explain why a table can be empty although its files exist. + +**Official documentation:** [DDL statements](https://hive.apache.org/docs/latest/language/languagemanual-ddl/) + +## 5. Read a table's definition + +**Purpose:** Obtain the CREATE TABLE statement that reproduces a table. Replace `database_name` and `table_name`. + +```sql +SHOW CREATE TABLE database_name.table_name; +``` + +**Result:** The CREATE TABLE statement, including the partition clauses and table properties. Hive has no `SHOW CREATE VIEW`; for a view, read its definition from the metastore instead. + +**Official documentation:** [DDL statements](https://hive.apache.org/docs/latest/language/languagemanual-ddl/) + +## 6. Inspect and repair partitions + +**Purpose:** See which partitions the metastore knows, and register files that were added outside Hive. Replace `database_name` and `table_name`. + +```sql +SHOW PARTITIONS database_name.table_name; +MSCK REPAIR TABLE database_name.table_name SYNC PARTITIONS; +``` + +**Result:** One partition per row for `SHOW PARTITIONS`. `MSCK REPAIR TABLE ... SYNC PARTITIONS` registers or drops partitions so the metastore matches the file system; Hive 4 uses the `SYNC PARTITIONS` spelling, and older versions accept `MSCK REPAIR TABLE table_name`. Neither statement moves data, and a repair over many partitions is slow. + +**Official documentation:** [DDL statements](https://hive.apache.org/docs/latest/language/languagemanual-ddl/) + +## 7. Bound every large read + +**Purpose:** Read a page of rows without scanning the whole table. Replace `database_name`, `table_name`, and `column_name`. + +```sql +SELECT column_name FROM database_name.table_name LIMIT 100; +SELECT count(*) AS row_count FROM database_name.table_name; +``` + +**Result:** Up to 100 rows for the first statement, one count row for the second. Hive has no safe unbounded scan: every `SELECT` compiles to a job on the cluster and reads the table's files, so always add `LIMIT` or restrict with a partition predicate. + +**Official documentation:** [Select](https://hive.apache.org/docs/latest/language/languagemanual-select/) · [DML](https://hive.apache.org/docs/latest/language/languagemanual-dml/) diff --git a/skills/sqlx/references/informix.md b/skills/sqlx/references/informix.md new file mode 100644 index 0000000..8759061 --- /dev/null +++ b/skills/sqlx/references/informix.md @@ -0,0 +1,99 @@ +# IBM Informix operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `table_name`, `column_name`, and `owner_name` with actual identifiers. Unquoted identifiers are case-insensitive and are stored in lower case, so `Orders` reaches the table stored as `orders`; a double-quoted delimited identifier keeps its exact case and must then always be quoted. + +`--type informix` runs on the JDBC worker and builds `jdbc:informix-sqli://host:port/database:INFORMIXSERVER=;`, where `--database` is the Informix database name and the default port is 9088. The server instance name comes from `--service `; without it the URL carries no `INFORMIXSERVER` and the driver refuses the connection. Informix is not on the list of engines whose driver SQLX may redistribute: run `sqlx driver add --type informix --jar ` once, and `sqlx driver list` to see where each engine's driver comes from. A username and a password are required, and the server authorizes a remote client as `user@host`: an instance whose configuration does not know the client's host refuses every network connection with "is not known on the database server" even when the credentials are right. The documentation root is https://www.ibm.com/docs/en/informix-servers. + +## 1. Identify the connection + +**Purpose:** Check the server version and the database the connection opened. No placeholders need replacement. + +```sql +SELECT DBINFO('version', 'full') AS engine_version FROM systables WHERE tabid = 1; +SELECT DBINFO('dbname') AS database_name FROM systables WHERE tabid = 1; +``` + +**Result:** One row each. `systables` with `tabid = 1` is the single row every Informix expression-only select uses, because this engine has no `DUAL`; the row exists in every database. + +**Official documentation:** [SYSTABLES](https://www.ibm.com/docs/en/informix-servers/14.10.0?topic=tables-systables) + +## 2. List tables, views and sequences + +**Purpose:** Discover the objects defined in the connected database. No placeholders need replacement. + +```sql +SELECT tabname, owner, tabtype FROM systables WHERE tabtype IN ('T', 'V') ORDER BY tabname; +``` + +**Result:** One object per row, with `T` for a table, `V` for a view, `Q` for a sequence, and `E` for an external table. The catalog is per database, so this lists only the database named by `--database`. + +**Official documentation:** [SYSTABLES](https://www.ibm.com/docs/en/informix-servers/14.10.0?topic=tables-systables) + +## 3. Inspect a table's columns + +**Purpose:** Read column names, type codes, and lengths for one table. Replace `table_name`. + +```sql +SELECT c.colno, c.colname, c.coltype, c.collength +FROM syscolumns c, systables t +WHERE c.tabid = t.tabid AND t.tabname = 'table_name' +ORDER BY c.colno; +``` + +**Result:** One row per column. `coltype` is a numeric code, not a type name, and it is incremented by 256 when the column is `NOT NULL`, so a code such as 262 means a `SERIAL` column that cannot be null. + +**Official documentation:** [SYSCOLUMNS](https://www.ibm.com/docs/en/informix-servers/14.10.0?topic=tables-syscolumns) + +## 4. Read row counts and statistics + +**Purpose:** Get an approximate row count without scanning the table. Replace `table_name`. + +```sql +SELECT tabname, nrows, npused, ustlowts FROM systables WHERE tabname = 'table_name'; +``` + +**Result:** One row. `nrows` and `npused` are estimates maintained by `UPDATE STATISTICS`, and `ustlowts` is when they were last recorded; without `UPDATE STATISTICS` the values are stale, so they must not be presented as an exact count. + +**Official documentation:** [SYSTABLES](https://www.ibm.com/docs/en/informix-servers/14.10.0?topic=tables-systables) + +## 5. Reconstruct a table's definition + +**Purpose:** Describe a table when no `SHOW CREATE TABLE` exists here. Replace `table_name`. + +```sql +SELECT c.colname, c.coltype, c.collength, t.ncols +FROM syscolumns c, systables t +WHERE c.tabid = t.tabid AND t.tabname = 'table_name' +ORDER BY c.colno; +``` + +**Result:** The catalog's view of the table, not DDL text. Informix generates DDL through the `dbschema` utility rather than through SQL, so a definition that includes constraints, indexes, and storage clauses has to come from that utility or from the other `sys*` catalogs. + +**Official documentation:** [SYSCOLUMNS](https://www.ibm.com/docs/en/informix-servers/14.10.0?topic=tables-syscolumns) + +## 6. Page large reads with SKIP and FIRST + +**Purpose:** Read a page of rows without loading a whole table. Replace `table_name` and `column_name`. + +```sql +SELECT SKIP 0 FIRST 100 column_name FROM table_name ORDER BY column_name; +``` + +**Result:** Up to 100 rows, and `SKIP` moves the window for the next page. Informix has no `LIMIT`; `FIRST` and `SKIP` are its row-limiting clauses, and both belong in the SELECT statement rather than appended to it. + +**Official documentation:** [IBM Informix Servers documentation](https://www.ibm.com/docs/en/informix-servers) + +## 7. Monitor the server through sysmaster + +**Purpose:** Read server-wide state that the per-database catalog cannot answer. No placeholders need replacement. + +```sql +SELECT * FROM sysmaster:syssqlstat; +SELECT * FROM sysmaster:syssessions; +``` + +**Result:** One row per statement or session on the whole server instance, not just this database. `sysmaster` is the shared monitoring database and is reached with the `sysmaster:` prefix; treat it as read-only and expect a large result, so bound it as in the previous section. + +**Official documentation:** [IBM Informix Servers documentation](https://www.ibm.com/docs/en/informix-servers) diff --git a/skills/sqlx/references/kylin.md b/skills/sqlx/references/kylin.md new file mode 100644 index 0000000..6c50696 --- /dev/null +++ b/skills/sqlx/references/kylin.md @@ -0,0 +1,82 @@ +# Apache Kylin operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `project`, `table_name`, and `column_name` with actual identifiers. Kylin converts unquoted identifiers to upper case before matching, and double quotes preserve the exact case as well as escape a keyword used as a name. String literals use single quotes. + +`--type kylin` runs on the JDBC worker with the driver `org.apache.kylin.jdbc.Driver`, which builds `jdbc:kylin://host:port/project`. `--database` carries the Kylin project name, for example `learn_kylin`, not a database in the relational sense; the default port is 7070. The default credentials are `ADMIN`/`KYLIN`, and they should be changed on a real deployment. Kylin is an OLAP engine over pre-built cubes: it answers `SELECT` with the ANSI dialect, exposes tables with `SHOW TABLES`, and accepts neither writes nor DDL. Official links target the current Kylin documentation, whose root is https://kylin.apache.org/. + +## 1. Confirm the project and the query engine + +**Purpose:** Check that the project named by `--database` accepts queries before reading a cube. No placeholders need replacement. + +```sql +SELECT 1 AS connected; +``` + +**Result:** One row. A failure here means the project name, the credentials, or the JDBC port is wrong; the statement itself needs no table and no cube. + +**Official documentation:** [SQL specification](https://kylin.apache.org/docs/query/specification/sql_spec/) + +## 2. List the tables the project exposes + +**Purpose:** Discover the tables the project and its models make queryable. No placeholders need replacement. + +```sql +SHOW TABLES; +``` + +**Result:** One table name per row from the project's data source and internal tables. A table that appears here is not guaranteed to be answerable from a cube; a query that matches no cube falls back to the data source or is refused. Not every Kylin build answers this: a 5.0.2 image rejects it with "This SQL is not supported at the moment", and the tables to query then come from the cube's model in the Kylin interface. + +**Official documentation:** [Internal table](https://kylin.apache.org/docs/internaltable/intro) · [Datasource](https://kylin.apache.org/docs/datasource/intro) + +## 3. Read a bounded sample to learn the columns + +**Purpose:** See the column names and the shape of one table without pulling a whole cube. Replace `table_name`. + +```sql +SELECT * FROM table_name LIMIT 5; +``` + +**Result:** At most five rows, and the CLI prints the column list of the result, which is how an agent learns a table's columns here. Do not assume a `DESCRIBE` statement: read the column names from a sample, or from the model that defines the cube in the Kylin interface. + +**Official documentation:** [Query](https://kylin.apache.org/docs/query/insight) · [SELECT syntax](https://kylin.apache.org/docs/query/specification/sql_spec/) + +## 4. Count and aggregate rows + +**Purpose:** Read a single number that Kylin can answer from a cube. Replace `table_name` and `column_name`. + +```sql +SELECT COUNT(*) AS row_count FROM table_name; +SELECT column_name, COUNT(*) AS row_count FROM table_name GROUP BY column_name; +``` + +**Result:** One row per group. Aggregations over cube dimensions and measures are what Kylin is built for; a query whose grouping does not match a cube is answered by pushing the work down to the data source or rejected, so it can be far slower than the same query on a matching cube. + +**Official documentation:** [Basic functions](https://kylin.apache.org/docs/query/specification/functions) · [Query pushdown](https://kylin.apache.org/docs/query/push_down) + +## 5. Escape keywords and preserve exact case + +**Purpose:** Query a column whose name collides with a Kylin keyword. Replace `YEAR` and `DATES` with the real names. + +```sql +SELECT "YEAR" FROM DATES LIMIT 5; +``` + +**Result:** One row per matching record. Without the double quotes the query fails with a parse error, because the engine cannot tell the column from the keyword. The same quoting rule applies when a name was created with mixed case. + +**Official documentation:** [Identifiers and escaping](https://kylin.apache.org/docs/query/specification/sql_spec/) + +## 6. Recognise that writes and DDL are rejected + +**Purpose:** Understand what happens before an agent tries to change data. No placeholders need replacement. + +```sql +CREATE TABLE sqlx_probe (id INTEGER); +INSERT INTO table_name (column_name) VALUES ('value'); +DELETE FROM table_name WHERE id = 1; +``` + +**Result:** All three statements fail: Kylin is a query engine, and the JDBC endpoint accepts only the `SELECT` grammar. Nothing is created, written, or deleted. Build or refresh the cube in the Kylin web interface or its REST API instead, then query it again. + +**Official documentation:** [SQL specification](https://kylin.apache.org/docs/query/specification/sql_spec/) · [REST API](https://kylin.apache.org/docs/restapi/intro) diff --git a/skills/sqlx/references/presto.md b/skills/sqlx/references/presto.md new file mode 100644 index 0000000..676b470 --- /dev/null +++ b/skills/sqlx/references/presto.md @@ -0,0 +1,95 @@ +# Presto operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Presto addresses objects with three-part names: `catalog.schema.table`. Replace `catalog`, `schema`, and `table_name` with actual identifiers. Quote identifiers with double quotes when they need case or special characters, and keep string literals in single quotes. + +`--type presto` runs on the JDBC worker with the PrestoDB driver `com.facebook.presto.jdbc.PrestoDriver`, which builds `jdbc:presto://host:port/catalog/schema`. `--database` carries `catalog.schema`, and each dot becomes a slash in that URL; the default port is 8080. A username is required; a password is only sent when TLS is enabled. This type is PrestoDB, not Trino: the two are separate products with separate drivers, the drivers are not interchangeable, and a Trino server is not reachable through `--type presto` (use `references/trino.md` for Trino). Official links target the current PrestoDB documentation, whose root is https://prestodb.io/docs/current/. + +## 1. Identify the current connection + +**Purpose:** Check the server version, the current catalog, schema, and user before operating on a target. No placeholders need replacement. + +```sql +SELECT version() AS server_version, + current_catalog, + current_schema, + current_user; +``` + +**Result:** One context row. `current_schema` is NULL when the connection selected no schema. + +**Official documentation:** [System functions](https://prestodb.io/docs/current/functions/system.html) · [Session information functions](https://prestodb.io/docs/current/functions/session.html) + +## 2. List available catalogs + +**Purpose:** Discover the catalogs this cluster exposes. No placeholders need replacement. + +```sql +SHOW CATALOGS; +``` + +**Result:** One catalog per row. Write support depends on the connector: `tpch`, `tpcds`, `jmx`, and `system` are read-only, while a database connector accepts writes only when the cluster configured it. + +**Official documentation:** [SHOW CATALOGS](https://prestodb.io/docs/current/sql/show-catalogs.html) + +## 3. List schemas and tables + +**Purpose:** Discover schemas, then the queryable tables and views inside one of them. Replace `catalog` and `schema`. + +```sql +SHOW SCHEMAS FROM catalog; +SHOW TABLES FROM catalog.schema; +``` + +**Result:** One schema name per row for the first statement, one table name per row for the second. Both report one catalog only; neither spans catalogs. `information_schema` and the connector's own schemas appear alongside the data schemas. + +**Official documentation:** [SHOW SCHEMAS](https://prestodb.io/docs/current/sql/show-schemas.html) · [SHOW TABLES](https://prestodb.io/docs/current/sql/show-tables.html) + +## 4. Inspect columns + +**Purpose:** Read column names, types, and extra attributes for one table. Replace the catalog, schema, and table identifiers. + +```sql +DESCRIBE catalog.schema.table_name; +``` + +**Result:** One row per column with `Column`, `Type`, `Extra`, and `Comment`. Partition columns are listed like ordinary columns. + +**Official documentation:** [DESCRIBE](https://prestodb.io/docs/current/sql/describe.html) + +## 5. Read a table's CREATE statement + +**Purpose:** Obtain the connector-generated DDL for a table. Replace the catalog, schema, and table identifiers. + +```sql +SHOW CREATE TABLE catalog.schema.table_name; +``` + +**Result:** The CREATE TABLE statement as the connector reports it. Connectors that cannot describe DDL, such as `tpch`, return an error instead. + +**Official documentation:** [SHOW CREATE TABLE](https://prestodb.io/docs/current/sql/show-create-table.html) + +## 6. Count rows without scanning the table + +**Purpose:** Get a row count and column statistics cheaply. Replace the catalog, schema, and table identifiers. + +```sql +SHOW STATS FOR catalog.schema.table_name; +``` + +**Result:** One row per column plus a summary row whose `row_count` is the table's row count. Only connectors that maintain table statistics fill it in; a NULL `row_count` means the connector has no statistics, so fall back to `SELECT count(*)`. This is not the same statement as Trino's. + +**Official documentation:** [SHOW STATS](https://prestodb.io/docs/current/sql/show-stats.html) · [SELECT](https://prestodb.io/docs/current/sql/select.html) + +## 7. Preview up to 100 rows + +**Purpose:** Inspect a small sample while controlling the agent's output volume. Replace the catalog, schema, and table identifiers. + +```sql +SELECT * FROM catalog.schema.table_name LIMIT 100; +``` + +**Result:** At most 100 rows. Without ORDER BY the order is unspecified; add ordering by a real key column when a repeatable sample matters. Always bound a preview, because a Presto query reads from remote connectors. + +**Official documentation:** [SELECT](https://prestodb.io/docs/current/sql/select.html) diff --git a/skills/sqlx/references/sundb.md b/skills/sqlx/references/sundb.md new file mode 100644 index 0000000..42eabdd --- /dev/null +++ b/skills/sqlx/references/sundb.md @@ -0,0 +1,80 @@ +# SUNDB operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `table_name` and `column_name` with actual identifiers. SUNDB runs the Goldilocks engine, which folds unquoted identifiers to upper case, so a name created unquoted is stored in upper case and must be written that way; double quotes preserve exact case and let a keyword stand as a name. System catalog views sit under the `SYSTEM_` prefix. + +`--type sundb` runs on the JDBC worker and builds `jdbc:goldilocks://host:port/database`, because the driver is the Goldilocks engine's and keeps its URL scheme. `--database` is the database name, the default database is `goldilocks`, the system user is `sys`, and the default system password is `gliese`; the default port is 22581. The driver is not bundled: run `sqlx driver add --type sundb --jar ` with the vendor's driver, which the SUNDB/Goldilocks image ships at `/goldilocks_home/lib/goldilocks8.jar`. The public vendor image's license expired in 2022, so testing against a real server needs a licensed installation. The vendor publishes manuals as PDFs rather than a browsable documentation site: the documentation root is https://www.sunjesoft.co.kr/en/, and the product site is https://www.sundb.com.cn/. + +## 1. Confirm the connection and list the system accounts + +**Purpose:** Check that the connection answers, then see which accounts exist. No placeholders need replacement. + +```sql +SELECT 1 AS connected FROM DUAL; +SELECT * FROM SYSTEM_.USERS; +``` + +**Result:** One row for the first statement and one row per account for the second. `DUAL` is the one-row table a statement selects from when it needs a `FROM` clause but no data, and `SYSTEM_.USERS` is the account catalog of the Goldilocks engine. + +**Official documentation:** [GOLDILOCKS 22c.1 User Manual](https://www.sunjesoft.co.kr/en/_files/ugd/216533_1b07b250f2c94a59a908abef0a0c5187.pdf) + +## 2. List the tables the server exposes + +**Purpose:** Discover the tables and views the catalog records. No placeholders need replacement. + +```sql +SELECT * FROM SYSTEM_.TABLES; +``` + +**Result:** One row per table object, and the CLI prints the column list, which tells you which column carries the schema and which carries the table name. Read that column list before filtering, then restrict the next query by schema so the catalog is not read in full. + +**Official documentation:** [GOLDILOCKS 22c.1 User Manual](https://www.sunjesoft.co.kr/en/_files/ugd/216533_1b07b250f2c94a59a908abef0a0c5187.pdf) + +## 3. Inspect a table's columns + +**Purpose:** Read the column catalog so the columns of one table can be seen. Replace `TABLE_NAME` once the catalog's table-name column is known, keeping its stored upper-case spelling. + +```sql +SELECT * FROM SYSTEM_.COLUMNS; +``` + +**Result:** One row per column of every table. The CLI prints the column list, so add a predicate such as `WHERE TABLE_NAME = 'TABLE_NAME'` in the next run, using the name the output shows for the table column; upper-case it unless the table was created with a quoted name. + +**Official documentation:** [GOLDILOCKS 3.2 User Manual](https://www.sunjesoft.co.kr/en/_files/ugd/216533_17279666fbdb415abc42c1087965007a.pdf) + +## 4. Read a bounded page + +**Purpose:** Read a page of rows without loading a whole table. Replace `table_name` and `column_name`. + +```sql +SELECT column_name FROM table_name ORDER BY column_name LIMIT 100; +``` + +**Result:** Up to 100 rows. Goldilocks accepts `LIMIT`; a server that rejects it accepts the standard `FETCH FIRST 100 ROWS ONLY` instead. Always bound a read, because the same statement without a bound returns every row and the CLI then has to store that whole result. + +**Official documentation:** [GOLDILOCKS 22c.1 User Manual](https://www.sunjesoft.co.kr/en/_files/ugd/216533_1b07b250f2c94a59a908abef0a0c5187.pdf) + +## 5. Count rows + +**Purpose:** Get a table's row count when a sample is not representative. Replace `table_name`. + +```sql +SELECT COUNT(*) AS row_count FROM table_name; +``` + +**Result:** One row with the exact count, which reads the table. There is no separate statistics catalog to consult first, so decide from the table's size whether the count is worth running. + +**Official documentation:** [GOLDILOCKS 3.2 User Manual](https://www.sunjesoft.co.kr/en/_files/ugd/216533_17279666fbdb415abc42c1087965007a.pdf) + +## 6. Evaluate expressions with DUAL + +**Purpose:** Run a statement that has no table, such as a connection check after a driver change. No placeholders need replacement. + +```sql +SELECT CURRENT_TIMESTAMP AS now FROM DUAL; +``` + +**Result:** One row. `DUAL` works for a function probe as well as for `SELECT 1`, which is the cheapest way to confirm that a freshly registered `goldilocks8.jar` really connects before a longer query. + +**Official documentation:** [GOLDILOCKS manuals](https://www.sunjesoft.co.kr/en/) · [SUNDB product site](https://www.sundb.com.cn/) diff --git a/skills/sqlx/references/xugu.md b/skills/sqlx/references/xugu.md new file mode 100644 index 0000000..a39d93a --- /dev/null +++ b/skills/sqlx/references/xugu.md @@ -0,0 +1,83 @@ +# XuguDB operations + +Each SQL block below performs one operation. Submit it through `sqlx sql execute --datasource --command "..."`. Repeat `--command` in the same invocation when operations need to share connection state. + +Replace `OWNER_NAME`, `TABLE_NAME`, and `COLUMN_NAME` with actual identifiers. XuguDB folds unquoted identifiers to upper case, so a name created unquoted is stored in upper case and is only reachable in that spelling; keep the stored case when you write the query. + +`--type xugu` runs on the JDBC worker with the driver `com.xugu.cloudjdbc.Driver`, which builds `jdbc:xugu://host:port/database`. `--database` is the database name, and `SYSTEM` is the system database; the default port is 5138. A username and a password are required, and the account must exist in the database being opened. XuguDB presents Oracle-style catalogs here: `ALL_TABLES`, `ALL_TAB_COLUMNS`, `USER_TABLES`, and `DUAL`. The vendor documentation root is https://www.xugudb.com/, and the documentation site is https://help.xugudb.com/. + +## 1. List the schemas the account can see + +**Purpose:** Find the owners whose tables are visible, before naming one. No placeholders need replacement. + +```sql +SELECT DISTINCT OWNER AS schema_name FROM ALL_TABLES ORDER BY OWNER; +``` + +**Result:** One owner per row. This is the account's visible set, not every schema in the database: a schema whose tables are not granted to the account does not appear. + +**Official documentation:** [ALL_TABLES](https://help.xugudb.com/content/reference/system-view/all/all_tables) · [Schemas](https://help.xugudb.com/content/reference/object/schema) + +## 2. List tables in a schema + +**Purpose:** Find the table to inspect or query. Replace `OWNER_NAME`, or drop the filter to use `USER_TABLES` for the account's own tables. + +```sql +SELECT OWNER, TABLE_NAME FROM ALL_TABLES WHERE OWNER = 'OWNER_NAME' ORDER BY TABLE_NAME; +SELECT TABLE_NAME FROM USER_TABLES ORDER BY TABLE_NAME; +``` + +**Result:** One table per row. `ALL_TABLES` covers the tables the account may read, `USER_TABLES` only those it owns. An empty `ALL_TABLES` result can mean a case mismatch in `OWNER_NAME` or missing privileges. + +**Official documentation:** [ALL_TABLES](https://help.xugudb.com/content/reference/system-view/all/all_tables) · [Tables](https://help.xugudb.com/content/reference/object/table/introduction) + +## 3. Inspect a table's columns + +**Purpose:** Read column names, types, lengths, and nullability of one table. Replace `OWNER_NAME` and `TABLE_NAME`. + +```sql +SELECT COLUMN_ID, COLUMN_NAME, DATA_TYPE, DATA_LENGTH, DATA_PRECISION, DATA_SCALE, NULLABLE +FROM ALL_TAB_COLUMNS +WHERE OWNER = 'OWNER_NAME' AND TABLE_NAME = 'TABLE_NAME' +ORDER BY COLUMN_ID; +``` + +**Result:** One row per column in declaration order. The catalog views keep Oracle's column names, and `DATA_LENGTH` is a byte length, so it must not be read as a character count for a character column. + +**Official documentation:** [System views](https://help.xugudb.com/content/reference/system-view/all/all_tables) · [Tables](https://help.xugudb.com/content/reference/object/table/introduction) + +## 4. Read a bounded page + +**Purpose:** Read rows without loading a whole table, with a repeatable order. Replace `OWNER_NAME`, `TABLE_NAME`, and `COLUMN_NAME`. + +```sql +SELECT COLUMN_NAME FROM OWNER_NAME.TABLE_NAME ORDER BY COLUMN_NAME LIMIT 100 OFFSET 0; +``` + +**Result:** Up to 100 rows. The result-set clause also accepts `FETCH FIRST n ROWS ONLY`, so a query copied from Oracle usually runs. Names are resolved case-sensitively against the stored upper-case spelling unless the identifier is quoted. + +**Official documentation:** [Restricting a result set](https://help.xugudb.com/content/reference/sql/select/resultset-restricted) · [SELECT](https://help.xugudb.com/content/reference/sql/select/select) + +## 5. Count rows + +**Purpose:** Get a table's row count when the sample is not representative. Replace `OWNER_NAME` and `TABLE_NAME`. + +```sql +SELECT COUNT(*) AS row_count FROM OWNER_NAME.TABLE_NAME; +``` + +**Result:** One row with the exact count, which reads the table. Use the catalog instead when an estimate is enough, or the table is large. + +**Official documentation:** [SELECT](https://help.xugudb.com/content/reference/sql/select/select) + +## 6. Evaluate an expression with DUAL + +**Purpose:** Run a statement that has no table, such as a connection check or a function probe. No placeholders need replacement. + +```sql +SELECT 1 AS connected FROM DUAL; +``` + +**Result:** One row. `DUAL` is the single-row table a query uses when it needs a `FROM` clause but no data, which is the Oracle spelling an agent should expect here. + +**Official documentation:** [FROM clause](https://help.xugudb.com/content/reference/sql/select/from) · [Databases](https://help.xugudb.com/content/reference/object/database) diff --git a/tests/compose.yaml b/tests/compose.yaml index 2c604c4..039bd0a 100644 --- a/tests/compose.yaml +++ b/tests/compose.yaml @@ -149,6 +149,74 @@ services: timeout: 5s retries: 60 + # Presto, Hive, Kylin, XuguDB, Db2 and Informix are the JDBC engines added after H2; each one + # stages its driver through scripts/jdbc-fixture.sh before tests/databases.py runs. + presto: + image: prestodb/presto:latest + ports: ["127.0.0.1:28083:8080"] + healthcheck: + # /v1/info answers before the coordinator accepts queries, so wait for starting:false. + test: ["CMD-SHELL", "curl -fsS http://127.0.0.1:8080/v1/info | grep -q '\"starting\":false'"] + interval: 5s + timeout: 5s + retries: 60 + hive: + image: apache/hive:4.0.1 + environment: + # The image starts HiveServer2 with an embedded metastore only when asked for it. + SERVICE_NAME: hiveserver2 + ports: ["127.0.0.1:21000:10000"] + healthcheck: + # The image has no client for the metastore schema, so wait for the listener. + test: ["CMD-SHELL", "timeout 3 bash -c '/dev/null 2>&1"] + interval: 15s + timeout: 20s + retries: 80 + informix: + image: ibmcom/informix-developer-database:latest + environment: + LICENSE: accept + # The image keeps its documented default password (in4mix) for the informix user; the + # DB_INFORMIX_PASSWORD variable does not change it. + DB_INFORMIX_PASSWORD: in4mix + ports: ["127.0.0.1:29088:9088"] + healthcheck: + test: ["CMD-SHELL", "timeout 3 bash -c ' = { sqlite: "SQLite", duckdb: "DuckDB", h2: "H2", + presto: "Presto", + hive: "Hive", + kylin: "Apache Kylin", + xugu: "XuguDB", + db2: "IBM Db2", + informix: "IBM Informix", + sundb: "SUNDB", + gbase8s: "GBase 8s", }; /** Engines that open a local file, or a local file until a host is given. */ export function isFileEngine(kind: string): boolean { @@ -100,8 +108,12 @@ export async function datasourcePage( ["Host", c.host], ["Port", String(c.port)], [ - c.database_type === "oracle" ? "Service name" : "Database", - c.database_type === "oracle" ? c.service : c.database || "Not specified", + c.database_type === "oracle" || c.database_type === "informix" || c.database_type === "gbase8s" + ? "Service name" + : "Database", + c.database_type === "oracle" || c.database_type === "informix" || c.database_type === "gbase8s" + ? c.service + : c.database || "Not specified", ], [ "Connection security", diff --git a/ui/src/setup.ts b/ui/src/setup.ts index 6de9711..f8be2f9 100644 --- a/ui/src/setup.ts +++ b/ui/src/setup.ts @@ -65,6 +65,14 @@ export async function setupPage( ["sqlite", "SQLite"], ["duckdb", "DuckDB"], ["h2", "H2"], + ["presto", "Presto"], + ["hive", "Hive"], + ["kylin", "Apache Kylin"], + ["xugu", "XuguDB"], + ["db2", "IBM Db2"], + ["informix", "IBM Informix"], + ["sundb", "SUNDB"], + ["gbase8s", "GBase 8s"], ], setup.connection.database_type, ); @@ -153,9 +161,11 @@ export async function setupPage( if (hostLabel && engine === "h2") hostLabel.textContent = "Host (empty for a local file)"; context.textContent = file ? `${kind.input.selectedOptions[0].text} · ${database.input.value}` - : `${kind.input.selectedOptions[0].text} · ${host.input.value}:${port.input.value} · ${engine === "oracle" ? service.input.value : database.input.value}`; - service.wrapper.hidden = engine !== "oracle"; - service.input.required = engine === "oracle"; + : `${kind.input.selectedOptions[0].text} · ${host.input.value}:${port.input.value} · ${service.input.value || database.input.value}`; + // Oracle names a service, and the Informix-derived engines name their server instance here. + const needsService = engine === "oracle" || engine === "informix" || engine === "gbase8s"; + service.wrapper.hidden = !needsService; + service.input.required = needsService; database.wrapper.hidden = engine === "oracle"; password.wrapper.hidden = passwordAction.input.value !== "replace"; password.input.required = passwordAction.input.value === "replace";