From e9f444c4ea5416ae7a1ca21fc467a4554d4a5a09 Mon Sep 17 00:00:00 2001 From: Dimitri Fontaine Date: Mon, 14 Sep 2026 01:28:11 +0000 Subject: [PATCH] fix(v4): MySQL/MSSQL migration bugs found while capturing the Migration course MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit MySQL: - Create target schemas before sequences, ENUM types and tables. ENUM types were created before their schema, so the table using them was lost, along with its FKs and sequence reset. - AUTO_INCREMENT integer columns become serial/bigserial (mirroring the CL default cast rules), instead of plain integer columns with no sequence. - DECODING TABLE NAMES MATCHING … AS now reads text columns as CAST(… AS BINARY) and decodes the stored bytes with that charset. SET NAMES asked the server to convert, which double-encoded the very data the clause is meant to repair. PostgreSQL target: - Fix the sequence reset off-by-one: setval(seq, MAX(col), true), or setval(seq, 1, false) on an empty table, so the first insert no longer collides with the last migrated row. - "Reset Sequences" counts sequences actually reset, not tables visited. SQL Server: - DEFAULT (N'new') lands as 'new', not as the literal text N'new'; doubled quotes in string literal defaults are unescaped. - Translate T-SQL default functions: GETDATE(), GETUTCDATE(), SYSDATETIME(), SYSUTCDATETIME(), SYSDATETIMEOFFSET() → CURRENT_TIMESTAMP, NEWID(), NEWSEQUENTIALID() → gen_random_uuid(). Logging: - Mask passwords in the Source/Target/Connected log lines, for pgloader URIs (user:****@) and JDBC URLs (password=****). Tests: unit tests for each fix; E2E coverage in the mysql suite (no more BEFORE LOAD DO create schema workaround, sequences, decoding) and a v4-only tsqldefaults database in the mssql suite. Baselines that recorded the missing nextval() defaults are updated. Co-Authored-By: Claude Opus 5 --- clojure/Makefile | 5 +- clojure/src/pgloader/core.clj | 70 +++++++--- clojure/src/pgloader/ddl/common.clj | 64 +++++++-- clojure/src/pgloader/log.clj | 12 ++ clojure/src/pgloader/source/mssql.clj | 32 ++++- clojure/src/pgloader/source/mssql.sql | 2 - clojure/src/pgloader/source/mysql.clj | 130 +++++++++++++----- clojure/src/pgloader/source/pgsql.clj | 3 +- clojure/test/pgloader/ddl_test.clj | 53 +++++++ clojure/test/pgloader/log_test.clj | 26 ++++ clojure/test/pgloader/source/mssql_test.clj | 30 ++++ clojure/test/pgloader/source/mysql_test.clj | 49 +++++++ clojure/tests/mssql/Makefile | 9 +- .../tests/mssql/expected/13-tsql-defaults.out | 21 +++ clojure/tests/mssql/init.sql | 32 +++++ clojure/tests/mssql/mssql-tsql-defaults.load | 8 ++ clojure/tests/mssql/sql/13-tsql-defaults.sql | 20 +++ .../02-datetime-precision.mariadb.out | 6 +- .../expected/02-datetime-precision.out | 6 +- .../expected/03-bit-defaults.mariadb.out | 6 +- .../expected/03-bit-defaults.out | 6 +- clojure/tests/mysql/Makefile | 6 +- .../tests/mysql/mytest/expected/01-tables.out | 3 +- .../mysql/mytest/expected/01-tables.v3.out | 3 +- .../mysql/mytest/expected/03-indexes.out | 3 +- .../mysql/mytest/expected/03-indexes.v3.out | 3 +- .../mytest/expected/06-datetime-precision.out | 6 +- .../mysql/mytest/expected/07-bit-defaults.out | 6 +- .../tests/mysql/mytest/expected/12-pkeys.out | 4 +- .../mysql/mytest/expected/12-pkeys.v3.out | 4 +- .../mysql/mytest/expected/18-sequences.out | 15 ++ .../mysql/mytest/expected/18-sequences.v3.out | 15 ++ .../mysql/mytest/expected/19-decoding-as.out | 6 + clojure/tests/mysql/mytest/mytest.load | 4 +- clojure/tests/mysql/mytest/mytest.sql | 20 +++ .../tests/mysql/mytest/sql/18-sequences.sql | 26 ++++ .../tests/mysql/mytest/sql/19-decoding-as.sql | 5 + 37 files changed, 618 insertions(+), 101 deletions(-) create mode 100644 clojure/test/pgloader/log_test.clj create mode 100644 clojure/test/pgloader/source/mssql_test.clj create mode 100644 clojure/test/pgloader/source/mysql_test.clj create mode 100644 clojure/tests/mssql/expected/13-tsql-defaults.out create mode 100644 clojure/tests/mssql/mssql-tsql-defaults.load create mode 100644 clojure/tests/mssql/sql/13-tsql-defaults.sql create mode 100644 clojure/tests/mysql/mytest/expected/18-sequences.out create mode 100644 clojure/tests/mysql/mytest/expected/18-sequences.v3.out create mode 100644 clojure/tests/mysql/mytest/expected/19-decoding-as.out create mode 100644 clojure/tests/mysql/mytest/sql/18-sequences.sql create mode 100644 clojure/tests/mysql/mytest/sql/19-decoding-as.sql diff --git a/clojure/Makefile b/clojure/Makefile index 9c917c58..07087261 100644 --- a/clojure/Makefile +++ b/clojure/Makefile @@ -53,7 +53,10 @@ test-unit: -n pgloader.load-file.parser-test \ -n pgloader.transforms-test \ -n pgloader.pg-service-test \ - -n pgloader.cli-test + -n pgloader.cli-test \ + -n pgloader.log-test \ + -n pgloader.source.mssql-test \ + -n pgloader.source.mysql-test # ─── E2E integration tests ──────────────────────────────────────────────────── # All suite management lives in tests/Makefile. diff --git a/clojure/src/pgloader/core.clj b/clojure/src/pgloader/core.clj index 45518433..c5c7e802 100644 --- a/clojure/src/pgloader/core.clj +++ b/clojure/src/pgloader/core.clj @@ -214,6 +214,34 @@ (.rollback pg-conn) (log/warn (str label " failed (skipping): " (.getMessage e))))))) +(defn- reset-sequences! + "Execute reset-sequences-sql statements, each in its own transaction. + Returns how many sequences were actually reset: setval() yields NULL for + columns that do not own a sequence, and those are not counted. + Errors are logged as warnings and skipped." + [^Connection pg-conn sqls label] + (reduce (fn [n sql] + (try + (let [row (first (jdbc/execute! pg-conn [sql]))] + (.commit pg-conn) + (if (some? (first (vals row))) (inc n) n)) + (catch Exception e + (.rollback pg-conn) + (log/warn (str label " failed (skipping): " (.getMessage e))) + n))) + 0 sqls)) + +(defn- create-schemas! + "Create every target schema up front, before sequences, types and tables." + [^Connection pg-conn schemas] + (when-let [sqls (seq (ddl/create-schemas-sql schemas))] + (stats/new-entry! :pre "Create Schemas") + (let [start (System/nanoTime)] + (exec-post-ddl! pg-conn sqls "CREATE SCHEMA") + (stats/update-entry! :pre "Create Schemas" + :rows (count sqls) + :total-nanos (- (System/nanoTime) start))))) + (defn- pg-major-version "Return the PostgreSQL major version as an integer (e.g. 13, 14, 15)." [^Connection pg-conn] @@ -422,9 +450,9 @@ ;; ── All other load types ────────────────────────────────────────────────── (let [source-uri (:source cmd) target-uri (get-in cmd [:target :target-uri]) - _ (log/debug (str "Connecting to PostgreSQL at " (:raw target-uri))) + _ (log/debug (str "Connecting to PostgreSQL at " (plog/redact-uri (:raw target-uri)))) ^Connection pg-conn (postgres-connection target-uri) - _ (log/info (str "Connected to PostgreSQL at " (:raw target-uri))) + _ (log/info (str "Connected to PostgreSQL at " (plog/redact-uri (:raw target-uri)))) source-overrides (select-keys source-uri [:inline-data]) commands-filters (:filters cmd) table-filter (when commands-filters @@ -441,8 +469,8 @@ source (source-from-uri source-uri table-spec (:with-options cmd) source-overrides (:decoding-as cmd)) verbose (or (:debug opts) (:verbose opts) false)] (log/info "pgloader v4") - (log/info "Source:" (source-name source)) - (log/info "Target:" (:raw target-uri)) + (log/info "Source:" (plog/redact-uri (source-name source))) + (log/info "Target:" (plog/redact-uri (:raw target-uri))) (if copy/*dry-run* ;; Dry run: verify both connections are reachable, then stop. ;; Mirrors v3 behaviour: no catalog fetch, no DDL, no COPY. @@ -467,8 +495,8 @@ (when-let [mysql-params (seq (filter :is-mysql (:set-parameters cmd)))] (log/debug "Sending MySQL SET parameters to source connection") (mysql-source/execute-set-params! source mysql-params))) - (let [_ (log/debug (str "Connecting to source: " (source-name source))) - _ (log/info (str "Fetching catalog from " (source-name source))) + (let [_ (log/debug (str "Connecting to source: " (plog/redact-uri (source-name source)))) + _ (log/info (str "Fetching catalog from " (plog/redact-uri (source-name source)))) fetch-t0 (System/nanoTime) cat (catalog source) ;; If MATERIALIZE ALL VIEWS, append view catalog entries to table catalog. @@ -707,10 +735,15 @@ ;; Ensure extensions required by column defaults exist ;; (e.g. pgcrypto for gen_random_uuid() on PG < 13). (ensure-uuid-extension! pg-conn cat) - ;; Create sequences before tables so that NEXT VALUE FOR - ;; defaults (translated to nextval()) resolve correctly. - (when (= :mssql (:type source-uri)) - (when-let [seqs (seq (mssql-source/catalog-sequences source))] + ;; Create schemas first: sequences, ENUM types and tables + ;; are all created inside them. + (let [seqs (when (= :mssql (:type source-uri)) + (seq (mssql-source/catalog-sequences source)))] + (create-schemas! pg-conn (concat (map #(or (:schema %) "public") cat) + (map :schema seqs))) + ;; Create sequences before tables so that NEXT VALUE FOR + ;; defaults (translated to nextval()) resolve correctly. + (when seqs (log/info (str "Creating " (count seqs) " sequence(s) from MS SQL")) (run-ddl-tx pg-conn (ddl/create-sequences-sql seqs)))) (stats/new-entry! :pre "Create tables") @@ -1049,15 +1082,16 @@ (log/info "Resetting sequences") (stats/new-entry! :post "Reset Sequences") (let [start (System/nanoTime) - n (atom 0)] - (doseq [t cat] - (let [schema (or (:schema t) "public") - table (:table-name t)] - (when-let [seqs (seq (ddl/reset-sequences-sql schema table (:columns t)))] - (exec-post-ddl! pg-conn seqs (str "SEQUENCE for " table)) - (swap! n inc)))) + n (reduce (fn [n t] + (+ n (reset-sequences! + pg-conn + (ddl/reset-sequences-sql (or (:schema t) "public") + (:table-name t) + (:columns t)) + (str "SEQUENCE for " (:table-name t))))) + 0 cat)] (stats/update-entry! :post "Reset Sequences" - :rows @n :total-nanos (- (System/nanoTime) start))))) + :rows n :total-nanos (- (System/nanoTime) start))))) ;; Execute AFTER LOAD DO statements — skip when the load failed (#930). (when-let [after-cmds (and (not @load-failed) (seq (:after-load cmd)))] (log/debug "Executing AFTER LOAD DO commands") diff --git a/clojure/src/pgloader/ddl/common.clj b/clojure/src/pgloader/ddl/common.clj index ce3a6623..6a60ce86 100644 --- a/clojure/src/pgloader/ddl/common.clj +++ b/clojure/src/pgloader/ddl/common.clj @@ -17,6 +17,28 @@ ([schema table] (str (identifier-quote schema) "." (identifier-quote table)))) +(defn- auto-increment? + "True when a column's :extra marks it as auto-increment." + [extra] + (str/includes? (str/lower-case (str extra)) "auto_increment")) + +(defn- serial-type? + [^String pg-type] + (boolean (re-matches #"(?i)(small|big)?serial" (str pg-type)))) + +(declare pg-type-for) + +(defn- auto-increment-pg-type + "Map an auto-increment integer column to serial/bigserial, mirroring the CL + default cast rules: types that map to bigint (or numeric, for bigint + unsigned) become bigserial, all smaller integer types become serial. + Returns nil when the source type is not an integer type." + [^String mysql-type] + (when (re-find #"(?i)^(tiny|small|medium|big)?int" mysql-type) + (case (pg-type-for mysql-type nil) + ("bigint" "numeric") "bigserial" + "serial"))) + (defn- pg-type-for "Map a MySQL type name to PostgreSQL type. Preserves precision modifiers for temporal types (#1629)." @@ -27,6 +49,9 @@ ;; Extract precision modifier like (6) from datetime(6) typemod (re-find #"\(\d+\)" mysql-type)] (cond + ;; AUTO_INCREMENT integers own a sequence on the target + (and (auto-increment? extra) (auto-increment-pg-type mysql-type)) + (auto-increment-pg-type mysql-type) ;; Pass-through: native PostgreSQL types emitted by non-MySQL sources (= lower "uuid") "uuid" (= lower "xml") "xml" @@ -202,6 +227,8 @@ (re-find #"(?i)^(timestamp|date|time)" (or pg-type "")) (re-matches #"^-?\d+$" (str coerced-default))) default-str (when (and coerced-default + ;; serial types come with their own nextval() default + (not (serial-type? pg-type)) (not= "NULL" (str coerced-default)) (not= "" (str coerced-default)) (not zero?) @@ -209,7 +236,7 @@ (str " DEFAULT " (format-default coerced-default)))] (str " " quoted-name " " pg-type (when (and (false? is-nullable) - (not (str/includes? (str extra) "auto_increment")) + (not (auto-increment? extra)) (not= "NULL" (str column-default)) (not zero?)) " NOT NULL") @@ -600,38 +627,55 @@ quoted-schema "." quoted-fn "();")] [fn-sql trg-sql])))) +(defn- sql-literal + [^String s] + (str "'" (str/replace s "'" "''") "'")) + (defn reset-sequences-sql "Generate SELECT setval() for auto-increment columns. Calls pg_catalog.setval with MAX(col) to advance the sequence - past any data that was bulk-loaded (bypassing the sequence). + past any data that was bulk-loaded (bypassing the sequence), so that the + next nextval() returns MAX(col) + 1, or 1 on an empty table. + + Each statement returns a single row whose setval value is NULL when the + column does not own a sequence (pg_get_serial_sequence returns NULL). Returns a vector of SQL strings (one per auto-increment column)." [schema table-name columns] (let [quoted-fqname (quote-fqname schema table-name)] (vec (keep (fn [col] (when (and (:column-name col) - (:extra col) - (str/includes? (str/lower-case (:extra col)) "auto_increment") + (auto-increment? (:extra col)) ;; Only reset sequences for integer-typed columns. ;; Cast rules may change the PG type (e.g. int → text); - ;; non-integer MAX() causes COALESCE type mismatch. + ;; non-integer MAX() causes a type mismatch. (let [src (or (:source-column-type col) (:column-type col) "") ct (or (:column-type col) "") pg-type (str/lower-case (if (not= src ct) ct - (pg-type-for ct nil)))] + (pg-type-for ct (:extra col))))] (some #(str/starts-with? pg-type %) ["integer" "bigint" "bigserial" "smallint" "serial" "int2" "int4" "int8"]))) - (let [col-name (:column-name col) - quoted-col (identifier-quote col-name)] + (let [quoted-col (identifier-quote (:column-name col)) + max-col (str "MAX(" quoted-col ")")] + ;; setval(seq, max, true) makes nextval() return max + 1; + ;; an empty table gets setval(seq, 1, false) so nextval() returns 1. (str "SELECT pg_catalog.setval(" - "pg_get_serial_sequence('" quoted-fqname "', '" col-name "')" - ", COALESCE(MAX(" quoted-col "), 1), false)" + "pg_get_serial_sequence(" (sql-literal quoted-fqname) + ", " (sql-literal (:column-name col)) ")" + ", GREATEST(" max-col ", 1), " max-col " IS NOT NULL)" " FROM " quoted-fqname ";\n")))) columns)))) +(defn create-schemas-sql + "Generate CREATE SCHEMA IF NOT EXISTS statements, one per distinct schema, + in first-seen order." + [schemas] + (mapv #(str "CREATE SCHEMA IF NOT EXISTS " (identifier-quote %) ";") + (distinct (remove nil? schemas)))) + (defn create-sequence-sql "Generate DROP … / CREATE SEQUENCE SQL for an MSSQL sequence descriptor. seq-map keys: :schema :name :start :increment :min :max :cycle? :cache :current diff --git a/clojure/src/pgloader/log.clj b/clojure/src/pgloader/log.clj index 63e6a7ee..16157284 100644 --- a/clojure/src/pgloader/log.clj +++ b/clojure/src/pgloader/log.clj @@ -1,8 +1,20 @@ (ns pgloader.log + (:require [clojure.string :as str]) (:import [java.util Locale])) (set! *warn-on-reflection* true) +(defn redact-uri + "Mask passwords in a connection URI or JDBC URL so it can be logged: + postgresql://user:secret@host/db → postgresql://user:****@host/db + jdbc:sqlserver://host;password=secret → jdbc:sqlserver://host;password=**** + jdbc:postgresql://host/db?password=x → jdbc:postgresql://host/db?password=****" + [s] + (when s + (-> (str s) + (str/replace #"(://[^:/@;?#\s]*):[^/?#;\s]*@" "$1:****@") + (str/replace #"(?i)([;?&]password=)[^;&\s]*" "$1****")))) + (defn- locale-format "Like clojure.core/format but with an explicit Locale to force '.' as decimal separator." [locale fmt & args] diff --git a/clojure/src/pgloader/source/mssql.clj b/clojure/src/pgloader/source/mssql.clj index 87661007..41e4cf06 100644 --- a/clojure/src/pgloader/source/mssql.clj +++ b/clojure/src/pgloader/source/mssql.clj @@ -1,5 +1,6 @@ (ns pgloader.source.mssql (:require [pgloader.source.protocol :refer [Source]] + [pgloader.log :as plog] [hugsql.core :as hugsql] [clojure.string :as str] [next.jdbc :as jdbc] @@ -81,22 +82,47 @@ "[schema].[sequence] syntax, dropping: " s)) nil))) +(def ^:private tsql-default-functions + "T-SQL niladic functions used in column defaults, and their PostgreSQL + equivalents. Keys are lower-case." + {"getdate()" "CURRENT_TIMESTAMP" + "getutcdate()" "CURRENT_TIMESTAMP" + "sysdatetime()" "CURRENT_TIMESTAMP" + "sysutcdatetime()" "CURRENT_TIMESTAMP" + "sysdatetimeoffset()" "CURRENT_TIMESTAMP" + "current_timestamp" "CURRENT_TIMESTAMP" + "newid()" "gen_random_uuid()" + "newsequentialid()" "gen_random_uuid()"}) + +(defn- unquote-string-literal + "Return the value of a T-SQL string literal ('abc' or N'abc'), with doubled + quotes unescaped, or nil when s is not a single string literal." + [^String s] + (when-let [[_ body] (re-matches #"(?s)[Nn]?'((?:[^']|'')*)'" s)] + (str/replace body "''" "'"))) + (defn- sanitize-default "Normalise MSSQL column defaults for PostgreSQL: - NEXT VALUE FOR [schema].[seq] → nextval('schema.seq') (#1497) + - T-SQL functions (GETDATE(), SYSUTCDATETIME(), NEWID(), …) → PostgreSQL + - String literals 'abc' and Unicode literals N'abc' → their value - CONVERT(…) expressions not already translated by mssql.sql → nil (#1409) - Empty-string defaults on numeric columns → nil (#1163)" [default pg-type] (when default (let [trimmed (str/trim default) - lower (str/lower-case trimmed)] + lower (str/lower-case trimmed) + literal (unquote-string-literal trimmed)] (cond ;; SQL Server sequence default → PostgreSQL nextval() (str/starts-with? lower "next value for") (translate-next-value-for trimmed) + (contains? tsql-default-functions lower) (get tsql-default-functions lower) ;; Any remaining CONVERT(…) that mssql.sql didn't map to a keyword (str/starts-with? lower "convert(") nil ;; Empty string on a numeric target type - (and (= trimmed "") (contains? numeric-pg-types pg-type)) nil + (and (or (= trimmed "") (= literal "")) + (contains? numeric-pg-types pg-type)) nil + literal literal :else default)))) (defn- connection @@ -353,4 +379,4 @@ (defn create-source [uri-map _table-spec] (let [conn (connection uri-map)] - (->MSSQLSource conn (:raw uri-map)))) + (->MSSQLSource conn (plog/redact-uri (:raw uri-map))))) diff --git a/clojure/src/pgloader/source/mssql.sql b/clojure/src/pgloader/source/mssql.sql index 33887d04..a8e86e1f 100644 --- a/clojure/src/pgloader/source/mssql.sql +++ b/clojure/src/pgloader/source/mssql.sql @@ -38,7 +38,6 @@ SELECT c.COLUMN_NAME, WHEN SUBSTRING(c.COLUMN_DEFAULT, 3, LEN(c.COLUMN_DEFAULT) - 4) LIKE 'convert(%varchar%,getdate(),%)' THEN 'CURRENT_DATE' WHEN SUBSTRING(c.COLUMN_DEFAULT, 3, LEN(c.COLUMN_DEFAULT) - 4) = 'getdate()' THEN 'CURRENT_TIMESTAMP' WHEN SUBSTRING(c.COLUMN_DEFAULT, 3, LEN(c.COLUMN_DEFAULT) - 4) = 'sysdatetimeoffset()' THEN 'CURRENT_TIMESTAMP' - WHEN SUBSTRING(c.COLUMN_DEFAULT, 3, LEN(c.COLUMN_DEFAULT) - 4) LIKE '''%''' THEN SUBSTRING(c.COLUMN_DEFAULT, 4, LEN(c.COLUMN_DEFAULT) - 6) ELSE SUBSTRING(c.COLUMN_DEFAULT, 3, LEN(c.COLUMN_DEFAULT) - 4) END WHEN c.COLUMN_DEFAULT LIKE '(%' AND c.COLUMN_DEFAULT LIKE '%)' THEN @@ -47,7 +46,6 @@ SELECT c.COLUMN_NAME, WHEN SUBSTRING(c.COLUMN_DEFAULT, 2, LEN(c.COLUMN_DEFAULT) - 2) LIKE 'convert(%varchar%,getdate(),%)' THEN 'CURRENT_DATE' WHEN SUBSTRING(c.COLUMN_DEFAULT, 2, LEN(c.COLUMN_DEFAULT) - 2) = 'getdate()' THEN 'CURRENT_TIMESTAMP' WHEN SUBSTRING(c.COLUMN_DEFAULT, 2, LEN(c.COLUMN_DEFAULT) - 2) = 'sysdatetimeoffset()' THEN 'CURRENT_TIMESTAMP' - WHEN SUBSTRING(c.COLUMN_DEFAULT, 2, LEN(c.COLUMN_DEFAULT) - 2) LIKE '''%''' THEN SUBSTRING(c.COLUMN_DEFAULT, 3, LEN(c.COLUMN_DEFAULT) - 4) ELSE SUBSTRING(c.COLUMN_DEFAULT, 2, LEN(c.COLUMN_DEFAULT) - 2) END ELSE c.COLUMN_DEFAULT diff --git a/clojure/src/pgloader/source/mysql.clj b/clojure/src/pgloader/source/mysql.clj index beca6569..d17beda4 100644 --- a/clojure/src/pgloader/source/mysql.clj +++ b/clojure/src/pgloader/source/mysql.clj @@ -4,7 +4,8 @@ [clojure.string :as str] [next.jdbc :as jdbc] [clojure.tools.logging :as log]) - (:import [java.sql Connection DriverManager PreparedStatement ResultSet])) + (:import [java.nio.charset Charset] + [java.sql Connection DriverManager PreparedStatement ResultSet])) (set! *warn-on-reflection* true) @@ -95,27 +96,63 @@ ["geometry" "point" "linestring" "polygon" "multipoint" "multilinestring" "multipolygon" "geometrycollection"]))) -;;; MySQL charset names that do not match the encoding they actually store. -;;; Applied to the charset returned by DECODING-AS rules before issuing SET NAMES, -;;; so the MySQL server sends bytes that Connector/J can correctly decode. +;;; DECODING TABLE NAMES MATCHING … AS means: take the bytes stored +;;; in the text columns as-is and decode them with , whatever charset +;;; the column is declared with (typically UTF-8 bytes stored in latin1 +;;; columns). Asking the server to convert (SET NAMES) would double-encode +;;; exactly that data, so the columns are read with CAST(… AS BINARY) and +;;; decoded on the client. +;;; +;;; MySQL charset names mapped to Java charsets. Some MySQL names do not match +;;; the encoding they actually store: ;;; ;;; latin1 — MySQL documents this as Windows cp1252 (not ISO 8859-1). ;;; They share 0x00-0x7F and 0xA0-0xFF but differ in 0x80-0x9F, ;;; where cp1252 has 27 printable characters (€, –, ™ …) that ;;; ISO 8859-1 leaves undefined as C1 control codes. ;;; -;;; latin5 — MySQL latin5 is ISO 8859-9 (Turkish). SET NAMES 'latin5' is -;;; already correct server-side (MySQL converts to UTF-8 properly), -;;; so no remapping is needed for the SET NAMES path. -(def ^:private mysql-charset-aliases - {"latin1" "cp1252"}) - -(defn- normalize-mysql-charset - "Map MySQL charset names that alias wrong encodings to their correct names - for the SET NAMES call. MySQL's latin1 is actually cp1252 (Windows-1252): - they share 0x00-0x7F and 0xA0-0xFF but differ in 0x80-0x9F." - [charset] - (get mysql-charset-aliases charset charset)) +;;; latin5 — MySQL latin5 is ISO 8859-9 (Turkish). +(def ^:private mysql-charsets + {"utf8" "UTF-8" + "utf8mb3" "UTF-8" + "utf8mb4" "UTF-8" + "latin1" "windows-1252" + "latin2" "ISO-8859-2" + "latin5" "ISO-8859-9" + "latin7" "ISO-8859-13" + "greek" "ISO-8859-7" + "hebrew" "ISO-8859-8" + "cp1250" "windows-1250" + "cp1251" "windows-1251" + "cp1256" "windows-1256" + "cp1257" "windows-1257" + "cp850" "IBM850" + "cp852" "IBM852" + "cp866" "IBM866" + "koi8r" "KOI8-R" + "koi8u" "KOI8-U" + "sjis" "Shift_JIS" + "cp932" "windows-31j" + "ujis" "EUC-JP" + "eucjpms" "x-eucJP-Open" + "euckr" "EUC-KR" + "big5" "Big5" + "tis620" "TIS-620" + "ucs2" "UTF-16BE" + "utf16" "UTF-16BE" + "utf16le" "UTF-16LE" + "utf32" "UTF-32BE"}) + +(defn- decoding-charset + "Resolve a DECODING … AS charset name (MySQL or Java spelling) to a + java.nio.charset.Charset. Throws when the charset is unknown." + ^Charset [^String charset] + (let [cs-name (str/lower-case (str/trim charset))] + (try + (Charset/forName ^String (get mysql-charsets cs-name cs-name)) + (catch Exception _ + (throw (ex-info (str "DECODING AS: unsupported charset " (pr-str charset)) + {:charset charset})))))) (defn- decoding-as-charset "Return the charset for TABLE-NAME by matching against DECODING-AS rules, @@ -161,13 +198,31 @@ (some #(str/starts-with? lower %) ["char" "varchar" "tinytext" "mediumtext" "longtext" "text"])))) +(defn- decodable-col-type? + "Return true for MySQL column types whose values are stored in a character + set, and so are subject to DECODING … AS rules." + [^String raw-col-type] + (when raw-col-type + (boolean (re-find #"(?i)^(char|varchar|tinytext|mediumtext|longtext|text|enum|set)\b" + raw-col-type)))) + +(defn- source-col-type + [col] + (or (:source-column-type col) (:column-type col))) + (defn- convert-mysql-value "Convert a raw JDBC value from MySQL to a String suitable for COPY TEXT. For text columns, reads bytes directly and re-encodes as UTF-8 to preserve - embedded NUL bytes that JDBC getString() would silently truncate (#1573)." - [v jdbc-type col ^java.sql.ResultSet rs col-idx] + embedded NUL bytes that JDBC getString() would silently truncate (#1573). + When decode-cs is given, text columns arrive as raw bytes (see + mysql-select-sql) and are decoded with that charset." + [v jdbc-type col ^java.sql.ResultSet rs col-idx & [^Charset decode-cs]] (when (some? v) - (let [raw-col-type (:column-type col)] + (let [raw-col-type (:column-type col) + decoded? (and decode-cs + (bytes? v) + (decodable-col-type? (source-col-type col))) + v (if decoded? (String. ^bytes v decode-cs) v)] (cond (instance? Boolean v) (if v "1" "0") (= "YEAR" jdbc-type) (str (if (instance? java.sql.Date v) @@ -188,20 +243,27 @@ ;; Text columns: use getString so the JDBC driver applies charset ;; conversion correctly (#1573). getString preserves NUL bytes in ;; MySQL Connector/J; getObject may truncate at the first NUL. + decoded? v (and (instance? String v) (text-col-type? raw-col-type)) (.getString rs (int col-idx)) :else (str v))))) (defn- mysql-select-sql - "Build base SELECT SQL for a table given its column metadata and MySQL table name." - [columns mysql-table] + "Build base SELECT SQL for a table given its column metadata and MySQL table name. + With raw-text? true, text columns are selected as CAST(… AS BINARY) so the + server sends the stored bytes without any charset conversion." + [columns mysql-table & [raw-text?]] (let [col-list (if (seq columns) (str/join ", " (mapv (fn [col] (let [cn (:column-name col) - ct (or (:source-column-type col) (:column-type col))] - (if (geometry-type? ct) + ct (source-col-type col)] + (cond + (geometry-type? ct) (str "ST_AsText(`" cn "`) AS `" cn "`") + (and raw-text? (decodable-col-type? ct)) + (str "CAST(`" cn "` AS BINARY) AS `" cn "`") + :else (str "`" cn "`")))) columns)) "*")] @@ -210,7 +272,7 @@ (defn- stream-ranges! "Execute sqls sequentially on conn, closing each RS/stmt before the next. Returns a lazy seq of all rows across all range queries." - [^Connection conn active-stmt active-rs sqls columns] + [^Connection conn active-stmt active-rs sqls columns decode-cs] (when (seq sqls) (let [sql (first sqls) rest-sql (rest sqls)] @@ -246,11 +308,11 @@ (.getObject rs i) (nth jtypes (dec i)) (nth columns (dec i) nil) - rs i))) + rs i decode-cs))) (persistent! result))) (next-row))) ;; RS exhausted — open next range - (stream-ranges! conn active-stmt active-rs rest-sql columns))))] + (stream-ranges! conn active-stmt active-rs rest-sql columns decode-cs))))] (next-row)))))) (deftype MySQLSource @@ -355,14 +417,12 @@ (read-rows [_ table-spec-entry] (let [{:keys [table-name source-table-name columns citus-read-sql]} table-spec-entry mysql-table (or source-table-name table-name)] - (when-let [charset (normalize-mysql-charset - (decoding-as-charset mysql-table decoding-rules))] - (try - (jdbc/execute! conn [(str "SET NAMES '" charset "'")]) - (log/info (str "SET NAMES '" charset "' for table " mysql-table)) - (catch Exception e - (log/warn (str "SET NAMES failed for " mysql-table ": " (.getMessage e)))))) - (let [base-sql (or citus-read-sql (mysql-select-sql columns mysql-table)) + (let [decode-cs (some-> (decoding-as-charset mysql-table decoding-rules) + decoding-charset) + _ (when decode-cs + (log/info (str "Decoding text columns of " mysql-table + " as " (.name decode-cs)))) + base-sql (or citus-read-sql (mysql-select-sql columns mysql-table (some? decode-cs))) sqls (if (seq ranges) (mapv (fn [[lo hi]] (str base-sql @@ -370,7 +430,7 @@ " AND `" range-col "` < " hi)) ranges) [base-sql])] - (stream-ranges! conn active-stmt active-rs sqls columns)))) + (stream-ranges! conn active-stmt active-rs sqls columns decode-cs)))) (read-query [_ sql] (let [stmt (.prepareStatement conn sql) _ (doto stmt diff --git a/clojure/src/pgloader/source/pgsql.clj b/clojure/src/pgloader/source/pgsql.clj index 6e94fb7e..f1b5c666 100644 --- a/clojure/src/pgloader/source/pgsql.clj +++ b/clojure/src/pgloader/source/pgsql.clj @@ -1,5 +1,6 @@ (ns pgloader.source.pgsql (:require [pgloader.source.protocol :refer [Source partition-source]] + [pgloader.log :as plog] [hugsql.core :as hugsql] [clojure.string :as str] [next.jdbc :as jdbc] @@ -308,7 +309,7 @@ (defn create-source [uri-map _table-spec] (let [conn (connection uri-map)] - (->PGSQLSource conn (:raw uri-map) uri-map nil nil))) + (->PGSQLSource conn (plog/redact-uri (:raw uri-map)) uri-map nil nil))) (defn pgsql-partition-source [^PGSQLSource src table-spec-entry n chunk-bytes] diff --git a/clojure/test/pgloader/ddl_test.clj b/clojure/test/pgloader/ddl_test.clj index d1356447..9dceefdb 100644 --- a/clojure/test/pgloader/ddl_test.clj +++ b/clojure/test/pgloader/ddl_test.clj @@ -433,3 +433,56 @@ (is (= 1 (count enum-types))) ;; type name uses the (already-renamed) table name (is (= "renamed_tbl_status_t" (:type-name (first enum-types)))))))) + +(deftest test-auto-increment-column-def + (testing "MySQL AUTO_INCREMENT integers become serial types (with their own default)" + (let [ai (fn [t] {:column-name "id" :column-type t :is-nullable false + :extra "auto_increment" :column-default nil})] + (is (= " \"id\" serial" (ddl/column-def (ai "int")))) + (is (= " \"id\" serial" (ddl/column-def (ai "int(9)")))) + (is (= " \"id\" serial" (ddl/column-def (ai "tinyint(4)")))) + (is (= " \"id\" serial" (ddl/column-def (ai "smallint")))) + (is (= " \"id\" serial" (ddl/column-def (ai "mediumint unsigned")))) + (is (= " \"id\" bigserial" (ddl/column-def (ai "int unsigned")))) + (is (= " \"id\" bigserial" (ddl/column-def (ai "int(11)")))) + (is (= " \"id\" bigserial" (ddl/column-def (ai "bigint")))) + (is (= " \"id\" bigserial" (ddl/column-def (ai "bigint(20) unsigned")))))) + + (testing "a serial column never carries a DEFAULT of its own" + (is (= " \"id\" serial" + (ddl/column-def {:column-name "id" :column-type "int" :is-nullable false + :extra "auto_increment" :column-default "0"})))) + + (testing "without auto_increment the plain integer mapping is unchanged" + (is (= " \"id\" bigint NOT NULL" + (ddl/column-def {:column-name "id" :column-type "int unsigned" + :is-nullable false :extra ""}))))) + +(deftest test-reset-sequences-sql + (testing "next nextval() returns MAX + 1, or 1 on an empty table" + (is (= [(str "SELECT pg_catalog.setval(" + "pg_get_serial_sequence('\"shopdb\".\"orders\"', 'id')" + ", GREATEST(MAX(\"id\"), 1), MAX(\"id\") IS NOT NULL)" + " FROM \"shopdb\".\"orders\";\n")] + (ddl/reset-sequences-sql "shopdb" "orders" + [{:column-name "id" :column-type "int unsigned" + :extra "auto_increment"} + {:column-name "total" :column-type "int" + :extra ""}])))) + + (testing "identifiers are escaped inside the string literals" + (let [[sql] (ddl/reset-sequences-sql "public" "o'brien" + [{:column-name "it's" :column-type "serial" + :extra "auto_increment"}])] + (is (str/includes? sql "pg_get_serial_sequence('\"public\".\"o''brien\"', 'it''s')")))) + + (testing "non-integer targets (cast rules) are skipped" + (is (empty? (ddl/reset-sequences-sql "public" "t" + [{:column-name "id" :column-type "text" + :source-column-type "int" + :extra "auto_increment"}]))))) + +(deftest test-create-schemas-sql + (is (= ["CREATE SCHEMA IF NOT EXISTS \"shopdb\";" + "CREATE SCHEMA IF NOT EXISTS \"public\";"] + (ddl/create-schemas-sql ["shopdb" "public" nil "shopdb"])))) diff --git a/clojure/test/pgloader/log_test.clj b/clojure/test/pgloader/log_test.clj new file mode 100644 index 00000000..a69edc0b --- /dev/null +++ b/clojure/test/pgloader/log_test.clj @@ -0,0 +1,26 @@ +(ns pgloader.log-test + (:require [clojure.test :refer [deftest is testing]] + [pgloader.log :as plog])) + +(deftest test-redact-uri + (testing "user:password@ in pgloader URIs" + (is (= "postgresql://postgres:****@v4target/shopdb" + (plog/redact-uri "postgresql://postgres:postgres@v4target/shopdb"))) + (is (= "mssql://pgloader_ro:****@mssql-source/shopdb" + (plog/redact-uri "mssql://pgloader_ro:pgloader_RO_dev_only1@mssql-source/shopdb"))) + (is (= "mysql://root:****@mysql:3306/db?useSSL=false" + (plog/redact-uri "mysql://root:p@ss:w0rd@mysql:3306/db?useSSL=false"))) + (is (= "jdbc:postgresql://u:****@host:5432/db" + (plog/redact-uri "jdbc:postgresql://u:secret@host:5432/db")))) + + (testing "password= parameters in JDBC URLs" + (is (= "jdbc:sqlserver://host:1433;databaseName=db;user=sa;password=****;encrypt=false" + (plog/redact-uri "jdbc:sqlserver://host:1433;databaseName=db;user=sa;password=pgloaderTest1!;encrypt=false"))) + (is (= "jdbc:postgresql://host/db?user=u&Password=****&ssl=true" + (plog/redact-uri "jdbc:postgresql://host/db?user=u&Password=secret&ssl=true")))) + + (testing "URIs without a password are unchanged" + (is (= "postgresql://postgres@host/db" (plog/redact-uri "postgresql://postgres@host/db"))) + (is (= "postgresql://host:5432/db" (plog/redact-uri "postgresql://host:5432/db"))) + (is (= "mysql://shopdb/" (plog/redact-uri "mysql://shopdb/"))) + (is (nil? (plog/redact-uri nil))))) diff --git a/clojure/test/pgloader/source/mssql_test.clj b/clojure/test/pgloader/source/mssql_test.clj new file mode 100644 index 00000000..f6f46a4a --- /dev/null +++ b/clojure/test/pgloader/source/mssql_test.clj @@ -0,0 +1,30 @@ +(ns pgloader.source.mssql-test + (:require [clojure.test :refer [deftest is testing]] + [pgloader.source.mssql :as mssql])) + +(defn- sanitize [default pg-type] + (#'mssql/sanitize-default default pg-type)) + +(deftest test-sanitize-default-string-literals + (testing "N'…' Unicode literals and '…' literals yield their value" + (is (= "new" (sanitize "N'new'" "text"))) + (is (= "new" (sanitize "'new'" "text"))) + (is (= "it's" (sanitize "N'it''s'" "text"))) + (is (= "" (sanitize "N''" "text")))) + + (testing "empty string literal on a numeric column is dropped (#1163)" + (is (nil? (sanitize "''" "integer"))) + (is (nil? (sanitize "N''" "integer"))))) + +(deftest test-sanitize-default-tsql-functions + (doseq [f ["SYSUTCDATETIME()" "sysutcdatetime()" "SYSDATETIME()" "GETDATE()" + "getutcdate()" "sysdatetimeoffset()" "CURRENT_TIMESTAMP"]] + (is (= "CURRENT_TIMESTAMP" (sanitize f "timestamptz")) f)) + (is (= "gen_random_uuid()" (sanitize "NEWID()" "uuid"))) + (is (= "gen_random_uuid()" (sanitize "newsequentialid()" "uuid")))) + +(deftest test-sanitize-default-passthrough + (is (= "0" (sanitize "0" "integer"))) + (is (= "nextval('public.order_seq')" (sanitize "NEXT VALUE FOR [dbo].[order_seq]" "bigint"))) + (is (nil? (sanitize "convert(datetime,'1753-01-01',0)" "timestamptz"))) + (is (nil? (sanitize nil "text")))) diff --git a/clojure/test/pgloader/source/mysql_test.clj b/clojure/test/pgloader/source/mysql_test.clj new file mode 100644 index 00000000..97c62915 --- /dev/null +++ b/clojure/test/pgloader/source/mysql_test.clj @@ -0,0 +1,49 @@ +(ns pgloader.source.mysql-test + (:require [clojure.test :refer [deftest is testing]] + [pgloader.source.mysql :as mysql]) + (:import [java.nio.charset StandardCharsets])) + +(deftest test-decoding-charset + (is (= "UTF-8" (.name (#'mysql/decoding-charset "utf8")))) + (is (= "UTF-8" (.name (#'mysql/decoding-charset "UTF8MB4")))) + (is (= "windows-1252" (.name (#'mysql/decoding-charset "latin1")))) + (is (= "ISO-8859-9" (.name (#'mysql/decoding-charset "latin5")))) + (is (= "UTF-8" (.name (#'mysql/decoding-charset "utf-8")))) + (is (thrown-with-msg? clojure.lang.ExceptionInfo #"unsupported charset" + (#'mysql/decoding-charset "no-such-charset")))) + +(deftest test-mysql-select-sql-raw-text + (let [cols [{:column-name "id" :column-type "int"} + {:column-name "name" :column-type "varchar(100)"} + {:column-name "notes" :column-type "text"} + {:column-name "tier" :column-type "\"shopdb\".\"t_tier_t\"" + :source-column-type "enum('a','b')"} + {:column-name "geo" :column-type "point"}]] + (testing "without decoding, columns are selected as-is" + (is (= "SELECT `id`, `name`, `notes`, `tier`, ST_AsText(`geo`) AS `geo` FROM `t`" + (#'mysql/mysql-select-sql cols "t")))) + (testing "with decoding, text columns are fetched as raw bytes" + (is (= (str "SELECT `id`, CAST(`name` AS BINARY) AS `name`, " + "CAST(`notes` AS BINARY) AS `notes`, CAST(`tier` AS BINARY) AS `tier`, " + "ST_AsText(`geo`) AS `geo` FROM `t`") + (#'mysql/mysql-select-sql cols "t" true)))))) + +(deftest test-convert-mysql-value-decoding + (let [utf8-bytes (.getBytes "Jean-François Ekström" StandardCharsets/UTF_8) + convert #'mysql/convert-mysql-value] + (testing "raw bytes of a text column are decoded with the DECODING charset" + (is (= "Jean-François Ekström" + (convert utf8-bytes "LONGBLOB" {:column-type "varchar(100)"} nil 1 + StandardCharsets/UTF_8)))) + (testing "the same bytes read as latin1 are mojibake — what the bug produced" + (is (= "Jean-François Ekström" + (convert utf8-bytes "LONGBLOB" {:column-type "varchar(100)"} nil 1 + (#'mysql/decoding-charset "latin1"))))) + (testing "SET columns are decoded then formatted as arrays" + (is (= "{a,b}" + (convert (.getBytes "a,b" StandardCharsets/UTF_8) "LONGBLOB" + {:column-type "set('a','b')"} nil 1 StandardCharsets/UTF_8)))) + (testing "binary columns are still hex-encoded" + (is (= "Xdead" + (convert (byte-array [(unchecked-byte 0xde) (unchecked-byte 0xad)]) "VARBINARY" + {:column-type "varbinary(2)"} nil 1 StandardCharsets/UTF_8)))))) diff --git a/clojure/tests/mssql/Makefile b/clojure/tests/mssql/Makefile index af8f69c5..095bab87 100644 --- a/clojure/tests/mssql/Makefile +++ b/clojure/tests/mssql/Makefile @@ -5,10 +5,15 @@ REGRESS_FLAGS ?= # 04-indexes: filtered index WHERE clause — v4 only. # v3 does not translate SQL Server partial index filters to PostgreSQL WHERE # clauses, so that test is excluded from v3 runs. +# 13-tsql-defaults: T-SQL default functions such as SYSUTCDATETIME() — v4 only. +# v3 copies them verbatim, fails to create the table and aborts the load, so +# mssql-tsql-defaults.load is not run with v3 either. ifeq ($(firstword $(PGLOADER_CMD)),pgloader) - _REGRESS_EXTRA := --exclude 04-indexes + _REGRESS_EXTRA := --exclude 04-indexes --exclude 13-tsql-defaults + _V4_ONLY_LOAD_CMD = else _REGRESS_EXTRA := + _V4_ONLY_LOAD_CMD = $(PGLOADER_CMD) /suite/mssql-tsql-defaults.load endif .PHONY: all run update-expected @@ -18,11 +23,13 @@ all: run run: $(PGLOADER_CMD) /suite/mssql.load $(PGLOADER_CMD) /suite/mssql-filter-views.load + $(_V4_ONLY_LOAD_CMD) java -jar $(JAR) regress $(REGRESS_FLAGS) $(_REGRESS_EXTRA) /suite update-expected: $(PGLOADER_CMD) /suite/mssql.load $(PGLOADER_CMD) /suite/mssql-filter-views.load + $(_V4_ONLY_LOAD_CMD) java -jar $(JAR) regress --update /suite %-v3: diff --git a/clojure/tests/mssql/expected/13-tsql-defaults.out b/clojure/tests/mssql/expected/13-tsql-defaults.out new file mode 100644 index 00000000..824aeffb --- /dev/null +++ b/clojure/tests/mssql/expected/13-tsql-defaults.out @@ -0,0 +1,21 @@ + column_name | column_default +-------------+------------------- + status | 'new'::text + label | 'it''s'::text + created_utc | CURRENT_TIMESTAMP + created_loc | CURRENT_TIMESTAMP + updated_utc | CURRENT_TIMESTAMP + row_guid | gen_random_uuid() +(6 rows) + + rows_loaded +------------- + 3 +(1 row) + + id | status | label | has_created_utc | has_row_guid +----+--------+-------+-----------------+-------------- + 4 | new | it's | t | t +(1 row) + +INSERT 0 1 diff --git a/clojure/tests/mssql/init.sql b/clojure/tests/mssql/init.sql index e005bf2b..879940c7 100644 --- a/clojure/tests/mssql/init.sql +++ b/clojure/tests/mssql/init.sql @@ -201,3 +201,35 @@ GO INSERT INTO filtered.order_items (product_id, qty) VALUES (1, 3), (2, 1), (1, 5); GO + +-- tsqldefaults: separate database for T-SQL default expressions, loaded by +-- mssql-tsql-defaults.load with v4 only — v3 copies SYSUTCDATETIME() verbatim, +-- fails to create the table and aborts the whole load. +-- Unicode string literals N'…' must land as the string value (not the literal +-- text N'new'), and T-SQL functions such as SYSUTCDATETIME() or +-- NEWSEQUENTIALID() must be translated to PostgreSQL. The table lives in a +-- non-dbo schema, which has to be created on the target before the table. +CREATE DATABASE tsqldefaults; +GO + +USE tsqldefaults; +GO + +CREATE SCHEMA shop; +GO + +CREATE TABLE shop.tsql_defaults ( + id INT IDENTITY(1,1) PRIMARY KEY, + status NVARCHAR(20) NOT NULL DEFAULT (N'new'), + label VARCHAR(20) DEFAULT ('it''s'), + created_utc DATETIME2(3) NOT NULL DEFAULT (SYSUTCDATETIME()), + created_loc DATETIME2 DEFAULT (SYSDATETIME()), + updated_utc DATETIME DEFAULT (GETUTCDATE()), + row_guid UNIQUEIDENTIFIER DEFAULT (NEWSEQUENTIALID()) +); +GO + +INSERT INTO shop.tsql_defaults DEFAULT VALUES; +INSERT INTO shop.tsql_defaults DEFAULT VALUES; +INSERT INTO shop.tsql_defaults DEFAULT VALUES; +GO diff --git a/clojure/tests/mssql/mssql-tsql-defaults.load b/clojure/tests/mssql/mssql-tsql-defaults.load new file mode 100644 index 00000000..734fb50b --- /dev/null +++ b/clojure/tests/mssql/mssql-tsql-defaults.load @@ -0,0 +1,8 @@ +-- T-SQL default expressions (N'…' literals, SYSUTCDATETIME(), NEWSEQUENTIALID()…) +-- translated to PostgreSQL, in a non-dbo schema. v4 only, see init.sql. +LOAD DATABASE + FROM mssql://sa:pgloaderTest1!@mssql:1433/tsqldefaults + INTO postgresql://pgloader:pgloader@postgres:5432/target + + WITH include drop, create tables, create indexes, reset sequences, + foreign keys; diff --git a/clojure/tests/mssql/sql/13-tsql-defaults.sql b/clojure/tests/mssql/sql/13-tsql-defaults.sql new file mode 100644 index 00000000..4eeb5fe8 --- /dev/null +++ b/clojure/tests/mssql/sql/13-tsql-defaults.sql @@ -0,0 +1,20 @@ +-- T-SQL defaults translated to PostgreSQL: +-- (N'new') → 'new'::text (not 'N''new''') +-- ('it''s') → 'it''s'::text +-- (SYSUTCDATETIME()), (SYSDATETIME()), (GETUTCDATE()) → CURRENT_TIMESTAMP +-- (NEWSEQUENTIALID()) → gen_random_uuid() +SELECT column_name, column_default +FROM information_schema.columns +WHERE table_schema = 'shop' + AND table_name = 'tsql_defaults' + AND column_name <> 'id' +ORDER BY ordinal_position; + +SELECT count(*) AS rows_loaded FROM shop.tsql_defaults; + +-- A new row gets the next identity value (sequence reset after 3 rows) and +-- the translated defaults. +INSERT INTO shop.tsql_defaults DEFAULT VALUES +RETURNING id, status, label, + created_utc IS NOT NULL AS has_created_utc, + row_guid IS NOT NULL AS has_row_guid; diff --git a/clojure/tests/mysql-unit-full/expected/02-datetime-precision.mariadb.out b/clojure/tests/mysql-unit-full/expected/02-datetime-precision.mariadb.out index d71afd21..0909c1fb 100644 --- a/clojure/tests/mysql-unit-full/expected/02-datetime-precision.mariadb.out +++ b/clojure/tests/mysql-unit-full/expected/02-datetime-precision.mariadb.out @@ -1,6 +1,6 @@ - column_name | data_type | datetime_precision | column_default --------------+--------------------------+--------------------+------------------- - id | bigint | | + column_name | data_type | datetime_precision | column_default +-------------+--------------------------+--------------------+------------------------------------------------------------ + id | bigint | | nextval('mysql_unit_full.type_precision_id_seq'::regclass) ts6 | timestamp with time zone | 6 | CURRENT_TIMESTAMP ts3 | timestamp with time zone | 3 | t3 | time without time zone | 3 | diff --git a/clojure/tests/mysql-unit-full/expected/02-datetime-precision.out b/clojure/tests/mysql-unit-full/expected/02-datetime-precision.out index 3c80bc23..9ceb0ab7 100644 --- a/clojure/tests/mysql-unit-full/expected/02-datetime-precision.out +++ b/clojure/tests/mysql-unit-full/expected/02-datetime-precision.out @@ -1,6 +1,6 @@ - column_name | data_type | datetime_precision | column_default --------------+--------------------------+--------------------+------------------- - id | integer | | + column_name | data_type | datetime_precision | column_default +-------------+--------------------------+--------------------+------------------------------------------------------------ + id | integer | | nextval('mysql_unit_full.type_precision_id_seq'::regclass) ts6 | timestamp with time zone | 6 | CURRENT_TIMESTAMP ts3 | timestamp with time zone | 3 | t3 | time without time zone | 3 | diff --git a/clojure/tests/mysql-unit-full/expected/03-bit-defaults.mariadb.out b/clojure/tests/mysql-unit-full/expected/03-bit-defaults.mariadb.out index 39b9c62c..78237619 100644 --- a/clojure/tests/mysql-unit-full/expected/03-bit-defaults.mariadb.out +++ b/clojure/tests/mysql-unit-full/expected/03-bit-defaults.mariadb.out @@ -1,6 +1,6 @@ - column_name | data_type | column_default --------------+--------------------------+------------------- - id | bigint | + column_name | data_type | column_default +-------------+--------------------------+---------------------------------------------------------- + id | bigint | nextval('mysql_unit_full.bit_defaults_id_seq'::regclass) flags | bit varying | '0'::"bit" single_bit | bit varying | '0'::"bit" created_at | timestamp with time zone | CURRENT_TIMESTAMP diff --git a/clojure/tests/mysql-unit-full/expected/03-bit-defaults.out b/clojure/tests/mysql-unit-full/expected/03-bit-defaults.out index 73213a10..7e9cef41 100644 --- a/clojure/tests/mysql-unit-full/expected/03-bit-defaults.out +++ b/clojure/tests/mysql-unit-full/expected/03-bit-defaults.out @@ -1,6 +1,6 @@ - column_name | data_type | column_default --------------+--------------------------+------------------- - id | integer | + column_name | data_type | column_default +-------------+--------------------------+---------------------------------------------------------- + id | integer | nextval('mysql_unit_full.bit_defaults_id_seq'::regclass) flags | bit varying | '0'::"bit" single_bit | bit varying | '0'::"bit" created_at | timestamp with time zone | CURRENT_TIMESTAMP diff --git a/clojure/tests/mysql/Makefile b/clojure/tests/mysql/Makefile index f9755bc4..62741ed2 100644 --- a/clojure/tests/mysql/Makefile +++ b/clojure/tests/mysql/Makefile @@ -27,10 +27,12 @@ my-cnf: # mytest-v3: same suite against the v3 binary, compared against # .v3 variant expected files that capture deliberate v3 vs v4 differences -# (serial vs identity, bit vs bit varying, trigger naming). +# (bit vs bit varying, trigger naming, sequence reset of empty tables). +# 19-decoding-as is v4 only: the v3 binary built from src/ does not apply +# DECODING TABLE NAMES MATCHING … AS utf8 to latin1 columns. mytest-v3: pgloader --client-min-messages notice mytest/mytest.load - java -jar $(JAR) regress $(REGRESS_FLAGS) --variant v3 mytest + java -jar $(JAR) regress $(REGRESS_FLAGS) --variant v3 --exclude 19-decoding-as mytest all-v3: mytest-v3 sakila-v3 f1db-v3 @: diff --git a/clojure/tests/mysql/mytest/expected/01-tables.out b/clojure/tests/mysql/mytest/expected/01-tables.out index 516dfb7b..5d10cb0b 100644 --- a/clojure/tests/mysql/mytest/expected/01-tables.out +++ b/clojure/tests/mysql/mytest/expected/01-tables.out @@ -19,6 +19,7 @@ history ip_addresses latin1_encoding + legacy_notes measurements on_update_ts onupdate @@ -35,5 +36,5 @@ utilisateurs__Yvelines2013-06-28 uw_defined_meaning zero_dates_notnull -(35 rows) +(36 rows) diff --git a/clojure/tests/mysql/mytest/expected/01-tables.v3.out b/clojure/tests/mysql/mytest/expected/01-tables.v3.out index 516dfb7b..5d10cb0b 100644 --- a/clojure/tests/mysql/mytest/expected/01-tables.v3.out +++ b/clojure/tests/mysql/mytest/expected/01-tables.v3.out @@ -19,6 +19,7 @@ history ip_addresses latin1_encoding + legacy_notes measurements on_update_ts onupdate @@ -35,5 +36,5 @@ utilisateurs__Yvelines2013-06-28 uw_defined_meaning zero_dates_notnull -(35 rows) +(36 rows) diff --git a/clojure/tests/mysql/mytest/expected/03-indexes.out b/clojure/tests/mysql/mytest/expected/03-indexes.out index 39e5f3df..5a4095f6 100644 --- a/clojure/tests/mysql/mytest/expected/03-indexes.out +++ b/clojure/tests/mysql/mytest/expected/03-indexes.out @@ -29,6 +29,7 @@ idx_{oid}_PRIMARY idx_{oid}_PRIMARY idx_{oid}_PRIMARY + idx_{oid}_PRIMARY idx_{oid}_ak_countdata_idx idx_{oid}_domain_filter idx_{oid}_domain_filter_unq @@ -41,5 +42,5 @@ idx_{oid}_search idx_{oid}_update_type idx_{oid}_url -(41 rows) +(42 rows) diff --git a/clojure/tests/mysql/mytest/expected/03-indexes.v3.out b/clojure/tests/mysql/mytest/expected/03-indexes.v3.out index 39e5f3df..5a4095f6 100644 --- a/clojure/tests/mysql/mytest/expected/03-indexes.v3.out +++ b/clojure/tests/mysql/mytest/expected/03-indexes.v3.out @@ -29,6 +29,7 @@ idx_{oid}_PRIMARY idx_{oid}_PRIMARY idx_{oid}_PRIMARY + idx_{oid}_PRIMARY idx_{oid}_ak_countdata_idx idx_{oid}_domain_filter idx_{oid}_domain_filter_unq @@ -41,5 +42,5 @@ idx_{oid}_search idx_{oid}_update_type idx_{oid}_url -(41 rows) +(42 rows) diff --git a/clojure/tests/mysql/mytest/expected/06-datetime-precision.out b/clojure/tests/mysql/mytest/expected/06-datetime-precision.out index 3c80bc23..9c416947 100644 --- a/clojure/tests/mysql/mytest/expected/06-datetime-precision.out +++ b/clojure/tests/mysql/mytest/expected/06-datetime-precision.out @@ -1,6 +1,6 @@ - column_name | data_type | datetime_precision | column_default --------------+--------------------------+--------------------+------------------- - id | integer | | + column_name | data_type | datetime_precision | column_default +-------------+--------------------------+--------------------+--------------------------------------------------- + id | integer | | nextval('mytest.type_precision_id_seq'::regclass) ts6 | timestamp with time zone | 6 | CURRENT_TIMESTAMP ts3 | timestamp with time zone | 3 | t3 | time without time zone | 3 | diff --git a/clojure/tests/mysql/mytest/expected/07-bit-defaults.out b/clojure/tests/mysql/mytest/expected/07-bit-defaults.out index 73213a10..7eb3f709 100644 --- a/clojure/tests/mysql/mytest/expected/07-bit-defaults.out +++ b/clojure/tests/mysql/mytest/expected/07-bit-defaults.out @@ -1,6 +1,6 @@ - column_name | data_type | column_default --------------+--------------------------+------------------- - id | integer | + column_name | data_type | column_default +-------------+--------------------------+------------------------------------------------- + id | integer | nextval('mytest.bit_defaults_id_seq'::regclass) flags | bit varying | '0'::"bit" single_bit | bit varying | '0'::"bit" created_at | timestamp with time zone | CURRENT_TIMESTAMP diff --git a/clojure/tests/mysql/mytest/expected/12-pkeys.out b/clojure/tests/mysql/mytest/expected/12-pkeys.out index c7ede454..7ded4d0a 100644 --- a/clojure/tests/mysql/mytest/expected/12-pkeys.out +++ b/clojure/tests/mysql/mytest/expected/12-pkeys.out @@ -1,10 +1,10 @@ pk_constraints ---------------- - 29 + 30 (1 row) total_indexes --------------- - 41 + 42 (1 row) diff --git a/clojure/tests/mysql/mytest/expected/12-pkeys.v3.out b/clojure/tests/mysql/mytest/expected/12-pkeys.v3.out index c7ede454..7ded4d0a 100644 --- a/clojure/tests/mysql/mytest/expected/12-pkeys.v3.out +++ b/clojure/tests/mysql/mytest/expected/12-pkeys.v3.out @@ -1,10 +1,10 @@ pk_constraints ---------------- - 29 + 30 (1 row) total_indexes --------------- - 41 + 42 (1 row) diff --git a/clojure/tests/mysql/mytest/expected/18-sequences.out b/clojure/tests/mysql/mytest/expected/18-sequences.out new file mode 100644 index 00000000..2ba40c26 --- /dev/null +++ b/clojure/tests/mysql/mytest/expected/18-sequences.out @@ -0,0 +1,15 @@ + table_name | column_name | data_type | has_sequence +--------------+-------------+-----------+-------------- + empty | id | integer | t + fcm_batches | id | bigint | t + legacy_notes | id | bigint | t + users | id | integer | t +(4 rows) + + table_name | next_id +--------------+--------- + empty | 1 + legacy_notes | 7 + users | 4 +(3 rows) + diff --git a/clojure/tests/mysql/mytest/expected/18-sequences.v3.out b/clojure/tests/mysql/mytest/expected/18-sequences.v3.out new file mode 100644 index 00000000..4cb4e9f0 --- /dev/null +++ b/clojure/tests/mysql/mytest/expected/18-sequences.v3.out @@ -0,0 +1,15 @@ + table_name | column_name | data_type | has_sequence +--------------+-------------+-----------+-------------- + empty | id | integer | t + fcm_batches | id | bigint | t + legacy_notes | id | bigint | t + users | id | integer | t +(4 rows) + + table_name | next_id +--------------+--------- + empty | 2 + legacy_notes | 7 + users | 4 +(3 rows) + diff --git a/clojure/tests/mysql/mytest/expected/19-decoding-as.out b/clojure/tests/mysql/mytest/expected/19-decoding-as.out new file mode 100644 index 00000000..f10089ba --- /dev/null +++ b/clojure/tests/mysql/mytest/expected/19-decoding-as.out @@ -0,0 +1,6 @@ + id | author | note +----+-----------------------+------------------ + 5 | Jean-François Ekström | Zoë – naïve café + 6 | ascii only | +(2 rows) + diff --git a/clojure/tests/mysql/mytest/mytest.load b/clojure/tests/mysql/mytest/mytest.load index 1ad07fee..50c888ae 100644 --- a/clojure/tests/mysql/mytest/mytest.load +++ b/clojure/tests/mysql/mytest/mytest.load @@ -90,5 +90,5 @@ load database column binary_types.uuid_col to uuid drop typemod using binary-to-uuid, column binary_types.data_col to bytea drop typemod using binary-to-bytea - BEFORE LOAD DO - $$ CREATE SCHEMA IF NOT EXISTS mytest; $$; + -- UTF-8 bytes stored in latin1 columns: decode the stored bytes as UTF-8. + DECODING TABLE NAMES MATCHING 'legacy_notes' AS utf8; diff --git a/clojure/tests/mysql/mytest/mytest.sql b/clojure/tests/mysql/mytest/mytest.sql index 0bff5632..a4a1b47d 100644 --- a/clojure/tests/mysql/mytest/mytest.sql +++ b/clojure/tests/mysql/mytest/mytest.sql @@ -454,6 +454,26 @@ INSERT INTO `latin1_encoding` (label, word) VALUES ('trademark', _latin1 x'99'), -- ™ U+2122 (cp1252 0x99, undefined in iso-8859-1) ('cafe_euro', _latin1 x'636166e980'); -- café€: c(63)a(61)f(66)é(e9)€(80) +-- ============================================================ +-- DECODING TABLE NAMES MATCHING 'legacy_notes' AS utf8: an application wrote +-- UTF-8 bytes into latin1 columns. Read through MySQL's charset conversion +-- they come out double-encoded ("Jean-François"); pgloader must decode the +-- stored bytes as UTF-8 instead. +-- The explicit ids leave a gap so the sequence reset is visible: the next +-- generated id on the target must be 7, not 6. +-- ============================================================ +CREATE TABLE `legacy_notes` ( + id INT UNSIGNED NOT NULL AUTO_INCREMENT, + author VARCHAR(100) NOT NULL, + note TEXT, + PRIMARY KEY (id) +) ENGINE=InnoDB DEFAULT CHARSET=latin1; + +INSERT INTO `legacy_notes` (id, author, note) VALUES + (5, _latin1 x'4a65616e2d4672616ec3a76f697320456b737472c3b66d', -- Jean-François Ekström + _latin1 x'5a6fc3ab20e28093206e61c3af766520636166c3a9'), -- Zoë – naïve café + (6, 'ascii only', NULL); + -- ============================================================ -- #1757: varbinary-to-inet — VARBINARY(16) storing raw IP bytes -- 4 bytes = IPv4, 16 bytes = IPv6, 0 bytes = NULL diff --git a/clojure/tests/mysql/mytest/sql/18-sequences.sql b/clojure/tests/mysql/mytest/sql/18-sequences.sql new file mode 100644 index 00000000..2c67ab7d --- /dev/null +++ b/clojure/tests/mysql/mytest/sql/18-sequences.sql @@ -0,0 +1,26 @@ +-- AUTO_INCREMENT columns become serial/bigserial columns owning a sequence +-- (int → integer, int unsigned → bigint), and "reset sequences" leaves each +-- sequence so that the next generated id is MAX(id) + 1, or 1 when empty. +SELECT table_name, + column_name, + data_type, + pg_get_serial_sequence(format('%I.%I', table_schema, table_name), + column_name) IS NOT NULL AS has_sequence +FROM information_schema.columns +WHERE table_schema = 'mytest' + AND table_name IN ('empty', 'fcm_batches', 'legacy_notes', 'users') + AND column_name = 'id' +ORDER BY table_name; + +-- Next value each sequence will hand out, without consuming it. +SELECT 'empty' AS table_name, + CASE WHEN is_called THEN last_value + 1 ELSE last_value END AS next_id + FROM mytest.empty_id_seq +UNION ALL +SELECT 'legacy_notes', + CASE WHEN is_called THEN last_value + 1 ELSE last_value END + FROM mytest.legacy_notes_id_seq +UNION ALL +SELECT 'users', + CASE WHEN is_called THEN last_value + 1 ELSE last_value END + FROM mytest.users_id_seq; diff --git a/clojure/tests/mysql/mytest/sql/19-decoding-as.sql b/clojure/tests/mysql/mytest/sql/19-decoding-as.sql new file mode 100644 index 00000000..874cd6b8 --- /dev/null +++ b/clojure/tests/mysql/mytest/sql/19-decoding-as.sql @@ -0,0 +1,5 @@ +-- DECODING TABLE NAMES MATCHING 'legacy_notes' AS utf8: UTF-8 bytes stored in +-- latin1 columns arrive decoded, not double-encoded ("Jean-François"). +SELECT id, author, note + FROM mytest.legacy_notes + ORDER BY id;