From 06391890d8d76692a2806306a22ef09116794a8c Mon Sep 17 00:00:00 2001 From: madtibo Date: Tue, 10 Dec 2019 11:42:17 +0100 Subject: [PATCH 01/29] help correction: si = SQLSERVER_INSTANCE --- sqlserver2pgsql.pl | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 0fa7a99..cfab8da 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -790,7 +790,7 @@ sub usage -sd SQLSERVER_DATABASE -sh SQLSERVER_HOST - -si SQLSERVER_HOST + -si SQLSERVER_INSTANCE -sp SQLSERVER_PORT -su SQLSERVER_USERNAME -sw SQLSERVER_PASSWORD From 78054012a2bc7a5417798e44589e6eb5418a3bf7 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Fri, 3 Apr 2020 17:23:55 +0200 Subject: [PATCH 02/29] permit other partition_scheme or filegroup than PRIMARY for object creation (#127) * permit other partition_scheme or filegroup than PRIMARY for object creation * specific env for PG to solve CI container pb with PG access --- .circleci/config.yml | 4 ++++ regression/reg_tests.sql | Bin 12596 -> 12608 bytes sqlserver2pgsql.pl | 2 +- 3 files changed, 5 insertions(+), 1 deletion(-) diff --git a/.circleci/config.yml b/.circleci/config.yml index e677c50..001c829 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -5,6 +5,10 @@ jobs: docker: - image: perl:5.24-threaded - image: postgres:10-alpine + environment: + POSTGRES_USER: postgres + POSTGRES_POSTGRES: postgres + POSTGRES_HOST_AUTH_METHOD: trust environment: # Inject APT packaged dependencies. PERL5LIB: /usr/lib/x86_64-linux-gnu/perl5/5.24:/usr/share/perl5 diff --git a/regression/reg_tests.sql b/regression/reg_tests.sql index 3e6b517666c89a927d837cb32f7c13678edee699..06af866668160b3d1df00f6e817c60b87df906f5 100644 GIT binary patch delta 46 zcmdmzbRcQNL|tYD2E)l8nB+n9=5@NI%%Z^zt_;o${tSK$E)0$gK@5=$u?$=QMOh0S delta 28 kcmX?*v?Xc7MBT~0N=lReDv3>&7vtM}U)P6uaxSYN0IuH(jQ{`u diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index cfab8da..f804ee9 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -1571,7 +1571,7 @@ sub parse_dump } } - elsif ($line =~ /^\s*\) ON \[PRIMARY\]/) + elsif ($line =~ /^\s*\) ON \[.*\]/) { # End of the table next MAIN; From 5e3e73838812835b397cc483d96a6724b9ab60ee Mon Sep 17 00:00:00 2001 From: Thibaut Date: Fri, 3 Apr 2020 17:34:42 +0200 Subject: [PATCH 03/29] update CI container to PG 12 (#128) --- .circleci/config.yml | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/.circleci/config.yml b/.circleci/config.yml index 001c829..2cab6cf 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -4,7 +4,7 @@ jobs: test: docker: - image: perl:5.24-threaded - - image: postgres:10-alpine + - image: postgres:12-alpine environment: POSTGRES_USER: postgres POSTGRES_POSTGRES: postgres From 7e65e475eddb4ff58cdd7ede80074bc0d9981b54 Mon Sep 17 00:00:00 2001 From: madtibo Date: Tue, 23 Jun 2020 10:16:33 +0200 Subject: [PATCH 04/29] exclude some more sp_addextendedproperty --- sqlserver2pgsql.pl | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index f804ee9..7c6ce76 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -2206,7 +2206,7 @@ sub parse_dump or croak "Cannot find a name for this extended property: $sqlproperty"; my $propertyname = $1; - if ($propertyname =~ /^(MS_DiagramPaneCount|MS_DiagramPane1|MS_DiagramPane2|Display Name|Description|Example Values|Source System|Table Description|Table Type|ETL Rules|Display Folder|SCD Type|Source Datatype)$/) + if ($propertyname =~ /^(AggregateType|AllowZeroLength|AppendOnly|Attributes|CollatingOrder|ColumnHidden|ColumnOrder|ColumnWidth|DataUpdatable|DateCreated|DefaultValue|Description|Display Folder|Display Name|DisplayViewsOnSharePointSite|ETL Rules|Example Values|FilterOnLoad|GUID|HideNewField|LastUpdated|MS_DecimalPlaces|MS_DefaultView|MS_DiagramPane1|MS_DiagramPane2|MS_DiagramPaneCount|MS_DisplayControl|MS_Format|MS_Hyperlink|MS_IMEMode|MS_IMESentMode|MS_InputMask|MS_OrderByOn|MS_Orientation|Name|OrderByOnLoad|OrdinalPosition|RecordCount|Required|SCD Type|ShowDatePicker|Size|Source Datatype|Source System|SourceField|SourceTable|Table Description|Table Type|TextAlign|TextFormat|TotalsRow|Type|UnicodeCompression|Updatable)$/) { # We don't dump these. They are graphical descriptions of the GUI next; From 2ef2175d56a5a5c954b1ce32a9ab97216f1f4a35 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Fri, 16 Oct 2020 16:30:11 +0200 Subject: [PATCH 05/29] 121 generated columns (#133) * Add generated columns support Many thanks to @alchemistmatt! Co-authored-by: alchemistmatt --- regression/reg_tests.sql | Bin 12608 -> 13250 bytes sqlserver2pgsql.pl | 71 +++++++++++++++++---------------------- 2 files changed, 31 insertions(+), 40 deletions(-) diff --git a/regression/reg_tests.sql b/regression/reg_tests.sql index 06af866668160b3d1df00f6e817c60b87df906f5..0ac88624209231fcf714edc7cdcd73024ded527b 100644 GIT binary patch delta 413 zcmX?*bSQlTtHI>IN@A1c#rP)Q*X7x)Zy?AxnM2Ib(48TbArFX)7!nyufOHCjGebT@ z4nrwJE>OIfA(labL4$#dfs-Mcp$w=dXYvDaX-31zaO-{=q)HR0E`Y(6AqeO;PaqBfvR%+zVM3H?U^giNy%PosFb154Pd=q&0RS?_ BO-29! delta 16 XcmX?}n}4FnkhK6(ZQ diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 7c6ce76..456fab8 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -1460,57 +1460,48 @@ sub parse_dump } } - # This is a calculated column. It doesn't exist in PG, it is not typed (I guess its type is the type of the returning function) - # So just put it as a varchar, and issue a warning in STDOUT - # FIXME this should exist in PG12 - elsif ($line =~ /^\s*\[(.*)\]\s+AS\s+\((.*)\)/) + # This is a computed column. PostgreSQL supports this as a generated column, starting with PG12 + # Will assume the data type is varchar, but this will need to be changed if the source columns are int, numeric, float, etc. + elsif ($line =~ /^\s*\[(.*)\]\s+AS\s+\((.*)\)(.*)/) { - # We just get the column name + # Get the column name my $colnumber=next_col_pos($schemaname,$tablename); my $colname = $1; my $code = $2; my $coltype = 'varchar'; + my $other_param = $3; + + # Replace square brackets in $code with double quotes + my $codequoted = $code =~ s/[\[\]]/"/gr; + my $generatedcode = " /* GENERATED ALWAYS AS ($codequoted)"; + if ($other_param =~ /PERSISTED/) { + $generatedcode .= " STORED"; + } + $generatedcode .= " */"; + $objects->{SCHEMAS}->{$schemaname}->{'TABLES'}->{$tablename}->{COLS} ->{$colname}->{POS} = $colnumber; $objects->{SCHEMAS}->{$schemaname}->{'TABLES'}->{$tablename}->{COLS} - ->{$colname}->{TYPE} = $coltype; - $objects->{SCHEMAS}->{$schemaname}->{'TABLES'}->{$tablename}->{COLS} - ->{$colname}->{NOT_NULL} = 0; + ->{$colname}->{TYPE} = $coltype . $generatedcode; + + if ($other_param =~ /NOT NULL/) { + $objects->{SCHEMAS}->{$schemaname}->{'TABLES'}->{$tablename}->{COLS} + ->{$colname}->{NOT_NULL} = 1; + } + else { + $objects->{SCHEMAS}->{$schemaname}->{'TABLES'}->{$tablename}->{COLS} + ->{$colname}->{NOT_NULL} = 0; + } - # Big fat warning + # Show a warning print STDERR - "Warning: There is a calculated column: $schemaname.$tablename.$colname. This isn't done the same way in PG at all\n"; + "\nWarning: There is a computed column: $schemaname.$tablename.$colname\n"; print STDERR - "\tFor now it has been declared as a varchar in PG, so that the values can be copied\n"; + "\tPostgreSQL 12 supports this via GENERATED ALWAYS AS (...)\n"; print STDERR - "\tYou should change its type manually in the dump (sorry for that),\n"; - print STDERR "\tA trigger has been written in the unsure file. It probably won't work as is.\n"; - print STDERR "\tPlease review it.\n"; - - # Try to correct what can be corrected from the AS : replace [COL] with NEW.COL - # It is obviously not going to work for anything a bit complicated - $code =~ s/\[(.*?)\]/NEW.$1/g; - my $triggerfunc = <{SCHEMAS}->{$schemaname}->{'TRIG_FUNCTIONS'} - ->{'trig_func_ins_or_upd' || $tablename}->{DEF} = - $triggerfunc; - $objects->{SCHEMAS}->{$schemaname}->{'TRIG_FUNCTIONS'} - ->{'trig_func_ins_or_upd' || $tablename}->{LANG} = - 'plpgsql'; - my %trigger; - $trigger{EVENTS} = 'before insert or update'; - $trigger{WHEN} = 'for each row'; - $trigger{FUNCTION} = - 'trig_func_ins_or_upd' || $tablename; # In the same schema - $trigger{NAME} = 'trig_ins_or_upd' || $tablename; - push @{$objects->{SCHEMAS}->{$schemaname}->{'TABLES'}->{$tablename} - ->{TRIGGERS}}, (\%trigger); - + "\tFor now it has been declared as a varchar and the calculation formula has been commented.\n"; + print STDERR + "\tThe formula will likely need to be manually fixed to properly refer to other columns.\n"; } elsif ($line =~ /^\s*(?:CONSTRAINT \[(.*)\] )?PRIMARY KEY (?:NON)?CLUSTERED(?: HASH)?/) @@ -2704,7 +2695,7 @@ sub generate_schema # the possible comment would go to unsure file $index_created = 2; } - + # Produce the comments for indexes if (defined $idxref->{COMMENT}) { From 8a3faa4b82f69611f2263aed07c0d7abf3543ce0 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Fri, 16 Oct 2020 16:43:07 +0200 Subject: [PATCH 06/29] 116 ignore grants (#134) * Ignore GRANT statements * add GRANT orders to regression tests Co-authored-by: alchemistmatt --- regression/basic_test/views.sql | Bin 6546 -> 6666 bytes regression/reg_tests.sql | Bin 13250 -> 13502 bytes sqlserver2pgsql.pl | 10 ++++++++++ 3 files changed, 10 insertions(+) diff --git a/regression/basic_test/views.sql b/regression/basic_test/views.sql index 093198a9d6485d163a802a6a2eeb81948a2c7139..f92587755af4174b18c7332453caa9f2f13f150f 100644 GIT binary patch delta 89 zcmbPa+-0(1lB6*&0~dokLlA=_gC9c(g91YsgC~P4LpYG-0;Ju5JU<`~0b+k3O94bq fo-HmZ8Ukb~Fhn!>GNdxZ0$Gj>!3+wM4LQXD&_NF# delta 7 OcmeA&nPj|Sk|Y2N_yXSm diff --git a/regression/reg_tests.sql b/regression/reg_tests.sql index 0ac88624209231fcf714edc7cdcd73024ded527b..f9c40f36fe419ddab81338b2a03829355b0ece1a 100644 GIT binary patch delta 192 zcmX? Date: Fri, 16 Oct 2020 16:52:35 +0200 Subject: [PATCH 07/29] 120 warn long name (#135) * Warn the user if a column or table name is more than 63 characters long Co-authored-by: alchemistmatt --- regression/reg_tests.sql | Bin 13502 -> 13694 bytes sqlserver2pgsql.pl | 5 +++++ 2 files changed, 5 insertions(+) diff --git a/regression/reg_tests.sql b/regression/reg_tests.sql index f9c40f36fe419ddab81338b2a03829355b0ece1a..bf771fc02b53c1b30aa3da8dadc1973b86d528be 100644 GIT binary patch delta 173 zcmdm&`7djOE@M3>Lo`DegDXQ2LnK2ygAap0gCB!CkmU?y`7nes_yYOyK 63) + { + print STDERR "WARNING: $identifier is more than 63 characters long; PostgreSQL will truncate the name internally\n"; + } + # Now, we protect the identifier (similar to quote_ident in PG) $identifier=~ s/"/""/g; $identifier='"'.$identifier.'"'; From bf08ca9ae5cedaaadcf4ccdc4a4c726e59b72d95 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Fri, 16 Oct 2020 18:06:36 +0200 Subject: [PATCH 08/29] Add option to skip checking the maximum length of citext columns (#136) * Add option to skip checking the maximum length of citext columns and camelcasetosnake option to the example conf file These options are commented out, to retain the current behaviour of the example conf file Co-authored-by: alchemistmatt --- example_conf_file | 12 +++++++----- sqlserver2pgsql.pl | 5 ++++- 2 files changed, 11 insertions(+), 6 deletions(-) diff --git a/example_conf_file b/example_conf_file index a2b734f..da617bb 100644 --- a/example_conf_file +++ b/example_conf_file @@ -23,12 +23,14 @@ parallelism_in=8 # Parallelism reading from SQL Server (where available) Default parallelism_out=8 # Default value is 8. Number of parallel connections used by kettle to insert data into the PostgreSQL database # Optional behaviour -case insensitive=0 # set it to 1 to generate a dump with citext and check constraints all over the place -no relabel dbo=1 # set it to 0 to convert the dbo schema to public -convert numeric to int=1 # set it to 0 to keep numeric(xx,0) as numeric(xx,0). Will be converted to smallint, int or bigint by default +case insensitive=0 # set it to 1 to generate a dump with citext and check constraints all over the place +no relabel dbo=1 # set it to 0 to convert the dbo schema to public +convert numeric to int=1 # set it to 0 to keep numeric(xx,0) as numeric(xx,0). Will be converted to smallint, int or bigint by default relabel schemas=dbo=>foo;schema1=>bar -keep identifier case=1 # keep case of database objects -validate constraints = yes # yes, after or no, should the constraints be validated by the dump ? (yes=validate during load, after after the load, no keep invalidated) +keep identifier case=1 # keep case of database objects; comment out to convert names to lowercase +#camelcasetosnake=1 # Uncomment to convert to snake case; comment out to leave names unchanged (or lowercase) +validate constraints = yes # yes, after or no, should the constraints be validated by the dump ? (yes=validate during load, after after the load, no keep invalidated) +#skip citext length check=1 # When defined, do not add a CHECK (char_length()) check for citext fields # Incremental job sort size=10000 # drives the amount of memory and temporary files that will be created by an incremental job diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index a5fd9f7..6ffe989 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -52,6 +52,7 @@ our $use_pk_if_possible; our $pforce_ssl; our $stringtype_unspecified; +our $skip_citext_length_check; # Will be set if we detect GIS objects our $requires_postgis=0; @@ -108,6 +109,7 @@ sub parse_conf_file 'ignore errors' => 'ignore_errors', 'postgresql force ssl' => 'pforce_ssl', 'stringtype unspecified' => 'stringtype_unspecified', + 'skip citext length check' => 'skip_citext_length_check', ); # Open the conf file or die @@ -158,6 +160,7 @@ sub set_default_conf_values $sp=1433 unless (defined ($sp)); $pforce_ssl=0 unless (defined ($pforce_ssl)); $stringtype_unspecified=0 unless (defined ($stringtype_unspecified)); + $skip_citext_length_check=0 unless (defined ($skip_citext_length_check)); } # Converts numeric(4,0) and similar to int, bigint, smallint @@ -326,7 +329,7 @@ sub convert_type $rettype = "citext"; # Do we have a SQL qualifier ? (we'll have to do check constraints then) - if ($sqlqual) + if ($sqlqual and not defined($skip_citext_length_check) or $sqlqual and $skip_citext_length_check == 0) { # Check we have a table name and a colname, or a typname From 898044b5128cffafac100f17c68bf071541df072 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Wed, 21 Oct 2020 11:53:23 +0200 Subject: [PATCH 09/29] parse index included columns (available for PG 11 onward) (#132) * parse index included columns (available for PG 11 onward) The created index will be written to AFTER file. --- regression/issue_59.sql | 48 +++++++++++++ sqlserver2pgsql.pl | 147 ++++++++++++++++++++-------------------- 2 files changed, 122 insertions(+), 73 deletions(-) diff --git a/regression/issue_59.sql b/regression/issue_59.sql index 3eda87a..db1073d 100644 --- a/regression/issue_59.sql +++ b/regression/issue_59.sql @@ -47,6 +47,15 @@ GO EXEC sys.sp_addextendedproperty @name=N'MS_SSMA_SOURCE', @value=N'ONEBANK.ACCOUNT.VERSION' , @level0type=N'SCHEMA',@level0name=N'dbo', @level1type=N'TABLE',@level1name=N'ACCOUNT', @level2type=N'INDEX',@level2name=N'IDX_ACCOUNT_VERSION' GO +CREATE NONCLUSTERED INDEX [IDX_ACCOUNT_BIC_IBAN] ON [dbo].[ACCOUNT] +( + [BIC] ASC, + [IBAN] ASC +) +INCLUDE ( [BACK_OFFICE_ACCOUNT_NUMBER], +[BANK_ACCOUNT_NUMBER]) WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, SORT_IN_TEMPDB = OFF, IGNORE_DUP_KEY = OFF, DROP_EXISTING = OFF, ONLINE = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [PRIMARY] +GO + CREATE TABLE [dbo].[ACCOUNT_CATEGORY]( [ID] [char](36) NOT NULL, [VERSION] [numeric](10, 0) NOT NULL, @@ -58,3 +67,42 @@ CREATE TABLE [dbo].[ACCOUNT_CATEGORY]( GO EXEC sys.sp_addextendedproperty @name=N'MS_SSMA_SOURCE', @value=N'ONEBANK.ACCOUNT_CATEGORY.UQ_INDEX' , @level0type=N'SCHEMA',@level0name=N'dbo', @level1type=N'TABLE',@level1name=N'ACCOUNT_CATEGORY', @level2type=N'INDEX',@level2name=N'UQ_INDEX' GO + +CREATE NONCLUSTERED INDEX [IDX_ACCOUNT_CATEGORY_ID] ON [dbo].[ACCOUNT_CATEGORY] +( + [ID] ASC +) +INCLUDE ( [VERSION]) WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, SORT_IN_TEMPDB = OFF, IGNORE_DUP_KEY = OFF, DROP_EXISTING = OFF, ONLINE = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [PRIMARY] +GO +/****** Object: Table [dbo].[IDX_TESTS] Script Date: 15/10/2020 14:34:08 ******/ +SET ANSI_NULLS ON +GO +SET QUOTED_IDENTIFIER ON +GO +CREATE TABLE [dbo].[IDX_TESTS]( + [I] [int] NULL, + [J] [int] NULL, + [K] [int] NULL, + [L] [int] NULL +) ON [PRIMARY] +GO +/****** Object: Index [idx_IDX_TESTS_i_part] Script Date: 15/10/2020 14:34:09 ******/ +CREATE NONCLUSTERED INDEX [IDX_IDX_TESTS_I_INCL_K_PART] ON [dbo].[IDX_TESTS] +( + [I] ASC +) +INCLUDE ( [K]) +WHERE ([L]>(10)) +WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, SORT_IN_TEMPDB = OFF, DROP_EXISTING = OFF, ONLINE = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [PRIMARY] +GO +/****** Object: Index [IDX_IDX_TESTS_I_J_PART] Script Date: 15/10/2020 14:34:09 ******/ +CREATE NONCLUSTERED INDEX [IDX_IDX_TESTS_I_J_INCL_K_L_PART] ON [dbo].[IDX_TESTS] +( + [I] ASC, + [J] ASC +) +INCLUDE ( [K], + [L]) +WHERE ([J]>(1)) +WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, SORT_IN_TEMPDB = OFF, DROP_EXISTING = OFF, ONLINE = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [PRIMARY] +GO diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 6ffe989..ca229cd 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -1917,41 +1917,45 @@ sub parse_dump next MAIN; } next - if ($idx =~ /^\(|^\)/) - ; # Begin/end of the columns declaration - if ($idx =~ /\t\[(.*)\] (ASC|DESC)(,)?/) - { - if (defined $2) - { - push @{$objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} + if ($idx =~ /^\(|^\)/) + ; # Begin/end of the columns declaration + if ($idx =~ /\t\[(.*)\] (ASC|DESC)(,)?/) { + if (defined $2) { + push @{$objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} ->{INDEXES}->{$idxname}->{COLS}}, ("$1 $2"); - } - else - { - push @{$objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} + } + else { + push @{$objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} ->{INDEXES}->{$idxname}->{COLS}}, ("$1"); - } + } } - if ($idx =~ /^INCLUDE \(/) - { - print STDERR - "Warning: This index ($schemaname.$tablename.$idxname) has some include columns. This isn't supported in PostgreSQL.\n"; - print STDERR - "\tThe columns in the INCLUDE clause have been ignored.\n"; - next - ; # Nothing equivalent in PG. Maybe if the index isn't unique, these columns should be added? + if ($idx =~ /^INCLUDE\s*\(\s*\[(.*?)\](.*)/) { + # INCLUDE coluns in indexes are available on PG11 onward + # if multiple included columns, there are declared one per line + push @{$objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} + ->{INDEXES}->{$idxname}->{INCLUDE}}, ($1); + if (index($2, ')') == -1) { + while (my $incl_line = read_and_clean($file)) { + if ($incl_line =~ /^\s*\[(.*?)\](.*)/) { + push @{$objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} + ->{INDEXES}->{$idxname}->{INCLUDE}}, ($1); + last if (index($2, ')') != -1); + } + } + } } - if ($idx =~ /^WHERE\s*\((.*)\)$/) - { - # This is a where clause. PostgreSQL has them too. But we cannot be sure this will be exactly the same. So if an index as a WHERE clause, it has to go to unsure - my $filter=$1; - $objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} - ->{INDEXES}->{$idxname}->{WHERE}="(".$filter.")"; + if ($idx =~ /^WHERE\s*\((.*)\)$/) { + # This is a where clause. PostgreSQL has them too. But we + # cannot be sure this will be exactly the same. So if an + # index as a WHERE clause, it has to go to unsure + my $filter=$1; + $objects->{SCHEMAS}->{$schemaname}->{TABLES}->{$tablename} + ->{INDEXES}->{$idxname}->{WHERE}="(".$filter.")"; } - } - } + } + } - # we do not take migrate spatial indexes + # we do not take migrate spatial indexes elsif ($line =~ /^CREATE SPATIAL INDEX/) { my $def=$line; @@ -2640,7 +2644,7 @@ sub generate_schema foreach my $table (sort keys %{$refschema->{TABLES}}) { foreach my $constraint ( - @{$refschema->{TABLES}->{$table}->{CONSTRAINTS}}) + @{$refschema->{TABLES}->{$table}->{CONSTRAINTS}}) { next unless ($constraint->{TYPE} eq 'UNIQUE'); my $consdef = "ALTER TABLE " . format_identifier($schema) . '.' . format_identifier($table) . " ADD"; @@ -2683,51 +2687,48 @@ sub generate_schema { $idxdef .= " INDEX " . format_identifier($index) . " ON " . format_identifier($schema) . '.' . format_identifier($table) . " (" . join(",", map{format_identifier_cols_index($_)} @{$idxref->{COLS}}) . ")"; - if (not defined $idxref->{WHERE} and not defined $idxref->{DISABLE}) - { - $idxdef .= ";\n"; - print AFTER $idxdef; - # the possible comment would go to after file - $index_created = 1; + + if (defined $idxref->{INCLUDE}) { + $idxdef .= " INCLUDE (" . + join(",", map{format_identifier_cols_index($_)} @{$idxref->{INCLUDE}}) + . ")"; } - else - { - # this is either a disabled index or an index with a where declaration - if (defined $idxref->{WHERE}) - { - print STDERR "Warning: index $schema.$index contains a where clause. It goes to unsure file\n"; - if ($idxref->{DISABLE}) - { - # if disabled, will be on the same line - $idxdef .= " "; - } - else - { - # otherwise, write condition on a new line - $idxdef .= "\n"; - } - $idxdef .= "WHERE (" . convert_transactsql_code($idxref->{WHERE}) . ")"; - } - $idxdef .= ";\n"; - print UNSURE $idxdef; - # the possible comment would go to unsure file - $index_created = 2; - } - - # Produce the comments for indexes - if (defined $idxref->{COMMENT}) - { - my $idxcomment = "COMMENT ON INDEX ". format_identifier($schema) . '.' . format_identifier($index) . " IS '" . $idxref->{COMMENT} . "';\n"; - if ($index_created == 1) - { - print AFTER $idxcomment; - } - elsif ($index_created == 2) - { - print UNSURE $idxcomment; - } - } + if (not defined $idxref->{WHERE} and not defined $idxref->{DISABLE}) { + $idxdef .= ";\n"; + print AFTER $idxdef; + # the possible comment would go to after file + $index_created = 1; + } + else { + + # this is either a disabled index or an index with a where declaration + if (defined $idxref->{WHERE}) { + print STDERR "Warning: index $schema.$index contains a where clause. It goes to unsure file\n"; + if ($idxref->{DISABLE}) { + # if disabled, will be on the same line + $idxdef .= " "; + } else { + # otherwise, write condition on a new line + $idxdef .= "\n"; + } + $idxdef .= "WHERE (" . convert_transactsql_code($idxref->{WHERE}) . ")"; + } + $idxdef .= ";\n"; + print UNSURE $idxdef; + # the possible comment would go to unsure file + $index_created = 2; + } + + # Produce the comments for indexes + if (defined $idxref->{COMMENT}) { + my $idxcomment = "COMMENT ON INDEX ". format_identifier($schema) . '.' . format_identifier($index) . " IS '" . $idxref->{COMMENT} . "';\n"; + if ($index_created == 1) { + print AFTER $idxcomment; + } elsif ($index_created == 2) { + print UNSURE $idxcomment; + } + } } } } From bfb908f9adbd4b1f6fb0449d4cc362d60337fa29 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Thu, 14 Jan 2021 14:54:06 +0100 Subject: [PATCH 10/29] exclude extendedproperty microsoft_database_tools_support (#138) --- sqlserver2pgsql.pl | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index ca229cd..e5a5ea8 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -2209,7 +2209,7 @@ sub parse_dump or croak "Cannot find a name for this extended property: $sqlproperty"; my $propertyname = $1; - if ($propertyname =~ /^(AggregateType|AllowZeroLength|AppendOnly|Attributes|CollatingOrder|ColumnHidden|ColumnOrder|ColumnWidth|DataUpdatable|DateCreated|DefaultValue|Description|Display Folder|Display Name|DisplayViewsOnSharePointSite|ETL Rules|Example Values|FilterOnLoad|GUID|HideNewField|LastUpdated|MS_DecimalPlaces|MS_DefaultView|MS_DiagramPane1|MS_DiagramPane2|MS_DiagramPaneCount|MS_DisplayControl|MS_Format|MS_Hyperlink|MS_IMEMode|MS_IMESentMode|MS_InputMask|MS_OrderByOn|MS_Orientation|Name|OrderByOnLoad|OrdinalPosition|RecordCount|Required|SCD Type|ShowDatePicker|Size|Source Datatype|Source System|SourceField|SourceTable|Table Description|Table Type|TextAlign|TextFormat|TotalsRow|Type|UnicodeCompression|Updatable)$/) + if ($propertyname =~ /^(AggregateType|AllowZeroLength|AppendOnly|Attributes|CollatingOrder|ColumnHidden|ColumnOrder|ColumnWidth|DataUpdatable|DateCreated|DefaultValue|Description|Display Folder|Display Name|DisplayViewsOnSharePointSite|ETL Rules|Example Values|FilterOnLoad|GUID|HideNewField|LastUpdated|microsoft_database_tools_support|MS_DecimalPlaces|MS_DefaultView|MS_DiagramPane1|MS_DiagramPane2|MS_DiagramPaneCount|MS_DisplayControl|MS_Format|MS_Hyperlink|MS_IMEMode|MS_IMESentMode|MS_InputMask|MS_OrderByOn|MS_Orientation|Name|OrderByOnLoad|OrdinalPosition|RecordCount|Required|SCD Type|ShowDatePicker|Size|Source Datatype|Source System|SourceField|SourceTable|Table Description|Table Type|TextAlign|TextFormat|TotalsRow|Type|UnicodeCompression|Updatable)$/) { # We don't dump these. They are graphical descriptions of the GUI next; From 97688656adeef78c2f5858c2c35070d7f90993ff Mon Sep 17 00:00:00 2001 From: Thibaut Date: Tue, 9 Feb 2021 11:50:59 +0100 Subject: [PATCH 11/29] add a section for operation order in the FAQ (#141) --- FAQ.md | 28 +++++++++++++++++++++++++++- 1 file changed, 27 insertions(+), 1 deletion(-) diff --git a/FAQ.md b/FAQ.md index 1d5686b..d712aba 100644 --- a/FAQ.md +++ b/FAQ.md @@ -16,6 +16,31 @@ variable to a higher value (4096) for 4GB for instance. There is another big advantage of using Kettle: you can tailor the scripts produced by sqlserver2pgsql to your needs, such as adding some conversions, schema changes. As Kettle is an ETL, it is a good tool for doing such conversions on the fly. +In which order should I run the operations? +---------------------------------- + +sqlserver2pgsql outputs several files: before, after and unsure SQL files, plus +the kettle jobs files. + +If you load all the SQL files before the data migration, you can experience +problems. For example, you can have errors when the kettle job truncates the +table at the start of the process. If a foreign key constraint is enforced, +PostgreSQL cannot truncate a table referenced in a foreign key constraint and +the job would error out. + +You should first check the unsure file, verify that the SQL is fine or correct +it if needed. Some SQL orders from the unsure files are to be run before the +data migration, for example, default column values, procedures or functions, +triggers. So move them to the before file. + +You can then load the before file (`before.sql`). Then use the kettle jobs to +migrate the data (`migration.kjb`). When this is done, load the rest of the +unsure and the after file (`unsure.sql` and `after.sql`). + +In case you are still using the original database, a specific kettle job is +created so that you can feed the change periodically to your PostgreSQL +database (`incremental.kjb`). + What is this IGNORE NULLS I have to change in kettle.properties ? ---------------------------------- @@ -39,4 +64,5 @@ doesn't exist either in PG. So more constraints will fail. Can this tool migrate functions and stored procedures? ---------------------------------- -No, Transact-SQL is very different from PostgreSQL's many PL languages. These would need a manual migration. \ No newline at end of file +No, Transact-SQL is very different from PostgreSQL's many PL languages. These would need a manual migration. + From e5c3c276063441a86280bec921e5407fb854eeff Mon Sep 17 00:00:00 2001 From: Thibaut Date: Wed, 24 Mar 2021 10:12:51 +0100 Subject: [PATCH 12/29] permit max in typequal in CREATE TYPE (#145) --- sqlserver2pgsql.pl | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index e5a5ea8..5311002 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -1824,14 +1824,18 @@ sub parse_dump my $newtype; TYPE: while (my $typeline= read_and_clean($file)) { - if ($typeline =~ /^\t\[(.*)\] \[(.*)\](?:\s*?\((\d+(?:,\d+)?)\))?(?:\s+?(?:NOT\s+?)?NULL),?$/) + if ($typeline =~ /^\t\[(.*)\] \[(.+?)\](?:\s*?\((\d+|max(?:,\d+)?)\))?(?:\s+?(?:NOT\s+?)?NULL),?$/) { # This is another column for this type $colname=$1; $type=$2; $typequal=$3; + if (defined $typequal and $typequal eq 'max') { + # max in SqlServer is the same as no typequal in pg + $typequal = undef; + } $newtype = - convert_type($type, $typequal, undef, undef, undef, undef); + convert_type($type, $typequal, undef, undef, undef, undef); push @cols_newbasetype,(format_identifier($colname) . ' ' . $newtype); } elsif ( $typeline =~ /PRIMARY KEY/) From 11f82dbd8a2f236727f31b3645981fd1916b3436 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Wed, 24 Mar 2021 15:30:11 +0100 Subject: [PATCH 13/29] update docker versions for CI (PG and perl) (#146) --- .circleci/config.yml | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/.circleci/config.yml b/.circleci/config.yml index 2cab6cf..dd16792 100644 --- a/.circleci/config.yml +++ b/.circleci/config.yml @@ -3,8 +3,8 @@ version: 2 jobs: test: docker: - - image: perl:5.24-threaded - - image: postgres:12-alpine + - image: perl:5-threaded + - image: postgres:13-alpine environment: POSTGRES_USER: postgres POSTGRES_POSTGRES: postgres From d93f0966e48442e21f4c38f07aa7e604c8841812 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Thu, 29 Apr 2021 11:14:54 +0200 Subject: [PATCH 14/29] #112 negative argument as IDENTITY (#149) * support IDENTITY with negative arguments; detect end of table more generically * Add tests * rebase patch by potrusil-osi Co-authored-by: Tomas Potrusil --- contributors | 1 + regression/issue_112.sql | 46 ++++++++++++++++++++++++++++++++++++++++ sqlserver2pgsql.pl | 10 ++++----- 3 files changed, 51 insertions(+), 6 deletions(-) create mode 100644 regression/issue_112.sql diff --git a/contributors b/contributors index 88b21c5..64b0657 100644 --- a/contributors +++ b/contributors @@ -16,6 +16,7 @@ bsacks99 keyjote mark-jay mikes-gh +postrusil-osi sebpcspkr stuey1978 diff --git a/regression/issue_112.sql b/regression/issue_112.sql new file mode 100644 index 0000000..181156c --- /dev/null +++ b/regression/issue_112.sql @@ -0,0 +1,46 @@ +CREATE TABLE [dbo].[AFElementAttributeCategory]( + [rid] [bigint] IDENTITY(-1,-1) NOT NULL, + [id] [uniqueidentifier] NOT NULL, + [rowversion] [timestamp] NOT NULL, + [fkelementversionid] [bigint] NOT NULL, + [fkparentattributeid] [uniqueidentifier] NULL, + [fkcategoryid] [uniqueidentifier] NOT NULL, + [changedby] [int] NOT NULL, + CONSTRAINT [PK_AFElementAttributeCategory] PRIMARY KEY CLUSTERED +( + [rid] ASC +)WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, IGNORE_DUP_KEY = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [ASSETS] +) ON [ASSETS] +GO + +CREATE TABLE [dbo].[AFCaseAdjustment]( + [rid] [bigint] IDENTITY(-1,-1) NOT NULL, + [id] [uniqueidentifier] NOT NULL, + [rowversion] [timestamp] NOT NULL, + [fkcaseid] [bigint] NOT NULL, + [attributeid] [uniqueidentifier] NOT NULL, + [adjustedvalue] [varbinary](max) NULL, + [comment] [nvarchar](1000) NULL, + [previousvalue] [varbinary](max) NULL, + [creator] [nvarchar](50) NULL, + [creationdate] [datetime2](7) NULL, + [changedby] [int] NOT NULL, + CONSTRAINT [PK_AFCaseAdjustment] PRIMARY KEY NONCLUSTERED +( + [rid] ASC +)WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, IGNORE_DUP_KEY = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [ANALYSIS] +) ON [ANALYSIS] TEXTIMAGE_ON [ANALYSIS] +GO + +CREATE TABLE [dbo].[sd]( + [rid] [int] IDENTITY(1000,1) NOT NULL, + [rowversion] [timestamp] NOT NULL, + [sd] [nvarchar](max) NOT NULL, + [ownerRights] [int] NOT NULL, + [lupd] [datetime2](7) NULL, + CONSTRAINT [pk_sd] PRIMARY KEY CLUSTERED +( + [rid] ASC +)WITH (PAD_INDEX = OFF, STATISTICS_NORECOMPUTE = OFF, IGNORE_DUP_KEY = OFF, ALLOW_ROW_LOCKS = ON, ALLOW_PAGE_LOCKS = ON) ON [ASSETS] +) ON [ASSETS] TEXTIMAGE_ON [ASSETS] +GO \ No newline at end of file diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 5311002..2e70cc9 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -1348,7 +1348,7 @@ sub add_column_to_table # We have an identity field. We remember the default value and # initialize the sequence correctly in the after script - $isidentity =~ /IDENTITY\s*\((\d+),\s*(\d+)\)/ + $isidentity =~ /IDENTITY\s*\((-?\d+),\s*(-?\d+)\)/ or die "Cannot understand <$isidentity>"; my $startseq = $1; my $stepseq = $2; @@ -1365,8 +1365,6 @@ sub add_column_to_table $objects->{SCHEMAS}->{$schemaname}->{SEQUENCES}->{$seqname}->{START} = $startseq; - $objects->{SCHEMAS}->{$schemaname}->{SEQUENCES}->{$seqname}->{MIN} - = $startseq; $objects->{SCHEMAS}->{$schemaname}->{SEQUENCES}->{$seqname}->{STEP} = $stepseq; $objects->{SCHEMAS}->{$schemaname}->{SEQUENCES}->{$seqname} @@ -1449,7 +1447,7 @@ sub parse_dump # column name, typical microsoft stuff :( ) # To make matters even worse, they seem to systematically add a space after it :) if ($line =~ - /^\s+\[(.*)\]\s*(?:\[(.*)\]\.)?\[(.*)\]\s*(\(.+?\))?(?: COLLATE (\S+))?( IDENTITY\s*\(\d+,\s*\d+\))?(?: ROWGUIDCOL ?)? (?:NOT FOR REPLICATION )?(?:SPARSE +)?(NOT NULL|NULL)(?:\s+CONSTRAINT \[.*\])?(?:\s+DEFAULT \((.*)\))?(?:,|$)?/ + /^\s+\[(.*)\]\s*(?:\[(.*)\]\.)?\[(.*)\]\s*(\(.+?\))?(?: COLLATE (\S+))?( IDENTITY\s*\(-?\d+,\s*-?\d+\))?(?: ROWGUIDCOL ?)? (?:NOT FOR REPLICATION )?(?:SPARSE +)?(NOT NULL|NULL)(?:\s+CONSTRAINT \[.*\])?(?:\s+DEFAULT \((.*)\))?(?:,|$)?/ ) { # Deported into a function because we can also meet alter table add columns on their own @@ -1983,7 +1981,7 @@ sub parse_dump # Added table columns… this seems to appear in SQL Server when some columns have ANSI padding, and some not. # PG follows ANSI, that is not an option. The end of the regexp is pasted from the create table elsif ($line =~ - /^ALTER TABLE \[(.*)\]\.\[(.*)\] ADD \[(.*)\] (?:\[(.*)\]\.)?\[(.*)\](\(.+?\))?( IDENTITY\(\d+,\s*\d+\))? (NOT NULL|NULL)(?: CONSTRAINT \[.*\] )?(?: DEFAULT \(.*\))?$/ + /^ALTER TABLE \[(.*)\]\.\[(.*)\] ADD \[(.*)\] (?:\[(.*)\]\.)?\[(.*)\](\(.+?\))?( IDENTITY\(-?\d+,\s*-?\d+\))? (NOT NULL|NULL)(?: CONSTRAINT \[.*\] )?(?: DEFAULT \(.*\))?$/ ) { my $schemaname=relabel_schemas($1); @@ -2883,7 +2881,7 @@ sub generate_schema # This may not be an identity. Skip it then next unless defined ($seqref->{OWNERCOL}); print AFTER "select setval('" . format_identifier($schema) . '.' - . format_identifier($sequence) . "',(select max(" + . format_identifier($sequence) . "',(select " . ($seqref->{STEP} > 0 ? "max" : "min") . "(" . format_identifier($seqref->{OWNERCOL}) .") from " . format_identifier($seqref->{OWNERSCHEMA}) . '.' . format_identifier($seqref->{OWNERTABLE}) . ")::bigint);\n"; From e4aa635046065574321e3da2bcd463547517918d Mon Sep 17 00:00:00 2001 From: Thibaut Date: Tue, 4 May 2021 11:57:11 +0200 Subject: [PATCH 15/29] Add option to convert Identity columns to "GENERATED ALWAYS AS IDENTITY" (#150) * Add option (`use_identity_column`) to convert Identity columns to "GENERATED ALWAYS AS IDENTITY" columns unset the option to use `CREATE SEQUENCE` statements instead. Co-authored-by: alchemistmatt --- example_conf_file | 1 + sqlserver2pgsql.pl | 131 ++++++++++++++++++++++++++++++--------------- 2 files changed, 90 insertions(+), 42 deletions(-) diff --git a/example_conf_file b/example_conf_file index da617bb..a93dea3 100644 --- a/example_conf_file +++ b/example_conf_file @@ -31,6 +31,7 @@ keep identifier case=1 # keep case of database objects; comment out to conv #camelcasetosnake=1 # Uncomment to convert to snake case; comment out to leave names unchanged (or lowercase) validate constraints = yes # yes, after or no, should the constraints be validated by the dump ? (yes=validate during load, after after the load, no keep invalidated) #skip citext length check=1 # When defined, do not add a CHECK (char_length()) check for citext fields +use identity column=1 # if set, use identity columns statements ('CREATE GENERATED ALWAYS') instead of creating a dedicated sequence ('CREATE SEQUENCE') # Incremental job sort size=10000 # drives the amount of memory and temporary files that will be created by an incremental job diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 2e70cc9..96b78fb 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -53,6 +53,7 @@ our $pforce_ssl; our $stringtype_unspecified; our $skip_citext_length_check; +our $use_identity_column; # Will be set if we detect GIS objects our $requires_postgis=0; @@ -109,7 +110,8 @@ sub parse_conf_file 'ignore errors' => 'ignore_errors', 'postgresql force ssl' => 'pforce_ssl', 'stringtype unspecified' => 'stringtype_unspecified', - 'skip citext length check' => 'skip_citext_length_check', + 'skip citext length check' => 'skip_citext_length_check', + 'use identity column' => 'use_identity_column', ); # Open the conf file or die @@ -161,6 +163,7 @@ sub set_default_conf_values $pforce_ssl=0 unless (defined ($pforce_ssl)); $stringtype_unspecified=0 unless (defined ($stringtype_unspecified)); $skip_citext_length_check=0 unless (defined ($skip_citext_length_check)); + $use_identity_column=0 unless (defined ($use_identity_column)); } # Converts numeric(4,0) and similar to int, bigint, smallint @@ -792,6 +795,11 @@ sub usage '1' sort all tables. LIST_OF_TABLES gives a comma separated list of tables to sort in the form 'schema1.table1,schema2.table2'. Cases are compared insensitively. + -skip_citext_length_check (Default 0) + if set, do not add a CHECK (char_length()) check for citext fields + -use_identity_column (Default 1) + if set, use identity columns statements (GENERATED ALWAYS AS + IDENTITY) instead of creating a dedicated sequence (CREATE SEQUENCE) Kettle options: if you are generating for kettle, you must provide connection information. @@ -2584,35 +2592,63 @@ sub generate_schema foreach my $sequence (sort keys %{$refschema->{SEQUENCES}}) { my $seqref = $refschema->{SEQUENCES}->{$sequence}; - print AFTER "CREATE SEQUENCE " . format_identifier($schema) . '.' . format_identifier($sequence); - if (defined $seqref->{STEP}) - { - print AFTER " INCREMENT BY ",$seqref->{STEP}; - } - if (defined $seqref->{MIN}) - { - print AFTER " MINVALUE ",$seqref->{MIN}; - } - if (defined $seqref->{MAX}) - { - print AFTER " MAXVALUE ",$seqref->{MAX}; - } - if (defined $seqref->{START}) - { - print AFTER " START WITH ",$seqref->{START}; - } - if (defined $seqref->{CACHE}) - { - print AFTER " CACHE ",$seqref->{CACHE}; - } - if (defined $seqref->{OWNERTABLE}) - { - print AFTER " OWNED BY ",format_identifier($seqref->{OWNERSCHEMA}), - '.',format_identifier($seqref->{OWNERTABLE}), - '.',format_identifier($seqref->{OWNERCOL}); - } - print AFTER ";\n"; - } + + if ($use_identity_column and defined $seqref->{OWNERTABLE}) + { + # Add a statement of the form + # ALTER TABLE "schema"."table_name" ALTER COLUMN "column_name" ADD GENERATED ALWAYS AS IDENTITY (start 1000); + + print AFTER "ALTER TABLE " . format_identifier($schema) . '.' . format_identifier($seqref->{OWNERTABLE}) . " "; + print AFTER "ALTER COLUMN " . format_identifier($seqref->{OWNERCOL}) . " ADD GENERATED ALWAYS AS IDENTITY"; + + if (defined $seqref->{START} or defined $seqref->{STEP}) + { + print AFTER " ("; + if (defined $seqref->{START}) + { + print AFTER " START WITH ",$seqref->{START}; + } + + if (defined $seqref->{STEP}) + { + print AFTER " INCREMENT BY ",$seqref->{STEP}; + } + print AFTER ")"; + } + } + else + { + print AFTER "CREATE SEQUENCE " . format_identifier($schema) . '.' . format_identifier($sequence); + if (defined $seqref->{STEP}) + { + print AFTER " INCREMENT BY ",$seqref->{STEP}; + } + if (defined $seqref->{MIN}) + { + print AFTER " MINVALUE ",$seqref->{MIN}; + } + if (defined $seqref->{MAX}) + { + print AFTER " MAXVALUE ",$seqref->{MAX}; + } + if (defined $seqref->{START}) + { + print AFTER " START WITH ",$seqref->{START}; + } + if (defined $seqref->{CACHE}) + { + print AFTER " CACHE ",$seqref->{CACHE}; + } + if (defined $seqref->{OWNERTABLE}) + { + print AFTER " OWNED BY ",format_identifier($seqref->{OWNERSCHEMA}), + '.',format_identifier($seqref->{OWNERTABLE}), + '.',format_identifier($seqref->{OWNERCOL}); + } + } + + print AFTER ";\n"; + } # Now PK. We have to go through all tables foreach my $table (sort keys %{$refschema->{TABLES}}) @@ -2861,12 +2897,20 @@ sub generate_schema . " ALTER COLUMN " . format_identifier($col) . " SET DEFAULT " . $default_value . ";\n"; if ($colref->{DEFAULT}->{UNSURE}) - { + { print UNSURE $definition; } else { - print AFTER $definition; + if ($use_identity_column and ($definition =~ /nextval.+_seq/i)) + { + # Skip this set default item + } + else + { + print AFTER $definition; + } + } } } @@ -2880,6 +2924,7 @@ sub generate_schema my $seqref = $refschema->{SEQUENCES}->{$sequence}; # This may not be an identity. Skip it then next unless defined ($seqref->{OWNERCOL}); + next if defined ($use_identity_column); print AFTER "select setval('" . format_identifier($schema) . '.' . format_identifier($sequence) . "',(select " . ($seqref->{STEP} > 0 ? "max" : "min") . "(" . format_identifier($seqref->{OWNERCOL}) .") from " @@ -3112,16 +3157,18 @@ sub resolve_name_conflicts "i" => \$case_insensitive, "nr" => \$norelabel_dbo, "num" => \$convert_numeric_to_int, - "drop_rowversion" => \$drop_rowversion, - "relabel_schemas=s" => \$relabel_schemas, - "keep_identifier_case" => \$keep_identifier_case, - "camel_to_snake" => \$camel_to_snake, - "validate_constraints=s" => \$validate_constraints, - "sort_size=i" => \$sort_size, - "use_pk_if_possible=s" => \$use_pk_if_possible, - "ignore_errors" => \$ignore_errors, - "pforce_ssl" => \$pforce_ssl, - "stringtype_unspecified" => \$stringtype_unspecified + "drop_rowversion" => \$drop_rowversion, + "relabel_schemas=s" => \$relabel_schemas, + "keep_identifier_case" => \$keep_identifier_case, + "camel_to_snake" => \$camel_to_snake, + "validate_constraints=s" => \$validate_constraints, + "sort_size=i" => \$sort_size, + "use_pk_if_possible=s" => \$use_pk_if_possible, + "ignore_errors" => \$ignore_errors, + "pforce_ssl" => \$pforce_ssl, + "stringtype_unspecified" => \$stringtype_unspecified, + "skip_citext_length_check" => \$skip_citext_length_check, + "use_identity_column" => \$use_identity_column ); # We don't understand command line or have been asked for usage From e01340a7bc74188168df939ea23c59037087778d Mon Sep 17 00:00:00 2001 From: Thibaut Date: Tue, 4 May 2021 12:20:06 +0200 Subject: [PATCH 16/29] test using sequences for identity columns (#151) --- regression/reg.pl | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/regression/reg.pl b/regression/reg.pl index f76e24d..c29b716 100755 --- a/regression/reg.pl +++ b/regression/reg.pl @@ -7,7 +7,7 @@ foreach my $file (<*.sql>) { - my @options_to_try=('-i','-nr','-num', '-keep_identifier_case', '-validate_constraints=after'); + my @options_to_try=('-i','-nr','-num', '-keep_identifier_case', '-validate_constraints=after', '-use_identity_column=0'); my @all_combinations=(''); foreach my $option (@options_to_try) { From 9027003f21fd4296cd2a9d80d652990c35a6706a Mon Sep 17 00:00:00 2001 From: madtibo Date: Tue, 4 May 2021 12:32:05 +0200 Subject: [PATCH 17/29] update contributors with alchemistmatt who made many great contributions to sqlserver2pgsql! Thank you very much for all of them! --- contributors | 1 + 1 file changed, 1 insertion(+) diff --git a/contributors b/contributors index 64b0657..7c4c9fb 100644 --- a/contributors +++ b/contributors @@ -5,6 +5,7 @@ Javier Callico (JCallico) Joshua F. Rountree (joshuairl) Julien Rouhaud (rjuju) Konstantin Mosolov (kmosolov) +Matthew Monroe (alchemistmatt) Philippe Baudoin (beaud76) Thibaut Madelaine (madtibo) Yann Verry (yanntech) From 340b38137b5b405b9b48dded5182fb39e2871843 Mon Sep 17 00:00:00 2001 From: Thibaut Date: Wed, 5 May 2021 12:53:44 +0200 Subject: [PATCH 18/29] View parsing upgrade (#152) * Add 3 functions in transact parsing (space, charindex and dateadd). * Update the regex to parse view to make it multiline --- regression/basic_test/views.sql | Bin 6666 -> 7868 bytes sqlserver2pgsql.pl | 6 ++++-- 2 files changed, 4 insertions(+), 2 deletions(-) diff --git a/regression/basic_test/views.sql b/regression/basic_test/views.sql index f92587755af4174b18c7332453caa9f2f13f150f..972c756cd9b9d6d09f7c63bdbe9967341c139df3 100644 GIT binary patch delta 310 zcmeA&*<-swN^o)suL83vgVAJbLD$K5c(;JrAGyRiQy3B%N*GcZVkawds|&(-B@CGi zxlmF0%`*HK7$=WMhmVW=xKkkmdvi4A5wxy2%&W#V5;1z5@VZ=`xrA delta 20 ccmdmE+hwvrN^tWkkss`n@3C`DJ}3JQ09vI8@c;k- diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 96b78fb..5f47f5b 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -521,7 +521,10 @@ sub convert_transact_function $code =~ s/ISNULL\s*\(/COALESCE(/gi; $code =~ s/getdate\s*\(\)/CURRENT_TIMESTAMP/gi; $code =~ s/user_name\s*\(\)/CURRENT_USER/gi; + $code =~ s/SPACE\s*\(/REPEAT(' ', /gi; + $code =~ s/charindex\s*\(\s*(.*?)\s*\,\s*(.*?)\s*\)/dPOSITION('$1' in $2)/gi; $code =~ s/datepart\s*\(\s*(.*?)\s*\,\s*(.*?)\s*\)/date_part('$1', $2)/gi; + $code =~ s/DATEADD\s*\(\s*(.*?)\s*\,\s*(.*?)\s*\,\s*(.*?)\s*\)/$3 + INTERVAL '$2 $1'/gi; $code =~ s/CONVERT\s*\(\s*NVARCHAR\s*(.*?)\s*\(\s*(.*?)\s*\s*\)\,\s*(.*?)\s*\)/CAST($3 AS varchar($2))/gi; $code =~ s/CONVERT\s*\(\s*(.*?)\s*\(\s*(.*?)\s*\s*\)\,\s*(.*?)\s*\)/CAST($3 AS $1($2))/gi; $code =~ s/CONVERT\s*\(\s*(.*?)\s*\,\s*(.*?)\s*\)/CAST($2 AS $1)/gi; @@ -1739,9 +1742,8 @@ sub parse_dump # We get rid of dbo. schemas $sql =~ s/(dbo)\./relabel_schemas($1) . '.'/eg ; # We put this in the replacement schema - # print STDERR "code view: ".$sql."\n"; # parse the query view - if ( $sql =~ /^\s*\(([^\)]+)\)\s*AS\s+SELECT\s+(.*)\s+FROM\s+(.*)$/i) { + if ( $sql =~ m/^\s*\(([^\)]+)\)\s*AS\s+SELECT\s+(.*)\s+FROM\s+(.*)$/is) { my $view_columns = $1; my $query_columns = $2; my $query_end = $3; From 397634493829d90f1f11cf29ea2aeceaaa3e4bcc Mon Sep 17 00:00:00 2001 From: Thibaut Date: Wed, 12 May 2021 09:23:37 +0200 Subject: [PATCH 19/29] Convert views (#153) * parse more transact functions * parse multiline views * update views test From 4e14df7ed8349fcd4b1df28bc7a2d6d620d07be1 Mon Sep 17 00:00:00 2001 From: "Amine B. Hassouna" <9788130+aminosbh@users.noreply.github.com> Date: Wed, 26 May 2021 14:19:53 +0100 Subject: [PATCH 20/29] Add support for 'ON UPDATE SET NULL' clause in FK (#156) --- sqlserver2pgsql.pl | 9 +++++++++ 1 file changed, 9 insertions(+) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 5f47f5b..7dfd8c4 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -2162,6 +2162,10 @@ sub parse_dump { $constraint->{ON_UPD_CASC} = 1; } + elsif ($fk =~ /^ON UPDATE SET NULL\s*$/) + { + $constraint->{ON_UPD_SET_NULL} = 1; + } elsif ($fk =~ /^NOT FOR REPLICATION\s*$/) { next; # We don't care for this, it has no meaning for PostgreSQL @@ -2817,6 +2821,11 @@ sub generate_schema { $consdef .= " ON UPDATE CASCADE"; } + if (defined $constraint->{ON_UPD_SET_NULL} + and $constraint->{ON_UPD_SET_NULL}) + { + $consdef .= " ON UPDATE SET NULL"; + } # We need a name on the constraint to be able to validate it later. Maybe it would be better to generate one # FIXME: we'll see later if a generator is needed (probably) if ($constraint->{TYPE} eq 'FK' and ($validate_constraints =~ /^after|no$/) and defined($constraint->{NAME})) From 2aaecc82b507aad379191976e6cf2a03af40f321 Mon Sep 17 00:00:00 2001 From: Ben Mares <15216687+maresb@users.noreply.github.com> Date: Wed, 26 May 2021 17:44:54 +0200 Subject: [PATCH 21/29] Remove UTF-8 BOM when necessary (#155) --- sqlserver2pgsql.pl | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 7dfd8c4..3a75ea3 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -1270,7 +1270,8 @@ sub generate_kettle my ($fd) = @_; my $line = <$fd>; return undef if (not defined $line); - $line =~ s/\r//g; # Remove \r from windows output + $line =~ s/^\x{FEFF}//; # Remove UTF-8 BOM + $line =~ s/\r//g; # Remove \r from windows output $line =~ s/EXEC(ute)?\s*(dbo|sys)\.sp_executesql( \@statement =)? N'//i ; # Remove executesql… it's a bit weird in the SQL Server's dump From 5b07d1ef0dddbb2ac309fbc3ed3d5678c9241da3 Mon Sep 17 00:00:00 2001 From: Philippe Beaudoin Date: Thu, 1 Jul 2021 10:30:35 +0200 Subject: [PATCH 22/29] Refactor the example_conf_file. Improve the global readibility by aligning the parameter values and the comments; move some parameters to a more convenient place; all optional parameters are commented with their default value; and add 3 missing parameters. Also fix a bug when either "keep identifier case" or "camelcasetosnake" are set to 0 --- example_conf_file | 104 +++++++++++++++++++++++++++------------------ sqlserver2pgsql.pl | 16 +++++-- 2 files changed, 75 insertions(+), 45 deletions(-) diff --git a/example_conf_file b/example_conf_file index a93dea3..1088657 100644 --- a/example_conf_file +++ b/example_conf_file @@ -1,41 +1,63 @@ -# Source SQL Server Dump. Obviously compulsory -sql server dump filename=/tmp/dump - -# These are used to generate SQL scripts (this is the only thing that is always done) -before file=/tmp/before -after file=/tmp/after -unsure file=/tmp/unsure - -kettle directory=/tmp/kettle # Comment this line if you don't want a kettle script to be generated -# These are ignored as long as kettle is not set -sql server database=foo -sql server host=foo_host -sql server host instance=my_instance # You can omit this if you use the default instance -sql server port=1433 -sql server username=foo_user -sql server password=foo_password -postgresql database=bar -postgresql host=bar_host -postgresql port=5432 -postgresql username=bar_user -postgresql password=bar_password -parallelism_in=8 # Parallelism reading from SQL Server (where available) Default value is 1 -parallelism_out=8 # Default value is 8. Number of parallel connections used by kettle to insert data into the PostgreSQL database - -# Optional behaviour -case insensitive=0 # set it to 1 to generate a dump with citext and check constraints all over the place -no relabel dbo=1 # set it to 0 to convert the dbo schema to public -convert numeric to int=1 # set it to 0 to keep numeric(xx,0) as numeric(xx,0). Will be converted to smallint, int or bigint by default -relabel schemas=dbo=>foo;schema1=>bar -keep identifier case=1 # keep case of database objects; comment out to convert names to lowercase -#camelcasetosnake=1 # Uncomment to convert to snake case; comment out to leave names unchanged (or lowercase) -validate constraints = yes # yes, after or no, should the constraints be validated by the dump ? (yes=validate during load, after after the load, no keep invalidated) -#skip citext length check=1 # When defined, do not add a CHECK (char_length()) check for citext fields -use identity column=1 # if set, use identity columns statements ('CREATE GENERATED ALWAYS') instead of creating a dedicated sequence ('CREATE SEQUENCE') - -# Incremental job -sort size=10000 # drives the amount of memory and temporary files that will be created by an incremental job -use pk if possible=0 # 1/list of tables, for tables where you want to try getting already sorted records - -# Ignore errors ? (will be slower, and you'll have to read the migration job's log throroughly). Ignored for incremental jobs -ignore errors=0 +# +# This file is an example of a sqlserver2pgsql configuration file. +# + +# +# Files location. +# + +# Input file. +sql server dump filename = /tmp/dump # the source SQL-Server Dump (obviously compulsory) + +# Output files. +before file = /tmp/before # SQL script to execute before loading the data +after file = /tmp/after # SQL script to execute after having loaded the data +unsure file = /tmp/unsure # SQL script containing statements to check or to adjust manually +kettle directory = /tmp/kettle # comment this line if you don't want kettle components to be generated + +# +# Optional parameters to setup specific behaviour. +# + +#case insensitive = 0 # set it to 1 to generate a dump with citext and check constraints all over the place +#skip citext length check = 0 # set it to 1 to not add a CHECK (char_length()) constraint for citext fields (when case insensitive is set to 1) +#no relabel dbo = 1 # set it to 0 to convert the dbo schema into public +#relabel schemas = dbo=>foo;schema1=>bar +#convert numeric to int = 1 # set it to 0 to keep numeric(xx,0) types as numeric(xx,0); they will be converted to smallint, int or bigint by default +#keep identifier case = 0 # set it to 1 to keep the case of database objects; by default identifier names are converted to lowercase +#camelcasetosnake = 0 # set it to 1 to convert identifiers from camelCase to snake_case; keep identifier case and camelcasetosnake cannot be both set to 1 +#validate constraints = yes # should the constraints be validated by the DDL scripts ? 'yes' = validated at constraint creation, + # 'no' = kept NOT VALID, 'after' = validated after the data load; 'yes' by default +#use identity column = 1 # set it to 0 to create an explicite SEQUENCE for identity columns (the old technic); by default, + # GENERATED ALWAYS clauses are generated +#drop rowversion = 0 # set it to 1 to ignore columns of SQL-Server type 'rowversion' or 'timestamp'; 0 by default + +# +# Parameters used for the data migration with Kettle. +# They are ignored as long as the 'kettle directory' parameter is not set. +# + +sql server database = foo +sql server host = foo_host +sql server host instance = my_instance # optional when the default instance is used +sql server port = 1433 +sql server username = foo_user +sql server password = foo_password + +postgresql database = bar +postgresql host = bar_host +postgresql port = 5432 +postgresql username = bar_user +postgresql password = bar_password + +#postgresql force ssl = 0 # set it to 1 to force a SSL session to PostgreSQL; 0 by default + +#parallelism_in = 1 # parallelism degree when reading from SQL-Server (where available); 1 by default +#parallelism_out = 8 # number of parallel connections used by kettle to insert data into the PostgreSQL database; 8 by default + +#stringtype unspecified = 0 # set it to 1 to let kettle process textual data as "not necessarily a strict PostgreSQL VARCHAR data"; 0 by default +#ignore errors = 0 # set it to 1 to not abort the data migration job when an error occurs; the parameter is ignored for incremental jobs + # warning: the migration will be slower and the job's log will need to be throroughly examined +# Incremental job parameters. +#sort size = 10000 # drives the amount of memory and temporary files that will be created by an incremental job; 10000 by default +#use pk if possible = 0 # set to either 1 or a space separated list of schema qualified table names, for tables candidated for getting already sorted rows diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 3a75ea3..c73ebfa 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -36,12 +36,12 @@ our $norelabel_dbo; # Passed as arg: should we convert DBO to public ? our $relabel_schemas; our $convert_numeric_to_int; # Should we convert numerics to int when possible ? (numeric (4,0) could be converted an int, for instance) -our $drop_rowversion; # Should we remove MSSQL timestamp/rowversion columns when converting +our $drop_rowversion; # Should we remove MSSQL timestamp/rowversion columns when converting our $kettle; our $before_file; our $after_file; our $unsure_file; -our $case_treatment=1; # 1=convert to lowercase, 2=convert to snake_case, 0 do nothing +our $case_treatment; # 0 do nothing, 1 = convert to lowercase (the default), 2 = convert to snake_case our $ignore_errors; our $keep_identifier_case; our $camel_to_snake; @@ -149,8 +149,8 @@ sub set_default_conf_values $norelabel_dbo=0 unless (defined ($norelabel_dbo)); $convert_numeric_to_int=0 unless (defined ($convert_numeric_to_int)); $drop_rowversion=0 unless (defined ($drop_rowversion)); - $case_treatment=0 if (defined ($keep_identifier_case)); - $case_treatment=2 if (defined ($camel_to_snake)); + $keep_identifier_case=0 unless (defined ($keep_identifier_case)); + $camel_to_snake=0 unless (defined ($camel_to_snake)); $parallelism_in=1 unless (defined ($parallelism_in));# the jdbc driver often errors when there are several sessions to sql server $parallelism_out=8 unless (defined ($parallelism_out)); $sort_size=10000 unless (defined ($sort_size)); @@ -164,6 +164,14 @@ sub set_default_conf_values $stringtype_unspecified=0 unless (defined ($stringtype_unspecified)); $skip_citext_length_check=0 unless (defined ($skip_citext_length_check)); $use_identity_column=0 unless (defined ($use_identity_column)); + + # Compute the case_treatment flag + $case_treatment = 1; + $case_treatment = 0 if ($keep_identifier_case); + $case_treatment = 2 if ($camel_to_snake); + if ($keep_identifier_case && $camel_to_snake) { + die "keep_identifier_case and camel_to_snake parameters cannot be both set to 1.\n"; + } } # Converts numeric(4,0) and similar to int, bigint, smallint From 83de15eb771b4a1618e500401c0938bc32468415 Mon Sep 17 00:00:00 2001 From: Philippe Beaudoin Date: Thu, 1 Jul 2021 10:55:18 +0200 Subject: [PATCH 23/29] group all the checks on parameters into a new dedicated process_check_parameters() function. --- sqlserver2pgsql.pl | 74 +++++++++++++++++++++++----------------------- 1 file changed, 37 insertions(+), 37 deletions(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index c73ebfa..c640bb2 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -142,6 +142,7 @@ sub parse_conf_file close CONF; } +# Set the default value for all parameters not set either in the configuration file or in the command line. sub set_default_conf_values { # Hard coded default values, set only if not passed or found in configuration @@ -164,13 +165,46 @@ sub set_default_conf_values $stringtype_unspecified=0 unless (defined ($stringtype_unspecified)); $skip_citext_length_check=0 unless (defined ($skip_citext_length_check)); $use_identity_column=0 unless (defined ($use_identity_column)); +} + +# Process and check the parameters. +sub process_check_parameters +{ + # We have no before, after, or unsure file + if (not $before_file or not $after_file or not $unsure_file or not $filename) { + usage(); + exit 1; + } + + if ($validate_constraints !~ '^(yes|after|no)$') { + die "'validate_constraints' should be either yes, after or no (default yes)\n"; + } + + # We have been asked for kettle, but the compulsory parameters are not there + if ($kettle + and ( not $sd + or not $sh + or not $sp + or not $su + or not defined($sw) # password can be empty, it just has to be defined + or not $pd + or not $ph + or not $pp + or not $pu + or not defined($pw) # password can be empty, it just has to be defined + )) { + usage(); + print + "You have to provide all connection information, if using -k or kettle directory set in configuration file\n"; + exit 1; + } # Compute the case_treatment flag $case_treatment = 1; $case_treatment = 0 if ($keep_identifier_case); $case_treatment = 2 if ($camel_to_snake); if ($keep_identifier_case && $camel_to_snake) { - die "keep_identifier_case and camel_to_snake parameters cannot be both set to 1.\n"; + die "'keep_identifier_case' and 'camel_to_snake parameters' cannot be both set to 1.\n"; } } @@ -3208,46 +3242,12 @@ sub resolve_name_conflicts # Set default values for anything not set yet set_default_conf_values(); -# We have no before, after, or unsure -if ( not $before_file - or not $after_file - or not $unsure_file - or not $filename) -{ - usage(); - exit 1; -} - -if ($validate_constraints !~ '^(yes|after|no)$') -{ - croak "validate_constraints should be yes, after or no (default yes)\n"; -} - -# We have been asked for kettle, but the compulsory parameters are not there -if ($kettle - and ( not $sd - or not $sh - or not $sp - or not $su - or not defined($sw) # password can be empty, it just has to be defined - or not $pd - or not $ph - or not $pp - or not $pu - or not defined($pw) # password can be empty, it just has to be defined - ) -) -{ - usage(); - print - "You have to provide all connection information, if using -k or kettle directory set in configuration file\n"; - exit 1; -} +# Perform checks on parameters +process_check_parameters(); # We need to build %relabel_schemas from $relabel_schemas build_relabel_schemas(); - # Read SQL Server's dump file parse_dump(); From 27014faba3a498508d1daeb6a56b6947a18db788 Mon Sep 17 00:00:00 2001 From: Philippe Beaudoin Date: Sat, 3 Jul 2021 20:59:04 +0200 Subject: [PATCH 24/29] Add options to let sqlserver2pgsql produce a file containing the list of all columns, with the SQL-Server and PostgreSQL names of the schemas, tables and columns names. "col_map_file" defines the output file name. No file is produced if the parameter is not set. When "col_map_file_header" is set, a header line is added to the file. "col_map_file_delimiter" defines the field delimiter. The parameter is a string of 1 or several characters, and may be \t, \n or \r. The default is \t. Per a proposal from alchemistmatt. Improved by me. --- README.md | 8 ++++- example_conf_file | 3 ++ sqlserver2pgsql.pl | 84 +++++++++++++++++++++++++++++++++++----------- 3 files changed, 74 insertions(+), 21 deletions(-) diff --git a/README.md b/README.md index bebf4f4..c426cfe 100644 --- a/README.md +++ b/README.md @@ -98,7 +98,13 @@ so the scale is often not important. `-keep_identifier_case`: don't convert the dump to all lower case. This is not recommended, as you'll have to put every identifier (column, table…) in double quotes… -`-camel_to_snake`: convert the object name (table, column, index...) from CamelCase to snake_case. Only do this if you are willing to change all your queries (or you use an ORM for instance) +`-camel_to_snake`: convert the object name (table, column, index...) from CamelCase to snake_case. Only do this if you are willing to change all your queries (or you use an ORM for instance). + +`-col_map_file`: specifies an output text file containing SQL-Server and PostgreSQL schemas, tables and columns names (1 line per column). + +`-col_map_file_header`: add a header line to the col_map_file (no header by default). + +`-col_map_file_delimiter`: specify a field delimiter for the col_map_file (TAB by default). `-validate_constraints=yes/after/no`: for foreign keys, if yes: foreign keys are created as valid in the after script (default) if no: they are created as not valid (enforced only for new rows) diff --git a/example_conf_file b/example_conf_file index 1088657..93fefa2 100644 --- a/example_conf_file +++ b/example_conf_file @@ -26,6 +26,9 @@ kettle directory = /tmp/kettle # comment this line if you don't w #convert numeric to int = 1 # set it to 0 to keep numeric(xx,0) types as numeric(xx,0); they will be converted to smallint, int or bigint by default #keep identifier case = 0 # set it to 1 to keep the case of database objects; by default identifier names are converted to lowercase #camelcasetosnake = 0 # set it to 1 to convert identifiers from camelCase to snake_case; keep identifier case and camelcasetosnake cannot be both set to 1 +#col map file = /tmp/map # text file with SQL-Server and PostgreSQL schemas, tables and columns names +#col map file header = 0 # set if to 1 to add a header line to the col map file +#col map file delimiter = \t # the fields delimiter in the col map file (TAB by default) #validate constraints = yes # should the constraints be validated by the DDL scripts ? 'yes' = validated at constraint creation, # 'no' = kept NOT VALID, 'after' = validated after the data load; 'yes' by default #use identity column = 1 # set it to 0 to create an explicite SEQUENCE for identity columns (the old technic); by default, diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index c640bb2..c1bff1b 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -41,6 +41,9 @@ our $before_file; our $after_file; our $unsure_file; +our $col_map_file; +our $col_map_file_header; +our $col_map_file_delimiter; our $case_treatment; # 0 do nothing, 1 = convert to lowercase (the default), 2 = convert to snake_case our $ignore_errors; our $keep_identifier_case; @@ -99,11 +102,14 @@ sub parse_conf_file 'sql server dump filename' => 'filename', 'case insensitive' => 'case_insensitive', 'no relabel dbo' => 'norelabel_dbo', - 'convert numeric to int' => 'convert_numeric_to_int', - 'drop rowversion' => 'drop_rowversion', 'relabel schemas' => 'relabel_schemas', 'keep identifier case' => 'keep_identifier_case', 'camelcasetosnake' => 'camel_to_snake', + 'col map file' => 'col_map_file', + 'col map file header' => 'col_map_file_header', + 'col map file delimiter' => 'col_map_file_delimiter', + 'convert numeric to int' => 'convert_numeric_to_int', + 'drop rowversion' => 'drop_rowversion', 'validate constraints' => 'validate_constraints', 'sort size' => 'sort_size', 'use pk if possible' => 'use_pk_if_possible', @@ -145,22 +151,23 @@ sub parse_conf_file # Set the default value for all parameters not set either in the configuration file or in the command line. sub set_default_conf_values { - # Hard coded default values, set only if not passed or found in configuration $case_insensitive=0 unless (defined ($case_insensitive)); $norelabel_dbo=0 unless (defined ($norelabel_dbo)); - $convert_numeric_to_int=0 unless (defined ($convert_numeric_to_int)); - $drop_rowversion=0 unless (defined ($drop_rowversion)); $keep_identifier_case=0 unless (defined ($keep_identifier_case)); $camel_to_snake=0 unless (defined ($camel_to_snake)); - $parallelism_in=1 unless (defined ($parallelism_in));# the jdbc driver often errors when there are several sessions to sql server + $col_map_file = "" unless (defined($col_map_file)); + $col_map_file_header = 0 unless (defined($col_map_file_header)); + $col_map_file_delimiter = '\t' unless (defined($col_map_file_delimiter)); + $convert_numeric_to_int=0 unless (defined ($convert_numeric_to_int)); + $drop_rowversion=0 unless (defined ($drop_rowversion)); + $parallelism_in=1 unless (defined ($parallelism_in)); # the jdbc driver often errors when there are several sessions to sql server $parallelism_out=8 unless (defined ($parallelism_out)); $sort_size=10000 unless (defined ($sort_size)); $use_pk_if_possible=0 unless (defined ($use_pk_if_possible)); $validate_constraints='yes' unless (defined ($validate_constraints)); $ignore_errors=0 unless (defined ($ignore_errors)); - # Default ports for PostgreSQL and SQL Server - $pp=5432 unless (defined ($pp)); - $sp=1433 unless (defined ($sp)); + $pp=5432 unless (defined ($pp)); # Default port for PostgreSQL + $sp=1433 unless (defined ($sp)); # Default port for SQL-Server $pforce_ssl=0 unless (defined ($pforce_ssl)); $stringtype_unspecified=0 unless (defined ($stringtype_unspecified)); $skip_citext_length_check=0 unless (defined ($skip_citext_length_check)); @@ -206,6 +213,11 @@ sub process_check_parameters if ($keep_identifier_case && $camel_to_snake) { die "'keep_identifier_case' and 'camel_to_snake parameters' cannot be both set to 1.\n"; } + + # In $col_map_file_delimiter, replace \t, \n and \r by the real equivalent characters + $col_map_file_delimiter =~ s/\\t/\t/g; + $col_map_file_delimiter =~ s/\\n/\n/g; + $col_map_file_delimiter =~ s/\\r/\r/g; } # Converts numeric(4,0) and similar to int, bigint, smallint @@ -779,7 +791,7 @@ sub usage { print qq{ Usage: - sqlserver2pgsql.pl -b BEFORE_FILE -a AFTER_FILE -u UNSURE_FILE -f SQLSERVER_SCHEMA_FILE + sqlserver2pgsql.pl -f SQLSERVER_SCHEMA_FILE -b BEFORE_FILE -a AFTER_FILE -u UNSURE_FILE ... OPTIONS Description: @@ -790,11 +802,11 @@ sub usage Optionnaly, using the '-k' option, it will generate a kettle job to transfer all data. -Mandatory options: +Mandatory parameters: SQL Server schema input file: -f SQLSERVER_SCHEMA_FILE - a readable SQL Server SQL structure dump. + a readable SQL-Server SQL structure dump. PostgreSQL output schema files: -b BEFORE_SCRIPT @@ -805,18 +817,13 @@ sub usage contains objects we attempt to migrate, but cannot guarantee, such as views or complex indexes. -Options: +Other options: -conf CONFIGURATION_FILE uses a configuration file. All options can be set there. Command line options will overwrite conf options. - -i the resulting PostgreSQL names will be case-insensitive. -nr the SQL Server 'dbo' schema will not be translated to PostgreSQL 'public' schema. 'dbo' will stay 'dbo'. - -camel_to_snake - all object names are converted from 'camelCase' to 'camel_case', - which is more often used in PostgreSQL. Do not use this unless you - are ready to do SQL query changes in the client. -relabel_schemas 'SOURCE1=>DEST1;SOURCE2=>DEST2' gives a list of schemas to rename. Quote this option to prevent the shell to alter it. The '-nr' option cancels the default 'dbo' to @@ -824,6 +831,17 @@ sub usage -keep_identifier_case keep the case of SQL server database objects. This option is not advised. Default is to lowercase everything. + -camel_to_snake + all object names are converted from 'camelCase' to 'camel_case', + which is more often used in PostgreSQL. Do not use this unless you + are ready to do SQL query changes in the client. + -col_map_file + optional text file with old and new schema, table and column names + -col_map_file_header + add a header line to the col_map_file + -col_map_file_delimiter + the field delimiter used in the col_map_file (TAB by default) + -i the resulting PostgreSQL names will be case-insensitive. -num convert numeric 'xxx,0' to int, bigint, etc. Will not keep numeric scale and precision for the converted. -drop_rowversion (Default 0) @@ -2530,6 +2548,11 @@ sub generate_schema open BEFORE, ">:utf8", $before_file or die "Cannot open $before_file, $!"; open AFTER, ">:utf8", $after_file or die "Cannot open $after_file, $!"; open UNSURE, ">:utf8", $unsure_file or die "Cannot open $unsure_file, $!"; + if ($col_map_file) + { + open NAMEMAP, ">:utf8", $col_map_file or die "Cannot open $col_map_file, $!"; + } + print BEFORE "\\set ON_ERROR_STOP\n"; print BEFORE "\\set ECHO all\n"; print BEFORE "BEGIN;\n"; @@ -2539,6 +2562,15 @@ sub generate_schema print UNSURE "\\set ON_ERROR_STOP\n"; print AFTER "\\set ECHO all\n"; print UNSURE "BEGIN;\n"; + if ($col_map_file && $col_map_file_header) + { + print NAMEMAP "Source_schema" . $col_map_file_delimiter . + "Source_table" . $col_map_file_delimiter . + "Source_column" . $col_map_file_delimiter . + "Schema" . $col_map_file_delimiter . + "Table" . $col_map_file_delimiter . + "Column\n"; + } # Are we case insensitive ? We have to install citext then # Won't work on pre-9.1 database. But as this is a migration tool @@ -2610,6 +2642,8 @@ sub generate_schema # The tables foreach my $table (sort keys %{$refschema->{TABLES}}) { + my $origschema = $refschema->{TABLES}->{$table}->{origschema}; + my $newtablename = format_identifier($table); my @colsdef; foreach my $col ( sort { @@ -2620,14 +2654,20 @@ sub generate_schema { my $colref = $refschema->{TABLES}->{$table}->{COLS}->{$col}; - my $coldef = format_identifier($col) . " " . $colref->{TYPE}; + my $newcolname = format_identifier($col); + my $coldef = $newcolname . " " . $colref->{TYPE}; if ($colref->{NOT_NULL}) { $coldef .= ' NOT NULL'; } push @colsdef, ($coldef); + if ($col_map_file) + { + print NAMEMAP $origschema . $col_map_file_delimiter . $table . $col_map_file_delimiter . $col . $col_map_file_delimiter + . $schema . $col_map_file_delimiter . $newtablename . $col_map_file_delimiter . $newcolname . "\n"; + } } - print BEFORE "CREATE TABLE " . format_identifier($schema) . '.' . format_identifier($table) . "( \n\t" + print BEFORE "CREATE TABLE " . format_identifier($schema) . '.' . $newtablename . "( \n\t" . join(",\n\t", @colsdef) . ");\n\n"; } @@ -3084,6 +3124,7 @@ sub generate_schema close BEFORE; close AFTER; close UNSURE; + close NAMEMAP if ($col_map_file); } @@ -3211,6 +3252,9 @@ sub resolve_name_conflicts "i" => \$case_insensitive, "nr" => \$norelabel_dbo, "num" => \$convert_numeric_to_int, + "col_map_file=s" => \$col_map_file, + "col_map_file_header" => \$col_map_file_header, + "col_map_file_delimiter=s" => \$col_map_file_delimiter, "drop_rowversion" => \$drop_rowversion, "relabel_schemas=s" => \$relabel_schemas, "keep_identifier_case" => \$keep_identifier_case, From 0f9e080738b595ba69632a6f01a99a181aa7cecd Mon Sep 17 00:00:00 2001 From: Philippe Beaudoin Date: Fri, 6 Aug 2021 09:18:23 +0200 Subject: [PATCH 25/29] When a constraint name is greater than 63 characters, do not use it in the generated scripts and let PostgreSQL rebuild the name at constraint creation time. --- regression/reg_tests.sql | Bin 13694 -> 13792 bytes sqlserver2pgsql.pl | 35 ++++++++++++++++++++++------------- 2 files changed, 22 insertions(+), 13 deletions(-) diff --git a/regression/reg_tests.sql b/regression/reg_tests.sql index bf771fc02b53c1b30aa3da8dadc1973b86d528be..123dd9098157e8ba5c62d66e4ba0c54048de81ed 100644 GIT binary patch delta 108 zcmeyD^&p$+|G&u>#WW_L=M>p!U&lS!kWpfCsiMH-Kt?gwa0XX~cm^LJ83LpO82lN6 xfh>21AfQMvLm*Hl9?W)P2xagD@?3y?Cx%EM=?PR93{>UA;I{c5_X15u1^}288R7r{ delta 37 tcmaEm{V$8@-~Y*m;+m8AI41vN=h 63) + { + print STDERR "Warning: because of its length, the constraint name $name is ignored and will be set by Postgres at execution time.\n"; + return 0; + } + return 1; +} + # This one will try to convert what can obviously be converted from transact to PG # Things such as getdate() which can become CURRENT_TIMESTAMP sub convert_transact_function @@ -2753,7 +2765,7 @@ sub generate_schema next; } my $pkdef = "ALTER TABLE " . format_identifier($schema) . '.' . format_identifier($table) . " ADD"; - if (defined $refpk->{NAME}) + if (defined $refpk->{NAME} and is_constraint_name_valid($refpk->{NAME})) { $pkdef .= " CONSTRAINT " . format_identifier($refpk->{NAME}); } @@ -2775,7 +2787,7 @@ sub generate_schema { next unless ($constraint->{TYPE} eq 'UNIQUE'); my $consdef = "ALTER TABLE " . format_identifier($schema) . '.' . format_identifier($table) . " ADD"; - if (defined $constraint->{NAME}) + if (defined $constraint->{NAME} and is_constraint_name_valid($constraint->{NAME})) { $consdef .= " CONSTRAINT " . format_identifier($constraint->{NAME}); } @@ -2872,12 +2884,11 @@ sub generate_schema { next if ($constraint->{TYPE} =~ /^UNIQUE|PK$/); my $consdef = "ALTER TABLE " . format_identifier($schema) . '.' . format_identifier($table) . " ADD"; - if (defined $constraint->{NAME}) + if (defined $constraint->{NAME} and is_constraint_name_valid($constraint->{NAME})) { $consdef .= " CONSTRAINT " . format_identifier($constraint->{NAME}); } - if ($constraint->{TYPE} eq - 'FK') # COLS are already a comma separated list + if ($constraint->{TYPE} eq 'FK') # COLS are already a comma separated list { # We need to convert the column list to protected names my @localcollist=map{format_identifier($_)} @{$constraint->{LOCAL_COLS}}; @@ -2911,23 +2922,22 @@ sub generate_schema } # We need a name on the constraint to be able to validate it later. Maybe it would be better to generate one # FIXME: we'll see later if a generator is needed (probably) - if ($constraint->{TYPE} eq 'FK' and ($validate_constraints =~ /^after|no$/) and defined($constraint->{NAME})) + if (($validate_constraints =~ /^after|no$/) and defined($constraint->{NAME}) and is_constraint_name_valid($constraint->{NAME})) { $consdef .= " NOT VALID"; } $consdef .= ";\n"; print AFTER $consdef; - if ($constraint->{TYPE} eq 'FK' and $validate_constraints eq 'after' and defined $constraint->{NAME}) + if ($validate_constraints eq 'after' and defined $constraint->{NAME} and is_constraint_name_valid($constraint->{NAME})) { print UNSURE "ALTER TABLE " . format_identifier($schema) . '.' . format_identifier($table) . " VALIDATE CONSTRAINT " . format_identifier($constraint->{NAME}) . ";\n"; } } elsif ($constraint->{TYPE} eq 'CHECK') - { - $consdef .= " CHECK (" . convert_transactsql_code($constraint->{TEXT}) . ");\n"; - print UNSURE $consdef - ; # Check constraints are SQL, so cannot be sure - } + { + $consdef .= " CHECK (" . convert_transactsql_code($constraint->{TEXT}) . ");\n"; + print UNSURE $consdef; # Check constraints are SQL, so cannot be sure + } elsif ($constraint->{TYPE} eq 'CHECK_CITEXT') { # These have been generated here, for citext mostly. So we know their syntax is ok @@ -3182,7 +3192,6 @@ sub resolve_name_conflicts } $known_names{format_identifier($domain."2pgd")}=1; } - } # Then we scan all indexes From 6f5af05b52a8361df4cd5c713c4c1fbcbf1b6dcc Mon Sep 17 00:00:00 2001 From: spanevin Date: Wed, 13 Jul 2022 13:36:11 +0300 Subject: [PATCH 26/29] Add support of schema-level comments --- sqlserver2pgsql.pl | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 08d7935..2f0a672 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -2313,7 +2313,7 @@ sub parse_dump # I hope it will be sufficient (won't be if someone decides to end a comment with a quote) unless ($sqlproperty =~ - /^EXEC sys.sp_addextendedproperty \@name=N'(.*?)'\s*,\s*\@value=N'(.*)'\s*,\s*\@level0type=N'(.*?)'\s*,\s*\@level0name=N'(.*?)'\s*(?:,\s*\@level1type=N'(.*?)'\s*,\s*\@level1name=N'(.*?)')\s*?(?:,\s*\@level2type=N'(.*?)'\s*,\s*\@level2name=N'(.*?)')?/s) + /^EXEC sys.sp_addextendedproperty \@name=N'(.*?)'\s*(?:,\s*\@value=N'(.*?)'\s*)?(?:,\s*\@level0type=N'(.*?)'\s*)?(?:,\s*\@level0name=N'(.*?)'\s*)?(?:,\s*\@level1type=N'(.*?)'\s*,\s*\@level1name=N'(.*?)')?\s*?(?:,\s*\@level2type=N'(.*?)'\s*,\s*\@level2name=N'(.*?)')?/s) { # Not parsing a comment should not stop print STDERR "Could not parse <$sqlproperty>. Ignored.\n"; @@ -2322,7 +2322,11 @@ sub parse_dump my ($comment, $schema, $obj, $objname, $subobj, $subobjname) = ($2, $4, $5, $6, $7, $8); $schema=relabel_schemas($schema); - if ($obj eq 'TABLE' and not defined $subobj) + if (not defined $obj) + { + $objects->{SCHEMAS}->{$schema}->{COMMENT} = $comment; + } + elsif ($obj eq 'TABLE' and not defined $subobj) { $objects->{SCHEMAS}->{$schema}->{TABLES}->{$objname}->{COMMENT} = $comment; @@ -3040,6 +3044,13 @@ sub generate_schema # Comments on tables and columns while (my ($schema, $refschema) = each %{$objects->{SCHEMAS}}) { + # Comments on schemas + if (defined($refschema->{COMMENT})) + { + print AFTER "COMMENT ON SCHEMA " . format_identifier($schema) . " IS '" + . $refschema->{COMMENT} . "';\n"; + } + # Comments on tables foreach my $table (sort keys %{$refschema->{TABLES}}) { From eea47867d755249ed73fa24d4f08df22a4ead60f Mon Sep 17 00:00:00 2001 From: Vladislav Moiseev Date: Thu, 17 Aug 2023 07:13:26 +0400 Subject: [PATCH 27/29] Added sforce_ssl flag for Kettle --- sqlserver2pgsql.pl | 26 ++++++++++++++++++++++++++ 1 file changed, 26 insertions(+) diff --git a/sqlserver2pgsql.pl b/sqlserver2pgsql.pl index 2f0a672..a25144d 100755 --- a/sqlserver2pgsql.pl +++ b/sqlserver2pgsql.pl @@ -53,6 +53,7 @@ our $parallelism_out; our $sort_size; our $use_pk_if_possible; +our $sforce_ssl; our $pforce_ssl; our $stringtype_unspecified; our $skip_citext_length_check; @@ -114,6 +115,7 @@ sub parse_conf_file 'sort size' => 'sort_size', 'use pk if possible' => 'use_pk_if_possible', 'ignore errors' => 'ignore_errors', + 'sql server force ssl' => 'sforce_ssl', 'postgresql force ssl' => 'pforce_ssl', 'stringtype unspecified' => 'stringtype_unspecified', 'skip citext length check' => 'skip_citext_length_check', @@ -168,6 +170,7 @@ sub set_default_conf_values $ignore_errors=0 unless (defined ($ignore_errors)); $pp=5432 unless (defined ($pp)); # Default port for PostgreSQL $sp=1433 unless (defined ($sp)); # Default port for SQL-Server + $sforce_ssl=0 unless (defined ($sforce_ssl)); $pforce_ssl=0 unless (defined ($pforce_ssl)); $stringtype_unspecified=0 unless (defined ($stringtype_unspecified)); $skip_citext_length_check=0 unless (defined ($skip_citext_length_check)); @@ -894,6 +897,8 @@ sub usage parallelism level for the kettle job (input, SQL Server). Default 1. -po PARALLELISM_OUT parallelism level for the kettle job (output, PostgreSQL). Default 8. + -sforce_ssl + force a SSL session to SQL Server -pforce_ssl force a SSL session to PostgreSQL -stringtype_unspecified @@ -1029,6 +1034,14 @@ sub generate_kettle $newtemplate =~ s/Y<\/use_batch>/N<\/use_batch>/g; # Cannot use batch mode with ignore errors } + if ($sforce_ssl) + { + $newtemplate =~ s/__sforce_ssl__/EXTRA_OPTION_MSSQL.ssl<\/code>require<\/attribute><\/attribute>/g; + } + else + { + $newtemplate =~ s/__sforce_ssl__//g; + } if ($pforce_ssl) { $newtemplate =~ s/__pforce_ssl__/EXTRA_OPTION_POSTGRESQL.ssl<\/code>true<\/attribute><\/attribute>\nEXTRA_OPTION_POSTGRESQL.sslfactory<\/code>org.postgresql.ssl.NonValidatingFactory<\/attribute><\/attribute>/g; @@ -1066,6 +1079,14 @@ sub generate_kettle $newincrementaltemplate =~ s/__PARALLELISM_OUT__/$parallelism_out/g; $newincrementaltemplate =~ s/__sort_size__/$sort_size/g; + if ($sforce_ssl) + { + $newincrementaltemplate =~ s/__sforce_ssl__/EXTRA_OPTION_MSSQL.ssl<\/code>require<\/attribute><\/attribute>/g; + } + else + { + $newincrementaltemplate =~ s/__sforce_ssl__//g; + } if ($pforce_ssl) { $newincrementaltemplate =~ s/__pforce_ssl__/EXTRA_OPTION_POSTGRESQL.ssl<\/code>true<\/attribute><\/attribute>\nEXTRA_OPTION_POSTGRESQL.sslfactory<\/code>org.postgresql.ssl.NonValidatingFactory<\/attribute><\/attribute>/g; @@ -3283,6 +3304,7 @@ sub resolve_name_conflicts "sort_size=i" => \$sort_size, "use_pk_if_possible=s" => \$use_pk_if_possible, "ignore_errors" => \$ignore_errors, + "sforce_ssl" => \$sforce_ssl, "pforce_ssl" => \$pforce_ssl, "stringtype_unspecified" => \$stringtype_unspecified, "skip_citext_length_check" => \$skip_citext_length_check, @@ -3420,6 +3442,7 @@ BEGIN + __sforce_ssl__ EXTRA_OPTION_MSSQL.instance__sqlserver_instance__ FORCE_IDENTIFIERS_TO_LOWERCASEN FORCE_IDENTIFIERS_TO_UPPERCASEN @@ -3782,6 +3805,7 @@ BEGIN + __sforce_ssl__ EXTRA_OPTION_MSSQL.instance__sqlserver_instance__ FORCE_IDENTIFIERS_TO_LOWERCASEN FORCE_IDENTIFIERS_TO_UPPERCASEN @@ -4351,6 +4375,7 @@ BEGIN + __sforce_ssl__ EXTRA_OPTION_MSSQL.instance__sqlserver_instance__ FORCE_IDENTIFIERS_TO_LOWERCASEN FORCE_IDENTIFIERS_TO_UPPERCASEN @@ -4781,6 +4806,7 @@ BEGIN + __sforce_ssl__ EXTRA_OPTION_MSSQL.instance__sqlserver_instance__ FORCE_IDENTIFIERS_TO_LOWERCASEN FORCE_IDENTIFIERS_TO_UPPERCASEN From 266465b178db3b5c755a1384fb05911b0ccc0c31 Mon Sep 17 00:00:00 2001 From: Vladislav Moiseev Date: Mon, 9 Oct 2023 20:21:55 +0400 Subject: [PATCH 28/29] Add info about -sforce_ssl flag into README.md --- README.md | 1 + 1 file changed, 1 insertion(+) diff --git a/README.md b/README.md index c426cfe..d8f04bc 100644 --- a/README.md +++ b/README.md @@ -135,6 +135,7 @@ cleartext, so don't make this directory public): `-pp` : postgresql port `-pu` : postgresql username `-pw` : postgresql password +`-sforce_ssl` : force a SSL connection to your SQL Server database. Required if ForceEncryption option is set to 'Yes' `-pforce_ssl` : force a SSL connection to your PostgreSQL database. ssl=on should be set on the PostgreSQL server `-f` : the SQL Server structure dump file -ignore_errors : ignore insert errors (not advised, you'll need to examine kettle's logs, and it will be slower) From d09f6cdf9fafda25c117e35fa5fe826e55d3c878 Mon Sep 17 00:00:00 2001 From: Florent Jardin Date: Tue, 10 Oct 2023 15:17:38 +0200 Subject: [PATCH 29/29] Update README.md --- README.md | 19 ++++++++++++++++++- 1 file changed, 18 insertions(+), 1 deletion(-) diff --git a/README.md b/README.md index d8f04bc..2b56f93 100644 --- a/README.md +++ b/README.md @@ -124,24 +124,41 @@ one for each table to copy, plus the one for the job) You'll also need to specify the connection parameters. They will be stored inside the kettle files (in cleartext, so don't make this directory public): + `-sd` : sql server database + `-sh` : sql server host + `-si` : sql server host instance + `-sp` : sql server port (usually 1433) + `-su` : sql server username + `-sw` : sql server password + `-pd` : postgresql database + `-ph` : postgresql host + `-pp` : postgresql port + `-pu` : postgresql username + `-pw` : postgresql password + `-sforce_ssl` : force a SSL connection to your SQL Server database. Required if ForceEncryption option is set to 'Yes' + `-pforce_ssl` : force a SSL connection to your PostgreSQL database. ssl=on should be set on the PostgreSQL server + `-f` : the SQL Server structure dump file --ignore_errors : ignore insert errors (not advised, you'll need to examine kettle's logs, and it will be slower) + +`-ignore_errors` : ignore insert errors (not advised, you'll need to examine kettle's logs, and it will be slower) `-pi` : The parallelism used in kettle jobs to read from SQL Server (1 by default, the jdbc driver frequently errors out when larger) + `-po` : The parallelism used in kettle jobs to write to PostgresSQL: there will be this amount of sessions used to insert into PostgreSQL. Default to 8 + `-sort_size=100000`: sort size to use for incremental jobs. Default is 10000, to try to be on the safe side (see below). We don't sort in databases for two reasons: the sort order (collation for strings for example) can be different between SQL Server