postgresql/contrib/unaccent/expected/unaccent.out

/*
 * This test must be run in a database with UTF-8 encoding,
 * because other encodings don't support all the characters used.
 */
SELECT getdatabaseencoding() <> 'UTF8'
       AS skip_test \gset
\if :skip_test
\quit
\endif
CREATE EXTENSION unaccent;
SET client_encoding TO 'UTF8';
SELECT unaccent('foobar');
 unaccent 
----------
 foobar
(1 row)

SELECT unaccent('ёлка');
 unaccent 
----------
 елка
(1 row)

SELECT unaccent('ЁЖИК');
 unaccent 
----------
 ЕЖИК
(1 row)

SELECT unaccent('˃˖˗˜');
 unaccent 
----------
 >+-~
(1 row)

SELECT unaccent('À');  -- Remove combining diacritical 0x0300
 unaccent 
----------
 A
(1 row)

SELECT unaccent('℃℉'); -- degree signs
 unaccent 
----------
 °C°F
(1 row)

SELECT unaccent('℗'); -- sound recording copyright
 unaccent 
----------
 (P)
(1 row)

SELECT unaccent('1½'); -- math expression with whitespace
 unaccent 
----------
 1 1/2
(1 row)

SELECT unaccent('〝'); -- quote
 unaccent 
----------
 "
(1 row)

SELECT unaccent('unaccent', 'foobar');
 unaccent 
----------
 foobar
(1 row)

SELECT unaccent('unaccent', 'ёлка');
 unaccent 
----------
 елка
(1 row)

SELECT unaccent('unaccent', 'ЁЖИК');
 unaccent 
----------
 ЕЖИК
(1 row)

SELECT unaccent('unaccent', '˃˖˗˜');
 unaccent 
----------
 >+-~
(1 row)

SELECT unaccent('unaccent', 'À');
 unaccent 
----------
 A
(1 row)

SELECT unaccent('unaccent', '℃℉');
 unaccent 
----------
 °C°F
(1 row)

SELECT unaccent('unaccent', '℗');
 unaccent 
----------
 (P)
(1 row)

SELECT unaccent('unaccent', '1½');
 unaccent 
----------
 1 1/2
(1 row)

SELECT unaccent('unaccent', '〝');
 unaccent 
----------
 "
(1 row)

SELECT ts_lexize('unaccent', 'foobar');
 ts_lexize 
-----------
 
(1 row)

SELECT ts_lexize('unaccent', 'ёлка');
 ts_lexize 
-----------
 {елка}
(1 row)

SELECT ts_lexize('unaccent', 'ЁЖИК');
 ts_lexize 
-----------
 {ЕЖИК}
(1 row)

SELECT ts_lexize('unaccent', '˃˖˗˜');
 ts_lexize 
-----------
 {>+-~}
(1 row)

SELECT ts_lexize('unaccent', 'À');
 ts_lexize 
-----------
 {A}
(1 row)

SELECT ts_lexize('unaccent', '℃℉');
 ts_lexize 
-----------
 {°C°F}
(1 row)

SELECT ts_lexize('unaccent', '℗');
 ts_lexize 
-----------
 {(P)}
(1 row)

SELECT ts_lexize('unaccent', '1½');
 ts_lexize 
-----------
 {"1 1/2"}
(1 row)

SELECT ts_lexize('unaccent', '〝');
 ts_lexize 
-----------
 {"\""}
(1 row)

-- Controversial case.  Black-Letter Capital H (U+210C) is translated by
-- Latin-ASCII.xml as 'x', but it should be 'H'.
SELECT unaccent('ℌ');
 unaccent 
----------
 x
(1 row)
-												Fix regression tests of unaccent to work without UTF8 support

The tests of unaccent rely on UTF8 characters, and unlike any other test
suite in the tree (fuzzystrmatch, citext, hstore, etc.), they would fail
if run on a database that does not support UTF8 encoding.

This commit fixes the tests of unaccent so as these are skipped when run
on a database without UTF8 support, using the same method as the other
test suits based on \if, getdatabaseencoding() and an alternate output
file.

This has been broken for a long time, but nobody has complained about
that either, so no backpatch is done.  This can be reproduced with
something like REGRESS_OPTS="--no-locale --encoding=sql_ascii", for
instance.  To defend against that, this module's Makefile and
meson.build enforced a UTF8 encoding without locales, but it did not
offer protection for options given by REGRESS_OPTS.  This switch makes
this regression test suite more consistent with all the others, as
well.

Reviewed-by: Peter Eisentraut
Discussion: https://postgr.es/m/ZIq1HUnIV2ksW85x@paquier.xyz

											
										
										
											2023-07-04 01:05:00 +02:00
+								/*
 								 * This test must be run in a database with UTF-8 encoding,
 								 * because other encodings don't support all the characters used.
 								 */
 								SELECT getdatabaseencoding() <> 'UTF8'
 								       AS skip_test \gset
 								\if :skip_test
 								\quit
 								\endif
-												Convert contrib modules to use the extension facility.

This isn't fully tested as yet, in particular I'm not sure that the
"foo--unpackaged--1.0.sql" scripts are OK.  But it's time to get some
buildfarm cycles on it.

sepgsql is not converted to an extension, mainly because it seems to
require a very nonstandard installation process.

Dimitri Fontaine and Tom Lane

											
										
										
											2011-02-14 02:06:41 +01:00
+								CREATE EXTENSION unaccent;
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SET client_encoding TO 'UTF8';
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								SELECT unaccent('foobar');
 								 unaccent
 								----------
 								 foobar
 								(1 row)
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SELECT unaccent('ёлка');
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								 unaccent
 								----------
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								 елка
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								(1 row)
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SELECT unaccent('ЁЖИК');
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								 unaccent
 								----------
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								 ЕЖИК
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								(1 row)
-												Update unaccent rules with release 34 of CLDR for Latin-ASCII.xml

This has required an update of the python script generating the rules,
as its format has changed in release 29.  This release has also added
new punctuation and symbols, and a new set of rules has been generated
to include them.  The way to find newest versions of Latin-ASCII gets
also more clearly documented.

Author: Hugh Ranalli, Michael Paquier
Discussion: https://postgr.es/m/15548-cef1b3f8de190d4f@postgresql.org

											
										
										
											2019-01-10 06:10:21 +01:00
+								SELECT unaccent('˃˖˗˜');
 								 unaccent
 								----------
 								 >+-~
 								(1 row)
-												Add combining characters to unaccent.rules.

Strip certain classes of combining characters, so that accents encoded
this way are removed.

Author: Hugh Ranalli
Discussion: https://postgr.es/m/15548-cef1b3f8de190d4f%40postgresql.org

											
										
										
											2019-02-01 15:23:01 +01:00
+								SELECT unaccent('À');  -- Remove combining diacritical 0x0300
 								 unaccent
 								----------
 								 A
 								(1 row)
-												Simplify a bit the special rules generating unaccent.rules

As noted by Thomas Munro, CLDR 36 has added SOUND RECORDING COPYRIGHT
(U+2117), and we use CLDR 41, so this can be removed from the set of
special cases.

The set of regression tests is expanded for degree signs, which are two
of the special cases, and a fancy case with U+210C in Latin-ASCII.xml
that we have discovered about when diving into what could be done for
Cyrillic characters (this last part is material for a future patch, not
tackled yet).

While on it, some of the assertions of generate_unaccent_rules.py are
expanded to report the codepoint on which a failure is found, something
useful for debugging.

Extracted from a larger patch by the same author.

Author: Przemysław Sztoch
Discussion: https://postgr.es/m/8478da0d-3b61-d24f-80b4-ce2f5e971c60@sztoch.pl

											
										
										
											2022-07-05 09:17:51 +02:00
+								SELECT unaccent('℃℉'); -- degree signs
 								 unaccent
 								----------
 								 °C°F
 								(1 row)
 								SELECT unaccent('℗'); -- sound recording copyright
 								 unaccent
 								----------
 								 (P)
 								(1 row)
-												unaccent: Add support for quoted translated characters

As reported in bug #18057, the extension unaccent removes in its rule
file whitespace characters that are intentionally specified when
building unaccent.rules from UnicodeData.txt, causing an incorrect
translation for some characters like numeric symbols.  This is caused by
the fact that all whitespaces before and after the origin and target
characters are all discarded (this limitation is documented).

This commit makes possible the use of quotes around target characters,
so as whitespaces can be considered part of target characters.  Some
target characters use a double quote, these require an extra double
quote.

The documentation is updated to show how to use quoted areas,
generate_unaccent_rules.py is updated to generate unaccent.rules and a
couple of tests are added for numeric symbols.  While working on this
patch, I have implemented a fake rule file to test the parsing logic
implemented, which is not included here as it would just consume extra
cycles in the tests, and it requires the manipulation of an installation
tree to be able to work correctly.

As this requires a change of format in unaccent.rules, this cannot be
backpatched, unfortunately.  The idea to use double quotes as escaped
characters comes from Tom Lane.

Reported-by: Martin Schlossarek
Author: Michael Paquier
Discussion: https://postgr.es/m/18057-62712cad01bd202c@postgresql.org

											
										
										
											2023-09-20 05:29:36 +02:00
+								SELECT unaccent('1½'); -- math expression with whitespace
 								 unaccent
 								----------
 1/2
 								(1 row)
 								SELECT unaccent('〝'); -- quote
 								 unaccent
 								----------
 								 "
 								(1 row)
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								SELECT unaccent('unaccent', 'foobar');
 								 unaccent
 								----------
 								 foobar
 								(1 row)
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SELECT unaccent('unaccent', 'ёлка');
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								 unaccent
 								----------
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								 елка
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								(1 row)
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SELECT unaccent('unaccent', 'ЁЖИК');
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								 unaccent
 								----------
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								 ЕЖИК
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								(1 row)
-												Update unaccent rules with release 34 of CLDR for Latin-ASCII.xml

This has required an update of the python script generating the rules,
as its format has changed in release 29.  This release has also added
new punctuation and symbols, and a new set of rules has been generated
to include them.  The way to find newest versions of Latin-ASCII gets
also more clearly documented.

Author: Hugh Ranalli, Michael Paquier
Discussion: https://postgr.es/m/15548-cef1b3f8de190d4f@postgresql.org

											
										
										
											2019-01-10 06:10:21 +01:00
+								SELECT unaccent('unaccent', '˃˖˗˜');
 								 unaccent
 								----------
 								 >+-~
 								(1 row)
-												Add combining characters to unaccent.rules.

Strip certain classes of combining characters, so that accents encoded
this way are removed.

Author: Hugh Ranalli
Discussion: https://postgr.es/m/15548-cef1b3f8de190d4f%40postgresql.org

											
										
										
											2019-02-01 15:23:01 +01:00
+								SELECT unaccent('unaccent', 'À');
 								 unaccent
 								----------
 								 A
 								(1 row)
-												Simplify a bit the special rules generating unaccent.rules

As noted by Thomas Munro, CLDR 36 has added SOUND RECORDING COPYRIGHT
(U+2117), and we use CLDR 41, so this can be removed from the set of
special cases.

The set of regression tests is expanded for degree signs, which are two
of the special cases, and a fancy case with U+210C in Latin-ASCII.xml
that we have discovered about when diving into what could be done for
Cyrillic characters (this last part is material for a future patch, not
tackled yet).

While on it, some of the assertions of generate_unaccent_rules.py are
expanded to report the codepoint on which a failure is found, something
useful for debugging.

Extracted from a larger patch by the same author.

Author: Przemysław Sztoch
Discussion: https://postgr.es/m/8478da0d-3b61-d24f-80b4-ce2f5e971c60@sztoch.pl

											
										
										
											2022-07-05 09:17:51 +02:00
+								SELECT unaccent('unaccent', '℃℉');
 								 unaccent
 								----------
 								 °C°F
 								(1 row)
 								SELECT unaccent('unaccent', '℗');
 								 unaccent
 								----------
 								 (P)
 								(1 row)
-												unaccent: Add support for quoted translated characters

As reported in bug #18057, the extension unaccent removes in its rule
file whitespace characters that are intentionally specified when
building unaccent.rules from UnicodeData.txt, causing an incorrect
translation for some characters like numeric symbols.  This is caused by
the fact that all whitespaces before and after the origin and target
characters are all discarded (this limitation is documented).

This commit makes possible the use of quotes around target characters,
so as whitespaces can be considered part of target characters.  Some
target characters use a double quote, these require an extra double
quote.

The documentation is updated to show how to use quoted areas,
generate_unaccent_rules.py is updated to generate unaccent.rules and a
couple of tests are added for numeric symbols.  While working on this
patch, I have implemented a fake rule file to test the parsing logic
implemented, which is not included here as it would just consume extra
cycles in the tests, and it requires the manipulation of an installation
tree to be able to work correctly.

As this requires a change of format in unaccent.rules, this cannot be
backpatched, unfortunately.  The idea to use double quotes as escaped
characters comes from Tom Lane.

Reported-by: Martin Schlossarek
Author: Michael Paquier
Discussion: https://postgr.es/m/18057-62712cad01bd202c@postgresql.org

											
										
										
											2023-09-20 05:29:36 +02:00
+								SELECT unaccent('unaccent', '1½');
 								 unaccent
 								----------
 1/2
 								(1 row)
 								SELECT unaccent('unaccent', '〝');
 								 unaccent
 								----------
 								 "
 								(1 row)
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								SELECT ts_lexize('unaccent', 'foobar');
 								 ts_lexize
 								-----------
 								(1 row)
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SELECT ts_lexize('unaccent', 'ёлка');
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								 ts_lexize
 								-----------
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								 {елка}
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								(1 row)
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								SELECT ts_lexize('unaccent', 'ЁЖИК');
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								 ts_lexize
 								-----------
-												Convert unaccent tests to UTF-8

This makes it easier to add new tests that are specific to Unicode
features.  The files were previously in KOI8-R.

Discussion: https://www.postgresql.org/message-id/8506.1545111362@sss.pgh.pa.us

											
										
										
											2019-01-02 18:36:05 +01:00
+								 {ЕЖИК}
-												Unaccent dictionary.

											
										
										
											2009-08-18 12:34:39 +02:00
+								(1 row)
-												Update unaccent rules with release 34 of CLDR for Latin-ASCII.xml

This has required an update of the python script generating the rules,
as its format has changed in release 29.  This release has also added
new punctuation and symbols, and a new set of rules has been generated
to include them.  The way to find newest versions of Latin-ASCII gets
also more clearly documented.

Author: Hugh Ranalli, Michael Paquier
Discussion: https://postgr.es/m/15548-cef1b3f8de190d4f@postgresql.org

											
										
										
											2019-01-10 06:10:21 +01:00
+								SELECT ts_lexize('unaccent', '˃˖˗˜');
 								 ts_lexize
 								-----------
 								 {>+-~}
 								(1 row)
-												Add combining characters to unaccent.rules.

Strip certain classes of combining characters, so that accents encoded
this way are removed.

Author: Hugh Ranalli
Discussion: https://postgr.es/m/15548-cef1b3f8de190d4f%40postgresql.org

											
										
										
											2019-02-01 15:23:01 +01:00
+								SELECT ts_lexize('unaccent', 'À');
 								 ts_lexize
 								-----------
 								 {A}
 								(1 row)
-												Simplify a bit the special rules generating unaccent.rules

As noted by Thomas Munro, CLDR 36 has added SOUND RECORDING COPYRIGHT
(U+2117), and we use CLDR 41, so this can be removed from the set of
special cases.

The set of regression tests is expanded for degree signs, which are two
of the special cases, and a fancy case with U+210C in Latin-ASCII.xml
that we have discovered about when diving into what could be done for
Cyrillic characters (this last part is material for a future patch, not
tackled yet).

While on it, some of the assertions of generate_unaccent_rules.py are
expanded to report the codepoint on which a failure is found, something
useful for debugging.

Extracted from a larger patch by the same author.

Author: Przemysław Sztoch
Discussion: https://postgr.es/m/8478da0d-3b61-d24f-80b4-ce2f5e971c60@sztoch.pl

											
										
										
											2022-07-05 09:17:51 +02:00
+								SELECT ts_lexize('unaccent', '℃℉');
 								 ts_lexize
 								-----------
 								 {°C°F}
 								(1 row)
 								SELECT ts_lexize('unaccent', '℗');
 								 ts_lexize
 								-----------
 								 {(P)}
 								(1 row)
-												unaccent: Add support for quoted translated characters

As reported in bug #18057, the extension unaccent removes in its rule
file whitespace characters that are intentionally specified when
building unaccent.rules from UnicodeData.txt, causing an incorrect
translation for some characters like numeric symbols.  This is caused by
the fact that all whitespaces before and after the origin and target
characters are all discarded (this limitation is documented).

This commit makes possible the use of quotes around target characters,
so as whitespaces can be considered part of target characters.  Some
target characters use a double quote, these require an extra double
quote.

The documentation is updated to show how to use quoted areas,
generate_unaccent_rules.py is updated to generate unaccent.rules and a
couple of tests are added for numeric symbols.  While working on this
patch, I have implemented a fake rule file to test the parsing logic
implemented, which is not included here as it would just consume extra
cycles in the tests, and it requires the manipulation of an installation
tree to be able to work correctly.

As this requires a change of format in unaccent.rules, this cannot be
backpatched, unfortunately.  The idea to use double quotes as escaped
characters comes from Tom Lane.

Reported-by: Martin Schlossarek
Author: Michael Paquier
Discussion: https://postgr.es/m/18057-62712cad01bd202c@postgresql.org

											
										
										
											2023-09-20 05:29:36 +02:00
+								SELECT ts_lexize('unaccent', '1½');
 								 ts_lexize
 								-----------
 								 {"1 1/2"}
 								(1 row)
 								SELECT ts_lexize('unaccent', '〝');
 								 ts_lexize
 								-----------
 								 {"\""}
 								(1 row)
-												Simplify a bit the special rules generating unaccent.rules

As noted by Thomas Munro, CLDR 36 has added SOUND RECORDING COPYRIGHT
(U+2117), and we use CLDR 41, so this can be removed from the set of
special cases.

The set of regression tests is expanded for degree signs, which are two
of the special cases, and a fancy case with U+210C in Latin-ASCII.xml
that we have discovered about when diving into what could be done for
Cyrillic characters (this last part is material for a future patch, not
tackled yet).

While on it, some of the assertions of generate_unaccent_rules.py are
expanded to report the codepoint on which a failure is found, something
useful for debugging.

Extracted from a larger patch by the same author.

Author: Przemysław Sztoch
Discussion: https://postgr.es/m/8478da0d-3b61-d24f-80b4-ce2f5e971c60@sztoch.pl

											
										
										
											2022-07-05 09:17:51 +02:00
+								-- Controversial case.  Black-Letter Capital H (U+210C) is translated by
 								-- Latin-ASCII.xml as 'x', but it should be 'H'.
 								SELECT unaccent('ℌ');
 								 unaccent
 								----------
 								 x
 								(1 row)