{"Entry":{"collection":"fts","key":"parser-default","name":"default","aliases":[],"metadata":{"aliases":[],"category":"Parsers","content_hash":"88c149eb359def8b50e7f6514182a1449d010e5adfab1fa6f8f918e45927b3f5","imported_at":"2026-09-30T00:40:47.651207+08:00","name":"default","name_zh":"","slug":"parser-default","summary":"default word parser"}},"Definition":{"Collection":"fts","Key":"parser-default","SourceDatabase":"center","Version":"18","SourceTable":"text_search_component","SourceKey":"parser-default","SourceRevision":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f","Facts":{"aliases":[],"attributes":{"prsend":"prsd_end","prsheadline":"prsd_headline","prslextype":"prsd_lextype","prsname":"default","prsnamespace":"pg_catalog","prsstart":"prsd_start","prstoken":"prsd_nexttoken"},"comparison_data":{"attributes":{"prsend":"prsd_end","prsheadline":"prsd_headline","prslextype":"prsd_lextype","prsname":"default","prsnamespace":"pg_catalog","prsstart":"prsd_start","prstoken":"prsd_nexttoken"},"tokens":[{"alias":"asciiword","description":"Word, all ASCII","id":"1"},{"alias":"word","description":"Word, all letters","id":"2"},{"alias":"numword","description":"Word, letters and digits","id":"3"},{"alias":"email","description":"Email address","id":"4"},{"alias":"url","description":"URL","id":"5"},{"alias":"host","description":"Host","id":"6"},{"alias":"sfloat","description":"Scientific notation","id":"7"},{"alias":"version","description":"Version number","id":"8"},{"alias":"hword_numpart","description":"Hyphenated word part, letters and digits","id":"9"},{"alias":"hword_part","description":"Hyphenated word part, all letters","id":"10"},{"alias":"hword_asciipart","description":"Hyphenated word part, all ASCII","id":"11"},{"alias":"blank","description":"Space symbols","id":"12"},{"alias":"tag","description":"XML tag","id":"13"},{"alias":"protocol","description":"Protocol head","id":"14"},{"alias":"numhword","description":"Hyphenated word, letters and digits","id":"15"},{"alias":"asciihword","description":"Hyphenated word, all ASCII","id":"16"},{"alias":"hword","description":"Hyphenated word, all letters","id":"17"},{"alias":"url_path","description":"URL path","id":"18"},{"alias":"file","description":"File or path name","id":"19"},{"alias":"float","description":"Decimal notation","id":"20"},{"alias":"int","description":"Signed integer","id":"21"},{"alias":"uint","description":"Unsigned integer","id":"22"},{"alias":"entity","description":"XML entity","id":"23"}]},"comparison_hash":"cc0b46768e43de6aa2e4460f1526f915791374259dbab054b02d6cc54582f89b","description":["default word parser"],"facts":[{"label":"Prsnamespace","value":"pg_catalog"},{"label":"Prsname","value":"default"},{"label":"Prsstart","value":"prsd_start"},{"label":"Prstoken","value":"prsd_nexttoken"},{"label":"Prsend","value":"prsd_end"},{"label":"Prsheadline","value":"prsd_headline"},{"label":"Prslextype","value":"prsd_lextype"}],"manual_html":"\u003cdiv class=\"sect1\" id=\"TEXTSEARCH-PARSERS\"\u003e\n\u003cdiv class=\"titlepage\"\u003e\n\u003cdiv\u003e\n\u003cdiv\u003e\n\u003ch2 class=\"title\"\u003e12.5. Parsers \u003c/h2\u003e\n\u003c/div\u003e\n\u003c/div\u003e\n\u003c/div\u003e\n\u003cp\u003eText search parsers are responsible for splitting raw document text into \u003cem class=\"firstterm\"\u003etokens\u003c/em\u003e and identifying each token's type, where the set of possible types is defined by the parser itself. Note that a parser does not modify the text at all — it simply identifies plausible word boundaries. Because of this limited scope, there is less need for application-specific custom parsers than there is for custom dictionaries. At present \u003cspan class=\"productname\"\u003ePostgreSQL\u003c/span\u003e provides just one built-in parser, which has been found to be useful for a wide range of applications.\u003c/p\u003e\n\u003cp\u003eThe built-in parser is named \u003ccode class=\"literal\"\u003epg_catalog.default\u003c/code\u003e. It recognizes 23 token types, shown in \u003ca class=\"xref\" href=\"/docs/18/textsearch-parsers.html#TEXTSEARCH-DEFAULT-PARSER\" title=\"Table 12.1. Default Parser's Token Types\"\u003eTable 12.1\u003c/a\u003e.\u003c/p\u003e\n\u003cdiv class=\"table\" id=\"TEXTSEARCH-DEFAULT-PARSER\"\u003e\n\u003cp class=\"title\"\u003e\u003cstrong\u003eTable 12.1. Default Parser's Token Types\u003c/strong\u003e\u003c/p\u003e\n\u003cdiv class=\"table-contents\"\u003e\n\u003ctable class=\"table\"\u003e\n\n\n\n\n\n\u003cthead\u003e\n\u003ctr\u003e\n\u003cth\u003eAlias\u003c/th\u003e\n\u003cth\u003eDescription\u003c/th\u003e\n\u003cth\u003eExample\u003c/th\u003e\n\u003c/tr\u003e\n\u003c/thead\u003e\n\u003ctbody\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003easciiword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, all ASCII letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eelephant\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003emañana\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003enumword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ebeta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003easciihword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, all ASCII\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eup-to-date\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003elógico-matemática\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003enumhword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword_asciipart\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, all ASCII\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003epostgresql\u003c/code\u003e in the context \u003ccode class=\"literal\"\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword_part\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003elógico\u003c/code\u003e or \u003ccode class=\"literal\"\u003ematemática\u003c/code\u003e in the context \u003ccode class=\"literal\"\u003elógico-matemática\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword_numpart\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ebeta1\u003c/code\u003e in the context \u003ccode class=\"literal\"\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eemail\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eEmail address\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003efoo@example.com\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eprotocol\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eProtocol head\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehttp://\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eurl\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eURL\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eexample.com/stuff/index.html\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehost\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHost\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eexample.com\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eurl_path\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eURL path\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e/stuff/index.html\u003c/code\u003e, in the context of a URL\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003efile\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eFile or path name\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e/usr/local/foo.txt\u003c/code\u003e, if not within a URL\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003esfloat\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eScientific notation\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e-1.234e56\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003efloat\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eDecimal notation\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e-1.234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eint\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eSigned integer\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e-1234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003euint\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eUnsigned integer\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e1234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eversion\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eVersion number\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e8.3.0\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003etag\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eXML tag\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e\u0026lt;a href=\"dictionaries.html\"\u0026gt;\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eentity\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eXML entity\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e\u0026amp;amp;\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eblank\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eSpace symbols\u003c/td\u003e\n\u003ctd\u003e(any whitespace or punctuation not otherwise recognized)\u003c/td\u003e\n\u003c/tr\u003e\n\u003c/tbody\u003e\n\u003c/table\u003e\n\u003c/div\u003e\n\u003c/div\u003e\u003cbr class=\"table-break\"\u003e\n\u003cdiv class=\"note\"\u003e\n\u003ch3 class=\"title\"\u003eNote\u003c/h3\u003e\n\u003cp\u003eThe parser's notion of a \u003cspan class=\"quote\"\u003e“\u003cspan class=\"quote\"\u003eletter\u003c/span\u003e”\u003c/span\u003e is determined by the database's locale setting, specifically \u003ccode class=\"varname\"\u003elc_ctype\u003c/code\u003e. Words containing only the basic ASCII letters are reported as a separate token type, since it is sometimes useful to distinguish them. In most European languages, token types \u003ccode class=\"literal\"\u003eword\u003c/code\u003e and \u003ccode class=\"literal\"\u003easciiword\u003c/code\u003e should be treated alike.\u003c/p\u003e\n\u003cp\u003e\u003ccode class=\"literal\"\u003eemail\u003c/code\u003e does not support all valid email characters as defined by \u003ca class=\"ulink\" href=\"https://datatracker.ietf.org/doc/html/rfc5322\"\u003eRFC 5322\u003c/a\u003e. Specifically, the only non-alphanumeric characters supported for email user names are period, dash, and underscore.\u003c/p\u003e\n\u003cp\u003e\u003ccode class=\"literal\"\u003etag\u003c/code\u003e does not support all valid tag names as defined by \u003ca class=\"ulink\" href=\"https://www.w3.org/TR/xml/\"\u003eW3C Recommendation, XML\u003c/a\u003e. Specifically, the only tag names supported are those starting with an ASCII letter, underscore, or colon, and containing only letters, digits, hyphens, underscores, periods, and colons. \u003ccode class=\"literal\"\u003etag\u003c/code\u003e also includes XML comments starting with \u003ccode class=\"literal\"\u003e\u0026lt;!--\u003c/code\u003e and ending with \u003ccode class=\"literal\"\u003e--\u0026gt;\u003c/code\u003e, and XML declarations (but note that this includes anything starting with \u003ccode class=\"literal\"\u003e\u0026lt;?x\u003c/code\u003e and ending with \u003ccode class=\"literal\"\u003e\u0026gt;\u003c/code\u003e).\u003c/p\u003e\n\u003c/div\u003e\n\u003cp\u003eIt is possible for the parser to produce overlapping tokens from the same piece of text. As an example, a hyphenated word will be reported both as the entire word and as each component:\u003c/p\u003e\n\u003cpre class=\"screen\"\u003eSELECT alias, description, token FROM ts_debug('foo-bar-beta1');\n      alias      |               description                |     token\n-----------------+------------------------------------------+---------------\n numhword        | Hyphenated word, letters and digits      | foo-bar-beta1\n hword_asciipart | Hyphenated word part, all ASCII          | foo\n blank           | Space symbols                            | -\n hword_asciipart | Hyphenated word part, all ASCII          | bar\n blank           | Space symbols                            | -\n hword_numpart   | Hyphenated word part, letters and digits | beta1\n\u003c/pre\u003e\n\u003cp\u003eThis behavior is desirable since it allows searches to work for both the whole compound word and for components. Here is another instructive example:\u003c/p\u003e\n\u003cpre class=\"screen\"\u003eSELECT alias, description, token FROM ts_debug('http://example.com/stuff/index.html');\n  alias   |  description  |            token\n----------+---------------+------------------------------\n protocol | Protocol head | http://\n url      | URL           | example.com/stuff/index.html\n host     | Host          | example.com\n url_path | URL path      | /stuff/index.html\n\u003c/pre\u003e\n\u003c/div\u003e","manual_path":"/docs/18/textsearch-parsers.html","related":[],"release":{"catalog_fingerprint":"65c93d6048ef30e61023a84f9680fa6a92b1c383b7eb226741170077eb078502","channel":"stable","label":"18.6","major":"18","ref":"https://ftp.postgresql.org/pub/source/v18.6/postgresql-18.6.tar.bz2","revision":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f","source_sha256":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f"},"sections":[],"signature":"","sources":[{"label":"Matching PostgreSQL source archive","sha256":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f","url":"https://ftp.postgresql.org/pub/source/v18.6/postgresql-18.6.tar.bz2"},{"label":"PostgreSQL 18 English manual","path":"textsearch-parsers.html","sha256":"97e57975afd39da6b33f440d01e36e79a6a57bc3e178c20dbb2f0b19e481e611","url":"/docs/18/textsearch-parsers.html"}],"tables":[{"columns":[{"key":"id","label":"Token ID"},{"key":"alias","label":"Alias"},{"key":"description","label":"Description"}],"key":"tokens","rows":[{"alias":"asciiword","description":"Word, all ASCII","id":"1"},{"alias":"word","description":"Word, all letters","id":"2"},{"alias":"numword","description":"Word, letters and digits","id":"3"},{"alias":"email","description":"Email address","id":"4"},{"alias":"url","description":"URL","id":"5"},{"alias":"host","description":"Host","id":"6"},{"alias":"sfloat","description":"Scientific notation","id":"7"},{"alias":"version","description":"Version number","id":"8"},{"alias":"hword_numpart","description":"Hyphenated word part, letters and digits","id":"9"},{"alias":"hword_part","description":"Hyphenated word part, all letters","id":"10"},{"alias":"hword_asciipart","description":"Hyphenated word part, all ASCII","id":"11"},{"alias":"blank","description":"Space symbols","id":"12"},{"alias":"tag","description":"XML tag","id":"13"},{"alias":"protocol","description":"Protocol head","id":"14"},{"alias":"numhword","description":"Hyphenated word, letters and digits","id":"15"},{"alias":"asciihword","description":"Hyphenated word, all ASCII","id":"16"},{"alias":"hword","description":"Hyphenated word, all letters","id":"17"},{"alias":"url_path","description":"URL path","id":"18"},{"alias":"file","description":"File or path name","id":"19"},{"alias":"float","description":"Decimal notation","id":"20"},{"alias":"int","description":"Signed integer","id":"21"},{"alias":"uint","description":"Unsigned integer","id":"22"},{"alias":"entity","description":"XML entity","id":"23"}],"title":"Parser token types"}]},"ManualEvidence":{"manual_path":"/docs/18/textsearch-parsers.html","release":{"catalog_fingerprint":"65c93d6048ef30e61023a84f9680fa6a92b1c383b7eb226741170077eb078502","channel":"stable","label":"18.6","major":"18","ref":"https://ftp.postgresql.org/pub/source/v18.6/postgresql-18.6.tar.bz2","revision":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f","source_sha256":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f"},"sources":[{"label":"Matching PostgreSQL source archive","sha256":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f","url":"https://ftp.postgresql.org/pub/source/v18.6/postgresql-18.6.tar.bz2"},{"label":"PostgreSQL 18 English manual","path":"textsearch-parsers.html","sha256":"97e57975afd39da6b33f440d01e36e79a6a57bc3e178c20dbb2f0b19e481e611","url":"/docs/18/textsearch-parsers.html"}]},"MeasuredEvidence":{}},"Text":{"Collection":"fts","Key":"parser-default","SourceDatabase":"center","Version":"18","Locale":"en","Title":"default","Summary":"default word parser","BodyHTML":"\u003cdiv id=\"TEXTSEARCH-PARSERS\"\u003e\n\u003cdiv\u003e\n\u003cdiv\u003e\n\u003cdiv\u003e\n\u003ch2\u003e12.5. Parsers \u003c/h2\u003e\n\u003c/div\u003e\n\u003c/div\u003e\n\u003c/div\u003e\n\u003cp\u003eText search parsers are responsible for splitting raw document text into \u003cem\u003etokens\u003c/em\u003e and identifying each token\u0026#39;s type, where the set of possible types is defined by the parser itself. Note that a parser does not modify the text at all — it simply identifies plausible word boundaries. Because of this limited scope, there is less need for application-specific custom parsers than there is for custom dictionaries. At present \u003cspan\u003ePostgreSQL\u003c/span\u003e provides just one built-in parser, which has been found to be useful for a wide range of applications.\u003c/p\u003e\n\u003cp\u003eThe built-in parser is named \u003ccode\u003epg_catalog.default\u003c/code\u003e. It recognizes 23 token types, shown in \u003ca href=\"/docs/18/textsearch-parsers.html#TEXTSEARCH-DEFAULT-PARSER\" rel=\"nofollow\"\u003eTable 12.1\u003c/a\u003e.\u003c/p\u003e\n\u003cdiv id=\"TEXTSEARCH-DEFAULT-PARSER\"\u003e\n\u003cp\u003e\u003cstrong\u003eTable 12.1. Default Parser\u0026#39;s Token Types\u003c/strong\u003e\u003c/p\u003e\n\u003cdiv\u003e\n\u003ctable\u003e\n\n\n\n\n\n\u003cthead\u003e\n\u003ctr\u003e\n\u003cth\u003eAlias\u003c/th\u003e\n\u003cth\u003eDescription\u003c/th\u003e\n\u003cth\u003eExample\u003c/th\u003e\n\u003c/tr\u003e\n\u003c/thead\u003e\n\u003ctbody\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003easciiword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, all ASCII letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003eelephant\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003emañana\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003enumword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003ebeta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003easciihword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, all ASCII\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003eup-to-date\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003ehword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003elógico-matemática\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003enumhword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003ehword_asciipart\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, all ASCII\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003epostgresql\u003c/code\u003e in the context \u003ccode\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003ehword_part\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003elógico\u003c/code\u003e or \u003ccode\u003ematemática\u003c/code\u003e in the context \u003ccode\u003elógico-matemática\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003ehword_numpart\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003ebeta1\u003c/code\u003e in the context \u003ccode\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eemail\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eEmail address\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003efoo@example.com\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eprotocol\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eProtocol head\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003ehttp://\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eurl\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eURL\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003eexample.com/stuff/index.html\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003ehost\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHost\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003eexample.com\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eurl_path\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eURL path\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e/stuff/index.html\u003c/code\u003e, in the context of a URL\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003efile\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eFile or path name\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e/usr/local/foo.txt\u003c/code\u003e, if not within a URL\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003esfloat\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eScientific notation\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e-1.234e56\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003efloat\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eDecimal notation\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e-1.234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eint\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eSigned integer\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e-1234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003euint\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eUnsigned integer\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e1234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eversion\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eVersion number\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e8.3.0\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003etag\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eXML tag\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e\u0026lt;a href=\u0026#34;dictionaries.html\u0026#34;\u0026gt;\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eentity\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eXML entity\u003c/td\u003e\n\u003ctd\u003e\u003ccode\u003e\u0026amp;amp;\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode\u003eblank\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eSpace symbols\u003c/td\u003e\n\u003ctd\u003e(any whitespace or punctuation not otherwise recognized)\u003c/td\u003e\n\u003c/tr\u003e\n\u003c/tbody\u003e\n\u003c/table\u003e\n\u003c/div\u003e\n\u003c/div\u003e\u003cbr\u003e\n\u003cdiv\u003e\n\u003ch3\u003eNote\u003c/h3\u003e\n\u003cp\u003eThe parser\u0026#39;s notion of a \u003cspan\u003e“\u003cspan\u003eletter\u003c/span\u003e”\u003c/span\u003e is determined by the database\u0026#39;s locale setting, specifically \u003ccode\u003elc_ctype\u003c/code\u003e. Words containing only the basic ASCII letters are reported as a separate token type, since it is sometimes useful to distinguish them. In most European languages, token types \u003ccode\u003eword\u003c/code\u003e and \u003ccode\u003easciiword\u003c/code\u003e should be treated alike.\u003c/p\u003e\n\u003cp\u003e\u003ccode\u003eemail\u003c/code\u003e does not support all valid email characters as defined by \u003ca href=\"https://datatracker.ietf.org/doc/html/rfc5322\" rel=\"nofollow\"\u003eRFC 5322\u003c/a\u003e. Specifically, the only non-alphanumeric characters supported for email user names are period, dash, and underscore.\u003c/p\u003e\n\u003cp\u003e\u003ccode\u003etag\u003c/code\u003e does not support all valid tag names as defined by \u003ca href=\"https://www.w3.org/TR/xml/\" rel=\"nofollow\"\u003eW3C Recommendation, XML\u003c/a\u003e. Specifically, the only tag names supported are those starting with an ASCII letter, underscore, or colon, and containing only letters, digits, hyphens, underscores, periods, and colons. \u003ccode\u003etag\u003c/code\u003e also includes XML comments starting with \u003ccode\u003e\u0026lt;!--\u003c/code\u003e and ending with \u003ccode\u003e--\u0026gt;\u003c/code\u003e, and XML declarations (but note that this includes anything starting with \u003ccode\u003e\u0026lt;?x\u003c/code\u003e and ending with \u003ccode\u003e\u0026gt;\u003c/code\u003e).\u003c/p\u003e\n\u003c/div\u003e\n\u003cp\u003eIt is possible for the parser to produce overlapping tokens from the same piece of text. As an example, a hyphenated word will be reported both as the entire word and as each component:\u003c/p\u003e\n\u003cpre\u003eSELECT alias, description, token FROM ts_debug(\u0026#39;foo-bar-beta1\u0026#39;);\n      alias      |               description                |     token\n-----------------+------------------------------------------+---------------\n numhword        | Hyphenated word, letters and digits      | foo-bar-beta1\n hword_asciipart | Hyphenated word part, all ASCII          | foo\n blank           | Space symbols                            | -\n hword_asciipart | Hyphenated word part, all ASCII          | bar\n blank           | Space symbols                            | -\n hword_numpart   | Hyphenated word part, letters and digits | beta1\n\u003c/pre\u003e\n\u003cp\u003eThis behavior is desirable since it allows searches to work for both the whole compound word and for components. Here is another instructive example:\u003c/p\u003e\n\u003cpre\u003eSELECT alias, description, token FROM ts_debug(\u0026#39;http://example.com/stuff/index.html\u0026#39;);\n  alias   |  description  |            token\n----------+---------------+------------------------------\n protocol | Protocol head | http://\n url      | URL           | example.com/stuff/index.html\n host     | Host          | example.com\n url_path | URL path      | /stuff/index.html\n\u003c/pre\u003e\n\u003c/div\u003e","SourceRevision":"555610c24d53e4316da5b7d3fc25c279d96856d5e0e23ee308c328c5fa881d9f","ContentHash":"e1708fdf43d9fd25fe640019e530472d928165e970dfd484b23e070f2845240b","Payload":{"description":["default word parser"],"manual_html":"\u003cdiv class=\"sect1\" id=\"TEXTSEARCH-PARSERS\"\u003e\n\u003cdiv class=\"titlepage\"\u003e\n\u003cdiv\u003e\n\u003cdiv\u003e\n\u003ch2 class=\"title\"\u003e12.5. Parsers \u003c/h2\u003e\n\u003c/div\u003e\n\u003c/div\u003e\n\u003c/div\u003e\n\u003cp\u003eText search parsers are responsible for splitting raw document text into \u003cem class=\"firstterm\"\u003etokens\u003c/em\u003e and identifying each token's type, where the set of possible types is defined by the parser itself. Note that a parser does not modify the text at all — it simply identifies plausible word boundaries. Because of this limited scope, there is less need for application-specific custom parsers than there is for custom dictionaries. At present \u003cspan class=\"productname\"\u003ePostgreSQL\u003c/span\u003e provides just one built-in parser, which has been found to be useful for a wide range of applications.\u003c/p\u003e\n\u003cp\u003eThe built-in parser is named \u003ccode class=\"literal\"\u003epg_catalog.default\u003c/code\u003e. It recognizes 23 token types, shown in \u003ca class=\"xref\" href=\"/docs/18/textsearch-parsers.html#TEXTSEARCH-DEFAULT-PARSER\" title=\"Table 12.1. Default Parser's Token Types\"\u003eTable 12.1\u003c/a\u003e.\u003c/p\u003e\n\u003cdiv class=\"table\" id=\"TEXTSEARCH-DEFAULT-PARSER\"\u003e\n\u003cp class=\"title\"\u003e\u003cstrong\u003eTable 12.1. Default Parser's Token Types\u003c/strong\u003e\u003c/p\u003e\n\u003cdiv class=\"table-contents\"\u003e\n\u003ctable class=\"table\"\u003e\n\n\n\n\n\n\u003cthead\u003e\n\u003ctr\u003e\n\u003cth\u003eAlias\u003c/th\u003e\n\u003cth\u003eDescription\u003c/th\u003e\n\u003cth\u003eExample\u003c/th\u003e\n\u003c/tr\u003e\n\u003c/thead\u003e\n\u003ctbody\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003easciiword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, all ASCII letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eelephant\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003emañana\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003enumword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eWord, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ebeta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003easciihword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, all ASCII\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eup-to-date\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003elógico-matemática\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003enumhword\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword_asciipart\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, all ASCII\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003epostgresql\u003c/code\u003e in the context \u003ccode class=\"literal\"\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword_part\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, all letters\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003elógico\u003c/code\u003e or \u003ccode class=\"literal\"\u003ematemática\u003c/code\u003e in the context \u003ccode class=\"literal\"\u003elógico-matemática\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehword_numpart\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHyphenated word part, letters and digits\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ebeta1\u003c/code\u003e in the context \u003ccode class=\"literal\"\u003epostgresql-beta1\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eemail\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eEmail address\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003efoo@example.com\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eprotocol\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eProtocol head\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehttp://\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eurl\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eURL\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eexample.com/stuff/index.html\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003ehost\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eHost\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eexample.com\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eurl_path\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eURL path\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e/stuff/index.html\u003c/code\u003e, in the context of a URL\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003efile\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eFile or path name\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e/usr/local/foo.txt\u003c/code\u003e, if not within a URL\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003esfloat\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eScientific notation\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e-1.234e56\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003efloat\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eDecimal notation\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e-1.234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eint\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eSigned integer\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e-1234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003euint\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eUnsigned integer\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e1234\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eversion\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eVersion number\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e8.3.0\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003etag\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eXML tag\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e\u0026lt;a href=\"dictionaries.html\"\u0026gt;\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eentity\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eXML entity\u003c/td\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003e\u0026amp;amp;\u003c/code\u003e\u003c/td\u003e\n\u003c/tr\u003e\n\u003ctr\u003e\n\u003ctd\u003e\u003ccode class=\"literal\"\u003eblank\u003c/code\u003e\u003c/td\u003e\n\u003ctd\u003eSpace symbols\u003c/td\u003e\n\u003ctd\u003e(any whitespace or punctuation not otherwise recognized)\u003c/td\u003e\n\u003c/tr\u003e\n\u003c/tbody\u003e\n\u003c/table\u003e\n\u003c/div\u003e\n\u003c/div\u003e\u003cbr class=\"table-break\"\u003e\n\u003cdiv class=\"note\"\u003e\n\u003ch3 class=\"title\"\u003eNote\u003c/h3\u003e\n\u003cp\u003eThe parser's notion of a \u003cspan class=\"quote\"\u003e“\u003cspan class=\"quote\"\u003eletter\u003c/span\u003e”\u003c/span\u003e is determined by the database's locale setting, specifically \u003ccode class=\"varname\"\u003elc_ctype\u003c/code\u003e. Words containing only the basic ASCII letters are reported as a separate token type, since it is sometimes useful to distinguish them. In most European languages, token types \u003ccode class=\"literal\"\u003eword\u003c/code\u003e and \u003ccode class=\"literal\"\u003easciiword\u003c/code\u003e should be treated alike.\u003c/p\u003e\n\u003cp\u003e\u003ccode class=\"literal\"\u003eemail\u003c/code\u003e does not support all valid email characters as defined by \u003ca class=\"ulink\" href=\"https://datatracker.ietf.org/doc/html/rfc5322\"\u003eRFC 5322\u003c/a\u003e. Specifically, the only non-alphanumeric characters supported for email user names are period, dash, and underscore.\u003c/p\u003e\n\u003cp\u003e\u003ccode class=\"literal\"\u003etag\u003c/code\u003e does not support all valid tag names as defined by \u003ca class=\"ulink\" href=\"https://www.w3.org/TR/xml/\"\u003eW3C Recommendation, XML\u003c/a\u003e. Specifically, the only tag names supported are those starting with an ASCII letter, underscore, or colon, and containing only letters, digits, hyphens, underscores, periods, and colons. \u003ccode class=\"literal\"\u003etag\u003c/code\u003e also includes XML comments starting with \u003ccode class=\"literal\"\u003e\u0026lt;!--\u003c/code\u003e and ending with \u003ccode class=\"literal\"\u003e--\u0026gt;\u003c/code\u003e, and XML declarations (but note that this includes anything starting with \u003ccode class=\"literal\"\u003e\u0026lt;?x\u003c/code\u003e and ending with \u003ccode class=\"literal\"\u003e\u0026gt;\u003c/code\u003e).\u003c/p\u003e\n\u003c/div\u003e\n\u003cp\u003eIt is possible for the parser to produce overlapping tokens from the same piece of text. As an example, a hyphenated word will be reported both as the entire word and as each component:\u003c/p\u003e\n\u003cpre class=\"screen\"\u003eSELECT alias, description, token FROM ts_debug('foo-bar-beta1');\n      alias      |               description                |     token\n-----------------+------------------------------------------+---------------\n numhword        | Hyphenated word, letters and digits      | foo-bar-beta1\n hword_asciipart | Hyphenated word part, all ASCII          | foo\n blank           | Space symbols                            | -\n hword_asciipart | Hyphenated word part, all ASCII          | bar\n blank           | Space symbols                            | -\n hword_numpart   | Hyphenated word part, letters and digits | beta1\n\u003c/pre\u003e\n\u003cp\u003eThis behavior is desirable since it allows searches to work for both the whole compound word and for components. Here is another instructive example:\u003c/p\u003e\n\u003cpre class=\"screen\"\u003eSELECT alias, description, token FROM ts_debug('http://example.com/stuff/index.html');\n  alias   |  description  |            token\n----------+---------------+------------------------------\n protocol | Protocol head | http://\n url      | URL           | example.com/stuff/index.html\n host     | Host          | example.com\n url_path | URL path      | /stuff/index.html\n\u003c/pre\u003e\n\u003c/div\u003e","related":[],"sections":[],"tables":[{"columns":[{"key":"id","label":"Token ID"},{"key":"alias","label":"Alias"},{"key":"description","label":"Description"}],"key":"tokens","rows":[{"alias":"asciiword","description":"Word, all ASCII","id":"1"},{"alias":"word","description":"Word, all letters","id":"2"},{"alias":"numword","description":"Word, letters and digits","id":"3"},{"alias":"email","description":"Email address","id":"4"},{"alias":"url","description":"URL","id":"5"},{"alias":"host","description":"Host","id":"6"},{"alias":"sfloat","description":"Scientific notation","id":"7"},{"alias":"version","description":"Version number","id":"8"},{"alias":"hword_numpart","description":"Hyphenated word part, letters and digits","id":"9"},{"alias":"hword_part","description":"Hyphenated word part, all letters","id":"10"},{"alias":"hword_asciipart","description":"Hyphenated word part, all ASCII","id":"11"},{"alias":"blank","description":"Space symbols","id":"12"},{"alias":"tag","description":"XML tag","id":"13"},{"alias":"protocol","description":"Protocol head","id":"14"},{"alias":"numhword","description":"Hyphenated word, letters and digits","id":"15"},{"alias":"asciihword","description":"Hyphenated word, all ASCII","id":"16"},{"alias":"hword","description":"Hyphenated word, all letters","id":"17"},{"alias":"url_path","description":"URL path","id":"18"},{"alias":"file","description":"File or path name","id":"19"},{"alias":"float","description":"Decimal notation","id":"20"},{"alias":"int","description":"Signed integer","id":"21"},{"alias":"uint","description":"Unsigned integer","id":"22"},{"alias":"entity","description":"XML entity","id":"23"}],"title":"Parser token types"}]}},"RequestedLocale":"zh-Hans","Fallback":true,"Versions":["10","11","12","13","14","15","16","17","18","19","20"],"Locales":["en"],"Signatures":null,"Spellings":null,"SQLState":null,"Evidence":null}
