AbstractUnicodeString.php 25 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576
  1. <?php
  2. /*
  3. * This file is part of the Symfony package.
  4. *
  5. * (c) Fabien Potencier <fabien@symfony.com>
  6. *
  7. * For the full copyright and license information, please view the LICENSE
  8. * file that was distributed with this source code.
  9. */
  10. namespace Symfony\Component\String;
  11. use Symfony\Component\String\Exception\ExceptionInterface;
  12. use Symfony\Component\String\Exception\InvalidArgumentException;
  13. use Symfony\Component\String\Exception\RuntimeException;
  14. /**
  15. * Represents a string of abstract Unicode characters.
  16. *
  17. * Unicode defines 3 types of "characters" (bytes, code points and grapheme clusters).
  18. * This class is the abstract type to use as a type-hint when the logic you want to
  19. * implement is Unicode-aware but doesn't care about code points vs grapheme clusters.
  20. *
  21. * @author Nicolas Grekas <p@tchwork.com>
  22. *
  23. * @throws ExceptionInterface
  24. */
  25. abstract class AbstractUnicodeString extends AbstractString
  26. {
  27. public const NFC = \Normalizer::NFC;
  28. public const NFD = \Normalizer::NFD;
  29. public const NFKC = \Normalizer::NFKC;
  30. public const NFKD = \Normalizer::NFKD;
  31. // all ASCII letters sorted by typical frequency of occurrence
  32. private const ASCII = "\x20\x65\x69\x61\x73\x6E\x74\x72\x6F\x6C\x75\x64\x5D\x5B\x63\x6D\x70\x27\x0A\x67\x7C\x68\x76\x2E\x66\x62\x2C\x3A\x3D\x2D\x71\x31\x30\x43\x32\x2A\x79\x78\x29\x28\x4C\x39\x41\x53\x2F\x50\x22\x45\x6A\x4D\x49\x6B\x33\x3E\x35\x54\x3C\x44\x34\x7D\x42\x7B\x38\x46\x77\x52\x36\x37\x55\x47\x4E\x3B\x4A\x7A\x56\x23\x48\x4F\x57\x5F\x26\x21\x4B\x3F\x58\x51\x25\x59\x5C\x09\x5A\x2B\x7E\x5E\x24\x40\x60\x7F\x00\x01\x02\x03\x04\x05\x06\x07\x08\x0B\x0C\x0D\x0E\x0F\x10\x11\x12\x13\x14\x15\x16\x17\x18\x19\x1A\x1B\x1C\x1D\x1E\x1F";
  33. // the subset of folded case mappings that is not in lower case mappings
  34. private const FOLD_FROM = ['İ', 'µ', 'ſ', "\xCD\x85", 'ς', 'ϐ', 'ϑ', 'ϕ', 'ϖ', 'ϰ', 'ϱ', 'ϵ', 'ẛ', "\xE1\xBE\xBE", 'ß', 'İ', 'ʼn', 'ǰ', 'ΐ', 'ΰ', 'և', 'ẖ', 'ẗ', 'ẘ', 'ẙ', 'ẚ', 'ẞ', 'ὐ', 'ὒ', 'ὔ', 'ὖ', 'ᾀ', 'ᾁ', 'ᾂ', 'ᾃ', 'ᾄ', 'ᾅ', 'ᾆ', 'ᾇ', 'ᾈ', 'ᾉ', 'ᾊ', 'ᾋ', 'ᾌ', 'ᾍ', 'ᾎ', 'ᾏ', 'ᾐ', 'ᾑ', 'ᾒ', 'ᾓ', 'ᾔ', 'ᾕ', 'ᾖ', 'ᾗ', 'ᾘ', 'ᾙ', 'ᾚ', 'ᾛ', 'ᾜ', 'ᾝ', 'ᾞ', 'ᾟ', 'ᾠ', 'ᾡ', 'ᾢ', 'ᾣ', 'ᾤ', 'ᾥ', 'ᾦ', 'ᾧ', 'ᾨ', 'ᾩ', 'ᾪ', 'ᾫ', 'ᾬ', 'ᾭ', 'ᾮ', 'ᾯ', 'ᾲ', 'ᾳ', 'ᾴ', 'ᾶ', 'ᾷ', 'ᾼ', 'ῂ', 'ῃ', 'ῄ', 'ῆ', 'ῇ', 'ῌ', 'ῒ', 'ΐ', 'ῖ', 'ῗ', 'ῢ', 'ΰ', 'ῤ', 'ῦ', 'ῧ', 'ῲ', 'ῳ', 'ῴ', 'ῶ', 'ῷ', 'ῼ', 'ff', 'fi', 'fl', 'ffi', 'ffl', 'ſt', 'st', 'ﬓ', 'ﬔ', 'ﬕ', 'ﬖ', 'ﬗ'];
  35. private const FOLD_TO = ['i̇', 'μ', 's', 'ι', 'σ', 'β', 'θ', 'φ', 'π', 'κ', 'ρ', 'ε', 'ṡ', 'ι', 'ss', 'i̇', 'ʼn', 'ǰ', 'ΐ', 'ΰ', 'եւ', 'ẖ', 'ẗ', 'ẘ', 'ẙ', 'aʾ', 'ss', 'ὐ', 'ὒ', 'ὔ', 'ὖ', 'ἀι', 'ἁι', 'ἂι', 'ἃι', 'ἄι', 'ἅι', 'ἆι', 'ἇι', 'ἀι', 'ἁι', 'ἂι', 'ἃι', 'ἄι', 'ἅι', 'ἆι', 'ἇι', 'ἠι', 'ἡι', 'ἢι', 'ἣι', 'ἤι', 'ἥι', 'ἦι', 'ἧι', 'ἠι', 'ἡι', 'ἢι', 'ἣι', 'ἤι', 'ἥι', 'ἦι', 'ἧι', 'ὠι', 'ὡι', 'ὢι', 'ὣι', 'ὤι', 'ὥι', 'ὦι', 'ὧι', 'ὠι', 'ὡι', 'ὢι', 'ὣι', 'ὤι', 'ὥι', 'ὦι', 'ὧι', 'ὰι', 'αι', 'άι', 'ᾶ', 'ᾶι', 'αι', 'ὴι', 'ηι', 'ήι', 'ῆ', 'ῆι', 'ηι', 'ῒ', 'ΐ', 'ῖ', 'ῗ', 'ῢ', 'ΰ', 'ῤ', 'ῦ', 'ῧ', 'ὼι', 'ωι', 'ώι', 'ῶ', 'ῶι', 'ωι', 'ff', 'fi', 'fl', 'ffi', 'ffl', 'st', 'st', 'մն', 'մե', 'մի', 'վն', 'մխ'];
  36. // the subset of upper case mappings that map one code point to many code points
  37. private const UPPER_FROM = ['ß', 'ff', 'fi', 'fl', 'ffi', 'ffl', 'ſt', 'st', 'և', 'ﬓ', 'ﬔ', 'ﬕ', 'ﬖ', 'ﬗ', 'ʼn', 'ΐ', 'ΰ', 'ǰ', 'ẖ', 'ẗ', 'ẘ', 'ẙ', 'ẚ', 'ὐ', 'ὒ', 'ὔ', 'ὖ', 'ᾶ', 'ῆ', 'ῒ', 'ΐ', 'ῖ', 'ῗ', 'ῢ', 'ΰ', 'ῤ', 'ῦ', 'ῧ', 'ῶ'];
  38. private const UPPER_TO = ['SS', 'FF', 'FI', 'FL', 'FFI', 'FFL', 'ST', 'ST', 'ԵՒ', 'ՄՆ', 'ՄԵ', 'ՄԻ', 'ՎՆ', 'ՄԽ', 'ʼN', 'Ϊ́', 'Ϋ́', 'J̌', 'H̱', 'T̈', 'W̊', 'Y̊', 'Aʾ', 'Υ̓', 'Υ̓̀', 'Υ̓́', 'Υ̓͂', 'Α͂', 'Η͂', 'Ϊ̀', 'Ϊ́', 'Ι͂', 'Ϊ͂', 'Ϋ̀', 'Ϋ́', 'Ρ̓', 'Υ͂', 'Ϋ͂', 'Ω͂'];
  39. // the subset of https://github.com/unicode-org/cldr/blob/master/common/transforms/Latin-ASCII.xml that is not in NFKD
  40. private const TRANSLIT_FROM = ['Æ', 'Ð', 'Ø', 'Þ', 'ß', 'æ', 'ð', 'ø', 'þ', 'Đ', 'đ', 'Ħ', 'ħ', 'ı', 'ĸ', 'Ŀ', 'ŀ', 'Ł', 'ł', 'ʼn', 'Ŋ', 'ŋ', 'Œ', 'œ', 'Ŧ', 'ŧ', 'ƀ', 'Ɓ', 'Ƃ', 'ƃ', 'Ƈ', 'ƈ', 'Ɖ', 'Ɗ', 'Ƌ', 'ƌ', 'Ɛ', 'Ƒ', 'ƒ', 'Ɠ', 'ƕ', 'Ɩ', 'Ɨ', 'Ƙ', 'ƙ', 'ƚ', 'Ɲ', 'ƞ', 'Ƣ', 'ƣ', 'Ƥ', 'ƥ', 'ƫ', 'Ƭ', 'ƭ', 'Ʈ', 'Ʋ', 'Ƴ', 'ƴ', 'Ƶ', 'ƶ', 'DŽ', 'Dž', 'dž', 'Ǥ', 'ǥ', 'ȡ', 'Ȥ', 'ȥ', 'ȴ', 'ȵ', 'ȶ', 'ȷ', 'ȸ', 'ȹ', 'Ⱥ', 'Ȼ', 'ȼ', 'Ƚ', 'Ⱦ', 'ȿ', 'ɀ', 'Ƀ', 'Ʉ', 'Ɇ', 'ɇ', 'Ɉ', 'ɉ', 'Ɍ', 'ɍ', 'Ɏ', 'ɏ', 'ɓ', 'ɕ', 'ɖ', 'ɗ', 'ɛ', 'ɟ', 'ɠ', 'ɡ', 'ɢ', 'ɦ', 'ɧ', 'ɨ', 'ɪ', 'ɫ', 'ɬ', 'ɭ', 'ɱ', 'ɲ', 'ɳ', 'ɴ', 'ɶ', 'ɼ', 'ɽ', 'ɾ', 'ʀ', 'ʂ', 'ʈ', 'ʉ', 'ʋ', 'ʏ', 'ʐ', 'ʑ', 'ʙ', 'ʛ', 'ʜ', 'ʝ', 'ʟ', 'ʠ', 'ʣ', 'ʥ', 'ʦ', 'ʪ', 'ʫ', 'ᴀ', 'ᴁ', 'ᴃ', 'ᴄ', 'ᴅ', 'ᴆ', 'ᴇ', 'ᴊ', 'ᴋ', 'ᴌ', 'ᴍ', 'ᴏ', 'ᴘ', 'ᴛ', 'ᴜ', 'ᴠ', 'ᴡ', 'ᴢ', 'ᵫ', 'ᵬ', 'ᵭ', 'ᵮ', 'ᵯ', 'ᵰ', 'ᵱ', 'ᵲ', 'ᵳ', 'ᵴ', 'ᵵ', 'ᵶ', 'ᵺ', 'ᵻ', 'ᵽ', 'ᵾ', 'ᶀ', 'ᶁ', 'ᶂ', 'ᶃ', 'ᶄ', 'ᶅ', 'ᶆ', 'ᶇ', 'ᶈ', 'ᶉ', 'ᶊ', 'ᶌ', 'ᶍ', 'ᶎ', 'ᶏ', 'ᶑ', 'ᶒ', 'ᶓ', 'ᶖ', 'ᶙ', 'ẚ', 'ẜ', 'ẝ', 'ẞ', 'Ỻ', 'ỻ', 'Ỽ', 'ỽ', 'Ỿ', 'ỿ', '©', '®', '₠', '₢', '₣', '₤', '₧', '₺', '₹', 'ℌ', '℞', '㎧', '㎮', '㏆', '㏗', '㏞', '㏟', '¼', '½', '¾', '⅓', '⅔', '⅕', '⅖', '⅗', '⅘', '⅙', '⅚', '⅛', '⅜', '⅝', '⅞', '⅟', '〇', '‘', '’', '‚', '‛', '“', '”', '„', '‟', '′', '″', '〝', '〞', '«', '»', '‹', '›', '‐', '‑', '‒', '–', '—', '―', '︱', '︲', '﹘', '‖', '⁄', '⁅', '⁆', '⁎', '、', '。', '〈', '〉', '《', '》', '〔', '〕', '〘', '〙', '〚', '〛', '︑', '︒', '︹', '︺', '︽', '︾', '︿', '﹀', '﹑', '﹝', '﹞', '⦅', '⦆', '。', '、', '×', '÷', '−', '∕', '∖', '∣', '∥', '≪', '≫', '⦅', '⦆'];
  41. private const TRANSLIT_TO = ['AE', 'D', 'O', 'TH', 'ss', 'ae', 'd', 'o', 'th', 'D', 'd', 'H', 'h', 'i', 'q', 'L', 'l', 'L', 'l', '\'n', 'N', 'n', 'OE', 'oe', 'T', 't', 'b', 'B', 'B', 'b', 'C', 'c', 'D', 'D', 'D', 'd', 'E', 'F', 'f', 'G', 'hv', 'I', 'I', 'K', 'k', 'l', 'N', 'n', 'OI', 'oi', 'P', 'p', 't', 'T', 't', 'T', 'V', 'Y', 'y', 'Z', 'z', 'DZ', 'Dz', 'dz', 'G', 'g', 'd', 'Z', 'z', 'l', 'n', 't', 'j', 'db', 'qp', 'A', 'C', 'c', 'L', 'T', 's', 'z', 'B', 'U', 'E', 'e', 'J', 'j', 'R', 'r', 'Y', 'y', 'b', 'c', 'd', 'd', 'e', 'j', 'g', 'g', 'G', 'h', 'h', 'i', 'I', 'l', 'l', 'l', 'm', 'n', 'n', 'N', 'OE', 'r', 'r', 'r', 'R', 's', 't', 'u', 'v', 'Y', 'z', 'z', 'B', 'G', 'H', 'j', 'L', 'q', 'dz', 'dz', 'ts', 'ls', 'lz', 'A', 'AE', 'B', 'C', 'D', 'D', 'E', 'J', 'K', 'L', 'M', 'O', 'P', 'T', 'U', 'V', 'W', 'Z', 'ue', 'b', 'd', 'f', 'm', 'n', 'p', 'r', 'r', 's', 't', 'z', 'th', 'I', 'p', 'U', 'b', 'd', 'f', 'g', 'k', 'l', 'm', 'n', 'p', 'r', 's', 'v', 'x', 'z', 'a', 'd', 'e', 'e', 'i', 'u', 'a', 's', 's', 'SS', 'LL', 'll', 'V', 'v', 'Y', 'y', '(C)', '(R)', 'CE', 'Cr', 'Fr.', 'L.', 'Pts', 'TL', 'Rs', 'x', 'Rx', 'm/s', 'rad/s', 'C/kg', 'pH', 'V/m', 'A/m', ' 1/4', ' 1/2', ' 3/4', ' 1/3', ' 2/3', ' 1/5', ' 2/5', ' 3/5', ' 4/5', ' 1/6', ' 5/6', ' 1/8', ' 3/8', ' 5/8', ' 7/8', ' 1/', '0', '\'', '\'', ',', '\'', '"', '"', ',,', '"', '\'', '"', '"', '"', '<<', '>>', '<', '>', '-', '-', '-', '-', '-', '-', '-', '-', '-', '||', '/', '[', ']', '*', ',', '.', '<', '>', '<<', '>>', '[', ']', '[', ']', '[', ']', ',', '.', '[', ']', '<<', '>>', '<', '>', ',', '[', ']', '((', '))', '.', ',', '*', '/', '-', '/', '\\', '|', '||', '<<', '>>', '((', '))'];
  42. private static $transliterators = [];
  43. /**
  44. * @return static
  45. */
  46. public static function fromCodePoints(int ...$codes): self
  47. {
  48. $string = '';
  49. foreach ($codes as $code) {
  50. if (0x80 > $code %= 0x200000) {
  51. $string .= \chr($code);
  52. } elseif (0x800 > $code) {
  53. $string .= \chr(0xC0 | $code >> 6).\chr(0x80 | $code & 0x3F);
  54. } elseif (0x10000 > $code) {
  55. $string .= \chr(0xE0 | $code >> 12).\chr(0x80 | $code >> 6 & 0x3F).\chr(0x80 | $code & 0x3F);
  56. } else {
  57. $string .= \chr(0xF0 | $code >> 18).\chr(0x80 | $code >> 12 & 0x3F).\chr(0x80 | $code >> 6 & 0x3F).\chr(0x80 | $code & 0x3F);
  58. }
  59. }
  60. return new static($string);
  61. }
  62. /**
  63. * Generic UTF-8 to ASCII transliteration.
  64. *
  65. * Install the intl extension for best results.
  66. *
  67. * @param string[]|\Transliterator[] $rules See "*-Latin" rules from Transliterator::listIDs()
  68. */
  69. public function ascii(array $rules = []): self
  70. {
  71. $str = clone $this;
  72. $s = $str->string;
  73. $str->string = '';
  74. array_unshift($rules, 'nfd');
  75. $rules[] = 'latin-ascii';
  76. if (\function_exists('transliterator_transliterate')) {
  77. $rules[] = 'any-latin/bgn';
  78. }
  79. $rules[] = 'nfkd';
  80. $rules[] = '[:nonspacing mark:] remove';
  81. while (\strlen($s) - 1 > $i = strspn($s, self::ASCII)) {
  82. if (0 < --$i) {
  83. $str->string .= substr($s, 0, $i);
  84. $s = substr($s, $i);
  85. }
  86. if (!$rule = array_shift($rules)) {
  87. $rules = []; // An empty rule interrupts the next ones
  88. }
  89. if ($rule instanceof \Transliterator) {
  90. $s = $rule->transliterate($s);
  91. } elseif ($rule) {
  92. if ('nfd' === $rule = strtolower($rule)) {
  93. normalizer_is_normalized($s, self::NFD) ?: $s = normalizer_normalize($s, self::NFD);
  94. } elseif ('nfkd' === $rule) {
  95. normalizer_is_normalized($s, self::NFKD) ?: $s = normalizer_normalize($s, self::NFKD);
  96. } elseif ('[:nonspacing mark:] remove' === $rule) {
  97. $s = preg_replace('/\p{Mn}++/u', '', $s);
  98. } elseif ('latin-ascii' === $rule) {
  99. $s = str_replace(self::TRANSLIT_FROM, self::TRANSLIT_TO, $s);
  100. } elseif ('de-ascii' === $rule) {
  101. $s = preg_replace("/([AUO])\u{0308}(?=\p{Ll})/u", '$1e', $s);
  102. $s = str_replace(["a\u{0308}", "o\u{0308}", "u\u{0308}", "A\u{0308}", "O\u{0308}", "U\u{0308}"], ['ae', 'oe', 'ue', 'AE', 'OE', 'UE'], $s);
  103. } elseif (\function_exists('transliterator_transliterate')) {
  104. if (null === $transliterator = self::$transliterators[$rule] ?? self::$transliterators[$rule] = \Transliterator::create($rule)) {
  105. if ('any-latin/bgn' === $rule) {
  106. $rule = 'any-latin';
  107. $transliterator = self::$transliterators[$rule] ?? self::$transliterators[$rule] = \Transliterator::create($rule);
  108. }
  109. if (null === $transliterator) {
  110. throw new InvalidArgumentException(sprintf('Unknown transliteration rule "%s".', $rule));
  111. }
  112. self::$transliterators['any-latin/bgn'] = $transliterator;
  113. }
  114. $s = $transliterator->transliterate($s);
  115. }
  116. } elseif (!\function_exists('iconv')) {
  117. $s = preg_replace('/[^\x00-\x7F]/u', '?', $s);
  118. } elseif (ICONV_IMPL === 'glibc') {
  119. $s = iconv('UTF-8', 'ASCII//TRANSLIT', $s);
  120. } else {
  121. $s = preg_replace_callback('/[^\x00-\x7F]/u', static function ($c) {
  122. $c = iconv('UTF-8', 'ASCII//IGNORE//TRANSLIT', $c[0]);
  123. return 1 < \strlen($c) ? ltrim($c, '\'`"^~') : (\strlen($c) ? $c : '?');
  124. }, $s);
  125. }
  126. }
  127. $str->string .= $s;
  128. return $str;
  129. }
  130. public function camel(): parent
  131. {
  132. $str = clone $this;
  133. $str->string = str_replace(' ', '', preg_replace_callback('/\b./u', static function ($m) use (&$i) {
  134. return 1 === ++$i ? ('İ' === $m[0] ? 'i̇' : mb_strtolower($m[0], 'UTF-8')) : mb_convert_case($m[0], MB_CASE_TITLE, 'UTF-8');
  135. }, preg_replace('/[^\pL0-9]++/u', ' ', $this->string)));
  136. return $str;
  137. }
  138. /**
  139. * @return int[]
  140. */
  141. public function codePointsAt(int $offset): array
  142. {
  143. $str = $this->slice($offset, 1);
  144. if ('' === $str->string) {
  145. return [];
  146. }
  147. $codePoints = [];
  148. foreach (preg_split('//u', $str->string, -1, PREG_SPLIT_NO_EMPTY) as $c) {
  149. $codePoints[] = mb_ord($c, 'UTF-8');
  150. }
  151. return $codePoints;
  152. }
  153. public function folded(bool $compat = true): parent
  154. {
  155. $str = clone $this;
  156. if (!$compat || \PHP_VERSION_ID < 70300 || !\defined('Normalizer::NFKC_CF')) {
  157. $str->string = normalizer_normalize($str->string, $compat ? \Normalizer::NFKC : \Normalizer::NFC);
  158. $str->string = mb_strtolower(str_replace(self::FOLD_FROM, self::FOLD_TO, $this->string), 'UTF-8');
  159. } else {
  160. $str->string = normalizer_normalize($str->string, \Normalizer::NFKC_CF);
  161. }
  162. return $str;
  163. }
  164. public function join(array $strings, string $lastGlue = null): parent
  165. {
  166. $str = clone $this;
  167. $tail = null !== $lastGlue && 1 < \count($strings) ? $lastGlue.array_pop($strings) : '';
  168. $str->string = implode($this->string, $strings).$tail;
  169. if (!preg_match('//u', $str->string)) {
  170. throw new InvalidArgumentException('Invalid UTF-8 string.');
  171. }
  172. return $str;
  173. }
  174. public function lower(): parent
  175. {
  176. $str = clone $this;
  177. $str->string = mb_strtolower(str_replace('İ', 'i̇', $str->string), 'UTF-8');
  178. return $str;
  179. }
  180. public function match(string $regexp, int $flags = 0, int $offset = 0): array
  181. {
  182. $match = ((PREG_PATTERN_ORDER | PREG_SET_ORDER) & $flags) ? 'preg_match_all' : 'preg_match';
  183. if ($this->ignoreCase) {
  184. $regexp .= 'i';
  185. }
  186. set_error_handler(static function ($t, $m) { throw new InvalidArgumentException($m); });
  187. try {
  188. if (false === $match($regexp.'u', $this->string, $matches, $flags | PREG_UNMATCHED_AS_NULL, $offset)) {
  189. $lastError = preg_last_error();
  190. foreach (get_defined_constants(true)['pcre'] as $k => $v) {
  191. if ($lastError === $v && '_ERROR' === substr($k, -6)) {
  192. throw new RuntimeException('Matching failed with '.$k.'.');
  193. }
  194. }
  195. throw new RuntimeException('Matching failed with unknown error code.');
  196. }
  197. } finally {
  198. restore_error_handler();
  199. }
  200. return $matches;
  201. }
  202. /**
  203. * @return static
  204. */
  205. public function normalize(int $form = self::NFC): self
  206. {
  207. if (!\in_array($form, [self::NFC, self::NFD, self::NFKC, self::NFKD])) {
  208. throw new InvalidArgumentException('Unsupported normalization form.');
  209. }
  210. $str = clone $this;
  211. normalizer_is_normalized($str->string, $form) ?: $str->string = normalizer_normalize($str->string, $form);
  212. return $str;
  213. }
  214. public function padBoth(int $length, string $padStr = ' '): parent
  215. {
  216. if ('' === $padStr || !preg_match('//u', $padStr)) {
  217. throw new InvalidArgumentException('Invalid UTF-8 string.');
  218. }
  219. $pad = clone $this;
  220. $pad->string = $padStr;
  221. return $this->pad($length, $pad, STR_PAD_BOTH);
  222. }
  223. public function padEnd(int $length, string $padStr = ' '): parent
  224. {
  225. if ('' === $padStr || !preg_match('//u', $padStr)) {
  226. throw new InvalidArgumentException('Invalid UTF-8 string.');
  227. }
  228. $pad = clone $this;
  229. $pad->string = $padStr;
  230. return $this->pad($length, $pad, STR_PAD_RIGHT);
  231. }
  232. public function padStart(int $length, string $padStr = ' '): parent
  233. {
  234. if ('' === $padStr || !preg_match('//u', $padStr)) {
  235. throw new InvalidArgumentException('Invalid UTF-8 string.');
  236. }
  237. $pad = clone $this;
  238. $pad->string = $padStr;
  239. return $this->pad($length, $pad, STR_PAD_LEFT);
  240. }
  241. public function replaceMatches(string $fromRegexp, $to): parent
  242. {
  243. if ($this->ignoreCase) {
  244. $fromRegexp .= 'i';
  245. }
  246. if (\is_array($to) || $to instanceof \Closure) {
  247. if (!\is_callable($to)) {
  248. throw new \TypeError(sprintf('Argument 2 passed to "%s::replaceMatches()" must be callable, array given.', static::class));
  249. }
  250. $replace = 'preg_replace_callback';
  251. $to = static function (array $m) use ($to): string {
  252. $to = $to($m);
  253. if ('' !== $to && (!\is_string($to) || !preg_match('//u', $to))) {
  254. throw new InvalidArgumentException('Replace callback must return a valid UTF-8 string.');
  255. }
  256. return $to;
  257. };
  258. } elseif ('' !== $to && !preg_match('//u', $to)) {
  259. throw new InvalidArgumentException('Invalid UTF-8 string.');
  260. } else {
  261. $replace = 'preg_replace';
  262. }
  263. set_error_handler(static function ($t, $m) { throw new InvalidArgumentException($m); });
  264. try {
  265. if (null === $string = $replace($fromRegexp.'u', $to, $this->string)) {
  266. $lastError = preg_last_error();
  267. foreach (get_defined_constants(true)['pcre'] as $k => $v) {
  268. if ($lastError === $v && '_ERROR' === substr($k, -6)) {
  269. throw new RuntimeException('Matching failed with '.$k.'.');
  270. }
  271. }
  272. throw new RuntimeException('Matching failed with unknown error code.');
  273. }
  274. } finally {
  275. restore_error_handler();
  276. }
  277. $str = clone $this;
  278. $str->string = $string;
  279. return $str;
  280. }
  281. public function reverse(): parent
  282. {
  283. $str = clone $this;
  284. $str->string = implode('', array_reverse(preg_split('/(\X)/u', $str->string, -1, PREG_SPLIT_DELIM_CAPTURE | PREG_SPLIT_NO_EMPTY)));
  285. return $str;
  286. }
  287. public function snake(): parent
  288. {
  289. $str = $this->camel()->title();
  290. $str->string = mb_strtolower(preg_replace(['/(\p{Lu}+)(\p{Lu}\p{Ll})/u', '/([\p{Ll}0-9])(\p{Lu})/u'], '\1_\2', $str->string), 'UTF-8');
  291. return $str;
  292. }
  293. public function title(bool $allWords = false): parent
  294. {
  295. $str = clone $this;
  296. $limit = $allWords ? -1 : 1;
  297. $str->string = preg_replace_callback('/\b./u', static function (array $m): string {
  298. return mb_convert_case($m[0], MB_CASE_TITLE, 'UTF-8');
  299. }, $str->string, $limit);
  300. return $str;
  301. }
  302. public function trim(string $chars = " \t\n\r\0\x0B\x0C\u{A0}\u{FEFF}"): parent
  303. {
  304. if (" \t\n\r\0\x0B\x0C\u{A0}\u{FEFF}" !== $chars && !preg_match('//u', $chars)) {
  305. throw new InvalidArgumentException('Invalid UTF-8 chars.');
  306. }
  307. $chars = preg_quote($chars);
  308. $str = clone $this;
  309. $str->string = preg_replace("{^[$chars]++|[$chars]++$}uD", '', $str->string);
  310. return $str;
  311. }
  312. public function trimEnd(string $chars = " \t\n\r\0\x0B\x0C\u{A0}\u{FEFF}"): parent
  313. {
  314. if (" \t\n\r\0\x0B\x0C\u{A0}\u{FEFF}" !== $chars && !preg_match('//u', $chars)) {
  315. throw new InvalidArgumentException('Invalid UTF-8 chars.');
  316. }
  317. $chars = preg_quote($chars);
  318. $str = clone $this;
  319. $str->string = preg_replace("{[$chars]++$}uD", '', $str->string);
  320. return $str;
  321. }
  322. public function trimStart(string $chars = " \t\n\r\0\x0B\x0C\u{A0}\u{FEFF}"): parent
  323. {
  324. if (" \t\n\r\0\x0B\x0C\u{A0}\u{FEFF}" !== $chars && !preg_match('//u', $chars)) {
  325. throw new InvalidArgumentException('Invalid UTF-8 chars.');
  326. }
  327. $chars = preg_quote($chars);
  328. $str = clone $this;
  329. $str->string = preg_replace("{^[$chars]++}uD", '', $str->string);
  330. return $str;
  331. }
  332. public function upper(): parent
  333. {
  334. $str = clone $this;
  335. $str->string = mb_strtoupper($str->string, 'UTF-8');
  336. if (\PHP_VERSION_ID < 70300) {
  337. $str->string = str_replace(self::UPPER_FROM, self::UPPER_TO, $str->string);
  338. }
  339. return $str;
  340. }
  341. public function width(bool $ignoreAnsiDecoration = true): int
  342. {
  343. $width = 0;
  344. $s = str_replace(["\x00", "\x05", "\x07"], '', $this->string);
  345. if (false !== strpos($s, "\r")) {
  346. $s = str_replace(["\r\n", "\r"], "\n", $s);
  347. }
  348. if (!$ignoreAnsiDecoration) {
  349. $s = preg_replace('/[\p{Cc}\x7F]++/u', '', $s);
  350. }
  351. foreach (explode("\n", $s) as $s) {
  352. if ($ignoreAnsiDecoration) {
  353. $s = preg_replace('/(?:\x1B(?:
  354. \[ [\x30-\x3F]*+ [\x20-\x2F]*+ [0x40-\x7E]
  355. | [P\]X^_] .*? \x1B\\\\
  356. | [\x41-\x7E]
  357. )|[\p{Cc}\x7F]++)/xu', '', $s);
  358. }
  359. // Non printable characters have been dropped, so wcswidth cannot logically return -1.
  360. $width += $this->wcswidth($s);
  361. }
  362. return $width;
  363. }
  364. /**
  365. * @return static
  366. */
  367. private function pad(int $len, self $pad, int $type): parent
  368. {
  369. $sLen = $this->length();
  370. if ($len <= $sLen) {
  371. return clone $this;
  372. }
  373. $padLen = $pad->length();
  374. $freeLen = $len - $sLen;
  375. $len = $freeLen % $padLen;
  376. switch ($type) {
  377. case STR_PAD_RIGHT:
  378. return $this->append(str_repeat($pad->string, $freeLen / $padLen).($len ? $pad->slice(0, $len) : ''));
  379. case STR_PAD_LEFT:
  380. return $this->prepend(str_repeat($pad->string, $freeLen / $padLen).($len ? $pad->slice(0, $len) : ''));
  381. case STR_PAD_BOTH:
  382. $freeLen /= 2;
  383. $rightLen = ceil($freeLen);
  384. $len = $rightLen % $padLen;
  385. $str = $this->append(str_repeat($pad->string, $rightLen / $padLen).($len ? $pad->slice(0, $len) : ''));
  386. $leftLen = floor($freeLen);
  387. $len = $leftLen % $padLen;
  388. return $str->prepend(str_repeat($pad->string, $leftLen / $padLen).($len ? $pad->slice(0, $len) : ''));
  389. default:
  390. throw new InvalidArgumentException('Invalid padding type.');
  391. }
  392. }
  393. /**
  394. * Based on https://github.com/jquast/wcwidth, a Python implementation of https://www.cl.cam.ac.uk/~mgk25/ucs/wcwidth.c.
  395. */
  396. private function wcswidth(string $string): int
  397. {
  398. $width = 0;
  399. foreach (preg_split('//u', $string, -1, PREG_SPLIT_NO_EMPTY) as $c) {
  400. $codePoint = mb_ord($c, 'UTF-8');
  401. if (0 === $codePoint // NULL
  402. || 0x034F === $codePoint // COMBINING GRAPHEME JOINER
  403. || (0x200B <= $codePoint && 0x200F >= $codePoint) // ZERO WIDTH SPACE to RIGHT-TO-LEFT MARK
  404. || 0x2028 === $codePoint // LINE SEPARATOR
  405. || 0x2029 === $codePoint // PARAGRAPH SEPARATOR
  406. || (0x202A <= $codePoint && 0x202E >= $codePoint) // LEFT-TO-RIGHT EMBEDDING to RIGHT-TO-LEFT OVERRIDE
  407. || (0x2060 <= $codePoint && 0x2063 >= $codePoint) // WORD JOINER to INVISIBLE SEPARATOR
  408. ) {
  409. continue;
  410. }
  411. // Non printable characters
  412. if (32 > $codePoint // C0 control characters
  413. || (0x07F <= $codePoint && 0x0A0 > $codePoint) // C1 control characters and DEL
  414. ) {
  415. return -1;
  416. }
  417. static $tableZero;
  418. if (null === $tableZero) {
  419. $tableZero = require __DIR__.'/Resources/data/wcswidth_table_zero.php';
  420. }
  421. if ($codePoint >= $tableZero[0][0] && $codePoint <= $tableZero[$ubound = \count($tableZero) - 1][1]) {
  422. $lbound = 0;
  423. while ($ubound >= $lbound) {
  424. $mid = floor(($lbound + $ubound) / 2);
  425. if ($codePoint > $tableZero[$mid][1]) {
  426. $lbound = $mid + 1;
  427. } elseif ($codePoint < $tableZero[$mid][0]) {
  428. $ubound = $mid - 1;
  429. } else {
  430. continue 2;
  431. }
  432. }
  433. }
  434. static $tableWide;
  435. if (null === $tableWide) {
  436. $tableWide = require __DIR__.'/Resources/data/wcswidth_table_wide.php';
  437. }
  438. if ($codePoint >= $tableWide[0][0] && $codePoint <= $tableWide[$ubound = \count($tableWide) - 1][1]) {
  439. $lbound = 0;
  440. while ($ubound >= $lbound) {
  441. $mid = floor(($lbound + $ubound) / 2);
  442. if ($codePoint > $tableWide[$mid][1]) {
  443. $lbound = $mid + 1;
  444. } elseif ($codePoint < $tableWide[$mid][0]) {
  445. $ubound = $mid - 1;
  446. } else {
  447. $width += 2;
  448. continue 2;
  449. }
  450. }
  451. }
  452. ++$width;
  453. }
  454. return $width;
  455. }
  456. }