note 105755 deleted from function.mb-strtolower by danbrown
| From: | danbrown@php.net | Date: | Tue, 13 Sep 2011 00:42:01 +0000 |
| Subject: | note 105755 deleted from function.mb-strtolower by danbrown | ||
| References: | 1 | Groups: | php.notes |
| Request: | Send a blank email to php-notes+get-183127@lists.php.net to get a copy of this message | ||
Note Submitter: akniep at rayo dot info
----
In addition to my last comment:
Since I was not finding any proper solution in the internet on how to map all UTF8-strings to their
lowercase counterpart in PHP, I offer the following hard-coded extended mb_strtolower function for
UTF-8 strings:
The function wraps the existing function mb_strtolower() and additionally replaces uppercase
UTF8-characters for which there is a lowercase representation. Since there is no proper Unicode
uppercase and lowercase character-table in the internet that I was able to find, I checked the first
million UTF8-characters against the Google-search and -KeywordTool and identified the following 49
characters as uppercase-characters, not being replaced by mb_strtolower, but having a UTF8 lowercase
counterpart.
<?php
// the numbers in the in-line-comments display the characters' Unicode code-points (CP).
function strtolower_utf8_extended( $utf8_string )
{
$additional_replacements = array
( "Ç
" => "Ç" // CP 453 -> 454
, "Ç" => "Ç" // CP 456 -> 457
, "Ç" => "Ç" // CP 459 -> 460
, "Dz" => "dz" // CP 498 -> 499
, "Ϸ" => "ϸ" // CP 1015 -> 1016
, "Ϲ" => "ϲ" // CP 1017 -> 1010
, "Ϻ" => "ϻ" // CP 1018 -> 1019
, "á¾" => "á¾" // CP 8075 -> 8067
, "á¾" => "á¾" // CP 8076 -> 8068
, "á¾" => "á¾
" // CP 8077 -> 8069
, "á¾" => "á¾" // CP 8078 -> 8070
, "á¾" => "á¾" // CP 8079 -> 8071
, "á¾" => "á¾" // CP 8088 -> 8080
, "á¾" => "á¾" // CP 8089 -> 8081
, "á¾" => "á¾" // CP 8090 -> 8082
, "á¾" => "á¾" // CP 8091 -> 8083
, "á¾" => "á¾" // CP 8092 -> 8084
, "á¾" => "á¾" // CP 8093 -> 8085
, "á¾" => "á¾" // CP 8094 -> 8086
, "á¾" => "á¾" // CP 8095 -> 8087
, "ᾨ" => "ᾠ" // CP 8104 -> 8096
, "ᾩ" => "ᾡ" // CP 8105 -> 8097
, "ᾪ" => "ᾢ" // CP 8106 -> 8098
, "ᾫ" => "ᾣ" // CP 8107 -> 8099
, "ᾬ" => "ᾤ" // CP 8108 -> 8100
, "á¾" => "á¾¥" // CP 8109 -> 8101
, "ᾮ" => "ᾦ" // CP 8110 -> 8102
, "ᾯ" => "ᾧ" // CP 8111 -> 8103
, "á¾¼" => "á¾³" // CP 8124 -> 8115
, "á¿" => "á¿" // CP 8140 -> 8131
, "ῼ" => "ῳ" // CP 8188 -> 8179
, "â
" => "â
°" // CP 8544 -> 8560
, "â
¡" => "â
±" // CP 8545 -> 8561
, "â
¢" => "â
²" // CP 8546 -> 8562
, "â
£" => "â
³" // CP 8547 -> 8563
, "â
¤" => "â
´" // CP 8548 -> 8564
, "â
¥" => "â
µ" // CP 8549 -> 8565
, "â
¦" => "â
¶" // CP 8550 -> 8566
, "â
§" => "â
·" // CP 8551 -> 8567
, "â
¨" => "â
¸" // CP 8552 -> 8568
, "â
©" => "â
¹" // CP 8553 -> 8569
, "â
ª" => "â
º" // CP 8554 -> 8570
, "â
«" => "â
»" // CP 8555 -> 8571
, "â
¬" => "â
¼" // CP 8556 -> 8572
, "â
" => "â
½" // CP 8557 -> 8573
, "â
®" => "â
¾" // CP 8558 -> 8574
, "â
¯" => "â
¿" // CP 8559 -> 8575
, "ð¦" => "ð" // CP 66598 -> 66638
, "ð§" => "ð" // CP 66599 -> 66639
);
$utf8_string = mb_strtolower( $utf8_string, "UTF-8");
$utf8_string = strtr( $utf8_string, $additional_replacements );
return $utf8_string;
} //strtolower_utf8_extended()
?>