Page Menu
Home
WickedGov Phorge
Search
Configure Global Search
Log In
Files
F5978227
Normalizer.php
No One
Temporary
Actions
Download File
Edit File
Delete File
View Transforms
Subscribe
Flag For Later
Award Token
Size
9 KB
Referenced Files
None
Subscribers
None
Normalizer.php
View Options
<?php
/*
* This file is part of the Symfony package.
*
* (c) Fabien Potencier <fabien@symfony.com>
*
* For the full copyright and license information, please view the LICENSE
* file that was distributed with this source code.
*/
namespace
Symfony\Polyfill\Intl\Normalizer
;
/**
* Normalizer is a PHP fallback implementation of the Normalizer class provided by the intl extension.
*
* It has been validated with Unicode 6.3 Normalization Conformance Test.
* See http://www.unicode.org/reports/tr15/ for detailed info about Unicode normalizations.
*
* @author Nicolas Grekas <p@tchwork.com>
*
* @internal
*/
class
Normalizer
{
public
const
FORM_D
=
\Normalizer
::
FORM_D
;
public
const
FORM_KD
=
\Normalizer
::
FORM_KD
;
public
const
FORM_C
=
\Normalizer
::
FORM_C
;
public
const
FORM_KC
=
\Normalizer
::
FORM_KC
;
public
const
NFD
=
\Normalizer
::
NFD
;
public
const
NFKD
=
\Normalizer
::
NFKD
;
public
const
NFC
=
\Normalizer
::
NFC
;
public
const
NFKC
=
\Normalizer
::
NFKC
;
private
static
$C
;
private
static
$D
;
private
static
$KD
;
private
static
$cC
;
private
static
$ulenMask
=
[
"
\x
C0"
=>
2
,
"
\x
D0"
=>
2
,
"
\x
E0"
=>
3
,
"
\x
F0"
=>
4
];
private
static
$ASCII
=
"
\x
20
\x
65
\x
69
\x
61
\x
73
\x
6E
\x
74
\x
72
\x
6F
\x
6C
\x
75
\x
64
\x
5D
\x
5B
\x
63
\x
6D
\x
70
\x
27
\x
0A
\x
67
\x
7C
\x
68
\x
76
\x
2E
\x
66
\x
62
\x
2C
\x
3A
\x
3D
\x
2D
\x
71
\x
31
\x
30
\x
43
\x
32
\x
2A
\x
79
\x
78
\x
29
\x
28
\x
4C
\x
39
\x
41
\x
53
\x
2F
\x
50
\x
22
\x
45
\x
6A
\x
4D
\x
49
\x
6B
\x
33
\x
3E
\x
35
\x
54
\x
3C
\x
44
\x
34
\x
7D
\x
42
\x
7B
\x
38
\x
46
\x
77
\x
52
\x
36
\x
37
\x
55
\x
47
\x
4E
\x
3B
\x
4A
\x
7A
\x
56
\x
23
\x
48
\x
4F
\x
57
\x
5F
\x
26
\x
21
\x
4B
\x
3F
\x
58
\x
51
\x
25
\x
59
\x
5C
\x
09
\x
5A
\x
2B
\x
7E
\x
5E
\x
24
\x
40
\x
60
\x
7F
\x
00
\x
01
\x
02
\x
03
\x
04
\x
05
\x
06
\x
07
\x
08
\x
0B
\x
0C
\x
0D
\x
0E
\x
0F
\x
10
\x
11
\x
12
\x
13
\x
14
\x
15
\x
16
\x
17
\x
18
\x
19
\x
1A
\x
1B
\x
1C
\x
1D
\x
1E
\x
1F"
;
public
static
function
isNormalized
(
string
$s
,
int
$form
=
self
::
FORM_C
)
{
if
(!
\in_array
(
$form
,
[
self
::
NFD
,
self
::
NFKD
,
self
::
NFC
,
self
::
NFKC
]))
{
return
false
;
}
if
(!
isset
(
$s
[
strspn
(
$s
,
self
::
$ASCII
)]))
{
return
true
;
}
if
(
self
::
NFC
==
$form
&&
preg_match
(
'//u'
,
$s
)
&&
!
preg_match
(
'/[^
\x
00-
\x
{2FF}]/u'
,
$s
))
{
return
true
;
}
return
self
::
normalize
(
$s
,
$form
)
===
$s
;
}
public
static
function
normalize
(
string
$s
,
int
$form
=
self
::
FORM_C
)
{
if
(!
preg_match
(
'//u'
,
$s
))
{
return
false
;
}
switch
(
$form
)
{
case
self
::
NFC
:
$C
=
true
;
$K
=
false
;
break
;
case
self
::
NFD
:
$C
=
false
;
$K
=
false
;
break
;
case
self
::
NFKC
:
$C
=
true
;
$K
=
true
;
break
;
case
self
::
NFKD
:
$C
=
false
;
$K
=
true
;
break
;
default
:
if
(
\defined
(
'Normalizer::NONE'
)
&&
\Normalizer
::
NONE
==
$form
)
{
return
$s
;
}
if
(
80000
>
\PHP_VERSION_ID
)
{
return
false
;
}
throw
new
\ValueError
(
'normalizer_normalize(): Argument #2 ($form) must be a a valid normalization form'
);
}
if
(
''
===
$s
)
{
return
''
;
}
if
(
$K
&&
null
===
self
::
$KD
)
{
self
::
$KD
=
self
::
getData
(
'compatibilityDecomposition'
);
}
if
(
null
===
self
::
$D
)
{
self
::
$D
=
self
::
getData
(
'canonicalDecomposition'
);
self
::
$cC
=
self
::
getData
(
'combiningClass'
);
}
if
(
null
!==
$mbEncoding
=
(
2
/* MB_OVERLOAD_STRING */
&
(
int
)
\ini_get
(
'mbstring.func_overload'
))
?
mb_internal_encoding
()
:
null
)
{
mb_internal_encoding
(
'8bit'
);
}
$r
=
self
::
decompose
(
$s
,
$K
);
if
(
$C
)
{
if
(
null
===
self
::
$C
)
{
self
::
$C
=
self
::
getData
(
'canonicalComposition'
);
}
$r
=
self
::
recompose
(
$r
);
}
if
(
null
!==
$mbEncoding
)
{
mb_internal_encoding
(
$mbEncoding
);
}
return
$r
;
}
private
static
function
recompose
(
$s
)
{
$ASCII
=
self
::
$ASCII
;
$compMap
=
self
::
$C
;
$combClass
=
self
::
$cC
;
$ulenMask
=
self
::
$ulenMask
;
$result
=
$tail
=
''
;
$i
=
$s
[
0
]
<
"
\x
80"
?
1
:
$ulenMask
[
$s
[
0
]
&
"
\x
F0"
];
$len
=
\strlen
(
$s
);
$lastUchr
=
substr
(
$s
,
0
,
$i
);
$lastUcls
=
isset
(
$combClass
[
$lastUchr
])
?
256
:
0
;
while
(
$i
<
$len
)
{
if
(
$s
[
$i
]
<
"
\x
80"
)
{
// ASCII chars
if
(
$tail
)
{
$lastUchr
.=
$tail
;
$tail
=
''
;
}
if
(
$j
=
strspn
(
$s
,
$ASCII
,
$i
+
1
))
{
$lastUchr
.=
substr
(
$s
,
$i
,
$j
);
$i
+=
$j
;
}
$result
.=
$lastUchr
;
$lastUchr
=
$s
[
$i
];
$lastUcls
=
0
;
++
$i
;
continue
;
}
$ulen
=
$ulenMask
[
$s
[
$i
]
&
"
\x
F0"
];
$uchr
=
substr
(
$s
,
$i
,
$ulen
);
if
(
$lastUchr
<
"
\x
E1
\x
84
\x
80"
||
"
\x
E1
\x
84
\x
92"
<
$lastUchr
||
$uchr
<
"
\x
E1
\x
85
\x
A1"
||
"
\x
E1
\x
85
\x
B5"
<
$uchr
||
$lastUcls
)
{
// Table lookup and combining chars composition
$ucls
=
$combClass
[
$uchr
]
??
0
;
if
(
isset
(
$compMap
[
$lastUchr
.
$uchr
])
&&
(!
$lastUcls
||
$lastUcls
<
$ucls
))
{
$lastUchr
=
$compMap
[
$lastUchr
.
$uchr
];
}
elseif
(
$lastUcls
=
$ucls
)
{
$tail
.=
$uchr
;
}
else
{
if
(
$tail
)
{
$lastUchr
.=
$tail
;
$tail
=
''
;
}
$result
.=
$lastUchr
;
$lastUchr
=
$uchr
;
}
}
else
{
// Hangul chars
$L
=
\ord
(
$lastUchr
[
2
])
-
0x80
;
$V
=
\ord
(
$uchr
[
2
])
-
0xA1
;
$T
=
0
;
$uchr
=
substr
(
$s
,
$i
+
$ulen
,
3
);
if
(
"
\x
E1
\x
86
\x
A7"
<=
$uchr
&&
$uchr
<=
"
\x
E1
\x
87
\x
82"
)
{
$T
=
\ord
(
$uchr
[
2
])
-
0xA7
;
0
>
$T
&&
$T
+=
0x40
;
$ulen
+=
3
;
}
$L
=
0xAC00
+
(
$L
*
21
+
$V
)
*
28
+
$T
;
$lastUchr
=
\chr
(
0xE0
|
$L
>>
12
).
\chr
(
0x80
|
$L
>>
6
&
0x3F
).
\chr
(
0x80
|
$L
&
0x3F
);
}
$i
+=
$ulen
;
}
return
$result
.
$lastUchr
.
$tail
;
}
private
static
function
decompose
(
$s
,
$c
)
{
$result
=
''
;
$ASCII
=
self
::
$ASCII
;
$decompMap
=
self
::
$D
;
$combClass
=
self
::
$cC
;
$ulenMask
=
self
::
$ulenMask
;
if
(
$c
)
{
$compatMap
=
self
::
$KD
;
}
$c
=
[];
$i
=
0
;
$len
=
\strlen
(
$s
);
while
(
$i
<
$len
)
{
if
(
$s
[
$i
]
<
"
\x
80"
)
{
// ASCII chars
if
(
$c
)
{
ksort
(
$c
);
$result
.=
implode
(
''
,
$c
);
$c
=
[];
}
$j
=
1
+
strspn
(
$s
,
$ASCII
,
$i
+
1
);
$result
.=
substr
(
$s
,
$i
,
$j
);
$i
+=
$j
;
continue
;
}
$ulen
=
$ulenMask
[
$s
[
$i
]
&
"
\x
F0"
];
$uchr
=
substr
(
$s
,
$i
,
$ulen
);
$i
+=
$ulen
;
if
(
$uchr
<
"
\x
EA
\x
B0
\x
80"
||
"
\x
ED
\x
9E
\x
A3"
<
$uchr
)
{
// Table lookup
if
(
$uchr
!==
$j
=
$compatMap
[
$uchr
]
??
(
$decompMap
[
$uchr
]
??
$uchr
))
{
$uchr
=
$j
;
$j
=
\strlen
(
$uchr
);
$ulen
=
$uchr
[
0
]
<
"
\x
80"
?
1
:
$ulenMask
[
$uchr
[
0
]
&
"
\x
F0"
];
if
(
$ulen
!=
$j
)
{
// Put trailing chars in $s
$j
-=
$ulen
;
$i
-=
$j
;
if
(
0
>
$i
)
{
$s
=
str_repeat
(
' '
,
-
$i
).
$s
;
$len
-=
$i
;
$i
=
0
;
}
while
(
$j
--)
{
$s
[
$i
+
$j
]
=
$uchr
[
$ulen
+
$j
];
}
$uchr
=
substr
(
$uchr
,
0
,
$ulen
);
}
}
if
(
isset
(
$combClass
[
$uchr
]))
{
// Combining chars, for sorting
if
(!
isset
(
$c
[
$combClass
[
$uchr
]]))
{
$c
[
$combClass
[
$uchr
]]
=
''
;
}
$c
[
$combClass
[
$uchr
]]
.=
$uchr
;
continue
;
}
}
else
{
// Hangul chars
$uchr
=
unpack
(
'C*'
,
$uchr
);
$j
=
((
$uchr
[
1
]
-
224
)
<<
12
)
+
((
$uchr
[
2
]
-
128
)
<<
6
)
+
$uchr
[
3
]
-
0xAC80
;
$uchr
=
"
\x
E1
\x
84"
.
\chr
(
0x80
+
(
int
)
(
$j
/
588
))
.
"
\x
E1
\x
85"
.
\chr
(
0xA1
+
(
int
)
((
$j
%
588
)
/
28
));
if
(
$j
%=
28
)
{
$uchr
.=
$j
<
25
?
(
"
\x
E1
\x
86"
.
\chr
(
0xA7
+
$j
))
:
(
"
\x
E1
\x
87"
.
\chr
(
0x67
+
$j
));
}
}
if
(
$c
)
{
ksort
(
$c
);
$result
.=
implode
(
''
,
$c
);
$c
=
[];
}
$result
.=
$uchr
;
}
if
(
$c
)
{
ksort
(
$c
);
$result
.=
implode
(
''
,
$c
);
}
return
$result
;
}
private
static
function
getData
(
$file
)
{
if
(
file_exists
(
$file
=
__DIR__
.
'/Resources/unidata/'
.
$file
.
'.php'
))
{
return
require
$file
;
}
return
false
;
}
}
File Metadata
Details
Attached
Mime Type
text/x-php
Expires
Sat, Oct 3, 20:18 (3 d, 14 h ago)
Storage Engine
local-disk
Storage Format
Raw Data
Storage Handle
71/55/6f8b975acc8dfa48b9831633ca26
Default Alt Text
Normalizer.php (9 KB)
Attached To
Mode
rMWPROD MediaWiki Production
Attached
Detach File
Event Timeline
Log In to Comment