| 1 |
<?php |
| 2 |
|
| 3 |
require_once __DIR__ . '/../MpdfException.php'; |
| 4 |
|
| 5 |
class INDIC |
| 6 |
{ |
| 7 |
/* FROM hb-ot-shape-complex-indic-private.hh */ |
| 8 |
|
| 9 |
// indic_category |
| 10 |
const OT_X = 0; |
| 11 |
const OT_C = 1; |
| 12 |
const OT_V = 2; |
| 13 |
const OT_N = 3; |
| 14 |
const OT_H = 4; |
| 15 |
const OT_ZWNJ = 5; |
| 16 |
const OT_ZWJ = 6; |
| 17 |
const OT_M = 7; /* Matra or Dependent Vowel */ |
| 18 |
const OT_SM = 8; |
| 19 |
const OT_VD = 9; |
| 20 |
const OT_A = 10; |
| 21 |
const OT_NBSP = 11; |
| 22 |
const OT_DOTTEDCIRCLE = 12; /* Not in the spec, but special in Uniscribe. /Very very/ special! */ |
| 23 |
const OT_RS = 13; /* Register Shifter, used in Khmer OT spec */ |
| 24 |
const OT_Coeng = 14; |
| 25 |
const OT_Repha = 15; |
| 26 |
|
| 27 |
const OT_Ra = 16; /* Not explicitly listed in the OT spec, but used in the grammar. */ |
| 28 |
const OT_CM = 17; |
| 29 |
|
| 30 |
// Based on indic_category used to make string to find syllables |
| 31 |
// OT_ to string character (using e.g. OT_C from INDIC) hb-ot-shape-complex-indic-private.hh |
| 32 |
public static $indic_category_char = array( |
| 33 |
'x', |
| 34 |
'C', |
| 35 |
'V', |
| 36 |
'N', |
| 37 |
'H', |
| 38 |
'Z', |
| 39 |
'J', |
| 40 |
'M', |
| 41 |
'S', |
| 42 |
'v', |
| 43 |
'A', /* Spec gives Andutta U+0952 as OT_A. However, testing shows that Uniscribe |
| 44 |
* treats U+0951..U+0952 all as OT_VD - see set_indic_properties */ |
| 45 |
's', |
| 46 |
'D', |
| 47 |
'F', /* Register shift Khmer only */ |
| 48 |
'G', /* Khmer only */ |
| 49 |
'r', /* 0D4E (dot reph) only one in Malayalam */ |
| 50 |
'R', |
| 51 |
'm', /* Consonant medial only used in Indic 0A75 in Gurmukhi (0A00..0A7F) : also in Lao, Myanmar, Tai Tham, Javanese & Cham */ |
| 52 |
); |
| 53 |
|
| 54 |
/* Visual positions in a syllable from left to right. */ |
| 55 |
/* FROM hb-ot-shape-complex-indic-private.hh */ |
| 56 |
|
| 57 |
// indic_position |
| 58 |
const POS_START = 0; |
| 59 |
|
| 60 |
const POS_RA_TO_BECOME_REPH = 1; |
| 61 |
const POS_PRE_M = 2; |
| 62 |
const POS_PRE_C = 3; |
| 63 |
|
| 64 |
const POS_BASE_C = 4; |
| 65 |
const POS_AFTER_MAIN = 5; |
| 66 |
|
| 67 |
const POS_ABOVE_C = 6; |
| 68 |
|
| 69 |
const POS_BEFORE_SUB = 7; |
| 70 |
const POS_BELOW_C = 8; |
| 71 |
const POS_AFTER_SUB = 9; |
| 72 |
|
| 73 |
const POS_BEFORE_POST = 10; |
| 74 |
const POS_POST_C = 11; |
| 75 |
const POS_AFTER_POST = 12; |
| 76 |
|
| 77 |
const POS_FINAL_C = 13; |
| 78 |
const POS_SMVD = 14; |
| 79 |
|
| 80 |
const POS_END = 15; |
| 81 |
|
| 82 |
/* |
| 83 |
* Basic features. |
| 84 |
* These features are applied in order, one at a time, after initial_reordering. |
| 85 |
*/ |
| 86 |
/* |
| 87 |
* Must be in the same order as the indic_features array. Ones starting with _ are F_GLOBAL |
| 88 |
* Ones without the _ are only applied where the mask says! |
| 89 |
*/ |
| 90 |
|
| 91 |
const _NUKT = 0; |
| 92 |
const _AKHN = 1; |
| 93 |
const RPHF = 2; |
| 94 |
const _RKRF = 3; |
| 95 |
const PREF = 4; |
| 96 |
const BLWF = 5; |
| 97 |
const HALF = 6; |
| 98 |
const ABVF = 7; |
| 99 |
const PSTF = 8; |
| 100 |
const CFAR = 9; // Khmer only |
| 101 |
const _VATU = 10; |
| 102 |
const _CJCT = 11; |
| 103 |
const INIT = 12; |
| 104 |
|
| 105 |
public static function set_indic_properties(&$info, $scriptblock) |
| 106 |
{ |
| 107 |
$u = $info['uni']; |
| 108 |
$type = self::indic_get_categories($u); |
| 109 |
$cat = ($type & 0x7F); |
| 110 |
$pos = ($type >> 8); |
| 111 |
|
| 112 |
/* |
| 113 |
* Re-assign category |
| 114 |
*/ |
| 115 |
|
| 116 |
if ($u == 0x17D1) |
| 117 |
$cat = self::OT_X; |
| 118 |
|
| 119 |
if ($cat == self::OT_X && self::in_range($u, 0x17CB, 0x17D3)) { /* Khmer Various signs */ |
| 120 |
/* These are like Top Matras. */ |
| 121 |
$cat = self::OT_M; |
| 122 |
$pos = self::POS_ABOVE_C; |
| 123 |
} |
| 124 |
|
| 125 |
if ($u == 0x17C6) |
| 126 |
$cat = self::OT_N; /* Khmer Bindu doesn't like to be repositioned. */ |
| 127 |
|
| 128 |
if ($u == 0x17D2) |
| 129 |
$cat = self::OT_Coeng; /* Khmer coeng */ |
| 130 |
|
| 131 |
/* The spec says U+0952 is OT_A. However, testing shows that Uniscribe |
| 132 |
* treats U+0951..U+0952 all as OT_VD. |
| 133 |
* TESTS: |
| 134 |
* U+092E,U+0947,U+0952 |
| 135 |
* U+092E,U+0952,U+0947 |
| 136 |
* U+092E,U+0947,U+0951 |
| 137 |
* U+092E,U+0951,U+0947 |
| 138 |
* */ |
| 139 |
//if ($u == 0x0952) $cat = self::OT_A; |
| 140 |
if (self::in_range($u, 0x0951, 0x0954)) |
| 141 |
$cat = self::OT_VD; |
| 142 |
|
| 143 |
if ($u == 0x200C) |
| 144 |
$cat = self::OT_ZWNJ; |
| 145 |
else if ($u == 0x200D) |
| 146 |
$cat = self::OT_ZWJ; |
| 147 |
else if ($u == 0x25CC) |
| 148 |
$cat = self::OT_DOTTEDCIRCLE; |
| 149 |
else if ($u == 0x0A71) |
| 150 |
$cat = self::OT_SM; /* GURMUKHI ADDAK. More like consonant medial. like 0A75. */ |
| 151 |
|
| 152 |
if ($cat == self::OT_Repha) { |
| 153 |
/* There are two kinds of characters marked as Repha: |
| 154 |
* - The ones that are GenCat=Mn are already positioned visually, ie. after base. (eg. Khmer) |
| 155 |
* - The ones that are GenCat=Lo is encoded logically, ie. beginning of syllable. (eg. Malayalam) |
| 156 |
* |
| 157 |
* We recategorize the first kind to look like a Nukta and attached to the base directly. |
| 158 |
*/ |
| 159 |
if ($info['general_category'] == UCDN::UNICODE_GENERAL_CATEGORY_NON_SPACING_MARK) |
| 160 |
$cat = self::OT_N; |
| 161 |
} |
| 162 |
|
| 163 |
/* |
| 164 |
* Re-assign position. |
| 165 |
*/ |
| 166 |
|
| 167 |
if ((self::FLAG($cat) & (self::FLAG(self::OT_C) | self::FLAG(self::OT_CM) | self::FLAG(self::OT_Ra) | self::FLAG(self::OT_V) | self::FLAG(self::OT_NBSP) | self::FLAG(self::OT_DOTTEDCIRCLE)))) { // = CONSONANT_FLAGS like is_consonant |
| 168 |
if ($scriptblock == UCDN::SCRIPT_KHMER) |
| 169 |
$pos = self::POS_BELOW_C; /* Khmer differs from Indic here. */ |
| 170 |
else |
| 171 |
$pos = self::POS_BASE_C; /* Will recategorize later based on font lookups. */ |
| 172 |
|
| 173 |
if (self::is_ra($u)) |
| 174 |
$cat = self::OT_Ra; |
| 175 |
} |
| 176 |
else if ($cat == self::OT_M) { |
| 177 |
$pos = self::matra_position($u, $pos); |
| 178 |
} else if ($cat == self::OT_SM || $cat == self::OT_VD) { |
| 179 |
$pos = self::POS_SMVD; |
| 180 |
} |
| 181 |
|
| 182 |
if ($u == 0x0B01) |
| 183 |
$pos = self::POS_BEFORE_SUB; /* Oriya Bindu is BeforeSub in the spec. */ |
| 184 |
|
| 185 |
$info['indic_category'] = $cat; |
| 186 |
$info['indic_position'] = $pos; |
| 187 |
} |
| 188 |
|
| 189 |
// syllable_type |
| 190 |
const CONSONANT_SYLLABLE = 0; |
| 191 |
const VOWEL_SYLLABLE = 1; |
| 192 |
const STANDALONE_CLUSTER = 2; |
| 193 |
const BROKEN_CLUSTER = 3; |
| 194 |
const NON_INDIC_CLUSTER = 4; |
| 195 |
|
| 196 |
public static function set_syllables(&$o, $s, &$broken_syllables) |
| 197 |
{ |
| 198 |
$ptr = 0; |
| 199 |
$syllable_serial = 1; |
| 200 |
$broken_syllables = false; |
| 201 |
|
| 202 |
while ($ptr < strlen($s)) { |
| 203 |
$match = ''; |
| 204 |
$syllable_length = 1; |
| 205 |
$syllable_type = self::NON_INDIC_CLUSTER; |
| 206 |
// CONSONANT_SYLLABLE Consonant syllable |
| 207 |
// From OT spec: |
| 208 |
if (preg_match('/^([CR]m*[N]?(H[ZJ]?|[ZJ]H))*[CR]m*[N]?[A]?(H[ZJ]?|[M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma)) { |
| 209 |
// From HarfBuzz: |
| 210 |
//if (preg_match('/^r?([CR]J?(Z?[N]{0,2})?[ZJ]?H(J[N]?)?){0,4}[CR]J?(Z?[N]{0,2})?A?((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})?(S[Z]?)?[v]{0,2}/', substr($s,$ptr), $ma)) { |
| 211 |
$syllable_length = strlen($ma[0]); |
| 212 |
$syllable_type = self::CONSONANT_SYLLABLE; |
| 213 |
} |
| 214 |
// VOWEL_SYLLABLE Vowel-based syllable |
| 215 |
// From OT spec: |
| 216 |
else if (preg_match('/^(RH|r)?V[N]?([ZJ]?H[CR]m*|J[CR]m*)?([M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma)) { |
| 217 |
// From HarfBuzz: |
| 218 |
//else if (preg_match('/^(RH|r)?V(Z?[N]{0,2})?(J|([ZJ]?H(J[N]?)?[CR]J?(Z?[N]{0,2})?){0,4}((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})?(S[Z]?)?[v]{0,2})/', substr($s,$ptr), $ma)) { |
| 219 |
$syllable_length = strlen($ma[0]); |
| 220 |
$syllable_type = self::VOWEL_SYLLABLE; |
| 221 |
} |
| 222 |
|
| 223 |
/* Apply only if it's a word start. */ |
| 224 |
// STANDALONE_CLUSTER Stand Alone syllable at start of word |
| 225 |
// From OT spec: |
| 226 |
else if (($ptr == 0 || |
| 227 |
$o[$ptr - 1]['general_category'] < UCDN::UNICODE_GENERAL_CATEGORY_LOWERCASE_LETTER || |
| 228 |
$o[$ptr - 1]['general_category'] > UCDN::UNICODE_GENERAL_CATEGORY_NON_SPACING_MARK |
| 229 |
) && (preg_match('/^(RH|r)?[sD][N]?([ZJ]?H[CR]m*)?([M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma))) { |
| 230 |
// From HarfBuzz: |
| 231 |
// && (preg_match('/^(RH|r)?[sD](Z?[N]{0,2})?(([ZJ]?H(J[N]?)?)[CR]J?(Z?[N]{0,2})?){0,4}((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})?(S[Z]?)?[v]{0,2}/', substr($s,$ptr), $ma)) { |
| 232 |
$syllable_length = strlen($ma[0]); |
| 233 |
$syllable_type = self::STANDALONE_CLUSTER; |
| 234 |
} |
| 235 |
|
| 236 |
// BROKEN_CLUSTER syllable |
| 237 |
else if (preg_match('/^(RH|r)?[N]?([ZJ]?H[CR])?([M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma)) { |
| 238 |
// From HarfBuzz: |
| 239 |
//else if (preg_match('/^(RH|r)?(Z?[N]{0,2})?(([ZJ]?H(J[N]?)?)[CR]J?(Z?[N]{0,2})?){0,4}((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})(S[Z]?)?[v]{0,2}/', substr($s,$ptr), $ma)) { |
| 240 |
if (strlen($ma[0])) { // May match blank |
| 241 |
$syllable_length = strlen($ma[0]); |
| 242 |
$syllable_type = self::BROKEN_CLUSTER; |
| 243 |
$broken_syllables = true; |
| 244 |
} |
| 245 |
} |
| 246 |
|
| 247 |
for ($i = $ptr; $i < $ptr + $syllable_length; $i++) { |
| 248 |
$o[$i]['syllable'] = ($syllable_serial << 4) | $syllable_type; |
| 249 |
} |
| 250 |
$ptr += $syllable_length; |
| 251 |
$syllable_serial++; |
| 252 |
if ($syllable_serial == 16) |
| 253 |
$syllable_serial = 1; |
| 254 |
} |
| 255 |
} |
| 256 |
|
| 257 |
public static function set_syllables_sinhala(&$o, $s, &$broken_syllables) |
| 258 |
{ |
| 259 |
$ptr = 0; |
| 260 |
$syllable_serial = 1; |
| 261 |
$broken_syllables = false; |
| 262 |
|
| 263 |
while ($ptr < strlen($s)) { |
| 264 |
$match = ''; |
| 265 |
$syllable_length = 1; |
| 266 |
$syllable_type = self::NON_INDIC_CLUSTER; |
| 267 |
// CONSONANT_SYLLABLE Consonant syllable |
| 268 |
// From OT spec: |
| 269 |
if (preg_match('/^([CR]HJ|[CR]JH){0,8}[CR][HM]{0,3}[S]{0,1}/', substr($s, $ptr), $ma)) { |
| 270 |
$syllable_length = strlen($ma[0]); |
| 271 |
$syllable_type = self::CONSONANT_SYLLABLE; |
| 272 |
} |
| 273 |
// VOWEL_SYLLABLE Vowel-based syllable |
| 274 |
// From OT spec: |
| 275 |
else if (preg_match('/^V[S]{0,1}/', substr($s, $ptr), $ma)) { |
| 276 |
$syllable_length = strlen($ma[0]); |
| 277 |
$syllable_type = self::VOWEL_SYLLABLE; |
| 278 |
} |
| 279 |
|
| 280 |
for ($i = $ptr; $i < $ptr + $syllable_length; $i++) { |
| 281 |
$o[$i]['syllable'] = ($syllable_serial << 4) | $syllable_type; |
| 282 |
} |
| 283 |
$ptr += $syllable_length; |
| 284 |
$syllable_serial++; |
| 285 |
if ($syllable_serial == 16) |
| 286 |
$syllable_serial = 1; |
| 287 |
} |
| 288 |
} |
| 289 |
|
| 290 |
public static function set_syllables_khmer(&$o, $s, &$broken_syllables) |
| 291 |
{ |
| 292 |
$ptr = 0; |
| 293 |
$syllable_serial = 1; |
| 294 |
$broken_syllables = false; |
| 295 |
|
| 296 |
while ($ptr < strlen($s)) { |
| 297 |
$match = ''; |
| 298 |
$syllable_length = 1; |
| 299 |
$syllable_type = self::NON_INDIC_CLUSTER; |
| 300 |
// CONSONANT_SYLLABLE Consonant syllable |
| 301 |
if (preg_match('/^r?([CR]J?((Z?F)?[N]{0,2})?[ZJ]?G(JN?)?){0,4}[CR]J?((Z?F)?[N]{0,2})?A?((([ZJ]?G(JN?)?)|GZ)|(GJ)?([ZJ]{0,3}MN?(H|JHJR)?){0,4})?(G([CR]J?((Z?F)?[N]{0,2})?|V))?(SZ?)?[v]{0,2}/', substr($s, $ptr), $ma)) { |
| 302 |
$syllable_length = strlen($ma[0]); |
| 303 |
$syllable_type = self::CONSONANT_SYLLABLE; |
| 304 |
} |
| 305 |
// VOWEL_SYLLABLE Vowel-based syllable |
| 306 |
else if (preg_match('/^(RH|r)?V((Z?F)?[N]{0,2})?(J|([ZJ]?G(JN?)?[CR]J?((Z?F)?[N]{0,2})?){0,4}((([ZJ]?G(JN?)?)|GZ)|(GJ)?([ZJ]{0,3}MN?(H|JHJR)?){0,4})?(G([CR]J?((Z?F)?[N]{0,2})?|V))?(SZ?)?[v]{0,2})/', substr($s, $ptr), $ma)) { |
| 307 |
$syllable_length = strlen($ma[0]); |
| 308 |
$syllable_type = self::VOWEL_SYLLABLE; |
| 309 |
} |
| 310 |
|
| 311 |
|
| 312 |
// BROKEN_CLUSTER syllable |
| 313 |
else if (preg_match('/^(RH|r)?((Z?F)?[N]{0,2})?(([ZJ]?G(JN?)?)[CR]J?((Z?F)?[N]{0,2})?){0,4}((([ZJ]?G(JN?)?)|GZ)|(GJ)?([ZJ]{0,3}MN?(H|JHJR)?){0,4})(G([CR]J?((Z?F)?[N]{0,2})?|V))?(SZ?)?[v]{0,2}/', substr($s, $ptr), $ma)) { |
| 314 |
if (strlen($ma[0])) { // May match blank |
| 315 |
$syllable_length = strlen($ma[0]); |
| 316 |
$syllable_type = self::BROKEN_CLUSTER; |
| 317 |
$broken_syllables = true; |
| 318 |
} |
| 319 |
} |
| 320 |
|
| 321 |
for ($i = $ptr; $i < $ptr + $syllable_length; $i++) { |
| 322 |
$o[$i]['syllable'] = ($syllable_serial << 4) | $syllable_type; |
| 323 |
} |
| 324 |
$ptr += $syllable_length; |
| 325 |
$syllable_serial++; |
| 326 |
if ($syllable_serial == 16) |
| 327 |
$syllable_serial = 1; |
| 328 |
} |
| 329 |
} |
| 330 |
|
| 331 |
public static function initial_reordering(&$info, $GSUBdata, $broken_syllables, $indic_config, $scriptblock, $is_old_spec, $dottedcircle) |
| 332 |
{ |
| 333 |
|
| 334 |
self::update_consonant_positions($info, $GSUBdata); |
| 335 |
|
| 336 |
if ($broken_syllables && $dottedcircle) { |
| 337 |
self::insert_dotted_circles($info, $dottedcircle); |
| 338 |
} |
| 339 |
|
| 340 |
$count = count($info); |
| 341 |
if (!$count) |
| 342 |
return; |
| 343 |
$last = 0; |
| 344 |
$last_syllable = $info[0]['syllable']; |
| 345 |
for ($i = 1; $i < $count; $i++) { |
| 346 |
if ($last_syllable != $info[$i]['syllable']) { |
| 347 |
self::initial_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $i); |
| 348 |
$last = $i; |
| 349 |
$last_syllable = $info[$last]['syllable']; |
| 350 |
} |
| 351 |
} |
| 352 |
self::initial_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $count); |
| 353 |
} |
| 354 |
|
| 355 |
public static function update_consonant_positions(&$info, $GSUBdata) |
| 356 |
{ |
| 357 |
$count = count($info); |
| 358 |
for ($i = 0; $i < $count; $i++) { |
| 359 |
if ($info[$i]['indic_position'] == self::POS_BASE_C) { |
| 360 |
$c = $info[$i]['uni']; |
| 361 |
// If would substitute... |
| 362 |
if (isset($GSUBdata['pref'][$c])) { |
| 363 |
$info[$i]['indic_position'] = self::POS_POST_C; |
| 364 |
} else if (isset($GSUBdata['blwf'][$c])) { |
| 365 |
$info[$i]['indic_position'] = self::POS_BELOW_C; |
| 366 |
} else if (isset($GSUBdata['pstf'][$c])) { |
| 367 |
$info[$i]['indic_position'] = self::POS_POST_C; |
| 368 |
} |
| 369 |
} |
| 370 |
} |
| 371 |
} |
| 372 |
|
| 373 |
public static function insert_dotted_circles(&$info, $dottedcircle) |
| 374 |
{ |
| 375 |
$idx = 0; |
| 376 |
$last_syllable = 0; |
| 377 |
while ($idx < count($info)) { |
| 378 |
$syllable = $info[$idx]['syllable']; |
| 379 |
$syllable_type = ($syllable & 0x0F); |
| 380 |
if ($last_syllable != $syllable && $syllable_type == self::BROKEN_CLUSTER) { |
| 381 |
$last_syllable = $syllable; |
| 382 |
|
| 383 |
$dottedcircle[0]['syllable'] = $info[$idx]['syllable']; |
| 384 |
|
| 385 |
/* Insert dottedcircle after possible Repha. */ |
| 386 |
while ($idx < count($info) && $last_syllable == $info[$idx]['syllable'] && $info[$idx]['indic_category'] == self::OT_Repha) |
| 387 |
$idx++; |
| 388 |
array_splice($info, $idx, 0, $dottedcircle); |
| 389 |
} else { |
| 390 |
$idx++; |
| 391 |
} |
| 392 |
} |
| 393 |
|
| 394 |
// I am not sue how this code below got in here, since $idx should now be > count($info) and thus invalid. |
| 395 |
// In case I am missing something(!) I'll leave a warning here for now: |
| 396 |
if (isset($info[$idx])) { |
| 397 |
throw new MpdfException('Unexpected error occured in Indic processing'); |
| 398 |
} |
| 399 |
// In case of final bloken cluster... |
| 400 |
//$syllable = $info[$idx]['syllable']; |
| 401 |
//$syllable_type = ($syllable & 0x0F); |
| 402 |
//if ($last_syllable != $syllable && $syllable_type == self::BROKEN_CLUSTER) { |
| 403 |
// $dottedcircle[0]['syllable'] = $info[$idx]['syllable']; |
| 404 |
// array_splice($info, $idx, 0, $dottedcircle); |
| 405 |
//} |
| 406 |
} |
| 407 |
|
| 408 |
/* Rules from: |
| 409 |
* https://www.microsoft.com/typography/otfntdev/devanot/shaping.aspx */ |
| 410 |
|
| 411 |
public static function initial_reordering_syllable(&$info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $start, $end) |
| 412 |
{ |
| 413 |
/* vowel_syllable: We made the vowels look like consonants. So uses the consonant logic! */ |
| 414 |
/* broken_cluster: We already inserted dotted-circles, so just call the standalone_cluster. */ |
| 415 |
/* standalone_cluster: We treat NBSP/dotted-circle as if they are consonants, so we should just chain. */ |
| 416 |
|
| 417 |
$syllable_type = ($info[$start]['syllable'] & 0x0F); |
| 418 |
if ($syllable_type == self::NON_INDIC_CLUSTER) { |
| 419 |
return; |
| 420 |
} |
| 421 |
if ($syllable_type == self::BROKEN_CLUSTER || $syllable_type == self::STANDALONE_CLUSTER) { |
| 422 |
//if ($uniscribe_bug_compatible) { |
| 423 |
/* For dotted-circle, this is what Uniscribe does: |
| 424 |
* If dotted-circle is the last glyph, it just does nothing. |
| 425 |
* i.e. It doesn't form Reph. */ |
| 426 |
if ($info[$end - 1]['indic_category'] == self::OT_DOTTEDCIRCLE) { |
| 427 |
return; |
| 428 |
} |
| 429 |
} |
| 430 |
|
| 431 |
/* 1. Find base consonant: |
| 432 |
* |
| 433 |
* The shaping engine finds the base consonant of the syllable, using the |
| 434 |
* following algorithm: starting from the end of the syllable, move backwards |
| 435 |
* until a consonant is found that does not have a below-base or post-base |
| 436 |
* form (post-base forms have to follow below-base forms), or that is not a |
| 437 |
* pre-base reordering Ra, or arrive at the first consonant. The consonant |
| 438 |
* stopped at will be the base. |
| 439 |
* |
| 440 |
* o If the syllable starts with Ra + Halant (in a script that has Reph) |
| 441 |
* and has more than one consonant, Ra is excluded from candidates for |
| 442 |
* base consonants. |
| 443 |
*/ |
| 444 |
|
| 445 |
$base = $end; |
| 446 |
$has_reph = false; |
| 447 |
$limit = $start; |
| 448 |
|
| 449 |
if ($scriptblock != UCDN::SCRIPT_KHMER) { |
| 450 |
/* -> If the syllable starts with Ra + Halant (in a script that has Reph) |
| 451 |
* and has more than one consonant, Ra is excluded from candidates for |
| 452 |
* base consonants. */ |
| 453 |
if (count($GSUBdata['rphf']) /* ?? $indic_plan->mask_array[RPHF] */ && $start + 3 <= $end && |
| 454 |
( |
| 455 |
($indic_config[4] == self::REPH_MODE_IMPLICIT && !self::is_joiner($info[$start + 2])) || |
| 456 |
($indic_config[4] == self::REPH_MODE_EXPLICIT && $info[$start + 2]['indic_category'] == self::OT_ZWJ) |
| 457 |
)) { |
| 458 |
/* See if it matches the 'rphf' feature. */ |
| 459 |
//$glyphs = array($info[$start]['uni'], $info[$start + 1]['uni']); |
| 460 |
//if ($indic_plan->rphf->would_substitute ($glyphs, count($glyphs), true, face)) { |
| 461 |
if (isset($GSUBdata['rphf'][$info[$start]['uni']]) && self::is_halant_or_coeng($info[$start + 1])) { |
| 462 |
$limit += 2; |
| 463 |
while ($limit < $end && self::is_joiner($info[$limit])) |
| 464 |
$limit++; |
| 465 |
$base = $start; |
| 466 |
$has_reph = true; |
| 467 |
} |
| 468 |
} else if ($indic_config[4] == self::REPH_MODE_LOG_REPHA && $info[$start]['indic_category'] == self::OT_Repha) { |
| 469 |
$limit += 1; |
| 470 |
while ($limit < $end && self::is_joiner($info[$limit])) |
| 471 |
$limit++; |
| 472 |
$base = $start; |
| 473 |
$has_reph = true; |
| 474 |
} |
| 475 |
} |
| 476 |
|
| 477 |
switch ($indic_config[2]) { // base_pos |
| 478 |
case self::BASE_POS_LAST: |
| 479 |
/* -> starting from the end of the syllable, move backwards */ |
| 480 |
$i = $end; |
| 481 |
$seen_below = false; |
| 482 |
do { |
| 483 |
$i--; |
| 484 |
/* -> until a consonant is found */ |
| 485 |
if (self::is_consonant($info[$i])) { |
| 486 |
/* -> that does not have a below-base or post-base form |
| 487 |
* (post-base forms have to follow below-base forms), */ |
| 488 |
if ($info[$i]['indic_position'] != self::POS_BELOW_C && ($info[$i]['indic_position'] != self::POS_POST_C || $seen_below)) { |
| 489 |
$base = $i; |
| 490 |
break; |
| 491 |
} |
| 492 |
if ($info[$i]['indic_position'] == self::POS_BELOW_C) |
| 493 |
$seen_below = true; |
| 494 |
|
| 495 |
/* -> or that is not a pre-base reordering Ra, |
| 496 |
* |
| 497 |
* IMPLEMENTATION NOTES: |
| 498 |
* |
| 499 |
* Our pre-base reordering Ra's are marked POS_POST_C, so will be skipped |
| 500 |
* by the logic above already. |
| 501 |
*/ |
| 502 |
|
| 503 |
/* -> or arrive at the first consonant. The consonant stopped at will |
| 504 |
* be the base. */ |
| 505 |
$base = $i; |
| 506 |
} |
| 507 |
else { |
| 508 |
/* A ZWJ after a Halant stops the base search, and requests an explicit |
| 509 |
* half form. |
| 510 |
* [A ZWJ before a Halant, requests a subjoined form instead, and hence |
| 511 |
* search continues. This is particularly important for Bengali |
| 512 |
* sequence Ra,H,Ya that should form Ya-Phalaa by subjoining Ya] */ |
| 513 |
if ($start < $i && $info[$i]['indic_category'] == self::OT_ZWJ && $info[$i - 1]['indic_category'] == self::OT_H) { |
| 514 |
if (!defined("OMIT_INDIC_FIX_1") || OMIT_INDIC_FIX_1 != 1) { |
| 515 |
$base = $i; |
| 516 |
} // INDIC_FIX_1 |
| 517 |
break; |
| 518 |
} |
| 519 |
// ZKI8 |
| 520 |
if ($start < $i && $info[$i]['indic_category'] == self::OT_ZWNJ) { |
| 521 |
break; |
| 522 |
} |
| 523 |
} |
| 524 |
} while ($i > $limit); |
| 525 |
break; |
| 526 |
|
| 527 |
case self::BASE_POS_FIRST: |
| 528 |
/* In scripts without half forms (eg. Khmer), the first consonant is always the base. */ |
| 529 |
|
| 530 |
if (!$has_reph) |
| 531 |
$base = $limit; |
| 532 |
|
| 533 |
/* Find the last base consonant that is not blocked by ZWJ. If there is |
| 534 |
* a ZWJ right before a base consonant, that would request a subjoined form. */ |
| 535 |
for ($i = $limit; $i < $end; $i++) { |
| 536 |
if (self::is_consonant($info[$i]) && $info[$i]['indic_position'] == self::POS_BASE_C) { |
| 537 |
if ($limit < $i && $info[$i - 1]['indic_category'] == self::OT_ZWJ) |
| 538 |
break; |
| 539 |
else |
| 540 |
$base = $i; |
| 541 |
} |
| 542 |
} |
| 543 |
|
| 544 |
/* Mark all subsequent consonants as below. */ |
| 545 |
for ($i = $base + 1; $i < $end; $i++) { |
| 546 |
if (self::is_consonant($info[$i]) && $info[$i]['indic_position'] == self::POS_BASE_C) |
| 547 |
$info[$i]['indic_position'] = self::POS_BELOW_C; |
| 548 |
} |
| 549 |
break; |
| 550 |
//default: |
| 551 |
//assert (false); |
| 552 |
/* fallthrough */ |
| 553 |
} |
| 554 |
|
| 555 |
/* -> If the syllable starts with Ra + Halant (in a script that has Reph) |
| 556 |
* and has more than one consonant, Ra is excluded from candidates for |
| 557 |
* base consonants. |
| 558 |
* |
| 559 |
* Only do this for unforced Reph. (ie. not for Ra,H,ZWJ. */ |
| 560 |
if ($scriptblock != UCDN::SCRIPT_KHMER) { |
| 561 |
if ($has_reph && $base == $start && $limit - $base <= 2) { |
| 562 |
/* Have no other consonant, so Reph is not formed and Ra becomes base. */ |
| 563 |
$has_reph = false; |
| 564 |
} |
| 565 |
} |
| 566 |
|
| 567 |
/* 2. Decompose and reorder Matras: |
| 568 |
* |
| 569 |
* Each matra and any syllable modifier sign in the cluster are moved to the |
| 570 |
* appropriate position relative to the consonant(s) in the cluster. The |
| 571 |
* shaping engine decomposes two- or three-part matras into their constituent |
| 572 |
* parts before any repositioning. Matra characters are classified by which |
| 573 |
* consonant in a conjunct they have affinity for and are reordered to the |
| 574 |
* following positions: |
| 575 |
* |
| 576 |
* o Before first half form in the syllable |
| 577 |
* o After subjoined consonants |
| 578 |
* o After post-form consonant |
| 579 |
* o After main consonant (for above marks) |
| 580 |
* |
| 581 |
* IMPLEMENTATION NOTES: |
| 582 |
* |
| 583 |
* The normalize() routine has already decomposed matras for us, so we don't |
| 584 |
* need to worry about that. |
| 585 |
*/ |
| 586 |
|
| 587 |
|
| 588 |
/* 3. Reorder marks to canonical order: |
| 589 |
* |
| 590 |
* Adjacent nukta and halant or nukta and vedic sign are always repositioned |
| 591 |
* if necessary, so that the nukta is first. |
| 592 |
* |
| 593 |
* IMPLEMENTATION NOTES: |
| 594 |
* |
| 595 |
* Use the combining Class from Unicode categories? to bubble_sort. |
| 596 |
*/ |
| 597 |
|
| 598 |
/* Reorder characters */ |
| 599 |
|
| 600 |
for ($i = $start; $i < $base; $i++) |
| 601 |
$info[$i]['indic_position'] = min(self::POS_PRE_C, $info[$i]['indic_position']); |
| 602 |
|
| 603 |
if ($base < $end) |
| 604 |
$info[$base]['indic_position'] = self::POS_BASE_C; |
| 605 |
|
| 606 |
/* Mark final consonants. A final consonant is one appearing after a matra, |
| 607 |
* ? only in Khmer. */ |
| 608 |
for ($i = $base + 1; $i < $end; $i++) |
| 609 |
if ($info[$i]['indic_category'] == self::OT_M) { |
| 610 |
for ($j = $i + 1; $j < $end; $j++) |
| 611 |
if (self::is_consonant($info[$j])) { |
| 612 |
$info[$j]['indic_position'] = self::POS_FINAL_C; |
| 613 |
break; |
| 614 |
} |
| 615 |
break; |
| 616 |
} |
| 617 |
|
| 618 |
/* Handle beginning Ra */ |
| 619 |
if ($scriptblock != UCDN::SCRIPT_KHMER) { |
| 620 |
if ($has_reph) |
| 621 |
$info[$start]['indic_position'] = self::POS_RA_TO_BECOME_REPH; |
| 622 |
} |
| 623 |
|
| 624 |
|
| 625 |
/* For old-style Indic script tags, move the first post-base Halant after |
| 626 |
* last consonant. Only do this if there is *not* a Halant after last |
| 627 |
* consonant. Otherwise it becomes messy. */ |
| 628 |
if ($is_old_spec) { |
| 629 |
for ($i = $base + 1; $i < $end; $i++) { |
| 630 |
if ($info[$i]['indic_category'] == self::OT_H) { |
| 631 |
for ($j = $end - 1; $j > $i; $j--) { |
| 632 |
if (self::is_consonant($info[$j]) || $info[$j]['indic_category'] == self::OT_H) { |
| 633 |
break; |
| 634 |
} |
| 635 |
} |
| 636 |
if ($info[$j]['indic_category'] != self::OT_H && $j > $i) { |
| 637 |
/* Move Halant to after last consonant. */ |
| 638 |
self::_move_info_pos($info, $i, $j + 1); |
| 639 |
} |
| 640 |
break; |
| 641 |
} |
| 642 |
} |
| 643 |
} |
| 644 |
|
| 645 |
/* Attach misc marks to previous char to move with them. */ |
| 646 |
$last_pos = self::POS_START; |
| 647 |
for ($i = $start; $i < $end; $i++) { |
| 648 |
if ((self::FLAG($info[$i]['indic_category']) & (self::FLAG(self::OT_ZWJ) | self::FLAG(self::OT_ZWNJ) | self::FLAG(self::OT_N) | self::FLAG(self::OT_RS) | self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng) ))) { |
| 649 |
$info[$i]['indic_position'] = $last_pos; |
| 650 |
if ($info[$i]['indic_category'] == self::OT_H && $info[$i]['indic_position'] == self::POS_PRE_M) { |
| 651 |
/* |
| 652 |
* Uniscribe doesn't move the Halant with Left Matra. |
| 653 |
* TEST: U+092B,U+093F,U+094DE |
| 654 |
* We follow. This is important for the Sinhala |
| 655 |
* U+0DDA split matra since it decomposes to U+0DD9,U+0DCA |
| 656 |
* where U+0DD9 is a left matra and U+0DCA is the virama. |
| 657 |
* We don't want to move the virama with the left matra. |
| 658 |
* TEST: U+0D9A,U+0DDA |
| 659 |
*/ |
| 660 |
for ($j = $i; $j > $start; $j--) |
| 661 |
if ($info[$j - 1]['indic_position'] != self::POS_PRE_M) { |
| 662 |
$info[$i]['indic_position'] = $info[$j - 1]['indic_position']; |
| 663 |
break; |
| 664 |
} |
| 665 |
} |
| 666 |
} else if ($info[$i]['indic_position'] != self::POS_SMVD) { |
| 667 |
$last_pos = $info[$i]['indic_position']; |
| 668 |
} |
| 669 |
} |
| 670 |
|
| 671 |
/* Re-attach ZWJ, ZWNJ, and halant to next char, for after-base consonants. */ |
| 672 |
$last_halant = $end; |
| 673 |
for ($i = $base + 1; $i < $end; $i++) { |
| 674 |
if (self::is_halant_or_coeng($info[$i])) |
| 675 |
$last_halant = $i; |
| 676 |
else if (self::is_consonant($info[$i])) { |
| 677 |
for ($j = $last_halant; $j < $i; $j++) |
| 678 |
if ($info[$j]['indic_position'] != self::POS_SMVD) |
| 679 |
$info[$j]['indic_position'] = $info[$i]['indic_position']; |
| 680 |
} |
| 681 |
} |
| 682 |
|
| 683 |
|
| 684 |
if ($scriptblock == UCDN::SCRIPT_KHMER) { |
| 685 |
/* KHMER_FIX_2 */ |
| 686 |
/* Move Coeng+RO (Halant,Ra) sequence before base consonant. */ |
| 687 |
for ($i = $base + 1; $i < $end; $i++) { |
| 688 |
if (self::is_halant_or_coeng($info[$i]) && self::is_ra($info[$i + 1]['uni'])) { |
| 689 |
$info[$i]['indic_position'] = self::POS_PRE_C; |
| 690 |
$info[$i + 1]['indic_position'] = self::POS_PRE_C; |
| 691 |
break; |
| 692 |
} |
| 693 |
} |
| 694 |
} |
| 695 |
|
| 696 |
|
| 697 |
/* |
| 698 |
if (!defined("OMIT_INDIC_FIX_2") || OMIT_INDIC_FIX_2 != 1) { |
| 699 |
// INDIC_FIX_2 |
| 700 |
$ZWNJ_found = false; |
| 701 |
$POST_ZWNJ_c_found = false; |
| 702 |
for ($i = $base + 1; $i < $end; $i++) { |
| 703 |
if ($info[$i]['indic_category'] == self::OT_ZWNJ) { $ZWNJ_found = true; } |
| 704 |
else if ($ZWNJ_found && $info[$i]['indic_category'] == self::OT_C) { $POST_ZWNJ_c_found = true; } |
| 705 |
else if ($POST_ZWNJ_c_found && $info[$i]['indic_position'] == self::POS_BEFORE_SUB) { $info[$i]['indic_position'] = self::POS_AFTER_SUB; } |
| 706 |
} |
| 707 |
} |
| 708 |
*/ |
| 709 |
|
| 710 |
/* Setup masks now */ |
| 711 |
for ($i = $start; $i < $end; $i++) { |
| 712 |
$info[$i]['mask'] = 0; |
| 713 |
} |
| 714 |
|
| 715 |
|
| 716 |
if ($scriptblock == UCDN::SCRIPT_KHMER) { |
| 717 |
/* Find a Coeng+RO (Halant,Ra) sequence and mark it for pre-base processing. */ |
| 718 |
$mask = self::FLAG(self::PREF); |
| 719 |
for ($i = $base; $i < $end - 1; $i++) { /* KHMER_FIX_1 From $start (not base) */ |
| 720 |
if (self::is_halant_or_coeng($info[$i]) && self::is_ra($info[$i + 1]['uni'])) { |
| 721 |
|
| 722 |
$info[$i]['mask'] |= self::FLAG(self::PREF); |
| 723 |
$info[$i + 1]['mask'] |= self::FLAG(self::PREF); |
| 724 |
|
| 725 |
/* Mark the subsequent stuff with 'cfar'. Used in Khmer. |
| 726 |
* Read the feature spec. |
| 727 |
* This allows distinguishing the following cases with MS Khmer fonts: |
| 728 |
* U+1784,U+17D2,U+179A,U+17D2,U+1782 [C+Coeng+RO+Coeng+C] => Should activate CFAR |
| 729 |
* U+1784,U+17D2,U+1782,U+17D2,U+179A [C+Coeng+C+Coeng+RO] => Should NOT activate CFAR |
| 730 |
*/ |
| 731 |
for ($j = ($i + 2); $j < $end; $j++) |
| 732 |
$info[$j]['mask'] |= self::FLAG(self::CFAR); |
| 733 |
|
| 734 |
break; |
| 735 |
} |
| 736 |
} |
| 737 |
} |
| 738 |
|
| 739 |
|
| 740 |
|
| 741 |
/* Sit tight, rock 'n roll! */ |
| 742 |
self::bubble_sort($info, $start, $end - $start); |
| 743 |
|
| 744 |
/* Find base again */ |
| 745 |
$base = $end; |
| 746 |
for ($i = $start; $i < $end; $i++) { |
| 747 |
if ($info[$i]['indic_position'] == self::POS_BASE_C) { |
| 748 |
$base = $i; |
| 749 |
break; |
| 750 |
} |
| 751 |
} |
| 752 |
|
| 753 |
if ($scriptblock != UCDN::SCRIPT_KHMER) { |
| 754 |
/* Reph */ |
| 755 |
for ($i = $start; $i < $end; $i++) { |
| 756 |
if ($info[$i]['indic_position'] == self::POS_RA_TO_BECOME_REPH) { |
| 757 |
$info[$i]['mask'] |= self::FLAG(self::RPHF); |
| 758 |
} |
| 759 |
} |
| 760 |
|
| 761 |
/* Pre-base */ |
| 762 |
$mask = self::FLAG(self::HALF); |
| 763 |
for ($i = $start; $i < $base; $i++) { |
| 764 |
$info[$i]['mask'] |= $mask; |
| 765 |
} |
| 766 |
} |
| 767 |
|
| 768 |
/* Post-base */ |
| 769 |
$mask = (self::FLAG(self::BLWF) | self::FLAG(self::ABVF) | self::FLAG(self::PSTF)); |
| 770 |
for ($i = $base + 1; $i < $end; $i++) { |
| 771 |
$info[$i]['mask'] |= $mask; |
| 772 |
} |
| 773 |
|
| 774 |
|
| 775 |
if ($scriptblock != UCDN::SCRIPT_KHMER) { |
| 776 |
if (!defined("OMIT_INDIC_FIX_3") || OMIT_INDIC_FIX_3 != 1) { |
| 777 |
/* INDIC_FIX_3 */ |
| 778 |
/* Find a (pre-base) Consonant, Halant,Ra sequence and mark Halant|Ra for below-base BLWF processing. */ |
| 779 |
// TEST CASE ক্র্ক in FreeSans versus Vrinda |
| 780 |
if (($base - $start) >= 3) { |
| 781 |
for ($i = $start; $i < ($base - 2); $i++) { |
| 782 |
if (self::is_consonant($info[$i])) { |
| 783 |
if (self::is_halant_or_coeng($info[$i + 1]) && self::is_ra($info[$i + 2]['uni'])) { |
| 784 |
// If would substitute Halant+Ra...BLWF |
| 785 |
if (isset($GSUBdata['blwf'][$info[$i + 2]['uni']])) { |
| 786 |
$info[$i + 1]['mask'] |= self::FLAG(self::BLWF); |
| 787 |
$info[$i + 2]['mask'] |= self::FLAG(self::BLWF); |
| 788 |
} |
| 789 |
/* If would not substitute as blwf, mark Ra+Halant for RPHF using following Halant (if present) */ else if (self::is_halant_or_coeng($info[$i + 3])) { |
| 790 |
$info[$i + 2]['mask'] |= self::FLAG(self::RPHF); |
| 791 |
$info[$i + 3]['mask'] |= self::FLAG(self::RPHF); |
| 792 |
} |
| 793 |
break; |
| 794 |
} |
| 795 |
} |
| 796 |
} |
| 797 |
} |
| 798 |
} |
| 799 |
} |
| 800 |
|
| 801 |
|
| 802 |
|
| 803 |
if ($is_old_spec && $scriptblock == UCDN::SCRIPT_DEVANAGARI) { |
| 804 |
/* Old-spec eye-lash Ra needs special handling. From the spec: |
| 805 |
* "The feature 'below-base form' is applied to consonants |
| 806 |
* having below-base forms and following the base consonant. |
| 807 |
* The exception is vattu, which may appear below half forms |
| 808 |
* as well as below the base glyph. The feature 'below-base |
| 809 |
* form' will be applied to all such occurrences of Ra as well." |
| 810 |
* |
| 811 |
* Test case: U+0924,U+094D,U+0930,U+094d,U+0915 |
| 812 |
* with Sanskrit 2003 font. |
| 813 |
* |
| 814 |
* However, note that Ra,Halant,ZWJ is the correct way to |
| 815 |
* request eyelash form of Ra, so we wouldbn't inhibit it |
| 816 |
* in that sequence. |
| 817 |
* |
| 818 |
* Test case: U+0924,U+094D,U+0930,U+094d,U+200D,U+0915 |
| 819 |
*/ |
| 820 |
for ($i = $start; ($i + 1) < $base; $i++) { |
| 821 |
if ($info[$i]['indic_category'] == self::OT_Ra && $info[$i + 1]['indic_category'] == self::OT_H && |
| 822 |
($i + 2 == $base || $info[$i + 2]['indic_category'] != self::OT_ZWJ)) { |
| 823 |
$info[$i]['mask'] |= self::FLAG(self::BLWF); |
| 824 |
$info[$i + 1]['mask'] |= self::FLAG(self::BLWF); |
| 825 |
} |
| 826 |
} |
| 827 |
} |
| 828 |
|
| 829 |
if ($scriptblock != UCDN::SCRIPT_KHMER) { |
| 830 |
if (count($GSUBdata['pref']) && $base + 2 < $end) { |
| 831 |
/* Find a Halant,Ra sequence and mark it for pre-base processing. */ |
| 832 |
for ($i = $base + 1; $i + 1 < $end; $i++) { |
| 833 |
// If old_spec find Ra-Halant... |
| 834 |
if ((isset($GSUBdata['pref'][$info[$i + 1]['uni']]) && self::is_halant_or_coeng($info[$i]) && self::is_ra($info[$i + 1]['uni']) ) || |
| 835 |
($is_old_spec && isset($GSUBdata['pref'][$info[$i]['uni']]) && self::is_halant_or_coeng($info[$i + 1]) && self::is_ra($info[$i]['uni']) ) |
| 836 |
) { |
| 837 |
$info[$i++]['mask'] |= self::FLAG(self::PREF); |
| 838 |
$info[$i++]['mask'] |= self::FLAG(self::PREF); |
| 839 |
break; |
| 840 |
} |
| 841 |
} |
| 842 |
} |
| 843 |
} |
| 844 |
|
| 845 |
|
| 846 |
/* Apply ZWJ/ZWNJ effects */ |
| 847 |
for ($i = $start + 1; $i < $end; $i++) { |
| 848 |
if (self::is_joiner($info[$i])) { |
| 849 |
$non_joiner = ($info[$i]['indic_category'] == self::OT_ZWNJ); |
| 850 |
$j = $i; |
| 851 |
while ($j > $start) { |
| 852 |
if (defined("OMIT_INDIC_FIX_4") && OMIT_INDIC_FIX_4 == 1) { |
| 853 |
// INDIC_FIX_4 = do nothing - carry on // |
| 854 |
// ZWNJ should block H C from forming blwf post-base - need to unmask backwards beyond first consonant arrived at // |
| 855 |
if (!self::is_consonant($info[$j])) { |
| 856 |
break; |
| 857 |
} |
| 858 |
} |
| 859 |
$j--; |
| 860 |
|
| 861 |
/* ZWJ/ZWNJ should disable CJCT. They do that by simply |
| 862 |
* being there, since we don't skip them for the CJCT |
| 863 |
* feature (ie. F_MANUAL_ZWJ) */ |
| 864 |
|
| 865 |
/* A ZWNJ disables HALF. */ |
| 866 |
if ($non_joiner) { |
| 867 |
$info[$j]['mask'] &= ~(self::FLAG(self::HALF) | self::FLAG(self::BLWF)); |
| 868 |
} |
| 869 |
} |
| 870 |
} |
| 871 |
} |
| 872 |
} |
| 873 |
|
| 874 |
public static function final_reordering(&$info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec) |
| 875 |
{ |
| 876 |
$count = count($info); |
| 877 |
if (!$count) |
| 878 |
return; |
| 879 |
$last = 0; |
| 880 |
$last_syllable = $info[0]['syllable']; |
| 881 |
for ($i = 1; $i < $count; $i++) { |
| 882 |
if ($last_syllable != $info[$i]['syllable']) { |
| 883 |
self::final_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $i); |
| 884 |
$last = $i; |
| 885 |
$last_syllable = $info[$last]['syllable']; |
| 886 |
} |
| 887 |
} |
| 888 |
self::final_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $count); |
| 889 |
} |
| 890 |
|
| 891 |
public static function final_reordering_syllable(&$info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $start, $end) |
| 892 |
{ |
| 893 |
|
| 894 |
/* 4. Final reordering: |
| 895 |
* |
| 896 |
* After the localized forms and basic shaping forms GSUB features have been |
| 897 |
* applied (see below), the shaping engine performs some final glyph |
| 898 |
* reordering before applying all the remaining font features to the entire |
| 899 |
* cluster. |
| 900 |
*/ |
| 901 |
|
| 902 |
/* Find base again */ |
| 903 |
for ($base = $start; $base < $end; $base++) |
| 904 |
if ($info[$base]['indic_position'] >= self::POS_BASE_C) { |
| 905 |
if ($start < $base && $info[$base]['indic_position'] > self::POS_BASE_C) |
| 906 |
$base--; |
| 907 |
break; |
| 908 |
} |
| 909 |
if ($base == $end && $start < $base && $info[$base - 1]['indic_category'] != self::OT_ZWJ) |
| 910 |
$base--; |
| 911 |
while ($start < $base && isset($info[$base]) && ($info[$base]['indic_category'] == self::OT_H || $info[$base]['indic_category'] == self::OT_N)) |
| 912 |
$base--; |
| 913 |
|
| 914 |
|
| 915 |
/* o Reorder matras: |
| 916 |
* |
| 917 |
* If a pre-base matra character had been reordered before applying basic |
| 918 |
* features, the glyph can be moved closer to the main consonant based on |
| 919 |
* whether half-forms had been formed. Actual position for the matra is |
| 920 |
* defined as "after last standalone halant glyph, after initial matra |
| 921 |
* position and before the main consonant". If ZWJ or ZWNJ follow this |
| 922 |
* halant, position is moved after it. |
| 923 |
*/ |
| 924 |
|
| 925 |
|
| 926 |
if ($start + 1 < $end && $start < $base) { /* Otherwise there can't be any pre-base matra characters. */ |
| 927 |
/* If we lost track of base, alas, position before last thingy. */ |
| 928 |
$new_pos = ($base == $end) ? $base - 2 : $base - 1; |
| 929 |
|
| 930 |
/* Malayalam / Tamil do not have "half" forms or explicit virama forms. |
| 931 |
* The glyphs formed by 'half' are Chillus or ligated explicit viramas. |
| 932 |
* We want to position matra after them. |
| 933 |
*/ |
| 934 |
if ($scriptblock != UCDN::SCRIPT_MALAYALAM && $scriptblock != UCDN::SCRIPT_TAMIL) { |
| 935 |
while ($new_pos > $start && !(self::is_one_of($info[$new_pos], (self::FLAG(self::OT_M) | self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng))))) |
| 936 |
$new_pos--; |
| 937 |
|
| 938 |
/* If we found no Halant we are done. |
| 939 |
* Otherwise only proceed if the Halant does |
| 940 |
* not belong to the Matra itself! */ |
| 941 |
if (self::is_halant_or_coeng($info[$new_pos]) && $info[$new_pos]['indic_position'] != self::POS_PRE_M) { |
| 942 |
/* -> If ZWJ or ZWNJ follow this halant, position is moved after it. */ |
| 943 |
if ($new_pos + 1 < $end && self::is_joiner($info[$new_pos + 1])) |
| 944 |
$new_pos++; |
| 945 |
} else |
| 946 |
$new_pos = $start; /* No move. */ |
| 947 |
} |
| 948 |
|
| 949 |
if ($start < $new_pos && $info[$new_pos]['indic_position'] != self::POS_PRE_M) { |
| 950 |
/* Now go see if there's actually any matras... */ |
| 951 |
for ($i = $new_pos; $i > $start; $i--) |
| 952 |
if ($info[$i - 1]['indic_position'] == self::POS_PRE_M) { |
| 953 |
$old_pos = $i - 1; |
| 954 |
//memmove (&info[$old_pos], &info[$old_pos + 1], ($new_pos - $old_pos) * sizeof ($info[0])); |
| 955 |
self::_move_info_pos($info, $old_pos, $new_pos + 1); |
| 956 |
|
| 957 |
if ($old_pos < $base && $base <= $new_pos) /* Shouldn't actually happen. */ |
| 958 |
$base--; |
| 959 |
$new_pos--; |
| 960 |
} |
| 961 |
} |
| 962 |
} |
| 963 |
|
| 964 |
|
| 965 |
/* o Reorder reph: |
| 966 |
* |
| 967 |
* Reph's original position is always at the beginning of the syllable, |
| 968 |
* (i.e. it is not reordered at the character reordering stage). However, |
| 969 |
* it will be reordered according to the basic-forms shaping results. |
| 970 |
* Possible positions for reph, depending on the script, are; after main, |
| 971 |
* before post-base consonant forms, and after post-base consonant forms. |
| 972 |
*/ |
| 973 |
|
| 974 |
/* If there's anything after the Ra that has the REPH pos, it ought to be halant. |
| 975 |
* Which means that the font has failed to ligate the Reph. In which case, we |
| 976 |
* shouldn't move. */ |
| 977 |
if ($start + 1 < $end && |
| 978 |
$info[$start]['indic_position'] == self::POS_RA_TO_BECOME_REPH && $info[$start + 1]['indic_position'] != self::POS_RA_TO_BECOME_REPH) { |
| 979 |
$reph_pos = $indic_config[3]; |
| 980 |
$skip_to_reph_step_5 = false; |
| 981 |
$skip_to_reph_move = false; |
| 982 |
|
| 983 |
/* 1. If reph should be positioned after post-base consonant forms, |
| 984 |
* proceed to step 5. |
| 985 |
*/ |
| 986 |
if ($reph_pos == self::REPH_POS_AFTER_POST) { |
| 987 |
$skip_to_reph_step_5 = true; |
| 988 |
} |
| 989 |
|
| 990 |
/* 2. If the reph repositioning class is not after post-base: target |
| 991 |
* position is after the first explicit halant glyph between the |
| 992 |
* first post-reph consonant and last main consonant. If ZWJ or ZWNJ |
| 993 |
* are following this halant, position is moved after it. If such |
| 994 |
* position is found, this is the target position. Otherwise, |
| 995 |
* proceed to the next step. |
| 996 |
* |
| 997 |
* Note: in old-implementation fonts, where classifications were |
| 998 |
* fixed in shaping engine, there was no case where reph position |
| 999 |
* will be found on this step. |
| 1000 |
*/ |
| 1001 |
|
| 1002 |
if (!$skip_to_reph_step_5) { |
| 1003 |
|
| 1004 |
$new_reph_pos = $start + 1; |
| 1005 |
|
| 1006 |
while ($new_reph_pos < $base && !self::is_halant_or_coeng($info[$new_reph_pos])) |
| 1007 |
$new_reph_pos++; |
| 1008 |
|
| 1009 |
if ($new_reph_pos < $base && self::is_halant_or_coeng($info[$new_reph_pos])) { |
| 1010 |
/* ->If ZWJ or ZWNJ are following this halant, position is moved after it. */ |
| 1011 |
if ($new_reph_pos + 1 < $base && self::is_joiner($info[$new_reph_pos + 1])) |
| 1012 |
$new_reph_pos++; |
| 1013 |
$skip_to_reph_move = true; |
| 1014 |
} |
| 1015 |
} |
| 1016 |
|
| 1017 |
/* 3. If reph should be repositioned after the main consonant: find the |
| 1018 |
* first consonant not ligated with main, or find the first |
| 1019 |
* consonant that is not a potential pre-base reordering Ra. |
| 1020 |
*/ |
| 1021 |
if ($reph_pos == self::REPH_POS_AFTER_MAIN && !$skip_to_reph_move && !$skip_to_reph_step_5) { |
| 1022 |
$new_reph_pos = $base; |
| 1023 |
/* XXX Skip potential pre-base reordering Ra. */ |
| 1024 |
while ($new_reph_pos + 1 < $end && $info[$new_reph_pos + 1]['indic_position'] <= self::POS_AFTER_MAIN) |
| 1025 |
$new_reph_pos++; |
| 1026 |
if ($new_reph_pos < $end) |
| 1027 |
$skip_to_reph_move = true; |
| 1028 |
} |
| 1029 |
|
| 1030 |
/* 4. If reph should be positioned before post-base consonant, find |
| 1031 |
* first post-base classified consonant not ligated with main. If no |
| 1032 |
* consonant is found, the target position should be before the |
| 1033 |
* first matra, syllable modifier sign or vedic sign. |
| 1034 |
*/ |
| 1035 |
/* This is our take on what step 4 is trying to say (and failing, BADLY). */ |
| 1036 |
if ($reph_pos == self::REPH_POS_AFTER_SUB && !$skip_to_reph_move && !$skip_to_reph_step_5) { |
| 1037 |
$new_reph_pos = $base; |
| 1038 |
while ($new_reph_pos < $end && isset($info[$new_reph_pos + 1]['indic_position']) && |
| 1039 |
!( self::FLAG($info[$new_reph_pos + 1]['indic_position']) & (self::FLAG(self::POS_POST_C) | self::FLAG(self::POS_AFTER_POST) | self::FLAG(self::POS_SMVD)))) { |
| 1040 |
$new_reph_pos++; |
| 1041 |
} |
| 1042 |
if ($new_reph_pos < $end) { |
| 1043 |
$skip_to_reph_move = true; |
| 1044 |
} |
| 1045 |
} |
| 1046 |
|
| 1047 |
/* 5. If no consonant is found in steps 3 or 4, move reph to a position |
| 1048 |
* immediately before the first post-base matra, syllable modifier |
| 1049 |
* sign or vedic sign that has a reordering class after the intended |
| 1050 |
* reph position. For example, if the reordering position for reph |
| 1051 |
* is post-main, it will skip above-base matras that also have a |
| 1052 |
* post-main position. |
| 1053 |
*/ |
| 1054 |
if (!$skip_to_reph_move) { |
| 1055 |
/* Copied from step 2. */ |
| 1056 |
$new_reph_pos = $start + 1; |
| 1057 |
while ($new_reph_pos < $base && !self::is_halant_or_coeng($info[$new_reph_pos])) |
| 1058 |
$new_reph_pos++; |
| 1059 |
|
| 1060 |
if ($new_reph_pos < $base && self::is_halant_or_coeng($info[$new_reph_pos])) { |
| 1061 |
/* ->If ZWJ or ZWNJ are following this halant, position is moved after it. */ |
| 1062 |
if ($new_reph_pos + 1 < $base && self::is_joiner($info[$new_reph_pos + 1])) |
| 1063 |
$new_reph_pos++; |
| 1064 |
$skip_to_reph_move = true; |
| 1065 |
} |
| 1066 |
} |
| 1067 |
|
| 1068 |
|
| 1069 |
/* 6. Otherwise, reorder reph to the end of the syllable. |
| 1070 |
*/ |
| 1071 |
if (!$skip_to_reph_move) { |
| 1072 |
$new_reph_pos = $end - 1; |
| 1073 |
while ($new_reph_pos > $start && $info[$new_reph_pos]['indic_position'] == self::POS_SMVD) |
| 1074 |
$new_reph_pos--; |
| 1075 |
|
| 1076 |
/* |
| 1077 |
* If the Reph is to be ending up after a Matra,Halant sequence, |
| 1078 |
* position it before that Halant so it can interact with the Matra. |
| 1079 |
* However, if it's a plain Consonant,Halant we shouldn't do that. |
| 1080 |
* Uniscribe doesn't do this. |
| 1081 |
* TEST: U+0930,U+094D,U+0915,U+094B,U+094D |
| 1082 |
*/ |
| 1083 |
//if (!$hb_options.uniscribe_bug_compatible && self::is_halant_or_coeng($info[$new_reph_pos])) { |
| 1084 |
if (self::is_halant_or_coeng($info[$new_reph_pos])) { |
| 1085 |
for ($i = $base + 1; $i < $new_reph_pos; $i++) |
| 1086 |
if ($info[$i]['indic_category'] == self::OT_M) { |
| 1087 |
/* Ok, got it. */ |
| 1088 |
$new_reph_pos--; |
| 1089 |
} |
| 1090 |
} |
| 1091 |
} |
| 1092 |
|
| 1093 |
|
| 1094 |
/* Move */ |
| 1095 |
self::_move_info_pos($info, $start, $new_reph_pos + 1); |
| 1096 |
|
| 1097 |
if ($start < $base && $base <= $new_reph_pos) { |
| 1098 |
$base--; |
| 1099 |
} |
| 1100 |
} |
| 1101 |
|
| 1102 |
|
| 1103 |
/* o Reorder pre-base reordering consonants: |
| 1104 |
* |
| 1105 |
* If a pre-base reordering consonant is found, reorder it according to |
| 1106 |
* the following rules: |
| 1107 |
*/ |
| 1108 |
|
| 1109 |
|
| 1110 |
if (count($GSUBdata['pref']) && $base + 1 < $end) { /* Otherwise there can't be any pre-base reordering Ra. */ |
| 1111 |
for ($i = $base + 1; $i < $end; $i++) { |
| 1112 |
if ($info[$i]['mask'] & self::FLAG(self::PREF)) { |
| 1113 |
/* 1. Only reorder a glyph produced by substitution during application |
| 1114 |
* of the <pref> feature. (Note that a font may shape a Ra consonant with |
| 1115 |
* the feature generally but block it in certain contexts.) |
| 1116 |
*/ |
| 1117 |
// ??? Need to TEST if actual substitution has occurred |
| 1118 |
if ($i + 1 == $end || ($info[$i + 1]['mask'] & self::FLAG(self::PREF)) == 0) { |
| 1119 |
/* |
| 1120 |
* 2. Try to find a target position the same way as for pre-base matra. |
| 1121 |
* If it is found, reorder pre-base consonant glyph. |
| 1122 |
* |
| 1123 |
* 3. If position is not found, reorder immediately before main |
| 1124 |
* consonant. |
| 1125 |
*/ |
| 1126 |
$new_pos = $base; |
| 1127 |
/* Malayalam / Tamil do not have "half" forms or explicit virama forms. |
| 1128 |
* The glyphs formed by 'half' are Chillus or ligated explicit viramas. |
| 1129 |
* We want to position matra after them. |
| 1130 |
*/ |
| 1131 |
if ($scriptblock != UCDN::SCRIPT_MALAYALAM && $scriptblock != UCDN::SCRIPT_TAMIL) { |
| 1132 |
while ($new_pos > $start && |
| 1133 |
!(self::is_one_of($info[$new_pos - 1], self::FLAG(self::OT_M) | self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng)))) |
| 1134 |
$new_pos--; |
| 1135 |
|
| 1136 |
/* In Khmer coeng model, a V,Ra can go *after* matras. If it goes after a |
| 1137 |
* split matra, it should be reordered to *before* the left part of such matra. */ |
| 1138 |
if ($new_pos > $start && $info[$new_pos - 1]['indic_category'] == self::OT_M) { |
| 1139 |
$old_pos = i; |
| 1140 |
for ($i = $base + 1; $i < $old_pos; $i++) |
| 1141 |
if ($info[$i]['indic_category'] == self::OT_M) { |
| 1142 |
$new_pos--; |
| 1143 |
break; |
| 1144 |
} |
| 1145 |
} |
| 1146 |
} |
| 1147 |
|
| 1148 |
if ($new_pos > $start && self::is_halant_or_coeng($info[$new_pos - 1])) { |
| 1149 |
/* -> If ZWJ or ZWNJ follow this halant, position is moved after it. */ |
| 1150 |
if ($new_pos < $end && self::is_joiner($info[$new_pos])) |
| 1151 |
$new_pos++; |
| 1152 |
} |
| 1153 |
|
| 1154 |
$old_pos = $i; |
| 1155 |
self::_move_info_pos($info, $old_pos, $new_pos); |
| 1156 |
|
| 1157 |
if ($new_pos <= $base && $base < $old_pos) |
| 1158 |
$base++; |
| 1159 |
} |
| 1160 |
|
| 1161 |
break; |
| 1162 |
} |
| 1163 |
} |
| 1164 |
} |
| 1165 |
|
| 1166 |
|
| 1167 |
/* Apply 'init' to the Left Matra if it's a word start. */ |
| 1168 |
if ($info[$start]['indic_position'] == self::POS_PRE_M && |
| 1169 |
($start == 0 || |
| 1170 |
($info[$start - 1]['general_category'] < UCDN::UNICODE_GENERAL_CATEGORY_FORMAT || $info[$start - 1]['general_category'] > UCDN::UNICODE_GENERAL_CATEGORY_NON_SPACING_MARK) |
| 1171 |
)) { |
| 1172 |
$info[$start]['mask'] |= self::FLAG(self::INIT); |
| 1173 |
} |
| 1174 |
|
| 1175 |
|
| 1176 |
/* |
| 1177 |
* Finish off and go home! |
| 1178 |
*/ |
| 1179 |
} |
| 1180 |
|
| 1181 |
public static function _move_info_pos(&$info, $from, $to) |
| 1182 |
{ |
| 1183 |
$t = array(); |
| 1184 |
$t[0] = $info[$from]; |
| 1185 |
if ($from > $to) { |
| 1186 |
array_splice($info, $from, 1); |
| 1187 |
array_splice($info, $to, 0, $t); |
| 1188 |
} else { |
| 1189 |
array_splice($info, $to, 0, $t); |
| 1190 |
array_splice($info, $from, 1); |
| 1191 |
} |
| 1192 |
} |
| 1193 |
|
| 1194 |
public static $ra_chars = array( |
| 1195 |
0x0930 => 1, /* Devanagari */ |
| 1196 |
0x09B0 => 1, /* Bengali */ |
| 1197 |
0x09F0 => 1, /* Bengali (Assamese) */ |
| 1198 |
0x0A30 => 1, /* Gurmukhi */ /* No Reph */ |
| 1199 |
0x0AB0 => 1, /* Gujarati */ |
| 1200 |
0x0B30 => 1, /* Oriya */ |
| 1201 |
0x0BB0 => 1, /* Tamil */ /* No Reph */ |
| 1202 |
0x0C30 => 1, /* Telugu */ /* Reph formed only with ZWJ */ |
| 1203 |
0x0CB0 => 1, /* Kannada */ |
| 1204 |
0x0D30 => 1, /* Malayalam */ /* No Reph, Logical Repha */ |
| 1205 |
0x0DBB => 1, /* Sinhala */ /* Reph formed only with ZWJ */ |
| 1206 |
0x179A => 1, /* Khmer */ /* No Reph, Visual Repha */ |
| 1207 |
); |
| 1208 |
|
| 1209 |
public static function is_ra($u) |
| 1210 |
{ |
| 1211 |
if (isset(self::$ra_chars[$u])) |
| 1212 |
return true; |
| 1213 |
return false; |
| 1214 |
} |
| 1215 |
|
| 1216 |
public static function is_one_of($info, $flags) |
| 1217 |
{ |
| 1218 |
if (isset($info['is_ligature']) && $info['is_ligature']) |
| 1219 |
return false; /* If it ligated, all bets are off. */ |
| 1220 |
return !!(self::FLAG($info['indic_category']) & $flags); |
| 1221 |
} |
| 1222 |
|
| 1223 |
public static function is_joiner($info) |
| 1224 |
{ |
| 1225 |
return self::is_one_of($info, (self::FLAG(self::OT_ZWJ) | self::FLAG(self::OT_ZWNJ))); |
| 1226 |
} |
| 1227 |
|
| 1228 |
/* Vowels and placeholders treated as if they were consonants. */ |
| 1229 |
|
| 1230 |
public static function is_consonant($info) |
| 1231 |
{ |
| 1232 |
return self::is_one_of($info, (self::FLAG(self::OT_C) | self::FLAG(self::OT_CM) | self::FLAG(self::OT_Ra) | self::FLAG(self::OT_V) | self::FLAG(self::OT_NBSP) | self::FLAG(self::OT_DOTTEDCIRCLE))); |
| 1233 |
} |
| 1234 |
|
| 1235 |
public static function is_halant_or_coeng($info) |
| 1236 |
{ |
| 1237 |
return self::is_one_of($info, (self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng))); |
| 1238 |
} |
| 1239 |
|
| 1240 |
// From hb-private.hh |
| 1241 |
public static function in_range($u, $lo, $hi) |
| 1242 |
{ |
| 1243 |
if ((($lo ^ $hi) & $lo) == 0 && (($lo ^ $hi) & $hi) == ($lo ^ $hi) && (($lo ^ $hi) & (($lo ^ $hi) + 1)) == 0) |
| 1244 |
return ($u & ~($lo ^ $hi)) == $lo; |
| 1245 |
else |
| 1246 |
return $lo <= $u && $u <= $hi; |
| 1247 |
} |
| 1248 |
|
| 1249 |
// From hb-private.hh |
| 1250 |
public static function FLAG($x) |
| 1251 |
{ |
| 1252 |
return (1 << ($x)); |
| 1253 |
} |
| 1254 |
|
| 1255 |
// BELOW from hb-ot-shape-complex-indic.cc |
| 1256 |
|
| 1257 |
/* |
| 1258 |
* Indic configurations. |
| 1259 |
*/ |
| 1260 |
|
| 1261 |
// base_position |
| 1262 |
const BASE_POS_FIRST = 0; |
| 1263 |
const BASE_POS_LAST = 1; |
| 1264 |
|
| 1265 |
// reph_position |
| 1266 |
const REPH_POS_DEFAULT = 10; // POS_BEFORE_POST, |
| 1267 |
|
| 1268 |
const REPH_POS_AFTER_MAIN = 5; // POS_AFTER_MAIN, |
| 1269 |
|
| 1270 |
const REPH_POS_BEFORE_SUB = 7; // POS_BEFORE_SUB, |
| 1271 |
const REPH_POS_AFTER_SUB = 9; // POS_AFTER_SUB, |
| 1272 |
const REPH_POS_BEFORE_POST = 10; // POS_BEFORE_POST, |
| 1273 |
const REPH_POS_AFTER_POST = 12; // POS_AFTER_POST |
| 1274 |
|
| 1275 |
// reph_mode |
| 1276 |
const REPH_MODE_IMPLICIT = 0; /* Reph formed out of initial Ra,H sequence. */ |
| 1277 |
const REPH_MODE_EXPLICIT = 1; /* Reph formed out of initial Ra,H,ZWJ sequence. */ |
| 1278 |
const REPH_MODE_VIS_REPHA = 2; /* Encoded Repha character, no reordering needed. */ |
| 1279 |
const REPH_MODE_LOG_REPHA = 3; /* Encoded Repha character, needs reordering. */ |
| 1280 |
|
| 1281 |
/* |
| 1282 |
struct of indic_configs{ |
| 1283 |
KEY - script; |
| 1284 |
0 - has_old_spec; |
| 1285 |
1 - virama; |
| 1286 |
2 - base_pos; |
| 1287 |
3 - reph_pos; |
| 1288 |
4 - reph_mode; |
| 1289 |
}; |
| 1290 |
*/ |
| 1291 |
|
| 1292 |
public static $indic_configs = array(/* index is SCRIPT_number from UCDN */ |
| 1293 |
9 => array(true, 0x094D, 1, 10, 0), |
| 1294 |
10 => array(true, 0x09CD, 1, 9, 0), |
| 1295 |
11 => array(true, 0x0A4D, 1, 7, 0), |
| 1296 |
12 => array(true, 0x0ACD, 1, 10, 0), |
| 1297 |
13 => array(true, 0x0B4D, 1, 5, 0), |
| 1298 |
14 => array(true, 0x0BCD, 1, 12, 0), |
| 1299 |
15 => array(true, 0x0C4D, 1, 12, 1), |
| 1300 |
16 => array(true, 0x0CCD, 1, 12, 0), |
| 1301 |
17 => array(true, 0x0D4D, 1, 5, 3), |
| 1302 |
18 => array(false, 0x0DCA, 0, 5, 1), /* Sinhala */ |
| 1303 |
30 => array(false, 0x17D2, 0, 10, 2), /* Khmer */ |
| 1304 |
84 => array(false, 0xA9C0, 1, 10, 0), /* Javanese */ |
| 1305 |
); |
| 1306 |
|
| 1307 |
|
| 1308 |
|
| 1309 |
/* |
| 1310 |
|
| 1311 |
// from "hb-ot-shape-complex-indic-table.cc" |
| 1312 |
|
| 1313 |
|
| 1314 |
const ISC_A = 0; // INDIC_SYLLABIC_CATEGORY_AVAGRAHA Avagraha |
| 1315 |
const ISC_Bi = 8; // INDIC_SYLLABIC_CATEGORY_BINDU Bindu |
| 1316 |
const ISC_C = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT Consonant |
| 1317 |
const ISC_CD = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT_DEAD Consonant_Dead |
| 1318 |
const ISC_CF = 17; // INDIC_SYLLABIC_CATEGORY_CONSONANT_FINAL Consonant_Final |
| 1319 |
const ISC_CHL = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT_HEAD_LETTER Consonant_Head_Letter |
| 1320 |
const ISC_CM = 17; // INDIC_SYLLABIC_CATEGORY_CONSONANT_MEDIAL Consonant_Medial |
| 1321 |
const ISC_CP = 11; // INDIC_SYLLABIC_CATEGORY_CONSONANT_PLACEHOLDER Consonant_Placeholder |
| 1322 |
const ISC_CR = 15; // INDIC_SYLLABIC_CATEGORY_CONSONANT_REPHA Consonant_Repha |
| 1323 |
const ISC_CS = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT_SUBJOINED Consonant_Subjoined |
| 1324 |
const ISC_ML = 0; // INDIC_SYLLABIC_CATEGORY_MODIFYING_LETTER Modifying_Letter |
| 1325 |
const ISC_N = 3; // INDIC_SYLLABIC_CATEGORY_NUKTA Nukta |
| 1326 |
const ISC_x = 0; // INDIC_SYLLABIC_CATEGORY_OTHER Other |
| 1327 |
const ISC_RS = 13; // INDIC_SYLLABIC_CATEGORY_REGISTER_SHIFTER Register_Shifter |
| 1328 |
const ISC_TL = 0; // INDIC_SYLLABIC_CATEGORY_TONE_LETTER Tone_Letter |
| 1329 |
const ISC_TM = 3; // INDIC_SYLLABIC_CATEGORY_TONE_MARK Tone_Mark |
| 1330 |
const ISC_V = 4; // INDIC_SYLLABIC_CATEGORY_VIRAMA Virama |
| 1331 |
const ISC_Vs = 8; // INDIC_SYLLABIC_CATEGORY_VISARGA Visarga |
| 1332 |
const ISC_Vo = 2; // INDIC_SYLLABIC_CATEGORY_VOWEL Vowel |
| 1333 |
const ISC_M = 7; // INDIC_SYLLABIC_CATEGORY_VOWEL_DEPENDENT Vowel_Dependent |
| 1334 |
const ISC_VI = 2; // INDIC_SYLLABIC_CATEGORY_VOWEL_INDEPENDENT Vowel_Independent |
| 1335 |
|
| 1336 |
const IMC_B = 8; // INDIC_MATRA_CATEGORY_BOTTOM Bottom |
| 1337 |
const IMC_BR = 11; // INDIC_MATRA_CATEGORY_BOTTOM_AND_RIGHT Bottom_And_Right |
| 1338 |
const IMC_I = 15; // INDIC_MATRA_CATEGORY_INVISIBLE Invisible |
| 1339 |
const IMC_L = 3; // INDIC_MATRA_CATEGORY_LEFT Left |
| 1340 |
const IMC_LR = 11; // INDIC_MATRA_CATEGORY_LEFT_AND_RIGHT Left_And_Right |
| 1341 |
const IMC_x = 15; // INDIC_MATRA_CATEGORY_NOT_APPLICABLE Not_Applicable |
| 1342 |
const IMC_O = 5; // INDIC_MATRA_CATEGORY_OVERSTRUCK Overstruck |
| 1343 |
const IMC_R = 11; // INDIC_MATRA_CATEGORY_RIGHT Right |
| 1344 |
const IMC_T = 6; // INDIC_MATRA_CATEGORY_TOP Top |
| 1345 |
const IMC_TB = 8; // INDIC_MATRA_CATEGORY_TOP_AND_BOTTOM Top_And_Bottom |
| 1346 |
const IMC_TBR = 11; // INDIC_MATRA_CATEGORY_TOP_AND_BOTTOM_AND_RIGHT Top_And_Bottom_And_Right |
| 1347 |
const IMC_TL = 6; // INDIC_MATRA_CATEGORY_TOP_AND_LEFT Top_And_Left |
| 1348 |
const IMC_TLR = 11; // INDIC_MATRA_CATEGORY_TOP_AND_LEFT_AND_RIGHT Top_And_Left_And_Right |
| 1349 |
const IMC_TR = 11; // INDIC_MATRA_CATEGORY_TOP_AND_RIGHT Top_And_Right |
| 1350 |
const IMC_VOL = 2; // INDIC_MATRA_CATEGORY_VISUAL_ORDER_LEFT Visual_Order_Left |
| 1351 |
|
| 1352 |
If in original table = _(C,x), that = ISC_C,IMC_x |
| 1353 |
Value is IMC_x << 8 (or IMC_x * 256) = 3840 |
| 1354 |
plus ISC_C = 1, so = 3841 |
| 1355 |
|
| 1356 |
*/ |
| 1357 |
|
| 1358 |
public static $indic_table = array( |
| 1359 |
/* Devanagari (0900..097F) */ |
| 1360 |
|
| 1361 |
/* 0900 */ 3848, 3848, 3848, 3848, 3842, 3842, 3842, 3842, |
| 1362 |
/* 0908 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3842, |
| 1363 |
/* 0910 */ 3842, 3842, 3842, 3842, 3842, 3841, 3841, 3841, |
| 1364 |
/* 0918 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1365 |
/* 0920 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1366 |
/* 0928 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1367 |
/* 0930 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1368 |
/* 0938 */ 3841, 3841, 1543, 2823, 3843, 3840, 2823, 775, |
| 1369 |
/* 0940 */ 2823, 2055, 2055, 2055, 2055, 1543, 1543, 1543, |
| 1370 |
/* 0948 */ 1543, 2823, 2823, 2823, 2823, 2052, 775, 2823, |
| 1371 |
/* 0950 */ 3840, 3840, 3840, 3840, 3840, 1543, 2055, 2055, |
| 1372 |
/* 0958 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1373 |
/* 0960 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1374 |
/* 0968 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1375 |
/* 0970 */ 3840, 3840, 3842, 3842, 3842, 3842, 3842, 3842, |
| 1376 |
/* 0978 */ 3840, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1377 |
/* Bengali (0980..09FF) */ |
| 1378 |
|
| 1379 |
/* 0980 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1380 |
/* 0988 */ 3842, 3842, 3842, 3842, 3842, 3840, 3840, 3842, |
| 1381 |
/* 0990 */ 3842, 3840, 3840, 3842, 3842, 3841, 3841, 3841, |
| 1382 |
/* 0998 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1383 |
/* 09A0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1384 |
/* 09A8 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1385 |
/* 09B0 */ 3841, 3840, 3841, 3840, 3840, 3840, 3841, 3841, |
| 1386 |
/* 09B8 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 775, |
| 1387 |
/* 09C0 */ 2823, 2055, 2055, 2055, 2055, 3840, 3840, 775, |
| 1388 |
/* 09C8 */ 775, 3840, 3840, 2823, 2823, 2052, 3841, 3840, |
| 1389 |
/* 09D0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 2823, |
| 1390 |
/* 09D8 */ 3840, 3840, 3840, 3840, 3841, 3841, 3840, 3841, |
| 1391 |
/* 09E0 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1392 |
/* 09E8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1393 |
/* 09F0 */ 3841, 3841, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1394 |
/* 09F8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1395 |
/* Gurmukhi (0A00..0A7F) */ |
| 1396 |
|
| 1397 |
/* 0A00 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1398 |
/* 0A08 */ 3842, 3842, 3842, 3840, 3840, 3840, 3840, 3842, |
| 1399 |
/* 0A10 */ 3842, 3840, 3840, 3842, 3842, 3841, 3841, 3841, |
| 1400 |
/* 0A18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1401 |
/* 0A20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1402 |
/* 0A28 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1403 |
/* 0A30 */ 3841, 3840, 3841, 3841, 3840, 3841, 3841, 3840, |
| 1404 |
/* 0A38 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 775, |
| 1405 |
/* 0A40 */ 2823, 2055, 2055, 3840, 3840, 3840, 3840, 1543, |
| 1406 |
/* 0A48 */ 1543, 3840, 3840, 1543, 1543, 2052, 3840, 3840, |
| 1407 |
/* 0A50 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1408 |
/* 0A58 */ 3840, 3841, 3841, 3841, 3841, 3840, 3841, 3840, |
| 1409 |
/* 0A60 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1410 |
/* 0A68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1411 |
/* 0A70 */ 3848, 3840, 13841, 13841, 3840, 3857, 3840, 3840, |
| 1412 |
/* 0A78 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1413 |
/* Gujarati (0A80..0AFF) */ |
| 1414 |
|
| 1415 |
/* 0A80 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1416 |
/* 0A88 */ 3842, 3842, 3842, 3842, 3842, 3842, 3840, 3842, |
| 1417 |
/* 0A90 */ 3842, 3842, 3840, 3842, 3842, 3841, 3841, 3841, |
| 1418 |
/* 0A98 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1419 |
/* 0AA0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1420 |
/* 0AA8 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1421 |
/* 0AB0 */ 3841, 3840, 3841, 3841, 3840, 3841, 3841, 3841, |
| 1422 |
/* 0AB8 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 775, |
| 1423 |
/* 0AC0 */ 2823, 2055, 2055, 2055, 2055, 1543, 3840, 1543, |
| 1424 |
/* 0AC8 */ 1543, 2823, 3840, 2823, 2823, 2052, 3840, 3840, |
| 1425 |
/* 0AD0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1426 |
/* 0AD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1427 |
/* 0AE0 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1428 |
/* 0AE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1429 |
/* 0AF0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1430 |
/* 0AF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1431 |
/* Oriya (0B00..0B7F) */ |
| 1432 |
|
| 1433 |
/* 0B00 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1434 |
/* 0B08 */ 3842, 3842, 3842, 3842, 3842, 3840, 3840, 3842, |
| 1435 |
/* 0B10 */ 3842, 3840, 3840, 3842, 3842, 3841, 3841, 3841, |
| 1436 |
/* 0B18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1437 |
/* 0B20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1438 |
/* 0B28 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1439 |
/* 0B30 */ 3841, 3840, 3841, 3841, 3840, 3841, 3841, 3841, |
| 1440 |
/* 0B38 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 1543, |
| 1441 |
/* 0B40 */ 2823, 2055, 2055, 2055, 2055, 3840, 3840, 775, |
| 1442 |
/* 0B48 */ 1543, 3840, 3840, 2823, 2823, 2052, 3840, 3840, |
| 1443 |
/* 0B50 */ 3840, 3840, 3840, 3840, 3840, 3840, 1543, 2823, |
| 1444 |
/* 0B58 */ 3840, 3840, 3840, 3840, 3841, 3841, 3840, 3841, |
| 1445 |
/* 0B60 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1446 |
/* 0B68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1447 |
/* 0B70 */ 3840, 3841, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1448 |
/* 0B78 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1449 |
/* Tamil (0B80..0BFF) */ |
| 1450 |
|
| 1451 |
/* 0B80 */ 3840, 3840, 3848, 3840, 3840, 3842, 3842, 3842, |
| 1452 |
/* 0B88 */ 3842, 3842, 3842, 3840, 3840, 3840, 3842, 3842, |
| 1453 |
/* 0B90 */ 3842, 3840, 3842, 3842, 3842, 3841, 3840, 3840, |
| 1454 |
/* 0B98 */ 3840, 3841, 3841, 3840, 3841, 3840, 3841, 3841, |
| 1455 |
/* 0BA0 */ 3840, 3840, 3840, 3841, 3841, 3840, 3840, 3840, |
| 1456 |
/* 0BA8 */ 3841, 3841, 3841, 3840, 3840, 3840, 3841, 3841, |
| 1457 |
/* 0BB0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1458 |
/* 0BB8 */ 3841, 3841, 3840, 3840, 3840, 3840, 2823, 2823, |
| 1459 |
/* 0BC0 */ 1543, 2055, 2055, 3840, 3840, 3840, 775, 775, |
| 1460 |
/* 0BC8 */ 775, 3840, 2823, 2823, 2823, 1540, 3840, 3840, |
| 1461 |
/* 0BD0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 2823, |
| 1462 |
/* 0BD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1463 |
/* 0BE0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1464 |
/* 0BE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1465 |
/* 0BF0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1466 |
/* 0BF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1467 |
/* Telugu (0C00..0C7F) */ |
| 1468 |
|
| 1469 |
/* 0C00 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1470 |
/* 0C08 */ 3842, 3842, 3842, 3842, 3842, 3840, 3842, 3842, |
| 1471 |
/* 0C10 */ 3842, 3840, 3842, 3842, 3842, 3841, 3841, 3841, |
| 1472 |
/* 0C18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1473 |
/* 0C20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1474 |
/* 0C28 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1475 |
/* 0C30 */ 3841, 3841, 3841, 3841, 3840, 3841, 3841, 3841, |
| 1476 |
/* 0C38 */ 3841, 3841, 3840, 3840, 3840, 3840, 1543, 1543, |
| 1477 |
/* 0C40 */ 1543, 2823, 2823, 2823, 2823, 3840, 1543, 1543, |
| 1478 |
/* 0C48 */ 2055, 3840, 1543, 1543, 1543, 1540, 3840, 3840, |
| 1479 |
/* 0C50 */ 3840, 3840, 3840, 3840, 3840, 1543, 2055, 3840, |
| 1480 |
/* 0C58 */ 3841, 3841, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1481 |
/* 0C60 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1482 |
/* 0C68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1483 |
/* 0C70 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1484 |
/* 0C78 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1485 |
/* Kannada (0C80..0CFF) */ |
| 1486 |
|
| 1487 |
/* 0C80 */ 3840, 3840, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1488 |
/* 0C88 */ 3842, 3842, 3842, 3842, 3842, 3840, 3842, 3842, |
| 1489 |
/* 0C90 */ 3842, 3840, 3842, 3842, 3842, 3841, 3841, 3841, |
| 1490 |
/* 0C98 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1491 |
/* 0CA0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1492 |
/* 0CA8 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1493 |
/* 0CB0 */ 3841, 3841, 3841, 3841, 3840, 3841, 3841, 3841, |
| 1494 |
/* 0CB8 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 1543, |
| 1495 |
/* 0CC0 */ 2823, 2823, 2823, 2823, 2823, 3840, 1543, 2823, |
| 1496 |
/* 0CC8 */ 2823, 3840, 2823, 2823, 1543, 1540, 3840, 3840, |
| 1497 |
/* 0CD0 */ 3840, 3840, 3840, 3840, 3840, 2823, 2823, 3840, |
| 1498 |
/* 0CD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3841, 3840, |
| 1499 |
/* 0CE0 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1500 |
/* 0CE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1501 |
/* 0CF0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1502 |
/* 0CF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1503 |
/* Malayalam (0D00..0D7F) */ |
| 1504 |
|
| 1505 |
/* 0D00 */ 3840, 3840, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1506 |
/* 0D08 */ 3842, 3842, 3842, 3842, 3842, 3840, 3842, 3842, |
| 1507 |
/* 0D10 */ 3842, 3840, 3842, 3842, 3842, 3841, 3841, 3841, |
| 1508 |
/* 0D18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1509 |
/* 0D20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1510 |
/* 0D28 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1511 |
/* 0D30 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1512 |
/* 0D38 */ 3841, 3841, 3841, 3840, 3840, 3840, 2823, 2823, |
| 1513 |
/* 0D40 */ 2823, 2823, 2823, 2055, 2055, 3840, 775, 775, |
| 1514 |
/* 0D48 */ 775, 3840, 2823, 2823, 2823, 1540, 3855, 3840, |
| 1515 |
/* 0D50 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 2823, |
| 1516 |
/* 0D58 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1517 |
/* 0D60 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, |
| 1518 |
/* 0D68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1519 |
/* 0D70 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1520 |
/* 0D78 */ 3840, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1521 |
/* Sinhala (0D80..0DFF) */ |
| 1522 |
|
| 1523 |
/* 0D80 */ 3840, 3840, 3848, 3848, 3840, 3842, 3842, 3842, |
| 1524 |
/* 0D88 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3842, |
| 1525 |
/* 0D90 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3840, |
| 1526 |
/* 0D98 */ 3840, 3840, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1527 |
/* 0DA0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1528 |
/* 0DA8 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1529 |
/* 0DB0 */ 3841, 3841, 3840, 3841, 3841, 3841, 3841, 3841, |
| 1530 |
/* 0DB8 */ 3841, 3841, 3841, 3841, 3840, 3841, 3840, 3840, |
| 1531 |
/* 0DC0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3840, |
| 1532 |
/* 0DC8 */ 3840, 3840, 1540, 3840, 3840, 3840, 3840, 2823, |
| 1533 |
/* 0DD0 */ 2823, 2823, 1543, 1543, 2055, 3840, 2055, 3840, |
| 1534 |
/* 0DD8 */ 2823, 775, 1543, 775, 2823, 2823, 2823, 2823, |
| 1535 |
/* 0DE0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1536 |
/* 0DE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1537 |
/* 0DF0 */ 3840, 3840, 2823, 2823, 3840, 3840, 3840, 3840, |
| 1538 |
/* 0DF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1539 |
/* Vedic Extensions (1CD0..1CFF) */ |
| 1540 |
|
| 1541 |
/* 1CD0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1542 |
/* 1CD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1543 |
/* 1CE0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1544 |
/* 1CE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1545 |
/* 1CF0 */ 3840, 3840, 3848, 3848, 3840, 3840, 3840, 3840, |
| 1546 |
/* 1CF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1547 |
); |
| 1548 |
|
| 1549 |
public static $khmer_table = array( |
| 1550 |
/* Khmer (1780..17FF) */ |
| 1551 |
|
| 1552 |
/* 1780 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1553 |
/* 1788 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1554 |
/* 1790 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1555 |
/* 1798 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, |
| 1556 |
/* 17A0 */ 3841, 3841, 3841, 3842, 3842, 3842, 3842, 3842, |
| 1557 |
/* 17A8 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3842, |
| 1558 |
/* 17B0 */ 3842, 3842, 3842, 3842, 3840, 3840, 2823, 1543, |
| 1559 |
/* 17B8 */ 1543, 1543, 1543, 2055, 2055, 2055, 1543, 2823, |
| 1560 |
/* 17C0 */ 2823, 775, 775, 775, 2823, 2823, 3848, 3848, |
| 1561 |
/* 17C8 */ 2823, 3853, 3853, 3840, 3855, 3840, 3840, 3840, |
| 1562 |
/* 17D0 */ 3840, 1540, 3844, 3840, 3840, 3840, 3840, 3840, |
| 1563 |
/* 17D8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1564 |
/* 17E0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1565 |
/* 17E8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1566 |
/* 17F0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1567 |
/* 17F8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, |
| 1568 |
); |
| 1569 |
|
| 1570 |
// from "hb-ot-shape-complex-indic-table.cc" |
| 1571 |
public static function indic_get_categories($u) |
| 1572 |
{ |
| 1573 |
if (0x0900 <= $u && $u <= 0x0DFF) |
| 1574 |
return self::$indic_table[$u - 0x0900 + 0]; // offset 0 for Most "indic" |
| 1575 |
if (0x1CD0 <= $u && $u <= 0x1D00) |
| 1576 |
return self::$indic_table[$u - 0x1CD0 + 1152]; // offset for Vedic extensions |
| 1577 |
if (0x1780 <= $u && $u <= 0x17FF) |
| 1578 |
return self::$khmer_table[$u - 0x1780]; // Khmer |
| 1579 |
if ($u == 0x00A0) |
| 1580 |
return 3851; // (ISC_CP | (IMC_x << 8)) |
| 1581 |
if ($u == 0x25CC) |
| 1582 |
return 3851; // (ISC_CP | (IMC_x << 8)) |
| 1583 |
return 3840; // (ISC_x | (IMC_x << 8)) |
| 1584 |
} |
| 1585 |
|
| 1586 |
// BELOW from hb-ot-shape-complex-indic.cc |
| 1587 |
/* |
| 1588 |
* Indic shaper. |
| 1589 |
*/ |
| 1590 |
|
| 1591 |
public static function IN_HALF_BLOCK($u, $Base) |
| 1592 |
{ |
| 1593 |
return (($u & ~0x7F) == $Base); |
| 1594 |
} |
| 1595 |
|
| 1596 |
public static function IS_DEVA($u) |
| 1597 |
{ |
| 1598 |
return self::IN_HALF_BLOCK($u, 0x0900); |
| 1599 |
} |
| 1600 |
|
| 1601 |
public static function IS_BENG($u) |
| 1602 |
{ |
| 1603 |
return self::IN_HALF_BLOCK($u, 0x0980); |
| 1604 |
} |
| 1605 |
|
| 1606 |
public static function IS_GURU($u) |
| 1607 |
{ |
| 1608 |
return self::IN_HALF_BLOCK($u, 0x0A00); |
| 1609 |
} |
| 1610 |
|
| 1611 |
public static function IS_GUJR($u) |
| 1612 |
{ |
| 1613 |
return self::IN_HALF_BLOCK($u, 0x0A80); |
| 1614 |
} |
| 1615 |
|
| 1616 |
public static function IS_ORYA($u) |
| 1617 |
{ |
| 1618 |
return self::IN_HALF_BLOCK($u, 0x0B00); |
| 1619 |
} |
| 1620 |
|
| 1621 |
public static function IS_TAML($u) |
| 1622 |
{ |
| 1623 |
return self::IN_HALF_BLOCK($u, 0x0B80); |
| 1624 |
} |
| 1625 |
|
| 1626 |
public static function IS_TELU($u) |
| 1627 |
{ |
| 1628 |
return self::IN_HALF_BLOCK($u, 0x0C00); |
| 1629 |
} |
| 1630 |
|
| 1631 |
public static function IS_KNDA($u) |
| 1632 |
{ |
| 1633 |
return self::IN_HALF_BLOCK($u, 0x0C80); |
| 1634 |
} |
| 1635 |
|
| 1636 |
public static function IS_MLYM($u) |
| 1637 |
{ |
| 1638 |
return self::IN_HALF_BLOCK($u, 0x0D00); |
| 1639 |
} |
| 1640 |
|
| 1641 |
public static function IS_SINH($u) |
| 1642 |
{ |
| 1643 |
return self::IN_HALF_BLOCK($u, 0x0D80); |
| 1644 |
} |
| 1645 |
|
| 1646 |
public static function IS_KHMR($u) |
| 1647 |
{ |
| 1648 |
return self::IN_HALF_BLOCK($u, 0x1780); |
| 1649 |
} |
| 1650 |
|
| 1651 |
public static function MATRA_POS_LEFT($u) |
| 1652 |
{ |
| 1653 |
return self::POS_PRE_M; |
| 1654 |
} |
| 1655 |
|
| 1656 |
public static function MATRA_POS_RIGHT($u) |
| 1657 |
{ |
| 1658 |
return |
| 1659 |
(self::IS_DEVA($u) ? self::POS_AFTER_SUB : |
| 1660 |
(self::IS_BENG($u) ? self::POS_AFTER_POST : |
| 1661 |
(self::IS_GURU($u) ? self::POS_AFTER_POST : |
| 1662 |
(self::IS_GUJR($u) ? self::POS_AFTER_POST : |
| 1663 |
(self::IS_ORYA($u) ? self::POS_AFTER_POST : |
| 1664 |
(self::IS_TAML($u) ? self::POS_AFTER_POST : |
| 1665 |
(self::IS_TELU($u) ? ($u <= 0x0C42 ? self::POS_BEFORE_SUB : self::POS_AFTER_SUB) : |
| 1666 |
(self::IS_KNDA($u) ? ($u < 0x0CC3 || $u > 0xCD6 ? self::POS_BEFORE_SUB : self::POS_AFTER_SUB) : |
| 1667 |
(self::IS_MLYM($u) ? self::POS_AFTER_POST : |
| 1668 |
(self::IS_SINH($u) ? self::POS_AFTER_SUB : |
| 1669 |
(self::IS_KHMR($u) ? self::POS_AFTER_POST : |
| 1670 |
self::POS_AFTER_SUB))))))))))); /* default */ |
| 1671 |
} |
| 1672 |
|
| 1673 |
public static function MATRA_POS_TOP($u) |
| 1674 |
{ |
| 1675 |
return /* BENG and MLYM don't have top matras. */ |
| 1676 |
(self::IS_DEVA($u) ? self::POS_AFTER_SUB : |
| 1677 |
(self::IS_GURU($u) ? self::POS_AFTER_POST : /* Deviate from spec */ |
| 1678 |
(self::IS_GUJR($u) ? self::POS_AFTER_SUB : |
| 1679 |
(self::IS_ORYA($u) ? self::POS_AFTER_MAIN : |
| 1680 |
(self::IS_TAML($u) ? self::POS_AFTER_SUB : |
| 1681 |
(self::IS_TELU($u) ? self::POS_BEFORE_SUB : |
| 1682 |
(self::IS_KNDA($u) ? self::POS_BEFORE_SUB : |
| 1683 |
(self::IS_SINH($u) ? self::POS_AFTER_SUB : |
| 1684 |
(self::IS_KHMR($u) ? self::POS_AFTER_POST : |
| 1685 |
self::POS_AFTER_SUB))))))))); /* default */ |
| 1686 |
} |
| 1687 |
|
| 1688 |
public static function MATRA_POS_BOTTOM($u) |
| 1689 |
{ |
| 1690 |
return |
| 1691 |
(self::IS_DEVA($u) ? self::POS_AFTER_SUB : |
| 1692 |
(self::IS_BENG($u) ? self::POS_AFTER_SUB : |
| 1693 |
(self::IS_GURU($u) ? self::POS_AFTER_POST : |
| 1694 |
(self::IS_GUJR($u) ? self::POS_AFTER_POST : |
| 1695 |
(self::IS_ORYA($u) ? self::POS_AFTER_SUB : |
| 1696 |
(self::IS_TAML($u) ? self::POS_AFTER_POST : |
| 1697 |
(self::IS_TELU($u) ? self::POS_BEFORE_SUB : |
| 1698 |
(self::IS_KNDA($u) ? self::POS_BEFORE_SUB : |
| 1699 |
(self::IS_MLYM($u) ? self::POS_AFTER_POST : |
| 1700 |
(self::IS_SINH($u) ? self::POS_AFTER_SUB : |
| 1701 |
(self::IS_KHMR($u) ? self::POS_AFTER_POST : |
| 1702 |
self::POS_AFTER_SUB))))))))))); /* default */ |
| 1703 |
} |
| 1704 |
|
| 1705 |
public static function matra_position($u, $side) |
| 1706 |
{ |
| 1707 |
switch ($side) { |
| 1708 |
case self::POS_PRE_C: return self::MATRA_POS_LEFT($u); |
| 1709 |
case self::POS_POST_C: return self::MATRA_POS_RIGHT($u); |
| 1710 |
case self::POS_ABOVE_C: return self::MATRA_POS_TOP($u); |
| 1711 |
case self::POS_BELOW_C: return self::MATRA_POS_BOTTOM($u); |
| 1712 |
} |
| 1713 |
return $side; |
| 1714 |
} |
| 1715 |
|
| 1716 |
// vowel matras that have to be split into two parts. |
| 1717 |
// From Harfbuzz (old) |
| 1718 |
// New HarfBuzz uses /src/hb-ucdn/ucdn.c and unicodedata_db.h for full method of decomposition for all characters |
| 1719 |
// Should always fully decompose and then recompose back, but we will just do the split matras |
| 1720 |
public static function decompose_indic($ab) |
| 1721 |
{ |
| 1722 |
$sub = array(); |
| 1723 |
switch ($ab) { |
| 1724 |
/* |
| 1725 |
* Decompose split matras. |
| 1726 |
*/ |
| 1727 |
/* bengali */ |
| 1728 |
case 0x9cb : $sub[0] = 0x9c7; |
| 1729 |
$sub[1] = 0x9be; |
| 1730 |
return $sub; |
| 1731 |
case 0x9cc : $sub[0] = 0x9c7; |
| 1732 |
$sub[1] = 0x9d7; |
| 1733 |
return $sub; |
| 1734 |
/* oriya */ |
| 1735 |
case 0xb48 : $sub[0] = 0xb47; |
| 1736 |
$sub[1] = 0xb56; |
| 1737 |
return $sub; |
| 1738 |
case 0xb4b : $sub[0] = 0xb47; |
| 1739 |
$sub[1] = 0xb3e; |
| 1740 |
return $sub; |
| 1741 |
case 0xb4c : $sub[0] = 0xb47; |
| 1742 |
$sub[1] = 0xb57; |
| 1743 |
return $sub; |
| 1744 |
/* tamil */ |
| 1745 |
case 0xbca : $sub[0] = 0xbc6; |
| 1746 |
$sub[1] = 0xbbe; |
| 1747 |
return $sub; |
| 1748 |
case 0xbcb : $sub[0] = 0xbc7; |
| 1749 |
$sub[1] = 0xbbe; |
| 1750 |
return $sub; |
| 1751 |
case 0xbcc : $sub[0] = 0xbc6; |
| 1752 |
$sub[1] = 0xbd7; |
| 1753 |
return $sub; |
| 1754 |
/* telugu */ |
| 1755 |
case 0xc48 : $sub[0] = 0xc46; |
| 1756 |
$sub[1] = 0xc56; |
| 1757 |
return $sub; |
| 1758 |
/* kannada */ |
| 1759 |
case 0xcc0 : $sub[0] = 0xcbf; |
| 1760 |
$sub[1] = 0xcd5; |
| 1761 |
return $sub; |
| 1762 |
case 0xcc7 : $sub[0] = 0xcc6; |
| 1763 |
$sub[1] = 0xcd5; |
| 1764 |
return $sub; |
| 1765 |
case 0xcc8 : $sub[0] = 0xcc6; |
| 1766 |
$sub[1] = 0xcd6; |
| 1767 |
return $sub; |
| 1768 |
case 0xcca : $sub[0] = 0xcc6; |
| 1769 |
$sub[1] = 0xcc2; |
| 1770 |
return $sub; |
| 1771 |
case 0xccb : $sub[0] = 0xcc6; |
| 1772 |
$sub[1] = 0xcc2; |
| 1773 |
$sub[2] = 0xcd5; |
| 1774 |
return $sub; |
| 1775 |
/* malayalam */ |
| 1776 |
case 0xd4a : $sub[0] = 0xd46; |
| 1777 |
$sub[1] = 0xd3e; |
| 1778 |
return $sub; |
| 1779 |
case 0xd4b : $sub[0] = 0xd47; |
| 1780 |
$sub[1] = 0xd3e; |
| 1781 |
return $sub; |
| 1782 |
case 0xd4c : $sub[0] = 0xd46; |
| 1783 |
$sub[1] = 0xd57; |
| 1784 |
return $sub; |
| 1785 |
/* sinhala */ |
| 1786 |
// NB Some fonts break with these Sinhala decomps (although this is Uniscribe spec) |
| 1787 |
// Can check if character would be substituted by pstf and only decompose if true |
| 1788 |
// e.g. if (isset($GSUBdata['pstf'][$ab])) - would need to pass $GSUBdata as parameter to this function |
| 1789 |
case 0xdda : $sub[0] = 0xdd9; |
| 1790 |
$sub[1] = 0xdca; |
| 1791 |
return $sub; |
| 1792 |
case 0xddc : $sub[0] = 0xdd9; |
| 1793 |
$sub[1] = 0xdcf; |
| 1794 |
return $sub; |
| 1795 |
case 0xddd : $sub[0] = 0xdd9; |
| 1796 |
$sub[1] = 0xdcf; |
| 1797 |
$sub[2] = 0xdca; |
| 1798 |
return $sub; |
| 1799 |
case 0xdde : $sub[0] = 0xdd9; |
| 1800 |
$sub[1] = 0xddf; |
| 1801 |
return $sub; |
| 1802 |
/* khmer */ |
| 1803 |
case 0x17be : $sub[0] = 0x17c1; |
| 1804 |
$sub[1] = 0x17be; |
| 1805 |
return $sub; |
| 1806 |
case 0x17bf : $sub[0] = 0x17c1; |
| 1807 |
$sub[1] = 0x17bf; |
| 1808 |
return $sub; |
| 1809 |
case 0x17c0 : $sub[0] = 0x17c1; |
| 1810 |
$sub[1] = 0x17c0; |
| 1811 |
return $sub; |
| 1812 |
|
| 1813 |
case 0x17c4 : $sub[0] = 0x17c1; |
| 1814 |
$sub[1] = 0x17c4; |
| 1815 |
return $sub; |
| 1816 |
case 0x17c5 : $sub[0] = 0x17c1; |
| 1817 |
$sub[1] = 0x17c5; |
| 1818 |
return $sub; |
| 1819 |
/* tibetan - included here although does not use Inidc shaper in other ways */ |
| 1820 |
case 0xf73 : $sub[0] = 0xf71; |
| 1821 |
$sub[1] = 0xf72; |
| 1822 |
return $sub; |
| 1823 |
case 0xf75 : $sub[0] = 0xf71; |
| 1824 |
$sub[1] = 0xf74; |
| 1825 |
return $sub; |
| 1826 |
case 0xf76 : $sub[0] = 0xfb2; |
| 1827 |
$sub[1] = 0xf80; |
| 1828 |
return $sub; |
| 1829 |
case 0xf77 : $sub[0] = 0xfb2; |
| 1830 |
$sub[1] = 0xf81; |
| 1831 |
return $sub; |
| 1832 |
case 0xf78 : $sub[0] = 0xfb3; |
| 1833 |
$sub[1] = 0xf80; |
| 1834 |
return $sub; |
| 1835 |
case 0xf79 : $sub[0] = 0xfb3; |
| 1836 |
$sub[1] = 0xf71; |
| 1837 |
$sub[2] = 0xf80; |
| 1838 |
return $sub; |
| 1839 |
case 0xf81 : $sub[0] = 0xf71; |
| 1840 |
$sub[1] = 0xf80; |
| 1841 |
return $sub; |
| 1842 |
} |
| 1843 |
return false; |
| 1844 |
} |
| 1845 |
|
| 1846 |
public static function bubble_sort(&$arr, $start, $len) |
| 1847 |
{ |
| 1848 |
if ($len < 2) { |
| 1849 |
return; |
| 1850 |
} |
| 1851 |
$k = $start + $len - 2; |
| 1852 |
while ($k >= $start) { |
| 1853 |
for ($j = $start; $j <= $k; $j++) { |
| 1854 |
if ($arr[$j]['indic_position'] > $arr[$j + 1]['indic_position']) { |
| 1855 |
$t = $arr[$j]; |
| 1856 |
$arr[$j] = $arr[$j + 1]; |
| 1857 |
$arr[$j + 1] = $t; |
| 1858 |
} |
| 1859 |
} |
| 1860 |
$k--; |
| 1861 |
} |
| 1862 |
} |
| 1863 |
|
| 1864 |
} |
| 1865 |
|