| @@ -1,433 +1,1864 @@ | ||
| 1 | -<?php | |
| 2 | - | |
| 3 | -class indic { | |
| 4 | - | |
| 5 | -function indic() { | |
| 6 | - | |
| 7 | -} | |
| 8 | - | |
| 9 | - | |
| 10 | -function substituteIndic($earr, $lang, $font) { | |
| 11 | - global $voltdata; | |
| 12 | - | |
| 13 | - if (!isset($voltdata[$font])) { | |
| 14 | - include_once(_MPDF_PATH.'includes/'.$font.'.volt.php'); | |
| 15 | - $voltdata[$font] = $volt; | |
| 16 | - } | |
| 17 | - | |
| 18 | - foreach($earr as $eid=>$char) { | |
| 19 | - $earr[$eid] = sprintf("%04s", strtoupper(dechex($char))); | |
| 20 | - } | |
| 21 | - $vstr = "0020 ".implode(" ",$earr)." 0020"; | |
| 22 | - //============================ | |
| 23 | - // Common Indic Punctuation marks | |
| 24 | - // If NOT devanagari | |
| 25 | - if ($lang!='hi') { | |
| 26 | - $vstr = str_replace('0964','007C', $vstr); // U+0964 replace with "|" | |
| 27 | - $vstr = str_replace('0965','007C 007C', $vstr); // U+0964 replace with "|" | |
| 28 | - } | |
| 29 | - //============================ | |
| 30 | - // Tamil numeral for Zero missing Added mPDF 4.2 | |
| 31 | - if ($lang=='ta') { | |
| 32 | - $vstr = str_replace('0BE6','0030', $vstr); // U+0BEB replace with "0" | |
| 33 | - } | |
| 34 | - | |
| 35 | - //============================ | |
| 36 | - // Re-order vowels | |
| 37 | - | |
| 38 | - // DEVANAGARI vowel sign matraI[093F] before consonant | |
| 39 | - if ($lang=='hi') { | |
| 40 | - $prebasedvowels = "(093F)"; | |
| 41 | - $nukta = "093C"; | |
| 42 | - $halant = "094D"; | |
| 43 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 44 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.' '.$nukta.'/','\\2 \\1 '.$nukta, $vstr); // before NUKTA | |
| 45 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$prebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT == VIRAMA | |
| 46 | - } | |
| 47 | - | |
| 48 | - // BENGALI vowels [09BF 09C7 09C8] | |
| 49 | - else if ($lang=='bn') { | |
| 50 | - | |
| 51 | - // Khanda Ta 09CE not in font -> replace with 09A4|09CD | |
| 52 | - $vstr = preg_replace('/09CE/','09A4 09CD 200D', $vstr); // mPDF 5.3.09 | |
| 53 | - | |
| 54 | - // BENGALI double-part vowels [09CB 09C7 09BE][09CC 09C7 09D7] | |
| 55 | - $vstr = str_replace('09CB','09C7 09BE', $vstr); // convert to 2 parts | |
| 56 | - $vstr = str_replace('09CC','09C7 09D7', $vstr); // 09C7 pre-based is then shifted below | |
| 57 | - $prebasedvowels = "(09BF|09C7|09C8)"; | |
| 58 | - $nukta = "09BC"; | |
| 59 | - $halant = "09CD"; | |
| 60 | - // mPDF 5.0.044 | |
| 61 | - $bnfullcons = "0995|0996|0997|0998|0999|099A|099B|099C|099D|099F|09A0|09A1|09A2|09A3|09A4|09A5|09A6|09A7|09A8|09AA|09AB|09AC|09AD|09AE|09AF|09B0|09B2|09B6|09B7|09B8|09B9|09DC|09DD|09DF"; | |
| 62 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 63 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.' '.$nukta.'/','\\2 \\1 '.$nukta, $vstr); // before NUKTA | |
| 64 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$prebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT | |
| 65 | - // mPDF 5.0.044 | |
| 66 | - // .. and shifting left-based vowel further to the left in case 3 consonants together. | |
| 67 | - $vstr = preg_replace('/('.$bnfullcons.') '.$halant.' '.$prebasedvowels.'/','\\2 \\1 '.$halant, $vstr); | |
| 68 | - | |
| 69 | - // mPDF 5.0.044 | |
| 70 | - // If left-based vowel has now been shifted to left of RA/Halant (09B0/09CD) | |
| 71 | - // Convert here to above-line form (E068) as it would get missed later | |
| 72 | - // e.g. 09B0 09CD 09AD 09C7 would be changed above => | |
| 73 | - // e.g. 09C7 09B0 09CD 09AD. The 09B0 09CD should => E068 | |
| 74 | - // ??? need to add 09BF as well (09BF|09C7|09C8) | |
| 75 | - $vstr = preg_replace('/(09C7|09C8) 09B0 09CD/', '\\1 E068', $vstr); | |
| 76 | - | |
| 77 | - } | |
| 78 | - | |
| 79 | - // GUJARATI pre-based vowel [0ABF] | |
| 80 | - else if ($lang=='gu') { | |
| 81 | - $prebasedvowels = "(0ABF)"; | |
| 82 | - $nukta = "0ABC"; | |
| 83 | - $halant = "0ACD"; | |
| 84 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 85 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.' '.$nukta.'/','\\2 \\1 '.$nukta, $vstr); // before NUKTA | |
| 86 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$prebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT | |
| 87 | - } | |
| 88 | - | |
| 89 | - // GURMUKHI/PUNJABI pre-based vowel [0ABF] | |
| 90 | - else if ($lang=='pa') { | |
| 91 | - $prebasedvowels = "(0A3F)"; | |
| 92 | - $nukta = "0A3C"; | |
| 93 | - $halant = "0A4D"; | |
| 94 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 95 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.' '.$nukta.'/','\\2 \\1 '.$nukta, $vstr); // before NUKTA | |
| 96 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$prebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT | |
| 97 | - } | |
| 98 | - | |
| 99 | - // TAMIL pre-based vowel [0ABF] | |
| 100 | - else if ($lang=='ta') { | |
| 101 | - // Shrii (Shree) | |
| 102 | - $vstr = preg_replace('/0BB6 0BCD 0BB0 0BC0/','E04B', $vstr); | |
| 103 | - | |
| 104 | - // TAMIL double-part vowels [0BCA 0BC6 0BBE][0BCB 0BC7 0BBE][0BCC 0BC6 0BD7] | |
| 105 | - $vstr = preg_replace('/0BCA/','0BC6 0BBE', $vstr); // convert to 2 parts | |
| 106 | - $vstr = preg_replace('/0BCB/','0BC7 0BBE', $vstr); // pre-based is then shifted below | |
| 107 | - $vstr = preg_replace('/0BCC/','0BC6 0BD7', $vstr); | |
| 108 | - $prebasedvowels = "(0BC6|0BC7|0BC8)"; | |
| 109 | - // No nukta | |
| 110 | - $halant = "0BCD"; // Doesn't seem to move most in front of halanted consonants | |
| 111 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$prebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 112 | - // ? Only for special case KSS (already moved to left of 0BB7) | |
| 113 | - $vstr = preg_replace('/0B95 '.$halant.' '.$prebasedvowels.' 0BB7/','\\1 0B95 '.$halant.' 0BB7', $vstr); | |
| 114 | - } | |
| 115 | - | |
| 116 | - // ORIYA | |
| 117 | - else if ($lang=='or') { | |
| 118 | - // ORIYA double-part vowels [] | |
| 119 | - $vstr = str_replace('0B48','0B47 0B56', $vstr); // 2-part Vowel | |
| 120 | - $vstr = str_replace('0B4B','0B47 0B3E', $vstr); // 2-part Vowel | |
| 121 | - $vstr = str_replace('0B4C','0B47 0B57', $vstr); // 2-part Vowel | |
| 122 | - $orprebasedvowels = "(0B47)"; | |
| 123 | - // No nukta | |
| 124 | - $halant = "0B4D"; | |
| 125 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$orprebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 126 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$orprebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT | |
| 127 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$orprebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT | |
| 128 | - } | |
| 129 | - | |
| 130 | - // MALAYALAM | |
| 131 | - else if ($lang=='ml') { | |
| 132 | - // Chillus - old forms - remove ZWNJ after | |
| 133 | - // This font Volt rules recognises e.g. "Na Halant(Virama)" as ChilluN | |
| 134 | - $vstr = preg_replace('/(0D23 0D4D|0D28 0D4D|0D30 0D4D|0D32 0D4D|0D33 0D4D) 200D/','\\1', $vstr); | |
| 135 | - // See Chillus in Unicode [http://en.wikipedia.org/wiki/Malayalam_script] | |
| 136 | - $vstr = str_replace('0D7A','0D23 0D4D', $vstr); // [mlymChilluNn] | |
| 137 | - $vstr = str_replace('0D7B','0D28 0D4D', $vstr); // [mlymChilluN] | |
| 138 | - $vstr = str_replace('0D7C','0D30 0D4D', $vstr); // [mlymChilluR] | |
| 139 | - $vstr = str_replace('0D7D','0D32 0D4D', $vstr); // [mlymChilluL] | |
| 140 | - $vstr = str_replace('0D7E','0D33 0D4D', $vstr); // [mlymChilluLl] | |
| 141 | -/* | |
| 142 | - // Chillus - 0D7A-0D7E not in font directly, but as E005-E009 | |
| 143 | - $vstr = preg_replace('/0D23 0D4D 200D/','0D7A', $vstr); | |
| 144 | - $vstr = preg_replace('/0D28 0D4D 200D/','0D7B', $vstr); | |
| 145 | - $vstr = preg_replace('/0D30 0D4D 200D/','0D7C', $vstr); | |
| 146 | - $vstr = preg_replace('/0D32 0D4D 200D/','0D7D', $vstr); | |
| 147 | - $vstr = preg_replace('/0D33 0D4D 200D/','0D7E', $vstr); | |
| 148 | - | |
| 149 | - $vstr = preg_replace('/0D7F/','E004', $vstr); // [mlymChilluK] | |
| 150 | - $vstr = preg_replace('/0D7A/','E005', $vstr); // [mlymChilluNn] | |
| 151 | - $vstr = preg_replace('/0D7B/','E006', $vstr); // [mlymChilluN] | |
| 152 | - $vstr = preg_replace('/0D7C/','E007', $vstr); // [mlymChilluR] | |
| 153 | - $vstr = preg_replace('/0D7D/','E008', $vstr); // [mlymChilluL] | |
| 154 | - $vstr = preg_replace('/0D7E/','E009', $vstr); // [mlymChilluLl] | |
| 155 | -*/ | |
| 156 | - | |
| 157 | - // MALAYALAM double-part vowels [] | |
| 158 | - $vstr = str_replace('0D4A','0D46 0D3E', $vstr); // 2-part Vowel | |
| 159 | - $vstr = str_replace('0D4B','0D47 0D3E', $vstr); // 2-part Vowel | |
| 160 | - $vstr = str_replace('0D4C','0D46 0D57', $vstr); // 2-part Vowel | |
| 161 | - $mlprebasedvowels = "(0D46|0D47|0D48)"; | |
| 162 | - // No nukta | |
| 163 | - $halant = "0D4D"; | |
| 164 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$mlprebasedvowels.'/','\\2 \\1', $vstr); // vowel sign pre-based shift left | |
| 165 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$mlprebasedvowels.'/','\\2 \\1 '.$halant, $vstr); // before CHAR HALANT | |
| 166 | - } | |
| 167 | - | |
| 168 | - // TELUGU | |
| 169 | - else if ($lang=='te') { | |
| 170 | - // TELUGU double-part vowels [0C48 -> 0C46 0C56] | |
| 171 | - $vstr = str_replace('0C48','0C46 0C56', $vstr); // 2-part Vowel | |
| 172 | - $prebasedvowels = "(0C46)"; | |
| 173 | - $abvvowels = "(0C3E|0C3F|0C40|0C46|0C47|0C4A|0C4B|0C4C|0C55)"; | |
| 174 | - // No nukta | |
| 175 | - $halant = "0C4D"; | |
| 176 | - $tefullforms = "0C15|0C17|0C18|0C1A|0C1B|0C1C|0C1D|0C20|0C21|0C22|0C24|0C25|0C26|0C27|0C28|0C2A|0C2B|0C2D|0C2E|0C2F|0C30|0C33|0C35|0C36|0C37|0C38|0C39|E028|E029|E02A|E02B|E078|E07A|E07B"; | |
| 177 | - $vstr = preg_replace('/('.$tefullforms .') '.$halant.' ('.$tefullforms .') '.$abvvowels .'/','\\1 \\3 '.$halant.' \\2', $vstr); // before HALANT | |
| 178 | - } | |
| 179 | - | |
| 180 | - | |
| 181 | - // KANNADA | |
| 182 | - else if ($lang=='kn') { | |
| 183 | - // KANNADA double-part vowels [0CC8 -> 0CC6 0CD6] | |
| 184 | - $vstr = str_replace('0CC0','0CBF 0CD5', $vstr); // 2-part Vowel | |
| 185 | - $vstr = str_replace('0CC7','0CC6 0CD5', $vstr); // 2-part Vowel | |
| 186 | - $vstr = str_replace('0CC8','0CC6 0CD6', $vstr); // 2-part Vowel AI - no glyph for single | |
| 187 | - $vstr = str_replace('0CCA','0CC6 0CC2', $vstr); // 2-part Vowel | |
| 188 | - $vstr = str_replace('0CCB','0CC6 0CC2 0CD5', $vstr); // 2-part Vowel | |
| 189 | - $prebasedvowels = "(0CBF|0CC6)"; | |
| 190 | - $halant = "0CCD"; | |
| 191 | - } | |
| 192 | - | |
| 193 | - | |
| 194 | - //============================ | |
| 195 | - | |
| 196 | - // SPECIALS | |
| 197 | - | |
| 198 | - // DEVANAGARI Ra Halant Ra | |
| 199 | - if ($lang=='hi') { | |
| 200 | - $vstr = str_replace('0930 094D 0930','E05D 0930', $vstr); // Ra Halant Ra => halfRa FullRa | |
| 201 | - } | |
| 202 | - | |
| 203 | - // GUJARATI | |
| 204 | - if ($lang=='gu') { | |
| 205 | - $vstr = str_replace('0AB0 0AC2','E02E', $vstr); // Ra VowelUu => SpecialForm RaUu | |
| 206 | - } | |
| 207 | - | |
| 208 | - // TELUGU Ra Halant <Consonant> Halant => halfRa Halant<Consonant> Halant | |
| 209 | - if ($lang=='te') { | |
| 210 | - $vstr = preg_replace('/0C30 0C4D ([A-F0-9]{4}) 0C4D/','E021 0C4D \\1 0C4D', $vstr); | |
| 211 | - } | |
| 212 | - | |
| 213 | - // KANNADA | |
| 214 | - // Reph at end of word becomes E0CC instead of E00B | |
| 215 | - if ($lang=='kn') { | |
| 216 | - $vstr = str_replace('0CB0 0CCD 0020','E0CC 0020', $vstr); // mPDF 5.3.87 | |
| 217 | - } | |
| 218 | - | |
| 219 | - | |
| 220 | - //============================ | |
| 221 | - // MAIN BIT FROM VOLT RULES | |
| 222 | - foreach($voltdata[$font] AS $rid=>$reps) { | |
| 223 | -//echo $rid . ': ' . $vstr.'<br />'; | |
| 224 | - $vstr = preg_replace('/'.$reps['match'].'/',$reps['replace'], $vstr); | |
| 225 | - } | |
| 226 | -//echo $vstr.'<br />'; exit; | |
| 227 | - | |
| 228 | - | |
| 229 | - //============================ | |
| 230 | - | |
| 231 | - // SPECIALS | |
| 232 | - | |
| 233 | - // KANNADA | |
| 234 | - // <Base> <BelowBase1> [<BelowBase2> ] MatraI -> <Base/MatraI ligature> <Belowbase1> etc | |
| 235 | - if ($lang=='kn') { | |
| 236 | - $matraI = "0CBF"; | |
| 237 | - $knbase = preg_split('/\|/', "0C95|0C96|0C97|0C98|0C9A|0C9B|0C9C|0C9D|0CA0|0CA1|0CA2|0CA3|0CA4|0CA5|0CA6|0CA7|0CA8|0CAA|0CAB|0CAC|0CAD|0CAE|0CAF|0CB0|0CB2|0CB3|0CB5|0CB6|0CB7|0CB8|0CB9|E0A3|E07D|E07E"); | |
| 238 | - $knmatraIligature = preg_split('/\|/', "E082|E083|E084|E085|E086|E087|E088|E089|E08A|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A4|E0A1|E0A2"); | |
| 239 | - $belowbase1 = "E02E|E02F|E030|E031|E032|E033|E034|E035|E036|E037|E038|E039|E03A|E03B|E03C|E03D|E03E|E03F|E040|E041|E042|E043|E044|E045|E046|E047|E048|E049|E04A|E04B|E04C|E04D|E04E|E04F|E050|E081"; | |
| 240 | - $belowbase2 = "E052|E053|E054|E055|E056|E057|E058|E059|E05A|E05B|E05C|E05D|E05E|E05F|E060|E061|E062|E063|E064|E065|E066|E067|E068|E069|E06A|E06B|E06C|E06D|E06E|E06F|E070|E071|E072|E073|E074|E081"; | |
| 241 | - for ($i=0; $i<count($knbase);$i++) { | |
| 242 | - $vstr = preg_replace('/'.$knbase[$i].' ('.$belowbase1.') ('.$belowbase2.') '.$matraI.'/', $knmatraIligature[$i].' \\1 \\2', $vstr); | |
| 243 | - $vstr = preg_replace('/'.$knbase[$i].' ('.$belowbase1.') '.$matraI.'/', $knmatraIligature[$i].' \\1', $vstr); | |
| 244 | - } | |
| 245 | - } | |
| 246 | - | |
| 247 | - // KANNADA | |
| 248 | - // [KanTtaFull] [matraI] => [KanTtaPartial] [matraI] | |
| 249 | - if ($lang=='kn') { | |
| 250 | - $vstr = preg_replace('/0C9F '.$matraI.'/', 'E015 '.$matraI, $vstr); | |
| 251 | - } | |
| 252 | - | |
| 253 | - // ORIYA | |
| 254 | - if ($lang=='or') { | |
| 255 | - // SpecialCase Ra[0B30] Halant still left before [oryaFullNnNna] => E00F | |
| 256 | - $vstr = preg_replace('/0B30 '.$halant.' E00F/','E00F E069', $vstr); // convert to Reph | |
| 257 | - } | |
| 258 | - | |
| 259 | - //============================ | |
| 260 | - // SHIFT REPH | |
| 261 | - | |
| 262 | - // DEVANAGARI Shift Reph [E015] | |
| 263 | - if ($lang=='hi') { | |
| 264 | - // FIRSTLY - halfRa = E05D - Change this to Reph [E015] | |
| 265 | - $himatchhalfforms = "E043|E044|E045|E046|E047|E048|E049|E04A|E04B|E04C|E04D|E04E|E04F|E050|E051|E052|E053|E054|E055|E056|E057|E058|E059|E05A|E05B|E05C|E05D|E05E|E05F|E060|E061|E062|E063|E064|E065|E066|E067|E068|E069|E06A|E06B|E06C|E06D|E06E|E06F|E070|E071|E072|E073|E074|E075|E076|E077|E078|E079|E07A|E07B|E07C|E07D|E07E|E07F|E080|E081|E082|E083|E084|E085|E086|E087|E088|E089|E08A|E0D3|E0D4|E0D5|E0D6|E0D7|E0D8|E0D9|E0DA|E0DB|E0DC|E0DD|E0DE|E0DF|E0E0|E0E1|E0E2|E0E3|E0E4|E0E5|E0E6|E0E7|E0E8|E0E9|E0EA|E0EB|E0EC|E0ED|E0EE|E0EF|E0F0|E0F1|E0F2|E0F3|E0F4|E0F5|E0F6|E0F7|E0F8|E0F9|E0FA|E0FB|E0FC|E0FD|E0FE|E0FF|E100|E101|E102|E103|E104|E105|E106|E107|E108|E109|E10A|E10B|E10C|E10D|E10E|E10F|E110|E111|E112|E113|E114|E115|E116|E117|E118|E119|E11A|E13D|E13E|E13F|E140|E141|E142|E143|E144|E145"; | |
| 266 | - $himatchfullforms = "0915|0916|0917|0918|0919|091A|091B|091C|091D|091E|091F|0920|0921|0922|0923|0924|0925|0926|0927|0928|092A|092B|092C|092D|092E|092F|0930|0932|0933|0935|0936|0937|0938|0939|E028|E029|0958|0959|095A|E02A|E02B|E02C|E02D|095B|E02E|E02F|E030|E031|095C|095D|E032|E033|E034|E035|E036|0929|E037|095E|E038|E039|E03A|095F|0931|E03B|0934|E03C|E03D|E03E|E03F|E040|E041|E042|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A1|E0A2|E0A3|E0A4|E0A5|E0A6|E0A7|E0A8|E0A9|E0AA|E0AB|E0AC|E0AD|E0AE|E0AF|E0B0|E0B1|E0B2|E0B3|E0B4|E0B5|E0B6|E0B7|E0B8|E0B9|E0BA|E0BB|E0BC|E0BD|E0BE|E0BF|E0C0|E0C1|E0C2|E0C3|E0C4|E0C5|E0C6|E0C7|E0C8|E0C9|E0CA|E0CB|E0CC|E0CD|E0CE|E0CF|E0D0|E0D1|E0D2|E11E|E11F|E120|E121|E122|E123|E124|E125|E126|E127|E128|E129|E12A|E12B|E12C|E12D|E12E|E12F|E130|E131|E132|E133"; | |
| 267 | - $vstr = preg_replace('/E05D ('.$himatchhalfforms.'|'.$himatchfullforms.')/', 'E015 \\1', $vstr); | |
| 268 | - | |
| 269 | - // Reph = E015 - Shift Right to just after end of syllable | |
| 270 | - // FullAllForms + HalfAllForms + 093E matraA | |
| 271 | - while(preg_match('/E015 ('.$himatchhalfforms.')/', $vstr)) { | |
| 272 | - $vstr = preg_replace('/E015 ('.$himatchhalfforms.')/', '\\1 E015', $vstr); | |
| 273 | - } | |
| 274 | - $vstr = preg_replace('/E015 ('.$himatchfullforms.')/', '\\1 E015', $vstr); | |
| 275 | - | |
| 276 | - // Now shift it beyond post-based vowels // ??? Need to add others e.g. 0949,094A,094B,094C + presentation forms like E198 | |
| 277 | - $vstr = str_replace('E015 093E', '093E E015', $vstr); | |
| 278 | - $vstr = preg_replace('/E015 (0940|E194|E195|E196|E197|E198)/', '\\1 E014', $vstr); // (Small) reph [E014] to Right of matraI | |
| 279 | - $vstr = str_replace('E015 0947', '0947 E014', $vstr); // (Small) reph [E014] to Right of matraI | |
| 280 | - } | |
| 281 | - | |
| 282 | - // BENGALI Shift Reph [E068] | |
| 283 | - else if ($lang=='bn') { | |
| 284 | - $bnfullconjuncts = "E002|E003|E004|E041|E042|E043|E044|E045|E046|E047|E048|E049|E04A|E04B|E04C|E04D|E04E|E04F|E050|E051|E052|E053|E054|E055|E056|E057|E058|E059|E05A|E05B|E05C|E05D|E05E|E05F|E060|E061|E062|E063|E064|E065|E06A|E06B|E06C|E06D|E06E|E06F|E070|E071|E072|E073|E074|E075|E076|E077|E078|E079|E07A|E07B|E07C|E07D|E07E|E07F|E080|E081|E082|E083|E084|E085|E086|E087|E088|E089|E08A|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A1|E0A2|E0A3|E0A4|E0A5|E0A6|E0A7|E0A8|E0A9|E0AA|E0AB|E0AC|E0AD|E0AE|E0AF|E0B0|E0B1|E0B2|E0B3|E0B4|E0B5|E0B6|E0B7|E0B8|E0B9|E0BA|E0BB|E0BC|E0BD|E0BE|E0BF|E0C0|E0C1|E0C2|E0C3|E0C4|E0C5|E0C6|E0C7|E0C8|E0C9|E0CA|E0CB|E0CC|E0CD|E0CE|E0CF|E0D0|E0D1|E0D2|E0D3|E0D4|E0D5|E0D6|E0D7|E0D8|E0D9|E0DA|E0DB|E0DC|E0DD|E0DE|E0DF|E0E0|E0E1|E0E2|E0E3|E0E4|E0E5|E0E6|E0E7|E0E8|E0E9|E0EA|E0EB|E0EC|E0ED|E0EE|E0EF|E0F0|E0F1|E0F2|E0F3|E0F4|E0F5|E0F6|E0F7|E0F8|E0F9|E0FA|E0FB|E0FC|E0FD|E0FE|E0FF|E100|E101|E102|E103|E104|E105|E106|E107|E108|E109|E10A|E10B|E10C|E10D|E10E|E10F|E110|E111|E112|E113|E114|E115|E116|E117|E118|E119|E11A|E11B|E11C|E11D|E11E|E11F|E120|E121|E122|E123|E124|E125|E126|E127|E128|E129|E12A|E12B|E12C|E12D|E12E|E12F|E130|E131|E132|E133|E134|E135|E136|E137|E138|E139|E13A|E13B|E13C|E13D|E13E|E13F|E140|E141|E142|E143|E144|E145|E146|E147|E148|E149|E14A|E14B|E14C|E14D|E14E|E14F|E150|E151|E152|E153|E154|E155|E156|E157|E158|E159|E15A|E15B|E15C|E15D|E15E|E15F|E160|E161|E162|E163|E164|E165|E166|E167|E168|E169|E16A|E16B|E16C|E16D|E16E|E16F|E170|E171|E172|E173|E174|E175|E176|E177|E178|E179|E17A|E17B|E17C|E17D|E17E|E17F|E180|E181|E182|E183|E184|E185|E186|E187|E188|E189|E18A|E18B|E18C|E18D|E18E|E18F|E190|E191|E192|E193|E194|E195|E196|E197|E198|E199|E19A"; | |
| 285 | - // $bnfullcons - set above; | |
| 286 | - $vstr = preg_replace('/E068 ('.$bnfullconjuncts.'|'.$bnfullcons.')/', '\\1 E068', $vstr); | |
| 287 | - // ? Need to shift it beyond post-base vowels 09BE, 09C0, 09D7 haven't found so can't test?? | |
| 288 | - $vstr = preg_replace('/E068 (09BE|09C0|09D7)/', '\\1 E068', $vstr); | |
| 289 | - } | |
| 290 | - | |
| 291 | - // GUJARATI Shift Reph [E032] | |
| 292 | - else if ($lang=='gu') { | |
| 293 | - $gufullforms = "0A95|0A96|0A97|0A98|0A99|0A9A|0A9B|0A9C|0A9D|0A9E|0A9F|0AA0|0AA1|0AA2|0AA3|0AA4|0AA5|0AA6|0AA7|0AA8|0AAA|0AAB|0AAC|0AAD|0AAE|0AAF|0AB0|0AB2|0AB3|0AB5|0AB6|0AB7|0AB8|0AB9|E002|E003|E004|E005|E006|E007|E008|E009|E00A|E00B|E00C|E00D|E00E|E00F|E010|E011|E012|E013|E014|E015|E016|E017|E018|E019|E01A|E01B|E01C|E01D|E01E|E01F|E020|E021|E022|E023|E024|E025|E026|E027|E05E|E05F|E060|E061|E062|E063|E064|E065|E066|E067|E068|E069|E06A|E06B|E06C|E06D|E06E|E06F|E070|E071|E072|E073|E074|E075|E076|E077|E078|E079|E07A|E07B|E07C|E07D|E07E|E07F|E080|E081|E082|E083|E084|E085|E086|E087|E088|E089|E08A|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A1|E0A2|E0A3|E0A4|E0A5"; | |
| 294 | - $vstr = preg_replace('/E032 ('.$gufullforms.')/', '\\1 E032', $vstr); | |
| 295 | - // Now shift it beyond post-based vowels // ??? Need to add others e.g. 0949,094A,094B,094C + presentation forms like E198 | |
| 296 | - // ? Need to shift it beyond post-base vowels 0ABE, 0AC0 haven't found so can't test?? | |
| 297 | - $vstr = preg_replace('/E032 (0ABE|0AC0)/', '\\1 E032', $vstr); | |
| 298 | - } | |
| 299 | - | |
| 300 | - | |
| 301 | - // TELUGU Shift Reph to LEFT [E046|E069|E077] [TelRaSmallOne] => E046 [TelRaSmallTwo] => E069 [TelRaSmallThree] => E077 | |
| 302 | - else if ($lang=='te') { | |
| 303 | - // tefullforms defined earlier | |
| 304 | - $tepartialforms = "E00D|E00E|E00F|E010|E011|E012|E013|E014|E015|E016|E017|E018|E019|E01A|E01B|E01C|E01D|E01E|E01F|E020|E021|E022|E023|E024|E025|E026|E027|E07C|E07D|E07E"; | |
| 305 | - $matraligs = "E07F|E080|E081|E082|E083|E084|E085|E086|E087|E088|E089|E08A|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A1|E0A2|E0A3|E0A4|E0A5|E0A6|E0A7|E0A8|E0A9|E0AA|E0AB|E0AC|E0AD|E0AE|E0AF"; | |
| 306 | - $tevowels = "0C3E|0C3F|0C40|0C46|0C47|0C56|0C4A|0C4B|0C4C" | |
| 307 | - ."|0C41|0C42|0C43|0C44"; // post matras | |
| 308 | - $vstr = preg_replace('/('.$tevowels.') (E046|E069|E077)/', '\\2 \\1', $vstr); | |
| 309 | - while(preg_match('/('.$tepartialforms.') (E046|E069|E077)/', $vstr)) { | |
| 310 | - $vstr = preg_replace('/('.$tepartialforms.') (E046|E069|E077)/', '\\2 \\1', $vstr); | |
| 311 | - } | |
| 312 | - $vstr = preg_replace('/('.$tefullforms .'|'.$matraligs.') (E046|E069|E077)/', '\\2 \\1', $vstr); | |
| 313 | - } | |
| 314 | - | |
| 315 | - | |
| 316 | - // KANNADA Shift Reph to RIGHT [E00B] | |
| 317 | - else if ($lang=='kn') { | |
| 318 | - $knfullforms = "0C95|0C96|0C97|0C98|0C99|0C9A|0C9B|0C9C|0C9D|0C9E|0C9F|0CA0|0CA1|0CA2|0CA3|0CA4|0CA5|0CA6|0CA7|0CA8|0CAA|0CAB|0CAC|0CAD|0CAE|0CAF|0CB0|0CB1|0CB2|0CB3|0CB5|0CB6|0CB7|0CB8|0CB9|E07D|E07E|E0A3"; | |
| 319 | - $knpartialforms = "E00C|E00D|E00E|E00F|E010|E011|E012|E013|E014|0C9E|E015|E016|E017|E018|E019|E01A|E01B|E01C|E01D|E01E|E01F|E020|E021|E022|E023|E024|E025|E026|E027|E028|E029|E02A|E02B|E02C|E02D|E07F"; | |
| 320 | - while(preg_match('/E00B ('.$knpartialforms.')/', $vstr)) { | |
| 321 | - $vstr = preg_replace('/E00B ('.$knpartialforms.')/', '\\1 E00B', $vstr); | |
| 322 | - } | |
| 323 | - // mPDF 5.3.47 Also move Reph to right of matraIligatures | |
| 324 | - $knfullforms .= "|E082|E083|E084|E085|E086|E087|E088|E089|E08A|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A4|E0A1|E0A2"; | |
| 325 | - $vstr = preg_replace('/E00B ('.$knfullforms.')/', '\\1 E00B', $vstr); | |
| 326 | - | |
| 327 | - // ? Need to shift it beyond base or below-base forms - haven't found so can't test?? | |
| 328 | - // mPDF 5.3.87 | |
| 329 | - // E004 added to list (which is a transformed version of 0CBE) | |
| 330 | - $knvowels = "0CBE|0CC0|0CC1|0CC2|0CC3|0CC4|0CC7|0CC8|0CCA|0CCB|0CD5|0CD6|E004"; | |
| 331 | - $vstr = preg_replace('/E00B ('.$knvowels.')/', '\\1 E00B', $vstr); | |
| 332 | - } | |
| 333 | - | |
| 334 | - | |
| 335 | - // ORIYA Shift Reph to RIGHT [E069|E06A|E06B|E06C] | |
| 336 | - else if ($lang=='or') { | |
| 337 | - $orrephs = "E069|E06A|E06B|E06C"; | |
| 338 | - $orfullforms = "0B15|0B16|0B17|0B18|0B19|0B1A|0B1B|0B1C|0B1D|0B1E|0B1F|0B20|0B21|0B22|0B23|0B24|0B25|0B26|0B27|0B28|0B29|0B2A|0B2B|0B2C|0B2D|0B2E|0B2F|0B30|0B31|0B32|0B33|0B34|0B35|0B36|0B37|0B38|E003|E004|E005|E006|E007|E008|E009|E00A|E00B|E00C|E00D|E00E|E00F|E010|E011|E012|E013|E014|E015|E016|E017|E018|E019|E01A|E01B|E01C|E01D|E01E|E01F|E020|E021|E022|E023|E024|E025|E026|E027|E028|E029|E02A|E02B|E02C|E02D|E02E|E02F|E030|E031|E032|E033|E034|E035|E036|E037"; | |
| 339 | - // E123 - E147 FullHalant forms ? add to FullForms | |
| 340 | - $orpartialforms = "E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A1|E0A2|E0A3|E0A4|E0A5|E0A6|E0A7|E0A8|E0A9|E0AA|E0AB|E0AC|E0AD|E0AE|E0AF|E0B0|E0B1|E0B2|E0B3|E0B4|E0B5|E0B6|E0B7|E0B8|E0B9|E0BA|E0BB|E0BC|E0BD|E0BE|E0BF|E0C0|E0C1|E0C2|E0C3|E0C4|E0C5|E0C6|E0C7|E0C8|E0C9|E0CA|E0CB|E0CC|E0CD|E0CE|E0CF|E0D0|E0D1|E0D2|E0D3|E0D4|E0D5|E0D6|E0D7|E0D8|E0D9|E0DA|E0DB|E0DC|E0DD|E0DE|E0DF|E0E0|E0E1|E0E2|E0E3|E0E4|E0E5|E0E6|E0E7|E0E8|E0E9|E0EA|E0EB|E0EC|E0ED|E0EE|E0EF|E0F0|E0F1|E0F2|E0F3|E0F4|E0F5"; | |
| 341 | - | |
| 342 | - // Combined MatraIReph[E06D] split [0B3F & E069] to allow reph to be shifted forwards | |
| 343 | - $vstr = preg_replace('/('.$orfullforms.') E06D ('.$orfullforms.') 0B3E/', '\\1 0B3F E069 \\2 0B3E', $vstr); | |
| 344 | - | |
| 345 | - | |
| 346 | - while(preg_match('/('.$orrephs.') ('.$orpartialforms.')/', $vstr)) { | |
| 347 | - $vstr = preg_replace('/('.$orrephs.') ('.$orpartialforms.')/', '\\2 \\1', $vstr); | |
| 348 | - } | |
| 349 | - $vstr = preg_replace('/('.$orrephs.') ('.$orfullforms.')/', '\\2 \\1', $vstr); | |
| 350 | - | |
| 351 | - | |
| 352 | - // Combine Reph and MatraI | |
| 353 | - $vstr = str_replace('E069 0B3F', 'E06D', $vstr); // Reph and MatraI -> MatraIReph | |
| 354 | - $vstr = str_replace('E06A 0B3F', 'E06E', $vstr); // Reph and MatraI -> MatraIReph | |
| 355 | - $vstr = str_replace('E06B 0B3F', 'E06F', $vstr); // Reph and MatraI -> MatraIReph | |
| 356 | - } | |
| 357 | - | |
| 358 | - | |
| 359 | - // MALAYALAM Shift Reph to LEFT [E00E] (mlylmRaVattu) | |
| 360 | - else if ($lang=='ml') { | |
| 361 | - $halant = "0D4D"; | |
| 362 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' 0D30/','E00E \\1', $vstr); // 0D30 = Ra | |
| 363 | - $vstr = preg_replace('/([A-F0-9]{4}) '.$halant.' '.$mlprebasedvowels .' 0D30/','\\2 E00E \\1', $vstr); // 0D30 = Ra | |
| 364 | - | |
| 365 | - $mlfullforms = "0D15|0D16|0D17|0D18|0D19|0D1A|0D1B|0D1C|0D1D|0D1E|0D1F|0D20|0D21|0D22|0D23|0D24|0D25|0D26|0D27|0D28|0D2A|0D2B|0D2C|0D2D|0D2E|0D2F|0D30|0D31|0D32|0D33|0D34|0D35|0D36|0D37|0D38|0D39" | |
| 366 | - ."|E010|E011|E012|E013|E014|E015|E016|E017|E018|E019|E01A|E01B|E01C|E01D|E01E|E01F|E020|E021|E022|E023|E024|E025|E026|E027|E028|E029|E02A|E02B|E02C|E02D|E02E|E02F|E030|E031|E032|E033|E034|E035|E036|E037|E038|E039|E03A|E03B|E03C|E03D|E03E|E03F|E040|E041|E042|E043|E044|E045|E046|E047|E048|E049|E04A|E04B|E04C|E04D|E04E|E04F|E050|E051|E052|E053|E054|E055|E056|E057|E058|E059|E05A|E05B|E05C|E05D|E05E|E05F|E060|E061|E062|E063|E064|E065|E066|E067|E068|E069|E06A|E06B|E06C|E06D|E06E|E06F|E070|E071|E072|E073|E074|E075|E076|E077|E078|E079|E07A|E07B|E07C|E07D"; | |
| 367 | - // = FullConsonants + FullConjuncts | |
| 368 | - | |
| 369 | - // = Add Chillu characters // mPDF 5.0.024 | |
| 370 | - $mlfullforms .= "|E004|E005|E006|E007|E008|E009"; | |
| 371 | - while(preg_match('/('.$mlfullforms.') E00E/', $vstr)) | |
| 372 | - $vstr = preg_replace('/('.$mlfullforms.') E00E/', 'E00E \\1', $vstr); | |
| 373 | - } | |
| 374 | - | |
| 375 | - //============================ | |
| 376 | - | |
| 377 | - // SHIFT post-based vowels to Left of SmallForms (NOT to left of full forms) | |
| 378 | - | |
| 379 | - // TELUGU Shift | |
| 380 | - if ($lang=='te') { | |
| 381 | - // NB $tevowels defined above | |
| 382 | - // NB $tefullforms defined above | |
| 383 | - $tebelowbase1 = "E02C|E02D|E02E|E02F|E030|E031|E032|E033|E034|E035|E036|E037|E038|E039|E03A|E03B|E03C|E03D|E03E|E03F|E040|E041|E042|E043|E044|E045|E046|E047|E048|E049|E04A|E04B|E04C|E04D|E04E"; //'Small1KaToHa' | |
| 384 | - $tebelowbase2 = "E04F|E050|E051|E052|E053|E054|E055|E056|E057|E058|E059|E05A|E05B|E05C|E05D|E05E|E05F|E060|E061|E062|E063|E064|E065|E066|E067|E068|E069|E06A|E06B|E06C|E06D|E06E|E06F|E070|E071"; // 'Small2KaToHa' | |
| 385 | - $vstr = preg_replace('/('.$tebelowbase2.') ('.$tevowels.')/', '\\2 \\1', $vstr); | |
| 386 | - $vstr = preg_replace('/('.$tebelowbase1.') ('.$tevowels.')/', '\\2 \\1', $vstr); | |
| 387 | - } | |
| 388 | - | |
| 389 | - | |
| 390 | - // KANNADA Shift | |
| 391 | - else if ($lang=='kn') { | |
| 392 | - $knvowels = "0CBE|0CC0|0CC1|0CC2|0CC3|0CC4|0CC7|0CC8|0CCA|0CCB|0CD5|0CD6" | |
| 393 | - // mPDF 5.3.87 Shouldn't swop E082 and E047 (belowbase1) below | |
| 394 | - // E082 is a matraIligature | |
| 395 | - // ."|E082|E083|E084|E085|E086|E087|E088|E089|E08A|E08B|E08C|E08D|E08E|E08F|E090|E091|E092|E093|E094|E095|E096|E097|E098|E099|E09A|E09B|E09C|E09D|E09E|E09F|E0A0|E0A1|E0A2|E0A3|E0A4|E0A5|E0A6|E0A7|E0A8|E0A9|E0AA|E0AB" | |
| 396 | - ."|E004|E007|E008|E009|E00A"; | |
| 397 | - | |
| 398 | - | |
| 399 | - // NB $knvowels defined above | |
| 400 | - // NB $fullforms defined above | |
| 401 | - // $belowbase1/2 defined above | |
| 402 | - $vstr = preg_replace('/('.$belowbase2.') ('.$knvowels.')/', '\\2 \\1', $vstr); | |
| 403 | - // mPDF 5.3.87 | |
| 404 | - $vstr = preg_replace('/('.$belowbase1.') ('.$knvowels.')/', '\\2 \\1', $vstr); | |
| 405 | - | |
| 406 | - //$vstr = preg_replace('/('.$fullforms.') ('.$knvowels.')/', '\\2 \\1', $vstr); | |
| 407 | - } | |
| 408 | - | |
| 409 | - //============================ | |
| 410 | - // Clear unwanted ZWJ, ZWNJ | |
| 411 | - // MALAYALAM | |
| 412 | - if ($lang=='ml') { | |
| 413 | - $vstr = preg_replace('/(200C|200D) /','', $vstr); | |
| 414 | - } | |
| 415 | - | |
| 416 | - //============================ | |
| 417 | - // END & PUT IT BACK TOGETHER | |
| 418 | - $vstr = preg_replace('/^0020 (.*) 0020$/', '\\1', $vstr); | |
| 419 | - | |
| 420 | - $varr = explode(" ",$vstr); | |
| 421 | - $e = ''; | |
| 422 | - foreach($varr AS $v) { | |
| 423 | - $e.=code2utf(hexdec($v)); | |
| 424 | - } | |
| 425 | - //============================ | |
| 426 | - | |
| 427 | - return $e; | |
| 428 | -} | |
| 429 | - | |
| 430 | - | |
| 431 | -} | |
| 432 | - | |
| 433 | -?> | |
| 1 | +<?php | |
| 2 | + | |
| 3 | +require_once __DIR__ . '/../MpdfException.php'; | |
| 4 | + | |
| 5 | +class INDIC | |
| 6 | +{ | |
| 7 | + /* FROM hb-ot-shape-complex-indic-private.hh */ | |
| 8 | + | |
| 9 | + // indic_category | |
| 10 | + const OT_X = 0; | |
| 11 | + const OT_C = 1; | |
| 12 | + const OT_V = 2; | |
| 13 | + const OT_N = 3; | |
| 14 | + const OT_H = 4; | |
| 15 | + const OT_ZWNJ = 5; | |
| 16 | + const OT_ZWJ = 6; | |
| 17 | + const OT_M = 7; /* Matra or Dependent Vowel */ | |
| 18 | + const OT_SM = 8; | |
| 19 | + const OT_VD = 9; | |
| 20 | + const OT_A = 10; | |
| 21 | + const OT_NBSP = 11; | |
| 22 | + const OT_DOTTEDCIRCLE = 12; /* Not in the spec, but special in Uniscribe. /Very very/ special! */ | |
| 23 | + const OT_RS = 13; /* Register Shifter, used in Khmer OT spec */ | |
| 24 | + const OT_Coeng = 14; | |
| 25 | + const OT_Repha = 15; | |
| 26 | + | |
| 27 | + const OT_Ra = 16; /* Not explicitly listed in the OT spec, but used in the grammar. */ | |
| 28 | + const OT_CM = 17; | |
| 29 | + | |
| 30 | + // Based on indic_category used to make string to find syllables | |
| 31 | + // OT_ to string character (using e.g. OT_C from INDIC) hb-ot-shape-complex-indic-private.hh | |
| 32 | + public static $indic_category_char = array( | |
| 33 | + 'x', | |
| 34 | + 'C', | |
| 35 | + 'V', | |
| 36 | + 'N', | |
| 37 | + 'H', | |
| 38 | + 'Z', | |
| 39 | + 'J', | |
| 40 | + 'M', | |
| 41 | + 'S', | |
| 42 | + 'v', | |
| 43 | + 'A', /* Spec gives Andutta U+0952 as OT_A. However, testing shows that Uniscribe | |
| 44 | + * treats U+0951..U+0952 all as OT_VD - see set_indic_properties */ | |
| 45 | + 's', | |
| 46 | + 'D', | |
| 47 | + 'F', /* Register shift Khmer only */ | |
| 48 | + 'G', /* Khmer only */ | |
| 49 | + 'r', /* 0D4E (dot reph) only one in Malayalam */ | |
| 50 | + 'R', | |
| 51 | + 'm', /* Consonant medial only used in Indic 0A75 in Gurmukhi (0A00..0A7F) : also in Lao, Myanmar, Tai Tham, Javanese & Cham */ | |
| 52 | + ); | |
| 53 | + | |
| 54 | + /* Visual positions in a syllable from left to right. */ | |
| 55 | + /* FROM hb-ot-shape-complex-indic-private.hh */ | |
| 56 | + | |
| 57 | + // indic_position | |
| 58 | + const POS_START = 0; | |
| 59 | + | |
| 60 | + const POS_RA_TO_BECOME_REPH = 1; | |
| 61 | + const POS_PRE_M = 2; | |
| 62 | + const POS_PRE_C = 3; | |
| 63 | + | |
| 64 | + const POS_BASE_C = 4; | |
| 65 | + const POS_AFTER_MAIN = 5; | |
| 66 | + | |
| 67 | + const POS_ABOVE_C = 6; | |
| 68 | + | |
| 69 | + const POS_BEFORE_SUB = 7; | |
| 70 | + const POS_BELOW_C = 8; | |
| 71 | + const POS_AFTER_SUB = 9; | |
| 72 | + | |
| 73 | + const POS_BEFORE_POST = 10; | |
| 74 | + const POS_POST_C = 11; | |
| 75 | + const POS_AFTER_POST = 12; | |
| 76 | + | |
| 77 | + const POS_FINAL_C = 13; | |
| 78 | + const POS_SMVD = 14; | |
| 79 | + | |
| 80 | + const POS_END = 15; | |
| 81 | + | |
| 82 | + /* | |
| 83 | + * Basic features. | |
| 84 | + * These features are applied in order, one at a time, after initial_reordering. | |
| 85 | + */ | |
| 86 | + /* | |
| 87 | + * Must be in the same order as the indic_features array. Ones starting with _ are F_GLOBAL | |
| 88 | + * Ones without the _ are only applied where the mask says! | |
| 89 | + */ | |
| 90 | + | |
| 91 | + const _NUKT = 0; | |
| 92 | + const _AKHN = 1; | |
| 93 | + const RPHF = 2; | |
| 94 | + const _RKRF = 3; | |
| 95 | + const PREF = 4; | |
| 96 | + const BLWF = 5; | |
| 97 | + const HALF = 6; | |
| 98 | + const ABVF = 7; | |
| 99 | + const PSTF = 8; | |
| 100 | + const CFAR = 9; // Khmer only | |
| 101 | + const _VATU = 10; | |
| 102 | + const _CJCT = 11; | |
| 103 | + const INIT = 12; | |
| 104 | + | |
| 105 | + public static function set_indic_properties(&$info, $scriptblock) | |
| 106 | + { | |
| 107 | + $u = $info['uni']; | |
| 108 | + $type = self::indic_get_categories($u); | |
| 109 | + $cat = ($type & 0x7F); | |
| 110 | + $pos = ($type >> 8); | |
| 111 | + | |
| 112 | + /* | |
| 113 | + * Re-assign category | |
| 114 | + */ | |
| 115 | + | |
| 116 | + if ($u == 0x17D1) | |
| 117 | + $cat = self::OT_X; | |
| 118 | + | |
| 119 | + if ($cat == self::OT_X && self::in_range($u, 0x17CB, 0x17D3)) { /* Khmer Various signs */ | |
| 120 | + /* These are like Top Matras. */ | |
| 121 | + $cat = self::OT_M; | |
| 122 | + $pos = self::POS_ABOVE_C; | |
| 123 | + } | |
| 124 | + | |
| 125 | + if ($u == 0x17C6) | |
| 126 | + $cat = self::OT_N; /* Khmer Bindu doesn't like to be repositioned. */ | |
| 127 | + | |
| 128 | + if ($u == 0x17D2) | |
| 129 | + $cat = self::OT_Coeng; /* Khmer coeng */ | |
| 130 | + | |
| 131 | + /* The spec says U+0952 is OT_A. However, testing shows that Uniscribe | |
| 132 | + * treats U+0951..U+0952 all as OT_VD. | |
| 133 | + * TESTS: | |
| 134 | + * U+092E,U+0947,U+0952 | |
| 135 | + * U+092E,U+0952,U+0947 | |
| 136 | + * U+092E,U+0947,U+0951 | |
| 137 | + * U+092E,U+0951,U+0947 | |
| 138 | + * */ | |
| 139 | + //if ($u == 0x0952) $cat = self::OT_A; | |
| 140 | + if (self::in_range($u, 0x0951, 0x0954)) | |
| 141 | + $cat = self::OT_VD; | |
| 142 | + | |
| 143 | + if ($u == 0x200C) | |
| 144 | + $cat = self::OT_ZWNJ; | |
| 145 | + else if ($u == 0x200D) | |
| 146 | + $cat = self::OT_ZWJ; | |
| 147 | + else if ($u == 0x25CC) | |
| 148 | + $cat = self::OT_DOTTEDCIRCLE; | |
| 149 | + else if ($u == 0x0A71) | |
| 150 | + $cat = self::OT_SM; /* GURMUKHI ADDAK. More like consonant medial. like 0A75. */ | |
| 151 | + | |
| 152 | + if ($cat == self::OT_Repha) { | |
| 153 | + /* There are two kinds of characters marked as Repha: | |
| 154 | + * - The ones that are GenCat=Mn are already positioned visually, ie. after base. (eg. Khmer) | |
| 155 | + * - The ones that are GenCat=Lo is encoded logically, ie. beginning of syllable. (eg. Malayalam) | |
| 156 | + * | |
| 157 | + * We recategorize the first kind to look like a Nukta and attached to the base directly. | |
| 158 | + */ | |
| 159 | + if ($info['general_category'] == UCDN::UNICODE_GENERAL_CATEGORY_NON_SPACING_MARK) | |
| 160 | + $cat = self::OT_N; | |
| 161 | + } | |
| 162 | + | |
| 163 | + /* | |
| 164 | + * Re-assign position. | |
| 165 | + */ | |
| 166 | + | |
| 167 | + if ((self::FLAG($cat) & (self::FLAG(self::OT_C) | self::FLAG(self::OT_CM) | self::FLAG(self::OT_Ra) | self::FLAG(self::OT_V) | self::FLAG(self::OT_NBSP) | self::FLAG(self::OT_DOTTEDCIRCLE)))) { // = CONSONANT_FLAGS like is_consonant | |
| 168 | + if ($scriptblock == UCDN::SCRIPT_KHMER) | |
| 169 | + $pos = self::POS_BELOW_C; /* Khmer differs from Indic here. */ | |
| 170 | + else | |
| 171 | + $pos = self::POS_BASE_C; /* Will recategorize later based on font lookups. */ | |
| 172 | + | |
| 173 | + if (self::is_ra($u)) | |
| 174 | + $cat = self::OT_Ra; | |
| 175 | + } | |
| 176 | + else if ($cat == self::OT_M) { | |
| 177 | + $pos = self::matra_position($u, $pos); | |
| 178 | + } else if ($cat == self::OT_SM || $cat == self::OT_VD) { | |
| 179 | + $pos = self::POS_SMVD; | |
| 180 | + } | |
| 181 | + | |
| 182 | + if ($u == 0x0B01) | |
| 183 | + $pos = self::POS_BEFORE_SUB; /* Oriya Bindu is BeforeSub in the spec. */ | |
| 184 | + | |
| 185 | + $info['indic_category'] = $cat; | |
| 186 | + $info['indic_position'] = $pos; | |
| 187 | + } | |
| 188 | + | |
| 189 | + // syllable_type | |
| 190 | + const CONSONANT_SYLLABLE = 0; | |
| 191 | + const VOWEL_SYLLABLE = 1; | |
| 192 | + const STANDALONE_CLUSTER = 2; | |
| 193 | + const BROKEN_CLUSTER = 3; | |
| 194 | + const NON_INDIC_CLUSTER = 4; | |
| 195 | + | |
| 196 | + public static function set_syllables(&$o, $s, &$broken_syllables) | |
| 197 | + { | |
| 198 | + $ptr = 0; | |
| 199 | + $syllable_serial = 1; | |
| 200 | + $broken_syllables = false; | |
| 201 | + | |
| 202 | + while ($ptr < strlen($s)) { | |
| 203 | + $match = ''; | |
| 204 | + $syllable_length = 1; | |
| 205 | + $syllable_type = self::NON_INDIC_CLUSTER; | |
| 206 | + // CONSONANT_SYLLABLE Consonant syllable | |
| 207 | + // From OT spec: | |
| 208 | + if (preg_match('/^([CR]m*[N]?(H[ZJ]?|[ZJ]H))*[CR]m*[N]?[A]?(H[ZJ]?|[M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma)) { | |
| 209 | + // From HarfBuzz: | |
| 210 | + //if (preg_match('/^r?([CR]J?(Z?[N]{0,2})?[ZJ]?H(J[N]?)?){0,4}[CR]J?(Z?[N]{0,2})?A?((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})?(S[Z]?)?[v]{0,2}/', substr($s,$ptr), $ma)) { | |
| 211 | + $syllable_length = strlen($ma[0]); | |
| 212 | + $syllable_type = self::CONSONANT_SYLLABLE; | |
| 213 | + } | |
| 214 | + // VOWEL_SYLLABLE Vowel-based syllable | |
| 215 | + // From OT spec: | |
| 216 | + else if (preg_match('/^(RH|r)?V[N]?([ZJ]?H[CR]m*|J[CR]m*)?([M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma)) { | |
| 217 | + // From HarfBuzz: | |
| 218 | + //else if (preg_match('/^(RH|r)?V(Z?[N]{0,2})?(J|([ZJ]?H(J[N]?)?[CR]J?(Z?[N]{0,2})?){0,4}((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})?(S[Z]?)?[v]{0,2})/', substr($s,$ptr), $ma)) { | |
| 219 | + $syllable_length = strlen($ma[0]); | |
| 220 | + $syllable_type = self::VOWEL_SYLLABLE; | |
| 221 | + } | |
| 222 | + | |
| 223 | + /* Apply only if it's a word start. */ | |
| 224 | + // STANDALONE_CLUSTER Stand Alone syllable at start of word | |
| 225 | + // From OT spec: | |
| 226 | + else if (($ptr == 0 || | |
| 227 | + $o[$ptr - 1]['general_category'] < UCDN::UNICODE_GENERAL_CATEGORY_LOWERCASE_LETTER || | |
| 228 | + $o[$ptr - 1]['general_category'] > UCDN::UNICODE_GENERAL_CATEGORY_NON_SPACING_MARK | |
| 229 | + ) && (preg_match('/^(RH|r)?[sD][N]?([ZJ]?H[CR]m*)?([M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma))) { | |
| 230 | + // From HarfBuzz: | |
| 231 | + // && (preg_match('/^(RH|r)?[sD](Z?[N]{0,2})?(([ZJ]?H(J[N]?)?)[CR]J?(Z?[N]{0,2})?){0,4}((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})?(S[Z]?)?[v]{0,2}/', substr($s,$ptr), $ma)) { | |
| 232 | + $syllable_length = strlen($ma[0]); | |
| 233 | + $syllable_type = self::STANDALONE_CLUSTER; | |
| 234 | + } | |
| 235 | + | |
| 236 | + // BROKEN_CLUSTER syllable | |
| 237 | + else if (preg_match('/^(RH|r)?[N]?([ZJ]?H[CR])?([M]*[N]?[H]?)?[S]?[v]{0,2}/', substr($s, $ptr), $ma)) { | |
| 238 | + // From HarfBuzz: | |
| 239 | + //else if (preg_match('/^(RH|r)?(Z?[N]{0,2})?(([ZJ]?H(J[N]?)?)[CR]J?(Z?[N]{0,2})?){0,4}((([ZJ]?H(J[N]?)?)|HZ)|(HJ)?([ZJ]{0,3}M[N]?(H|JHJR)?){0,4})(S[Z]?)?[v]{0,2}/', substr($s,$ptr), $ma)) { | |
| 240 | + if (strlen($ma[0])) { // May match blank | |
| 241 | + $syllable_length = strlen($ma[0]); | |
| 242 | + $syllable_type = self::BROKEN_CLUSTER; | |
| 243 | + $broken_syllables = true; | |
| 244 | + } | |
| 245 | + } | |
| 246 | + | |
| 247 | + for ($i = $ptr; $i < $ptr + $syllable_length; $i++) { | |
| 248 | + $o[$i]['syllable'] = ($syllable_serial << 4) | $syllable_type; | |
| 249 | + } | |
| 250 | + $ptr += $syllable_length; | |
| 251 | + $syllable_serial++; | |
| 252 | + if ($syllable_serial == 16) | |
| 253 | + $syllable_serial = 1; | |
| 254 | + } | |
| 255 | + } | |
| 256 | + | |
| 257 | + public static function set_syllables_sinhala(&$o, $s, &$broken_syllables) | |
| 258 | + { | |
| 259 | + $ptr = 0; | |
| 260 | + $syllable_serial = 1; | |
| 261 | + $broken_syllables = false; | |
| 262 | + | |
| 263 | + while ($ptr < strlen($s)) { | |
| 264 | + $match = ''; | |
| 265 | + $syllable_length = 1; | |
| 266 | + $syllable_type = self::NON_INDIC_CLUSTER; | |
| 267 | + // CONSONANT_SYLLABLE Consonant syllable | |
| 268 | + // From OT spec: | |
| 269 | + if (preg_match('/^([CR]HJ|[CR]JH){0,8}[CR][HM]{0,3}[S]{0,1}/', substr($s, $ptr), $ma)) { | |
| 270 | + $syllable_length = strlen($ma[0]); | |
| 271 | + $syllable_type = self::CONSONANT_SYLLABLE; | |
| 272 | + } | |
| 273 | + // VOWEL_SYLLABLE Vowel-based syllable | |
| 274 | + // From OT spec: | |
| 275 | + else if (preg_match('/^V[S]{0,1}/', substr($s, $ptr), $ma)) { | |
| 276 | + $syllable_length = strlen($ma[0]); | |
| 277 | + $syllable_type = self::VOWEL_SYLLABLE; | |
| 278 | + } | |
| 279 | + | |
| 280 | + for ($i = $ptr; $i < $ptr + $syllable_length; $i++) { | |
| 281 | + $o[$i]['syllable'] = ($syllable_serial << 4) | $syllable_type; | |
| 282 | + } | |
| 283 | + $ptr += $syllable_length; | |
| 284 | + $syllable_serial++; | |
| 285 | + if ($syllable_serial == 16) | |
| 286 | + $syllable_serial = 1; | |
| 287 | + } | |
| 288 | + } | |
| 289 | + | |
| 290 | + public static function set_syllables_khmer(&$o, $s, &$broken_syllables) | |
| 291 | + { | |
| 292 | + $ptr = 0; | |
| 293 | + $syllable_serial = 1; | |
| 294 | + $broken_syllables = false; | |
| 295 | + | |
| 296 | + while ($ptr < strlen($s)) { | |
| 297 | + $match = ''; | |
| 298 | + $syllable_length = 1; | |
| 299 | + $syllable_type = self::NON_INDIC_CLUSTER; | |
| 300 | + // CONSONANT_SYLLABLE Consonant syllable | |
| 301 | + if (preg_match('/^r?([CR]J?((Z?F)?[N]{0,2})?[ZJ]?G(JN?)?){0,4}[CR]J?((Z?F)?[N]{0,2})?A?((([ZJ]?G(JN?)?)|GZ)|(GJ)?([ZJ]{0,3}MN?(H|JHJR)?){0,4})?(G([CR]J?((Z?F)?[N]{0,2})?|V))?(SZ?)?[v]{0,2}/', substr($s, $ptr), $ma)) { | |
| 302 | + $syllable_length = strlen($ma[0]); | |
| 303 | + $syllable_type = self::CONSONANT_SYLLABLE; | |
| 304 | + } | |
| 305 | + // VOWEL_SYLLABLE Vowel-based syllable | |
| 306 | + else if (preg_match('/^(RH|r)?V((Z?F)?[N]{0,2})?(J|([ZJ]?G(JN?)?[CR]J?((Z?F)?[N]{0,2})?){0,4}((([ZJ]?G(JN?)?)|GZ)|(GJ)?([ZJ]{0,3}MN?(H|JHJR)?){0,4})?(G([CR]J?((Z?F)?[N]{0,2})?|V))?(SZ?)?[v]{0,2})/', substr($s, $ptr), $ma)) { | |
| 307 | + $syllable_length = strlen($ma[0]); | |
| 308 | + $syllable_type = self::VOWEL_SYLLABLE; | |
| 309 | + } | |
| 310 | + | |
| 311 | + | |
| 312 | + // BROKEN_CLUSTER syllable | |
| 313 | + else if (preg_match('/^(RH|r)?((Z?F)?[N]{0,2})?(([ZJ]?G(JN?)?)[CR]J?((Z?F)?[N]{0,2})?){0,4}((([ZJ]?G(JN?)?)|GZ)|(GJ)?([ZJ]{0,3}MN?(H|JHJR)?){0,4})(G([CR]J?((Z?F)?[N]{0,2})?|V))?(SZ?)?[v]{0,2}/', substr($s, $ptr), $ma)) { | |
| 314 | + if (strlen($ma[0])) { // May match blank | |
| 315 | + $syllable_length = strlen($ma[0]); | |
| 316 | + $syllable_type = self::BROKEN_CLUSTER; | |
| 317 | + $broken_syllables = true; | |
| 318 | + } | |
| 319 | + } | |
| 320 | + | |
| 321 | + for ($i = $ptr; $i < $ptr + $syllable_length; $i++) { | |
| 322 | + $o[$i]['syllable'] = ($syllable_serial << 4) | $syllable_type; | |
| 323 | + } | |
| 324 | + $ptr += $syllable_length; | |
| 325 | + $syllable_serial++; | |
| 326 | + if ($syllable_serial == 16) | |
| 327 | + $syllable_serial = 1; | |
| 328 | + } | |
| 329 | + } | |
| 330 | + | |
| 331 | + public static function initial_reordering(&$info, $GSUBdata, $broken_syllables, $indic_config, $scriptblock, $is_old_spec, $dottedcircle) | |
| 332 | + { | |
| 333 | + | |
| 334 | + self::update_consonant_positions($info, $GSUBdata); | |
| 335 | + | |
| 336 | + if ($broken_syllables && $dottedcircle) { | |
| 337 | + self::insert_dotted_circles($info, $dottedcircle); | |
| 338 | + } | |
| 339 | + | |
| 340 | + $count = count($info); | |
| 341 | + if (!$count) | |
| 342 | + return; | |
| 343 | + $last = 0; | |
| 344 | + $last_syllable = $info[0]['syllable']; | |
| 345 | + for ($i = 1; $i < $count; $i++) { | |
| 346 | + if ($last_syllable != $info[$i]['syllable']) { | |
| 347 | + self::initial_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $i); | |
| 348 | + $last = $i; | |
| 349 | + $last_syllable = $info[$last]['syllable']; | |
| 350 | + } | |
| 351 | + } | |
| 352 | + self::initial_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $count); | |
| 353 | + } | |
| 354 | + | |
| 355 | + public static function update_consonant_positions(&$info, $GSUBdata) | |
| 356 | + { | |
| 357 | + $count = count($info); | |
| 358 | + for ($i = 0; $i < $count; $i++) { | |
| 359 | + if ($info[$i]['indic_position'] == self::POS_BASE_C) { | |
| 360 | + $c = $info[$i]['uni']; | |
| 361 | + // If would substitute... | |
| 362 | + if (isset($GSUBdata['pref'][$c])) { | |
| 363 | + $info[$i]['indic_position'] = self::POS_POST_C; | |
| 364 | + } else if (isset($GSUBdata['blwf'][$c])) { | |
| 365 | + $info[$i]['indic_position'] = self::POS_BELOW_C; | |
| 366 | + } else if (isset($GSUBdata['pstf'][$c])) { | |
| 367 | + $info[$i]['indic_position'] = self::POS_POST_C; | |
| 368 | + } | |
| 369 | + } | |
| 370 | + } | |
| 371 | + } | |
| 372 | + | |
| 373 | + public static function insert_dotted_circles(&$info, $dottedcircle) | |
| 374 | + { | |
| 375 | + $idx = 0; | |
| 376 | + $last_syllable = 0; | |
| 377 | + while ($idx < count($info)) { | |
| 378 | + $syllable = $info[$idx]['syllable']; | |
| 379 | + $syllable_type = ($syllable & 0x0F); | |
| 380 | + if ($last_syllable != $syllable && $syllable_type == self::BROKEN_CLUSTER) { | |
| 381 | + $last_syllable = $syllable; | |
| 382 | + | |
| 383 | + $dottedcircle[0]['syllable'] = $info[$idx]['syllable']; | |
| 384 | + | |
| 385 | + /* Insert dottedcircle after possible Repha. */ | |
| 386 | + while ($idx < count($info) && $last_syllable == $info[$idx]['syllable'] && $info[$idx]['indic_category'] == self::OT_Repha) | |
| 387 | + $idx++; | |
| 388 | + array_splice($info, $idx, 0, $dottedcircle); | |
| 389 | + } else { | |
| 390 | + $idx++; | |
| 391 | + } | |
| 392 | + } | |
| 393 | + | |
| 394 | + // I am not sue how this code below got in here, since $idx should now be > count($info) and thus invalid. | |
| 395 | + // In case I am missing something(!) I'll leave a warning here for now: | |
| 396 | + if (isset($info[$idx])) { | |
| 397 | + throw new MpdfException('Unexpected error occured in Indic processing'); | |
| 398 | + } | |
| 399 | + // In case of final bloken cluster... | |
| 400 | + //$syllable = $info[$idx]['syllable']; | |
| 401 | + //$syllable_type = ($syllable & 0x0F); | |
| 402 | + //if ($last_syllable != $syllable && $syllable_type == self::BROKEN_CLUSTER) { | |
| 403 | + // $dottedcircle[0]['syllable'] = $info[$idx]['syllable']; | |
| 404 | + // array_splice($info, $idx, 0, $dottedcircle); | |
| 405 | + //} | |
| 406 | + } | |
| 407 | + | |
| 408 | + /* Rules from: | |
| 409 | + * https://www.microsoft.com/typography/otfntdev/devanot/shaping.aspx */ | |
| 410 | + | |
| 411 | + public static function initial_reordering_syllable(&$info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $start, $end) | |
| 412 | + { | |
| 413 | + /* vowel_syllable: We made the vowels look like consonants. So uses the consonant logic! */ | |
| 414 | + /* broken_cluster: We already inserted dotted-circles, so just call the standalone_cluster. */ | |
| 415 | + /* standalone_cluster: We treat NBSP/dotted-circle as if they are consonants, so we should just chain. */ | |
| 416 | + | |
| 417 | + $syllable_type = ($info[$start]['syllable'] & 0x0F); | |
| 418 | + if ($syllable_type == self::NON_INDIC_CLUSTER) { | |
| 419 | + return; | |
| 420 | + } | |
| 421 | + if ($syllable_type == self::BROKEN_CLUSTER || $syllable_type == self::STANDALONE_CLUSTER) { | |
| 422 | + //if ($uniscribe_bug_compatible) { | |
| 423 | + /* For dotted-circle, this is what Uniscribe does: | |
| 424 | + * If dotted-circle is the last glyph, it just does nothing. | |
| 425 | + * i.e. It doesn't form Reph. */ | |
| 426 | + if ($info[$end - 1]['indic_category'] == self::OT_DOTTEDCIRCLE) { | |
| 427 | + return; | |
| 428 | + } | |
| 429 | + } | |
| 430 | + | |
| 431 | + /* 1. Find base consonant: | |
| 432 | + * | |
| 433 | + * The shaping engine finds the base consonant of the syllable, using the | |
| 434 | + * following algorithm: starting from the end of the syllable, move backwards | |
| 435 | + * until a consonant is found that does not have a below-base or post-base | |
| 436 | + * form (post-base forms have to follow below-base forms), or that is not a | |
| 437 | + * pre-base reordering Ra, or arrive at the first consonant. The consonant | |
| 438 | + * stopped at will be the base. | |
| 439 | + * | |
| 440 | + * o If the syllable starts with Ra + Halant (in a script that has Reph) | |
| 441 | + * and has more than one consonant, Ra is excluded from candidates for | |
| 442 | + * base consonants. | |
| 443 | + */ | |
| 444 | + | |
| 445 | + $base = $end; | |
| 446 | + $has_reph = false; | |
| 447 | + $limit = $start; | |
| 448 | + | |
| 449 | + if ($scriptblock != UCDN::SCRIPT_KHMER) { | |
| 450 | + /* -> If the syllable starts with Ra + Halant (in a script that has Reph) | |
| 451 | + * and has more than one consonant, Ra is excluded from candidates for | |
| 452 | + * base consonants. */ | |
| 453 | + if (count($GSUBdata['rphf']) /* ?? $indic_plan->mask_array[RPHF] */ && $start + 3 <= $end && | |
| 454 | + ( | |
| 455 | + ($indic_config[4] == self::REPH_MODE_IMPLICIT && !self::is_joiner($info[$start + 2])) || | |
| 456 | + ($indic_config[4] == self::REPH_MODE_EXPLICIT && $info[$start + 2]['indic_category'] == self::OT_ZWJ) | |
| 457 | + )) { | |
| 458 | + /* See if it matches the 'rphf' feature. */ | |
| 459 | + //$glyphs = array($info[$start]['uni'], $info[$start + 1]['uni']); | |
| 460 | + //if ($indic_plan->rphf->would_substitute ($glyphs, count($glyphs), true, face)) { | |
| 461 | + if (isset($GSUBdata['rphf'][$info[$start]['uni']]) && self::is_halant_or_coeng($info[$start + 1])) { | |
| 462 | + $limit += 2; | |
| 463 | + while ($limit < $end && self::is_joiner($info[$limit])) | |
| 464 | + $limit++; | |
| 465 | + $base = $start; | |
| 466 | + $has_reph = true; | |
| 467 | + } | |
| 468 | + } else if ($indic_config[4] == self::REPH_MODE_LOG_REPHA && $info[$start]['indic_category'] == self::OT_Repha) { | |
| 469 | + $limit += 1; | |
| 470 | + while ($limit < $end && self::is_joiner($info[$limit])) | |
| 471 | + $limit++; | |
| 472 | + $base = $start; | |
| 473 | + $has_reph = true; | |
| 474 | + } | |
| 475 | + } | |
| 476 | + | |
| 477 | + switch ($indic_config[2]) { // base_pos | |
| 478 | + case self::BASE_POS_LAST: | |
| 479 | + /* -> starting from the end of the syllable, move backwards */ | |
| 480 | + $i = $end; | |
| 481 | + $seen_below = false; | |
| 482 | + do { | |
| 483 | + $i--; | |
| 484 | + /* -> until a consonant is found */ | |
| 485 | + if (self::is_consonant($info[$i])) { | |
| 486 | + /* -> that does not have a below-base or post-base form | |
| 487 | + * (post-base forms have to follow below-base forms), */ | |
| 488 | + if ($info[$i]['indic_position'] != self::POS_BELOW_C && ($info[$i]['indic_position'] != self::POS_POST_C || $seen_below)) { | |
| 489 | + $base = $i; | |
| 490 | + break; | |
| 491 | + } | |
| 492 | + if ($info[$i]['indic_position'] == self::POS_BELOW_C) | |
| 493 | + $seen_below = true; | |
| 494 | + | |
| 495 | + /* -> or that is not a pre-base reordering Ra, | |
| 496 | + * | |
| 497 | + * IMPLEMENTATION NOTES: | |
| 498 | + * | |
| 499 | + * Our pre-base reordering Ra's are marked POS_POST_C, so will be skipped | |
| 500 | + * by the logic above already. | |
| 501 | + */ | |
| 502 | + | |
| 503 | + /* -> or arrive at the first consonant. The consonant stopped at will | |
| 504 | + * be the base. */ | |
| 505 | + $base = $i; | |
| 506 | + } | |
| 507 | + else { | |
| 508 | + /* A ZWJ after a Halant stops the base search, and requests an explicit | |
| 509 | + * half form. | |
| 510 | + * [A ZWJ before a Halant, requests a subjoined form instead, and hence | |
| 511 | + * search continues. This is particularly important for Bengali | |
| 512 | + * sequence Ra,H,Ya that should form Ya-Phalaa by subjoining Ya] */ | |
| 513 | + if ($start < $i && $info[$i]['indic_category'] == self::OT_ZWJ && $info[$i - 1]['indic_category'] == self::OT_H) { | |
| 514 | + if (!defined("OMIT_INDIC_FIX_1") || OMIT_INDIC_FIX_1 != 1) { | |
| 515 | + $base = $i; | |
| 516 | + } // INDIC_FIX_1 | |
| 517 | + break; | |
| 518 | + } | |
| 519 | + // ZKI8 | |
| 520 | + if ($start < $i && $info[$i]['indic_category'] == self::OT_ZWNJ) { | |
| 521 | + break; | |
| 522 | + } | |
| 523 | + } | |
| 524 | + } while ($i > $limit); | |
| 525 | + break; | |
| 526 | + | |
| 527 | + case self::BASE_POS_FIRST: | |
| 528 | + /* In scripts without half forms (eg. Khmer), the first consonant is always the base. */ | |
| 529 | + | |
| 530 | + if (!$has_reph) | |
| 531 | + $base = $limit; | |
| 532 | + | |
| 533 | + /* Find the last base consonant that is not blocked by ZWJ. If there is | |
| 534 | + * a ZWJ right before a base consonant, that would request a subjoined form. */ | |
| 535 | + for ($i = $limit; $i < $end; $i++) { | |
| 536 | + if (self::is_consonant($info[$i]) && $info[$i]['indic_position'] == self::POS_BASE_C) { | |
| 537 | + if ($limit < $i && $info[$i - 1]['indic_category'] == self::OT_ZWJ) | |
| 538 | + break; | |
| 539 | + else | |
| 540 | + $base = $i; | |
| 541 | + } | |
| 542 | + } | |
| 543 | + | |
| 544 | + /* Mark all subsequent consonants as below. */ | |
| 545 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 546 | + if (self::is_consonant($info[$i]) && $info[$i]['indic_position'] == self::POS_BASE_C) | |
| 547 | + $info[$i]['indic_position'] = self::POS_BELOW_C; | |
| 548 | + } | |
| 549 | + break; | |
| 550 | + //default: | |
| 551 | + //assert (false); | |
| 552 | + /* fallthrough */ | |
| 553 | + } | |
| 554 | + | |
| 555 | + /* -> If the syllable starts with Ra + Halant (in a script that has Reph) | |
| 556 | + * and has more than one consonant, Ra is excluded from candidates for | |
| 557 | + * base consonants. | |
| 558 | + * | |
| 559 | + * Only do this for unforced Reph. (ie. not for Ra,H,ZWJ. */ | |
| 560 | + if ($scriptblock != UCDN::SCRIPT_KHMER) { | |
| 561 | + if ($has_reph && $base == $start && $limit - $base <= 2) { | |
| 562 | + /* Have no other consonant, so Reph is not formed and Ra becomes base. */ | |
| 563 | + $has_reph = false; | |
| 564 | + } | |
| 565 | + } | |
| 566 | + | |
| 567 | + /* 2. Decompose and reorder Matras: | |
| 568 | + * | |
| 569 | + * Each matra and any syllable modifier sign in the cluster are moved to the | |
| 570 | + * appropriate position relative to the consonant(s) in the cluster. The | |
| 571 | + * shaping engine decomposes two- or three-part matras into their constituent | |
| 572 | + * parts before any repositioning. Matra characters are classified by which | |
| 573 | + * consonant in a conjunct they have affinity for and are reordered to the | |
| 574 | + * following positions: | |
| 575 | + * | |
| 576 | + * o Before first half form in the syllable | |
| 577 | + * o After subjoined consonants | |
| 578 | + * o After post-form consonant | |
| 579 | + * o After main consonant (for above marks) | |
| 580 | + * | |
| 581 | + * IMPLEMENTATION NOTES: | |
| 582 | + * | |
| 583 | + * The normalize() routine has already decomposed matras for us, so we don't | |
| 584 | + * need to worry about that. | |
| 585 | + */ | |
| 586 | + | |
| 587 | + | |
| 588 | + /* 3. Reorder marks to canonical order: | |
| 589 | + * | |
| 590 | + * Adjacent nukta and halant or nukta and vedic sign are always repositioned | |
| 591 | + * if necessary, so that the nukta is first. | |
| 592 | + * | |
| 593 | + * IMPLEMENTATION NOTES: | |
| 594 | + * | |
| 595 | + * Use the combining Class from Unicode categories? to bubble_sort. | |
| 596 | + */ | |
| 597 | + | |
| 598 | + /* Reorder characters */ | |
| 599 | + | |
| 600 | + for ($i = $start; $i < $base; $i++) | |
| 601 | + $info[$i]['indic_position'] = min(self::POS_PRE_C, $info[$i]['indic_position']); | |
| 602 | + | |
| 603 | + if ($base < $end) | |
| 604 | + $info[$base]['indic_position'] = self::POS_BASE_C; | |
| 605 | + | |
| 606 | + /* Mark final consonants. A final consonant is one appearing after a matra, | |
| 607 | + * ? only in Khmer. */ | |
| 608 | + for ($i = $base + 1; $i < $end; $i++) | |
| 609 | + if ($info[$i]['indic_category'] == self::OT_M) { | |
| 610 | + for ($j = $i + 1; $j < $end; $j++) | |
| 611 | + if (self::is_consonant($info[$j])) { | |
| 612 | + $info[$j]['indic_position'] = self::POS_FINAL_C; | |
| 613 | + break; | |
| 614 | + } | |
| 615 | + break; | |
| 616 | + } | |
| 617 | + | |
| 618 | + /* Handle beginning Ra */ | |
| 619 | + if ($scriptblock != UCDN::SCRIPT_KHMER) { | |
| 620 | + if ($has_reph) | |
| 621 | + $info[$start]['indic_position'] = self::POS_RA_TO_BECOME_REPH; | |
| 622 | + } | |
| 623 | + | |
| 624 | + | |
| 625 | + /* For old-style Indic script tags, move the first post-base Halant after | |
| 626 | + * last consonant. Only do this if there is *not* a Halant after last | |
| 627 | + * consonant. Otherwise it becomes messy. */ | |
| 628 | + if ($is_old_spec) { | |
| 629 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 630 | + if ($info[$i]['indic_category'] == self::OT_H) { | |
| 631 | + for ($j = $end - 1; $j > $i; $j--) { | |
| 632 | + if (self::is_consonant($info[$j]) || $info[$j]['indic_category'] == self::OT_H) { | |
| 633 | + break; | |
| 634 | + } | |
| 635 | + } | |
| 636 | + if ($info[$j]['indic_category'] != self::OT_H && $j > $i) { | |
| 637 | + /* Move Halant to after last consonant. */ | |
| 638 | + self::_move_info_pos($info, $i, $j + 1); | |
| 639 | + } | |
| 640 | + break; | |
| 641 | + } | |
| 642 | + } | |
| 643 | + } | |
| 644 | + | |
| 645 | + /* Attach misc marks to previous char to move with them. */ | |
| 646 | + $last_pos = self::POS_START; | |
| 647 | + for ($i = $start; $i < $end; $i++) { | |
| 648 | + if ((self::FLAG($info[$i]['indic_category']) & (self::FLAG(self::OT_ZWJ) | self::FLAG(self::OT_ZWNJ) | self::FLAG(self::OT_N) | self::FLAG(self::OT_RS) | self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng) ))) { | |
| 649 | + $info[$i]['indic_position'] = $last_pos; | |
| 650 | + if ($info[$i]['indic_category'] == self::OT_H && $info[$i]['indic_position'] == self::POS_PRE_M) { | |
| 651 | + /* | |
| 652 | + * Uniscribe doesn't move the Halant with Left Matra. | |
| 653 | + * TEST: U+092B,U+093F,U+094DE | |
| 654 | + * We follow. This is important for the Sinhala | |
| 655 | + * U+0DDA split matra since it decomposes to U+0DD9,U+0DCA | |
| 656 | + * where U+0DD9 is a left matra and U+0DCA is the virama. | |
| 657 | + * We don't want to move the virama with the left matra. | |
| 658 | + * TEST: U+0D9A,U+0DDA | |
| 659 | + */ | |
| 660 | + for ($j = $i; $j > $start; $j--) | |
| 661 | + if ($info[$j - 1]['indic_position'] != self::POS_PRE_M) { | |
| 662 | + $info[$i]['indic_position'] = $info[$j - 1]['indic_position']; | |
| 663 | + break; | |
| 664 | + } | |
| 665 | + } | |
| 666 | + } else if ($info[$i]['indic_position'] != self::POS_SMVD) { | |
| 667 | + $last_pos = $info[$i]['indic_position']; | |
| 668 | + } | |
| 669 | + } | |
| 670 | + | |
| 671 | + /* Re-attach ZWJ, ZWNJ, and halant to next char, for after-base consonants. */ | |
| 672 | + $last_halant = $end; | |
| 673 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 674 | + if (self::is_halant_or_coeng($info[$i])) | |
| 675 | + $last_halant = $i; | |
| 676 | + else if (self::is_consonant($info[$i])) { | |
| 677 | + for ($j = $last_halant; $j < $i; $j++) | |
| 678 | + if ($info[$j]['indic_position'] != self::POS_SMVD) | |
| 679 | + $info[$j]['indic_position'] = $info[$i]['indic_position']; | |
| 680 | + } | |
| 681 | + } | |
| 682 | + | |
| 683 | + | |
| 684 | + if ($scriptblock == UCDN::SCRIPT_KHMER) { | |
| 685 | + /* KHMER_FIX_2 */ | |
| 686 | + /* Move Coeng+RO (Halant,Ra) sequence before base consonant. */ | |
| 687 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 688 | + if (self::is_halant_or_coeng($info[$i]) && self::is_ra($info[$i + 1]['uni'])) { | |
| 689 | + $info[$i]['indic_position'] = self::POS_PRE_C; | |
| 690 | + $info[$i + 1]['indic_position'] = self::POS_PRE_C; | |
| 691 | + break; | |
| 692 | + } | |
| 693 | + } | |
| 694 | + } | |
| 695 | + | |
| 696 | + | |
| 697 | + /* | |
| 698 | + if (!defined("OMIT_INDIC_FIX_2") || OMIT_INDIC_FIX_2 != 1) { | |
| 699 | + // INDIC_FIX_2 | |
| 700 | + $ZWNJ_found = false; | |
| 701 | + $POST_ZWNJ_c_found = false; | |
| 702 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 703 | + if ($info[$i]['indic_category'] == self::OT_ZWNJ) { $ZWNJ_found = true; } | |
| 704 | + else if ($ZWNJ_found && $info[$i]['indic_category'] == self::OT_C) { $POST_ZWNJ_c_found = true; } | |
| 705 | + else if ($POST_ZWNJ_c_found && $info[$i]['indic_position'] == self::POS_BEFORE_SUB) { $info[$i]['indic_position'] = self::POS_AFTER_SUB; } | |
| 706 | + } | |
| 707 | + } | |
| 708 | + */ | |
| 709 | + | |
| 710 | + /* Setup masks now */ | |
| 711 | + for ($i = $start; $i < $end; $i++) { | |
| 712 | + $info[$i]['mask'] = 0; | |
| 713 | + } | |
| 714 | + | |
| 715 | + | |
| 716 | + if ($scriptblock == UCDN::SCRIPT_KHMER) { | |
| 717 | + /* Find a Coeng+RO (Halant,Ra) sequence and mark it for pre-base processing. */ | |
| 718 | + $mask = self::FLAG(self::PREF); | |
| 719 | + for ($i = $base; $i < $end - 1; $i++) { /* KHMER_FIX_1 From $start (not base) */ | |
| 720 | + if (self::is_halant_or_coeng($info[$i]) && self::is_ra($info[$i + 1]['uni'])) { | |
| 721 | + | |
| 722 | + $info[$i]['mask'] |= self::FLAG(self::PREF); | |
| 723 | + $info[$i + 1]['mask'] |= self::FLAG(self::PREF); | |
| 724 | + | |
| 725 | + /* Mark the subsequent stuff with 'cfar'. Used in Khmer. | |
| 726 | + * Read the feature spec. | |
| 727 | + * This allows distinguishing the following cases with MS Khmer fonts: | |
| 728 | + * U+1784,U+17D2,U+179A,U+17D2,U+1782 [C+Coeng+RO+Coeng+C] => Should activate CFAR | |
| 729 | + * U+1784,U+17D2,U+1782,U+17D2,U+179A [C+Coeng+C+Coeng+RO] => Should NOT activate CFAR | |
| 730 | + */ | |
| 731 | + for ($j = ($i + 2); $j < $end; $j++) | |
| 732 | + $info[$j]['mask'] |= self::FLAG(self::CFAR); | |
| 733 | + | |
| 734 | + break; | |
| 735 | + } | |
| 736 | + } | |
| 737 | + } | |
| 738 | + | |
| 739 | + | |
| 740 | + | |
| 741 | + /* Sit tight, rock 'n roll! */ | |
| 742 | + self::bubble_sort($info, $start, $end - $start); | |
| 743 | + | |
| 744 | + /* Find base again */ | |
| 745 | + $base = $end; | |
| 746 | + for ($i = $start; $i < $end; $i++) { | |
| 747 | + if ($info[$i]['indic_position'] == self::POS_BASE_C) { | |
| 748 | + $base = $i; | |
| 749 | + break; | |
| 750 | + } | |
| 751 | + } | |
| 752 | + | |
| 753 | + if ($scriptblock != UCDN::SCRIPT_KHMER) { | |
| 754 | + /* Reph */ | |
| 755 | + for ($i = $start; $i < $end; $i++) { | |
| 756 | + if ($info[$i]['indic_position'] == self::POS_RA_TO_BECOME_REPH) { | |
| 757 | + $info[$i]['mask'] |= self::FLAG(self::RPHF); | |
| 758 | + } | |
| 759 | + } | |
| 760 | + | |
| 761 | + /* Pre-base */ | |
| 762 | + $mask = self::FLAG(self::HALF); | |
| 763 | + for ($i = $start; $i < $base; $i++) { | |
| 764 | + $info[$i]['mask'] |= $mask; | |
| 765 | + } | |
| 766 | + } | |
| 767 | + | |
| 768 | + /* Post-base */ | |
| 769 | + $mask = (self::FLAG(self::BLWF) | self::FLAG(self::ABVF) | self::FLAG(self::PSTF)); | |
| 770 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 771 | + $info[$i]['mask'] |= $mask; | |
| 772 | + } | |
| 773 | + | |
| 774 | + | |
| 775 | + if ($scriptblock != UCDN::SCRIPT_KHMER) { | |
| 776 | + if (!defined("OMIT_INDIC_FIX_3") || OMIT_INDIC_FIX_3 != 1) { | |
| 777 | + /* INDIC_FIX_3 */ | |
| 778 | + /* Find a (pre-base) Consonant, Halant,Ra sequence and mark Halant|Ra for below-base BLWF processing. */ | |
| 779 | + // TEST CASE ক্র্ক in FreeSans versus Vrinda | |
| 780 | + if (($base - $start) >= 3) { | |
| 781 | + for ($i = $start; $i < ($base - 2); $i++) { | |
| 782 | + if (self::is_consonant($info[$i])) { | |
| 783 | + if (self::is_halant_or_coeng($info[$i + 1]) && self::is_ra($info[$i + 2]['uni'])) { | |
| 784 | + // If would substitute Halant+Ra...BLWF | |
| 785 | + if (isset($GSUBdata['blwf'][$info[$i + 2]['uni']])) { | |
| 786 | + $info[$i + 1]['mask'] |= self::FLAG(self::BLWF); | |
| 787 | + $info[$i + 2]['mask'] |= self::FLAG(self::BLWF); | |
| 788 | + } | |
| 789 | + /* If would not substitute as blwf, mark Ra+Halant for RPHF using following Halant (if present) */ else if (self::is_halant_or_coeng($info[$i + 3])) { | |
| 790 | + $info[$i + 2]['mask'] |= self::FLAG(self::RPHF); | |
| 791 | + $info[$i + 3]['mask'] |= self::FLAG(self::RPHF); | |
| 792 | + } | |
| 793 | + break; | |
| 794 | + } | |
| 795 | + } | |
| 796 | + } | |
| 797 | + } | |
| 798 | + } | |
| 799 | + } | |
| 800 | + | |
| 801 | + | |
| 802 | + | |
| 803 | + if ($is_old_spec && $scriptblock == UCDN::SCRIPT_DEVANAGARI) { | |
| 804 | + /* Old-spec eye-lash Ra needs special handling. From the spec: | |
| 805 | + * "The feature 'below-base form' is applied to consonants | |
| 806 | + * having below-base forms and following the base consonant. | |
| 807 | + * The exception is vattu, which may appear below half forms | |
| 808 | + * as well as below the base glyph. The feature 'below-base | |
| 809 | + * form' will be applied to all such occurrences of Ra as well." | |
| 810 | + * | |
| 811 | + * Test case: U+0924,U+094D,U+0930,U+094d,U+0915 | |
| 812 | + * with Sanskrit 2003 font. | |
| 813 | + * | |
| 814 | + * However, note that Ra,Halant,ZWJ is the correct way to | |
| 815 | + * request eyelash form of Ra, so we wouldbn't inhibit it | |
| 816 | + * in that sequence. | |
| 817 | + * | |
| 818 | + * Test case: U+0924,U+094D,U+0930,U+094d,U+200D,U+0915 | |
| 819 | + */ | |
| 820 | + for ($i = $start; ($i + 1) < $base; $i++) { | |
| 821 | + if ($info[$i]['indic_category'] == self::OT_Ra && $info[$i + 1]['indic_category'] == self::OT_H && | |
| 822 | + ($i + 2 == $base || $info[$i + 2]['indic_category'] != self::OT_ZWJ)) { | |
| 823 | + $info[$i]['mask'] |= self::FLAG(self::BLWF); | |
| 824 | + $info[$i + 1]['mask'] |= self::FLAG(self::BLWF); | |
| 825 | + } | |
| 826 | + } | |
| 827 | + } | |
| 828 | + | |
| 829 | + if ($scriptblock != UCDN::SCRIPT_KHMER) { | |
| 830 | + if (count($GSUBdata['pref']) && $base + 2 < $end) { | |
| 831 | + /* Find a Halant,Ra sequence and mark it for pre-base processing. */ | |
| 832 | + for ($i = $base + 1; $i + 1 < $end; $i++) { | |
| 833 | + // If old_spec find Ra-Halant... | |
| 834 | + if ((isset($GSUBdata['pref'][$info[$i + 1]['uni']]) && self::is_halant_or_coeng($info[$i]) && self::is_ra($info[$i + 1]['uni']) ) || | |
| 835 | + ($is_old_spec && isset($GSUBdata['pref'][$info[$i]['uni']]) && self::is_halant_or_coeng($info[$i + 1]) && self::is_ra($info[$i]['uni']) ) | |
| 836 | + ) { | |
| 837 | + $info[$i++]['mask'] |= self::FLAG(self::PREF); | |
| 838 | + $info[$i++]['mask'] |= self::FLAG(self::PREF); | |
| 839 | + break; | |
| 840 | + } | |
| 841 | + } | |
| 842 | + } | |
| 843 | + } | |
| 844 | + | |
| 845 | + | |
| 846 | + /* Apply ZWJ/ZWNJ effects */ | |
| 847 | + for ($i = $start + 1; $i < $end; $i++) { | |
| 848 | + if (self::is_joiner($info[$i])) { | |
| 849 | + $non_joiner = ($info[$i]['indic_category'] == self::OT_ZWNJ); | |
| 850 | + $j = $i; | |
| 851 | + while ($j > $start) { | |
| 852 | + if (defined("OMIT_INDIC_FIX_4") && OMIT_INDIC_FIX_4 == 1) { | |
| 853 | + // INDIC_FIX_4 = do nothing - carry on // | |
| 854 | + // ZWNJ should block H C from forming blwf post-base - need to unmask backwards beyond first consonant arrived at // | |
| 855 | + if (!self::is_consonant($info[$j])) { | |
| 856 | + break; | |
| 857 | + } | |
| 858 | + } | |
| 859 | + $j--; | |
| 860 | + | |
| 861 | + /* ZWJ/ZWNJ should disable CJCT. They do that by simply | |
| 862 | + * being there, since we don't skip them for the CJCT | |
| 863 | + * feature (ie. F_MANUAL_ZWJ) */ | |
| 864 | + | |
| 865 | + /* A ZWNJ disables HALF. */ | |
| 866 | + if ($non_joiner) { | |
| 867 | + $info[$j]['mask'] &= ~(self::FLAG(self::HALF) | self::FLAG(self::BLWF)); | |
| 868 | + } | |
| 869 | + } | |
| 870 | + } | |
| 871 | + } | |
| 872 | + } | |
| 873 | + | |
| 874 | + public static function final_reordering(&$info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec) | |
| 875 | + { | |
| 876 | + $count = count($info); | |
| 877 | + if (!$count) | |
| 878 | + return; | |
| 879 | + $last = 0; | |
| 880 | + $last_syllable = $info[0]['syllable']; | |
| 881 | + for ($i = 1; $i < $count; $i++) { | |
| 882 | + if ($last_syllable != $info[$i]['syllable']) { | |
| 883 | + self::final_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $i); | |
| 884 | + $last = $i; | |
| 885 | + $last_syllable = $info[$last]['syllable']; | |
| 886 | + } | |
| 887 | + } | |
| 888 | + self::final_reordering_syllable($info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $last, $count); | |
| 889 | + } | |
| 890 | + | |
| 891 | + public static function final_reordering_syllable(&$info, $GSUBdata, $indic_config, $scriptblock, $is_old_spec, $start, $end) | |
| 892 | + { | |
| 893 | + | |
| 894 | + /* 4. Final reordering: | |
| 895 | + * | |
| 896 | + * After the localized forms and basic shaping forms GSUB features have been | |
| 897 | + * applied (see below), the shaping engine performs some final glyph | |
| 898 | + * reordering before applying all the remaining font features to the entire | |
| 899 | + * cluster. | |
| 900 | + */ | |
| 901 | + | |
| 902 | + /* Find base again */ | |
| 903 | + for ($base = $start; $base < $end; $base++) | |
| 904 | + if ($info[$base]['indic_position'] >= self::POS_BASE_C) { | |
| 905 | + if ($start < $base && $info[$base]['indic_position'] > self::POS_BASE_C) | |
| 906 | + $base--; | |
| 907 | + break; | |
| 908 | + } | |
| 909 | + if ($base == $end && $start < $base && $info[$base - 1]['indic_category'] != self::OT_ZWJ) | |
| 910 | + $base--; | |
| 911 | + while ($start < $base && isset($info[$base]) && ($info[$base]['indic_category'] == self::OT_H || $info[$base]['indic_category'] == self::OT_N)) | |
| 912 | + $base--; | |
| 913 | + | |
| 914 | + | |
| 915 | + /* o Reorder matras: | |
| 916 | + * | |
| 917 | + * If a pre-base matra character had been reordered before applying basic | |
| 918 | + * features, the glyph can be moved closer to the main consonant based on | |
| 919 | + * whether half-forms had been formed. Actual position for the matra is | |
| 920 | + * defined as "after last standalone halant glyph, after initial matra | |
| 921 | + * position and before the main consonant". If ZWJ or ZWNJ follow this | |
| 922 | + * halant, position is moved after it. | |
| 923 | + */ | |
| 924 | + | |
| 925 | + | |
| 926 | + if ($start + 1 < $end && $start < $base) { /* Otherwise there can't be any pre-base matra characters. */ | |
| 927 | + /* If we lost track of base, alas, position before last thingy. */ | |
| 928 | + $new_pos = ($base == $end) ? $base - 2 : $base - 1; | |
| 929 | + | |
| 930 | + /* Malayalam / Tamil do not have "half" forms or explicit virama forms. | |
| 931 | + * The glyphs formed by 'half' are Chillus or ligated explicit viramas. | |
| 932 | + * We want to position matra after them. | |
| 933 | + */ | |
| 934 | + if ($scriptblock != UCDN::SCRIPT_MALAYALAM && $scriptblock != UCDN::SCRIPT_TAMIL) { | |
| 935 | + while ($new_pos > $start && !(self::is_one_of($info[$new_pos], (self::FLAG(self::OT_M) | self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng))))) | |
| 936 | + $new_pos--; | |
| 937 | + | |
| 938 | + /* If we found no Halant we are done. | |
| 939 | + * Otherwise only proceed if the Halant does | |
| 940 | + * not belong to the Matra itself! */ | |
| 941 | + if (self::is_halant_or_coeng($info[$new_pos]) && $info[$new_pos]['indic_position'] != self::POS_PRE_M) { | |
| 942 | + /* -> If ZWJ or ZWNJ follow this halant, position is moved after it. */ | |
| 943 | + if ($new_pos + 1 < $end && self::is_joiner($info[$new_pos + 1])) | |
| 944 | + $new_pos++; | |
| 945 | + } else | |
| 946 | + $new_pos = $start; /* No move. */ | |
| 947 | + } | |
| 948 | + | |
| 949 | + if ($start < $new_pos && $info[$new_pos]['indic_position'] != self::POS_PRE_M) { | |
| 950 | + /* Now go see if there's actually any matras... */ | |
| 951 | + for ($i = $new_pos; $i > $start; $i--) | |
| 952 | + if ($info[$i - 1]['indic_position'] == self::POS_PRE_M) { | |
| 953 | + $old_pos = $i - 1; | |
| 954 | + //memmove (&info[$old_pos], &info[$old_pos + 1], ($new_pos - $old_pos) * sizeof ($info[0])); | |
| 955 | + self::_move_info_pos($info, $old_pos, $new_pos + 1); | |
| 956 | + | |
| 957 | + if ($old_pos < $base && $base <= $new_pos) /* Shouldn't actually happen. */ | |
| 958 | + $base--; | |
| 959 | + $new_pos--; | |
| 960 | + } | |
| 961 | + } | |
| 962 | + } | |
| 963 | + | |
| 964 | + | |
| 965 | + /* o Reorder reph: | |
| 966 | + * | |
| 967 | + * Reph's original position is always at the beginning of the syllable, | |
| 968 | + * (i.e. it is not reordered at the character reordering stage). However, | |
| 969 | + * it will be reordered according to the basic-forms shaping results. | |
| 970 | + * Possible positions for reph, depending on the script, are; after main, | |
| 971 | + * before post-base consonant forms, and after post-base consonant forms. | |
| 972 | + */ | |
| 973 | + | |
| 974 | + /* If there's anything after the Ra that has the REPH pos, it ought to be halant. | |
| 975 | + * Which means that the font has failed to ligate the Reph. In which case, we | |
| 976 | + * shouldn't move. */ | |
| 977 | + if ($start + 1 < $end && | |
| 978 | + $info[$start]['indic_position'] == self::POS_RA_TO_BECOME_REPH && $info[$start + 1]['indic_position'] != self::POS_RA_TO_BECOME_REPH) { | |
| 979 | + $reph_pos = $indic_config[3]; | |
| 980 | + $skip_to_reph_step_5 = false; | |
| 981 | + $skip_to_reph_move = false; | |
| 982 | + | |
| 983 | + /* 1. If reph should be positioned after post-base consonant forms, | |
| 984 | + * proceed to step 5. | |
| 985 | + */ | |
| 986 | + if ($reph_pos == self::REPH_POS_AFTER_POST) { | |
| 987 | + $skip_to_reph_step_5 = true; | |
| 988 | + } | |
| 989 | + | |
| 990 | + /* 2. If the reph repositioning class is not after post-base: target | |
| 991 | + * position is after the first explicit halant glyph between the | |
| 992 | + * first post-reph consonant and last main consonant. If ZWJ or ZWNJ | |
| 993 | + * are following this halant, position is moved after it. If such | |
| 994 | + * position is found, this is the target position. Otherwise, | |
| 995 | + * proceed to the next step. | |
| 996 | + * | |
| 997 | + * Note: in old-implementation fonts, where classifications were | |
| 998 | + * fixed in shaping engine, there was no case where reph position | |
| 999 | + * will be found on this step. | |
| 1000 | + */ | |
| 1001 | + | |
| 1002 | + if (!$skip_to_reph_step_5) { | |
| 1003 | + | |
| 1004 | + $new_reph_pos = $start + 1; | |
| 1005 | + | |
| 1006 | + while ($new_reph_pos < $base && !self::is_halant_or_coeng($info[$new_reph_pos])) | |
| 1007 | + $new_reph_pos++; | |
| 1008 | + | |
| 1009 | + if ($new_reph_pos < $base && self::is_halant_or_coeng($info[$new_reph_pos])) { | |
| 1010 | + /* ->If ZWJ or ZWNJ are following this halant, position is moved after it. */ | |
| 1011 | + if ($new_reph_pos + 1 < $base && self::is_joiner($info[$new_reph_pos + 1])) | |
| 1012 | + $new_reph_pos++; | |
| 1013 | + $skip_to_reph_move = true; | |
| 1014 | + } | |
| 1015 | + } | |
| 1016 | + | |
| 1017 | + /* 3. If reph should be repositioned after the main consonant: find the | |
| 1018 | + * first consonant not ligated with main, or find the first | |
| 1019 | + * consonant that is not a potential pre-base reordering Ra. | |
| 1020 | + */ | |
| 1021 | + if ($reph_pos == self::REPH_POS_AFTER_MAIN && !$skip_to_reph_move && !$skip_to_reph_step_5) { | |
| 1022 | + $new_reph_pos = $base; | |
| 1023 | + /* XXX Skip potential pre-base reordering Ra. */ | |
| 1024 | + while ($new_reph_pos + 1 < $end && $info[$new_reph_pos + 1]['indic_position'] <= self::POS_AFTER_MAIN) | |
| 1025 | + $new_reph_pos++; | |
| 1026 | + if ($new_reph_pos < $end) | |
| 1027 | + $skip_to_reph_move = true; | |
| 1028 | + } | |
| 1029 | + | |
| 1030 | + /* 4. If reph should be positioned before post-base consonant, find | |
| 1031 | + * first post-base classified consonant not ligated with main. If no | |
| 1032 | + * consonant is found, the target position should be before the | |
| 1033 | + * first matra, syllable modifier sign or vedic sign. | |
| 1034 | + */ | |
| 1035 | + /* This is our take on what step 4 is trying to say (and failing, BADLY). */ | |
| 1036 | + if ($reph_pos == self::REPH_POS_AFTER_SUB && !$skip_to_reph_move && !$skip_to_reph_step_5) { | |
| 1037 | + $new_reph_pos = $base; | |
| 1038 | + while ($new_reph_pos < $end && isset($info[$new_reph_pos + 1]['indic_position']) && | |
| 1039 | + !( self::FLAG($info[$new_reph_pos + 1]['indic_position']) & (self::FLAG(self::POS_POST_C) | self::FLAG(self::POS_AFTER_POST) | self::FLAG(self::POS_SMVD)))) { | |
| 1040 | + $new_reph_pos++; | |
| 1041 | + } | |
| 1042 | + if ($new_reph_pos < $end) { | |
| 1043 | + $skip_to_reph_move = true; | |
| 1044 | + } | |
| 1045 | + } | |
| 1046 | + | |
| 1047 | + /* 5. If no consonant is found in steps 3 or 4, move reph to a position | |
| 1048 | + * immediately before the first post-base matra, syllable modifier | |
| 1049 | + * sign or vedic sign that has a reordering class after the intended | |
| 1050 | + * reph position. For example, if the reordering position for reph | |
| 1051 | + * is post-main, it will skip above-base matras that also have a | |
| 1052 | + * post-main position. | |
| 1053 | + */ | |
| 1054 | + if (!$skip_to_reph_move) { | |
| 1055 | + /* Copied from step 2. */ | |
| 1056 | + $new_reph_pos = $start + 1; | |
| 1057 | + while ($new_reph_pos < $base && !self::is_halant_or_coeng($info[$new_reph_pos])) | |
| 1058 | + $new_reph_pos++; | |
| 1059 | + | |
| 1060 | + if ($new_reph_pos < $base && self::is_halant_or_coeng($info[$new_reph_pos])) { | |
| 1061 | + /* ->If ZWJ or ZWNJ are following this halant, position is moved after it. */ | |
| 1062 | + if ($new_reph_pos + 1 < $base && self::is_joiner($info[$new_reph_pos + 1])) | |
| 1063 | + $new_reph_pos++; | |
| 1064 | + $skip_to_reph_move = true; | |
| 1065 | + } | |
| 1066 | + } | |
| 1067 | + | |
| 1068 | + | |
| 1069 | + /* 6. Otherwise, reorder reph to the end of the syllable. | |
| 1070 | + */ | |
| 1071 | + if (!$skip_to_reph_move) { | |
| 1072 | + $new_reph_pos = $end - 1; | |
| 1073 | + while ($new_reph_pos > $start && $info[$new_reph_pos]['indic_position'] == self::POS_SMVD) | |
| 1074 | + $new_reph_pos--; | |
| 1075 | + | |
| 1076 | + /* | |
| 1077 | + * If the Reph is to be ending up after a Matra,Halant sequence, | |
| 1078 | + * position it before that Halant so it can interact with the Matra. | |
| 1079 | + * However, if it's a plain Consonant,Halant we shouldn't do that. | |
| 1080 | + * Uniscribe doesn't do this. | |
| 1081 | + * TEST: U+0930,U+094D,U+0915,U+094B,U+094D | |
| 1082 | + */ | |
| 1083 | + //if (!$hb_options.uniscribe_bug_compatible && self::is_halant_or_coeng($info[$new_reph_pos])) { | |
| 1084 | + if (self::is_halant_or_coeng($info[$new_reph_pos])) { | |
| 1085 | + for ($i = $base + 1; $i < $new_reph_pos; $i++) | |
| 1086 | + if ($info[$i]['indic_category'] == self::OT_M) { | |
| 1087 | + /* Ok, got it. */ | |
| 1088 | + $new_reph_pos--; | |
| 1089 | + } | |
| 1090 | + } | |
| 1091 | + } | |
| 1092 | + | |
| 1093 | + | |
| 1094 | + /* Move */ | |
| 1095 | + self::_move_info_pos($info, $start, $new_reph_pos + 1); | |
| 1096 | + | |
| 1097 | + if ($start < $base && $base <= $new_reph_pos) { | |
| 1098 | + $base--; | |
| 1099 | + } | |
| 1100 | + } | |
| 1101 | + | |
| 1102 | + | |
| 1103 | + /* o Reorder pre-base reordering consonants: | |
| 1104 | + * | |
| 1105 | + * If a pre-base reordering consonant is found, reorder it according to | |
| 1106 | + * the following rules: | |
| 1107 | + */ | |
| 1108 | + | |
| 1109 | + | |
| 1110 | + if (count($GSUBdata['pref']) && $base + 1 < $end) { /* Otherwise there can't be any pre-base reordering Ra. */ | |
| 1111 | + for ($i = $base + 1; $i < $end; $i++) { | |
| 1112 | + if ($info[$i]['mask'] & self::FLAG(self::PREF)) { | |
| 1113 | + /* 1. Only reorder a glyph produced by substitution during application | |
| 1114 | + * of the <pref> feature. (Note that a font may shape a Ra consonant with | |
| 1115 | + * the feature generally but block it in certain contexts.) | |
| 1116 | + */ | |
| 1117 | +// ??? Need to TEST if actual substitution has occurred | |
| 1118 | + if ($i + 1 == $end || ($info[$i + 1]['mask'] & self::FLAG(self::PREF)) == 0) { | |
| 1119 | + /* | |
| 1120 | + * 2. Try to find a target position the same way as for pre-base matra. | |
| 1121 | + * If it is found, reorder pre-base consonant glyph. | |
| 1122 | + * | |
| 1123 | + * 3. If position is not found, reorder immediately before main | |
| 1124 | + * consonant. | |
| 1125 | + */ | |
| 1126 | + $new_pos = $base; | |
| 1127 | + /* Malayalam / Tamil do not have "half" forms or explicit virama forms. | |
| 1128 | + * The glyphs formed by 'half' are Chillus or ligated explicit viramas. | |
| 1129 | + * We want to position matra after them. | |
| 1130 | + */ | |
| 1131 | + if ($scriptblock != UCDN::SCRIPT_MALAYALAM && $scriptblock != UCDN::SCRIPT_TAMIL) { | |
| 1132 | + while ($new_pos > $start && | |
| 1133 | + !(self::is_one_of($info[$new_pos - 1], self::FLAG(self::OT_M) | self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng)))) | |
| 1134 | + $new_pos--; | |
| 1135 | + | |
| 1136 | + /* In Khmer coeng model, a V,Ra can go *after* matras. If it goes after a | |
| 1137 | + * split matra, it should be reordered to *before* the left part of such matra. */ | |
| 1138 | + if ($new_pos > $start && $info[$new_pos - 1]['indic_category'] == self::OT_M) { | |
| 1139 | + $old_pos = i; | |
| 1140 | + for ($i = $base + 1; $i < $old_pos; $i++) | |
| 1141 | + if ($info[$i]['indic_category'] == self::OT_M) { | |
| 1142 | + $new_pos--; | |
| 1143 | + break; | |
| 1144 | + } | |
| 1145 | + } | |
| 1146 | + } | |
| 1147 | + | |
| 1148 | + if ($new_pos > $start && self::is_halant_or_coeng($info[$new_pos - 1])) { | |
| 1149 | + /* -> If ZWJ or ZWNJ follow this halant, position is moved after it. */ | |
| 1150 | + if ($new_pos < $end && self::is_joiner($info[$new_pos])) | |
| 1151 | + $new_pos++; | |
| 1152 | + } | |
| 1153 | + | |
| 1154 | + $old_pos = $i; | |
| 1155 | + self::_move_info_pos($info, $old_pos, $new_pos); | |
| 1156 | + | |
| 1157 | + if ($new_pos <= $base && $base < $old_pos) | |
| 1158 | + $base++; | |
| 1159 | + } | |
| 1160 | + | |
| 1161 | + break; | |
| 1162 | + } | |
| 1163 | + } | |
| 1164 | + } | |
| 1165 | + | |
| 1166 | + | |
| 1167 | + /* Apply 'init' to the Left Matra if it's a word start. */ | |
| 1168 | + if ($info[$start]['indic_position'] == self::POS_PRE_M && | |
| 1169 | + ($start == 0 || | |
| 1170 | + ($info[$start - 1]['general_category'] < UCDN::UNICODE_GENERAL_CATEGORY_FORMAT || $info[$start - 1]['general_category'] > UCDN::UNICODE_GENERAL_CATEGORY_NON_SPACING_MARK) | |
| 1171 | + )) { | |
| 1172 | + $info[$start]['mask'] |= self::FLAG(self::INIT); | |
| 1173 | + } | |
| 1174 | + | |
| 1175 | + | |
| 1176 | + /* | |
| 1177 | + * Finish off and go home! | |
| 1178 | + */ | |
| 1179 | + } | |
| 1180 | + | |
| 1181 | + public static function _move_info_pos(&$info, $from, $to) | |
| 1182 | + { | |
| 1183 | + $t = array(); | |
| 1184 | + $t[0] = $info[$from]; | |
| 1185 | + if ($from > $to) { | |
| 1186 | + array_splice($info, $from, 1); | |
| 1187 | + array_splice($info, $to, 0, $t); | |
| 1188 | + } else { | |
| 1189 | + array_splice($info, $to, 0, $t); | |
| 1190 | + array_splice($info, $from, 1); | |
| 1191 | + } | |
| 1192 | + } | |
| 1193 | + | |
| 1194 | + public static $ra_chars = array( | |
| 1195 | + 0x0930 => 1, /* Devanagari */ | |
| 1196 | + 0x09B0 => 1, /* Bengali */ | |
| 1197 | + 0x09F0 => 1, /* Bengali (Assamese) */ | |
| 1198 | + 0x0A30 => 1, /* Gurmukhi */ /* No Reph */ | |
| 1199 | + 0x0AB0 => 1, /* Gujarati */ | |
| 1200 | + 0x0B30 => 1, /* Oriya */ | |
| 1201 | + 0x0BB0 => 1, /* Tamil */ /* No Reph */ | |
| 1202 | + 0x0C30 => 1, /* Telugu */ /* Reph formed only with ZWJ */ | |
| 1203 | + 0x0CB0 => 1, /* Kannada */ | |
| 1204 | + 0x0D30 => 1, /* Malayalam */ /* No Reph, Logical Repha */ | |
| 1205 | + 0x0DBB => 1, /* Sinhala */ /* Reph formed only with ZWJ */ | |
| 1206 | + 0x179A => 1, /* Khmer */ /* No Reph, Visual Repha */ | |
| 1207 | + ); | |
| 1208 | + | |
| 1209 | + public static function is_ra($u) | |
| 1210 | + { | |
| 1211 | + if (isset(self::$ra_chars[$u])) | |
| 1212 | + return true; | |
| 1213 | + return false; | |
| 1214 | + } | |
| 1215 | + | |
| 1216 | + public static function is_one_of($info, $flags) | |
| 1217 | + { | |
| 1218 | + if (isset($info['is_ligature']) && $info['is_ligature']) | |
| 1219 | + return false; /* If it ligated, all bets are off. */ | |
| 1220 | + return !!(self::FLAG($info['indic_category']) & $flags); | |
| 1221 | + } | |
| 1222 | + | |
| 1223 | + public static function is_joiner($info) | |
| 1224 | + { | |
| 1225 | + return self::is_one_of($info, (self::FLAG(self::OT_ZWJ) | self::FLAG(self::OT_ZWNJ))); | |
| 1226 | + } | |
| 1227 | + | |
| 1228 | + /* Vowels and placeholders treated as if they were consonants. */ | |
| 1229 | + | |
| 1230 | + public static function is_consonant($info) | |
| 1231 | + { | |
| 1232 | + return self::is_one_of($info, (self::FLAG(self::OT_C) | self::FLAG(self::OT_CM) | self::FLAG(self::OT_Ra) | self::FLAG(self::OT_V) | self::FLAG(self::OT_NBSP) | self::FLAG(self::OT_DOTTEDCIRCLE))); | |
| 1233 | + } | |
| 1234 | + | |
| 1235 | + public static function is_halant_or_coeng($info) | |
| 1236 | + { | |
| 1237 | + return self::is_one_of($info, (self::FLAG(self::OT_H) | self::FLAG(self::OT_Coeng))); | |
| 1238 | + } | |
| 1239 | + | |
| 1240 | + // From hb-private.hh | |
| 1241 | + public static function in_range($u, $lo, $hi) | |
| 1242 | + { | |
| 1243 | + if ((($lo ^ $hi) & $lo) == 0 && (($lo ^ $hi) & $hi) == ($lo ^ $hi) && (($lo ^ $hi) & (($lo ^ $hi) + 1)) == 0) | |
| 1244 | + return ($u & ~($lo ^ $hi)) == $lo; | |
| 1245 | + else | |
| 1246 | + return $lo <= $u && $u <= $hi; | |
| 1247 | + } | |
| 1248 | + | |
| 1249 | + // From hb-private.hh | |
| 1250 | + public static function FLAG($x) | |
| 1251 | + { | |
| 1252 | + return (1 << ($x)); | |
| 1253 | + } | |
| 1254 | + | |
| 1255 | + // BELOW from hb-ot-shape-complex-indic.cc | |
| 1256 | + | |
| 1257 | + /* | |
| 1258 | + * Indic configurations. | |
| 1259 | + */ | |
| 1260 | + | |
| 1261 | + // base_position | |
| 1262 | + const BASE_POS_FIRST = 0; | |
| 1263 | + const BASE_POS_LAST = 1; | |
| 1264 | + | |
| 1265 | + // reph_position | |
| 1266 | + const REPH_POS_DEFAULT = 10; // POS_BEFORE_POST, | |
| 1267 | + | |
| 1268 | + const REPH_POS_AFTER_MAIN = 5; // POS_AFTER_MAIN, | |
| 1269 | + | |
| 1270 | + const REPH_POS_BEFORE_SUB = 7; // POS_BEFORE_SUB, | |
| 1271 | + const REPH_POS_AFTER_SUB = 9; // POS_AFTER_SUB, | |
| 1272 | + const REPH_POS_BEFORE_POST = 10; // POS_BEFORE_POST, | |
| 1273 | + const REPH_POS_AFTER_POST = 12; // POS_AFTER_POST | |
| 1274 | + | |
| 1275 | + // reph_mode | |
| 1276 | + const REPH_MODE_IMPLICIT = 0; /* Reph formed out of initial Ra,H sequence. */ | |
| 1277 | + const REPH_MODE_EXPLICIT = 1; /* Reph formed out of initial Ra,H,ZWJ sequence. */ | |
| 1278 | + const REPH_MODE_VIS_REPHA = 2; /* Encoded Repha character, no reordering needed. */ | |
| 1279 | + const REPH_MODE_LOG_REPHA = 3; /* Encoded Repha character, needs reordering. */ | |
| 1280 | + | |
| 1281 | + /* | |
| 1282 | + struct of indic_configs{ | |
| 1283 | + KEY - script; | |
| 1284 | + 0 - has_old_spec; | |
| 1285 | + 1 - virama; | |
| 1286 | + 2 - base_pos; | |
| 1287 | + 3 - reph_pos; | |
| 1288 | + 4 - reph_mode; | |
| 1289 | + }; | |
| 1290 | + */ | |
| 1291 | + | |
| 1292 | + public static $indic_configs = array(/* index is SCRIPT_number from UCDN */ | |
| 1293 | + 9 => array(true, 0x094D, 1, 10, 0), | |
| 1294 | + 10 => array(true, 0x09CD, 1, 9, 0), | |
| 1295 | + 11 => array(true, 0x0A4D, 1, 7, 0), | |
| 1296 | + 12 => array(true, 0x0ACD, 1, 10, 0), | |
| 1297 | + 13 => array(true, 0x0B4D, 1, 5, 0), | |
| 1298 | + 14 => array(true, 0x0BCD, 1, 12, 0), | |
| 1299 | + 15 => array(true, 0x0C4D, 1, 12, 1), | |
| 1300 | + 16 => array(true, 0x0CCD, 1, 12, 0), | |
| 1301 | + 17 => array(true, 0x0D4D, 1, 5, 3), | |
| 1302 | + 18 => array(false, 0x0DCA, 0, 5, 1), /* Sinhala */ | |
| 1303 | + 30 => array(false, 0x17D2, 0, 10, 2), /* Khmer */ | |
| 1304 | + 84 => array(false, 0xA9C0, 1, 10, 0), /* Javanese */ | |
| 1305 | + ); | |
| 1306 | + | |
| 1307 | + | |
| 1308 | + | |
| 1309 | + /* | |
| 1310 | + | |
| 1311 | + // from "hb-ot-shape-complex-indic-table.cc" | |
| 1312 | + | |
| 1313 | + | |
| 1314 | + const ISC_A = 0; // INDIC_SYLLABIC_CATEGORY_AVAGRAHA Avagraha | |
| 1315 | + const ISC_Bi = 8; // INDIC_SYLLABIC_CATEGORY_BINDU Bindu | |
| 1316 | + const ISC_C = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT Consonant | |
| 1317 | + const ISC_CD = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT_DEAD Consonant_Dead | |
| 1318 | + const ISC_CF = 17; // INDIC_SYLLABIC_CATEGORY_CONSONANT_FINAL Consonant_Final | |
| 1319 | + const ISC_CHL = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT_HEAD_LETTER Consonant_Head_Letter | |
| 1320 | + const ISC_CM = 17; // INDIC_SYLLABIC_CATEGORY_CONSONANT_MEDIAL Consonant_Medial | |
| 1321 | + const ISC_CP = 11; // INDIC_SYLLABIC_CATEGORY_CONSONANT_PLACEHOLDER Consonant_Placeholder | |
| 1322 | + const ISC_CR = 15; // INDIC_SYLLABIC_CATEGORY_CONSONANT_REPHA Consonant_Repha | |
| 1323 | + const ISC_CS = 1; // INDIC_SYLLABIC_CATEGORY_CONSONANT_SUBJOINED Consonant_Subjoined | |
| 1324 | + const ISC_ML = 0; // INDIC_SYLLABIC_CATEGORY_MODIFYING_LETTER Modifying_Letter | |
| 1325 | + const ISC_N = 3; // INDIC_SYLLABIC_CATEGORY_NUKTA Nukta | |
| 1326 | + const ISC_x = 0; // INDIC_SYLLABIC_CATEGORY_OTHER Other | |
| 1327 | + const ISC_RS = 13; // INDIC_SYLLABIC_CATEGORY_REGISTER_SHIFTER Register_Shifter | |
| 1328 | + const ISC_TL = 0; // INDIC_SYLLABIC_CATEGORY_TONE_LETTER Tone_Letter | |
| 1329 | + const ISC_TM = 3; // INDIC_SYLLABIC_CATEGORY_TONE_MARK Tone_Mark | |
| 1330 | + const ISC_V = 4; // INDIC_SYLLABIC_CATEGORY_VIRAMA Virama | |
| 1331 | + const ISC_Vs = 8; // INDIC_SYLLABIC_CATEGORY_VISARGA Visarga | |
| 1332 | + const ISC_Vo = 2; // INDIC_SYLLABIC_CATEGORY_VOWEL Vowel | |
| 1333 | + const ISC_M = 7; // INDIC_SYLLABIC_CATEGORY_VOWEL_DEPENDENT Vowel_Dependent | |
| 1334 | + const ISC_VI = 2; // INDIC_SYLLABIC_CATEGORY_VOWEL_INDEPENDENT Vowel_Independent | |
| 1335 | + | |
| 1336 | + const IMC_B = 8; // INDIC_MATRA_CATEGORY_BOTTOM Bottom | |
| 1337 | + const IMC_BR = 11; // INDIC_MATRA_CATEGORY_BOTTOM_AND_RIGHT Bottom_And_Right | |
| 1338 | + const IMC_I = 15; // INDIC_MATRA_CATEGORY_INVISIBLE Invisible | |
| 1339 | + const IMC_L = 3; // INDIC_MATRA_CATEGORY_LEFT Left | |
| 1340 | + const IMC_LR = 11; // INDIC_MATRA_CATEGORY_LEFT_AND_RIGHT Left_And_Right | |
| 1341 | + const IMC_x = 15; // INDIC_MATRA_CATEGORY_NOT_APPLICABLE Not_Applicable | |
| 1342 | + const IMC_O = 5; // INDIC_MATRA_CATEGORY_OVERSTRUCK Overstruck | |
| 1343 | + const IMC_R = 11; // INDIC_MATRA_CATEGORY_RIGHT Right | |
| 1344 | + const IMC_T = 6; // INDIC_MATRA_CATEGORY_TOP Top | |
| 1345 | + const IMC_TB = 8; // INDIC_MATRA_CATEGORY_TOP_AND_BOTTOM Top_And_Bottom | |
| 1346 | + const IMC_TBR = 11; // INDIC_MATRA_CATEGORY_TOP_AND_BOTTOM_AND_RIGHT Top_And_Bottom_And_Right | |
| 1347 | + const IMC_TL = 6; // INDIC_MATRA_CATEGORY_TOP_AND_LEFT Top_And_Left | |
| 1348 | + const IMC_TLR = 11; // INDIC_MATRA_CATEGORY_TOP_AND_LEFT_AND_RIGHT Top_And_Left_And_Right | |
| 1349 | + const IMC_TR = 11; // INDIC_MATRA_CATEGORY_TOP_AND_RIGHT Top_And_Right | |
| 1350 | + const IMC_VOL = 2; // INDIC_MATRA_CATEGORY_VISUAL_ORDER_LEFT Visual_Order_Left | |
| 1351 | + | |
| 1352 | + If in original table = _(C,x), that = ISC_C,IMC_x | |
| 1353 | + Value is IMC_x << 8 (or IMC_x * 256) = 3840 | |
| 1354 | + plus ISC_C = 1, so = 3841 | |
| 1355 | + | |
| 1356 | + */ | |
| 1357 | + | |
| 1358 | + public static $indic_table = array( | |
| 1359 | + /* Devanagari (0900..097F) */ | |
| 1360 | + | |
| 1361 | + /* 0900 */ 3848, 3848, 3848, 3848, 3842, 3842, 3842, 3842, | |
| 1362 | + /* 0908 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3842, | |
| 1363 | + /* 0910 */ 3842, 3842, 3842, 3842, 3842, 3841, 3841, 3841, | |
| 1364 | + /* 0918 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1365 | + /* 0920 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1366 | + /* 0928 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1367 | + /* 0930 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1368 | + /* 0938 */ 3841, 3841, 1543, 2823, 3843, 3840, 2823, 775, | |
| 1369 | + /* 0940 */ 2823, 2055, 2055, 2055, 2055, 1543, 1543, 1543, | |
| 1370 | + /* 0948 */ 1543, 2823, 2823, 2823, 2823, 2052, 775, 2823, | |
| 1371 | + /* 0950 */ 3840, 3840, 3840, 3840, 3840, 1543, 2055, 2055, | |
| 1372 | + /* 0958 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1373 | + /* 0960 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1374 | + /* 0968 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1375 | + /* 0970 */ 3840, 3840, 3842, 3842, 3842, 3842, 3842, 3842, | |
| 1376 | + /* 0978 */ 3840, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1377 | + /* Bengali (0980..09FF) */ | |
| 1378 | + | |
| 1379 | + /* 0980 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1380 | + /* 0988 */ 3842, 3842, 3842, 3842, 3842, 3840, 3840, 3842, | |
| 1381 | + /* 0990 */ 3842, 3840, 3840, 3842, 3842, 3841, 3841, 3841, | |
| 1382 | + /* 0998 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1383 | + /* 09A0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1384 | + /* 09A8 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1385 | + /* 09B0 */ 3841, 3840, 3841, 3840, 3840, 3840, 3841, 3841, | |
| 1386 | + /* 09B8 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 775, | |
| 1387 | + /* 09C0 */ 2823, 2055, 2055, 2055, 2055, 3840, 3840, 775, | |
| 1388 | + /* 09C8 */ 775, 3840, 3840, 2823, 2823, 2052, 3841, 3840, | |
| 1389 | + /* 09D0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 2823, | |
| 1390 | + /* 09D8 */ 3840, 3840, 3840, 3840, 3841, 3841, 3840, 3841, | |
| 1391 | + /* 09E0 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1392 | + /* 09E8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1393 | + /* 09F0 */ 3841, 3841, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1394 | + /* 09F8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1395 | + /* Gurmukhi (0A00..0A7F) */ | |
| 1396 | + | |
| 1397 | + /* 0A00 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1398 | + /* 0A08 */ 3842, 3842, 3842, 3840, 3840, 3840, 3840, 3842, | |
| 1399 | + /* 0A10 */ 3842, 3840, 3840, 3842, 3842, 3841, 3841, 3841, | |
| 1400 | + /* 0A18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1401 | + /* 0A20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1402 | + /* 0A28 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1403 | + /* 0A30 */ 3841, 3840, 3841, 3841, 3840, 3841, 3841, 3840, | |
| 1404 | + /* 0A38 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 775, | |
| 1405 | + /* 0A40 */ 2823, 2055, 2055, 3840, 3840, 3840, 3840, 1543, | |
| 1406 | + /* 0A48 */ 1543, 3840, 3840, 1543, 1543, 2052, 3840, 3840, | |
| 1407 | + /* 0A50 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1408 | + /* 0A58 */ 3840, 3841, 3841, 3841, 3841, 3840, 3841, 3840, | |
| 1409 | + /* 0A60 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1410 | + /* 0A68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1411 | + /* 0A70 */ 3848, 3840, 13841, 13841, 3840, 3857, 3840, 3840, | |
| 1412 | + /* 0A78 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1413 | + /* Gujarati (0A80..0AFF) */ | |
| 1414 | + | |
| 1415 | + /* 0A80 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1416 | + /* 0A88 */ 3842, 3842, 3842, 3842, 3842, 3842, 3840, 3842, | |
| 1417 | + /* 0A90 */ 3842, 3842, 3840, 3842, 3842, 3841, 3841, 3841, | |
| 1418 | + /* 0A98 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1419 | + /* 0AA0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1420 | + /* 0AA8 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1421 | + /* 0AB0 */ 3841, 3840, 3841, 3841, 3840, 3841, 3841, 3841, | |
| 1422 | + /* 0AB8 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 775, | |
| 1423 | + /* 0AC0 */ 2823, 2055, 2055, 2055, 2055, 1543, 3840, 1543, | |
| 1424 | + /* 0AC8 */ 1543, 2823, 3840, 2823, 2823, 2052, 3840, 3840, | |
| 1425 | + /* 0AD0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1426 | + /* 0AD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1427 | + /* 0AE0 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1428 | + /* 0AE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1429 | + /* 0AF0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1430 | + /* 0AF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1431 | + /* Oriya (0B00..0B7F) */ | |
| 1432 | + | |
| 1433 | + /* 0B00 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1434 | + /* 0B08 */ 3842, 3842, 3842, 3842, 3842, 3840, 3840, 3842, | |
| 1435 | + /* 0B10 */ 3842, 3840, 3840, 3842, 3842, 3841, 3841, 3841, | |
| 1436 | + /* 0B18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1437 | + /* 0B20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1438 | + /* 0B28 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1439 | + /* 0B30 */ 3841, 3840, 3841, 3841, 3840, 3841, 3841, 3841, | |
| 1440 | + /* 0B38 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 1543, | |
| 1441 | + /* 0B40 */ 2823, 2055, 2055, 2055, 2055, 3840, 3840, 775, | |
| 1442 | + /* 0B48 */ 1543, 3840, 3840, 2823, 2823, 2052, 3840, 3840, | |
| 1443 | + /* 0B50 */ 3840, 3840, 3840, 3840, 3840, 3840, 1543, 2823, | |
| 1444 | + /* 0B58 */ 3840, 3840, 3840, 3840, 3841, 3841, 3840, 3841, | |
| 1445 | + /* 0B60 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1446 | + /* 0B68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1447 | + /* 0B70 */ 3840, 3841, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1448 | + /* 0B78 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1449 | + /* Tamil (0B80..0BFF) */ | |
| 1450 | + | |
| 1451 | + /* 0B80 */ 3840, 3840, 3848, 3840, 3840, 3842, 3842, 3842, | |
| 1452 | + /* 0B88 */ 3842, 3842, 3842, 3840, 3840, 3840, 3842, 3842, | |
| 1453 | + /* 0B90 */ 3842, 3840, 3842, 3842, 3842, 3841, 3840, 3840, | |
| 1454 | + /* 0B98 */ 3840, 3841, 3841, 3840, 3841, 3840, 3841, 3841, | |
| 1455 | + /* 0BA0 */ 3840, 3840, 3840, 3841, 3841, 3840, 3840, 3840, | |
| 1456 | + /* 0BA8 */ 3841, 3841, 3841, 3840, 3840, 3840, 3841, 3841, | |
| 1457 | + /* 0BB0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1458 | + /* 0BB8 */ 3841, 3841, 3840, 3840, 3840, 3840, 2823, 2823, | |
| 1459 | + /* 0BC0 */ 1543, 2055, 2055, 3840, 3840, 3840, 775, 775, | |
| 1460 | + /* 0BC8 */ 775, 3840, 2823, 2823, 2823, 1540, 3840, 3840, | |
| 1461 | + /* 0BD0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 2823, | |
| 1462 | + /* 0BD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1463 | + /* 0BE0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1464 | + /* 0BE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1465 | + /* 0BF0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1466 | + /* 0BF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1467 | + /* Telugu (0C00..0C7F) */ | |
| 1468 | + | |
| 1469 | + /* 0C00 */ 3840, 3848, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1470 | + /* 0C08 */ 3842, 3842, 3842, 3842, 3842, 3840, 3842, 3842, | |
| 1471 | + /* 0C10 */ 3842, 3840, 3842, 3842, 3842, 3841, 3841, 3841, | |
| 1472 | + /* 0C18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1473 | + /* 0C20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1474 | + /* 0C28 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1475 | + /* 0C30 */ 3841, 3841, 3841, 3841, 3840, 3841, 3841, 3841, | |
| 1476 | + /* 0C38 */ 3841, 3841, 3840, 3840, 3840, 3840, 1543, 1543, | |
| 1477 | + /* 0C40 */ 1543, 2823, 2823, 2823, 2823, 3840, 1543, 1543, | |
| 1478 | + /* 0C48 */ 2055, 3840, 1543, 1543, 1543, 1540, 3840, 3840, | |
| 1479 | + /* 0C50 */ 3840, 3840, 3840, 3840, 3840, 1543, 2055, 3840, | |
| 1480 | + /* 0C58 */ 3841, 3841, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1481 | + /* 0C60 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1482 | + /* 0C68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1483 | + /* 0C70 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1484 | + /* 0C78 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1485 | + /* Kannada (0C80..0CFF) */ | |
| 1486 | + | |
| 1487 | + /* 0C80 */ 3840, 3840, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1488 | + /* 0C88 */ 3842, 3842, 3842, 3842, 3842, 3840, 3842, 3842, | |
| 1489 | + /* 0C90 */ 3842, 3840, 3842, 3842, 3842, 3841, 3841, 3841, | |
| 1490 | + /* 0C98 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1491 | + /* 0CA0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1492 | + /* 0CA8 */ 3841, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1493 | + /* 0CB0 */ 3841, 3841, 3841, 3841, 3840, 3841, 3841, 3841, | |
| 1494 | + /* 0CB8 */ 3841, 3841, 3840, 3840, 3843, 3840, 2823, 1543, | |
| 1495 | + /* 0CC0 */ 2823, 2823, 2823, 2823, 2823, 3840, 1543, 2823, | |
| 1496 | + /* 0CC8 */ 2823, 3840, 2823, 2823, 1543, 1540, 3840, 3840, | |
| 1497 | + /* 0CD0 */ 3840, 3840, 3840, 3840, 3840, 2823, 2823, 3840, | |
| 1498 | + /* 0CD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3841, 3840, | |
| 1499 | + /* 0CE0 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1500 | + /* 0CE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1501 | + /* 0CF0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1502 | + /* 0CF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1503 | + /* Malayalam (0D00..0D7F) */ | |
| 1504 | + | |
| 1505 | + /* 0D00 */ 3840, 3840, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1506 | + /* 0D08 */ 3842, 3842, 3842, 3842, 3842, 3840, 3842, 3842, | |
| 1507 | + /* 0D10 */ 3842, 3840, 3842, 3842, 3842, 3841, 3841, 3841, | |
| 1508 | + /* 0D18 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1509 | + /* 0D20 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1510 | + /* 0D28 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1511 | + /* 0D30 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1512 | + /* 0D38 */ 3841, 3841, 3841, 3840, 3840, 3840, 2823, 2823, | |
| 1513 | + /* 0D40 */ 2823, 2823, 2823, 2055, 2055, 3840, 775, 775, | |
| 1514 | + /* 0D48 */ 775, 3840, 2823, 2823, 2823, 1540, 3855, 3840, | |
| 1515 | + /* 0D50 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 2823, | |
| 1516 | + /* 0D58 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1517 | + /* 0D60 */ 3842, 3842, 2055, 2055, 3840, 3840, 3840, 3840, | |
| 1518 | + /* 0D68 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1519 | + /* 0D70 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1520 | + /* 0D78 */ 3840, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1521 | + /* Sinhala (0D80..0DFF) */ | |
| 1522 | + | |
| 1523 | + /* 0D80 */ 3840, 3840, 3848, 3848, 3840, 3842, 3842, 3842, | |
| 1524 | + /* 0D88 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3842, | |
| 1525 | + /* 0D90 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3840, | |
| 1526 | + /* 0D98 */ 3840, 3840, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1527 | + /* 0DA0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1528 | + /* 0DA8 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1529 | + /* 0DB0 */ 3841, 3841, 3840, 3841, 3841, 3841, 3841, 3841, | |
| 1530 | + /* 0DB8 */ 3841, 3841, 3841, 3841, 3840, 3841, 3840, 3840, | |
| 1531 | + /* 0DC0 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3840, | |
| 1532 | + /* 0DC8 */ 3840, 3840, 1540, 3840, 3840, 3840, 3840, 2823, | |
| 1533 | + /* 0DD0 */ 2823, 2823, 1543, 1543, 2055, 3840, 2055, 3840, | |
| 1534 | + /* 0DD8 */ 2823, 775, 1543, 775, 2823, 2823, 2823, 2823, | |
| 1535 | + /* 0DE0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1536 | + /* 0DE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1537 | + /* 0DF0 */ 3840, 3840, 2823, 2823, 3840, 3840, 3840, 3840, | |
| 1538 | + /* 0DF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1539 | + /* Vedic Extensions (1CD0..1CFF) */ | |
| 1540 | + | |
| 1541 | + /* 1CD0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1542 | + /* 1CD8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1543 | + /* 1CE0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1544 | + /* 1CE8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1545 | + /* 1CF0 */ 3840, 3840, 3848, 3848, 3840, 3840, 3840, 3840, | |
| 1546 | + /* 1CF8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1547 | + ); | |
| 1548 | + | |
| 1549 | + public static $khmer_table = array( | |
| 1550 | + /* Khmer (1780..17FF) */ | |
| 1551 | + | |
| 1552 | + /* 1780 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1553 | + /* 1788 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1554 | + /* 1790 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1555 | + /* 1798 */ 3841, 3841, 3841, 3841, 3841, 3841, 3841, 3841, | |
| 1556 | + /* 17A0 */ 3841, 3841, 3841, 3842, 3842, 3842, 3842, 3842, | |
| 1557 | + /* 17A8 */ 3842, 3842, 3842, 3842, 3842, 3842, 3842, 3842, | |
| 1558 | + /* 17B0 */ 3842, 3842, 3842, 3842, 3840, 3840, 2823, 1543, | |
| 1559 | + /* 17B8 */ 1543, 1543, 1543, 2055, 2055, 2055, 1543, 2823, | |
| 1560 | + /* 17C0 */ 2823, 775, 775, 775, 2823, 2823, 3848, 3848, | |
| 1561 | + /* 17C8 */ 2823, 3853, 3853, 3840, 3855, 3840, 3840, 3840, | |
| 1562 | + /* 17D0 */ 3840, 1540, 3844, 3840, 3840, 3840, 3840, 3840, | |
| 1563 | + /* 17D8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1564 | + /* 17E0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1565 | + /* 17E8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1566 | + /* 17F0 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1567 | + /* 17F8 */ 3840, 3840, 3840, 3840, 3840, 3840, 3840, 3840, | |
| 1568 | + ); | |
| 1569 | + | |
| 1570 | + // from "hb-ot-shape-complex-indic-table.cc" | |
| 1571 | + public static function indic_get_categories($u) | |
| 1572 | + { | |
| 1573 | + if (0x0900 <= $u && $u <= 0x0DFF) | |
| 1574 | + return self::$indic_table[$u - 0x0900 + 0]; // offset 0 for Most "indic" | |
| 1575 | + if (0x1CD0 <= $u && $u <= 0x1D00) | |
| 1576 | + return self::$indic_table[$u - 0x1CD0 + 1152]; // offset for Vedic extensions | |
| 1577 | + if (0x1780 <= $u && $u <= 0x17FF) | |
| 1578 | + return self::$khmer_table[$u - 0x1780]; // Khmer | |
| 1579 | + if ($u == 0x00A0) | |
| 1580 | + return 3851; // (ISC_CP | (IMC_x << 8)) | |
| 1581 | + if ($u == 0x25CC) | |
| 1582 | + return 3851; // (ISC_CP | (IMC_x << 8)) | |
| 1583 | + return 3840; // (ISC_x | (IMC_x << 8)) | |
| 1584 | + } | |
| 1585 | + | |
| 1586 | + // BELOW from hb-ot-shape-complex-indic.cc | |
| 1587 | + /* | |
| 1588 | + * Indic shaper. | |
| 1589 | + */ | |
| 1590 | + | |
| 1591 | + public static function IN_HALF_BLOCK($u, $Base) | |
| 1592 | + { | |
| 1593 | + return (($u & ~0x7F) == $Base); | |
| 1594 | + } | |
| 1595 | + | |
| 1596 | + public static function IS_DEVA($u) | |
| 1597 | + { | |
| 1598 | + return self::IN_HALF_BLOCK($u, 0x0900); | |
| 1599 | + } | |
| 1600 | + | |
| 1601 | + public static function IS_BENG($u) | |
| 1602 | + { | |
| 1603 | + return self::IN_HALF_BLOCK($u, 0x0980); | |
| 1604 | + } | |
| 1605 | + | |
| 1606 | + public static function IS_GURU($u) | |
| 1607 | + { | |
| 1608 | + return self::IN_HALF_BLOCK($u, 0x0A00); | |
| 1609 | + } | |
| 1610 | + | |
| 1611 | + public static function IS_GUJR($u) | |
| 1612 | + { | |
| 1613 | + return self::IN_HALF_BLOCK($u, 0x0A80); | |
| 1614 | + } | |
| 1615 | + | |
| 1616 | + public static function IS_ORYA($u) | |
| 1617 | + { | |
| 1618 | + return self::IN_HALF_BLOCK($u, 0x0B00); | |
| 1619 | + } | |
| 1620 | + | |
| 1621 | + public static function IS_TAML($u) | |
| 1622 | + { | |
| 1623 | + return self::IN_HALF_BLOCK($u, 0x0B80); | |
| 1624 | + } | |
| 1625 | + | |
| 1626 | + public static function IS_TELU($u) | |
| 1627 | + { | |
| 1628 | + return self::IN_HALF_BLOCK($u, 0x0C00); | |
| 1629 | + } | |
| 1630 | + | |
| 1631 | + public static function IS_KNDA($u) | |
| 1632 | + { | |
| 1633 | + return self::IN_HALF_BLOCK($u, 0x0C80); | |
| 1634 | + } | |
| 1635 | + | |
| 1636 | + public static function IS_MLYM($u) | |
| 1637 | + { | |
| 1638 | + return self::IN_HALF_BLOCK($u, 0x0D00); | |
| 1639 | + } | |
| 1640 | + | |
| 1641 | + public static function IS_SINH($u) | |
| 1642 | + { | |
| 1643 | + return self::IN_HALF_BLOCK($u, 0x0D80); | |
| 1644 | + } | |
| 1645 | + | |
| 1646 | + public static function IS_KHMR($u) | |
| 1647 | + { | |
| 1648 | + return self::IN_HALF_BLOCK($u, 0x1780); | |
| 1649 | + } | |
| 1650 | + | |
| 1651 | + public static function MATRA_POS_LEFT($u) | |
| 1652 | + { | |
| 1653 | + return self::POS_PRE_M; | |
| 1654 | + } | |
| 1655 | + | |
| 1656 | + public static function MATRA_POS_RIGHT($u) | |
| 1657 | + { | |
| 1658 | + return | |
| 1659 | + (self::IS_DEVA($u) ? self::POS_AFTER_SUB : | |
| 1660 | + (self::IS_BENG($u) ? self::POS_AFTER_POST : | |
| 1661 | + (self::IS_GURU($u) ? self::POS_AFTER_POST : | |
| 1662 | + (self::IS_GUJR($u) ? self::POS_AFTER_POST : | |
| 1663 | + (self::IS_ORYA($u) ? self::POS_AFTER_POST : | |
| 1664 | + (self::IS_TAML($u) ? self::POS_AFTER_POST : | |
| 1665 | + (self::IS_TELU($u) ? ($u <= 0x0C42 ? self::POS_BEFORE_SUB : self::POS_AFTER_SUB) : | |
| 1666 | + (self::IS_KNDA($u) ? ($u < 0x0CC3 || $u > 0xCD6 ? self::POS_BEFORE_SUB : self::POS_AFTER_SUB) : | |
| 1667 | + (self::IS_MLYM($u) ? self::POS_AFTER_POST : | |
| 1668 | + (self::IS_SINH($u) ? self::POS_AFTER_SUB : | |
| 1669 | + (self::IS_KHMR($u) ? self::POS_AFTER_POST : | |
| 1670 | + self::POS_AFTER_SUB))))))))))); /* default */ | |
| 1671 | + } | |
| 1672 | + | |
| 1673 | + public static function MATRA_POS_TOP($u) | |
| 1674 | + { | |
| 1675 | + return /* BENG and MLYM don't have top matras. */ | |
| 1676 | + (self::IS_DEVA($u) ? self::POS_AFTER_SUB : | |
| 1677 | + (self::IS_GURU($u) ? self::POS_AFTER_POST : /* Deviate from spec */ | |
| 1678 | + (self::IS_GUJR($u) ? self::POS_AFTER_SUB : | |
| 1679 | + (self::IS_ORYA($u) ? self::POS_AFTER_MAIN : | |
| 1680 | + (self::IS_TAML($u) ? self::POS_AFTER_SUB : | |
| 1681 | + (self::IS_TELU($u) ? self::POS_BEFORE_SUB : | |
| 1682 | + (self::IS_KNDA($u) ? self::POS_BEFORE_SUB : | |
| 1683 | + (self::IS_SINH($u) ? self::POS_AFTER_SUB : | |
| 1684 | + (self::IS_KHMR($u) ? self::POS_AFTER_POST : | |
| 1685 | + self::POS_AFTER_SUB))))))))); /* default */ | |
| 1686 | + } | |
| 1687 | + | |
| 1688 | + public static function MATRA_POS_BOTTOM($u) | |
| 1689 | + { | |
| 1690 | + return | |
| 1691 | + (self::IS_DEVA($u) ? self::POS_AFTER_SUB : | |
| 1692 | + (self::IS_BENG($u) ? self::POS_AFTER_SUB : | |
| 1693 | + (self::IS_GURU($u) ? self::POS_AFTER_POST : | |
| 1694 | + (self::IS_GUJR($u) ? self::POS_AFTER_POST : | |
| 1695 | + (self::IS_ORYA($u) ? self::POS_AFTER_SUB : | |
| 1696 | + (self::IS_TAML($u) ? self::POS_AFTER_POST : | |
| 1697 | + (self::IS_TELU($u) ? self::POS_BEFORE_SUB : | |
| 1698 | + (self::IS_KNDA($u) ? self::POS_BEFORE_SUB : | |
| 1699 | + (self::IS_MLYM($u) ? self::POS_AFTER_POST : | |
| 1700 | + (self::IS_SINH($u) ? self::POS_AFTER_SUB : | |
| 1701 | + (self::IS_KHMR($u) ? self::POS_AFTER_POST : | |
| 1702 | + self::POS_AFTER_SUB))))))))))); /* default */ | |
| 1703 | + } | |
| 1704 | + | |
| 1705 | + public static function matra_position($u, $side) | |
| 1706 | + { | |
| 1707 | + switch ($side) { | |
| 1708 | + case self::POS_PRE_C: return self::MATRA_POS_LEFT($u); | |
| 1709 | + case self::POS_POST_C: return self::MATRA_POS_RIGHT($u); | |
| 1710 | + case self::POS_ABOVE_C: return self::MATRA_POS_TOP($u); | |
| 1711 | + case self::POS_BELOW_C: return self::MATRA_POS_BOTTOM($u); | |
| 1712 | + } | |
| 1713 | + return $side; | |
| 1714 | + } | |
| 1715 | + | |
| 1716 | + // vowel matras that have to be split into two parts. | |
| 1717 | + // From Harfbuzz (old) | |
| 1718 | + // New HarfBuzz uses /src/hb-ucdn/ucdn.c and unicodedata_db.h for full method of decomposition for all characters | |
| 1719 | + // Should always fully decompose and then recompose back, but we will just do the split matras | |
| 1720 | + public static function decompose_indic($ab) | |
| 1721 | + { | |
| 1722 | + $sub = array(); | |
| 1723 | + switch ($ab) { | |
| 1724 | + /* | |
| 1725 | + * Decompose split matras. | |
| 1726 | + */ | |
| 1727 | + /* bengali */ | |
| 1728 | + case 0x9cb : $sub[0] = 0x9c7; | |
| 1729 | + $sub[1] = 0x9be; | |
| 1730 | + return $sub; | |
| 1731 | + case 0x9cc : $sub[0] = 0x9c7; | |
| 1732 | + $sub[1] = 0x9d7; | |
| 1733 | + return $sub; | |
| 1734 | + /* oriya */ | |
| 1735 | + case 0xb48 : $sub[0] = 0xb47; | |
| 1736 | + $sub[1] = 0xb56; | |
| 1737 | + return $sub; | |
| 1738 | + case 0xb4b : $sub[0] = 0xb47; | |
| 1739 | + $sub[1] = 0xb3e; | |
| 1740 | + return $sub; | |
| 1741 | + case 0xb4c : $sub[0] = 0xb47; | |
| 1742 | + $sub[1] = 0xb57; | |
| 1743 | + return $sub; | |
| 1744 | + /* tamil */ | |
| 1745 | + case 0xbca : $sub[0] = 0xbc6; | |
| 1746 | + $sub[1] = 0xbbe; | |
| 1747 | + return $sub; | |
| 1748 | + case 0xbcb : $sub[0] = 0xbc7; | |
| 1749 | + $sub[1] = 0xbbe; | |
| 1750 | + return $sub; | |
| 1751 | + case 0xbcc : $sub[0] = 0xbc6; | |
| 1752 | + $sub[1] = 0xbd7; | |
| 1753 | + return $sub; | |
| 1754 | + /* telugu */ | |
| 1755 | + case 0xc48 : $sub[0] = 0xc46; | |
| 1756 | + $sub[1] = 0xc56; | |
| 1757 | + return $sub; | |
| 1758 | + /* kannada */ | |
| 1759 | + case 0xcc0 : $sub[0] = 0xcbf; | |
| 1760 | + $sub[1] = 0xcd5; | |
| 1761 | + return $sub; | |
| 1762 | + case 0xcc7 : $sub[0] = 0xcc6; | |
| 1763 | + $sub[1] = 0xcd5; | |
| 1764 | + return $sub; | |
| 1765 | + case 0xcc8 : $sub[0] = 0xcc6; | |
| 1766 | + $sub[1] = 0xcd6; | |
| 1767 | + return $sub; | |
| 1768 | + case 0xcca : $sub[0] = 0xcc6; | |
| 1769 | + $sub[1] = 0xcc2; | |
| 1770 | + return $sub; | |
| 1771 | + case 0xccb : $sub[0] = 0xcc6; | |
| 1772 | + $sub[1] = 0xcc2; | |
| 1773 | + $sub[2] = 0xcd5; | |
| 1774 | + return $sub; | |
| 1775 | + /* malayalam */ | |
| 1776 | + case 0xd4a : $sub[0] = 0xd46; | |
| 1777 | + $sub[1] = 0xd3e; | |
| 1778 | + return $sub; | |
| 1779 | + case 0xd4b : $sub[0] = 0xd47; | |
| 1780 | + $sub[1] = 0xd3e; | |
| 1781 | + return $sub; | |
| 1782 | + case 0xd4c : $sub[0] = 0xd46; | |
| 1783 | + $sub[1] = 0xd57; | |
| 1784 | + return $sub; | |
| 1785 | + /* sinhala */ | |
| 1786 | + // NB Some fonts break with these Sinhala decomps (although this is Uniscribe spec) | |
| 1787 | + // Can check if character would be substituted by pstf and only decompose if true | |
| 1788 | + // e.g. if (isset($GSUBdata['pstf'][$ab])) - would need to pass $GSUBdata as parameter to this function | |
| 1789 | + case 0xdda : $sub[0] = 0xdd9; | |
| 1790 | + $sub[1] = 0xdca; | |
| 1791 | + return $sub; | |
| 1792 | + case 0xddc : $sub[0] = 0xdd9; | |
| 1793 | + $sub[1] = 0xdcf; | |
| 1794 | + return $sub; | |
| 1795 | + case 0xddd : $sub[0] = 0xdd9; | |
| 1796 | + $sub[1] = 0xdcf; | |
| 1797 | + $sub[2] = 0xdca; | |
| 1798 | + return $sub; | |
| 1799 | + case 0xdde : $sub[0] = 0xdd9; | |
| 1800 | + $sub[1] = 0xddf; | |
| 1801 | + return $sub; | |
| 1802 | + /* khmer */ | |
| 1803 | + case 0x17be : $sub[0] = 0x17c1; | |
| 1804 | + $sub[1] = 0x17be; | |
| 1805 | + return $sub; | |
| 1806 | + case 0x17bf : $sub[0] = 0x17c1; | |
| 1807 | + $sub[1] = 0x17bf; | |
| 1808 | + return $sub; | |
| 1809 | + case 0x17c0 : $sub[0] = 0x17c1; | |
| 1810 | + $sub[1] = 0x17c0; | |
| 1811 | + return $sub; | |
| 1812 | + | |
| 1813 | + case 0x17c4 : $sub[0] = 0x17c1; | |
| 1814 | + $sub[1] = 0x17c4; | |
| 1815 | + return $sub; | |
| 1816 | + case 0x17c5 : $sub[0] = 0x17c1; | |
| 1817 | + $sub[1] = 0x17c5; | |
| 1818 | + return $sub; | |
| 1819 | + /* tibetan - included here although does not use Inidc shaper in other ways */ | |
| 1820 | + case 0xf73 : $sub[0] = 0xf71; | |
| 1821 | + $sub[1] = 0xf72; | |
| 1822 | + return $sub; | |
| 1823 | + case 0xf75 : $sub[0] = 0xf71; | |
| 1824 | + $sub[1] = 0xf74; | |
| 1825 | + return $sub; | |
| 1826 | + case 0xf76 : $sub[0] = 0xfb2; | |
| 1827 | + $sub[1] = 0xf80; | |
| 1828 | + return $sub; | |
| 1829 | + case 0xf77 : $sub[0] = 0xfb2; | |
| 1830 | + $sub[1] = 0xf81; | |
| 1831 | + return $sub; | |
| 1832 | + case 0xf78 : $sub[0] = 0xfb3; | |
| 1833 | + $sub[1] = 0xf80; | |
| 1834 | + return $sub; | |
| 1835 | + case 0xf79 : $sub[0] = 0xfb3; | |
| 1836 | + $sub[1] = 0xf71; | |
| 1837 | + $sub[2] = 0xf80; | |
| 1838 | + return $sub; | |
| 1839 | + case 0xf81 : $sub[0] = 0xf71; | |
| 1840 | + $sub[1] = 0xf80; | |
| 1841 | + return $sub; | |
| 1842 | + } | |
| 1843 | + return false; | |
| 1844 | + } | |
| 1845 | + | |
| 1846 | + public static function bubble_sort(&$arr, $start, $len) | |
| 1847 | + { | |
| 1848 | + if ($len < 2) { | |
| 1849 | + return; | |
| 1850 | + } | |
| 1851 | + $k = $start + $len - 2; | |
| 1852 | + while ($k >= $start) { | |
| 1853 | + for ($j = $start; $j <= $k; $j++) { | |
| 1854 | + if ($arr[$j]['indic_position'] > $arr[$j + 1]['indic_position']) { | |
| 1855 | + $t = $arr[$j]; | |
| 1856 | + $arr[$j] = $arr[$j + 1]; | |
| 1857 | + $arr[$j + 1] = $t; | |
| 1858 | + } | |
| 1859 | + } | |
| 1860 | + $k--; | |
| 1861 | + } | |
| 1862 | + } | |
| 1863 | + | |
| 1864 | +} | |