| 1 |
<?php |
| 2 |
|
| 3 |
namespace TablePress\PhpOffice\PhpSpreadsheet\Shared; |
| 4 |
|
| 5 |
use TablePress\Composer\Pcre\Preg; |
| 6 |
use IntlCalendar; |
| 7 |
use NumberFormatter; |
| 8 |
use TablePress\PhpOffice\PhpSpreadsheet\Calculation\Calculation; |
| 9 |
use TablePress\PhpOffice\PhpSpreadsheet\Exception as SpreadsheetException; |
| 10 |
use Stringable; |
| 11 |
|
| 12 |
class StringHelper |
| 13 |
{ |
| 14 |
private const CONTROL_CHARACTERS_KEYS = [ |
| 15 |
"\x00", |
| 16 |
"\x01", |
| 17 |
"\x02", |
| 18 |
"\x03", |
| 19 |
"\x04", |
| 20 |
"\x05", |
| 21 |
"\x06", |
| 22 |
"\x07", |
| 23 |
"\x08", |
| 24 |
"\x0b", |
| 25 |
"\x0c", |
| 26 |
"\x0e", |
| 27 |
"\x0f", |
| 28 |
"\x10", |
| 29 |
"\x11", |
| 30 |
"\x12", |
| 31 |
"\x13", |
| 32 |
"\x14", |
| 33 |
"\x15", |
| 34 |
"\x16", |
| 35 |
"\x17", |
| 36 |
"\x18", |
| 37 |
"\x19", |
| 38 |
"\x1a", |
| 39 |
"\x1b", |
| 40 |
"\x1c", |
| 41 |
"\x1d", |
| 42 |
"\x1e", |
| 43 |
"\x1f", |
| 44 |
]; |
| 45 |
private const CONTROL_CHARACTERS_VALUES = [ |
| 46 |
'_x0000_', |
| 47 |
'_x0001_', |
| 48 |
'_x0002_', |
| 49 |
'_x0003_', |
| 50 |
'_x0004_', |
| 51 |
'_x0005_', |
| 52 |
'_x0006_', |
| 53 |
'_x0007_', |
| 54 |
'_x0008_', |
| 55 |
'_x000B_', |
| 56 |
'_x000C_', |
| 57 |
'_x000E_', |
| 58 |
'_x000F_', |
| 59 |
'_x0010_', |
| 60 |
'_x0011_', |
| 61 |
'_x0012_', |
| 62 |
'_x0013_', |
| 63 |
'_x0014_', |
| 64 |
'_x0015_', |
| 65 |
'_x0016_', |
| 66 |
'_x0017_', |
| 67 |
'_x0018_', |
| 68 |
'_x0019_', |
| 69 |
'_x001A_', |
| 70 |
'_x001B_', |
| 71 |
'_x001C_', |
| 72 |
'_x001D_', |
| 73 |
'_x001E_', |
| 74 |
'_x001F_', |
| 75 |
]; |
| 76 |
|
| 77 |
/** |
| 78 |
* SYLK Characters array. |
| 79 |
*/ |
| 80 |
private const SYLK_CHARACTERS = [ |
| 81 |
"\x1B 0" => "\x00", |
| 82 |
"\x1B 1" => "\x01", |
| 83 |
"\x1B 2" => "\x02", |
| 84 |
"\x1B 3" => "\x03", |
| 85 |
"\x1B 4" => "\x04", |
| 86 |
"\x1B 5" => "\x05", |
| 87 |
"\x1B 6" => "\x06", |
| 88 |
"\x1B 7" => "\x07", |
| 89 |
"\x1B 8" => "\x08", |
| 90 |
"\x1B 9" => "\x09", |
| 91 |
"\x1B :" => "\x0a", |
| 92 |
"\x1B ;" => "\x0b", |
| 93 |
"\x1B <" => "\x0c", |
| 94 |
"\x1B =" => "\x0d", |
| 95 |
"\x1B >" => "\x0e", |
| 96 |
"\x1B ?" => "\x0f", |
| 97 |
"\x1B!0" => "\x10", |
| 98 |
"\x1B!1" => "\x11", |
| 99 |
"\x1B!2" => "\x12", |
| 100 |
"\x1B!3" => "\x13", |
| 101 |
"\x1B!4" => "\x14", |
| 102 |
"\x1B!5" => "\x15", |
| 103 |
"\x1B!6" => "\x16", |
| 104 |
"\x1B!7" => "\x17", |
| 105 |
"\x1B!8" => "\x18", |
| 106 |
"\x1B!9" => "\x19", |
| 107 |
"\x1B!:" => "\x1a", |
| 108 |
"\x1B!;" => "\x1b", |
| 109 |
"\x1B!<" => "\x1c", |
| 110 |
"\x1B!=" => "\x1d", |
| 111 |
"\x1B!>" => "\x1e", |
| 112 |
"\x1B!?" => "\x1f", |
| 113 |
"\x1B'?" => "\x7f", |
| 114 |
"\x1B(0" => '€', // 128 in CP1252 |
| 115 |
"\x1B(2" => '‚', // 130 in CP1252 |
| 116 |
"\x1B(3" => 'ƒ', // 131 in CP1252 |
| 117 |
"\x1B(4" => '„', // 132 in CP1252 |
| 118 |
"\x1B(5" => '…', // 133 in CP1252 |
| 119 |
"\x1B(6" => '†', // 134 in CP1252 |
| 120 |
"\x1B(7" => '‡', // 135 in CP1252 |
| 121 |
"\x1B(8" => 'ˆ', // 136 in CP1252 |
| 122 |
"\x1B(9" => '‰', // 137 in CP1252 |
| 123 |
"\x1B(:" => 'Š', // 138 in CP1252 |
| 124 |
"\x1B(;" => '‹', // 139 in CP1252 |
| 125 |
"\x1BNj" => 'Œ', // 140 in CP1252 |
| 126 |
"\x1B(>" => 'Ž', // 142 in CP1252 |
| 127 |
"\x1B)1" => '‘', // 145 in CP1252 |
| 128 |
"\x1B)2" => '’', // 146 in CP1252 |
| 129 |
"\x1B)3" => '“', // 147 in CP1252 |
| 130 |
"\x1B)4" => '”', // 148 in CP1252 |
| 131 |
"\x1B)5" => '•', // 149 in CP1252 |
| 132 |
"\x1B)6" => '–', // 150 in CP1252 |
| 133 |
"\x1B)7" => '—', // 151 in CP1252 |
| 134 |
"\x1B)8" => '˜', // 152 in CP1252 |
| 135 |
"\x1B)9" => '™', // 153 in CP1252 |
| 136 |
"\x1B):" => 'š', // 154 in CP1252 |
| 137 |
"\x1B);" => '›', // 155 in CP1252 |
| 138 |
"\x1BNz" => 'œ', // 156 in CP1252 |
| 139 |
"\x1B)>" => 'ž', // 158 in CP1252 |
| 140 |
"\x1B)?" => 'Ÿ', // 159 in CP1252 |
| 141 |
"\x1B*0" => ' ', // 160 in CP1252 |
| 142 |
"\x1BN!" => '¡', // 161 in CP1252 |
| 143 |
"\x1BN\"" => '¢', // 162 in CP1252 |
| 144 |
"\x1BN#" => '£', // 163 in CP1252 |
| 145 |
"\x1BN(" => '¤', // 164 in CP1252 |
| 146 |
"\x1BN%" => '¥', // 165 in CP1252 |
| 147 |
"\x1B*6" => '¦', // 166 in CP1252 |
| 148 |
"\x1BN'" => '§', // 167 in CP1252 |
| 149 |
"\x1BNH " => '¨', // 168 in CP1252 |
| 150 |
"\x1BNS" => '©', // 169 in CP1252 |
| 151 |
"\x1BNc" => 'ª', // 170 in CP1252 |
| 152 |
"\x1BN+" => '«', // 171 in CP1252 |
| 153 |
"\x1B*<" => '¬', // 172 in CP1252 |
| 154 |
"\x1B*=" => '', // 173 in CP1252 |
| 155 |
"\x1BNR" => '®', // 174 in CP1252 |
| 156 |
"\x1B*?" => '¯', // 175 in CP1252 |
| 157 |
"\x1BN0" => '°', // 176 in CP1252 |
| 158 |
"\x1BN1" => '±', // 177 in CP1252 |
| 159 |
"\x1BN2" => '²', // 178 in CP1252 |
| 160 |
"\x1BN3" => '³', // 179 in CP1252 |
| 161 |
"\x1BNB " => '´', // 180 in CP1252 |
| 162 |
"\x1BN5" => 'µ', // 181 in CP1252 |
| 163 |
"\x1BN6" => '¶', // 182 in CP1252 |
| 164 |
"\x1BN7" => '·', // 183 in CP1252 |
| 165 |
"\x1B+8" => '¸', // 184 in CP1252 |
| 166 |
"\x1BNQ" => '¹', // 185 in CP1252 |
| 167 |
"\x1BNk" => 'º', // 186 in CP1252 |
| 168 |
"\x1BN;" => '»', // 187 in CP1252 |
| 169 |
"\x1BN<" => '¼', // 188 in CP1252 |
| 170 |
"\x1BN=" => '½', // 189 in CP1252 |
| 171 |
"\x1BN>" => '¾', // 190 in CP1252 |
| 172 |
"\x1BN?" => '¿', // 191 in CP1252 |
| 173 |
"\x1BNAA" => 'À', // 192 in CP1252 |
| 174 |
"\x1BNBA" => 'Á', // 193 in CP1252 |
| 175 |
"\x1BNCA" => 'Â', // 194 in CP1252 |
| 176 |
"\x1BNDA" => 'Ã', // 195 in CP1252 |
| 177 |
"\x1BNHA" => 'Ä', // 196 in CP1252 |
| 178 |
"\x1BNJA" => '� |
| 179 |
', // 197 in CP1252 |
| 180 |
"\x1BNa" => 'Æ', // 198 in CP1252 |
| 181 |
"\x1BNKC" => 'Ç', // 199 in CP1252 |
| 182 |
"\x1BNAE" => 'È', // 200 in CP1252 |
| 183 |
"\x1BNBE" => 'É', // 201 in CP1252 |
| 184 |
"\x1BNCE" => 'Ê', // 202 in CP1252 |
| 185 |
"\x1BNHE" => 'Ë', // 203 in CP1252 |
| 186 |
"\x1BNAI" => 'Ì', // 204 in CP1252 |
| 187 |
"\x1BNBI" => 'Í', // 205 in CP1252 |
| 188 |
"\x1BNCI" => 'Î', // 206 in CP1252 |
| 189 |
"\x1BNHI" => 'Ï', // 207 in CP1252 |
| 190 |
"\x1BNb" => 'Ð', // 208 in CP1252 |
| 191 |
"\x1BNDN" => 'Ñ', // 209 in CP1252 |
| 192 |
"\x1BNAO" => 'Ò', // 210 in CP1252 |
| 193 |
"\x1BNBO" => 'Ó', // 211 in CP1252 |
| 194 |
"\x1BNCO" => 'Ô', // 212 in CP1252 |
| 195 |
"\x1BNDO" => 'Õ', // 213 in CP1252 |
| 196 |
"\x1BNHO" => 'Ö', // 214 in CP1252 |
| 197 |
"\x1B-7" => '×', // 215 in CP1252 |
| 198 |
"\x1BNi" => 'Ø', // 216 in CP1252 |
| 199 |
"\x1BNAU" => 'Ù', // 217 in CP1252 |
| 200 |
"\x1BNBU" => 'Ú', // 218 in CP1252 |
| 201 |
"\x1BNCU" => 'Û', // 219 in CP1252 |
| 202 |
"\x1BNHU" => 'Ü', // 220 in CP1252 |
| 203 |
"\x1B-=" => 'Ý', // 221 in CP1252 |
| 204 |
"\x1BNl" => 'Þ', // 222 in CP1252 |
| 205 |
"\x1BN{" => 'ß', // 223 in CP1252 |
| 206 |
"\x1BNAa" => 'à', // 224 in CP1252 |
| 207 |
"\x1BNBa" => 'á', // 225 in CP1252 |
| 208 |
"\x1BNCa" => 'â', // 226 in CP1252 |
| 209 |
"\x1BNDa" => 'ã', // 227 in CP1252 |
| 210 |
"\x1BNHa" => 'ä', // 228 in CP1252 |
| 211 |
"\x1BNJa" => 'å', // 229 in CP1252 |
| 212 |
"\x1BNq" => 'æ', // 230 in CP1252 |
| 213 |
"\x1BNKc" => 'ç', // 231 in CP1252 |
| 214 |
"\x1BNAe" => 'è', // 232 in CP1252 |
| 215 |
"\x1BNBe" => 'é', // 233 in CP1252 |
| 216 |
"\x1BNCe" => 'ê', // 234 in CP1252 |
| 217 |
"\x1BNHe" => 'ë', // 235 in CP1252 |
| 218 |
"\x1BNAi" => 'ì', // 236 in CP1252 |
| 219 |
"\x1BNBi" => 'í', // 237 in CP1252 |
| 220 |
"\x1BNCi" => 'î', // 238 in CP1252 |
| 221 |
"\x1BNHi" => 'ï', // 239 in CP1252 |
| 222 |
"\x1BNs" => 'ð', // 240 in CP1252 |
| 223 |
"\x1BNDn" => 'ñ', // 241 in CP1252 |
| 224 |
"\x1BNAo" => 'ò', // 242 in CP1252 |
| 225 |
"\x1BNBo" => 'ó', // 243 in CP1252 |
| 226 |
"\x1BNCo" => 'ô', // 244 in CP1252 |
| 227 |
"\x1BNDo" => 'õ', // 245 in CP1252 |
| 228 |
"\x1BNHo" => 'ö', // 246 in CP1252 |
| 229 |
"\x1B/7" => '÷', // 247 in CP1252 |
| 230 |
"\x1BNy" => 'ø', // 248 in CP1252 |
| 231 |
"\x1BNAu" => 'ù', // 249 in CP1252 |
| 232 |
"\x1BNBu" => 'ú', // 250 in CP1252 |
| 233 |
"\x1BNCu" => 'û', // 251 in CP1252 |
| 234 |
"\x1BNHu" => 'ü', // 252 in CP1252 |
| 235 |
"\x1B/=" => 'ý', // 253 in CP1252 |
| 236 |
"\x1BN|" => 'þ', // 254 in CP1252 |
| 237 |
"\x1BNHy" => 'ÿ', // 255 in CP1252 |
| 238 |
]; |
| 239 |
|
| 240 |
/** |
| 241 |
* Decimal separator. |
| 242 |
*/ |
| 243 |
protected static ?string $decimalSeparator = null; |
| 244 |
|
| 245 |
/** |
| 246 |
* Thousands separator. |
| 247 |
*/ |
| 248 |
protected static ?string $thousandsSeparator = null; |
| 249 |
|
| 250 |
/** |
| 251 |
* Currency code. |
| 252 |
*/ |
| 253 |
protected static ?string $currencyCode = null; |
| 254 |
|
| 255 |
/** |
| 256 |
* Is iconv extension available? |
| 257 |
*/ |
| 258 |
protected static ?bool $isIconvEnabled = null; |
| 259 |
|
| 260 |
/** |
| 261 |
* iconv options. |
| 262 |
*/ |
| 263 |
protected static string $iconvOptions = '//IGNORE//TRANSLIT'; |
| 264 |
|
| 265 |
/** @var string[] */ |
| 266 |
protected static array $iconvOptionsArray = ['//IGNORE//TRANSLIT', '//IGNORE']; |
| 267 |
|
| 268 |
/** @internal */ |
| 269 |
protected static string $iconvName = 'iconv'; |
| 270 |
|
| 271 |
/** @internal */ |
| 272 |
protected static bool $iconvTest2 = false; |
| 273 |
|
| 274 |
/** @internal */ |
| 275 |
protected static bool $iconvTest3 = false; |
| 276 |
|
| 277 |
/** |
| 278 |
* Get whether iconv extension is available. |
| 279 |
*/ |
| 280 |
public static function getIsIconvEnabled(): bool |
| 281 |
{ |
| 282 |
if (isset(static::$isIconvEnabled)) { |
| 283 |
return static::$isIconvEnabled; |
| 284 |
} |
| 285 |
|
| 286 |
// Assume no problems with iconv |
| 287 |
static::$isIconvEnabled = true; |
| 288 |
|
| 289 |
// Fail if iconv doesn't exist |
| 290 |
if (!function_exists(static::$iconvName)) { |
| 291 |
static::$isIconvEnabled = false; |
| 292 |
} elseif (static::$iconvTest2 || !@iconv('UTF-8', 'UTF-16LE', 'x')) { |
| 293 |
// Sometimes iconv is not working, and e.g. iconv('UTF-8', 'UTF-16LE', 'x') just returns false, |
| 294 |
static::$isIconvEnabled = false; |
| 295 |
} elseif (static::$iconvTest3 || (defined('PHP_OS') && @stristr(PHP_OS, 'AIX') && defined('ICONV_IMPL') && (@strcasecmp(ICONV_IMPL, 'unknown') == 0) && defined('ICONV_VERSION') && (@strcasecmp(ICONV_VERSION, 'unknown') == 0))) { |
| 296 |
// CUSTOM: IBM AIX iconv() does not work |
| 297 |
static::$isIconvEnabled = false; |
| 298 |
} |
| 299 |
|
| 300 |
// Deactivate iconv default options if they fail (as seen on IBM i-series) |
| 301 |
if (static::$isIconvEnabled) { |
| 302 |
static::$iconvOptions = ''; |
| 303 |
foreach (static::$iconvOptionsArray as $option) { |
| 304 |
if (@iconv('UTF-8', 'UTF-16LE' . $option, 'x') !== false) { |
| 305 |
static::$iconvOptions = $option; |
| 306 |
|
| 307 |
break; |
| 308 |
} |
| 309 |
} |
| 310 |
} |
| 311 |
|
| 312 |
return static::$isIconvEnabled; |
| 313 |
} |
| 314 |
|
| 315 |
/** |
| 316 |
* Convert from OpenXML escaped control character to PHP control character. |
| 317 |
* |
| 318 |
* Excel 2007 team: |
| 319 |
* ---------------- |
| 320 |
* That's correct, control characters are stored directly in the shared-strings table. |
| 321 |
* We do encode characters that cannot be represented in XML using the following escape sequence: |
| 322 |
* _xHHHH_ where H represents a hexadecimal character in the character's value... |
| 323 |
* So you could end up with something like _x0008_ in a string (either in a cell value (<v>) |
| 324 |
* element or in the shared string <t> element. |
| 325 |
* |
| 326 |
* @param string $textValue Value to unescape |
| 327 |
*/ |
| 328 |
public static function controlCharacterOOXML2PHP(string $textValue): string |
| 329 |
{ |
| 330 |
return Preg::replaceCallback('/_x[0-9A-F]{4}_(_xD[CDEF][0-9A-F]{2}_)?/', \Closure::fromCallable([self::class, 'toOutChar']), $textValue); |
| 331 |
} |
| 332 |
|
| 333 |
private static function toHexVal(string $char): int |
| 334 |
{ |
| 335 |
if ($char >= '0' && $char <= '9') { |
| 336 |
return ord($char) - ord('0'); |
| 337 |
} |
| 338 |
|
| 339 |
return ord($char) - ord('A') + 10; |
| 340 |
} |
| 341 |
|
| 342 |
/** @param array<?string> $match */ |
| 343 |
private static function toOutChar(array $match): string |
| 344 |
{ |
| 345 |
/** @var string */ |
| 346 |
$chars = $match[0]; |
| 347 |
$h = ((self::toHexVal($chars[2]) << 12) |
| 348 |
| (self::toHexVal($chars[3]) << 8) |
| 349 |
| (self::toHexVal($chars[4]) << 4) |
| 350 |
| (self::toHexVal($chars[5]))); |
| 351 |
if (strlen($chars) === 7) { // no low surrogate |
| 352 |
if ($chars[2] === 'D' && in_array($chars[3], ['8', '9', 'A', 'B', 'C', 'D', 'E', 'F'], true)) { |
| 353 |
return '�'; |
| 354 |
} |
| 355 |
|
| 356 |
return mb_chr($h, 'UTF-8'); |
| 357 |
} |
| 358 |
if ($chars[2] === 'D' && in_array($chars[3], ['C', 'D', 'D', 'F'], true)) { |
| 359 |
return '�'; // Excel interprets as one substitute, not 2 |
| 360 |
} |
| 361 |
if ($chars[2] !== 'D' || !in_array($chars[3], ['8', '9', 'A', 'B'], true)) { |
| 362 |
return mb_chr($h, 'UTF-8') . '�'; |
| 363 |
} |
| 364 |
$l = ((self::toHexVal($chars[9]) << 12) |
| 365 |
| (self::toHexVal($chars[10]) << 8) |
| 366 |
| (self::toHexVal($chars[11]) << 4) |
| 367 |
| (self::toHexVal($chars[12]))); |
| 368 |
$result = 0x10000 + ($h - 0xD800) * 0x400 + ($l - 0xDC00); |
| 369 |
|
| 370 |
return mb_chr($result, 'UTF-8'); |
| 371 |
} |
| 372 |
|
| 373 |
/** |
| 374 |
* Convert from PHP control character to OpenXML escaped control character. |
| 375 |
* |
| 376 |
* Excel 2007 team: |
| 377 |
* ---------------- |
| 378 |
* That's correct, control characters are stored directly in the shared-strings table. |
| 379 |
* We do encode characters that cannot be represented in XML using the following escape sequence: |
| 380 |
* _xHHHH_ where H represents a hexadecimal character in the character's value... |
| 381 |
* So you could end up with something like _x0008_ in a string (either in a cell value (<v>) |
| 382 |
* element or in the shared string <t> element. |
| 383 |
* |
| 384 |
* @param string $textValue Value to escape |
| 385 |
*/ |
| 386 |
public static function controlCharacterPHP2OOXML(string $textValue): string |
| 387 |
{ |
| 388 |
$textValue = Preg::replace('/_(x[0-9A-F]{4}_)/', '_x005F_$1', $textValue); |
| 389 |
|
| 390 |
return str_replace(self::CONTROL_CHARACTERS_KEYS, self::CONTROL_CHARACTERS_VALUES, $textValue); |
| 391 |
} |
| 392 |
|
| 393 |
/** |
| 394 |
* Try to sanitize UTF8, replacing invalid sequences with Unicode substitution characters. |
| 395 |
*/ |
| 396 |
public static function sanitizeUTF8(string $textValue): string |
| 397 |
{ |
| 398 |
$textValue = str_replace(["\xef\xbf\xbe", "\xef\xbf\xbf"], "\xef\xbf\xbd", $textValue); |
| 399 |
$subst = mb_substitute_character(); // default is question mark |
| 400 |
mb_substitute_character(65533); // Unicode substitution character |
| 401 |
$returnValue = (string) mb_convert_encoding($textValue, 'UTF-8', 'UTF-8'); |
| 402 |
mb_substitute_character($subst); |
| 403 |
|
| 404 |
return $returnValue; |
| 405 |
} |
| 406 |
|
| 407 |
/** |
| 408 |
* Check if a string contains UTF8 data. |
| 409 |
*/ |
| 410 |
public static function isUTF8(string $textValue): bool |
| 411 |
{ |
| 412 |
return $textValue === self::sanitizeUTF8($textValue); |
| 413 |
} |
| 414 |
|
| 415 |
/** |
| 416 |
* Formats a numeric value as a string for output in various output writers forcing |
| 417 |
* point as decimal separator in case locale is other than English. |
| 418 |
* @param float|int|string|null $numericValue |
| 419 |
*/ |
| 420 |
public static function formatNumber($numericValue): string |
| 421 |
{ |
| 422 |
if (is_float($numericValue)) { |
| 423 |
return str_replace(',', '.', (string) $numericValue); |
| 424 |
} |
| 425 |
|
| 426 |
return (string) $numericValue; |
| 427 |
} |
| 428 |
|
| 429 |
/** |
| 430 |
* Converts a UTF-8 string into BIFF8 Unicode string data (8-bit string length) |
| 431 |
* Writes the string using uncompressed notation, no rich text, no Asian phonetics |
| 432 |
* If mbstring extension is not available, ASCII is assumed, and compressed notation is used |
| 433 |
* although this will give wrong results for non-ASCII strings |
| 434 |
* see OpenOffice.org's Documentation of the Microsoft Excel File Format, sect. 2.5.3. |
| 435 |
* |
| 436 |
* @param string $textValue UTF-8 encoded string |
| 437 |
* @param array<int, array{strlen: int, fontidx: int}> $arrcRuns Details of rich text runs in $value |
| 438 |
*/ |
| 439 |
public static function UTF8toBIFF8UnicodeShort(string $textValue, array $arrcRuns = []): string |
| 440 |
{ |
| 441 |
// character count |
| 442 |
$ln = self::countCharacters($textValue, 'UTF-8'); |
| 443 |
// option flags |
| 444 |
if (empty($arrcRuns)) { |
| 445 |
$data = pack('CC', $ln, 0x0001); |
| 446 |
// characters |
| 447 |
$data .= self::convertEncoding($textValue, 'UTF-16LE', 'UTF-8'); |
| 448 |
} else { |
| 449 |
$data = pack('vC', $ln, 0x09); |
| 450 |
$data .= pack('v', count($arrcRuns)); |
| 451 |
// characters |
| 452 |
$data .= self::convertEncoding($textValue, 'UTF-16LE', 'UTF-8'); |
| 453 |
foreach ($arrcRuns as $cRun) { |
| 454 |
$data .= pack('v', $cRun['strlen']); |
| 455 |
$data .= pack('v', $cRun['fontidx']); |
| 456 |
} |
| 457 |
} |
| 458 |
|
| 459 |
return $data; |
| 460 |
} |
| 461 |
|
| 462 |
/** |
| 463 |
* Converts a UTF-8 string into BIFF8 Unicode string data (16-bit string length) |
| 464 |
* Writes the string using uncompressed notation, no rich text, no Asian phonetics |
| 465 |
* If mbstring extension is not available, ASCII is assumed, and compressed notation is used |
| 466 |
* although this will give wrong results for non-ASCII strings |
| 467 |
* see OpenOffice.org's Documentation of the Microsoft Excel File Format, sect. 2.5.3. |
| 468 |
* |
| 469 |
* @param string $textValue UTF-8 encoded string |
| 470 |
*/ |
| 471 |
public static function UTF8toBIFF8UnicodeLong(string $textValue): string |
| 472 |
{ |
| 473 |
// characters |
| 474 |
$chars = self::convertEncoding($textValue, 'UTF-16LE', 'UTF-8'); |
| 475 |
$ln = (int) (strlen($chars) / 2); // N.B. - strlen, not mb_strlen issue #642 |
| 476 |
|
| 477 |
return pack('vC', $ln, 0x0001) . $chars; |
| 478 |
} |
| 479 |
|
| 480 |
/** |
| 481 |
* Convert string from one encoding to another. |
| 482 |
* |
| 483 |
* @param string $to Encoding to convert to, e.g. 'UTF-8' |
| 484 |
* @param string $from Encoding to convert from, e.g. 'UTF-16LE' |
| 485 |
*/ |
| 486 |
public static function convertEncoding(string $textValue, string $to, string $from, ?string $options = null): string |
| 487 |
{ |
| 488 |
if (static::getIsIconvEnabled()) { |
| 489 |
$result = iconv($from, $to . ($options ?? static::$iconvOptions), $textValue); |
| 490 |
if (false !== $result) { |
| 491 |
return $result; |
| 492 |
} |
| 493 |
} |
| 494 |
|
| 495 |
return (string) mb_convert_encoding($textValue, $to, $from); |
| 496 |
} |
| 497 |
|
| 498 |
/** |
| 499 |
* Get character count. |
| 500 |
* |
| 501 |
* @param string $encoding Encoding |
| 502 |
* |
| 503 |
* @return int Character count |
| 504 |
*/ |
| 505 |
public static function countCharacters(string $textValue, string $encoding = 'UTF-8'): int |
| 506 |
{ |
| 507 |
return mb_strlen($textValue, $encoding); |
| 508 |
} |
| 509 |
|
| 510 |
/** |
| 511 |
* Get character count using mb_strwidth rather than mb_strlen. |
| 512 |
* |
| 513 |
* @param string $encoding Encoding |
| 514 |
* |
| 515 |
* @return int Character count |
| 516 |
*/ |
| 517 |
public static function countCharactersDbcs(string $textValue, string $encoding = 'UTF-8'): int |
| 518 |
{ |
| 519 |
return mb_strwidth($textValue, $encoding); |
| 520 |
} |
| 521 |
|
| 522 |
/** |
| 523 |
* Get a substring of a UTF-8 encoded string. |
| 524 |
* |
| 525 |
* @param string $textValue UTF-8 encoded string |
| 526 |
* @param int $offset Start offset |
| 527 |
* @param ?int $length Maximum number of characters in substring |
| 528 |
*/ |
| 529 |
public static function substring(string $textValue, int $offset, ?int $length = 0): string |
| 530 |
{ |
| 531 |
return mb_substr($textValue, $offset, $length, 'UTF-8'); |
| 532 |
} |
| 533 |
|
| 534 |
/** |
| 535 |
* Convert a UTF-8 encoded string to upper case. |
| 536 |
* |
| 537 |
* @param string $textValue UTF-8 encoded string |
| 538 |
*/ |
| 539 |
public static function strToUpper(string $textValue): string |
| 540 |
{ |
| 541 |
return mb_convert_case($textValue, MB_CASE_UPPER, 'UTF-8'); |
| 542 |
} |
| 543 |
|
| 544 |
/** |
| 545 |
* Convert a UTF-8 encoded string to lower case. |
| 546 |
* |
| 547 |
* @param string $textValue UTF-8 encoded string |
| 548 |
*/ |
| 549 |
public static function strToLower(string $textValue): string |
| 550 |
{ |
| 551 |
return mb_convert_case($textValue, MB_CASE_LOWER, 'UTF-8'); |
| 552 |
} |
| 553 |
|
| 554 |
/** |
| 555 |
* Convert a UTF-8 encoded string to title/proper case |
| 556 |
* (uppercase every first character in each word, lower case all other characters). |
| 557 |
* |
| 558 |
* @param string $textValue UTF-8 encoded string |
| 559 |
*/ |
| 560 |
public static function strToTitle(string $textValue): string |
| 561 |
{ |
| 562 |
return mb_convert_case($textValue, MB_CASE_TITLE, 'UTF-8'); |
| 563 |
} |
| 564 |
|
| 565 |
public static function mbIsUpper(string $character): bool |
| 566 |
{ |
| 567 |
return mb_strtolower($character, 'UTF-8') !== $character; |
| 568 |
} |
| 569 |
|
| 570 |
/** |
| 571 |
* Splits a UTF-8 string into an array of individual characters. |
| 572 |
* |
| 573 |
* @return string[] |
| 574 |
*/ |
| 575 |
public static function mbStrSplit(string $string): array |
| 576 |
{ |
| 577 |
// Split at all position not after the start: ^ |
| 578 |
// and not before the end: $ |
| 579 |
$split = Preg::split('/(?<!^)(?!$)/u', $string); |
| 580 |
|
| 581 |
return $split; |
| 582 |
} |
| 583 |
|
| 584 |
/** |
| 585 |
* Reverse the case of a string, so that all uppercase characters become lowercase |
| 586 |
* and all lowercase characters become uppercase. |
| 587 |
* |
| 588 |
* @param string $textValue UTF-8 encoded string |
| 589 |
*/ |
| 590 |
public static function strCaseReverse(string $textValue): string |
| 591 |
{ |
| 592 |
$characters = self::mbStrSplit($textValue); |
| 593 |
foreach ($characters as &$character) { |
| 594 |
if (self::mbIsUpper($character)) { |
| 595 |
$character = mb_strtolower($character, 'UTF-8'); |
| 596 |
} else { |
| 597 |
$character = mb_strtoupper($character, 'UTF-8'); |
| 598 |
} |
| 599 |
} |
| 600 |
|
| 601 |
return implode('', $characters); |
| 602 |
} |
| 603 |
|
| 604 |
private static function useAlt(string $altValue, string $default, bool $trimAlt): string |
| 605 |
{ |
| 606 |
return ($trimAlt ? trim($altValue) : $altValue) ?: $default; |
| 607 |
} |
| 608 |
|
| 609 |
private static function getLocaleValue(string $key, string $altKey, string $default, bool $trimAlt = false): string |
| 610 |
{ |
| 611 |
/** @var string[] */ |
| 612 |
$localeconv = localeconv(); |
| 613 |
$rslt = $localeconv[$key]; |
| 614 |
// win-1252 implements Euro as 0x80 plus other symbols |
| 615 |
// Not suitable for Composer\Pcre\Preg |
| 616 |
if (preg_match('//u', $rslt) !== 1) { |
| 617 |
$rslt = ''; |
| 618 |
} |
| 619 |
|
| 620 |
return $rslt ?: self::useAlt($localeconv[$altKey], $default, $trimAlt); |
| 621 |
} |
| 622 |
|
| 623 |
/** |
| 624 |
* Get the decimal separator. If it has not yet been set explicitly, try to obtain number |
| 625 |
* formatting information from locale. |
| 626 |
*/ |
| 627 |
public static function getDecimalSeparator(): string |
| 628 |
{ |
| 629 |
if (!isset(static::$decimalSeparator)) { |
| 630 |
static::$decimalSeparator = self::getLocaleValue('decimal_point', 'mon_decimal_point', '.'); |
| 631 |
} |
| 632 |
|
| 633 |
return static::$decimalSeparator; |
| 634 |
} |
| 635 |
|
| 636 |
/** |
| 637 |
* Set the decimal separator. Only used by NumberFormat::toFormattedString() |
| 638 |
* to format output by \PhpOffice\PhpSpreadsheet\Writer\Html and \PhpOffice\PhpSpreadsheet\Writer\Pdf. |
| 639 |
* |
| 640 |
* @param ?string $separator Character for decimal separator |
| 641 |
*/ |
| 642 |
public static function setDecimalSeparator(?string $separator): void |
| 643 |
{ |
| 644 |
static::$decimalSeparator = $separator; |
| 645 |
} |
| 646 |
|
| 647 |
/** |
| 648 |
* Get the thousands separator. If it has not yet been set explicitly, try to obtain number |
| 649 |
* formatting information from locale. |
| 650 |
*/ |
| 651 |
public static function getThousandsSeparator(): string |
| 652 |
{ |
| 653 |
if (!isset(static::$thousandsSeparator)) { |
| 654 |
static::$thousandsSeparator = self::getLocaleValue('thousands_sep', 'mon_thousands_sep', ','); |
| 655 |
} |
| 656 |
|
| 657 |
return static::$thousandsSeparator; |
| 658 |
} |
| 659 |
|
| 660 |
/** |
| 661 |
* Set the thousands separator. Only used by NumberFormat::toFormattedString() |
| 662 |
* to format output by \PhpOffice\PhpSpreadsheet\Writer\Html and \PhpOffice\PhpSpreadsheet\Writer\Pdf. |
| 663 |
* |
| 664 |
* @param ?string $separator Character for thousands separator |
| 665 |
*/ |
| 666 |
public static function setThousandsSeparator(?string $separator): void |
| 667 |
{ |
| 668 |
static::$thousandsSeparator = $separator; |
| 669 |
} |
| 670 |
|
| 671 |
/** |
| 672 |
* Get the currency code. If it has not yet been set explicitly, try to obtain the |
| 673 |
* symbol information from locale. |
| 674 |
*/ |
| 675 |
public static function getCurrencyCode(bool $trimAlt = false): string |
| 676 |
{ |
| 677 |
if (!isset(static::$currencyCode)) { |
| 678 |
static::$currencyCode = self::getLocaleValue('currency_symbol', 'int_curr_symbol', '$', $trimAlt); |
| 679 |
} |
| 680 |
|
| 681 |
return static::$currencyCode; |
| 682 |
} |
| 683 |
|
| 684 |
/** |
| 685 |
* Set the currency code. Only used by NumberFormat::toFormattedString() |
| 686 |
* to format output by \PhpOffice\PhpSpreadsheet\Writer\Html and \PhpOffice\PhpSpreadsheet\Writer\Pdf. |
| 687 |
* |
| 688 |
* @param ?string $currencyCode Character for currency code |
| 689 |
*/ |
| 690 |
public static function setCurrencyCode(?string $currencyCode): void |
| 691 |
{ |
| 692 |
static::$currencyCode = $currencyCode; |
| 693 |
} |
| 694 |
|
| 695 |
/** |
| 696 |
* Convert SYLK encoded string to UTF-8. |
| 697 |
* |
| 698 |
* @param string $textValue SYLK encoded string |
| 699 |
* |
| 700 |
* @return string UTF-8 encoded string |
| 701 |
*/ |
| 702 |
public static function SYLKtoUTF8(string $textValue): string |
| 703 |
{ |
| 704 |
// If there is no escape character in the string there is nothing to do |
| 705 |
if (!str_contains($textValue, "\x1b")) { |
| 706 |
return $textValue; |
| 707 |
} |
| 708 |
|
| 709 |
foreach (self::SYLK_CHARACTERS as $k => $v) { |
| 710 |
$textValue = str_replace($k, $v, $textValue); |
| 711 |
} |
| 712 |
|
| 713 |
return $textValue; |
| 714 |
} |
| 715 |
|
| 716 |
/** |
| 717 |
* Retrieve any leading numeric part of a string, or return the full string if no leading numeric |
| 718 |
* (handles basic integer or float, but not exponent or non decimal). |
| 719 |
* |
| 720 |
* @return float|string string or only the leading numeric part of the string |
| 721 |
*/ |
| 722 |
public static function testStringAsNumeric(string $textValue) |
| 723 |
{ |
| 724 |
if (is_numeric($textValue)) { |
| 725 |
return $textValue; |
| 726 |
} |
| 727 |
$v = (float) $textValue; |
| 728 |
|
| 729 |
return (is_numeric(substr($textValue, 0, strlen((string) $v)))) ? $v : $textValue; |
| 730 |
} |
| 731 |
|
| 732 |
public static function strlenAllowNull(?string $string): int |
| 733 |
{ |
| 734 |
return strlen("$string"); |
| 735 |
} |
| 736 |
|
| 737 |
/** |
| 738 |
* @param bool $convertBool If true, convert bool to locale-aware TRUE/FALSE rather than 1/null-string |
| 739 |
* @param bool $lessFloatPrecision If true, floats will be converted to a more human-friendly but less computationally accurate value |
| 740 |
* @param mixed $value |
| 741 |
*/ |
| 742 |
public static function convertToString($value, bool $throw = true, string $default = '', bool $convertBool = false, bool $lessFloatPrecision = false): string |
| 743 |
{ |
| 744 |
if ($convertBool && is_bool($value)) { |
| 745 |
return $value ? Calculation::getTRUE() : Calculation::getFALSE(); |
| 746 |
} |
| 747 |
if (is_float($value) && !$lessFloatPrecision) { |
| 748 |
$string = (string) $value; |
| 749 |
// look out for scientific notation |
| 750 |
if (!Preg::isMatch('/[^-+0-9.]/', $string)) { |
| 751 |
$minus = $value < 0 ? '-' : ''; |
| 752 |
$positive = abs($value); |
| 753 |
$floor = floor($positive); |
| 754 |
$oldFrac = (string) ($positive - $floor); |
| 755 |
$frac = Preg::replace('/^0[.](\d+)$/', '$1', $oldFrac); |
| 756 |
if ($frac !== $oldFrac) { |
| 757 |
return "$minus$floor.$frac"; |
| 758 |
} |
| 759 |
} |
| 760 |
|
| 761 |
return $string; |
| 762 |
} |
| 763 |
if ($value === null || is_scalar($value) || is_object($value) && method_exists($value, '__toString')) { |
| 764 |
return (string) $value; |
| 765 |
} |
| 766 |
|
| 767 |
if ($throw) { |
| 768 |
throw new SpreadsheetException('Unable to convert to string'); |
| 769 |
} |
| 770 |
|
| 771 |
return $default; |
| 772 |
} |
| 773 |
|
| 774 |
/** |
| 775 |
* Assist with POST items when samples are run in browser. |
| 776 |
* Never run as part of unit tests, which are command line. |
| 777 |
* |
| 778 |
* @codeCoverageIgnore |
| 779 |
*/ |
| 780 |
public static function convertPostToString(string $index, string $default = ''): string |
| 781 |
{ |
| 782 |
if (isset($_POST[$index])) { |
| 783 |
return htmlentities(self::convertToString($_POST[$index], false, $default)); |
| 784 |
} |
| 785 |
|
| 786 |
return $default; |
| 787 |
} |
| 788 |
|
| 789 |
/** |
| 790 |
* Php introduced str_increment with Php8.3, |
| 791 |
* but didn't issue deprecation notices till 8.5. |
| 792 |
* |
| 793 |
* @param-out string $str |
| 794 |
* |
| 795 |
* @codeCoverageIgnore |
| 796 |
*/ |
| 797 |
public static function stringIncrement(string &$str): string |
| 798 |
{ |
| 799 |
if (function_exists('str_increment')) { |
| 800 |
/** @var non-empty-string $str */ |
| 801 |
$str2 = str_increment($str); |
| 802 |
} else { |
| 803 |
$str1 = $str; |
| 804 |
/** |
| 805 |
* This is an outright lie, but I don't know how else to satisfy Phpstan for Php8.5+. |
| 806 |
* |
| 807 |
* @var numeric-string $str1 |
| 808 |
*/ |
| 809 |
++$str1; |
| 810 |
$str2 = "$str1"; |
| 811 |
} |
| 812 |
|
| 813 |
/** |
| 814 |
* This is demonstrably the case. I don't know why Phpstan thinks $str2 is mixed at this point. |
| 815 |
* |
| 816 |
* @var string $str2 set above by str_increment or ="..." |
| 817 |
*/ |
| 818 |
$str = $str2; |
| 819 |
|
| 820 |
return $str; |
| 821 |
} |
| 822 |
|
| 823 |
/** @internal */ |
| 824 |
protected static string $testClass = IntlCalendar::class; |
| 825 |
|
| 826 |
/** |
| 827 |
* Set all of currencyCode, thousandsSeparator, decimalSeparator, |
| 828 |
* and Calculation locale with a single call. |
| 829 |
* The main point here is avoid the use of Php setlocale, |
| 830 |
* which is not threadsafe. It uses the Intl extension instead, |
| 831 |
* which is not a requirement for PhpSpreadsheet. |
| 832 |
* Because of that, the function returns a bool which will |
| 833 |
* be false if Intl is not available, or the supplied locale |
| 834 |
* is not valid according to Intl. |
| 835 |
*/ |
| 836 |
public static function setLocale(?string $locale): bool |
| 837 |
{ |
| 838 |
if ($locale === null) { |
| 839 |
self::$currencyCode = null; |
| 840 |
self::$thousandsSeparator = null; |
| 841 |
self::$decimalSeparator = null; |
| 842 |
Calculation::getInstance()->setLocale('en_us'); |
| 843 |
|
| 844 |
return true; |
| 845 |
} |
| 846 |
$localeCalc = $locale; |
| 847 |
if (Preg::isMatch('/^([a-z][a-z])_([a-z][a-z])(?:[.]utf-8)?$/i', $locale, $matches)) { |
| 848 |
$locale = strtolower($matches[1]) . '_' . strtoupper($matches[2]); |
| 849 |
$localeCalc = strtolower($matches[1]) . '_' . strtolower($matches[2]); |
| 850 |
} |
| 851 |
if (!class_exists(static::$testClass)) { |
| 852 |
return false; |
| 853 |
} |
| 854 |
// NumberFormatter constructor succeeds even with |
| 855 |
// bad locale before Php8.4, so try to validate |
| 856 |
// the locale beforehand. |
| 857 |
$locales = IntlCalendar::getAvailableLocales(); |
| 858 |
if (!in_array($locale, $locales, true)) { |
| 859 |
return false; |
| 860 |
} |
| 861 |
$formatter = new NumberFormatter( |
| 862 |
$locale, |
| 863 |
NumberFormatter::CURRENCY |
| 864 |
); |
| 865 |
$currency = $formatter->getSymbol( |
| 866 |
NumberFormatter::CURRENCY_SYMBOL |
| 867 |
); |
| 868 |
$formatter = new NumberFormatter( |
| 869 |
$locale, |
| 870 |
NumberFormatter::DECIMAL |
| 871 |
); |
| 872 |
$thousands = $formatter->getSymbol( |
| 873 |
NumberFormatter::GROUPING_SEPARATOR_SYMBOL |
| 874 |
); |
| 875 |
$decimal = $formatter->getSymbol( |
| 876 |
NumberFormatter::DECIMAL_SEPARATOR_SYMBOL |
| 877 |
); |
| 878 |
self::$currencyCode = $currency; |
| 879 |
self::$thousandsSeparator = $thousands; |
| 880 |
self::$decimalSeparator = $decimal; |
| 881 |
Calculation::getInstance()->setLocale($localeCalc); |
| 882 |
|
| 883 |
return true; |
| 884 |
} |
| 885 |
} |
| 886 |
|