Кодировки: основы
Компьютеры хранят текст как числа. Кодировка определяет, какое число соответствует какому символу.
ASCII (1963):
128 символов, 7 бит
Только латиница, цифры, знаки препинания
'A' = 65, 'a' = 97, '0' = 48
ISO-8859-1 (Latin-1):
256 символов, 8 бит (1 байт на символ)
ASCII + западноевропейские символы (ñ, ü, ø)
Нет кириллицы!
Windows-1251 (CP1251):
256 символов, 8 бит
ASCII + кириллица
Несовместима с ISO-8859-1!
UTF-8 (1993 — стандарт):
1,112,064 символа Unicode
Переменная длина: 1-4 байта на символ
ASCII-совместима (первые 128 символов = 1 байт)
'A' = 1 байт (0x41)
'Я' = 2 байта (0xD0 0xAF)
'€' = 3 байта (0xE2 0x82 0xAC)
'😀' = 4 байта (0xF0 0x9F 0x98 0x80)
'👨👩👧👦' = 25 байт (7 code points, ZWJ sequence!)
Правило: Всегда используйте UTF-8. Не ISO-8859-1, не Windows-1251, не ASCII. UTF-8 — единственный разумный выбор в 2024+.
BOM (Byte Order Mark)
<?php
declare(strict_types=1);
// BOM = invisible character at the start of a file
// UTF-8 BOM: 0xEF 0xBB 0xBF (3 bytes)
// BOM can cause problems:
// - Headers already sent (BOM output before header())
// - JSON parsing errors
// - XML parsing errors
// Detect BOM
function hasBom(string $content): bool
{
return str_starts_with($content, "\xEF\xBB\xBF");
}
// Remove BOM
function removeBom(string $content): string
{
if (str_starts_with($content, "\xEF\xBB\xBF")) {
return substr($content, 3);
}
return $content;
}
// Best practice: Save PHP files as UTF-8 WITHOUT BOM
mb_* функции
Стандартные строковые функции PHP (strlen, substr, strpos) работают с байтами, а не символами. Для UTF-8 строк используйте mb_* функции.
<?php
declare(strict_types=1);
// Standard functions count BYTES, not characters
$text = 'Привет';
echo strlen($text) . "\n"; // 12 (6 кириллических символов × 2 байта)
echo mb_strlen($text) . "\n"; // 6 (6 символов)
$emoji = '👋🌍';
echo strlen($emoji) . "\n"; // 8 (2 emoji × 4 байта)
echo mb_strlen($emoji) . "\n"; // 2 (2 символа)
Основные mb_* функции
<?php
declare(strict_types=1);
$text = 'Привет, мир!';
// mb_strlen() — длина в символах
echo mb_strlen($text) . "\n"; // 12
// mb_substr() — подстрока
echo mb_substr($text, 0, 6) . "\n"; // 'Привет'
echo mb_substr($text, 8) . "\n"; // 'мир!'
echo mb_substr($text, -4) . "\n"; // 'мир!'
// mb_strpos() / mb_strrpos() — позиция подстроки
echo mb_strpos($text, 'мир') . "\n"; // 8
echo mb_strrpos($text, ',') . "\n"; // 6
// mb_stripos() — позиция без учёта регистра
echo mb_stripos('Привет', 'привет') . "\n"; // 0
// mb_strtolower() / mb_strtoupper()
echo mb_strtolower('ПРИВЕТ') . "\n"; // 'привет'
echo mb_strtoupper('привет') . "\n"; // 'ПРИВЕТ'
// Important: strtolower('ПРИВЕТ') returns 'ПРИВЕТ' — doesn't work for non-ASCII!
// mb_str_split() — split into array of characters (PHP 7.4+)
$chars = mb_str_split('Привет');
print_r($chars); // ['П', 'р', 'и', 'в', 'е', 'т']
$chunks = mb_str_split('Привет', 2);
print_r($chunks); // ['Пр', 'ив', 'ет']
// mb_detect_encoding() — detect encoding
$unknown = file_get_contents('file.txt');
$encoding = mb_detect_encoding($unknown, ['UTF-8', 'Windows-1251', 'ISO-8859-1']);
echo "Detected: {$encoding}\n";
// mb_convert_encoding() — convert between encodings
$win1251 = mb_convert_encoding('Привет', 'Windows-1251', 'UTF-8');
$utf8 = mb_convert_encoding($win1251, 'UTF-8', 'Windows-1251');
echo $utf8 . "\n"; // 'Привет'
// mb_internal_encoding() — set default encoding for mb_* functions
mb_internal_encoding('UTF-8'); // Set once at startup
// mb_http_output() — encoding for HTTP output
mb_http_output('UTF-8');
// mb_str_pad() — pad string (PHP 8.3+)
echo mb_str_pad('Привет', 20, '.') . "\n"; // 'Привет..............'
echo mb_str_pad('Привет', 20, '.', STR_PAD_LEFT) . "\n"; // '..............Привет'
// mb_ucfirst() / mb_lcfirst() — PHP 8.4+
echo mb_ucfirst('привет') . "\n"; // 'Привет'
echo mb_lcfirst('Привет') . "\n"; // 'привет'
// mb_trim() / mb_ltrim() / mb_rtrim() — PHP 8.4+
echo mb_trim(' Привет ') . "\n"; // 'Привет'
mb_convert_case — преобразование регистра
<?php
declare(strict_types=1);
// mb_convert_case supports Unicode case mapping
$text = 'привет мир';
// Title Case
echo mb_convert_case($text, MB_CASE_TITLE) . "\n";
// 'Привет Мир'
// Upper Case
echo mb_convert_case($text, MB_CASE_UPPER) . "\n";
// 'ПРИВЕТ МИР'
// Lower Case
echo mb_convert_case('ПРИВЕТ МИР', MB_CASE_LOWER) . "\n";
// 'привет мир'
// Fold Case (for case-insensitive comparison, PHP 7.3+)
echo mb_convert_case('Straße', MB_CASE_FOLD) . "\n";
// 'strasse' (ß → ss for folding)
// Special cases:
echo mb_strtolower('İSTANBUL') . "\n"; // Depends on locale
echo mb_strtoupper('istanbul') . "\n";
Безопасное обрезание строк
<?php
declare(strict_types=1);
/**
* Safely truncate a multibyte string with ellipsis.
*/
function truncate(
string $text,
int $maxLength,
string $ellipsis = '...',
): string {
if (mb_strlen($text) <= $maxLength) {
return $text;
}
$ellipsisLength = mb_strlen($ellipsis);
$truncatedLength = $maxLength - $ellipsisLength;
if ($truncatedLength <= 0) {
return mb_substr($ellipsis, 0, $maxLength);
}
return mb_substr($text, 0, $truncatedLength) . $ellipsis;
}
echo truncate('Привет, как дела?', 10) . "\n"; // 'Привет,...'
echo truncate('Hello', 10) . "\n"; // 'Hello'
echo truncate('Длинный текст для тестирования', 15) . "\n"; // 'Длинный тек...'
grapheme_* функции
Grapheme (графема) — то, что пользователь воспринимает как один символ. Один grapheme может состоять из нескольких Unicode code points.
<?php
declare(strict_types=1);
// Grapheme cluster examples:
// 'é' = 'e' + combining accent (2 code points, 1 grapheme)
// '👨👩👧👦' = 7 code points (4 emoji + 3 ZWJ), 1 grapheme
// '🇷🇺' = 2 code points (regional indicators), 1 grapheme
$flag = '🇷🇺';
echo strlen($flag) . "\n"; // 8 bytes
echo mb_strlen($flag) . "\n"; // 2 code points
echo grapheme_strlen($flag) . "\n"; // 1 grapheme (what user sees)
$family = '👨👩👧👦';
echo strlen($family) . "\n"; // 25 bytes
echo mb_strlen($family) . "\n"; // 7 code points
echo grapheme_strlen($family) . "\n"; // 1 grapheme
// grapheme_substr() — substring by graphemes
$text = 'Привет 🇷🇺!';
echo grapheme_substr($text, 0, 7) . "\n"; // 'Привет '
echo grapheme_substr($text, 7, 1) . "\n"; // '🇷🇺' (1 grapheme = full flag)
// grapheme_strpos()
echo grapheme_strpos($text, '🇷🇺') . "\n"; // 7
// grapheme_strrpos()
echo grapheme_strrpos($text, '!') . "\n"; // 8
// grapheme_extract() — extract by grapheme count
$extracted = grapheme_extract('Привет!', 3, GRAPHEME_EXTR_COUNT);
echo $extracted . "\n"; // 'При'
grapheme_levenshtein() — PHP 8.5+
<?php
declare(strict_types=1);
// grapheme_levenshtein() — Levenshtein distance for grapheme clusters
// Available in PHP 8.5+
// Standard levenshtein() works with bytes, not graphemes
// This causes incorrect results for multibyte strings
// PHP 8.5+:
$distance = grapheme_levenshtein('Привет', 'Привед');
echo "Distance: {$distance}\n"; // 1 (one grapheme changed: т → д)
// Without grapheme support, levenshtein('Привет', 'Привед')
// would return 2 (because each Cyrillic char = 2 bytes)
Intl Extension — NumberFormatter
NumberFormatter форматирует числа, валюты и проценты с учётом локали.
<?php
declare(strict_types=1);
// Number formatting
$number = 1234567.89;
// Decimal format
$fmt = new NumberFormatter('ru_RU', NumberFormatter::DECIMAL);
echo $fmt->format($number) . "\n"; // '1 234 567,89'
$fmt = new NumberFormatter('en_US', NumberFormatter::DECIMAL);
echo $fmt->format($number) . "\n"; // '1,234,567.89'
$fmt = new NumberFormatter('de_DE', NumberFormatter::DECIMAL);
echo $fmt->format($number) . "\n"; // '1.234.567,89'
// Currency
$fmt = new NumberFormatter('ru_RU', NumberFormatter::CURRENCY);
echo $fmt->formatCurrency(1234.56, 'RUB') . "\n"; // '1 234,56 ₽'
echo $fmt->formatCurrency(1234.56, 'USD') . "\n"; // '1 234,56 $'
echo $fmt->formatCurrency(1234.56, 'EUR') . "\n"; // '1 234,56 €'
$fmt = new NumberFormatter('en_US', NumberFormatter::CURRENCY);
echo $fmt->formatCurrency(1234.56, 'USD') . "\n"; // '$1,234.56'
echo $fmt->formatCurrency(1234.56, 'RUB') . "\n"; // 'RUB 1,234.56'
// Percentage
$fmt = new NumberFormatter('ru_RU', NumberFormatter::PERCENT);
echo $fmt->format(0.856) . "\n"; // '86 %'
$fmt = new NumberFormatter('en_US', NumberFormatter::PERCENT);
echo $fmt->format(0.856) . "\n"; // '86%'
// Spelled out (words)
$fmt = new NumberFormatter('ru_RU', NumberFormatter::SPELLOUT);
echo $fmt->format(42) . "\n"; // 'сорок два'
echo $fmt->format(1001) . "\n"; // 'одна тысяча один'
echo $fmt->format(3.14) . "\n"; // 'три целых четырнадцать сотых'
$fmt = new NumberFormatter('en_US', NumberFormatter::SPELLOUT);
echo $fmt->format(42) . "\n"; // 'forty-two'
// Ordinal
$fmt = new NumberFormatter('en_US', NumberFormatter::ORDINAL);
echo $fmt->format(1) . "\n"; // '1st'
echo $fmt->format(2) . "\n"; // '2nd'
echo $fmt->format(42) . "\n"; // '42nd'
// Duration (seconds → hours:minutes:seconds)
$fmt = new NumberFormatter('en_US', NumberFormatter::DURATION);
echo $fmt->format(3661) . "\n"; // '1:01:01'
// Scientific notation
$fmt = new NumberFormatter('en_US', NumberFormatter::SCIENTIFIC);
echo $fmt->format(1234567) . "\n"; // '1.234567E6'
// Parsing (string → number)
$fmt = new NumberFormatter('ru_RU', NumberFormatter::DECIMAL);
$parsed = $fmt->parse('1 234,56');
echo $parsed . "\n"; // 1234.56 (float)
$fmt = new NumberFormatter('en_US', NumberFormatter::CURRENCY);
$parsed = $fmt->parseCurrency('$1,234.56', $currency);
echo "{$parsed} ({$currency})\n"; // 1234.56 (USD)
Настройка атрибутов
<?php
declare(strict_types=1);
$fmt = new NumberFormatter('ru_RU', NumberFormatter::DECIMAL);
// Set minimum fraction digits
$fmt->setAttribute(NumberFormatter::MIN_FRACTION_DIGITS, 2);
echo $fmt->format(42) . "\n"; // '42,00'
// Set maximum fraction digits
$fmt->setAttribute(NumberFormatter::MAX_FRACTION_DIGITS, 4);
echo $fmt->format(3.14159) . "\n"; // '3,1416'
// Grouping separator
$fmt->setAttribute(NumberFormatter::GROUPING_USED, 0);
echo $fmt->format(1234567) . "\n"; // '1234567'
// Rounding mode
$fmt->setAttribute(NumberFormatter::ROUNDING_MODE, NumberFormatter::ROUND_HALFUP);
echo $fmt->format(2.555) . "\n";
Intl Extension — IntlDateFormatter
<?php
declare(strict_types=1);
$timestamp = time();
// Short date
$fmt = new IntlDateFormatter(
locale: 'ru_RU',
dateType: IntlDateFormatter::SHORT,
timeType: IntlDateFormatter::SHORT,
);
echo $fmt->format($timestamp) . "\n"; // '24.02.2026, 14:30'
// Long date
$fmt = new IntlDateFormatter(
locale: 'ru_RU',
dateType: IntlDateFormatter::LONG,
timeType: IntlDateFormatter::LONG,
);
echo $fmt->format($timestamp) . "\n"; // '24 февраля 2026 г., 14:30:00 MSK'
// Full date
$fmt = new IntlDateFormatter(
locale: 'ru_RU',
dateType: IntlDateFormatter::FULL,
timeType: IntlDateFormatter::FULL,
);
echo $fmt->format($timestamp) . "\n";
// 'вторник, 24 февраля 2026 г., 14:30:00 Москва, стандартное время'
// English locale
$fmt = new IntlDateFormatter(
locale: 'en_US',
dateType: IntlDateFormatter::FULL,
timeType: IntlDateFormatter::SHORT,
);
echo $fmt->format($timestamp) . "\n";
// 'Tuesday, February 24, 2026 at 2:30 PM'
// Custom pattern
$fmt = new IntlDateFormatter(
locale: 'ru_RU',
dateType: IntlDateFormatter::NONE,
timeType: IntlDateFormatter::NONE,
pattern: "d MMMM yyyy 'г.', EEEE",
);
echo $fmt->format($timestamp) . "\n"; // '24 февраля 2026 г., вторник'
// Relative dates
$fmt = new IntlDateFormatter(
locale: 'ru_RU',
dateType: IntlDateFormatter::RELATIVE_SHORT,
timeType: IntlDateFormatter::SHORT,
);
echo $fmt->format(time()) . "\n"; // 'сегодня, 14:30'
echo $fmt->format(time() - 86400) . "\n"; // 'вчера, 14:30'
echo $fmt->format(time() + 86400) . "\n"; // 'завтра, 14:30'
// DateTimeImmutable works too
$dt = new DateTimeImmutable('2026-12-31 23:59:59');
$fmt = new IntlDateFormatter('ru_RU', IntlDateFormatter::LONG, IntlDateFormatter::SHORT);
echo $fmt->format($dt) . "\n"; // '31 декабря 2026 г., 23:59'
Intl Extension — Collator
Collator выполняет сортировку строк с учётом языковых правил (а не по кодам символов).
<?php
declare(strict_types=1);
// Default sort (by byte values) — WRONG for many languages
$cities = ['Ёбург', 'Архангельск', 'Якутск', 'Барнаул', 'Ёж'];
sort($cities);
print_r($cities);
// Wrong! Ё is sorted incorrectly (different Unicode position)
// Collator — linguistically correct sort
$collator = new Collator('ru_RU');
$collator->sort($cities);
print_r($cities);
// ['Архангельск', 'Барнаул', 'Ёбург', 'Ёж', 'Якутск']
// Case-insensitive comparison
$collator = new Collator('ru_RU');
$collator->setStrength(Collator::SECONDARY); // Ignore case
echo $collator->compare('привет', 'Привет') . "\n"; // 0 (equal)
echo $collator->compare('а', 'б') . "\n"; // -1 (а < б)
echo $collator->compare('я', 'а') . "\n"; // 1 (я > а)
// Sort with keys preserved (asort)
$items = ['c' => 'яблоко', 'a' => 'арбуз', 'b' => 'банан'];
$collator->asort($items);
print_r($items);
// ['a' => 'арбуз', 'b' => 'банан', 'c' => 'яблоко']
// usort with Collator
$names = ['Ёлкин', 'Абрамов', 'Яковлев', 'Борисов'];
usort($names, [$collator, 'compare']);
print_r($names);
// ['Абрамов', 'Борисов', 'Ёлкин', 'Яковлев']
// German: ä, ö, ü treated as a, o, u (not separate letters)
$collator = new Collator('de_DE');
$words = ['Übung', 'Apfel', 'Öl', 'Zucker'];
$collator->sort($words);
print_r($words);
// ['Apfel', 'Öl', 'Übung', 'Zucker']
// Swedish: ö is a separate letter AFTER z
$collator = new Collator('sv_SE');
$words = ['öl', 'zoo', 'äpple'];
$collator->sort($words);
print_r($words);
// ['zoo', 'äpple', 'öl'] — ä and ö after z in Swedish!
Intl Extension — MessageFormatter
MessageFormatter реализует ICU Message Format — стандарт для интернационализированных строк с плюрализацией и условными подстановками.
<?php
declare(strict_types=1);
// Simple substitution
$msg = MessageFormatter::formatMessage(
'ru_RU',
'Привет, {0}! Тебе {1} лет.',
['Иван', 30],
);
echo $msg . "\n"; // 'Привет, Иван! Тебе 30 лет.'
// Named arguments
$msg = MessageFormatter::formatMessage(
'ru_RU',
'Привет, {name}! Тебе {age, number} лет.',
['name' => 'Иван', 'age' => 30],
);
echo $msg . "\n";
// Pluralization
$pattern = '{count, plural, =0{Нет сообщений} =1{Одно сообщение} one{# сообщение} few{# сообщения} many{# сообщений} other{# сообщений}}';
foreach ([0, 1, 2, 5, 21, 42, 111] as $count) {
$msg = MessageFormatter::formatMessage('ru_RU', $pattern, ['count' => $count]);
echo "{$count}: {$msg}\n";
}
// 0: Нет сообщений
// 1: Одно сообщение
// 2: 2 сообщения
// 5: 5 сообщений
// 21: 21 сообщение
// 42: 42 сообщения
// 111: 111 сообщений
// Select (gender/choice)
$pattern = '{gender, select, male{Он отправил} female{Она отправила} other{Отправлено}} {count, plural, =1{# сообщение} few{# сообщения} other{# сообщений}}';
echo MessageFormatter::formatMessage('ru_RU', $pattern, [
'gender' => 'female',
'count' => 5,
]) . "\n";
// 'Она отправила 5 сообщений'
// Number formatting inside messages
$pattern = 'Цена: {price, number, currency}';
$msg = MessageFormatter::formatMessage('ru_RU', $pattern, ['price' => 1234.56]);
echo $msg . "\n"; // 'Цена: 1 234,56 ₽'
// Date formatting inside messages
$pattern = 'Создано: {date, date, long}';
$msg = MessageFormatter::formatMessage('ru_RU', $pattern, ['date' => time()]);
echo $msg . "\n"; // 'Создано: 24 февраля 2026 г.'
Intl Extension — Transliterator
Transliterator конвертирует текст между скриптами (системами письма) по правилам ICU.
<?php
declare(strict_types=1);
// Cyrillic → Latin (transliteration)
$transliterator = Transliterator::create('Russian-Latin/BGN');
echo $transliterator->transliterate('Привет, мир!') . "\n";
// 'Privet, mir!'
// Any Cyrillic → Latin
$transliterator = Transliterator::create('Cyrillic-Latin');
echo $transliterator->transliterate('Привет, мир!') . "\n";
// 'Privet, mir!'
// Latin → Cyrillic
$transliterator = Transliterator::create('Latin-Cyrillic');
echo $transliterator->transliterate('Privet') . "\n";
// 'Привет'
// Remove diacritics (accents)
$transliterator = Transliterator::create('NFD; [:Nonspacing Mark:] Remove; NFC');
echo $transliterator->transliterate('café résumé naïve') . "\n";
// 'cafe resume naive'
// To ASCII (slug generation)
$transliterator = Transliterator::create('Any-Latin; Latin-ASCII; Lower()');
$text = 'Привет, мир! Ёлочка & Ёжик';
$slug = $transliterator->transliterate($text);
$slug = preg_replace('/[^a-z0-9]+/', '-', $slug);
$slug = trim($slug, '-');
echo $slug . "\n";
// 'privet-mir-elochka-ezhik'
// Greek → Latin
$transliterator = Transliterator::create('Greek-Latin');
echo $transliterator->transliterate('Αλφάβητο') . "\n";
// 'Alfávīto'
// Japanese Katakana → Latin
$transliterator = Transliterator::create('Katakana-Latin');
echo $transliterator->transliterate('カタカナ') . "\n";
// 'katakana'
// Available transliterator IDs
$ids = Transliterator::listIDs();
echo "Available: " . count($ids) . " transliterators\n";
// Usually 500+
Intl Extension — IntlBreakIterator
IntlBreakIterator разбивает текст на слова, предложения, строки и grapheme clusters по правилам Unicode.
<?php
declare(strict_types=1);
// Word boundary detection
$text = 'Привет, мир! Как дела? Everything is fine.';
$bi = IntlBreakIterator::createWordInstance('ru_RU');
$bi->setText($text);
$words = [];
$prev = 0;
foreach ($bi as $pos) {
$word = mb_substr($text, $prev, $pos - $prev);
if (trim($word) !== '' && preg_match('/\w/u', $word)) {
$words[] = $word;
}
$prev = $pos;
}
print_r($words);
// ['Привет', 'мир', 'Как', 'дела', 'Everything', 'is', 'fine']
// Sentence boundary detection
$bi = IntlBreakIterator::createSentenceInstance('ru_RU');
$bi->setText($text);
$sentences = [];
$prev = 0;
foreach ($bi as $pos) {
$sentence = trim(mb_substr($text, $prev, $pos - $prev));
if ($sentence !== '') {
$sentences[] = $sentence;
}
$prev = $pos;
}
print_r($sentences);
// ['Привет, мир!', 'Как дела?', 'Everything is fine.']
// Line break opportunities
$bi = IntlBreakIterator::createLineInstance('ru_RU');
$bi->setText('Длинный текст для переноса строки');
echo "Line breaks at positions: ";
foreach ($bi as $pos) {
echo "{$pos} ";
}
echo "\n";
IntlListFormatter (PHP 8.5+)
<?php
declare(strict_types=1);
// IntlListFormatter formats lists according to locale conventions
// Available in PHP 8.5+
// Russian: "и" as conjunction
$formatter = new IntlListFormatter('ru_RU', IntlListFormatter::TYPE_CONJUNCTION);
echo $formatter->format(['яблоки', 'бананы', 'апельсины']) . "\n";
// 'яблоки, бананы и апельсины'
// English: "and"
$formatter = new IntlListFormatter('en_US', IntlListFormatter::TYPE_CONJUNCTION);
echo $formatter->format(['apples', 'bananas', 'oranges']) . "\n";
// 'apples, bananas, and oranges' (Oxford comma!)
// Disjunction: "или" / "or"
$formatter = new IntlListFormatter('ru_RU', IntlListFormatter::TYPE_DISJUNCTION);
echo $formatter->format(['красный', 'синий', 'зелёный']) . "\n";
// 'красный, синий или зелёный'
// Unit list
$formatter = new IntlListFormatter('ru_RU', IntlListFormatter::TYPE_UNIT);
echo $formatter->format(['5 кг', '200 г']) . "\n";
// '5 кг, 200 г'
Locale class
<?php
declare(strict_types=1);
// Get default locale
echo Locale::getDefault() . "\n"; // 'en_US_POSIX' or system locale
// Set default locale
Locale::setDefault('ru_RU');
// Parse locale string
$locale = Locale::parseLocale('ru_RU.UTF-8@calendar=gregorian');
print_r($locale);
// ['language' => 'ru', 'region' => 'RU']
// Get display names
echo Locale::getDisplayName('ru_RU', 'en') . "\n"; // 'Russian (Russia)'
echo Locale::getDisplayName('ru_RU', 'ru') . "\n"; // 'русский (Россия)'
echo Locale::getDisplayLanguage('ru_RU', 'en') . "\n"; // 'Russian'
echo Locale::getDisplayRegion('ru_RU', 'en') . "\n"; // 'Russia'
// Lookup best match from available locales
$available = ['en_US', 'ru_RU', 'de_DE', 'fr_FR'];
$best = Locale::lookup($available, 'ru_UA', false, 'en_US');
echo "Best match: {$best}\n"; // 'ru_RU' (Russian, closest to ru_UA)
// Accept from HTTP header
$httpAccept = 'ru-RU,ru;q=0.9,en-US;q=0.8,en;q=0.7';
$best = Locale::acceptFromHttp($httpAccept);
echo "HTTP best: {$best}\n"; // 'ru_RU'
// Compose locale
$locale = Locale::composeLocale([
'language' => 'ru',
'region' => 'RU',
]);
echo "Composed: {$locale}\n"; // 'ru_RU'
iconv — конвертация кодировок (fallback)
<?php
declare(strict_types=1);
// iconv() — basic encoding conversion
$windows1251 = iconv('UTF-8', 'Windows-1251', 'Привет');
$utf8 = iconv('Windows-1251', 'UTF-8', $windows1251);
// With transliteration (replace unknown chars)
$ascii = iconv('UTF-8', 'ASCII//TRANSLIT', 'café');
echo $ascii . "\n"; // "cafe'" or "cafe" depending on system
// With ignore (skip unknown chars)
$ascii = iconv('UTF-8', 'ASCII//IGNORE', 'Привет hello');
echo $ascii . "\n"; // " hello"
// iconv_strlen, iconv_substr, iconv_strpos
echo iconv_strlen('Привет', 'UTF-8') . "\n"; // 6
// Prefer mb_* functions over iconv — more reliable and feature-rich
Практические примеры
Валидация многоязычной формы
<?php
declare(strict_types=1);
final class MultilingualValidator
{
/**
* Validate a name (letters from any script + spaces/hyphens).
*/
public function validateName(string $name): bool
{
// \p{L} matches any Unicode letter (Cyrillic, Latin, etc.)
return (bool) preg_match('/^[\p{L}\s\'-]{2,100}$/u', $name);
}
/**
* Validate that string length is within bounds (grapheme-based).
*/
public function validateLength(
string $value,
int $min,
int $max,
): bool {
$length = grapheme_strlen($value);
return $length >= $min && $length <= $max;
}
/**
* Normalize whitespace in multibyte string.
*/
public function normalizeWhitespace(string $text): string
{
// Replace various Unicode whitespace with regular space
$text = preg_replace('/[\p{Zs}\t]+/u', ' ', $text);
return trim($text);
}
/**
* Generate URL slug from any language.
*/
public function slugify(string $text): string
{
$transliterator = Transliterator::create(
'Any-Latin; Latin-ASCII; Lower()',
);
$slug = $transliterator->transliterate($text);
$slug = preg_replace('/[^a-z0-9]+/', '-', $slug);
return trim($slug, '-');
}
}
$validator = new MultilingualValidator();
echo $validator->validateName('Иван Петров-Водкин') ? "valid\n" : "invalid\n"; // valid
echo $validator->validateName('名前太郎') ? "valid\n" : "invalid\n"; // valid
echo $validator->validateName('123') ? "valid\n" : "invalid\n"; // invalid
echo $validator->slugify('Привет мир!') . "\n"; // 'privet-mir'
echo $validator->slugify('Über uns') . "\n"; // 'uber-uns'
Форматирование чисел и валют по локали
<?php
declare(strict_types=1);
final class LocaleFormatter
{
private NumberFormatter $numberFmt;
private NumberFormatter $currencyFmt;
private IntlDateFormatter $dateFmt;
public function __construct(
private readonly string $locale,
) {
$this->numberFmt = new NumberFormatter($locale, NumberFormatter::DECIMAL);
$this->currencyFmt = new NumberFormatter($locale, NumberFormatter::CURRENCY);
$this->dateFmt = new IntlDateFormatter(
$locale,
IntlDateFormatter::LONG,
IntlDateFormatter::SHORT,
);
}
public function number(float|int $value, int $decimals = 2): string
{
$this->numberFmt->setAttribute(
NumberFormatter::MAX_FRACTION_DIGITS,
$decimals,
);
return $this->numberFmt->format($value);
}
public function currency(float $amount, string $currencyCode): string
{
return $this->currencyFmt->formatCurrency($amount, $currencyCode);
}
public function date(DateTimeInterface $date): string
{
return $this->dateFmt->format($date);
}
public function pluralize(int $count, string $pattern): string
{
return MessageFormatter::formatMessage(
$this->locale,
$pattern,
['count' => $count],
);
}
}
// Russian
$ru = new LocaleFormatter('ru_RU');
echo $ru->number(1234567.89) . "\n"; // '1 234 567,89'
echo $ru->currency(99.90, 'RUB') . "\n"; // '99,90 ₽'
echo $ru->date(new DateTimeImmutable()) . "\n"; // '24 февраля 2026 г., 14:30'
echo $ru->pluralize(5, '{count, plural, =1{# товар} few{# товара} other{# товаров}}') . "\n";
// '5 товаров'
// English
$en = new LocaleFormatter('en_US');
echo $en->number(1234567.89) . "\n"; // '1,234,567.89'
echo $en->currency(99.90, 'USD') . "\n"; // '$99.90'
echo $en->date(new DateTimeImmutable()) . "\n"; // 'February 24, 2026 at 2:30 PM'
echo $en->pluralize(5, '{count, plural, =1{# item} other{# items}}') . "\n";
// '5 items'