MidТеория10 min

Парсинг XML

SimpleXML, DOM, XMLReader, XPath, namespaces, PHP 8.4 Dom\HTMLDocument

PHP предоставляет три основных API для разбора XML: SimpleXML (простой и удобный), DOM (полный стандарт W3C) и XMLReader (потоковый для больших файлов). Каждый подходит для своих задач. В PHP 8.4 появился новый DOM API -- Dom\HTMLDocument и Dom\XMLDocument.

SimpleXML -- простой и быстрый

SimpleXML преобразует XML в объект с интуитивным доступом к элементам через свойства и итерацию.

<?php
declare(strict_types=1);

// === Load from string ===
$xmlString = <<<XML
<?xml version="1.0" encoding="UTF-8"?>
<catalog>
    <book id="1" lang="en">
        <title>PHP 8 in Action</title>
        <author>John Doe</author>
        <price currency="USD">39.99</price>
        <tags>
            <tag>php</tag>
            <tag>programming</tag>
        </tags>
    </book>
    <book id="2" lang="ru">
        <title>Symfony Deep Dive</title>
        <author>Jane Smith</author>
        <price currency="EUR">44.99</price>
        <tags>
            <tag>symfony</tag>
            <tag>framework</tag>
        </tags>
    </book>
</catalog>
XML;

$xml = simplexml_load_string($xmlString);

if ($xml === false) {
    $errors = libxml_get_errors();
    foreach ($errors as $error) {
        echo "XML Error: {$error->message} on line {$error->line}\n";
    }
    libxml_clear_errors();
    throw new RuntimeException('Invalid XML');
}

// === Access elements ===
echo $xml->book[0]->title;     // 'PHP 8 in Action'
echo $xml->book[1]->author;    // 'Jane Smith'

// === Access attributes ===
echo $xml->book[0]['id'];           // '1'
echo $xml->book[0]['lang'];         // 'en'
echo $xml->book[0]->price['currency']; // 'USD'

// ⚠️ SimpleXML returns SimpleXMLElement objects, not strings!
$title = $xml->book[0]->title;
echo get_class($title); // 'SimpleXMLElement'

// Cast to string explicitly
$titleStr = (string) $xml->book[0]->title;

// Cast to other types
$price = (float) $xml->book[0]->price;    // 39.99
$id = (int) $xml->book[0]['id'];           // 1

// === Load from file ===
$xml = simplexml_load_file('/path/to/catalog.xml');

// With options (suppress warnings, handle entities)
$xml = simplexml_load_string($xmlString, SimpleXMLElement::class, LIBXML_NOCDATA | LIBXML_NOERROR);

SimpleXML -- итерация и поиск

<?php
declare(strict_types=1);

$xml = simplexml_load_string($xmlString);

// === Iterate over elements ===
foreach ($xml->book as $book) {
    echo sprintf(
        "Book #%d: %s by %s (%.2f %s)\n",
        (int) $book['id'],
        (string) $book->title,
        (string) $book->author,
        (float) $book->price,
        (string) $book->price['currency'],
    );
}

// === Count elements ===
echo count($xml->book); // 2
echo $xml->book->count(); // 2

// === children() — access child elements ===
$firstBook = $xml->book[0];
foreach ($firstBook->children() as $name => $child) {
    echo "$name: $child\n";
    // title: PHP 8 in Action
    // author: John Doe
    // price: 39.99
    // tags: [SimpleXMLElement]
}

// === attributes() — access all attributes ===
foreach ($firstBook->attributes() as $name => $value) {
    echo "$name = $value\n";
    // id = 1
    // lang = en
}

// === Nested iteration ===
foreach ($xml->book[0]->tags->tag as $tag) {
    echo "Tag: $tag\n";
}

// === Check if element/attribute exists ===
if (isset($xml->book[0]->subtitle)) {
    echo "Has subtitle\n";
} else {
    echo "No subtitle\n"; // This one
}

// === XPath queries ===
// Find all books with price > 40
$expensive = $xml->xpath('//book[price > 40]');
foreach ($expensive as $book) {
    echo "Expensive: {$book->title}\n"; // 'Symfony Deep Dive'
}

// Find book by attribute
$ruBooks = $xml->xpath('//book[@lang="ru"]');

// Find all tags
$allTags = $xml->xpath('//tag');

// Find books by author
$janeBooks = $xml->xpath('//book[author="Jane Smith"]');

// XPath with text functions
$phpBooks = $xml->xpath('//book[contains(title, "PHP")]');

// XPath predicates
$firstBook = $xml->xpath('//book[1]');       // First book
$lastBook = $xml->xpath('//book[last()]');    // Last book
$prices = $xml->xpath('//book/price/text()'); // All prices as text

SimpleXML -- пространства имен

<?php
declare(strict_types=1);

$xmlWithNs = <<<XML
<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
    <title>My Blog</title>
    <entry>
        <title>First Post</title>
        <dc:creator>John</dc:creator>
        <dc:date>2024-01-15</dc:date>
    </entry>
</feed>
XML;

$xml = simplexml_load_string($xmlWithNs);

// Default namespace — must register to query
$xml->registerXPathNamespace('atom', 'http://www.w3.org/2005/Atom');
$xml->registerXPathNamespace('dc', 'http://purl.org/dc/elements/1.1/');

// Access elements in default namespace
$titles = $xml->xpath('//atom:title');
echo (string) $titles[0]; // 'My Blog'

// Access namespaced elements
$creators = $xml->xpath('//dc:creator');
echo (string) $creators[0]; // 'John'

// children() with namespace
$entry = $xml->entry[0] ?? null;
if ($entry !== null) {
    $dcChildren = $entry->children('http://purl.org/dc/elements/1.1/');
    echo (string) $dcChildren->creator; // 'John'
    echo (string) $dcChildren->date;    // '2024-01-15'
}

// Get namespaces used in document
$namespaces = $xml->getNamespaces(true); // recursive = true
print_r($namespaces);
// ['http://www.w3.org/2005/Atom', 'dc' => 'http://purl.org/dc/elements/1.1/']

SimpleXML -- конвертация в массив

<?php
declare(strict_types=1);

// Convert SimpleXML to array (common need)
function xmlToArray(SimpleXMLElement $xml): array
{
    $result = [];

    // Add attributes
    foreach ($xml->attributes() as $name => $value) {
        $result['@' . $name] = (string) $value;
    }

    // Add children
    foreach ($xml->children() as $name => $child) {
        $childArray = xmlToArray($child);

        if (isset($result[$name])) {
            // Multiple children with same name -> make array
            if (!isset($result[$name][0])) {
                $result[$name] = [$result[$name]];
            }
            $result[$name][] = $childArray;
        } else {
            $result[$name] = $childArray;
        }
    }

    // If no children, add text content
    $text = trim((string) $xml);
    if ($result === [] && $text !== '') {
        return $text;
    }

    if ($text !== '') {
        $result['#text'] = $text;
    }

    return $result;
}

// Quick and dirty (loses attributes):
$json = json_encode($xml);
$array = json_decode($json, true);

DOM -- полный контроль (W3C стандарт)

<?php
declare(strict_types=1);

$xmlString = <<<XML
<?xml version="1.0" encoding="UTF-8"?>
<catalog>
    <book id="1">
        <title>PHP 8 in Action</title>
        <author>John Doe</author>
        <price currency="USD">39.99</price>
    </book>
    <book id="2">
        <title>Symfony Guide</title>
        <author>Jane Smith</author>
        <price currency="EUR">44.99</price>
    </book>
</catalog>
XML;

// === Load XML ===
$dom = new DOMDocument('1.0', 'UTF-8');
$dom->preserveWhiteSpace = false;
$dom->formatOutput = true;

$loaded = $dom->loadXML($xmlString);
if (!$loaded) {
    throw new RuntimeException('Failed to parse XML');
}

// Load from file
// $dom->load('/path/to/file.xml');

// === Access elements ===
$books = $dom->getElementsByTagName('book');
echo $books->length; // 2

foreach ($books as $book) {
    /** @var DOMElement $book */
    echo $book->getAttribute('id') . ': ';
    echo $book->getElementsByTagName('title')->item(0)?->textContent . "\n";
}

// === getElementById (requires DTD or schema) ===
// Alternatively, use XPath for id-based lookup

// === DOMNodeList iteration ===
$titles = $dom->getElementsByTagName('title');
for ($i = 0; $i < $titles->length; $i++) {
    $node = $titles->item($i);
    echo $node->textContent . "\n";
}

// === Node properties ===
$firstBook = $books->item(0);
echo $firstBook->nodeName;       // 'book'
echo $firstBook->nodeType;       // XML_ELEMENT_NODE (1)
echo $firstBook->textContent;    // All text content recursively
echo $firstBook->nodeValue;      // null for elements

// Child nodes
echo $firstBook->childNodes->length; // Number of child nodes
echo $firstBook->firstChild->nodeName;
echo $firstBook->lastChild->nodeName;

// Parent and siblings
echo $firstBook->parentNode->nodeName;    // 'catalog'
echo $firstBook->nextSibling?->nodeName;  // 'book' (second book)

// Check node type
if ($firstBook->nodeType === XML_ELEMENT_NODE) {
    echo "It's an element!\n";
}

// Has children/attributes
echo $firstBook->hasChildNodes() ? 'Yes' : 'No';
echo $firstBook->hasAttributes() ? 'Yes' : 'No';

DOMXPath -- мощные запросы

<?php
declare(strict_types=1);

$dom = new DOMDocument();
$dom->loadXML($xmlString);

$xpath = new DOMXPath($dom);

// === Basic queries ===
// Find all book titles
$titles = $xpath->query('//book/title');
foreach ($titles as $title) {
    echo $title->textContent . "\n";
}

// Find books with specific attribute
$book1 = $xpath->query('//book[@id="1"]')->item(0);
echo $book1->getElementsByTagName('title')->item(0)->textContent;

// Find with conditions
$expensive = $xpath->query('//book[price > 40]');
$usdBooks = $xpath->query('//book[price/@currency="USD"]');

// === evaluate() — returns typed values ===
$count = $xpath->evaluate('count(//book)');
echo $count; // 2

$totalPrice = $xpath->evaluate('sum(//book/price)');
echo $totalPrice; // 84.98

$hasBooks = $xpath->evaluate('boolean(//book)');
echo $hasBooks ? 'Yes' : 'No'; // Yes

// Find text content directly
$firstTitle = $xpath->evaluate('string(//book[1]/title)');
echo $firstTitle; // 'PHP 8 in Action'

// === Context node ===
$firstBook = $dom->getElementsByTagName('book')->item(0);
$title = $xpath->query('title', $firstBook)->item(0);
echo $title->textContent; // 'PHP 8 in Action'

// === With namespaces ===
$xpath->registerNamespace('atom', 'http://www.w3.org/2005/Atom');
$entries = $xpath->query('//atom:entry');

PHP 8.4 -- новый DOM API

<?php
declare(strict_types=1);

// PHP 8.4+: Modern DOM API with spec-compliant behavior

// === Dom\XMLDocument — new XML parser ===
$doc = Dom\XMLDocument::createFromString($xmlString);

// querySelector / querySelectorAll (CSS selectors!)
$firstBook = $doc->querySelector('book');
echo $firstBook?->getAttribute('id'); // '1'

$allBooks = $doc->querySelectorAll('book');
foreach ($allBooks as $book) {
    $title = $book->querySelector('title');
    echo $title?->textContent . "\n";
}

// CSS selectors work like in browsers
$usdPrices = $doc->querySelectorAll('price[currency="USD"]');
$firstBookTitle = $doc->querySelector('book:first-child > title');

// Create from file
$doc = Dom\XMLDocument::createFromFile('/path/to/file.xml');

// === Dom\HTMLDocument — HTML5 parser ===
$html = '<div class="user"><h1>John</h1><p>Developer</p></div>';
$htmlDoc = Dom\HTMLDocument::createFromString($html);

$name = $htmlDoc->querySelector('.user h1');
echo $name?->textContent; // 'John'

// Parse full HTML pages
$htmlDoc = Dom\HTMLDocument::createFromFile('https://example.com');
$title = $htmlDoc->querySelector('title');
$links = $htmlDoc->querySelectorAll('a[href]');

foreach ($links as $link) {
    echo $link->getAttribute('href') . "\n";
}

// innerHTML support
$div = $htmlDoc->querySelector('div');
echo $div?->innerHTML;

// Serialization
echo $doc->saveXML();       // XMLDocument
echo $htmlDoc->saveHTML();   // HTMLDocument

XMLReader -- потоковый парсинг больших файлов

<?php
declare(strict_types=1);

// XMLReader is a pull parser — reads XML node by node
// Memory efficient: doesn't load entire document into memory
// Perfect for large XML files (100MB+)

// === Basic usage ===
$reader = new XMLReader();
$reader->open('/path/to/large-catalog.xml');
// Or from string: $reader->XML($xmlString);

while ($reader->read()) {
    // Process each node
    if ($reader->nodeType === XMLReader::ELEMENT && $reader->name === 'book') {
        echo "Book found!\n";

        // Read attribute
        $id = $reader->getAttribute('id');
        echo "ID: $id\n";
    }

    if ($reader->nodeType === XMLReader::ELEMENT && $reader->name === 'title') {
        // Move to text content
        $reader->read();
        echo "Title: {$reader->value}\n";
    }
}

$reader->close();

// === Process specific elements ===
$reader = new XMLReader();
$reader->open('/path/to/products.xml');

$products = [];

while ($reader->read()) {
    if ($reader->nodeType !== XMLReader::ELEMENT || $reader->name !== 'product') {
        continue;
    }

    // expand() converts current node to DOMNode for easy access
    $node = $reader->expand();

    $product = [
        'id'    => $reader->getAttribute('id'),
        'name'  => $node->getElementsByTagName('name')->item(0)?->textContent,
        'price' => (float) ($node->getElementsByTagName('price')->item(0)?->textContent ?? 0),
    ];

    $products[] = $product;

    // Process in batches to save memory
    if (count($products) >= 1000) {
        processBatch($products);
        $products = [];
    }
}

if ($products !== []) {
    processBatch($products);
}

$reader->close();

function processBatch(array $products): void
{
    // Save to database, process, etc.
    echo sprintf("Processing batch of %d products\n", count($products));
}

// === moveToAttribute / moveToElement ===
$reader = new XMLReader();
$reader->XML('<items><item id="1" type="book" active="true"/></items>');

while ($reader->read()) {
    if ($reader->nodeType === XMLReader::ELEMENT && $reader->name === 'item') {
        // Iterate over attributes
        if ($reader->hasAttributes) {
            while ($reader->moveToNextAttribute()) {
                echo "{$reader->name} = {$reader->value}\n";
            }
            $reader->moveToElement(); // Back to element
        }
    }
}

$reader->close();

// === Node types ===
// XMLReader::ELEMENT = 1        — opening tag
// XMLReader::END_ELEMENT = 15   — closing tag
// XMLReader::TEXT = 3           — text content
// XMLReader::CDATA = 4         — CDATA section
// XMLReader::COMMENT = 8       — comment
// XMLReader::ATTRIBUTE = 2     — attribute

Сравнение: SimpleXML vs DOM vs XMLReader

Критерий SimpleXML DOM XMLReader
Простота Очень просто Средне Сложнее
Память Весь документ в RAM Весь документ в RAM Потоковый, минимум RAM
Модификация Ограниченная Полная Только чтение
XPath Да Да (DOMXPath) Нет
CSS селекторы Нет Нет (PHP 8.4 -- да) Нет
Namespaces Да Да Да
Большие файлы Нет Нет Да
Валидация Нет DTD, Schema DTD, Schema
Когда использовать Простой XML, конфиги, API Сложный XML, модификация Файлы > 10MB

Практический пример: RSS/Atom парсер

<?php
declare(strict_types=1);

final readonly class FeedItem
{
    public function __construct(
        public string $title,
        public string $link,
        public string $description,
        public ?DateTimeImmutable $pubDate,
        public string $author,
    ) {}
}

function parseRssFeed(string $xmlContent): array
{
    $xml = simplexml_load_string($xmlContent, SimpleXMLElement::class, LIBXML_NOCDATA);

    if ($xml === false) {
        throw new RuntimeException('Invalid RSS XML');
    }

    $items = [];

    // RSS 2.0 format
    foreach ($xml->channel->item as $item) {
        $items[] = new FeedItem(
            title: (string) $item->title,
            link: (string) $item->link,
            description: (string) $item->description,
            pubDate: $item->pubDate
                ? new DateTimeImmutable((string) $item->pubDate)
                : null,
            author: (string) ($item->author ?? $item->children('dc', true)->creator ?? ''),
        );
    }

    return $items;
}

function parseAtomFeed(string $xmlContent): array
{
    $xml = simplexml_load_string($xmlContent);

    if ($xml === false) {
        throw new RuntimeException('Invalid Atom XML');
    }

    $xml->registerXPathNamespace('atom', 'http://www.w3.org/2005/Atom');
    $items = [];

    foreach ($xml->entry as $entry) {
        $link = '';
        foreach ($entry->link as $l) {
            if ((string) $l['rel'] === 'alternate' || (string) $l['rel'] === '') {
                $link = (string) $l['href'];
                break;
            }
        }

        $items[] = new FeedItem(
            title: (string) $entry->title,
            link: $link,
            description: (string) ($entry->summary ?? $entry->content ?? ''),
            pubDate: $entry->updated
                ? new DateTimeImmutable((string) $entry->updated)
                : null,
            author: (string) ($entry->author->name ?? ''),
        );
    }

    return $items;
}

// Auto-detect feed type
function parseFeed(string $xmlContent): array
{
    $xml = simplexml_load_string($xmlContent);

    if ($xml === false) {
        throw new RuntimeException('Invalid XML');
    }

    return match ($xml->getName()) {
        'rss'  => parseRssFeed($xmlContent),
        'feed' => parseAtomFeed($xmlContent),
        default => throw new RuntimeException("Unknown feed format: {$xml->getName()}"),
    };
}

Практический пример: SOAP ответ

<?php
declare(strict_types=1);

function parseSoapResponse(string $soapXml): array
{
    $dom = new DOMDocument();
    $dom->loadXML($soapXml);

    $xpath = new DOMXPath($dom);

    // Register SOAP namespaces
    $xpath->registerNamespace('soap', 'http://schemas.xmlsoap.org/soap/envelope/');
    $xpath->registerNamespace('ns', 'http://example.com/webservice');

    // Check for SOAP fault
    $fault = $xpath->query('//soap:Fault');
    if ($fault->length > 0) {
        $faultCode = $xpath->evaluate('string(//soap:Fault/faultcode)');
        $faultString = $xpath->evaluate('string(//soap:Fault/faultstring)');
        throw new RuntimeException("SOAP Fault [$faultCode]: $faultString");
    }

    // Extract response data
    $body = $xpath->query('//soap:Body/*');
    if ($body->length === 0) {
        throw new RuntimeException('Empty SOAP response body');
    }

    // Convert body to SimpleXML for easier access
    $bodyXml = simplexml_import_dom($body->item(0));

    return xmlToArray($bodyXml);
}

Тесты

Вопросы с экзамена ZCE

Проверь себя

5 из 27

Что из следующего извлекает дочерние узлы указанного XML-узла?

Как выполнить XPath запрос в SimpleXML?

Какова роль simplexml_import_dom() в следующем PHP-коде? ```php $dom = new domDocument; $dom->loadXML('<email><from>John</from></email>'); $xml = simplexml_import_dom($dom); echo $xml->from; ```

Какая из следующих функций устанавливает обработчики начала и конца элементов?

Какой парсер подходит для XML файлов размером 500MB?