PHP предоставляет три основных API для разбора XML: SimpleXML (простой и удобный), DOM (полный стандарт W3C) и XMLReader (потоковый для больших файлов). Каждый подходит для своих задач. В PHP 8.4 появился новый DOM API -- Dom\HTMLDocument и Dom\XMLDocument.
SimpleXML -- простой и быстрый
SimpleXML преобразует XML в объект с интуитивным доступом к элементам через свойства и итерацию.
<?php
declare(strict_types=1);
// === Load from string ===
$xmlString = <<<XML
<?xml version="1.0" encoding="UTF-8"?>
<catalog>
<book id="1" lang="en">
<title>PHP 8 in Action</title>
<author>John Doe</author>
<price currency="USD">39.99</price>
<tags>
<tag>php</tag>
<tag>programming</tag>
</tags>
</book>
<book id="2" lang="ru">
<title>Symfony Deep Dive</title>
<author>Jane Smith</author>
<price currency="EUR">44.99</price>
<tags>
<tag>symfony</tag>
<tag>framework</tag>
</tags>
</book>
</catalog>
XML;
$xml = simplexml_load_string($xmlString);
if ($xml === false) {
$errors = libxml_get_errors();
foreach ($errors as $error) {
echo "XML Error: {$error->message} on line {$error->line}\n";
}
libxml_clear_errors();
throw new RuntimeException('Invalid XML');
}
// === Access elements ===
echo $xml->book[0]->title; // 'PHP 8 in Action'
echo $xml->book[1]->author; // 'Jane Smith'
// === Access attributes ===
echo $xml->book[0]['id']; // '1'
echo $xml->book[0]['lang']; // 'en'
echo $xml->book[0]->price['currency']; // 'USD'
// ⚠️ SimpleXML returns SimpleXMLElement objects, not strings!
$title = $xml->book[0]->title;
echo get_class($title); // 'SimpleXMLElement'
// Cast to string explicitly
$titleStr = (string) $xml->book[0]->title;
// Cast to other types
$price = (float) $xml->book[0]->price; // 39.99
$id = (int) $xml->book[0]['id']; // 1
// === Load from file ===
$xml = simplexml_load_file('/path/to/catalog.xml');
// With options (suppress warnings, handle entities)
$xml = simplexml_load_string($xmlString, SimpleXMLElement::class, LIBXML_NOCDATA | LIBXML_NOERROR);
SimpleXML -- итерация и поиск
<?php
declare(strict_types=1);
$xml = simplexml_load_string($xmlString);
// === Iterate over elements ===
foreach ($xml->book as $book) {
echo sprintf(
"Book #%d: %s by %s (%.2f %s)\n",
(int) $book['id'],
(string) $book->title,
(string) $book->author,
(float) $book->price,
(string) $book->price['currency'],
);
}
// === Count elements ===
echo count($xml->book); // 2
echo $xml->book->count(); // 2
// === children() — access child elements ===
$firstBook = $xml->book[0];
foreach ($firstBook->children() as $name => $child) {
echo "$name: $child\n";
// title: PHP 8 in Action
// author: John Doe
// price: 39.99
// tags: [SimpleXMLElement]
}
// === attributes() — access all attributes ===
foreach ($firstBook->attributes() as $name => $value) {
echo "$name = $value\n";
// id = 1
// lang = en
}
// === Nested iteration ===
foreach ($xml->book[0]->tags->tag as $tag) {
echo "Tag: $tag\n";
}
// === Check if element/attribute exists ===
if (isset($xml->book[0]->subtitle)) {
echo "Has subtitle\n";
} else {
echo "No subtitle\n"; // This one
}
// === XPath queries ===
// Find all books with price > 40
$expensive = $xml->xpath('//book[price > 40]');
foreach ($expensive as $book) {
echo "Expensive: {$book->title}\n"; // 'Symfony Deep Dive'
}
// Find book by attribute
$ruBooks = $xml->xpath('//book[@lang="ru"]');
// Find all tags
$allTags = $xml->xpath('//tag');
// Find books by author
$janeBooks = $xml->xpath('//book[author="Jane Smith"]');
// XPath with text functions
$phpBooks = $xml->xpath('//book[contains(title, "PHP")]');
// XPath predicates
$firstBook = $xml->xpath('//book[1]'); // First book
$lastBook = $xml->xpath('//book[last()]'); // Last book
$prices = $xml->xpath('//book/price/text()'); // All prices as text
SimpleXML -- пространства имен
<?php
declare(strict_types=1);
$xmlWithNs = <<<XML
<?xml version="1.0"?>
<feed xmlns="http://www.w3.org/2005/Atom" xmlns:dc="http://purl.org/dc/elements/1.1/">
<title>My Blog</title>
<entry>
<title>First Post</title>
<dc:creator>John</dc:creator>
<dc:date>2024-01-15</dc:date>
</entry>
</feed>
XML;
$xml = simplexml_load_string($xmlWithNs);
// Default namespace — must register to query
$xml->registerXPathNamespace('atom', 'http://www.w3.org/2005/Atom');
$xml->registerXPathNamespace('dc', 'http://purl.org/dc/elements/1.1/');
// Access elements in default namespace
$titles = $xml->xpath('//atom:title');
echo (string) $titles[0]; // 'My Blog'
// Access namespaced elements
$creators = $xml->xpath('//dc:creator');
echo (string) $creators[0]; // 'John'
// children() with namespace
$entry = $xml->entry[0] ?? null;
if ($entry !== null) {
$dcChildren = $entry->children('http://purl.org/dc/elements/1.1/');
echo (string) $dcChildren->creator; // 'John'
echo (string) $dcChildren->date; // '2024-01-15'
}
// Get namespaces used in document
$namespaces = $xml->getNamespaces(true); // recursive = true
print_r($namespaces);
// ['http://www.w3.org/2005/Atom', 'dc' => 'http://purl.org/dc/elements/1.1/']
SimpleXML -- конвертация в массив
<?php
declare(strict_types=1);
// Convert SimpleXML to array (common need)
function xmlToArray(SimpleXMLElement $xml): array
{
$result = [];
// Add attributes
foreach ($xml->attributes() as $name => $value) {
$result['@' . $name] = (string) $value;
}
// Add children
foreach ($xml->children() as $name => $child) {
$childArray = xmlToArray($child);
if (isset($result[$name])) {
// Multiple children with same name -> make array
if (!isset($result[$name][0])) {
$result[$name] = [$result[$name]];
}
$result[$name][] = $childArray;
} else {
$result[$name] = $childArray;
}
}
// If no children, add text content
$text = trim((string) $xml);
if ($result === [] && $text !== '') {
return $text;
}
if ($text !== '') {
$result['#text'] = $text;
}
return $result;
}
// Quick and dirty (loses attributes):
$json = json_encode($xml);
$array = json_decode($json, true);
DOM -- полный контроль (W3C стандарт)
<?php
declare(strict_types=1);
$xmlString = <<<XML
<?xml version="1.0" encoding="UTF-8"?>
<catalog>
<book id="1">
<title>PHP 8 in Action</title>
<author>John Doe</author>
<price currency="USD">39.99</price>
</book>
<book id="2">
<title>Symfony Guide</title>
<author>Jane Smith</author>
<price currency="EUR">44.99</price>
</book>
</catalog>
XML;
// === Load XML ===
$dom = new DOMDocument('1.0', 'UTF-8');
$dom->preserveWhiteSpace = false;
$dom->formatOutput = true;
$loaded = $dom->loadXML($xmlString);
if (!$loaded) {
throw new RuntimeException('Failed to parse XML');
}
// Load from file
// $dom->load('/path/to/file.xml');
// === Access elements ===
$books = $dom->getElementsByTagName('book');
echo $books->length; // 2
foreach ($books as $book) {
/** @var DOMElement $book */
echo $book->getAttribute('id') . ': ';
echo $book->getElementsByTagName('title')->item(0)?->textContent . "\n";
}
// === getElementById (requires DTD or schema) ===
// Alternatively, use XPath for id-based lookup
// === DOMNodeList iteration ===
$titles = $dom->getElementsByTagName('title');
for ($i = 0; $i < $titles->length; $i++) {
$node = $titles->item($i);
echo $node->textContent . "\n";
}
// === Node properties ===
$firstBook = $books->item(0);
echo $firstBook->nodeName; // 'book'
echo $firstBook->nodeType; // XML_ELEMENT_NODE (1)
echo $firstBook->textContent; // All text content recursively
echo $firstBook->nodeValue; // null for elements
// Child nodes
echo $firstBook->childNodes->length; // Number of child nodes
echo $firstBook->firstChild->nodeName;
echo $firstBook->lastChild->nodeName;
// Parent and siblings
echo $firstBook->parentNode->nodeName; // 'catalog'
echo $firstBook->nextSibling?->nodeName; // 'book' (second book)
// Check node type
if ($firstBook->nodeType === XML_ELEMENT_NODE) {
echo "It's an element!\n";
}
// Has children/attributes
echo $firstBook->hasChildNodes() ? 'Yes' : 'No';
echo $firstBook->hasAttributes() ? 'Yes' : 'No';
DOMXPath -- мощные запросы
<?php
declare(strict_types=1);
$dom = new DOMDocument();
$dom->loadXML($xmlString);
$xpath = new DOMXPath($dom);
// === Basic queries ===
// Find all book titles
$titles = $xpath->query('//book/title');
foreach ($titles as $title) {
echo $title->textContent . "\n";
}
// Find books with specific attribute
$book1 = $xpath->query('//book[@id="1"]')->item(0);
echo $book1->getElementsByTagName('title')->item(0)->textContent;
// Find with conditions
$expensive = $xpath->query('//book[price > 40]');
$usdBooks = $xpath->query('//book[price/@currency="USD"]');
// === evaluate() — returns typed values ===
$count = $xpath->evaluate('count(//book)');
echo $count; // 2
$totalPrice = $xpath->evaluate('sum(//book/price)');
echo $totalPrice; // 84.98
$hasBooks = $xpath->evaluate('boolean(//book)');
echo $hasBooks ? 'Yes' : 'No'; // Yes
// Find text content directly
$firstTitle = $xpath->evaluate('string(//book[1]/title)');
echo $firstTitle; // 'PHP 8 in Action'
// === Context node ===
$firstBook = $dom->getElementsByTagName('book')->item(0);
$title = $xpath->query('title', $firstBook)->item(0);
echo $title->textContent; // 'PHP 8 in Action'
// === With namespaces ===
$xpath->registerNamespace('atom', 'http://www.w3.org/2005/Atom');
$entries = $xpath->query('//atom:entry');
PHP 8.4 -- новый DOM API
<?php
declare(strict_types=1);
// PHP 8.4+: Modern DOM API with spec-compliant behavior
// === Dom\XMLDocument — new XML parser ===
$doc = Dom\XMLDocument::createFromString($xmlString);
// querySelector / querySelectorAll (CSS selectors!)
$firstBook = $doc->querySelector('book');
echo $firstBook?->getAttribute('id'); // '1'
$allBooks = $doc->querySelectorAll('book');
foreach ($allBooks as $book) {
$title = $book->querySelector('title');
echo $title?->textContent . "\n";
}
// CSS selectors work like in browsers
$usdPrices = $doc->querySelectorAll('price[currency="USD"]');
$firstBookTitle = $doc->querySelector('book:first-child > title');
// Create from file
$doc = Dom\XMLDocument::createFromFile('/path/to/file.xml');
// === Dom\HTMLDocument — HTML5 parser ===
$html = '<div class="user"><h1>John</h1><p>Developer</p></div>';
$htmlDoc = Dom\HTMLDocument::createFromString($html);
$name = $htmlDoc->querySelector('.user h1');
echo $name?->textContent; // 'John'
// Parse full HTML pages
$htmlDoc = Dom\HTMLDocument::createFromFile('https://example.com');
$title = $htmlDoc->querySelector('title');
$links = $htmlDoc->querySelectorAll('a[href]');
foreach ($links as $link) {
echo $link->getAttribute('href') . "\n";
}
// innerHTML support
$div = $htmlDoc->querySelector('div');
echo $div?->innerHTML;
// Serialization
echo $doc->saveXML(); // XMLDocument
echo $htmlDoc->saveHTML(); // HTMLDocument
XMLReader -- потоковый парсинг больших файлов
<?php
declare(strict_types=1);
// XMLReader is a pull parser — reads XML node by node
// Memory efficient: doesn't load entire document into memory
// Perfect for large XML files (100MB+)
// === Basic usage ===
$reader = new XMLReader();
$reader->open('/path/to/large-catalog.xml');
// Or from string: $reader->XML($xmlString);
while ($reader->read()) {
// Process each node
if ($reader->nodeType === XMLReader::ELEMENT && $reader->name === 'book') {
echo "Book found!\n";
// Read attribute
$id = $reader->getAttribute('id');
echo "ID: $id\n";
}
if ($reader->nodeType === XMLReader::ELEMENT && $reader->name === 'title') {
// Move to text content
$reader->read();
echo "Title: {$reader->value}\n";
}
}
$reader->close();
// === Process specific elements ===
$reader = new XMLReader();
$reader->open('/path/to/products.xml');
$products = [];
while ($reader->read()) {
if ($reader->nodeType !== XMLReader::ELEMENT || $reader->name !== 'product') {
continue;
}
// expand() converts current node to DOMNode for easy access
$node = $reader->expand();
$product = [
'id' => $reader->getAttribute('id'),
'name' => $node->getElementsByTagName('name')->item(0)?->textContent,
'price' => (float) ($node->getElementsByTagName('price')->item(0)?->textContent ?? 0),
];
$products[] = $product;
// Process in batches to save memory
if (count($products) >= 1000) {
processBatch($products);
$products = [];
}
}
if ($products !== []) {
processBatch($products);
}
$reader->close();
function processBatch(array $products): void
{
// Save to database, process, etc.
echo sprintf("Processing batch of %d products\n", count($products));
}
// === moveToAttribute / moveToElement ===
$reader = new XMLReader();
$reader->XML('<items><item id="1" type="book" active="true"/></items>');
while ($reader->read()) {
if ($reader->nodeType === XMLReader::ELEMENT && $reader->name === 'item') {
// Iterate over attributes
if ($reader->hasAttributes) {
while ($reader->moveToNextAttribute()) {
echo "{$reader->name} = {$reader->value}\n";
}
$reader->moveToElement(); // Back to element
}
}
}
$reader->close();
// === Node types ===
// XMLReader::ELEMENT = 1 — opening tag
// XMLReader::END_ELEMENT = 15 — closing tag
// XMLReader::TEXT = 3 — text content
// XMLReader::CDATA = 4 — CDATA section
// XMLReader::COMMENT = 8 — comment
// XMLReader::ATTRIBUTE = 2 — attribute
Сравнение: SimpleXML vs DOM vs XMLReader
| Критерий | SimpleXML | DOM | XMLReader |
|---|---|---|---|
| Простота | Очень просто | Средне | Сложнее |
| Память | Весь документ в RAM | Весь документ в RAM | Потоковый, минимум RAM |
| Модификация | Ограниченная | Полная | Только чтение |
| XPath | Да | Да (DOMXPath) | Нет |
| CSS селекторы | Нет | Нет (PHP 8.4 -- да) | Нет |
| Namespaces | Да | Да | Да |
| Большие файлы | Нет | Нет | Да |
| Валидация | Нет | DTD, Schema | DTD, Schema |
| Когда использовать | Простой XML, конфиги, API | Сложный XML, модификация | Файлы > 10MB |
Практический пример: RSS/Atom парсер
<?php
declare(strict_types=1);
final readonly class FeedItem
{
public function __construct(
public string $title,
public string $link,
public string $description,
public ?DateTimeImmutable $pubDate,
public string $author,
) {}
}
function parseRssFeed(string $xmlContent): array
{
$xml = simplexml_load_string($xmlContent, SimpleXMLElement::class, LIBXML_NOCDATA);
if ($xml === false) {
throw new RuntimeException('Invalid RSS XML');
}
$items = [];
// RSS 2.0 format
foreach ($xml->channel->item as $item) {
$items[] = new FeedItem(
title: (string) $item->title,
link: (string) $item->link,
description: (string) $item->description,
pubDate: $item->pubDate
? new DateTimeImmutable((string) $item->pubDate)
: null,
author: (string) ($item->author ?? $item->children('dc', true)->creator ?? ''),
);
}
return $items;
}
function parseAtomFeed(string $xmlContent): array
{
$xml = simplexml_load_string($xmlContent);
if ($xml === false) {
throw new RuntimeException('Invalid Atom XML');
}
$xml->registerXPathNamespace('atom', 'http://www.w3.org/2005/Atom');
$items = [];
foreach ($xml->entry as $entry) {
$link = '';
foreach ($entry->link as $l) {
if ((string) $l['rel'] === 'alternate' || (string) $l['rel'] === '') {
$link = (string) $l['href'];
break;
}
}
$items[] = new FeedItem(
title: (string) $entry->title,
link: $link,
description: (string) ($entry->summary ?? $entry->content ?? ''),
pubDate: $entry->updated
? new DateTimeImmutable((string) $entry->updated)
: null,
author: (string) ($entry->author->name ?? ''),
);
}
return $items;
}
// Auto-detect feed type
function parseFeed(string $xmlContent): array
{
$xml = simplexml_load_string($xmlContent);
if ($xml === false) {
throw new RuntimeException('Invalid XML');
}
return match ($xml->getName()) {
'rss' => parseRssFeed($xmlContent),
'feed' => parseAtomFeed($xmlContent),
default => throw new RuntimeException("Unknown feed format: {$xml->getName()}"),
};
}
Практический пример: SOAP ответ
<?php
declare(strict_types=1);
function parseSoapResponse(string $soapXml): array
{
$dom = new DOMDocument();
$dom->loadXML($soapXml);
$xpath = new DOMXPath($dom);
// Register SOAP namespaces
$xpath->registerNamespace('soap', 'http://schemas.xmlsoap.org/soap/envelope/');
$xpath->registerNamespace('ns', 'http://example.com/webservice');
// Check for SOAP fault
$fault = $xpath->query('//soap:Fault');
if ($fault->length > 0) {
$faultCode = $xpath->evaluate('string(//soap:Fault/faultcode)');
$faultString = $xpath->evaluate('string(//soap:Fault/faultstring)');
throw new RuntimeException("SOAP Fault [$faultCode]: $faultString");
}
// Extract response data
$body = $xpath->query('//soap:Body/*');
if ($body->length === 0) {
throw new RuntimeException('Empty SOAP response body');
}
// Convert body to SimpleXML for easier access
$bodyXml = simplexml_import_dom($body->item(0));
return xmlToArray($bodyXml);
}