Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
212 changes: 212 additions & 0 deletions src/wp-includes/class-wp-encoding.php
Original file line number Diff line number Diff line change
@@ -0,0 +1,212 @@
<?php

/**
* WP_Encoding class for interacting with character encoding names and byte streams,
* according to the WHATWG Encoding specification.
*
* @link https://encoding.spec.whatwg.org/
*
* @since 7.2.0
*
* @group encoding
*/
final class WP_Encoding {
private static ?bool $mb_present = null;

/**
* Returns a UTF-8 string loaded from an encoded byte stream.
*
* @see self::export_from_utf8() for exporting from UTF-8.
*
* @since 7.2.0
*
* @param string $source_encoding
* @param string $encoded_bytes
* @return string|null
*/
public static function load_into_utf8( string $source_encoding, string $encoded_bytes ): ?string {
$encoding = self::encoding_from_label( $source_encoding );
if ( null === $encoding ) {
return null;
}

if ( 'replacement' === $encoding ) {
return '';
}

self::$mb_present ??= \function_exists( '\mb_convert_encoding' );

/*
* In a future expansion, when the decoders from the WHATWG spec are
* introduced, this will return according to those fallbacks. For now,
* without the `mbstring` extension, this will fail.
*/
if ( ! self::$mb_present ) {
return null;
}

try {
return \mb_convert_encoding( $encoded_bytes, $encoding, 'UTF-8' );
} catch ( \Throwable $e ) {
return null;
}
}

/**
* Returns an encoded byte stream from a source UTF-8 string.
*
* @see self::load_into_utf8() for importing into UTF-8.
*
* @since 7.2.0
*
* @param string $destination_encoding
* @param string $utf8_bytes
* @return string|null
*/
public static function export_from_utf8( string $destination_encoding, string $utf8_bytes ): ?string {
$encoding = self::encoding_from_label( $destination_encoding );
if ( null === $encoding || 'replacement' === $encoding ) {
return null;
}

self::$mb_present ??= \function_exists( '\mb_convert_encoding' );

/*
* In a future expansion, when the encoders from the WHATWG spec are
* introduced, this will return according to those fallbacks. For now,
* without the `mbstring` extension, this will fail.
*/
if ( ! self::$mb_present ) {
return null;
}

try {
return \mb_convert_encoding( $utf8_bytes, 'UTF-8', $encoding );
} catch ( \Throwable $e ) {
return null;
}
}

/**
* Canonicalizes a character-encoding label into a supported encoding name.
*
* Note! “Supported” does not mean that there’s a way to definitive way to
* convert the encoding; it means that the encoding is not rejected
* by WordPress as unsupported. Actual support currently depends on
* the presence of the `mbstring` extension with support built-in
* for the returned encoding.
*
* The `replacement` encoding implies that a decoder always returns the empty
* string. This catches encodings which pose security risks and should never
* be attempted to be decoded. The `x-user-defined` is a special form of
* decoding byte streams into UTF-8 while preserving the source bytes.
*
* Example:
*
* // Common labels map to well-known encoding names.
* 'UTF-8' === WP_Encoding::encoding_from_label( 'utf8' );
* 'windows-1252' === WP_Encoding::encoding_from_label( 'ISO-8859-1' );
* 'windows-1252' === WP_Encoding::encoding_from_label( 'cp1252' );
*
* // Leading and trailing whitespace is trimmed; matches are ASCII-case-insensitive.
* 'UTF-8' === WP_Encoding::encoding_from_label( 'utf8' );
* 'UTF-8' === WP_Encoding::encoding_from_label( 'UTF8' );
* 'UTF-8' === WP_Encoding::encoding_from_label( " utf-8\t" );
*
* // For security purposes, some labels map to a rejected pseudo-encoding.
* 'replacement' === WP_Encoding::encoding_from_label( 'iso-2022-cn' );
*
* // Unknown or unrecognized labels are rejected entirely.
* null === WP_Encoding::encoding_from_label( 'utf-7' );
* null === WP_Encoding::encoding_from_label( 'UTF-8; latin1' );
* null === WP_Encoding::encoding_from_label( '<meta charset="utf-8">' );
*
* @see self::load_into_utf8() for importing into UTF-8 from these encodings.
* @see self::export_from_utf8() for exporting from UTF-8 into these encodings.
*
* @link https://encoding.spec.whatwg.org/#names-and-labels
*
* @since 7.2.0
*
* @param string $label Candidate encoding name, e.g. "latin1" or "UTF-8".
* @return string|'replacment'|null Canonical name of encoding, if supported, else `null`.
* `'replacement'` means that the decode should always be
* the empty string, as a security precaution.
*/
public static function encoding_from_label( string $label ): ?string {
/*
* > To get an encoding from a string label, run these steps:
* > 1. Remove any leading and trailing ASCII whitespace from label.
* > 2. If label is an ASCII case-insensitive match for any of the labels listed in the table below,
* > then return the corresponding encoding; otherwise return failure.
*/
$label = \trim( $label, " \t\f\r\n" );
$label = \strtolower( $label );

/*
* Pre-filter on the most-common reported names to avoid additional processing.
* A survey of 231,566 root-domain pages from the top ten-million visited sites
* revealed that 13,402 (5.8%) reported a charset. Among those, these case-sensitive
* labels accounted for the overwhelming majority of reports:
*
* | Label | Count | % |
* | UTF-8 | 9,295 | 69 % |
* | iso-8859-1 | 2,150 | 16 % |
* | windows-1252 | 663 | 4.9% |
* | us-ascii | 470 | 3.5% |
* | windows-1251 | 162 | 1.2% |
*
* These account for 95% of reported charset labels, comprising only three distinct
* supported charsets. This is worth the quick-pass even though it induces a tiny
* amount of re-computation in the case that none of these match.
*/
if ( 'utf-8' === $label ) {
return 'UTF-8';
}

if ( 'iso-8859-1' === $label || 'windows-1252' === $label || 'us-ascii' === $label ) {
return 'windows-1252';
}

if ( 'windows-1251' === $label ) {
return 'windows-1251';
}

/*
* If whitespace exists within the label, it’s probably the accidental concatenation
* of multiple labels and no disambiguation is possible. If a semicolon is trapped
* in the name then this probably comes from mis-parsing of a `charset` property on
* a MIME content-type.
*/
if ( \strlen( $label ) !== \strcspn( $label, " \t\f\r\n;" ) ) {
return null;
}

$label = " {$label} ";

/**
* Mapping from encoding name to space-separated list of its labels.
*
* > The table below lists all encodings and their labels user agents must
* > support. User agents must not support any other encodings or labels.
*
* Every label will be surrounded on each side by spaces, as labels containing
* spaces are rejected. This allows for efficient string lookup.
*
* @see \generate_charset_labels_table()
*
* @since 7.2.0
*
* @type $table non-empty-array<non-empty-string, non-empty-string>
*/
$table = require __DIR__ . '/wp-supported-encodings.php';

foreach ( $table as $name => $labels ) {
if ( \str_contains( $labels, $label ) ) {
return $name;
}
}

return null;
}
}
77 changes: 77 additions & 0 deletions src/wp-includes/wp-supported-encodings.php
Original file line number Diff line number Diff line change
@@ -0,0 +1,77 @@
<?php

/**
* Auto-generated class for mapping character set labels to supported names.
*
* ⚠️ !!! THIS ENTIRE FILE IS AUTOMATICALLY GENERATED !!! ⚠️
* Do not modify this file directly.
*
* To regenerate, run the generation script directly.
*
* Example:
*
* php tests/phpunit/data/encoding/generate-labels-table.php
*
* @package WordPress
* @since 7.2.0
*/

// phpcs:disable

/**
* Mapping from encoding name to space-separated list of its labels.
*
* > The table below lists all encodings and their labels user agents must
* > support. User agents must not support any other encodings or labels.
*
* Every label will be surrounded on each side by spaces, as labels containing
* spaces are rejected. This allows for efficient string lookup.
*
* @see \generate_charset_labels_table()
*
* @since 7.2.0
*
* @type array<non-empty-string, non-empty-string>
*/
return array(
'UTF-8' => ' unicode-1-1-utf-8 unicode11utf8 unicode20utf8 utf-8 utf8 x-unicode20utf8 ',
'IBM866' => ' 866 cp866 csibm866 ibm866 ',
'ISO-8859-2' => ' csisolatin2 iso-8859-2 iso-ir-101 iso8859-2 iso88592 iso_8859-2 iso_8859-2:1987 l2 latin2 ',
'ISO-8859-3' => ' csisolatin3 iso-8859-3 iso-ir-109 iso8859-3 iso88593 iso_8859-3 iso_8859-3:1988 l3 latin3 ',
'ISO-8859-4' => ' csisolatin4 iso-8859-4 iso-ir-110 iso8859-4 iso88594 iso_8859-4 iso_8859-4:1988 l4 latin4 ',
'ISO-8859-5' => ' csisolatincyrillic cyrillic iso-8859-5 iso-ir-144 iso8859-5 iso88595 iso_8859-5 iso_8859-5:1988 ',
'ISO-8859-6' => ' arabic asmo-708 csiso88596e csiso88596i csisolatinarabic ecma-114 iso-8859-6 iso-8859-6-e iso-8859-6-i iso-ir-127 iso8859-6 iso88596 iso_8859-6 iso_8859-6:1987 ',
'ISO-8859-7' => ' csisolatingreek ecma-118 elot_928 greek greek8 iso-8859-7 iso-ir-126 iso8859-7 iso88597 iso_8859-7 iso_8859-7:1987 sun_eu_greek ',
'ISO-8859-8' => ' csiso88598e csisolatinhebrew hebrew iso-8859-8 iso-8859-8-e iso-ir-138 iso8859-8 iso88598 iso_8859-8 iso_8859-8:1988 visual ',
'ISO-8859-8-I' => ' csiso88598i iso-8859-8-i logical ',
'ISO-8859-10' => ' csisolatin6 iso-8859-10 iso-ir-157 iso8859-10 iso885910 l6 latin6 ',
'ISO-8859-13' => ' iso-8859-13 iso8859-13 iso885913 ',
'ISO-8859-14' => ' iso-8859-14 iso8859-14 iso885914 ',
'ISO-8859-15' => ' csisolatin9 iso-8859-15 iso8859-15 iso885915 iso_8859-15 l9 ',
'ISO-8859-16' => ' iso-8859-16 ',
'KOI8-R' => ' cskoi8r koi koi8 koi8-r koi8_r ',
'KOI8-U' => ' koi8-ru koi8-u ',
'macintosh' => ' csmacintosh mac macintosh x-mac-roman ',
'windows-874' => ' dos-874 iso-8859-11 iso8859-11 iso885911 tis-620 windows-874 ',
'windows-1250' => ' cp1250 windows-1250 x-cp1250 ',
'windows-1251' => ' cp1251 windows-1251 x-cp1251 ',
'windows-1252' => ' ansi_x3.4-1968 ascii cp1252 cp819 csisolatin1 ibm819 iso-8859-1 iso-ir-100 iso8859-1 iso88591 iso_8859-1 iso_8859-1:1987 l1 latin1 us-ascii windows-1252 x-cp1252 ',
'windows-1253' => ' cp1253 windows-1253 x-cp1253 ',
'windows-1254' => ' cp1254 csisolatin5 iso-8859-9 iso-ir-148 iso8859-9 iso88599 iso_8859-9 iso_8859-9:1989 l5 latin5 windows-1254 x-cp1254 ',
'windows-1255' => ' cp1255 windows-1255 x-cp1255 ',
'windows-1256' => ' cp1256 windows-1256 x-cp1256 ',
'windows-1257' => ' cp1257 windows-1257 x-cp1257 ',
'windows-1258' => ' cp1258 windows-1258 x-cp1258 ',
'x-mac-cyrillic' => ' x-mac-cyrillic x-mac-ukrainian ',
'GBK' => ' chinese csgb2312 csiso58gb231280 gb2312 gb_2312 gb_2312-80 gbk iso-ir-58 x-gbk ',
'gb18030' => ' gb18030 ',
'Big5' => ' big5 big5-hkscs cn-big5 csbig5 x-x-big5 ',
'EUC-JP' => ' cseucpkdfmtjapanese euc-jp x-euc-jp ',
'ISO-2022-JP' => ' csiso2022jp iso-2022-jp ',
'Shift_JIS' => ' csshiftjis ms932 ms_kanji shift-jis shift_jis sjis windows-31j x-sjis ',
'EUC-KR' => ' cseuckr csksc56011987 euc-kr iso-ir-149 korean ks_c_5601-1987 ks_c_5601-1989 ksc5601 ksc_5601 windows-949 ',
'replacement' => ' csiso2022kr hz-gb-2312 iso-2022-cn iso-2022-cn-ext iso-2022-kr replacement ',
'UTF-16BE' => ' unicodefffe utf-16be ',
'UTF-16LE' => ' csunicode iso-10646-ucs-2 ucs-2 unicode unicodefeff utf-16 utf-16le ',
'x-user-defined' => ' x-user-defined ',
);
21 changes: 21 additions & 0 deletions tests/phpunit/data/encoding/README.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,21 @@
# WHATWG Encodings and Labels

This directory contains the listing of encoding labels and their associated encoding name.

- https://encoding.spec.whatwg.org/#names-and-labels

The non-normative [`encodings.json`](https://encoding.spec.whatwg.org/encodings.json) file comes from the WHATWG server, and
is cached here in the test directory so that it doesn't need to be constantly re-downloaded.

## Updating the optimized lookup class.

The [`class-wp-encodings.php`][1] file contains an optimized lookup map for the names and lables in `encodings.json`.
Run the [`generate-labels-table.php`][2] file to update the auto-generated Core module.

```bash
~$ php tests/phpunit/data/html5-entities/generate-html5-named-character-references.php
OK: Successfully generated optimized lookup class.
```

[1]: ../../../../src/wp-includes/class-wp-encodings.php
[2]: generate-labels-table.php
Loading
Loading