| @@ 2727-2733 (lines=7) @@ | ||
| 2724 | if (0 === (0x80 & $in)) { |
|
| 2725 | // US-ASCII, pass straight through. |
|
| 2726 | $mBytes = 1; |
|
| 2727 | } elseif (0xC0 === (0xE0 & $in)) { |
|
| 2728 | // First octet of 2 octet sequence. |
|
| 2729 | $mUcs4 = $in; |
|
| 2730 | $mUcs4 = ($mUcs4 & 0x1F) << 6; |
|
| 2731 | $mState = 1; |
|
| 2732 | $mBytes = 2; |
|
| 2733 | } elseif (0xE0 === (0xF0 & $in)) { |
|
| 2734 | // First octet of 3 octet sequence. |
|
| 2735 | $mUcs4 = $in; |
|
| 2736 | $mUcs4 = ($mUcs4 & 0x0F) << 12; |
|
| @@ 2739-2745 (lines=7) @@ | ||
| 2736 | $mUcs4 = ($mUcs4 & 0x0F) << 12; |
|
| 2737 | $mState = 2; |
|
| 2738 | $mBytes = 3; |
|
| 2739 | } elseif (0xF0 === (0xF8 & $in)) { |
|
| 2740 | // First octet of 4 octet sequence. |
|
| 2741 | $mUcs4 = $in; |
|
| 2742 | $mUcs4 = ($mUcs4 & 0x07) << 18; |
|
| 2743 | $mState = 3; |
|
| 2744 | $mBytes = 4; |
|
| 2745 | } elseif (0xF8 === (0xFC & $in)) { |
|
| 2746 | /* First octet of 5 octet sequence. |
|
| 2747 | * |
|
| 2748 | * This is illegal because the encoded codepoint must be either |
|
| @@ 2758-2764 (lines=7) @@ | ||
| 2755 | $mUcs4 = ($mUcs4 & 0x03) << 24; |
|
| 2756 | $mState = 4; |
|
| 2757 | $mBytes = 5; |
|
| 2758 | } elseif (0xFC === (0xFE & $in)) { |
|
| 2759 | // First octet of 6 octet sequence, see comments for 5 octet sequence. |
|
| 2760 | $mUcs4 = $in; |
|
| 2761 | $mUcs4 = ($mUcs4 & 1) << 30; |
|
| 2762 | $mState = 5; |
|
| 2763 | $mBytes = 6; |
|
| 2764 | } else { |
|
| 2765 | /* Current octet is neither in the US-ASCII range nor a legal first |
|
| 2766 | * octet of a multi-octet sequence. |
|
| 2767 | */ |
|