| @@ 3679-3685 (lines=7) @@ | ||
| 3676 | if (0 === (0x80 & $in)) { |
|
| 3677 | // US-ASCII, pass straight through. |
|
| 3678 | $mBytes = 1; |
|
| 3679 | } elseif (0xC0 === (0xE0 & $in)) { |
|
| 3680 | // First octet of 2 octet sequence. |
|
| 3681 | $mUcs4 = $in; |
|
| 3682 | $mUcs4 = ($mUcs4 & 0x1F) << 6; |
|
| 3683 | $mState = 1; |
|
| 3684 | $mBytes = 2; |
|
| 3685 | } elseif (0xE0 === (0xF0 & $in)) { |
|
| 3686 | // First octet of 3 octet sequence. |
|
| 3687 | $mUcs4 = $in; |
|
| 3688 | $mUcs4 = ($mUcs4 & 0x0F) << 12; |
|
| @@ 3691-3697 (lines=7) @@ | ||
| 3688 | $mUcs4 = ($mUcs4 & 0x0F) << 12; |
|
| 3689 | $mState = 2; |
|
| 3690 | $mBytes = 3; |
|
| 3691 | } elseif (0xF0 === (0xF8 & $in)) { |
|
| 3692 | // First octet of 4 octet sequence. |
|
| 3693 | $mUcs4 = $in; |
|
| 3694 | $mUcs4 = ($mUcs4 & 0x07) << 18; |
|
| 3695 | $mState = 3; |
|
| 3696 | $mBytes = 4; |
|
| 3697 | } elseif (0xF8 === (0xFC & $in)) { |
|
| 3698 | /* First octet of 5 octet sequence. |
|
| 3699 | * |
|
| 3700 | * This is illegal because the encoded codepoint must be either |
|
| @@ 3710-3716 (lines=7) @@ | ||
| 3707 | $mUcs4 = ($mUcs4 & 0x03) << 24; |
|
| 3708 | $mState = 4; |
|
| 3709 | $mBytes = 5; |
|
| 3710 | } elseif (0xFC === (0xFE & $in)) { |
|
| 3711 | // First octet of 6 octet sequence, see comments for 5 octet sequence. |
|
| 3712 | $mUcs4 = $in; |
|
| 3713 | $mUcs4 = ($mUcs4 & 1) << 30; |
|
| 3714 | $mState = 5; |
|
| 3715 | $mBytes = 6; |
|
| 3716 | } else { |
|
| 3717 | /* Current octet is neither in the US-ASCII range nor a legal first |
|
| 3718 | * octet of a multi-octet sequence. |
|
| 3719 | */ |
|