diff options
| author | Simon Sapin <simon.sapin@exyr.org> | 2015-08-13 18:39:46 +0200 |
|---|---|---|
| committer | Simon Sapin <simon.sapin@exyr.org> | 2015-08-23 00:28:56 +0200 |
| commit | 6174b8d726ed5764694e5404329d8b5e66517ed5 (patch) | |
| tree | 47dd9d787f7550a5d47301652fc5081446f6495e /src/librustc_unicode | |
| parent | c408b7863389aa2bdb253ffa363e693bcd02439f (diff) | |
Refactor low-level UTF-16 decoding.
* Rename `utf16_items` to `decode_utf16`. "Items" is meaningless. * Move it to `rustc_unicode::char`, exposed in `std::char`. * Generalize it to any `u16` iterable, not just `&[u16]`. * Make it yield `Result` instead of a custom `Utf16Item` enum that was isomorphic to `Result`. This enable using the `FromIterator for Result` impl. * Add a `REPLACEMENT_CHARACTER` constant. * Document how `result.unwrap_or(REPLACEMENT_CHARACTER)` replaces `Utf16Item::to_char_lossy`.
Diffstat (limited to 'src/librustc_unicode')
| -rw-r--r-- | src/librustc_unicode/char.rs | 113 | ||||
| -rw-r--r-- | src/librustc_unicode/lib.rs | 1 | ||||
| -rw-r--r-- | src/librustc_unicode/u_str.rs | 63 |
3 files changed, 138 insertions, 39 deletions
diff --git a/src/librustc_unicode/char.rs b/src/librustc_unicode/char.rs index 780f8aa5be9..e08b3244109 100644 --- a/src/librustc_unicode/char.rs +++ b/src/librustc_unicode/char.rs @@ -503,3 +503,116 @@ impl char { ToUppercase(CaseMappingIter::new(conversions::to_upper(self))) } } + +/// An iterator that decodes UTF-16 encoded codepoints from an iterator of `u16`s. +#[unstable(feature = "decode_utf16", reason = "recently exposed", issue = "27830")] +#[derive(Clone)] +pub struct DecodeUtf16<I> where I: Iterator<Item=u16> { + iter: I, + buf: Option<u16>, +} + +/// Create an iterator over the UTF-16 encoded codepoints in `iterable`, +/// returning unpaired surrogates as `Err`s. +/// +/// # Examples +/// +/// ``` +/// #![feature(decode_utf16)] +/// +/// use std::char::decode_utf16; +/// +/// fn main() { +/// // 𝄞mus<invalid>ic<invalid> +/// let v = [0xD834, 0xDD1E, 0x006d, 0x0075, +/// 0x0073, 0xDD1E, 0x0069, 0x0063, +/// 0xD834]; +/// +/// assert_eq!(decode_utf16(v.iter().cloned()).collect::<Vec<_>>(), +/// vec![Ok('𝄞'), +/// Ok('m'), Ok('u'), Ok('s'), +/// Err(0xDD1E), +/// Ok('i'), Ok('c'), +/// Err(0xD834)]); +/// } +/// ``` +/// +/// A lossy decoder can be obtained by replacing `Err` results with the replacement character: +/// +/// ``` +/// #![feature(decode_utf16)] +/// +/// use std::char::{decode_utf16, REPLACEMENT_CHARACTER}; +/// +/// fn main() { +/// // 𝄞mus<invalid>ic<invalid> +/// let v = [0xD834, 0xDD1E, 0x006d, 0x0075, +/// 0x0073, 0xDD1E, 0x0069, 0x0063, +/// 0xD834]; +/// +/// assert_eq!(decode_utf16(v.iter().cloned()) +/// .map(|r| r.unwrap_or(REPLACEMENT_CHARACTER)) +/// .collect::<String>(), +/// "𝄞mus�ic�"); +/// } +/// ``` +#[unstable(feature = "decode_utf16", reason = "recently exposed", issue = "27830")] +#[inline] +pub fn decode_utf16<I: IntoIterator<Item=u16>>(iterable: I) -> DecodeUtf16<I::IntoIter> { + DecodeUtf16 { + iter: iterable.into_iter(), + buf: None, + } +} + +#[unstable(feature = "decode_utf16", reason = "recently exposed", issue = "27830")] +impl<I: Iterator<Item=u16>> Iterator for DecodeUtf16<I> { + type Item = Result<char, u16>; + + fn next(&mut self) -> Option<Result<char, u16>> { + let u = match self.buf.take() { + Some(buf) => buf, + None => match self.iter.next() { + Some(u) => u, + None => return None + } + }; + + if u < 0xD800 || 0xDFFF < u { + // not a surrogate + Some(Ok(unsafe { from_u32_unchecked(u as u32) })) + } else if u >= 0xDC00 { + // a trailing surrogate + Some(Err(u)) + } else { + let u2 = match self.iter.next() { + Some(u2) => u2, + // eof + None => return Some(Err(u)) + }; + if u2 < 0xDC00 || u2 > 0xDFFF { + // not a trailing surrogate so we're not a valid + // surrogate pair, so rewind to redecode u2 next time. + self.buf = Some(u2); + return Some(Err(u)) + } + + // all ok, so lets decode it. + let c = (((u - 0xD800) as u32) << 10 | (u2 - 0xDC00) as u32) + 0x1_0000; + Some(Ok(unsafe { from_u32_unchecked(c) })) + } + } + + #[inline] + fn size_hint(&self) -> (usize, Option<usize>) { + let (low, high) = self.iter.size_hint(); + // we could be entirely valid surrogates (2 elements per + // char), or entirely non-surrogates (1 element per char) + (low / 2, high) + } +} + +/// U+FFFD REPLACEMENT CHARACTER (�) is used in Unicode to represent a decoding error. +/// It can occur, for example, when giving ill-formed UTF-8 bytes to `String::from_utf8_lossy`. +#[unstable(feature = "decode_utf16", reason = "recently added", issue = "27830")] +pub const REPLACEMENT_CHARACTER: char = '\u{FFFD}'; diff --git a/src/librustc_unicode/lib.rs b/src/librustc_unicode/lib.rs index d046393cdeb..4f0aa69d771 100644 --- a/src/librustc_unicode/lib.rs +++ b/src/librustc_unicode/lib.rs @@ -46,6 +46,7 @@ mod tables; mod u_str; pub mod char; +#[allow(deprecated)] pub mod str { pub use u_str::{UnicodeStr, SplitWhitespace}; pub use u_str::{utf8_char_width, is_utf16, Utf16Items, Utf16Item}; diff --git a/src/librustc_unicode/u_str.rs b/src/librustc_unicode/u_str.rs index f6e6ac508a7..67333c98fcf 100644 --- a/src/librustc_unicode/u_str.rs +++ b/src/librustc_unicode/u_str.rs @@ -13,8 +13,9 @@ //! This module provides functionality to `str` that requires the Unicode methods provided by the //! unicode parts of the CharExt trait. +use char::{DecodeUtf16, decode_utf16}; use core::char; -use core::iter::Filter; +use core::iter::{Cloned, Filter}; use core::slice; use core::str::Split; @@ -119,11 +120,18 @@ pub fn is_utf16(v: &[u16]) -> bool { /// An iterator that decodes UTF-16 encoded codepoints from a vector /// of `u16`s. +#[deprecated(since = "1.4.0", reason = "renamed to `char::DecodeUtf16`")] +#[unstable(feature = "decode_utf16", reason = "not exposed in std", issue = "27830")] +#[allow(deprecated)] #[derive(Clone)] pub struct Utf16Items<'a> { - iter: slice::Iter<'a, u16> + decoder: DecodeUtf16<Cloned<slice::Iter<'a, u16>>> } + /// The possibilities for values decoded from a `u16` stream. +#[deprecated(since = "1.4.0", reason = "`char::DecodeUtf16` uses `Result<char, u16>` instead")] +#[unstable(feature = "decode_utf16", reason = "not exposed in std", issue = "27830")] +#[allow(deprecated)] #[derive(Copy, PartialEq, Eq, Clone, Debug)] pub enum Utf16Item { /// A valid codepoint. @@ -132,6 +140,7 @@ pub enum Utf16Item { LoneSurrogate(u16) } +#[allow(deprecated)] impl Utf16Item { /// Convert `self` to a `char`, taking `LoneSurrogate`s to the /// replacement character (U+FFFD). @@ -144,49 +153,22 @@ impl Utf16Item { } } +#[deprecated(since = "1.4.0", reason = "use `char::DecodeUtf16` instead")] +#[unstable(feature = "decode_utf16", reason = "not exposed in std", issue = "27830")] +#[allow(deprecated)] impl<'a> Iterator for Utf16Items<'a> { type Item = Utf16Item; fn next(&mut self) -> Option<Utf16Item> { - let u = match self.iter.next() { - Some(u) => *u, - None => return None - }; - - if u < 0xD800 || 0xDFFF < u { - // not a surrogate - Some(Utf16Item::ScalarValue(unsafe { char::from_u32_unchecked(u as u32) })) - } else if u >= 0xDC00 { - // a trailing surrogate - Some(Utf16Item::LoneSurrogate(u)) - } else { - // preserve state for rewinding. - let old = self.iter.clone(); - - let u2 = match self.iter.next() { - Some(u2) => *u2, - // eof - None => return Some(Utf16Item::LoneSurrogate(u)) - }; - if u2 < 0xDC00 || u2 > 0xDFFF { - // not a trailing surrogate so we're not a valid - // surrogate pair, so rewind to redecode u2 next time. - self.iter = old.clone(); - return Some(Utf16Item::LoneSurrogate(u)) - } - - // all ok, so lets decode it. - let c = (((u - 0xD800) as u32) << 10 | (u2 - 0xDC00) as u32) + 0x1_0000; - Some(Utf16Item::ScalarValue(unsafe { char::from_u32_unchecked(c) })) - } + self.decoder.next().map(|result| match result { + Ok(c) => Utf16Item::ScalarValue(c), + Err(s) => Utf16Item::LoneSurrogate(s), + }) } #[inline] fn size_hint(&self) -> (usize, Option<usize>) { - let (low, high) = self.iter.size_hint(); - // we could be entirely valid surrogates (2 elements per - // char), or entirely non-surrogates (1 element per char) - (low / 2, high) + self.decoder.size_hint() } } @@ -196,7 +178,7 @@ impl<'a> Iterator for Utf16Items<'a> { /// # Examples /// /// ``` -/// #![feature(unicode)] +/// #![feature(unicode, decode_utf16)] /// /// extern crate rustc_unicode; /// @@ -216,8 +198,11 @@ impl<'a> Iterator for Utf16Items<'a> { /// LoneSurrogate(0xD834)]); /// } /// ``` +#[deprecated(since = "1.4.0", reason = "renamed to `char::decode_utf16`")] +#[unstable(feature = "decode_utf16", reason = "not exposed in std", issue = "27830")] +#[allow(deprecated)] pub fn utf16_items<'a>(v: &'a [u16]) -> Utf16Items<'a> { - Utf16Items { iter : v.iter() } + Utf16Items { decoder: decode_utf16(v.iter().cloned()) } } /// Iterator adaptor for encoding `char`s to UTF-16. |
