usecrate::parser::*; usecrate::subtags::Subtag; usecrate::{extensions, subtags, LanguageIdentifier}; #[cfg(feature = "alloc")] use alloc::borrow::Cow; use core::cmp::Ordering; #[cfg(feature = "alloc")] use core::str::FromStr;
/// A core struct representing a [`Unicode Locale Identifier`]. /// /// A locale is made of two parts: /// * Unicode Language Identifier /// * A set of Unicode Extensions /// /// [`Locale`] exposes all of the same fields and methods as [`LanguageIdentifier`], and /// on top of that is able to parse, manipulate and serialize unicode extension fields. /// /// # Ordering /// /// This type deliberately does not implement `Ord` or `PartialOrd` because there are /// multiple possible orderings. Depending on your use case, two orderings are available: /// /// 1. A string ordering, suitable for stable serialization: [`Locale::strict_cmp`] /// 2. A struct ordering, suitable for use with a BTreeSet: [`Locale::total_cmp`] /// /// See issue: <https://github.com/unicode-org/icu4x/issues/1215> /// /// # Parsing /// /// Unicode recognizes three levels of standard conformance for a locale: /// /// * *well-formed* - syntactically correct /// * *valid* - well-formed and only uses registered language subtags, extensions, keywords, types... /// * *canonical* - valid and no deprecated codes or structure. /// /// Any syntactically invalid subtags will cause the parsing to fail with an error. /// /// This operation normalizes syntax to be well-formed. No legacy subtag replacements is performed. /// For validation and canonicalization, see `LocaleCanonicalizer`. /// /// ICU4X's Locale parsing does not allow for non-BCP-47-compatible locales [allowed by UTS 35 for backwards compatability][tr35-bcp]. /// Furthermore, it currently does not allow for language tags to have more than three characters. /// /// # Serde /// /// This type implements `serde::Serialize` and `serde::Deserialize` if the /// `"serde"` Cargo feature is enabled on the crate. /// /// The value will be serialized as a string and parsed when deserialized. /// For tips on efficient storage and retrieval of locales, see [`crate::zerovec`]. /// /// # Examples /// /// Simple example: /// /// ``` /// use icu::locale::{ /// extensions::unicode::{key, value}, /// locale, /// subtags::{language, region}, /// }; /// /// let loc = locale!("en-US-u-ca-buddhist"); /// /// assert_eq!(loc.id.language, language!("en")); /// assert_eq!(loc.id.script, None); /// assert_eq!(loc.id.region, Some(region!("US"))); /// assert_eq!(loc.id.variants.len(), 0); /// assert_eq!( /// loc.extensions.unicode.keywords.get(&key!("ca")), /// Some(&value!("buddhist")) /// ); /// ``` /// /// More complex example: /// /// ``` /// use icu::locale::{subtags::*, Locale}; /// /// let loc: Locale = "eN-latn-Us-Valencia-u-hC-H12" /// .parse() /// .expect("Failed to parse."); /// /// assert_eq!(loc.id.language, "en".parse::<Language>().unwrap()); /// assert_eq!(loc.id.script, "Latn".parse::<Script>().ok()); /// assert_eq!(loc.id.region, "US".parse::<Region>().ok()); /// assert_eq!( /// loc.id.variants.first(), /// "valencia".parse::<Variant>().ok().as_ref() /// ); /// ``` /// /// [`Unicode Locale Identifier`]: https://unicode.org/reports/tr35/tr35.html#Unicode_locale_identifier /// [tr35-bcp]: https://unicode.org/reports/tr35/#BCP_47_Conformance #[derive(PartialEq, Eq, Clone, Hash)] // no Ord or PartialOrd: see docs #[allow(clippy::exhaustive_structs)] // This struct is stable (and invoked by a macro) pubstruct Locale { /// The basic language/script/region components in the locale identifier along with any variants. pub id: LanguageIdentifier, /// Any extensions present in the locale identifier. pub extensions: extensions::Extensions,
}
#[test] // Expected sizes are based on a 64-bit architecture #[cfg(target_pointer_width = "64")] fn test_sizes() {
assert_eq!(core::mem::size_of::<subtags::Language>(), 3);
assert_eq!(core::mem::size_of::<subtags::Script>(), 4);
assert_eq!(core::mem::size_of::<subtags::Region>(), 3);
assert_eq!(core::mem::size_of::<subtags::Variant>(), 8);
assert_eq!(core::mem::size_of::<subtags::Variants>(), 16);
assert_eq!(core::mem::size_of::<LanguageIdentifier>(), 32);
/// A constructor which takes a utf8 slice, parses it and /// produces a well-formed [`Locale`]. /// /// ✨ *Enabled with the `alloc` Cargo feature.* /// /// # Examples /// /// ``` /// use icu::locale::Locale; /// /// Locale::try_from_str("en-US-u-hc-h12").unwrap(); /// ``` #[inline] #[cfg(feature = "alloc")] pubfn try_from_str(s: &str) -> Result<Self, ParseError> { Self::try_from_utf8(s.as_bytes())
}
/// See [`Self::try_from_str`] /// /// ✨ *Enabled with the `alloc` Cargo feature.* #[cfg(feature = "alloc")] pubfn try_from_utf8(code_units: &[u8]) -> Result<Self, ParseError> {
parse_locale(code_units)
}
/// Normalize the locale (operating on UTF-8 formatted byte slices) /// /// This operation will normalize casing and the separator. /// /// ✨ *Enabled with the `alloc` Cargo feature.* /// /// # Examples /// /// ``` /// use icu::locale::Locale; /// /// assert_eq!( /// Locale::normalize_utf8(b"pL-latn-pl-U-HC-H12").as_deref(), /// Ok("pl-Latn-PL-u-hc-h12") /// ); /// ``` #[cfg(feature = "alloc")] pubfn normalize_utf8(input: &[u8]) -> Result<Cow<'_, str>, ParseError> { let locale = Self::try_from_utf8(input)?;
Ok(writeable::to_string_or_borrow(&locale, input))
}
/// Normalize the locale (operating on strings) /// /// This operation will normalize casing and the separator. /// /// ✨ *Enabled with the `alloc` Cargo feature.* /// /// # Examples /// /// ``` /// use icu::locale::Locale; /// /// assert_eq!( /// Locale::normalize("pL-latn-pl-U-HC-H12").as_deref(), /// Ok("pl-Latn-PL-u-hc-h12") /// ); /// ``` #[cfg(feature = "alloc")] pubfn normalize(input: &str) -> Result<Cow<'_, str>, ParseError> { Self::normalize_utf8(input.as_bytes())
}
/// Compare this [`Locale`] with BCP-47 bytes. /// /// The return value is equivalent to what would happen if you first converted this /// [`Locale`] to a BCP-47 string and then performed a byte comparison. /// /// This function is case-sensitive and results in a *total order*, so it is appropriate for /// binary search. The only argument producing [`Ordering::Equal`] is `self.to_string()`. /// /// # Examples /// /// Sorting a list of locales with this method requires converting one of them to a string: /// /// ``` /// use icu::locale::Locale; /// use std::cmp::Ordering; /// use writeable::Writeable; /// /// // Random input order: /// let bcp47_strings: &[&str] = &[ /// "und-u-ca-hebrew", /// "ar-Latn", /// "zh-Hant-TW", /// "zh-TW", /// "und-fonipa", /// "zh-Hant", /// "ar-SA", /// ]; /// /// let mut locales = bcp47_strings /// .iter() /// .map(|s| s.parse().unwrap()) /// .collect::<Vec<Locale>>(); /// locales.sort_by(|a, b| { /// let b = b.write_to_string(); /// a.strict_cmp(b.as_bytes()) /// }); /// let strict_cmp_strings = locales /// .iter() /// .map(|l| l.to_string()) /// .collect::<Vec<String>>(); /// /// // Output ordering, sorted alphabetically /// let expected_ordering: &[&str] = &[ /// "ar-Latn", /// "ar-SA", /// "und-fonipa", /// "und-u-ca-hebrew", /// "zh-Hant", /// "zh-Hant-TW", /// "zh-TW", /// ]; /// /// assert_eq!(expected_ordering, strict_cmp_strings); /// ``` pubfn strict_cmp(&self, other: &[u8]) -> Ordering {
writeable::cmp_utf8(self, other)
}
/// Returns an ordering suitable for use in [`BTreeSet`]. /// /// Unlike [`Locale::strict_cmp`], the ordering may or may not be equivalent /// to string ordering, and it may or may not be stable across ICU4X releases. /// /// # Examples /// /// This method returns a nonsensical ordering derived from the fields of the struct: /// /// ``` /// use icu::locale::Locale; /// use std::cmp::Ordering; /// /// // Input strings, sorted alphabetically /// let bcp47_strings: &[&str] = &[ /// "ar-Latn", /// "ar-SA", /// "und-fonipa", /// "und-u-ca-hebrew", /// "zh-Hant", /// "zh-Hant-TW", /// "zh-TW", /// ]; /// assert!(bcp47_strings.windows(2).all(|w| w[0] < w[1])); /// /// let mut locales = bcp47_strings /// .iter() /// .map(|s| s.parse().unwrap()) /// .collect::<Vec<Locale>>(); /// locales.sort_by(Locale::total_cmp); /// let total_cmp_strings = locales /// .iter() /// .map(|l| l.to_string()) /// .collect::<Vec<String>>(); /// /// // Output ordering, sorted arbitrarily /// let expected_ordering: &[&str] = &[ /// "ar-SA", /// "ar-Latn", /// "und-u-ca-hebrew", /// "und-fonipa", /// "zh-TW", /// "zh-Hant", /// "zh-Hant-TW", /// ]; /// /// assert_eq!(expected_ordering, total_cmp_strings); /// ``` /// /// Use a wrapper to add a [`Locale`] to a [`BTreeSet`]: /// /// ```no_run /// use icu::locale::Locale; /// use std::cmp::Ordering; /// use std::collections::BTreeSet; /// /// #[derive(PartialEq, Eq)] /// struct LocaleTotalOrd(Locale); /// /// impl Ord for LocaleTotalOrd { /// fn cmp(&self, other: &Self) -> Ordering { /// self.0.total_cmp(&other.0) /// } /// } /// /// impl PartialOrd for LocaleTotalOrd { /// fn partial_cmp(&self, other: &Self) -> Option<Ordering> { /// Some(self.cmp(other)) /// } /// } /// /// let _: BTreeSet<LocaleTotalOrd> = unimplemented!(); /// ``` /// /// [`BTreeSet`]: alloc::collections::BTreeSet pubfn total_cmp(&self, other: &Self) -> Ordering { self.as_tuple().cmp(&other.as_tuple())
}
/// Compare this `Locale` with a potentially unnormalized BCP-47 string. /// /// The return value is equivalent to what would happen if you first parsed the /// BCP-47 string to a `Locale` and then performed a structural comparison. /// /// ✨ *Enabled with the `alloc` Cargo feature.* /// /// # Examples /// /// ``` /// use icu::locale::Locale; /// /// let bcp47_strings: &[&str] = &[ /// "pl-LaTn-pL", /// "uNd", /// "UND-FONIPA", /// "UnD-t-m0-TrUe", /// "uNd-u-CA-Japanese", /// "ZH", /// ]; /// /// for a in bcp47_strings { /// assert!(a.parse::<Locale>().unwrap().normalizing_eq(a)); /// } /// ``` #[cfg(feature = "alloc")] pubfn normalizing_eq(&self, other: &str) -> bool {
macro_rules! subtag_matches {
($T:ty, $iter:ident, $expected:expr) => {
$iter
.next()
.map(|b| <$T>::try_from_utf8(b) == Ok($expected))
.unwrap_or(false)
};
}
letmut iter = SubtagIterator::new(other.as_bytes()); if !subtag_matches!(subtags::Language, iter, self.id.language) { returnfalse;
} iflet Some(ref script) = self.id.script { if !subtag_matches!(subtags::Script, iter, *script) { returnfalse;
}
} iflet Some(ref region) = self.id.region { if !subtag_matches!(subtags::Region, iter, *region) { returnfalse;
}
} for variant inself.id.variants.iter() { if !subtag_matches!(subtags::Variant, iter, *variant) { returnfalse;
}
} if !self.extensions.is_empty() { match extensions::Extensions::try_from_iter(&mut iter) {
Ok(exts) => { ifself.extensions != exts { returnfalse;
}
}
Err(_) => { returnfalse;
}
}
}
iter.next().is_none()
}
Die Informationen auf dieser Webseite wurden
nach bestem Wissen sorgfältig zusammengestellt. Es wird jedoch weder Vollständigkeit, noch Richtigkeit,
noch Qualität der bereit gestellten Informationen zugesichert.
Bemerkung:
Die farbliche Syntaxdarstellung und die Messung sind noch experimentell.