| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320 |
- // Copyright 2013-2014 The rust-url developers.
- //
- // Licensed under the Apache License, Version 2.0 <LICENSE-APACHE or
- // http://www.apache.org/licenses/LICENSE-2.0> or the MIT license
- // <LICENSE-MIT or http://opensource.org/licenses/MIT>, at your
- // option. This file may not be copied, modified, or distributed
- // except according to those terms.
- //! [*Unicode IDNA Compatibility Processing*
- //! (Unicode Technical Standard #46)](http://www.unicode.org/reports/tr46/)
- use self::Mapping::*;
- use punycode;
- use std::ascii::AsciiExt;
- use unicode_normalization::UnicodeNormalization;
- use unicode_normalization::char::is_combining_mark;
- use unicode_bidi::{BidiClass, bidi_class};
- include!("uts46_mapping_table.rs");
- #[derive(Debug)]
- enum Mapping {
- Valid,
- Ignored,
- Mapped(&'static str),
- Deviation(&'static str),
- Disallowed,
- DisallowedStd3Valid,
- DisallowedStd3Mapped(&'static str),
- }
- struct Range {
- from: char,
- to: char,
- mapping: Mapping,
- }
- fn find_char(codepoint: char) -> &'static Mapping {
- let mut min = 0;
- let mut max = TABLE.len() - 1;
- while max > min {
- let mid = (min + max) >> 1;
- if codepoint > TABLE[mid].to {
- min = mid;
- } else if codepoint < TABLE[mid].from {
- max = mid;
- } else {
- min = mid;
- max = mid;
- }
- }
- &TABLE[min].mapping
- }
- fn map_char(codepoint: char, flags: Flags, output: &mut String, errors: &mut Vec<Error>) {
- match *find_char(codepoint) {
- Mapping::Valid => output.push(codepoint),
- Mapping::Ignored => {},
- Mapping::Mapped(mapping) => output.push_str(mapping),
- Mapping::Deviation(mapping) => {
- if flags.transitional_processing {
- output.push_str(mapping)
- } else {
- output.push(codepoint)
- }
- }
- Mapping::Disallowed => {
- errors.push(Error::DissallowedCharacter);
- output.push(codepoint);
- }
- Mapping::DisallowedStd3Valid => {
- if flags.use_std3_ascii_rules {
- errors.push(Error::DissallowedByStd3AsciiRules);
- }
- output.push(codepoint)
- }
- Mapping::DisallowedStd3Mapped(mapping) => {
- if flags.use_std3_ascii_rules {
- errors.push(Error::DissallowedMappedInStd3);
- }
- output.push_str(mapping)
- }
- }
- }
- // http://tools.ietf.org/html/rfc5893#section-2
- fn passes_bidi(label: &str, transitional_processing: bool) -> bool {
- let mut chars = label.chars();
- let class = match chars.next() {
- Some(c) => bidi_class(c),
- None => return true, // empty string
- };
- if class == BidiClass::L
- || (class == BidiClass::ON && transitional_processing) // starts with \u200D
- || (class == BidiClass::ES && transitional_processing) // hack: 1.35.+33.49
- || class == BidiClass::EN // hack: starts with number 0à.\u05D0
- { // LTR
- // Rule 5
- loop {
- match chars.next() {
- Some(c) => {
- let c = bidi_class(c);
- if !matches!(c, BidiClass::L | BidiClass::EN |
- BidiClass::ES | BidiClass::CS |
- BidiClass::ET | BidiClass::ON |
- BidiClass::BN | BidiClass::NSM) {
- return false;
- }
- },
- None => { break; },
- }
- }
- // Rule 6
- let mut rev_chars = label.chars().rev();
- let mut last = rev_chars.next();
- loop { // must end in L or EN followed by 0 or more NSM
- match last {
- Some(c) if bidi_class(c) == BidiClass::NSM => {
- last = rev_chars.next();
- continue;
- }
- _ => { break; },
- }
- }
- // TODO: does not pass for àˇ.\u05D0
- // match last {
- // Some(c) if bidi_class(c) == BidiClass::L
- // || bidi_class(c) == BidiClass::EN => {},
- // Some(c) => { return false; },
- // _ => {}
- // }
- } else if class == BidiClass::R || class == BidiClass::AL { // RTL
- let mut found_en = false;
- let mut found_an = false;
- // Rule 2
- loop {
- match chars.next() {
- Some(c) => {
- let char_class = bidi_class(c);
- if char_class == BidiClass::EN {
- found_en = true;
- }
- if char_class == BidiClass::AN {
- found_an = true;
- }
- if !matches!(char_class, BidiClass::R | BidiClass::AL |
- BidiClass::AN | BidiClass::EN |
- BidiClass::ES | BidiClass::CS |
- BidiClass::ET | BidiClass::ON |
- BidiClass::BN | BidiClass::NSM) {
- return false;
- }
- },
- None => { break; },
- }
- }
- // Rule 3
- let mut rev_chars = label.chars().rev();
- let mut last = rev_chars.next();
- loop { // must end in L or EN followed by 0 or more NSM
- match last {
- Some(c) if bidi_class(c) == BidiClass::NSM => {
- last = rev_chars.next();
- continue;
- }
- _ => { break; },
- }
- }
- match last {
- Some(c) if matches!(bidi_class(c), BidiClass::R | BidiClass::AL |
- BidiClass::EN | BidiClass::AN) => {},
- _ => { return false; }
- }
- // Rule 4
- if found_an && found_en {
- return false;
- }
- } else {
- // Rule 2: Should start with L or R/AL
- return false;
- }
- return true;
- }
- /// http://www.unicode.org/reports/tr46/#Validity_Criteria
- fn validate(label: &str, flags: Flags, errors: &mut Vec<Error>) {
- if label.nfc().ne(label.chars()) {
- errors.push(Error::ValidityCriteria);
- }
- // Can not contain '.' since the input is from .split('.')
- // Spec says that the label must not contain a HYPHEN-MINUS character in both the
- // third and fourth positions. But nobody follows this criteria. See the spec issue below:
- // https://github.com/whatwg/url/issues/53
- if label.starts_with("-")
- || label.ends_with("-")
- || label.chars().next().map_or(false, is_combining_mark)
- || label.chars().any(|c| match *find_char(c) {
- Mapping::Valid => false,
- Mapping::Deviation(_) => flags.transitional_processing,
- Mapping::DisallowedStd3Valid => flags.use_std3_ascii_rules,
- _ => true,
- })
- || !passes_bidi(label, flags.transitional_processing)
- {
- errors.push(Error::ValidityCriteria)
- }
- }
- /// http://www.unicode.org/reports/tr46/#Processing
- fn processing(domain: &str, flags: Flags, errors: &mut Vec<Error>) -> String {
- let mut mapped = String::new();
- for c in domain.chars() {
- map_char(c, flags, &mut mapped, errors)
- }
- let normalized: String = mapped.nfc().collect();
- let mut validated = String::new();
- for label in normalized.split('.') {
- if validated.len() > 0 {
- validated.push('.');
- }
- if label.starts_with("xn--") {
- match punycode::decode_to_string(&label["xn--".len()..]) {
- Some(decoded_label) => {
- let flags = Flags { transitional_processing: false, ..flags };
- validate(&decoded_label, flags, errors);
- validated.push_str(&decoded_label)
- }
- None => errors.push(Error::PunycodeError)
- }
- } else {
- validate(label, flags, errors);
- validated.push_str(label)
- }
- }
- validated
- }
- #[derive(Copy, Clone)]
- pub struct Flags {
- pub use_std3_ascii_rules: bool,
- pub transitional_processing: bool,
- pub verify_dns_length: bool,
- }
- #[derive(PartialEq, Eq, Clone, Copy, Debug)]
- enum Error {
- PunycodeError,
- ValidityCriteria,
- DissallowedByStd3AsciiRules,
- DissallowedMappedInStd3,
- DissallowedCharacter,
- TooLongForDns,
- }
- /// Errors recorded during UTS #46 processing.
- ///
- /// This is opaque for now, only indicating the presence of at least one error.
- /// More details may be exposed in the future.
- #[derive(Debug)]
- pub struct Errors(Vec<Error>);
- /// http://www.unicode.org/reports/tr46/#ToASCII
- pub fn to_ascii(domain: &str, flags: Flags) -> Result<String, Errors> {
- let mut errors = Vec::new();
- let mut result = String::new();
- for label in processing(domain, flags, &mut errors).split('.') {
- if result.len() > 0 {
- result.push('.');
- }
- if label.is_ascii() {
- result.push_str(label);
- } else {
- match punycode::encode_str(label) {
- Some(x) => {
- result.push_str("xn--");
- result.push_str(&x);
- },
- None => errors.push(Error::PunycodeError)
- }
- }
- }
- if flags.verify_dns_length {
- let domain = if result.ends_with(".") { &result[..result.len()-1] } else { &*result };
- if domain.len() < 1 || domain.len() > 253 ||
- domain.split('.').any(|label| label.len() < 1 || label.len() > 63) {
- errors.push(Error::TooLongForDns)
- }
- }
- if errors.is_empty() {
- Ok(result)
- } else {
- Err(Errors(errors))
- }
- }
- /// http://www.unicode.org/reports/tr46/#ToUnicode
- ///
- /// Only `use_std3_ascii_rules` is used in `flags`.
- pub fn to_unicode(domain: &str, mut flags: Flags) -> (String, Result<(), Errors>) {
- flags.transitional_processing = false;
- let mut errors = Vec::new();
- let domain = processing(domain, flags, &mut errors);
- let errors = if errors.is_empty() {
- Ok(())
- } else {
- Err(Errors(errors))
- };
- (domain, errors)
- }
|