percent_encoding.rs 4.9 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146
  1. // Copyright 2013-2014 Simon Sapin.
  2. //
  3. // Licensed under the Apache License, Version 2.0 <LICENSE-APACHE or
  4. // http://www.apache.org/licenses/LICENSE-2.0> or the MIT license
  5. // <LICENSE-MIT or http://opensource.org/licenses/MIT>, at your
  6. // option. This file may not be copied, modified, or distributed
  7. // except according to those terms.
  8. #[path = "encode_sets.rs"]
  9. mod encode_sets;
  10. /// Represents a set of characters / bytes that should be percent-encoded.
  11. ///
  12. /// See [encode sets specification](http://url.spec.whatwg.org/#simple-encode-set).
  13. ///
  14. /// Different characters need to be encoded in different parts of an URL.
  15. /// For example, a literal `?` question mark in an URL’s path would indicate
  16. /// the start of the query string.
  17. /// A question mark meant to be part of the path therefore needs to be percent-encoded.
  18. /// In the query string however, a question mark does not have any special meaning
  19. /// and does not need to be percent-encoded.
  20. ///
  21. /// Since the implementation details of `EncodeSet` are private,
  22. /// the set of available encode sets is not extensible beyond the ones
  23. /// provided here.
  24. /// If you need a different encode set,
  25. /// please [file a bug](https://github.com/servo/rust-url/issues)
  26. /// explaining the use case.
  27. #[derive(Copy, Clone)]
  28. pub struct EncodeSet {
  29. map: &'static [&'static str; 256],
  30. }
  31. /// This encode set is used for fragment identifier and non-relative scheme data.
  32. pub static SIMPLE_ENCODE_SET: EncodeSet = EncodeSet { map: &encode_sets::SIMPLE };
  33. /// This encode set is used in the URL parser for query strings.
  34. pub static QUERY_ENCODE_SET: EncodeSet = EncodeSet { map: &encode_sets::QUERY };
  35. /// This encode set is used for path components.
  36. pub static DEFAULT_ENCODE_SET: EncodeSet = EncodeSet { map: &encode_sets::DEFAULT };
  37. /// This encode set is used in the URL parser for usernames and passwords.
  38. pub static USERINFO_ENCODE_SET: EncodeSet = EncodeSet { map: &encode_sets::USERINFO };
  39. /// This encode set should be used when setting the password field of a parsed URL.
  40. pub static PASSWORD_ENCODE_SET: EncodeSet = EncodeSet { map: &encode_sets::PASSWORD };
  41. /// This encode set should be used when setting the username field of a parsed URL.
  42. pub static USERNAME_ENCODE_SET: EncodeSet = EncodeSet { map: &encode_sets::USERNAME };
  43. /// This encode set is used in `application/x-www-form-urlencoded` serialization.
  44. pub static FORM_URLENCODED_ENCODE_SET: EncodeSet = EncodeSet {
  45. map: &encode_sets::FORM_URLENCODED,
  46. };
  47. /// Percent-encode the given bytes, and push the result to `output`.
  48. ///
  49. /// The pushed strings are within the ASCII range.
  50. #[inline]
  51. pub fn percent_encode_to(input: &[u8], encode_set: EncodeSet, output: &mut String) {
  52. for &byte in input {
  53. output.push_str(encode_set.map[byte as usize])
  54. }
  55. }
  56. /// Percent-encode the given bytes.
  57. ///
  58. /// The returned string is within the ASCII range.
  59. #[inline]
  60. pub fn percent_encode(input: &[u8], encode_set: EncodeSet) -> String {
  61. let mut output = String::new();
  62. percent_encode_to(input, encode_set, &mut output);
  63. output
  64. }
  65. /// Percent-encode the UTF-8 encoding of the given string, and push the result to `output`.
  66. ///
  67. /// The pushed strings are within the ASCII range.
  68. #[inline]
  69. pub fn utf8_percent_encode_to(input: &str, encode_set: EncodeSet, output: &mut String) {
  70. percent_encode_to(input.as_bytes(), encode_set, output)
  71. }
  72. /// Percent-encode the UTF-8 encoding of the given string.
  73. ///
  74. /// The returned string is within the ASCII range.
  75. #[inline]
  76. pub fn utf8_percent_encode(input: &str, encode_set: EncodeSet) -> String {
  77. let mut output = String::new();
  78. utf8_percent_encode_to(input, encode_set, &mut output);
  79. output
  80. }
  81. /// Percent-decode the given bytes, and push the result to `output`.
  82. pub fn percent_decode_to(input: &[u8], output: &mut Vec<u8>) {
  83. let mut i = 0;
  84. while i < input.len() {
  85. let c = input[i];
  86. if c == b'%' && i + 2 < input.len() {
  87. if let (Some(h), Some(l)) = (from_hex(input[i + 1]), from_hex(input[i + 2])) {
  88. output.push(h * 0x10 + l);
  89. i += 3;
  90. continue
  91. }
  92. }
  93. output.push(c);
  94. i += 1;
  95. }
  96. }
  97. /// Percent-decode the given bytes.
  98. #[inline]
  99. pub fn percent_decode(input: &[u8]) -> Vec<u8> {
  100. let mut output = Vec::new();
  101. percent_decode_to(input, &mut output);
  102. output
  103. }
  104. /// Percent-decode the given bytes, and decode the result as UTF-8.
  105. ///
  106. /// This is “lossy”: invalid UTF-8 percent-encoded byte sequences
  107. /// will be replaced � U+FFFD, the replacement character.
  108. #[inline]
  109. pub fn lossy_utf8_percent_decode(input: &[u8]) -> String {
  110. String::from_utf8_lossy(&percent_decode(input)).to_string()
  111. }
  112. #[inline]
  113. pub fn from_hex(byte: u8) -> Option<u8> {
  114. match byte {
  115. b'0' ... b'9' => Some(byte - b'0'), // 0..9
  116. b'A' ... b'F' => Some(byte + 10 - b'A'), // A..F
  117. b'a' ... b'f' => Some(byte + 10 - b'a'), // a..f
  118. _ => None
  119. }
  120. }