Skip to main content

headless_lms_utils/
url_encoding.rs

1use bytes::Bytes;
2use percent_encoding::{AsciiSet, NON_ALPHANUMERIC, percent_decode_str, utf8_percent_encode};
3
4/// URL-encodes a string value for use in HTTP headers or other contexts requiring ASCII-compatibility.
5/// Percent-encodes all non-alphanumeric characters (including spaces, punctuation, ASCII special
6/// characters, non-ASCII characters, and control characters) to preserve the original information
7/// while making the value ASCII-safe for use in HTTP headers or other contexts requiring ASCII-compatibility.
8pub fn url_encode(value: &str) -> Bytes {
9    percent_encode_component(value).into()
10}
11
12/// Percent-encodes everything but ASCII alphanumerics, for a value placed in one URL path
13/// segment or query parameter. Same encoding as [`url_encode`], as a `String`.
14pub fn percent_encode_component(value: &str) -> String {
15    utf8_percent_encode(value, NON_ALPHANUMERIC).to_string()
16}
17
18/// URL-decodes a percent-encoded string back to its original UTF-8 representation.
19/// Decodes percent-encoded values back to their original UTF-8 strings.
20pub fn url_decode(encoded: &str) -> anyhow::Result<String> {
21    percent_decode_str(encoded)
22        .decode_utf8()
23        .map_err(|e| anyhow::anyhow!("Failed to decode URL-encoded value: {}", e))
24        .map(|s| s.to_string())
25}
26
27/// Percent-encodes the characters RFC 3986 forbids in a URI reference's fragment.
28///
29/// Unlike [`url_encode`], the fragment's own syntax survives: the leading `#`, the `/` and `?`
30/// separators, the sub-delimiters and any existing `%XX` escape all pass through. Use it for
31/// values that are URI references rather than opaque data, such as JSON Schema `$ref`s.
32pub fn percent_encode_fragment(reference: &str) -> String {
33    utf8_percent_encode(reference, FRAGMENT_FORBIDDEN).to_string()
34}
35
36/// What to encode: everything the `fragment = *( pchar / "/" / "?" )` production disallows, except
37/// `#` and `%`, which are left alone so that a whole URI reference and its existing escapes survive.
38const FRAGMENT_FORBIDDEN: &AsciiSet = &NON_ALPHANUMERIC
39    .remove(b'-')
40    .remove(b'.')
41    .remove(b'_')
42    .remove(b'~')
43    .remove(b'!')
44    .remove(b'$')
45    .remove(b'&')
46    .remove(b'\'')
47    .remove(b'(')
48    .remove(b')')
49    .remove(b'*')
50    .remove(b'+')
51    .remove(b',')
52    .remove(b';')
53    .remove(b'=')
54    .remove(b':')
55    .remove(b'@')
56    .remove(b'/')
57    .remove(b'?')
58    .remove(b'#')
59    .remove(b'%');
60
61#[cfg(test)]
62mod tests {
63    use super::*;
64
65    #[test]
66    fn round_trips_a_value_through_encoding() {
67        let value = "Hello, wörld! / 100%";
68        let encoded = url_encode(value);
69        assert_eq!(
70            url_decode(std::str::from_utf8(&encoded).unwrap()).unwrap(),
71            value
72        );
73    }
74
75    #[test]
76    fn fragment_encoding_keeps_the_reference_syntax() {
77        assert_eq!(
78            percent_encode_fragment("#/definitions/TopLevelSpec"),
79            "#/definitions/TopLevelSpec"
80        );
81    }
82
83    #[test]
84    fn fragment_encoding_escapes_what_a_fragment_may_not_contain() {
85        assert_eq!(
86            percent_encode_fragment("#/definitions/MarkPropDef<(Gradient|string|null)>"),
87            "#/definitions/MarkPropDef%3C(Gradient%7Cstring%7Cnull)%3E"
88        );
89    }
90
91    #[test]
92    fn fragment_encoding_leaves_an_existing_escape_alone() {
93        assert_eq!(
94            percent_encode_fragment("#/definitions/A%20B"),
95            "#/definitions/A%20B"
96        );
97    }
98}