diff --git a/bin/Cargo.toml b/bin/Cargo.toml index 981068a..23883b6 100644 --- a/bin/Cargo.toml +++ b/bin/Cargo.toml @@ -16,5 +16,5 @@ name = "data-encoding" path = "src/main.rs" [dependencies] -data-encoding = { version = "2.11.2-git", path = "../lib" } +data-encoding = { version = "2.12.0-git", path = "../lib" } getopts = "0.2" diff --git a/lib/CHANGELOG.md b/lib/CHANGELOG.md index 087a505..9c2b3dd 100644 --- a/lib/CHANGELOG.md +++ b/lib/CHANGELOG.md @@ -1,6 +1,11 @@ # Changelog -## 2.11.2-git +## 2.12.0-git + +### Minor + +- Document the input chunk alignment for `Encoding::decode_mut()` +- Add `Encoding::decode_align()` to decide where to split long inputs ### Patch diff --git a/lib/Cargo.toml b/lib/Cargo.toml index 1288e60..ff6d396 100644 --- a/lib/Cargo.toml +++ b/lib/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "data-encoding" -version = "2.11.2-git" +version = "2.12.0-git" license = "MIT" edition = "2018" rust-version = "1.48" diff --git a/lib/macro/Cargo.toml b/lib/macro/Cargo.toml index f89fa0a..83c69b7 100644 --- a/lib/macro/Cargo.toml +++ b/lib/macro/Cargo.toml @@ -13,5 +13,5 @@ description = "Macros for data-encoding" include = ["Cargo.toml", "LICENSE", "README.md", "src/lib.rs"] [dependencies] -data-encoding = { version = "2.11.2-git", path = "..", default-features = false } +data-encoding = { version = "2.12.0-git", path = "..", default-features = false } data-encoding-macro-internal = { version = "0.1.20-git", path = "internal" } diff --git a/lib/macro/internal/Cargo.toml b/lib/macro/internal/Cargo.toml index 8c3e517..eab1d2a 100644 --- a/lib/macro/internal/Cargo.toml +++ b/lib/macro/internal/Cargo.toml @@ -13,7 +13,7 @@ include = ["Cargo.toml", "LICENSE", "README.md", "src/lib.rs"] proc-macro = true [dependencies.data-encoding] -version = "2.11.2-git" +version = "2.12.0-git" path = "../.." default-features = false features = ["alloc"] diff --git a/lib/src/lib.rs b/lib/src/lib.rs index 225d6dc..6628de8 100644 --- a/lib/src/lib.rs +++ b/lib/src/lib.rs @@ -1530,6 +1530,11 @@ impl Encoding { /// - `Err(DecodePartial { read, written, .. })` means that `read` bytes have been read and /// `written` bytes written (the error can be ignored) /// + /// The number of non-ignored characters of each input chunk must be a multiple of + /// [`decode_align`]. Bases whose bit-width does not divide 8 and that don't use padding + /// otherwise decode a trailing partial group of symbols as the end of the input, so + /// `Ok(written)` is returned and the remaining symbols of that group are lost. + /// /// Note that this function only _may_ panic in those cases. The function may also return the /// correct value in some cases depending on the implementation. In other words, those limits /// are the guarantee below which the function will not panic, and not the guarantee above which @@ -1540,6 +1545,7 @@ impl Encoding { /// Returns an error if `len` is invalid. The error kind is [`Length`] and the [position] is the /// greatest valid input length. /// + /// [`decode_align`]: struct.Encoding.html#method.decode_align /// [`decode_mut`]: struct.Encoding.html#method.decode_mut /// [`Length`]: enum.DecodeKind.html#variant.Length /// [position]: struct.DecodeError.html#structfield.position @@ -1557,6 +1563,16 @@ impl Encoding { Ok(olen) } + /// Returns the minimum alignment when chunking a long input + /// + /// See [`decode_len`] for context. + /// + /// [`decode_len`]: struct.Encoding.html#method.decode_len + #[must_use] + pub fn decode_align(&self) -> usize { + dec(self.bit()) + } + /// Decodes `input` in `output` /// /// Returns the length of the decoded output. This length may be smaller than the output length diff --git a/lib/tests/lib.rs b/lib/tests/lib.rs index 22aca36..24f721f 100644 --- a/lib/tests/lib.rs +++ b/lib/tests/lib.rs @@ -701,3 +701,57 @@ fn encoder() { test(&[b"foob", b"a"], "Zm9vYmE="); test(&[b"foob", b"ar"], "Zm9vYmFy"); } + +#[test] +fn decode_align() { + // Decodes chunk by chunk as documented by `decode_len`, cutting each chunk after + // `decode_align` non-ignored characters. + #[track_caller] + fn test(base: &Encoding, align: usize) { + assert_eq!(base.decode_align(), align); + for input in [b"" as &[u8], b"f", b"fo", b"foo", b"foob", b"fooba", b"foobar", &[0; 25]] { + let encoded = base.encode(input); + let encoded = encoded.as_bytes(); + let mut output = Vec::new(); + let mut pos = 0; + while pos < encoded.len() { + let mut end = pos; + let mut count = 0; + while end < encoded.len() + && (count < align || base.interpret_byte(encoded[end]).is_ignored()) + { + count += usize::from(!base.interpret_byte(encoded[end]).is_ignored()); + end += 1; + } + let chunk = &encoded[pos .. end]; + pos = end; + let mut buffer = vec![0; base.decode_len(chunk.len()).unwrap()]; + let written = base.decode_mut(chunk, &mut buffer).unwrap(); + output.extend_from_slice(&buffer[.. written]); + } + assert_eq!(output, input); + } + } + let custom = |symbols| { + let mut spec = Specification::new(); + spec.symbols.push_str(symbols); + spec.encoding().unwrap() + }; + test(&custom("01"), 8); + test(&custom("0123"), 4); + test(&custom("01234567"), 8); + test(&data_encoding::HEXLOWER, 2); + test(&data_encoding::HEXUPPER, 2); + test(&data_encoding::BASE32, 8); + test(&data_encoding::BASE32_NOPAD, 8); + test(&data_encoding::BASE32HEX, 8); + test(&data_encoding::BASE32HEX_NOPAD, 8); + test(&data_encoding::BASE32_DNSSEC, 8); + test(&data_encoding::BASE32_DNSCURVE, 8); + test(&data_encoding::BASE64, 4); + test(&data_encoding::BASE64_NOPAD, 4); + test(&data_encoding::BASE64_MIME, 4); + test(&data_encoding::BASE64_MIME_PERMISSIVE, 4); + test(&data_encoding::BASE64URL, 4); + test(&data_encoding::BASE64URL_NOPAD, 4); +}