From 45d2c1d8ec22fc66e26ffb56af098fb85f183f5d Mon Sep 17 00:00:00 2001 From: Kirill Date: Fri, 24 Jul 2026 20:57:32 +0200 Subject: [PATCH] =?UTF-8?q?ChaCha20=E2=80=91Poly1305=20encryption?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit --- CMakeLists.txt | 14 +- README.md | 60 ++--- examples/client.conf | 2 +- examples/client_udp.conf | 2 +- examples/server.conf | 2 +- examples/server_udp.conf | 2 +- include/ntptun/Carrier.hpp | 24 +- include/ntptun/ChaCha20Poly1305.hpp | 66 +++++ include/ntptun/Client.hpp | 2 + include/ntptun/Config.hpp | 5 +- include/ntptun/ReplayWindow.hpp | 50 ++++ include/ntptun/Server.hpp | 2 + include/ntptun/Sha256.hpp | 9 +- src/Carrier.cpp | 130 +++++----- src/Client.cpp | 6 +- src/Config.cpp | 3 + src/Server.cpp | 10 +- tests/CMakeLists.txt | 8 + tests/test_aead.cpp | 92 +++++++ tests/test_carrier.cpp | 6 +- tests/test_replay.cpp | 69 ++++++ third_party/portable8439/LICENSE | 121 ++++++++++ .../src/chacha-portable/chacha-portable.c | 226 ++++++++++++++++++ .../src/chacha-portable/chacha-portable.h | 45 ++++ .../src/poly1305-donna/poly1305-donna-16.h | 196 +++++++++++++++ .../src/poly1305-donna/poly1305-donna-32.h | 214 +++++++++++++++++ .../src/poly1305-donna/poly1305-donna-64.h | 219 +++++++++++++++++ .../src/poly1305-donna/poly1305-donna-8.h | 180 ++++++++++++++ .../src/poly1305-donna/poly1305-donna.c | 69 ++++++ .../src/poly1305-donna/poly1305-donna.h | 18 ++ third_party/portable8439/src/portable8439.c | 119 +++++++++ third_party/portable8439/src/portable8439.h | 104 ++++++++ 32 files changed, 1949 insertions(+), 126 deletions(-) create mode 100644 include/ntptun/ChaCha20Poly1305.hpp create mode 100644 include/ntptun/ReplayWindow.hpp create mode 100644 tests/test_aead.cpp create mode 100644 tests/test_replay.cpp create mode 100644 third_party/portable8439/LICENSE create mode 100644 third_party/portable8439/src/chacha-portable/chacha-portable.c create mode 100644 third_party/portable8439/src/chacha-portable/chacha-portable.h create mode 100644 third_party/portable8439/src/poly1305-donna/poly1305-donna-16.h create mode 100644 third_party/portable8439/src/poly1305-donna/poly1305-donna-32.h create mode 100644 third_party/portable8439/src/poly1305-donna/poly1305-donna-64.h create mode 100644 third_party/portable8439/src/poly1305-donna/poly1305-donna-8.h create mode 100644 third_party/portable8439/src/poly1305-donna/poly1305-donna.c create mode 100644 third_party/portable8439/src/poly1305-donna/poly1305-donna.h create mode 100644 third_party/portable8439/src/portable8439.c create mode 100644 third_party/portable8439/src/portable8439.h diff --git a/CMakeLists.txt b/CMakeLists.txt index cfe0785..be8e473 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,5 +1,5 @@ cmake_minimum_required(VERSION 3.16) -project(ntptun LANGUAGES CXX) +project(ntptun LANGUAGES C CXX) # ---- Standard / defaults --------------------------------------------------- set(CMAKE_CXX_STANDARD 17) @@ -30,6 +30,17 @@ else() -Wunused -Wnull-dereference -Wdouble-promotion -Wformat=2) endif() +# ---- Vendored crypto ------------------------------------------------------- +# RFC 8439 (ChaCha20-Poly1305) AEAD, public-domain (CC0) reference sources under +# third_party/portable8439. Built as a separate C library so the project's +# strict C++ warning set is not applied to third-party code. +add_library(ntptun_crypto STATIC + third_party/portable8439/src/portable8439.c + third_party/portable8439/src/chacha-portable/chacha-portable.c + third_party/portable8439/src/poly1305-donna/poly1305-donna.c) +target_include_directories(ntptun_crypto PUBLIC + ${CMAKE_CURRENT_SOURCE_DIR}/third_party/portable8439/src) + # ---- Core library ---------------------------------------------------------- set(ntptun_core_sources src/Ntp.cpp @@ -48,6 +59,7 @@ endif() add_library(ntptun_core STATIC ${ntptun_core_sources}) target_include_directories(ntptun_core PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/include) +target_link_libraries(ntptun_core PUBLIC ntptun_crypto) target_link_libraries(ntptun_core PRIVATE ntptun_warnings) if(WIN32) target_link_libraries(ntptun_core PUBLIC ws2_32) diff --git a/README.md b/README.md index 5ad7d08..ad23e44 100644 --- a/README.md +++ b/README.md @@ -2,21 +2,16 @@ (and UDP too) A small, dependency-free C++17 implementation of the IP-over-NTP tunnel and UDP-over-NTP: it carries one complete inner -IPv4/IPv6 or UDP datagram inside one NTPv4 packet, using an NTP extension field with a -lightweight SHA‑256‑based XOR obfuscation layer for traffic classification and -admission filtering. - -> **Scope and honesty.** This is a *datagram* tunnel and an *obfuscation* layer, -> not a security layer. The XOR construction provides **no confidentiality and -> no authentication**. If you need those properties, run WireGuard or IPsec -> *inside* the tunnel. +IPv4/IPv6 or UDP datagram inside one NTPv4 packet, using an NTP extension field +sealed with ChaCha20‑Poly1305 (RFC 8439) authenticated encryption. - **Targets:** Linux (full features) and Windows/MSVC (UDP transport only — see below). The TUN transport uses `/dev/net/tun` and is Linux-only; sockets and the event loop are portable over Winsock and BSD sockets. -- **Dependencies:** none. Only the C++17 standard library. SHA‑256 is a - self-contained header ([include/ntptun/Sha256.hpp](include/ntptun/Sha256.hpp), - FIPS 180‑4). +- **Dependencies:** none. Only the C++17 standard library. The carrier's + ChaCha20‑Poly1305 (RFC 8439) AEAD is a vendored, public‑domain (CC0) reference + implementation under `third_party/portable8439/`; no external crypto library + is required. - **Interface model:** the app attaches to an **existing** TUN interface that you create and configure out of band (see below), or — in UDP transport — it needs no interface at all. @@ -51,10 +46,10 @@ by `transport =` in the config: Every datagram the inner app sends must fit one NTP carrier: its UDP payload must be ≤ `inner_mtu`, or ntptun drops it (with an error logged). The carrier's -own overhead is a fixed **64 bytes** (`48` NTP header + `4` ext type/len + `12` -carrier header), so the outer packet is `inner_mtu + 64`; keep -`ntp_payload_cap ≥ inner_mtu + 64`. Configure each protocol so its packets stay -under `inner_mtu`: +own overhead is a fixed **80 bytes** (`48` NTP header + `4` ext type/len + `2` +Client ID + `10` protected header + `16` Poly1305 tag), so the outer packet is +`inner_mtu + 80`; keep `ntp_payload_cap ≥ inner_mtu + 80`. Configure each +protocol so its packets stay under `inner_mtu`: | Protocol | Hard min packet? | Path MTU discovery | What you configure | |---|---|---|---| @@ -174,7 +169,7 @@ sudo ip addr add 10.9.0.2/24 dev tun0 Notes: -- **MTU.** With `ntp_payload_cap = 1200` the largest inner datagram is 1136 +- **MTU.** With `ntp_payload_cap = 1216` the largest inner datagram is 1136 bytes. Set the TUN MTU to `inner_mtu` (1136) or lower. - **Port 123.** Binding the standard NTP port needs root or `CAP_NET_BIND_SERVICE`. For unprivileged testing use a high port (the example @@ -226,7 +221,8 @@ and `udp_base_port`. | `tun` | TUN, both | — | existing TUN interface name (required when `transport = tun`) | | `ext_type` | both | `0x0F4E` | NTP extension type used for the tunnel | | `inner_mtu` | both | `1136` | max inner IP datagram (`tun`) or UDP payload (`udp`) size | -| `ntp_payload_cap` | both | `1200` | max NTP UDP payload; must accommodate `inner_mtu + 64` bytes of framing | +| `ntp_payload_cap` | both | `1216` | max NTP UDP payload; must accommodate `inner_mtu + 80` bytes of framing | +| `replay_window` | both | `1024` | anti‑replay window: recent carrier timestamps remembered per association to reject replays (`0` disables) | | `log_level` | both | `info` | `error`/`warn`/`info`/`debug` | | `default_key` | both | — | shared 32‑byte key (hex) | | `client_key` | both | — | `id:hex` per‑client override (repeatable) | @@ -358,23 +354,31 @@ Takeaways: ## Design notes -- **Carrier & obfuscation** — [src/Carrier.cpp](src/Carrier.cpp): 12‑byte carrier - header (clear Client ID + obfuscated Magic/Version/Kind/Payload‑Length/Flags), - per‑offset SHA‑256 mask `M_j = SHA256("ntp-xor-v1" ‖ K_I ‖ I ‖ D ‖ T ‖ j)`, - 4‑byte alignment padding, and all decode‑side validation rules. +- **Carrier & encryption** — [src/Carrier.cpp](src/Carrier.cpp): a clear 2‑byte + Client ID followed by a ChaCha20‑Poly1305 (RFC 8439) seal of the + Magic/Version/Kind/Payload‑Length/Flags header, the inner datagram and 4‑byte + alignment padding, plus the 16‑byte Poly1305 tag. The clear Client ID is the + AEAD associated data and the 96‑bit nonce is `direction ‖ Client ID ‖ Transmit + Timestamp`; decode authenticates before applying all validation rules. - **NTP framing** — [src/Ntp.cpp](src/Ntp.cpp): 48‑byte header and 4‑byte‑aligned extension‑field parsing/assembly. - **Classification & relay** — [src/Server.cpp](src/Server.cpp): tunnel vs. real‑NTP path, request/response correlation (`Origin == Transmit`), non‑blocking relay with amplification and rate‑limit guards. - **Transmit‑timestamp uniqueness** — each endpoint uses a monotonic generator so - a `T` (and therefore an XOR mask) is never reused in a direction; + a `T` (and therefore an AEAD nonce) is never reused in a direction; responses keep `Receive ≤ Transmit`. -## Limitations +## Credits + +The carrier's ChaCha20‑Poly1305 (RFC 8439) AEAD uses the vendored, public‑domain +(CC0) **portable8439** by Davy Landman — + — which bundles the +`chacha-portable` (Davy Landman) and `poly1305-donna` +([Andrew Moon](https://github.com/floodyberry/poly1305-donna)) reference +implementations. The vendored sources live under +[third_party/portable8439/](third_party/portable8439/); their upstream license is +[third_party/portable8439/LICENSE](third_party/portable8439/LICENSE). + +ntptun itself is licensed under the GNU GPL v2 (see [LICENSE](LICENSE)). -No cryptographic confidentiality or authentication, no integrity beyond UDP/IP -checksums, no replay protection, no retransmission, no ordering, no congestion -control, and no tunnel‑level fragmentation. An inner datagram larger than -`inner_mtu` is dropped, not fragmented. The private extension type, packet -sizes and traffic pattern remain visible to a passive observer. diff --git a/examples/client.conf b/examples/client.conf index 20c71aa..9766094 100644 --- a/examples/client.conf +++ b/examples/client.conf @@ -10,7 +10,7 @@ server = 203.0.113.10:12300 ext_type = 0x0F4E inner_mtu = 1136 -ntp_payload_cap = 1200 +ntp_payload_cap = 1216 # Downstream delivery mode (must match the server): # poll (push=false, default): NTP-realistic. The client keeps `poll_window` diff --git a/examples/client_udp.conf b/examples/client_udp.conf index 78e754b..aabee85 100644 --- a/examples/client_udp.conf +++ b/examples/client_udp.conf @@ -23,7 +23,7 @@ ext_type = 0x0F4E # that is MTU = 1104 (put `MTU = 1104` under [Interface] in the WireGuard conf). # The client also logs this recommended maximum at startup. inner_mtu = 1136 -ntp_payload_cap = 1200 +ntp_payload_cap = 1216 # push = false poll_interval_ms = 250 diff --git a/examples/server.conf b/examples/server.conf index 1335097..9329368 100644 --- a/examples/server.conf +++ b/examples/server.conf @@ -17,7 +17,7 @@ upstream = 192.168.1.1:123 ext_type = 0x0F4E inner_mtu = 1136 -ntp_payload_cap = 1200 +ntp_payload_cap = 1216 relay_timeout_ms = 1000 relay_rate_per_sec = 10 log_level = info diff --git a/examples/server_udp.conf b/examples/server_udp.conf index 3925aca..bb1c566 100644 --- a/examples/server_udp.conf +++ b/examples/server_udp.conf @@ -24,7 +24,7 @@ ext_type = 0x0F4E # Largest UDP payload (each direction) that fits one NTP carrier. Larger # datagrams are discarded with an error in the log. inner_mtu = 1136 -ntp_payload_cap = 1200 +ntp_payload_cap = 1216 relay_timeout_ms = 1000 relay_rate_per_sec = 10 log_level = info diff --git a/include/ntptun/Carrier.hpp b/include/ntptun/Carrier.hpp index bd3fe7f..4028eb0 100644 --- a/include/ntptun/Carrier.hpp +++ b/include/ntptun/Carrier.hpp @@ -1,5 +1,5 @@ -// Tunnel carrier: the 12-byte carrier header, the SHA-256-based XOR obfuscation -// mask, and encode/decode of the NTP extension value. +// Tunnel carrier: the carrier header, ChaCha20-Poly1305 (RFC 8439) AEAD sealing +// of the protected region, and encode/decode of the NTP extension value. #ifndef NTPTUN_CARRIER_HPP #define NTPTUN_CARRIER_HPP @@ -12,7 +12,7 @@ namespace ntptun { // Fixed deployment marker distinguishing tunnel carriers from unrelated NTP -// extensions ("NTX1"). Obfuscated on the wire. +// extensions ("NTX1"). Encrypted on the wire (inside the AEAD-sealed region). constexpr std::uint32_t kCarrierMagic = 0x4E545831u; constexpr std::uint8_t kCarrierVersion = 1; @@ -31,25 +31,19 @@ constexpr std::size_t kWireGuardOverhead = 32; // Carrier layout sizes. constexpr std::size_t kClientIdSize = 2; // clear -constexpr std::size_t kProtectedHeaderSize = 10; // obfuscated +constexpr std::size_t kProtectedHeaderSize = 10; // sealed (AEAD plaintext) constexpr std::size_t kCarrierHeaderSize = kClientIdSize + kProtectedHeaderSize; // 12 -// Obfuscation direction byte: 0 = client-to-server, 1 = server-to-client. +// Direction byte (also feeds the AEAD nonce): 0 = client-to-server, +// 1 = server-to-client. enum class Direction : std::uint8_t { ClientToServer = 0, ServerToClient = 1, }; -// Apply (or, being XOR, remove) the keystream mask in place over `len` bytes of -// the protected region, using the key, clear Client ID, direction and the outer -// packet's own Transmit Timestamp. -void apply_xor_mask(Byte* data, std::size_t len, const Key& key, - std::uint16_t client_id, Direction dir, - std::uint64_t transmit_ts); - -// Build a complete extension value (clear Client ID followed by the obfuscated -// protected header, inner datagram and alignment padding). `inner_ip` is empty -// for an empty poll. +// Build a complete extension value (clear Client ID followed by the AEAD-sealed +// protected header, inner datagram, alignment padding and the 16-byte Poly1305 +// tag). `inner_ip` is empty for an empty poll. ByteVector encode_carrier_value(std::uint16_t client_id, std::uint8_t kind, ByteSpan inner_ip, const Key& key, Direction dir, std::uint64_t transmit_ts); diff --git a/include/ntptun/ChaCha20Poly1305.hpp b/include/ntptun/ChaCha20Poly1305.hpp new file mode 100644 index 0000000..120621c --- /dev/null +++ b/include/ntptun/ChaCha20Poly1305.hpp @@ -0,0 +1,66 @@ +// ChaCha20-Poly1305 (RFC 8439) authenticated encryption. +// +// A thin C++ facade over the vendored, public-domain reference implementation +// in third_party/portable8439 (portable8439 + chacha-portable + poly1305-donna, +// all CC0). This provides real AEAD confidentiality and integrity for the +// tunnel carrier; it is not a from-scratch reimplementation of the primitives. +#ifndef NTPTUN_CHACHA20POLY1305_HPP +#define NTPTUN_CHACHA20POLY1305_HPP + +#include +#include +#include +#include + +#include "ntptun/Bytes.hpp" + +extern "C" { +#include "portable8439.h" +} +// portable8439.h defines `restrict` as a macro under C++; do not leak it. +#ifdef restrict +#undef restrict +#endif + +namespace ntptun { + +constexpr std::size_t kAeadKeySize = RFC_8439_KEY_SIZE; // 32 +constexpr std::size_t kAeadNonceSize = RFC_8439_NONCE_SIZE; // 12 +constexpr std::size_t kAeadTagSize = RFC_8439_TAG_SIZE; // 16 + +using AeadNonce = std::array; + +// Seal `plaintext` under `key` (kAeadKeySize bytes) and `nonce`, authenticating +// `aad`. The (key, nonce) pair must never repeat. Returns the ciphertext +// followed by the 16-byte Poly1305 tag (plaintext.size() + kAeadTagSize bytes). +inline ByteVector aead_seal(const Byte* key, const AeadNonce& nonce, + ByteSpan aad, ByteSpan plaintext) { + ByteVector out(plaintext.size() + kAeadTagSize); + portable_chacha20_poly1305_encrypt(out.data(), key, nonce.data(), aad.data(), + aad.size(), plaintext.data(), + plaintext.size()); + return out; +} + +// Verify and decrypt `sealed` (ciphertext followed by its 16-byte tag) under +// `key`, `nonce` and `aad`. Returns the recovered plaintext, or nullopt if +// authentication fails (wrong key/nonce/aad, or the carrier was modified). +inline std::optional aead_open(const Byte* key, + const AeadNonce& nonce, ByteSpan aad, + ByteSpan sealed) { + if (sealed.size() < kAeadTagSize) { + return std::nullopt; + } + ByteVector out(sealed.size() - kAeadTagSize); + const std::size_t n = portable_chacha20_poly1305_decrypt( + out.data(), key, nonce.data(), aad.data(), aad.size(), sealed.data(), + sealed.size()); + if (n == static_cast(-1)) { + return std::nullopt; + } + return out; +} + +} // namespace ntptun + +#endif // NTPTUN_CHACHA20POLY1305_HPP diff --git a/include/ntptun/Client.hpp b/include/ntptun/Client.hpp index 9605c29..72c23cf 100644 --- a/include/ntptun/Client.hpp +++ b/include/ntptun/Client.hpp @@ -16,6 +16,7 @@ #include "ntptun/Config.hpp" #include "ntptun/Endpoint.hpp" #include "ntptun/Platform.hpp" +#include "ntptun/ReplayWindow.hpp" #include "ntptun/TunDevice.hpp" #include "ntptun/UdpSocket.hpp" @@ -52,6 +53,7 @@ private: std::uint64_t last_transmit_ts_ = 0; std::deque pending_; // outstanding request Transmit Timestamps std::uint64_t last_send_ms_ = 0; // time of the last packet sent (keepalive) + ReplayWindow replay_; // anti-replay for server->client carriers }; } // namespace ntptun diff --git a/include/ntptun/Config.hpp b/include/ntptun/Config.hpp index ba0f4d6..08e679f 100644 --- a/include/ntptun/Config.hpp +++ b/include/ntptun/Config.hpp @@ -24,7 +24,10 @@ struct CommonConfig { std::string tun; std::uint16_t ext_type = 0x0F4E; std::size_t inner_mtu = 1136; - std::size_t ntp_payload_cap = 1200; + std::size_t ntp_payload_cap = 1216; + // Anti-replay window: number of recently accepted carrier Transmit + // Timestamps remembered per association to reject replays. 0 disables it. + std::size_t replay_window = 1024; KeyStore keys; LogLevel log_level = LogLevel::Info; // When true the server pushes downstream datagrams to the client's last diff --git a/include/ntptun/ReplayWindow.hpp b/include/ntptun/ReplayWindow.hpp new file mode 100644 index 0000000..f4ed0cf --- /dev/null +++ b/include/ntptun/ReplayWindow.hpp @@ -0,0 +1,50 @@ +// Anti-replay window for the AEAD nonce (the outer NTP Transmit Timestamp). +// +// Each sender emits Transmit Timestamps strictly increasing per direction (see +// Client/Server next_transmit_ts). A receiver remembers the most recent accepted +// timestamps so a replayed or excessively delayed carrier is rejected, while +// still tolerating up to `capacity` packets of network reordering. +#ifndef NTPTUN_REPLAYWINDOW_HPP +#define NTPTUN_REPLAYWINDOW_HPP + +#include +#include +#include + +namespace ntptun { + +class ReplayWindow { +public: + // Test `ts` against the window and, if fresh, record it. Returns true to + // accept (fresh) or false to reject (a replay of an already-seen timestamp, + // or one older than the retained window). `capacity` is the number of recent + // timestamps to remember; 0 disables the check (always accept). The value is + // passed per call so the window needs no configuration at construction. + bool accept(std::uint64_t ts, std::size_t capacity) { + if (capacity == 0) { + return true; // disabled + } + // Below the window floor once the window is full: a replay of an evicted + // timestamp or one delayed beyond the reorder horizon. + if (seen_.size() >= capacity && ts < *seen_.begin()) { + return false; + } + if (!seen_.insert(ts).second) { + return false; // already accepted -> replay + } + // Keep only the `capacity` largest (most recent) timestamps. + while (seen_.size() > capacity) { + seen_.erase(seen_.begin()); + } + return true; + } + + void clear() noexcept { seen_.clear(); } + +private: + std::set seen_; +}; + +} // namespace ntptun + +#endif // NTPTUN_REPLAYWINDOW_HPP diff --git a/include/ntptun/Server.hpp b/include/ntptun/Server.hpp index 3eef1eb..105b66c 100644 --- a/include/ntptun/Server.hpp +++ b/include/ntptun/Server.hpp @@ -20,6 +20,7 @@ #include "ntptun/IpPacket.hpp" #include "ntptun/Ntp.hpp" #include "ntptun/Platform.hpp" +#include "ntptun/ReplayWindow.hpp" #include "ntptun/TunDevice.hpp" #include "ntptun/UdpSocket.hpp" @@ -37,6 +38,7 @@ private: bool has_src = false; // whether last_src is known yet std::uint64_t last_request_ts = 0; // echoed as Origin in pushed packets std::deque downstream; // queued datagrams awaiting a poll + ReplayWindow replay; // anti-replay for client->server carriers }; struct PendingRelay { diff --git a/include/ntptun/Sha256.hpp b/include/ntptun/Sha256.hpp index b0326eb..e73c6e9 100644 --- a/include/ntptun/Sha256.hpp +++ b/include/ntptun/Sha256.hpp @@ -1,9 +1,10 @@ // Self-contained SHA-256 (FIPS 180-4), header-only. // -// This is used ONLY as a mask expander for the tunnel's XOR obfuscation layer. -// It is a straightforward, portable implementation and is NOT hardened against -// timing side-channels. The tunnel makes no cryptographic-confidentiality or -// authentication claims. +// A straightforward, portable implementation retained as a general-purpose +// utility (and exercised by the unit tests). The tunnel carrier is sealed with +// ChaCha20-Poly1305 (see ChaCha20Poly1305.hpp), so SHA-256 is no longer on the +// carrier's encode/decode path. This implementation is NOT hardened against +// timing side-channels. #ifndef NTPTUN_SHA256_HPP #define NTPTUN_SHA256_HPP diff --git a/src/Carrier.cpp b/src/Carrier.cpp index 0309f37..c008533 100644 --- a/src/Carrier.cpp +++ b/src/Carrier.cpp @@ -2,94 +2,74 @@ #include +#include "ntptun/ChaCha20Poly1305.hpp" #include "ntptun/IpPacket.hpp" -#include "ntptun/Sha256.hpp" namespace ntptun { namespace { -// Domain-separation label for the mask. -constexpr char kMaskLabel[] = "ntp-xor-v1"; -constexpr std::size_t kMaskLabelLen = sizeof(kMaskLabel) - 1; // exclude NUL - -// Padding needed so that (kClientIdSize + kProtectedHeaderSize + P) is a -// multiple of four. The 12-byte header is already aligned, so this aligns P. +// Padding needed so that (kClientIdSize + kProtectedHeaderSize + tag + P) is a +// multiple of four. The clear Client ID (2), protected header (10) and Poly1305 +// tag (16) already sum to a multiple of four, so this only aligns the payload P. std::size_t alignment_padding(std::size_t payload_len) { return (4u - (payload_len % 4u)) % 4u; } -} // namespace - -void apply_xor_mask(Byte* data, std::size_t len, const Key& key, - std::uint16_t client_id, Direction dir, - std::uint64_t transmit_ts) { - Byte id_be[2]; - store_be16(id_be, client_id); - Byte ts_be[8]; - store_be64(ts_be, transmit_ts); - const Byte direction_byte = static_cast(dir); - - std::size_t offset = 0; - std::uint32_t block_index = 0; - while (offset < len) { - Sha256 hash; - hash.update(reinterpret_cast(kMaskLabel), kMaskLabelLen); - hash.update(key.data(), key.size()); - hash.update(id_be, sizeof(id_be)); - hash.update(&direction_byte, 1); - hash.update(ts_be, sizeof(ts_be)); - Byte index_be[4]; - store_be32(index_be, block_index); - hash.update(index_be, sizeof(index_be)); - const Sha256::Digest block = hash.finish(); - - const std::size_t n = - (len - offset < Sha256::kDigestSize) ? (len - offset) - : Sha256::kDigestSize; - for (std::size_t i = 0; i < n; ++i) { - data[offset + i] ^= block[i]; - } - offset += n; - ++block_index; - } +// Build the 96-bit AEAD nonce from the fields that make each carrier unique: +// direction, clear Client ID and the outer packet's Transmit Timestamp. A +// sender must never reuse a Transmit Timestamp in the same direction under the +// same key, which keeps every (key, nonce) pair unique. +AeadNonce make_nonce(std::uint16_t client_id, Direction dir, + std::uint64_t transmit_ts) { + AeadNonce nonce{}; // zero-filled; the final byte stays 0 + nonce[0] = static_cast(dir); + store_be16(nonce.data() + 1, client_id); + store_be64(nonce.data() + 3, transmit_ts); + return nonce; } +} // namespace + ByteVector encode_carrier_value(std::uint16_t client_id, std::uint8_t kind, ByteSpan inner_ip, const Key& key, Direction dir, std::uint64_t transmit_ts) { const std::size_t payload_len = inner_ip.size(); const std::size_t pad = alignment_padding(payload_len); - ByteVector value(kCarrierHeaderSize + payload_len + pad); // zero-initialized - - // Clear Client ID. - store_be16(value.data(), client_id); - - // Protected header (Magic || Version || Kind || Payload Length || Flags). - Byte* ph = value.data() + kClientIdSize; - store_be32(ph, kCarrierMagic); - ph[4] = kCarrierVersion; - ph[5] = kind; - store_be16(ph + 6, static_cast(payload_len)); - store_be16(ph + 8, 0); // Flags - + // Protected plaintext (Magic || Version || Kind || Payload Length || Flags), + // then the inner datagram and zero alignment padding. + ByteVector plaintext(kProtectedHeaderSize + payload_len + pad); // zeroed + store_be32(plaintext.data(), kCarrierMagic); + plaintext[4] = kCarrierVersion; + plaintext[5] = kind; + store_be16(plaintext.data() + 6, static_cast(payload_len)); + store_be16(plaintext.data() + 8, 0); // Flags if (payload_len > 0) { - std::memcpy(value.data() + kCarrierHeaderSize, inner_ip.data(), + std::memcpy(plaintext.data() + kProtectedHeaderSize, inner_ip.data(), payload_len); } - // Obfuscate everything after the clear Client ID. - apply_xor_mask(value.data() + kClientIdSize, - kProtectedHeaderSize + payload_len + pad, key, client_id, dir, - transmit_ts); + // The clear Client ID is authenticated as associated data so it cannot be + // swapped to redirect a carrier to a different key without detection. + Byte id_be[2]; + store_be16(id_be, client_id); + const AeadNonce nonce = make_nonce(client_id, dir, transmit_ts); + const ByteVector sealed = + aead_seal(key.data(), nonce, ByteSpan(id_be, 2), ByteSpan(plaintext)); + + // Extension value: clear Client ID followed by ciphertext || tag. + ByteVector value(kClientIdSize + sealed.size()); + store_be16(value.data(), client_id); + std::memcpy(value.data() + kClientIdSize, sealed.data(), sealed.size()); return value; } CarrierStatus decode_carrier_value(ByteSpan value, const KeyStore& keys, Direction dir, std::uint64_t transmit_ts, DecodedCarrier& out) { - // Structural gate: at least a full header, 4-byte aligned value. - if (value.size() < kCarrierHeaderSize || (value.size() % 4) != 0) { + // Structural gate: clear Client ID + protected header + tag, 4-byte aligned. + if (value.size() < kCarrierHeaderSize + kAeadTagSize || + (value.size() % 4) != 0) { return CarrierStatus::NotTunnel; } @@ -102,16 +82,28 @@ CarrierStatus decode_carrier_value(ByteSpan value, const KeyStore& keys, return CarrierStatus::NotTunnel; // no key configured -> relay path } - // Deobfuscate the protected region into a scratch buffer. - const std::size_t protected_len = value.size() - kClientIdSize; - ByteVector buf(value.data() + kClientIdSize, value.data() + value.size()); - apply_xor_mask(buf.data(), protected_len, *key, client_id, dir, transmit_ts); + // Authenticate and decrypt. A failure is cryptographically indistinguishable + // from unrelated traffic (wrong key, wrong nonce, or a modified carrier), so + // it falls through to the real-NTP relay path. + Byte id_be[2]; + store_be16(id_be, client_id); + const AeadNonce nonce = make_nonce(client_id, dir, transmit_ts); + const ByteSpan sealed(value.data() + kClientIdSize, + value.size() - kClientIdSize); + const std::optional opened = + aead_open(key->data(), nonce, ByteSpan(id_be, 2), sealed); + if (!opened) { + return CarrierStatus::NotTunnel; + } + const ByteVector& buf = *opened; + // From here the carrier is authenticated as ours; any inconsistency is a + // hard error (dropped), not a relay candidate. if (load_be32(buf.data()) != kCarrierMagic) { - return CarrierStatus::NotTunnel; // wrong marker -> relay path + return CarrierStatus::Invalid; // wrong marker } if (buf[4] != kCarrierVersion) { - return CarrierStatus::NotTunnel; // unknown version -> relay path + return CarrierStatus::Invalid; // unknown version } const std::uint8_t kind = buf[5]; @@ -126,9 +118,9 @@ CarrierStatus decode_carrier_value(ByteSpan value, const KeyStore& keys, return CarrierStatus::Invalid; } - // The obfuscated region must be exactly header + payload + alignment pad. + // The plaintext must be exactly header + payload + alignment pad. const std::size_t pad = alignment_padding(payload_len); - if (kProtectedHeaderSize + payload_len + pad != protected_len) { + if (kProtectedHeaderSize + payload_len + pad != buf.size()) { return CarrierStatus::Invalid; } for (std::size_t i = 0; i < pad; ++i) { diff --git a/src/Client.cpp b/src/Client.cpp index 3ab0145..2151f0e 100644 --- a/src/Client.cpp +++ b/src/Client.cpp @@ -144,7 +144,7 @@ void Client::drain_inner() { " bytes (max payload is ", cfg_.common.inner_mtu, "); lower the inner app MTU, or for QUIC/Hysteria2 " "raise inner_mtu >= 1200 and ntp_payload_cap >= " - "inner_mtu+64"); + "inner_mtu+80"); continue; } send_carrier(kKindUdpDatagram, @@ -289,6 +289,10 @@ void Client::handle_response(ByteSpan packet) { if (decoded.client_id != cfg_.client_id) { return; // belongs to another logical association } + if (!replay_.accept(header->transmit_ts, cfg_.common.replay_window)) { + log_debug("replay: dropping downstream carrier ts=", header->transmit_ts); + return; + } if ((decoded.kind == kKindIpDatagram || decoded.kind == kKindUdpDatagram) && !decoded.inner_ip.empty()) { deliver_inner(decoded.kind, decoded.inner_ip); diff --git a/src/Config.cpp b/src/Config.cpp index ea0157f..8b2f7b6 100644 --- a/src/Config.cpp +++ b/src/Config.cpp @@ -172,6 +172,9 @@ Config load_config(const std::string& path) { if (const auto* v = get("ntp_payload_cap")) { common.ntp_payload_cap = parse_uint("ntp_payload_cap", *v, 65535); } + if (const auto* v = get("replay_window")) { + common.replay_window = parse_uint("replay_window", *v, 1048576); + } if (const auto* v = get("log_level")) { common.log_level = parse_log_level(*v); } diff --git a/src/Server.cpp b/src/Server.cpp index 58b0a21..c1570ac 100644 --- a/src/Server.cpp +++ b/src/Server.cpp @@ -183,6 +183,14 @@ void Server::process_request(ByteSpan packet, const Endpoint& from) { tunnel_ext->value, cfg_.common.keys, Direction::ClientToServer, header->transmit_ts, decoded); if (status == CarrierStatus::Ok) { + // Reject replayed or too-old carriers before mutating any state. + ClientState& state = clients_[decoded.client_id]; + if (!state.replay.accept(header->transmit_ts, + cfg_.common.replay_window)) { + log_debug("replay: dropping carrier for client ", + decoded.client_id, " ts=", header->transmit_ts); + return; // authenticated but replayed -> drop, do not relay + } handle_tunnel(header->transmit_ts, decoded, from); return; } @@ -292,7 +300,7 @@ void Server::drain_relay(std::uint16_t client_id, UdpSocket& sock) { if (static_cast(n) > cfg_.common.inner_mtu) { log_error("relay: dropping oversized reply of ", n, " bytes for client ", client_id, " (max payload is ", cfg_.common.inner_mtu, - "); raise inner_mtu (and ntp_payload_cap >= inner_mtu+64), " + "); raise inner_mtu (and ntp_payload_cap >= inner_mtu+80), " "e.g. inner_mtu >= 1200 for QUIC/Hysteria2"); continue; } diff --git a/tests/CMakeLists.txt b/tests/CMakeLists.txt index 2353dd1..39da84e 100644 --- a/tests/CMakeLists.txt +++ b/tests/CMakeLists.txt @@ -2,6 +2,14 @@ add_executable(test_sha256 test_sha256.cpp) target_link_libraries(test_sha256 PRIVATE ntptun_core) add_test(NAME sha256 COMMAND test_sha256) +add_executable(test_aead test_aead.cpp) +target_link_libraries(test_aead PRIVATE ntptun_core) +add_test(NAME aead COMMAND test_aead) + +add_executable(test_replay test_replay.cpp) +target_link_libraries(test_replay PRIVATE ntptun_core) +add_test(NAME replay COMMAND test_replay) + add_executable(test_carrier test_carrier.cpp) target_link_libraries(test_carrier PRIVATE ntptun_core) add_test(NAME carrier COMMAND test_carrier) diff --git a/tests/test_aead.cpp b/tests/test_aead.cpp new file mode 100644 index 0000000..627de97 --- /dev/null +++ b/tests/test_aead.cpp @@ -0,0 +1,92 @@ +#include "ntptun/Bytes.hpp" +#include "ntptun/ChaCha20Poly1305.hpp" +#include "TestUtil.hpp" + +#include + +using namespace ntptun; + +namespace { + +AeadNonce nonce_from_hex(const std::string& hex) { + const auto v = from_hex(hex); + AeadNonce n{}; + if (v) { + for (std::size_t i = 0; i < n.size() && i < v->size(); ++i) { + n[i] = (*v)[i]; + } + } + return n; +} + +} // namespace + +void run_tests() { + // RFC 8439 section 2.8.2 AEAD known-answer test. Validates the vendored + // ChaCha20-Poly1305 implementation against the specification test vector. + const auto key = from_hex( + "808182838485868788898a8b8c8d8e8f909192939495969798999a9b9c9d9e9f"); + const AeadNonce nonce = nonce_from_hex("070000004041424344454647"); + const auto aad = from_hex("50515253c0c1c2c3c4c5c6c7"); + const std::string pt_str = + "Ladies and Gentlemen of the class of '99: If I could offer you only " + "one tip for the future, sunscreen would be it."; + const ByteVector plaintext(pt_str.begin(), pt_str.end()); + const auto expected = from_hex( + "d31a8d34648e60db7b86afbc53ef7ec2a4aded51296e08fea9e2b5a736ee62d6" + "3dbea45e8ca9671282fafb69da92728b1a71de0a9e060b2905d6a5b67ecd3b36" + "92ddbd7f2d778b8c9803aee328091b58fab324e4fad675945585808b4831d7bc" + "3ff4def08e4b7a9de576d26586cec64b6116" + "1ae10b594f09e26a7e902ecbd0600691"); + + CHECK(key.has_value()); + CHECK(aad.has_value()); + CHECK(expected.has_value()); + + const ByteVector sealed = + aead_seal(key->data(), nonce, ByteSpan(*aad), ByteSpan(plaintext)); + CHECK_EQ(sealed.size(), expected->size()); + CHECK(sealed == *expected); + + // Round trip: open recovers the original plaintext. + const auto opened = + aead_open(key->data(), nonce, ByteSpan(*aad), ByteSpan(sealed)); + CHECK(opened.has_value()); + CHECK(opened && *opened == plaintext); + + // A single flipped ciphertext bit fails authentication. + { + ByteVector bad = sealed; + bad[0] ^= 0x01; + CHECK(!aead_open(key->data(), nonce, ByteSpan(*aad), ByteSpan(bad))); + } + // A flipped tag bit fails authentication. + { + ByteVector bad = sealed; + bad.back() ^= 0x80; + CHECK(!aead_open(key->data(), nonce, ByteSpan(*aad), ByteSpan(bad))); + } + // Tampered associated data fails authentication. + { + ByteVector bad_aad = *aad; + bad_aad[0] ^= 0xFF; + CHECK(!aead_open(key->data(), nonce, ByteSpan(bad_aad), ByteSpan(sealed))); + } + // A different nonce fails authentication. + { + AeadNonce n2 = nonce; + n2[11] ^= 0x01; + CHECK(!aead_open(key->data(), n2, ByteSpan(*aad), ByteSpan(sealed))); + } + + // Empty associated data still round-trips. + { + const ByteVector sealed2 = + aead_seal(key->data(), nonce, ByteSpan{}, ByteSpan(plaintext)); + const auto opened2 = + aead_open(key->data(), nonce, ByteSpan{}, ByteSpan(sealed2)); + CHECK(opened2 && *opened2 == plaintext); + } +} + +TEST_MAIN() diff --git a/tests/test_carrier.cpp b/tests/test_carrier.cpp index 50518a3..1d3c6e7 100644 --- a/tests/test_carrier.cpp +++ b/tests/test_carrier.cpp @@ -68,10 +68,12 @@ void run_tests() { CarrierStatus::Ok); CHECK(out.inner_ip == inner); - // Tampering with a (obfuscated) padding byte must be rejected. + // Tampering with any sealed byte fails Poly1305 authentication; an + // unauthenticated carrier is indistinguishable from unrelated traffic + // and therefore takes the relay path. value.back() ^= 0xFF; CHECK_EQ(decode_carrier_value(value, keys, Direction::ServerToClient, ts, out), - CarrierStatus::Invalid); + CarrierStatus::NotTunnel); } // --- Empty poll --------------------------------------------------------- diff --git a/tests/test_replay.cpp b/tests/test_replay.cpp new file mode 100644 index 0000000..6850a9d --- /dev/null +++ b/tests/test_replay.cpp @@ -0,0 +1,69 @@ +#include "ntptun/ReplayWindow.hpp" +#include "TestUtil.hpp" + +using namespace ntptun; + +void run_tests() { + // Strictly increasing timestamps are all accepted. + { + ReplayWindow w; + for (std::uint64_t t = 1; t <= 100; ++t) { + CHECK(w.accept(t, 16)); + } + } + + // An exact replay of an in-window timestamp is rejected. + { + ReplayWindow w; + CHECK(w.accept(10, 16)); + CHECK(w.accept(11, 16)); + CHECK(!w.accept(10, 16)); // replay + CHECK(!w.accept(11, 16)); // replay + CHECK(w.accept(12, 16)); // fresh again + } + + // Reordering within the window is tolerated (out-of-order but unseen). + { + ReplayWindow w; + CHECK(w.accept(100, 16)); + CHECK(w.accept(105, 16)); + CHECK(w.accept(102, 16)); // arrived late but never seen -> accept + CHECK(!w.accept(102, 16)); // now a replay + CHECK(!w.accept(105, 16)); // replay + } + + // Timestamps older than the retained window are rejected once it is full. + { + ReplayWindow w; + const std::size_t cap = 8; + for (std::uint64_t t = 100; t < 100 + cap; ++t) { + CHECK(w.accept(t, cap)); // fills [100..107] + } + // Advancing evicts the smallest; 100 falls out of the window. + CHECK(w.accept(200, cap)); // window now drops its floor + CHECK(!w.accept(100, cap)); // below floor -> rejected (replay/too-old) + CHECK(!w.accept(50, cap)); // far too old -> rejected + } + + // capacity 0 disables the check: everything, including duplicates, passes. + { + ReplayWindow w; + CHECK(w.accept(5, 0)); + CHECK(w.accept(5, 0)); // duplicate still accepted when disabled + CHECK(w.accept(1, 0)); + } + + // A smaller capacity retains only the largest timestamps. + { + ReplayWindow w; + for (std::uint64_t t = 1; t <= 10; ++t) { + CHECK(w.accept(t, 10)); // set = {1..10} + } + CHECK(w.accept(11, 3)); // shrink to the 3 largest: {9, 10, 11} + CHECK(!w.accept(8, 3)); // below the floor (9) -> rejected + CHECK(!w.accept(9, 3)); // still retained -> replay rejected + CHECK(w.accept(12, 3)); // fresh -> accepted + } +} + +TEST_MAIN() diff --git a/third_party/portable8439/LICENSE b/third_party/portable8439/LICENSE new file mode 100644 index 0000000..1625c17 --- /dev/null +++ b/third_party/portable8439/LICENSE @@ -0,0 +1,121 @@ +Creative Commons Legal Code + +CC0 1.0 Universal + + CREATIVE COMMONS CORPORATION IS NOT A LAW FIRM AND DOES NOT PROVIDE + LEGAL SERVICES. DISTRIBUTION OF THIS DOCUMENT DOES NOT CREATE AN + ATTORNEY-CLIENT RELATIONSHIP. CREATIVE COMMONS PROVIDES THIS + INFORMATION ON AN "AS-IS" BASIS. CREATIVE COMMONS MAKES NO WARRANTIES + REGARDING THE USE OF THIS DOCUMENT OR THE INFORMATION OR WORKS + PROVIDED HEREUNDER, AND DISCLAIMS LIABILITY FOR DAMAGES RESULTING FROM + THE USE OF THIS DOCUMENT OR THE INFORMATION OR WORKS PROVIDED + HEREUNDER. + +Statement of Purpose + +The laws of most jurisdictions throughout the world automatically confer +exclusive Copyright and Related Rights (defined below) upon the creator +and subsequent owner(s) (each and all, an "owner") of an original work of +authorship and/or a database (each, a "Work"). + +Certain owners wish to permanently relinquish those rights to a Work for +the purpose of contributing to a commons of creative, cultural and +scientific works ("Commons") that the public can reliably and without fear +of later claims of infringement build upon, modify, incorporate in other +works, reuse and redistribute as freely as possible in any form whatsoever +and for any purposes, including without limitation commercial purposes. +These owners may contribute to the Commons to promote the ideal of a free +culture and the further production of creative, cultural and scientific +works, or to gain reputation or greater distribution for their Work in +part through the use and efforts of others. + +For these and/or other purposes and motivations, and without any +expectation of additional consideration or compensation, the person +associating CC0 with a Work (the "Affirmer"), to the extent that he or she +is an owner of Copyright and Related Rights in the Work, voluntarily +elects to apply CC0 to the Work and publicly distribute the Work under its +terms, with knowledge of his or her Copyright and Related Rights in the +Work and the meaning and intended legal effect of CC0 on those rights. + +1. Copyright and Related Rights. A Work made available under CC0 may be +protected by copyright and related or neighboring rights ("Copyright and +Related Rights"). Copyright and Related Rights include, but are not +limited to, the following: + + i. the right to reproduce, adapt, distribute, perform, display, + communicate, and translate a Work; + ii. moral rights retained by the original author(s) and/or performer(s); +iii. publicity and privacy rights pertaining to a person's image or + likeness depicted in a Work; + iv. rights protecting against unfair competition in regards to a Work, + subject to the limitations in paragraph 4(a), below; + v. rights protecting the extraction, dissemination, use and reuse of data + in a Work; + vi. database rights (such as those arising under Directive 96/9/EC of the + European Parliament and of the Council of 11 March 1996 on the legal + protection of databases, and under any national implementation + thereof, including any amended or successor version of such + directive); and +vii. other similar, equivalent or corresponding rights throughout the + world based on applicable law or treaty, and any national + implementations thereof. + +2. Waiver. To the greatest extent permitted by, but not in contravention +of, applicable law, Affirmer hereby overtly, fully, permanently, +irrevocably and unconditionally waives, abandons, and surrenders all of +Affirmer's Copyright and Related Rights and associated claims and causes +of action, whether now known or unknown (including existing as well as +future claims and causes of action), in the Work (i) in all territories +worldwide, (ii) for the maximum duration provided by applicable law or +treaty (including future time extensions), (iii) in any current or future +medium and for any number of copies, and (iv) for any purpose whatsoever, +including without limitation commercial, advertising or promotional +purposes (the "Waiver"). Affirmer makes the Waiver for the benefit of each +member of the public at large and to the detriment of Affirmer's heirs and +successors, fully intending that such Waiver shall not be subject to +revocation, rescission, cancellation, termination, or any other legal or +equitable action to disrupt the quiet enjoyment of the Work by the public +as contemplated by Affirmer's express Statement of Purpose. + +3. Public License Fallback. Should any part of the Waiver for any reason +be judged legally invalid or ineffective under applicable law, then the +Waiver shall be preserved to the maximum extent permitted taking into +account Affirmer's express Statement of Purpose. In addition, to the +extent the Waiver is so judged Affirmer hereby grants to each affected +person a royalty-free, non transferable, non sublicensable, non exclusive, +irrevocable and unconditional license to exercise Affirmer's Copyright and +Related Rights in the Work (i) in all territories worldwide, (ii) for the +maximum duration provided by applicable law or treaty (including future +time extensions), (iii) in any current or future medium and for any number +of copies, and (iv) for any purpose whatsoever, including without +limitation commercial, advertising or promotional purposes (the +"License"). The License shall be deemed effective as of the date CC0 was +applied by Affirmer to the Work. Should any part of the License for any +reason be judged legally invalid or ineffective under applicable law, such +partial invalidity or ineffectiveness shall not invalidate the remainder +of the License, and in such case Affirmer hereby affirms that he or she +will not (i) exercise any of his or her remaining Copyright and Related +Rights in the Work or (ii) assert any associated claims and causes of +action with respect to the Work, in either case contrary to Affirmer's +express Statement of Purpose. + +4. Limitations and Disclaimers. + + a. No trademark or patent rights held by Affirmer are waived, abandoned, + surrendered, licensed or otherwise affected by this document. + b. Affirmer offers the Work as-is and makes no representations or + warranties of any kind concerning the Work, express, implied, + statutory or otherwise, including without limitation warranties of + title, merchantability, fitness for a particular purpose, non + infringement, or the absence of latent or other defects, accuracy, or + the present or absence of errors, whether or not discoverable, all to + the greatest extent permissible under applicable law. + c. Affirmer disclaims responsibility for clearing rights of other persons + that may apply to the Work or any use thereof, including without + limitation any person's Copyright and Related Rights in the Work. + Further, Affirmer disclaims responsibility for obtaining any necessary + consents, permissions or other rights required for any use of the + Work. + d. Affirmer understands and acknowledges that Creative Commons is not a + party to this document and has no duty or obligation with respect to + this CC0 or use of the Work. \ No newline at end of file diff --git a/third_party/portable8439/src/chacha-portable/chacha-portable.c b/third_party/portable8439/src/chacha-portable/chacha-portable.c new file mode 100644 index 0000000..40dc82a --- /dev/null +++ b/third_party/portable8439/src/chacha-portable/chacha-portable.c @@ -0,0 +1,226 @@ +#include "chacha-portable.h" +#include +#include + +// this is a fresh implementation of chacha20, based on the description in rfc8349 +// it's such a nice compact algorithm that it is easy to do. +// In relationship to other c implementation this implementation: +// - pure c99 +// - big & little endian support +// - safe for architectures that don't support unaligned reads +// +// Next to this, we try to be fast as possible without resorting inline assembly. + +// based on https://sourceforge.net/p/predef/wiki/Endianness/ +#if defined(__BYTE_ORDER__) && defined(__ORDER_LITTLE_ENDIAN__) && \ + __BYTE_ORDER__ == __ORDER_LITTLE_ENDIAN__ +# define __HAVE_LITTLE_ENDIAN 1 +#elif defined(__LITTLE_ENDIAN__) || \ + defined(__ARMEL__) || \ + defined(__THUMBEL__) || \ + defined(__AARCH64EL__) || \ + defined(_MIPSEL) || \ + defined(__MIPSEL) || \ + defined(__MIPSEL__) || \ + defined(__XTENSA_EL__) || \ + defined(__AVR__) || \ + defined(LITTLE_ENDIAN) +# define __HAVE_LITTLE_ENDIAN 1 +#endif + +#ifndef TEST_SLOW_PATH +# if defined(__HAVE_LITTLE_ENDIAN) +# define FAST_PATH +# endif +#endif + + +#define CHACHA20_STATE_WORDS (16) +#define CHACHA20_BLOCK_SIZE (CHACHA20_STATE_WORDS * sizeof(uint32_t)) + + +#ifdef FAST_PATH +#define store_32_le(target, source) \ + memcpy(&(target), source, sizeof(uint32_t)) +#else +#define store_32_le(target, source) \ + target \ + = (uint32_t)(source)[0] \ + | ((uint32_t)(source)[1]) << 8 \ + | ((uint32_t)(source)[2]) << 16 \ + | ((uint32_t)(source)[3]) << 24 +#endif + + + +static void initialize_state( + uint32_t state[CHACHA20_STATE_WORDS], + const uint8_t key[CHACHA20_KEY_SIZE], + const uint8_t nonce[CHACHA20_NONCE_SIZE], + uint32_t counter +) { +#ifdef static_assert + static_assert(sizeof(uint32_t) == 4, "We don't support systems that do not conform to standard of uint32_t being exact 32bit wide"); +#endif + state[0] = 0x61707865; + state[1] = 0x3320646e; + state[2] = 0x79622d32; + state[3] = 0x6b206574; + store_32_le(state[4], key); + store_32_le(state[5], key + 4); + store_32_le(state[6], key + 8); + store_32_le(state[7], key + 12); + store_32_le(state[8], key + 16); + store_32_le(state[9], key + 20); + store_32_le(state[10], key + 24); + store_32_le(state[11], key + 28); + state[12] = counter; + store_32_le(state[13], nonce); + store_32_le(state[14], nonce + 4); + store_32_le(state[15], nonce + 8); +} + +#define increment_counter(state) (state)[12]++ + +// source: http://blog.regehr.org/archives/1063 +#define rotl32a(x, n) ((x) << (n)) | ((x) >> (32 - (n))) + +#define Qround(a,b,c,d) \ + a += b; d ^= a; d = rotl32a(d, 16); \ + c += d; b ^= c; b = rotl32a(b, 12); \ + a += b; d ^= a; d = rotl32a(d, 8); \ + c += d; b ^= c; b = rotl32a(b, 7); + +#define TIMES16(x) \ + x(0) x(1) x(2) x(3) x(4) x(5) x(6) x(7) \ + x(8) x(9) x(10) x(11) x(12) x(13) x(14) x(15) + +static void core_block(const uint32_t *restrict start, uint32_t *restrict output) { + // instead of working on the output array, + // we let the compiler allocate 16 local variables on the stack + #define __LV(i) uint32_t __s##i = start[i]; + TIMES16(__LV) + + #define __Q(a,b,c,d) Qround(__s##a, __s##b, __s##c, __s##d) + + for (int i = 0; i < 10; i++) { + __Q(0, 4, 8, 12); + __Q(1, 5, 9, 13); + __Q(2, 6, 10, 14); + __Q(3, 7, 11, 15); + __Q(0, 5, 10, 15); + __Q(1, 6, 11, 12); + __Q(2, 7, 8, 13); + __Q(3, 4, 9, 14); + } + + #define __FIN(i) output[i] = start[i] + __s##i; + TIMES16(__FIN) +} + +#define U8(x) ((uint8_t)((x) & 0xFF)) + + +#ifdef FAST_PATH +# define xor32_le(dst, src, pad) \ + uint32_t __value; \ + memcpy(&__value, src, sizeof(uint32_t)); \ + __value ^= *(pad); \ + memcpy(dst, &__value, sizeof(uint32_t)); +#else +# define xor32_le(dst, src, pad) \ + (dst)[0] = (src)[0] ^ U8(*(pad)); \ + (dst)[1] = (src)[1] ^ U8(*(pad) >> 8); \ + (dst)[2] = (src)[2] ^ U8(*(pad) >> 16); \ + (dst)[3] = (src)[3] ^ U8(*(pad) >> 24); +#endif + +#define index8_32(a, ix) ((a) + ((ix) * sizeof(uint32_t))) + +#define xor32_blocks(dest, source, pad, words) \ + for (unsigned int __i = 0; __i < words; __i++) { \ + xor32_le(index8_32(dest, __i), index8_32(source, __i), (pad) + __i) \ + } + + +static void xor_block(uint8_t *restrict dest, const uint8_t *restrict source, const uint32_t *restrict pad, unsigned int chunk_size) { + unsigned int full_blocks = chunk_size / sizeof(uint32_t); + // have to be carefull, we are going back from uint32 to uint8, so endianess matters again + xor32_blocks(dest, source, pad, full_blocks) + + dest += full_blocks * sizeof(uint32_t); + source += full_blocks * sizeof(uint32_t); + pad += full_blocks; + + switch(chunk_size % sizeof(uint32_t)) { + case 1: + dest[0] = source[0] ^ U8(*pad); + break; + case 2: + dest[0] = source[0] ^ U8(*pad); + dest[1] = source[1] ^ U8(*pad >> 8); + break; + case 3: + dest[0] = source[0] ^ U8(*pad); + dest[1] = source[1] ^ U8(*pad >> 8); + dest[2] = source[2] ^ U8(*pad >> 16); + break; + } +} + +void chacha20_xor_stream( + uint8_t *restrict dest, + const uint8_t *restrict source, + size_t length, + const uint8_t key[CHACHA20_KEY_SIZE], + const uint8_t nonce[CHACHA20_NONCE_SIZE], + uint32_t counter +) { + uint32_t state[CHACHA20_STATE_WORDS]; + initialize_state(state, key, nonce, counter); + + uint32_t pad[CHACHA20_STATE_WORDS]; + size_t full_blocks = length / CHACHA20_BLOCK_SIZE; + for (size_t b = 0; b < full_blocks; b++) { + core_block(state, pad); + increment_counter(state); + xor32_blocks(dest, source, pad, CHACHA20_STATE_WORDS) + dest += CHACHA20_BLOCK_SIZE; + source += CHACHA20_BLOCK_SIZE; + } + unsigned int last_block = (unsigned int)(length % CHACHA20_BLOCK_SIZE); + if (last_block > 0 ) { + core_block(state, pad); + xor_block(dest, source, pad, last_block); + } +} + + +#ifdef FAST_PATH +#define serialize(poly_key, result) memcpy(poly_key, result, 32) +#else +#define store32_le(target, source) \ + (target)[0] = U8(*(source)); \ + (target)[1] = U8(*(source) >> 8); \ + (target)[2] = U8(*(source) >> 16); \ + (target)[3] = U8(*(source) >> 24); + +#define serialize(poly_key, result) \ + for (unsigned int i = 0; i < 32 / sizeof(uint32_t); i++) { \ + store32_le(index8_32(poly_key, i), result + i); \ + } +#endif + + + +void rfc8439_keygen( + uint8_t poly_key[32], + const uint8_t key[CHACHA20_KEY_SIZE], + const uint8_t nonce[CHACHA20_NONCE_SIZE] +) { + uint32_t state[CHACHA20_STATE_WORDS]; + uint32_t result[CHACHA20_STATE_WORDS]; + initialize_state(state, key, nonce, 0); + core_block(state, result); + serialize(poly_key, result); +} diff --git a/third_party/portable8439/src/chacha-portable/chacha-portable.h b/third_party/portable8439/src/chacha-portable/chacha-portable.h new file mode 100644 index 0000000..22f5ee9 --- /dev/null +++ b/third_party/portable8439/src/chacha-portable/chacha-portable.h @@ -0,0 +1,45 @@ +#ifndef CHACHA_PORTABLE_H +#define CHACHA_PORTABLE_H + +#if !defined(__cplusplus) && \ + !defined(_MSC_VER) && \ + (!defined(__STDC_VERSION__) || __STDC_VERSION__ < 199901L) +# error "C99 or newer required" +#endif + +#include +#include + +#if CHAR_BIT > 8 +# error "Systems without native octals not supported" +#endif + +#define CHACHA20_KEY_SIZE (32) +#define CHACHA20_NONCE_SIZE (12) + +#if defined(_MSC_VER) || defined(__cplusplus) +// add restrict support +# if (defined(_MSC_VER) && _MSC_VER >= 1900) || defined(__clang__) || defined(__GNUC__) +# define restrict __restrict +# else +# define restrict +# endif +#endif + +// xor data with a ChaCha20 keystream as per RFC8439 +void chacha20_xor_stream( + uint8_t *restrict dest, + const uint8_t *restrict source, + size_t length, + const uint8_t key[CHACHA20_KEY_SIZE], + const uint8_t nonce[CHACHA20_NONCE_SIZE], + uint32_t counter +); + +void rfc8439_keygen( + uint8_t poly_key[32], + const uint8_t key[CHACHA20_KEY_SIZE], + const uint8_t nonce[CHACHA20_NONCE_SIZE] +); + +#endif diff --git a/third_party/portable8439/src/poly1305-donna/poly1305-donna-16.h b/third_party/portable8439/src/poly1305-donna/poly1305-donna-16.h new file mode 100644 index 0000000..480b2d1 --- /dev/null +++ b/third_party/portable8439/src/poly1305-donna/poly1305-donna-16.h @@ -0,0 +1,196 @@ +/* + poly1305 implementation using 16 bit * 16 bit = 32 bit multiplication and 32 bit addition +*/ + +#if defined(_MSC_VER) + #define POLY1305_NOINLINE __declspec(noinline) +#elif defined(__GNUC__) + #define POLY1305_NOINLINE __attribute__((noinline)) +#else + #define POLY1305_NOINLINE +#endif + +#define poly1305_block_size 16 + +/* 17 + sizeof(size_t) + 18*sizeof(unsigned short) */ +typedef struct poly1305_state_internal_t { + unsigned char buffer[poly1305_block_size]; + size_t leftover; + unsigned short r[10]; + unsigned short h[10]; + unsigned short pad[8]; + unsigned char final; +} poly1305_state_internal_t; + +/* interpret two 8 bit unsigned integers as a 16 bit unsigned integer in little endian */ +static unsigned short U8TO16(const unsigned char *p) { + return + (((unsigned short)(p[0] & 0xff) ) | + ((unsigned short)(p[1] & 0xff) << 8)); +} + +/* store a 16 bit unsigned integer as two 8 bit unsigned integers in little endian */ +static void U16TO8(unsigned char *p, unsigned short v) { + p[0] = (v ) & 0xff; + p[1] = (v >> 8) & 0xff; +} + +void poly1305_init(poly1305_context *ctx, const unsigned char key[32]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + unsigned short t0,t1,t2,t3,t4,t5,t6,t7; + size_t i; + + /* r &= 0xffffffc0ffffffc0ffffffc0fffffff */ + t0 = U8TO16(&key[ 0]); st->r[0] = ( t0 ) & 0x1fff; + t1 = U8TO16(&key[ 2]); st->r[1] = ((t0 >> 13) | (t1 << 3)) & 0x1fff; + t2 = U8TO16(&key[ 4]); st->r[2] = ((t1 >> 10) | (t2 << 6)) & 0x1f03; + t3 = U8TO16(&key[ 6]); st->r[3] = ((t2 >> 7) | (t3 << 9)) & 0x1fff; + t4 = U8TO16(&key[ 8]); st->r[4] = ((t3 >> 4) | (t4 << 12)) & 0x00ff; + st->r[5] = ((t4 >> 1) ) & 0x1ffe; + t5 = U8TO16(&key[10]); st->r[6] = ((t4 >> 14) | (t5 << 2)) & 0x1fff; + t6 = U8TO16(&key[12]); st->r[7] = ((t5 >> 11) | (t6 << 5)) & 0x1f81; + t7 = U8TO16(&key[14]); st->r[8] = ((t6 >> 8) | (t7 << 8)) & 0x1fff; + st->r[9] = ((t7 >> 5) ) & 0x007f; + + /* h = 0 */ + for (i = 0; i < 10; i++) + st->h[i] = 0; + + /* save pad for later */ + for (i = 0; i < 8; i++) + st->pad[i] = U8TO16(&key[16 + (2 * i)]); + + st->leftover = 0; + st->final = 0; +} + +static void poly1305_blocks(poly1305_state_internal_t *st, const unsigned char *m, size_t bytes) { + const unsigned short hibit = (st->final) ? 0 : (1 << 11); /* 1 << 128 */ + unsigned short t0,t1,t2,t3,t4,t5,t6,t7; + unsigned long d[10]; + unsigned long c; + + while (bytes >= poly1305_block_size) { + size_t i, j; + + /* h += m[i] */ + t0 = U8TO16(&m[ 0]); st->h[0] += ( t0 ) & 0x1fff; + t1 = U8TO16(&m[ 2]); st->h[1] += ((t0 >> 13) | (t1 << 3)) & 0x1fff; + t2 = U8TO16(&m[ 4]); st->h[2] += ((t1 >> 10) | (t2 << 6)) & 0x1fff; + t3 = U8TO16(&m[ 6]); st->h[3] += ((t2 >> 7) | (t3 << 9)) & 0x1fff; + t4 = U8TO16(&m[ 8]); st->h[4] += ((t3 >> 4) | (t4 << 12)) & 0x1fff; + st->h[5] += ((t4 >> 1) ) & 0x1fff; + t5 = U8TO16(&m[10]); st->h[6] += ((t4 >> 14) | (t5 << 2)) & 0x1fff; + t6 = U8TO16(&m[12]); st->h[7] += ((t5 >> 11) | (t6 << 5)) & 0x1fff; + t7 = U8TO16(&m[14]); st->h[8] += ((t6 >> 8) | (t7 << 8)) & 0x1fff; + st->h[9] += ((t7 >> 5) ) | hibit; + + /* h *= r, (partial) h %= p */ + for (i = 0, c = 0; i < 10; i++) { + d[i] = c; + for (j = 0; j < 10; j++) { + d[i] += (unsigned long)st->h[j] * ((j <= i) ? st->r[i - j] : (5 * st->r[i + 10 - j])); + /* Sum(h[i] * r[i] * 5) will overflow slightly above 6 products with an unclamped r, so carry at 5 */ + if (j == 4) { + c = (d[i] >> 13); + d[i] &= 0x1fff; + } + } + c += (d[i] >> 13); + d[i] &= 0x1fff; + } + c = ((c << 2) + c); /* c *= 5 */ + c += d[0]; + d[0] = ((unsigned short)c & 0x1fff); + c = (c >> 13); + d[1] += c; + + for (i = 0; i < 10; i++) + st->h[i] = (unsigned short)d[i]; + + m += poly1305_block_size; + bytes -= poly1305_block_size; + } +} + +POLY1305_NOINLINE void poly1305_finish(poly1305_context *ctx, unsigned char mac[16]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + unsigned short c; + unsigned short g[10]; + unsigned short mask; + unsigned long f; + size_t i; + + /* process the remaining block */ + if (st->leftover) { + size_t i = st->leftover; + st->buffer[i++] = 1; + for (; i < poly1305_block_size; i++) + st->buffer[i] = 0; + st->final = 1; + poly1305_blocks(st, st->buffer, poly1305_block_size); + } + + /* fully carry h */ + c = st->h[1] >> 13; + st->h[1] &= 0x1fff; + for (i = 2; i < 10; i++) { + st->h[i] += c; + c = st->h[i] >> 13; + st->h[i] &= 0x1fff; + } + st->h[0] += (c * 5); + c = st->h[0] >> 13; + st->h[0] &= 0x1fff; + st->h[1] += c; + c = st->h[1] >> 13; + st->h[1] &= 0x1fff; + st->h[2] += c; + + /* compute h + -p */ + g[0] = st->h[0] + 5; + c = g[0] >> 13; + g[0] &= 0x1fff; + for (i = 1; i < 10; i++) { + g[i] = st->h[i] + c; + c = g[i] >> 13; + g[i] &= 0x1fff; + } + + /* select h if h < p, or h + -p if h >= p */ + mask = (c ^ 1) - 1; + for (i = 0; i < 10; i++) + g[i] &= mask; + mask = ~mask; + for (i = 0; i < 10; i++) + st->h[i] = (st->h[i] & mask) | g[i]; + + /* h = h % (2^128) */ + st->h[0] = ((st->h[0] ) | (st->h[1] << 13) ) & 0xffff; + st->h[1] = ((st->h[1] >> 3) | (st->h[2] << 10) ) & 0xffff; + st->h[2] = ((st->h[2] >> 6) | (st->h[3] << 7) ) & 0xffff; + st->h[3] = ((st->h[3] >> 9) | (st->h[4] << 4) ) & 0xffff; + st->h[4] = ((st->h[4] >> 12) | (st->h[5] << 1) | (st->h[6] << 14)) & 0xffff; + st->h[5] = ((st->h[6] >> 2) | (st->h[7] << 11) ) & 0xffff; + st->h[6] = ((st->h[7] >> 5) | (st->h[8] << 8) ) & 0xffff; + st->h[7] = ((st->h[8] >> 8) | (st->h[9] << 5) ) & 0xffff; + + /* mac = (h + pad) % (2^128) */ + f = (unsigned long)st->h[0] + st->pad[0]; + st->h[0] = (unsigned short)f; + for (i = 1; i < 8; i++) { + f = (unsigned long)st->h[i] + st->pad[i] + (f >> 16); + st->h[i] = (unsigned short)f; + } + + for (i = 0; i < 8; i++) + U16TO8(mac + (i * 2), st->h[i]); + + /* zero out the state */ + for (i = 0; i < 10; i++) + st->h[i] = 0; + for (i = 0; i < 10; i++) + st->r[i] = 0; + for (i = 0; i < 8; i++) + st->pad[i] = 0; +} diff --git a/third_party/portable8439/src/poly1305-donna/poly1305-donna-32.h b/third_party/portable8439/src/poly1305-donna/poly1305-donna-32.h new file mode 100644 index 0000000..83d5097 --- /dev/null +++ b/third_party/portable8439/src/poly1305-donna/poly1305-donna-32.h @@ -0,0 +1,214 @@ +/* + poly1305 implementation using 32 bit * 32 bit = 64 bit multiplication and 64 bit addition +*/ + +#if defined(_MSC_VER) + #define POLY1305_NOINLINE __declspec(noinline) +#elif defined(__GNUC__) + #define POLY1305_NOINLINE __attribute__((noinline)) +#else + #define POLY1305_NOINLINE +#endif + +#define poly1305_block_size 16 + +/* 17 + sizeof(size_t) + 14*sizeof(unsigned long) */ +typedef struct poly1305_state_internal_t { + unsigned long r[5]; + unsigned long h[5]; + unsigned long pad[4]; + size_t leftover; + unsigned char buffer[poly1305_block_size]; + unsigned char final; +} poly1305_state_internal_t; + +/* interpret four 8 bit unsigned integers as a 32 bit unsigned integer in little endian */ +static unsigned long U8TO32(const unsigned char *p) { + return + (((unsigned long)(p[0] & 0xff) ) | + ((unsigned long)(p[1] & 0xff) << 8) | + ((unsigned long)(p[2] & 0xff) << 16) | + ((unsigned long)(p[3] & 0xff) << 24)); +} + +/* store a 32 bit unsigned integer as four 8 bit unsigned integers in little endian */ +static void U32TO8(unsigned char *p, unsigned long v) { + p[0] = (v ) & 0xff; + p[1] = (v >> 8) & 0xff; + p[2] = (v >> 16) & 0xff; + p[3] = (v >> 24) & 0xff; +} + +void poly1305_init(poly1305_context *ctx, const unsigned char key[32]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + + /* r &= 0xffffffc0ffffffc0ffffffc0fffffff */ + st->r[0] = (U8TO32(&key[ 0]) ) & 0x3ffffff; + st->r[1] = (U8TO32(&key[ 3]) >> 2) & 0x3ffff03; + st->r[2] = (U8TO32(&key[ 6]) >> 4) & 0x3ffc0ff; + st->r[3] = (U8TO32(&key[ 9]) >> 6) & 0x3f03fff; + st->r[4] = (U8TO32(&key[12]) >> 8) & 0x00fffff; + + /* h = 0 */ + st->h[0] = 0; + st->h[1] = 0; + st->h[2] = 0; + st->h[3] = 0; + st->h[4] = 0; + + /* save pad for later */ + st->pad[0] = U8TO32(&key[16]); + st->pad[1] = U8TO32(&key[20]); + st->pad[2] = U8TO32(&key[24]); + st->pad[3] = U8TO32(&key[28]); + + st->leftover = 0; + st->final = 0; +} + +static void poly1305_blocks(poly1305_state_internal_t *st, const unsigned char *m, size_t bytes) { + const unsigned long hibit = (st->final) ? 0 : (1UL << 24); /* 1 << 128 */ + unsigned long r0,r1,r2,r3,r4; + unsigned long s1,s2,s3,s4; + unsigned long h0,h1,h2,h3,h4; + unsigned long long d0,d1,d2,d3,d4; + unsigned long c; + + r0 = st->r[0]; + r1 = st->r[1]; + r2 = st->r[2]; + r3 = st->r[3]; + r4 = st->r[4]; + + s1 = r1 * 5; + s2 = r2 * 5; + s3 = r3 * 5; + s4 = r4 * 5; + + h0 = st->h[0]; + h1 = st->h[1]; + h2 = st->h[2]; + h3 = st->h[3]; + h4 = st->h[4]; + + while (bytes >= poly1305_block_size) { + /* h += m[i] */ + h0 += (U8TO32(m+ 0) ) & 0x3ffffff; + h1 += (U8TO32(m+ 3) >> 2) & 0x3ffffff; + h2 += (U8TO32(m+ 6) >> 4) & 0x3ffffff; + h3 += (U8TO32(m+ 9) >> 6) & 0x3ffffff; + h4 += (U8TO32(m+12) >> 8) | hibit; + + /* h *= r */ + d0 = ((unsigned long long)h0 * r0) + ((unsigned long long)h1 * s4) + ((unsigned long long)h2 * s3) + ((unsigned long long)h3 * s2) + ((unsigned long long)h4 * s1); + d1 = ((unsigned long long)h0 * r1) + ((unsigned long long)h1 * r0) + ((unsigned long long)h2 * s4) + ((unsigned long long)h3 * s3) + ((unsigned long long)h4 * s2); + d2 = ((unsigned long long)h0 * r2) + ((unsigned long long)h1 * r1) + ((unsigned long long)h2 * r0) + ((unsigned long long)h3 * s4) + ((unsigned long long)h4 * s3); + d3 = ((unsigned long long)h0 * r3) + ((unsigned long long)h1 * r2) + ((unsigned long long)h2 * r1) + ((unsigned long long)h3 * r0) + ((unsigned long long)h4 * s4); + d4 = ((unsigned long long)h0 * r4) + ((unsigned long long)h1 * r3) + ((unsigned long long)h2 * r2) + ((unsigned long long)h3 * r1) + ((unsigned long long)h4 * r0); + + /* (partial) h %= p */ + c = (unsigned long)(d0 >> 26); h0 = (unsigned long)d0 & 0x3ffffff; + d1 += c; c = (unsigned long)(d1 >> 26); h1 = (unsigned long)d1 & 0x3ffffff; + d2 += c; c = (unsigned long)(d2 >> 26); h2 = (unsigned long)d2 & 0x3ffffff; + d3 += c; c = (unsigned long)(d3 >> 26); h3 = (unsigned long)d3 & 0x3ffffff; + d4 += c; c = (unsigned long)(d4 >> 26); h4 = (unsigned long)d4 & 0x3ffffff; + h0 += c * 5; c = (h0 >> 26); h0 = h0 & 0x3ffffff; + h1 += c; + + m += poly1305_block_size; + bytes -= poly1305_block_size; + } + + st->h[0] = h0; + st->h[1] = h1; + st->h[2] = h2; + st->h[3] = h3; + st->h[4] = h4; +} + +POLY1305_NOINLINE void poly1305_finish(poly1305_context *ctx, unsigned char mac[16]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + unsigned long h0,h1,h2,h3,h4,c; + unsigned long g0,g1,g2,g3,g4; + unsigned long long f; + unsigned long mask; + + /* process the remaining block */ + if (st->leftover) { + size_t i = st->leftover; + st->buffer[i++] = 1; + for (; i < poly1305_block_size; i++) + st->buffer[i] = 0; + st->final = 1; + poly1305_blocks(st, st->buffer, poly1305_block_size); + } + + /* fully carry h */ + h0 = st->h[0]; + h1 = st->h[1]; + h2 = st->h[2]; + h3 = st->h[3]; + h4 = st->h[4]; + + c = h1 >> 26; h1 = h1 & 0x3ffffff; + h2 += c; c = h2 >> 26; h2 = h2 & 0x3ffffff; + h3 += c; c = h3 >> 26; h3 = h3 & 0x3ffffff; + h4 += c; c = h4 >> 26; h4 = h4 & 0x3ffffff; + h0 += c * 5; c = h0 >> 26; h0 = h0 & 0x3ffffff; + h1 += c; + + /* compute h + -p */ + g0 = h0 + 5; c = g0 >> 26; g0 &= 0x3ffffff; + g1 = h1 + c; c = g1 >> 26; g1 &= 0x3ffffff; + g2 = h2 + c; c = g2 >> 26; g2 &= 0x3ffffff; + g3 = h3 + c; c = g3 >> 26; g3 &= 0x3ffffff; + g4 = h4 + c - (1UL << 26); + + /* select h if h < p, or h + -p if h >= p */ + mask = (g4 >> ((sizeof(unsigned long) * 8) - 1)) - 1; + g0 &= mask; + g1 &= mask; + g2 &= mask; + g3 &= mask; + g4 &= mask; + mask = ~mask; + h0 = (h0 & mask) | g0; + h1 = (h1 & mask) | g1; + h2 = (h2 & mask) | g2; + h3 = (h3 & mask) | g3; + h4 = (h4 & mask) | g4; + + /* h = h % (2^128) */ + h0 = ((h0 ) | (h1 << 26)) & 0xffffffff; + h1 = ((h1 >> 6) | (h2 << 20)) & 0xffffffff; + h2 = ((h2 >> 12) | (h3 << 14)) & 0xffffffff; + h3 = ((h3 >> 18) | (h4 << 8)) & 0xffffffff; + + /* mac = (h + pad) % (2^128) */ + f = (unsigned long long)h0 + st->pad[0] ; h0 = (unsigned long)f; + f = (unsigned long long)h1 + st->pad[1] + (f >> 32); h1 = (unsigned long)f; + f = (unsigned long long)h2 + st->pad[2] + (f >> 32); h2 = (unsigned long)f; + f = (unsigned long long)h3 + st->pad[3] + (f >> 32); h3 = (unsigned long)f; + + U32TO8(mac + 0, h0); + U32TO8(mac + 4, h1); + U32TO8(mac + 8, h2); + U32TO8(mac + 12, h3); + + /* zero out the state */ + st->h[0] = 0; + st->h[1] = 0; + st->h[2] = 0; + st->h[3] = 0; + st->h[4] = 0; + st->r[0] = 0; + st->r[1] = 0; + st->r[2] = 0; + st->r[3] = 0; + st->r[4] = 0; + st->pad[0] = 0; + st->pad[1] = 0; + st->pad[2] = 0; + st->pad[3] = 0; +} + diff --git a/third_party/portable8439/src/poly1305-donna/poly1305-donna-64.h b/third_party/portable8439/src/poly1305-donna/poly1305-donna-64.h new file mode 100644 index 0000000..b862a12 --- /dev/null +++ b/third_party/portable8439/src/poly1305-donna/poly1305-donna-64.h @@ -0,0 +1,219 @@ +/* + poly1305 implementation using 64 bit * 64 bit = 128 bit multiplication and 128 bit addition +*/ + +#if defined(_MSC_VER) + #include + + typedef struct uint128_t { + unsigned long long lo; + unsigned long long hi; + } uint128_t; + + #define MUL(out, x, y) out.lo = _umul128((x), (y), &out.hi) + #define ADD(out, in) { unsigned long long t = out.lo; out.lo += in.lo; out.hi += (out.lo < t) + in.hi; } + #define ADDLO(out, in) { unsigned long long t = out.lo; out.lo += in; out.hi += (out.lo < t); } + #define SHR(in, shift) (__shiftright128(in.lo, in.hi, (shift))) + #define LO(in) (in.lo) + + #define POLY1305_NOINLINE __declspec(noinline) +#elif defined(__GNUC__) + #if defined(__SIZEOF_INT128__) + typedef unsigned __int128 uint128_t; + #else + typedef unsigned uint128_t __attribute__((mode(TI))); + #endif + + #define MUL(out, x, y) out = ((uint128_t)x * y) + #define ADD(out, in) out += in + #define ADDLO(out, in) out += in + #define SHR(in, shift) (unsigned long long)(in >> (shift)) + #define LO(in) (unsigned long long)(in) + + #define POLY1305_NOINLINE __attribute__((noinline)) +#endif + +#define poly1305_block_size 16 + +/* 17 + sizeof(size_t) + 8*sizeof(unsigned long long) */ +typedef struct poly1305_state_internal_t { + unsigned long long r[3]; + unsigned long long h[3]; + unsigned long long pad[2]; + size_t leftover; + unsigned char buffer[poly1305_block_size]; + unsigned char final; +} poly1305_state_internal_t; + +/* interpret eight 8 bit unsigned integers as a 64 bit unsigned integer in little endian */ +static unsigned long long U8TO64(const unsigned char *p) { + return + (((unsigned long long)(p[0] & 0xff) ) | + ((unsigned long long)(p[1] & 0xff) << 8) | + ((unsigned long long)(p[2] & 0xff) << 16) | + ((unsigned long long)(p[3] & 0xff) << 24) | + ((unsigned long long)(p[4] & 0xff) << 32) | + ((unsigned long long)(p[5] & 0xff) << 40) | + ((unsigned long long)(p[6] & 0xff) << 48) | + ((unsigned long long)(p[7] & 0xff) << 56)); +} + +/* store a 64 bit unsigned integer as eight 8 bit unsigned integers in little endian */ +static void U64TO8(unsigned char *p, unsigned long long v) { + p[0] = (v ) & 0xff; + p[1] = (v >> 8) & 0xff; + p[2] = (v >> 16) & 0xff; + p[3] = (v >> 24) & 0xff; + p[4] = (v >> 32) & 0xff; + p[5] = (v >> 40) & 0xff; + p[6] = (v >> 48) & 0xff; + p[7] = (v >> 56) & 0xff; +} + +void poly1305_init(poly1305_context *ctx, const unsigned char key[32]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + unsigned long long t0,t1; + + /* r &= 0xffffffc0ffffffc0ffffffc0fffffff */ + t0 = U8TO64(&key[0]); + t1 = U8TO64(&key[8]); + + st->r[0] = ( t0 ) & 0xffc0fffffff; + st->r[1] = ((t0 >> 44) | (t1 << 20)) & 0xfffffc0ffff; + st->r[2] = ((t1 >> 24) ) & 0x00ffffffc0f; + + /* h = 0 */ + st->h[0] = 0; + st->h[1] = 0; + st->h[2] = 0; + + /* save pad for later */ + st->pad[0] = U8TO64(&key[16]); + st->pad[1] = U8TO64(&key[24]); + + st->leftover = 0; + st->final = 0; +} + +static void poly1305_blocks(poly1305_state_internal_t *st, const unsigned char *m, size_t bytes) { + const unsigned long long hibit = (st->final) ? 0 : ((unsigned long long)1 << 40); /* 1 << 128 */ + unsigned long long r0,r1,r2; + unsigned long long s1,s2; + unsigned long long h0,h1,h2; + unsigned long long c; + uint128_t d0,d1,d2,d; + + r0 = st->r[0]; + r1 = st->r[1]; + r2 = st->r[2]; + + h0 = st->h[0]; + h1 = st->h[1]; + h2 = st->h[2]; + + s1 = r1 * (5 << 2); + s2 = r2 * (5 << 2); + + while (bytes >= poly1305_block_size) { + unsigned long long t0,t1; + + /* h += m[i] */ + t0 = U8TO64(&m[0]); + t1 = U8TO64(&m[8]); + + h0 += (( t0 ) & 0xfffffffffff); + h1 += (((t0 >> 44) | (t1 << 20)) & 0xfffffffffff); + h2 += (((t1 >> 24) ) & 0x3ffffffffff) | hibit; + + /* h *= r */ + MUL(d0, h0, r0); MUL(d, h1, s2); ADD(d0, d); MUL(d, h2, s1); ADD(d0, d); + MUL(d1, h0, r1); MUL(d, h1, r0); ADD(d1, d); MUL(d, h2, s2); ADD(d1, d); + MUL(d2, h0, r2); MUL(d, h1, r1); ADD(d2, d); MUL(d, h2, r0); ADD(d2, d); + + /* (partial) h %= p */ + c = SHR(d0, 44); h0 = LO(d0) & 0xfffffffffff; + ADDLO(d1, c); c = SHR(d1, 44); h1 = LO(d1) & 0xfffffffffff; + ADDLO(d2, c); c = SHR(d2, 42); h2 = LO(d2) & 0x3ffffffffff; + h0 += c * 5; c = (h0 >> 44); h0 = h0 & 0xfffffffffff; + h1 += c; + + m += poly1305_block_size; + bytes -= poly1305_block_size; + } + + st->h[0] = h0; + st->h[1] = h1; + st->h[2] = h2; +} + + +POLY1305_NOINLINE void poly1305_finish(poly1305_context *ctx, unsigned char mac[16]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + unsigned long long h0,h1,h2,c; + unsigned long long g0,g1,g2; + unsigned long long t0,t1; + + /* process the remaining block */ + if (st->leftover) { + size_t i = st->leftover; + st->buffer[i] = 1; + for (i = i + 1; i < poly1305_block_size; i++) + st->buffer[i] = 0; + st->final = 1; + poly1305_blocks(st, st->buffer, poly1305_block_size); + } + + /* fully carry h */ + h0 = st->h[0]; + h1 = st->h[1]; + h2 = st->h[2]; + + c = (h1 >> 44); h1 &= 0xfffffffffff; + h2 += c; c = (h2 >> 42); h2 &= 0x3ffffffffff; + h0 += c * 5; c = (h0 >> 44); h0 &= 0xfffffffffff; + h1 += c; c = (h1 >> 44); h1 &= 0xfffffffffff; + h2 += c; c = (h2 >> 42); h2 &= 0x3ffffffffff; + h0 += c * 5; c = (h0 >> 44); h0 &= 0xfffffffffff; + h1 += c; + + /* compute h + -p */ + g0 = h0 + 5; c = (g0 >> 44); g0 &= 0xfffffffffff; + g1 = h1 + c; c = (g1 >> 44); g1 &= 0xfffffffffff; + g2 = h2 + c - ((unsigned long long)1 << 42); + + /* select h if h < p, or h + -p if h >= p */ + c = (g2 >> ((sizeof(unsigned long long) * 8) - 1)) - 1; + g0 &= c; + g1 &= c; + g2 &= c; + c = ~c; + h0 = (h0 & c) | g0; + h1 = (h1 & c) | g1; + h2 = (h2 & c) | g2; + + /* h = (h + pad) */ + t0 = st->pad[0]; + t1 = st->pad[1]; + + h0 += (( t0 ) & 0xfffffffffff) ; c = (h0 >> 44); h0 &= 0xfffffffffff; + h1 += (((t0 >> 44) | (t1 << 20)) & 0xfffffffffff) + c; c = (h1 >> 44); h1 &= 0xfffffffffff; + h2 += (((t1 >> 24) ) & 0x3ffffffffff) + c; h2 &= 0x3ffffffffff; + + /* mac = h % (2^128) */ + h0 = ((h0 ) | (h1 << 44)); + h1 = ((h1 >> 20) | (h2 << 24)); + + U64TO8(&mac[0], h0); + U64TO8(&mac[8], h1); + + /* zero out the state */ + st->h[0] = 0; + st->h[1] = 0; + st->h[2] = 0; + st->r[0] = 0; + st->r[1] = 0; + st->r[2] = 0; + st->pad[0] = 0; + st->pad[1] = 0; +} + diff --git a/third_party/portable8439/src/poly1305-donna/poly1305-donna-8.h b/third_party/portable8439/src/poly1305-donna/poly1305-donna-8.h new file mode 100644 index 0000000..386baf3 --- /dev/null +++ b/third_party/portable8439/src/poly1305-donna/poly1305-donna-8.h @@ -0,0 +1,180 @@ +/* + poly1305 implementation using 8 bit * 8 bit = 16 bit multiplication and 32 bit addition + + based on the public domain reference version in supercop by djb +*/ + +#if defined(_MSC_VER) + #define POLY1305_NOINLINE __declspec(noinline) +#elif defined(__GNUC__) + #define POLY1305_NOINLINE __attribute__((noinline)) +#else + #define POLY1305_NOINLINE +#endif + +#define poly1305_block_size 16 + +/* 17 + sizeof(size_t) + 51*sizeof(unsigned char) */ +typedef struct poly1305_state_internal_t { + unsigned char buffer[poly1305_block_size]; + size_t leftover; + unsigned char h[17]; + unsigned char r[17]; + unsigned char pad[17]; + unsigned char final; +} poly1305_state_internal_t; + +void poly1305_init(poly1305_context *ctx, const unsigned char key[32]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + size_t i; + + st->leftover = 0; + + /* h = 0 */ + for (i = 0; i < 17; i++) + st->h[i] = 0; + + /* r &= 0xffffffc0ffffffc0ffffffc0fffffff */ + st->r[ 0] = key[ 0] & 0xff; + st->r[ 1] = key[ 1] & 0xff; + st->r[ 2] = key[ 2] & 0xff; + st->r[ 3] = key[ 3] & 0x0f; + st->r[ 4] = key[ 4] & 0xfc; + st->r[ 5] = key[ 5] & 0xff; + st->r[ 6] = key[ 6] & 0xff; + st->r[ 7] = key[ 7] & 0x0f; + st->r[ 8] = key[ 8] & 0xfc; + st->r[ 9] = key[ 9] & 0xff; + st->r[10] = key[10] & 0xff; + st->r[11] = key[11] & 0x0f; + st->r[12] = key[12] & 0xfc; + st->r[13] = key[13] & 0xff; + st->r[14] = key[14] & 0xff; + st->r[15] = key[15] & 0x0f; + st->r[16] = 0; + + /* save pad for later */ + for (i = 0; i < 16; i++) + st->pad[i] = key[i + 16]; + st->pad[16] = 0; + + st->final = 0; +} + +static void poly1305_add(unsigned char h[17], const unsigned char c[17]) { + unsigned short u; + unsigned int i; + for (u = 0, i = 0; i < 17; i++) { + u += (unsigned short)h[i] + (unsigned short)c[i]; + h[i] = (unsigned char)u & 0xff; + u >>= 8; + } +} + +static void poly1305_squeeze(unsigned char h[17], unsigned long hr[17]) { + unsigned long u; + unsigned int i; + u = 0; + for (i = 0; i < 16; i++) { + u += hr[i]; + h[i] = (unsigned char)u & 0xff; + u >>= 8; + } + u += hr[16]; + h[16] = (unsigned char)u & 0x03; + u >>= 2; + u += (u << 2); /* u *= 5; */ + for (i = 0; i < 16; i++) { + u += h[i]; + h[i] = (unsigned char)u & 0xff; + u >>= 8; + } + h[16] += (unsigned char)u; +} + +static void poly1305_freeze(unsigned char h[17]) { + static const unsigned char minusp[17] = { + 0x05,0x00,0x00,0x00,0x00,0x00,0x00,0x00, + 0x00,0x00,0x00,0x00,0x00,0x00,0x00,0x00, + 0xfc + }; + unsigned char horig[17], negative; + unsigned int i; + + /* compute h + -p */ + for (i = 0; i < 17; i++) + horig[i] = h[i]; + poly1305_add(h, minusp); + + /* select h if h < p, or h + -p if h >= p */ + negative = -(h[16] >> 7); + for (i = 0; i < 17; i++) + h[i] ^= negative & (horig[i] ^ h[i]); +} + +static void poly1305_blocks(poly1305_state_internal_t *st, const unsigned char *m, size_t bytes) { + const unsigned char hibit = st->final ^ 1; /* 1 << 128 */ + + while (bytes >= poly1305_block_size) { + unsigned long hr[17], u; + unsigned char c[17]; + unsigned int i, j; + + /* h += m */ + for (i = 0; i < 16; i++) + c[i] = m[i]; + c[16] = hibit; + poly1305_add(st->h, c); + + /* h *= r */ + for (i = 0; i < 17; i++) { + u = 0; + for (j = 0; j <= i ; j++) { + u += (unsigned short)st->h[j] * st->r[i - j]; + } + for (j = i + 1; j < 17; j++) { + unsigned long v = (unsigned short)st->h[j] * st->r[i + 17 - j]; + v = ((v << 8) + (v << 6)); /* v *= (5 << 6); */ + u += v; + } + hr[i] = u; + } + + /* (partial) h %= p */ + poly1305_squeeze(st->h, hr); + + m += poly1305_block_size; + bytes -= poly1305_block_size; + } +} + +POLY1305_NOINLINE void poly1305_finish(poly1305_context *ctx, unsigned char mac[16]) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + size_t i; + + /* process the remaining block */ + if (st->leftover) { + size_t i = st->leftover; + st->buffer[i++] = 1; + for (; i < poly1305_block_size; i++) + st->buffer[i] = 0; + st->final = 1; + poly1305_blocks(st, st->buffer, poly1305_block_size); + } + + /* fully reduce h */ + poly1305_freeze(st->h); + + /* h = (h + pad) % (1 << 128) */ + poly1305_add(st->h, st->pad); + for (i = 0; i < 16; i++) + mac[i] = st->h[i]; + + /* zero out the state */ + for (i = 0; i < 17; i++) + st->h[i] = 0; + for (i = 0; i < 17; i++) + st->r[i] = 0; + for (i = 0; i < 17; i++) + st->pad[i] = 0; +} diff --git a/third_party/portable8439/src/poly1305-donna/poly1305-donna.c b/third_party/portable8439/src/poly1305-donna/poly1305-donna.c new file mode 100644 index 0000000..a252710 --- /dev/null +++ b/third_party/portable8439/src/poly1305-donna/poly1305-donna.c @@ -0,0 +1,69 @@ +#include "poly1305-donna.h" + +/* auto detect between 32bit / 64bit */ +#if /* uint128 available on 64bit system*/ \ + (defined(__SIZEOF_INT128__) && defined(__LP64__)) \ + /* MSVC 64bit compiler */ \ + || (defined(_MSC_VER) && defined(_M_X64)) \ + /* gcc >= 4.4 64bit */ \ + || (defined(__GNUC__) && defined(__LP64__) && \ + ((__GNUC__ > 4) || ((__GNUC__ == 4) && (__GNUC_MINOR__ >= 4)))) +# define __GUESS64 +#else +# define __GUESS32 +#endif + +#if defined(POLY1305_8BIT) +# include "poly1305-donna-8.h" +#elif defined(POLY1305_16BIT) +# include "poly1305-donna-16.h" +#elif defined(POLY1305_32BIT) || (!defined(POLY1305_64BIT) && defined(__GUESS32)) +# include "poly1305-donna-32.h" +#else +# include "poly1305-donna-64.h" +#endif + +void poly1305_update(poly1305_context *ctx, const unsigned char *m, size_t bytes) { + poly1305_state_internal_t *st = (poly1305_state_internal_t *)ctx; + size_t i; + + /* handle leftover */ + if (st->leftover) { + size_t want = (poly1305_block_size - st->leftover); + if (want > bytes) + want = bytes; + for (i = 0; i < want; i++) + st->buffer[st->leftover + i] = m[i]; + bytes -= want; + m += want; + st->leftover += want; + if (st->leftover < poly1305_block_size) + return; + poly1305_blocks(st, st->buffer, poly1305_block_size); + st->leftover = 0; + } + + /* process full blocks */ + if (bytes >= poly1305_block_size) { + size_t want = (bytes & ~(poly1305_block_size - 1)); + poly1305_blocks(st, m, want); + m += want; + bytes -= want; + } + + /* store leftover */ + if (bytes) { + for (i = 0; i < bytes; i++) + st->buffer[st->leftover + i] = m[i]; + st->leftover += bytes; + } +} + +int poly1305_verify(const unsigned char mac1[16], const unsigned char mac2[16]) { + size_t i; + unsigned int dif = 0; + for (i = 0; i < 16; i++) + dif |= (mac1[i] ^ mac2[i]); + dif = (dif - 1) >> ((sizeof(unsigned int) * 8) - 1); + return (dif & 1); +} diff --git a/third_party/portable8439/src/poly1305-donna/poly1305-donna.h b/third_party/portable8439/src/poly1305-donna/poly1305-donna.h new file mode 100644 index 0000000..90c2a0d --- /dev/null +++ b/third_party/portable8439/src/poly1305-donna/poly1305-donna.h @@ -0,0 +1,18 @@ +#ifndef POLY1305_DONNA_H +#define POLY1305_DONNA_H + +#include + +typedef struct poly1305_context { + size_t aligner; + unsigned char opaque[136]; +} poly1305_context; + +void poly1305_init(poly1305_context *ctx, const unsigned char key[32]); +void poly1305_update(poly1305_context *ctx, const unsigned char *m, size_t bytes); +void poly1305_finish(poly1305_context *ctx, unsigned char mac[16]); + +int poly1305_verify(const unsigned char mac1[16], const unsigned char mac2[16]); + +#endif /* POLY1305_DONNA_H */ + diff --git a/third_party/portable8439/src/portable8439.c b/third_party/portable8439/src/portable8439.c new file mode 100644 index 0000000..fe98f4c --- /dev/null +++ b/third_party/portable8439/src/portable8439.c @@ -0,0 +1,119 @@ +#include "portable8439.h" +#include "chacha-portable/chacha-portable.h" +#include "poly1305-donna/poly1305-donna.h" + +#define __CHACHA20_BLOCK_SIZE (64) +#define __POLY1305_KEY_SIZE (32) + +static uint8_t __ZEROES[16] = { 0 }; +static void pad_if_needed(poly1305_context *ctx, size_t size) { + size_t padding = size % 16; + if (padding != 0) { + poly1305_update(ctx, __ZEROES, 16 - padding); + } +} + +#define __u8(v) ((uint8_t)((v) & 0xFF)) + +// TODO: make this depending on the unaligned/native read size possible +static void write_64bit_int(poly1305_context *ctx, uint64_t value) { + uint8_t result[8]; + result[0] = __u8(value); + result[1] = __u8(value >> 8); + result[2] = __u8(value >> 16); + result[3] = __u8(value >> 24); + result[4] = __u8(value >> 32); + result[5] = __u8(value >> 40); + result[6] = __u8(value >> 48); + result[7] = __u8(value >> 56); + poly1305_update(ctx, result, 8); +} + +static void poly1305_calculate_mac( + uint8_t *mac, + const uint8_t *cipher_text, + size_t cipher_text_size, + const uint8_t key[RFC_8439_KEY_SIZE], + const uint8_t nonce[RFC_8439_NONCE_SIZE], + const uint8_t *ad, + size_t ad_size +) { + // init poly key (section 2.6) + uint8_t poly_key[__POLY1305_KEY_SIZE] = {0}; + rfc8439_keygen(poly_key, key, nonce); + // start poly1305 mac + poly1305_context poly_ctx; + poly1305_init(&poly_ctx, poly_key); + + if (ad != NULL && ad_size > 0) { + // write AD if present + poly1305_update(&poly_ctx, ad, ad_size); + pad_if_needed(&poly_ctx, ad_size); + } + + // now write the cipher text + poly1305_update(&poly_ctx, cipher_text, cipher_text_size); + pad_if_needed(&poly_ctx, cipher_text_size); + + // write sizes + write_64bit_int(&poly_ctx, ad_size); + write_64bit_int(&poly_ctx, cipher_text_size); + + // calculate MAC + poly1305_finish(&poly_ctx, mac); +} + + +#define PM(p) ((uintptr_t)(p)) + +// pointers overlap if the smaller either ahead of the end, +// or its end is before the start of the other +// +// s_size should be smaller or equal to b_size +#define OVERLAPPING(s, s_size, b, b_size) \ + (PM(s) < PM((b) + (b_size))) \ + && (PM(b) < PM((s) + (s_size))) + +size_t portable_chacha20_poly1305_encrypt( + uint8_t *restrict cipher_text, + const uint8_t key[RFC_8439_KEY_SIZE], + const uint8_t nonce[RFC_8439_NONCE_SIZE], + const uint8_t *restrict ad, + size_t ad_size, + const uint8_t *restrict plain_text, + size_t plain_text_size +) { + size_t new_size = plain_text_size + RFC_8439_TAG_SIZE; + if (OVERLAPPING(plain_text, plain_text_size, cipher_text, new_size)) { + return -1; + } + chacha20_xor_stream(cipher_text, plain_text, plain_text_size, key, nonce, 1); + poly1305_calculate_mac(cipher_text + plain_text_size, cipher_text, plain_text_size, key, nonce, ad, ad_size); + return new_size; +} + +size_t portable_chacha20_poly1305_decrypt( + uint8_t *restrict plain_text, + const uint8_t key[RFC_8439_KEY_SIZE], + const uint8_t nonce[RFC_8439_NONCE_SIZE], + const uint8_t *restrict ad, + size_t ad_size, + const uint8_t *restrict cipher_text, + size_t cipher_text_size +) { + // first we calculate the mac and see if it lines up, only then do we decrypt + uint8_t actual_mac[RFC_8439_TAG_SIZE]; + size_t actual_size = cipher_text_size - RFC_8439_TAG_SIZE; + if (OVERLAPPING(plain_text, actual_size, cipher_text, cipher_text_size)) { + return -1; + } + + poly1305_calculate_mac(actual_mac, cipher_text, actual_size, key, nonce, ad, ad_size); + + if (poly1305_verify(cipher_text + actual_size, actual_mac)) { + // valid mac, so decrypt cipher_text + chacha20_xor_stream(plain_text, cipher_text, actual_size, key, nonce, 1); + return actual_size; + } + return -1; +} diff --git a/third_party/portable8439/src/portable8439.h b/third_party/portable8439/src/portable8439.h new file mode 100644 index 0000000..b857d2a --- /dev/null +++ b/third_party/portable8439/src/portable8439.h @@ -0,0 +1,104 @@ +#ifndef COMPACT_8439_H +#define COMPACT_8439_H +/* + This library implements RFC 8439 a.k.a. ChaCha20-Poly1305 AEAD + + You can use this library to avoid attackers mutating or reusing your + encrypted messages. This does assume you never reuse a nonce+key pair and, + if possible, carefully pick your associated data. +*/ + +// Make sure we are either nested in C++ or running in a C99+ compiler +#if !defined(__cplusplus) && \ + !defined(_MSC_VER) && \ + (!defined(__STDC_VERSION__) || __STDC_VERSION__ < 199901L) +# error "C99 or newer required" +#endif + +#include +#include + +#if CHAR_BIT > 8 +# error "Systems without native octals not suppoted" +#endif + +#if defined(_MSC_VER) || defined(__cplusplus) +// add restrict support is possible +# if (defined(_MSC_VER) && _MSC_VER >= 1900) \ + || defined(__clang__) || defined(__GNUC__) +# define restrict __restrict +# else +# define restrict +# endif +#endif + +#define RFC_8439_TAG_SIZE (16) +#define RFC_8439_KEY_SIZE (32) +#define RFC_8439_NONCE_SIZE (12) + +/* + Encrypt/Seal plain text bytes into a cipher text that can only be + decrypted by knowing the key, nonce and associated data. + + input: + - key: RFC_8439_KEY_SIZE bytes that all parties have agreed + upon beforehand + - nonce: RFC_8439_NONCE_SIZE bytes that should never be repeated + for the same key. A counter or a pseudo-random value are fine. + - ad: associated data to include with calculating the tag of the + cipher text. Can be null for empty. + - plain_text: data to be encrypted, pointer + size should not overlap + with cipher_text pointer + + output: + - cipher_text: encrypted plain_text with a tag appended. Make sure to + allocate at least plain_text_size + RFC_8439_TAG_SIZE + + returns: + - size of bytes written to cipher_text, can be -1 if overlapping + pointers are passed for plain_text and cipher_text +*/ +size_t portable_chacha20_poly1305_encrypt( + uint8_t *restrict cipher_text, + const uint8_t key[RFC_8439_KEY_SIZE], + const uint8_t nonce[RFC_8439_NONCE_SIZE], + const uint8_t *restrict ad, + size_t ad_size, + const uint8_t *restrict plain_text, + size_t plain_text_size +); + + +/* + Decrypt/unseal cipher text given the right key, nonce, and additional data. + + input: + - key: RFC_8439_KEY_SIZE bytes that all parties have agreed + upon beforehand + - nonce: RFC_8439_NONCE_SIZE bytes that should never be repeated for + the same key. A counter or a pseudo-random value are fine. + - ad: associated data to include with calculating the tag of the + cipher text. Can be null for empty. + - cipher_text: encrypted message. + + output: + - plain_text: data to be encrypted, pointer + size should not overlap + with cipher_text pointer, leave at least enough room for + cipher_text_size - RFC_8439_TAG_SIZE + + returns: + - size of bytes written to plain_text, -1 signals either: + - incorrect key/nonce/ad + - corrupted cipher_text + - overlapping pointers are passed for plain_text and cipher_text +*/ +size_t portable_chacha20_poly1305_decrypt( + uint8_t *restrict plain_text, + const uint8_t key[RFC_8439_KEY_SIZE], + const uint8_t nonce[RFC_8439_NONCE_SIZE], + const uint8_t *restrict ad, + size_t ad_size, + const uint8_t *restrict cipher_text, + size_t cipher_text_size +); +#endif