diff options
Diffstat (limited to 'include/wireframe/net')
| -rw-r--r-- | include/wireframe/net/checksum.hpp | 91 | ||||
| -rw-r--r-- | include/wireframe/net/icmp.hpp | 84 | ||||
| -rw-r--r-- | include/wireframe/net/tcp_reassembly.hpp | 125 |
3 files changed, 300 insertions, 0 deletions
diff --git a/include/wireframe/net/checksum.hpp b/include/wireframe/net/checksum.hpp new file mode 100644 index 0000000..97e5254 --- /dev/null +++ b/include/wireframe/net/checksum.hpp @@ -0,0 +1,91 @@ +#pragma once + +#include <cstdint> +#include <span> +#include <vector> + +#include "wireframe/net/ipv4.hpp" + +// RFC 1071 Internet checksum, and the IPv4/TCP/UDP verification built +// on it. Not wired into summarize_packet(): on loopback, and for many +// packets captured right as they leave the local machine, the +// transmitted checksum is legitimately 0x0000 or garbage - modern +// NICs compute it in hardware ("checksum offload") only once the frame +// actually reaches them, which is *after* most capture points see it. +// Flagging that as "BAD" by default would be noise, not signal, on +// exactly the interfaces this project has been tested against all +// session (lo, tailscale0). Wireshark makes this opt-in for the same +// reason; so does this (CLI's -c flag calls these directly). +namespace wireframe::net { + +// One's-complement sum of 16-bit big-endian words, folded back into 16 +// bits, then complemented. Used identically by IPv4's header checksum +// and, over a pseudo-header + segment instead of a plain header, by +// TCP/UDP. +inline std::uint16_t internet_checksum(std::span<const unsigned char> data) { + std::uint32_t sum = 0; + std::size_t i = 0; + for (; i + 1 < data.size(); i += 2) { + sum += (static_cast<std::uint32_t>(data[i]) << 8) | data[i + 1]; + } + if (i < data.size()) { + sum += static_cast<std::uint32_t>(data[i]) << 8; // odd trailing byte: high half only + } + while (sum >> 16) { + sum = (sum & 0xFFFFu) + (sum >> 16); + } + return static_cast<std::uint16_t>(~sum & 0xFFFFu); +} + +// `header_bytes` must be exactly the IPv4 header as it appeared on the +// wire (IHL*4 bytes, options included, checksum field included as its +// real transmitted value - not zeroed). Summing a header that already +// contains its own valid checksum comes out to exactly 0; that's the +// verification, no need for a mutable copy with the field zeroed out. +inline bool verify_ipv4_checksum(std::span<const unsigned char> header_bytes) { + return internet_checksum(header_bytes) == 0; +} + +enum class ChecksumResult { kValid, kInvalid, kNotPresent }; + +namespace detail { + +inline std::vector<unsigned char> build_ipv4_pseudo_header(const Ipv4Address& src, + const Ipv4Address& dst, + std::uint8_t protocol, + std::span<const unsigned char> segment) { + std::vector<unsigned char> buf; + buf.reserve(12 + segment.size()); + buf.insert(buf.end(), src.bytes.begin(), src.bytes.end()); + buf.insert(buf.end(), dst.bytes.begin(), dst.bytes.end()); + buf.push_back(0); + buf.push_back(protocol); + std::uint16_t len = static_cast<std::uint16_t>(segment.size()); + buf.push_back(static_cast<unsigned char>(len >> 8)); + buf.push_back(static_cast<unsigned char>(len & 0xFF)); + buf.insert(buf.end(), segment.begin(), segment.end()); + return buf; +} + +} // namespace detail + +// TCP's checksum is mandatory - always kValid or kInvalid. +inline ChecksumResult verify_tcp_checksum_ipv4(const Ipv4Address& src, const Ipv4Address& dst, + std::span<const unsigned char> tcp_segment) { + auto buf = detail::build_ipv4_pseudo_header(src, dst, kProtoTcp, tcp_segment); + return internet_checksum(buf) == 0 ? ChecksumResult::kValid : ChecksumResult::kInvalid; +} + +// UDP's checksum is optional over IPv4 (RFC 768): a transmitted value +// of exactly 0x0000 means "no checksum was computed", not "checksum is +// zero" - that's kNotPresent, not a failure. +inline ChecksumResult verify_udp_checksum_ipv4(const Ipv4Address& src, const Ipv4Address& dst, + std::span<const unsigned char> udp_datagram) { + if (udp_datagram.size() >= 8 && udp_datagram[6] == 0 && udp_datagram[7] == 0) { + return ChecksumResult::kNotPresent; + } + auto buf = detail::build_ipv4_pseudo_header(src, dst, kProtoUdp, udp_datagram); + return internet_checksum(buf) == 0 ? ChecksumResult::kValid : ChecksumResult::kInvalid; +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/icmp.hpp b/include/wireframe/net/icmp.hpp new file mode 100644 index 0000000..af83916 --- /dev/null +++ b/include/wireframe/net/icmp.hpp @@ -0,0 +1,84 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> +#include <string> + +#include "wireframe/byteio.hpp" + +// ICMPv4 (RFC 792) and ICMPv6 (RFC 4443) share the same first-4-byte +// shape (Type, Code, Checksum) but a completely different type +// namespace - the same numeric type means something different in each +// - so they get separate parse functions and separate type-name +// tables, sharing only the header struct shape. Neither protocol has +// ports, so this doesn't fit L7Registry's port-keyed dispatch at all; +// it's handled directly by protocol number in summarize.hpp instead. +namespace wireframe::net { + +struct IcmpHeader { + std::uint8_t type; + std::uint8_t code; + std::optional<std::uint16_t> identifier; // echo request/reply only + std::optional<std::uint16_t> sequence; // echo request/reply only +}; + +inline std::optional<IcmpHeader> parse_icmpv4(std::span<const unsigned char> bytes) { + if (bytes.size() < 4) return std::nullopt; + + IcmpHeader header{}; + header.type = bytes[0]; + header.code = bytes[1]; + if ((header.type == 8 || header.type == 0) && bytes.size() >= 8) { // echo request/reply + header.identifier = read_be16(bytes, 4); + header.sequence = read_be16(bytes, 6); + } + return header; +} + +inline std::string icmpv4_type_name(std::uint8_t type) { + switch (type) { + case 0: return "Echo Reply"; + case 3: return "Destination Unreachable"; + case 4: return "Source Quench"; + case 5: return "Redirect"; + case 8: return "Echo Request"; + case 11: return "Time Exceeded"; + case 12: return "Parameter Problem"; + case 13: return "Timestamp Request"; + case 14: return "Timestamp Reply"; + default: return "type=" + std::to_string(type); + } +} + +inline std::optional<IcmpHeader> parse_icmpv6(std::span<const unsigned char> bytes) { + if (bytes.size() < 4) return std::nullopt; + + IcmpHeader header{}; + header.type = bytes[0]; + header.code = bytes[1]; + if ((header.type == 128 || header.type == 129) && bytes.size() >= 8) { // echo request/reply + header.identifier = read_be16(bytes, 4); + header.sequence = read_be16(bytes, 6); + } + return header; +} + +inline std::string icmpv6_type_name(std::uint8_t type) { + switch (type) { + case 1: return "Destination Unreachable"; + case 2: return "Packet Too Big"; + case 3: return "Time Exceeded"; + case 4: return "Parameter Problem"; + case 128: return "Echo Request"; + case 129: return "Echo Reply"; + case 133: return "Router Solicitation"; + case 134: return "Router Advertisement"; + case 135: return "Neighbor Solicitation"; + case 136: return "Neighbor Advertisement"; + case 137: return "Redirect"; + default: return "type=" + std::to_string(type); + } +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/tcp_reassembly.hpp b/include/wireframe/net/tcp_reassembly.hpp new file mode 100644 index 0000000..90824a4 --- /dev/null +++ b/include/wireframe/net/tcp_reassembly.hpp @@ -0,0 +1,125 @@ +#pragma once + +#include <cstdint> +#include <map> +#include <optional> +#include <span> +#include <tuple> +#include <vector> + +#include "wireframe/net/ipv4.hpp" + +// Minimal, in-order-only TCP stream reassembly: tracks each flow's two +// directions separately, accumulating payload bytes as segments arrive +// exactly in sequence order. Out-of-order segments and retransmissions +// are dropped rather than buffered for later reordering - a real +// limitation, but a reasonable one for a learning-focused reassembler +// capturing directly on an endpoint (this project's demonstrated use +// all session: lo, wlp1s0, tailscale0), where segments mostly do +// arrive in order. A capture point far from either endpoint (e.g. a +// middlebox) would need real out-of-order buffering this doesn't do. +// +// The point: HTTP's dissector (wireframe/l7/http.hpp) only ever sees +// one segment at a time, so a request/response split across TCP +// segments - a Host: header landing in the second packet of a +// request, say - is invisible to it. Feeding the *reassembled* stream +// back through the same parse_http() lets it see what single-segment +// dissection structurally can't. +namespace wireframe::net { + +struct FlowKey { + Ipv4Address ip_a; + std::uint16_t port_a; + Ipv4Address ip_b; + std::uint16_t port_b; + + bool operator<(const FlowKey& other) const { + return std::tie(ip_a.bytes, port_a, ip_b.bytes, port_b) < + std::tie(other.ip_a.bytes, other.port_a, other.ip_b.bytes, other.port_b); + } +}; + +// Canonicalizes a (src, dst) pair into a direction-independent +// FlowKey - both directions of the same connection map to the same +// key - plus whether this segment's source was the "a" side. +inline std::pair<FlowKey, bool> canonicalize_flow(const Ipv4Address& src_ip, + std::uint16_t src_port, + const Ipv4Address& dst_ip, + std::uint16_t dst_port) { + bool src_is_a = std::tie(src_ip.bytes, src_port) < std::tie(dst_ip.bytes, dst_port); + FlowKey key = src_is_a ? FlowKey{src_ip, src_port, dst_ip, dst_port} + : FlowKey{dst_ip, dst_port, src_ip, src_port}; + return {key, src_is_a}; +} + +struct DirectionState { + bool syn_seen = false; + std::uint32_t next_seq = 0; + std::vector<unsigned char> buffer; +}; + +struct FlowState { + DirectionState a_to_b; + DirectionState b_to_a; +}; + +class TcpReassembler { +public: + explicit TcpReassembler(std::size_t max_buffer_per_direction = 65536, + std::size_t max_flows = 4096) + : max_buffer_(max_buffer_per_direction), max_flows_(max_flows) {} + + // Feeds one TCP segment in. Returns a snapshot of the *sender's* + // accumulated stream so far if this segment extended it + // contiguously in order; nullopt if the segment was out of order, + // a retransmission, a control segment with no payload, or the flow + // table was full and this would be a brand new flow. Returned by + // value rather than by reference: the buffer this points at can + // grow/move on the next call, and bounded copies (max 64 KiB by + // default) are cheap enough that this isn't worth the lifetime risk. + std::optional<std::vector<unsigned char>> process_segment( + const Ipv4Address& src_ip, std::uint16_t src_port, const Ipv4Address& dst_ip, + std::uint16_t dst_port, std::uint32_t seq, std::uint8_t flags, + std::span<const unsigned char> payload) { + auto [key, src_is_a] = canonicalize_flow(src_ip, src_port, dst_ip, dst_port); + + auto it = flows_.find(key); + if (it == flows_.end()) { + if (flows_.size() >= max_flows_) return std::nullopt; // table full: drop new flows + it = flows_.emplace(key, FlowState{}).first; + } + DirectionState& dir = src_is_a ? it->second.a_to_b : it->second.b_to_a; + + constexpr std::uint8_t kSyn = 0x02; + if (flags & kSyn) { + dir.syn_seen = true; + dir.next_seq = seq + 1; // the SYN itself consumes one sequence number + return std::nullopt; + } + + // seq != dir.next_seq covers both out-of-order segments and + // retransmissions (a retransmit repeats a seq already below + // next_seq) - unsigned wraparound makes plain equality correct + // even across a sequence-number wrap, no need for RFC 1982 + // serial-number comparison for an exact-match check like this. + if (!dir.syn_seen || payload.empty() || seq != dir.next_seq) { + return std::nullopt; + } + + if (dir.buffer.size() + payload.size() <= max_buffer_) { + dir.buffer.insert(dir.buffer.end(), payload.begin(), payload.end()); + } + dir.next_seq = seq + static_cast<std::uint32_t>(payload.size()); + + return dir.buffer; + } + + std::size_t flow_count() const { return flows_.size(); } + +private: + std::map<FlowKey, FlowState> flows_; + std::size_t max_buffer_; + std::size_t max_flows_; +}; + +} // namespace wireframe::net |