srdusr
aboutsummaryrefslogtreecommitdiffstats
path: root/include/wireframe/net
diff options
context:
space:
mode:
Diffstat (limited to 'include/wireframe/net')
-rw-r--r--include/wireframe/net/checksum.hpp91
-rw-r--r--include/wireframe/net/icmp.hpp84
-rw-r--r--include/wireframe/net/tcp_reassembly.hpp125
3 files changed, 300 insertions, 0 deletions
diff --git a/include/wireframe/net/checksum.hpp b/include/wireframe/net/checksum.hpp
new file mode 100644
index 0000000..97e5254
--- /dev/null
+++ b/include/wireframe/net/checksum.hpp
@@ -0,0 +1,91 @@
+#pragma once
+
+#include <cstdint>
+#include <span>
+#include <vector>
+
+#include "wireframe/net/ipv4.hpp"
+
+// RFC 1071 Internet checksum, and the IPv4/TCP/UDP verification built
+// on it. Not wired into summarize_packet(): on loopback, and for many
+// packets captured right as they leave the local machine, the
+// transmitted checksum is legitimately 0x0000 or garbage - modern
+// NICs compute it in hardware ("checksum offload") only once the frame
+// actually reaches them, which is *after* most capture points see it.
+// Flagging that as "BAD" by default would be noise, not signal, on
+// exactly the interfaces this project has been tested against all
+// session (lo, tailscale0). Wireshark makes this opt-in for the same
+// reason; so does this (CLI's -c flag calls these directly).
+namespace wireframe::net {
+
+// One's-complement sum of 16-bit big-endian words, folded back into 16
+// bits, then complemented. Used identically by IPv4's header checksum
+// and, over a pseudo-header + segment instead of a plain header, by
+// TCP/UDP.
+inline std::uint16_t internet_checksum(std::span<const unsigned char> data) {
+ std::uint32_t sum = 0;
+ std::size_t i = 0;
+ for (; i + 1 < data.size(); i += 2) {
+ sum += (static_cast<std::uint32_t>(data[i]) << 8) | data[i + 1];
+ }
+ if (i < data.size()) {
+ sum += static_cast<std::uint32_t>(data[i]) << 8; // odd trailing byte: high half only
+ }
+ while (sum >> 16) {
+ sum = (sum & 0xFFFFu) + (sum >> 16);
+ }
+ return static_cast<std::uint16_t>(~sum & 0xFFFFu);
+}
+
+// `header_bytes` must be exactly the IPv4 header as it appeared on the
+// wire (IHL*4 bytes, options included, checksum field included as its
+// real transmitted value - not zeroed). Summing a header that already
+// contains its own valid checksum comes out to exactly 0; that's the
+// verification, no need for a mutable copy with the field zeroed out.
+inline bool verify_ipv4_checksum(std::span<const unsigned char> header_bytes) {
+ return internet_checksum(header_bytes) == 0;
+}
+
+enum class ChecksumResult { kValid, kInvalid, kNotPresent };
+
+namespace detail {
+
+inline std::vector<unsigned char> build_ipv4_pseudo_header(const Ipv4Address& src,
+ const Ipv4Address& dst,
+ std::uint8_t protocol,
+ std::span<const unsigned char> segment) {
+ std::vector<unsigned char> buf;
+ buf.reserve(12 + segment.size());
+ buf.insert(buf.end(), src.bytes.begin(), src.bytes.end());
+ buf.insert(buf.end(), dst.bytes.begin(), dst.bytes.end());
+ buf.push_back(0);
+ buf.push_back(protocol);
+ std::uint16_t len = static_cast<std::uint16_t>(segment.size());
+ buf.push_back(static_cast<unsigned char>(len >> 8));
+ buf.push_back(static_cast<unsigned char>(len & 0xFF));
+ buf.insert(buf.end(), segment.begin(), segment.end());
+ return buf;
+}
+
+} // namespace detail
+
+// TCP's checksum is mandatory - always kValid or kInvalid.
+inline ChecksumResult verify_tcp_checksum_ipv4(const Ipv4Address& src, const Ipv4Address& dst,
+ std::span<const unsigned char> tcp_segment) {
+ auto buf = detail::build_ipv4_pseudo_header(src, dst, kProtoTcp, tcp_segment);
+ return internet_checksum(buf) == 0 ? ChecksumResult::kValid : ChecksumResult::kInvalid;
+}
+
+// UDP's checksum is optional over IPv4 (RFC 768): a transmitted value
+// of exactly 0x0000 means "no checksum was computed", not "checksum is
+// zero" - that's kNotPresent, not a failure.
+inline ChecksumResult verify_udp_checksum_ipv4(const Ipv4Address& src, const Ipv4Address& dst,
+ std::span<const unsigned char> udp_datagram) {
+ if (udp_datagram.size() >= 8 && udp_datagram[6] == 0 && udp_datagram[7] == 0) {
+ return ChecksumResult::kNotPresent;
+ }
+ auto buf = detail::build_ipv4_pseudo_header(src, dst, kProtoUdp, udp_datagram);
+ return internet_checksum(buf) == 0 ? ChecksumResult::kValid : ChecksumResult::kInvalid;
+}
+
+} // namespace wireframe::net
diff --git a/include/wireframe/net/icmp.hpp b/include/wireframe/net/icmp.hpp
new file mode 100644
index 0000000..af83916
--- /dev/null
+++ b/include/wireframe/net/icmp.hpp
@@ -0,0 +1,84 @@
+#pragma once
+
+#include <cstdint>
+#include <optional>
+#include <span>
+#include <string>
+
+#include "wireframe/byteio.hpp"
+
+// ICMPv4 (RFC 792) and ICMPv6 (RFC 4443) share the same first-4-byte
+// shape (Type, Code, Checksum) but a completely different type
+// namespace - the same numeric type means something different in each
+// - so they get separate parse functions and separate type-name
+// tables, sharing only the header struct shape. Neither protocol has
+// ports, so this doesn't fit L7Registry's port-keyed dispatch at all;
+// it's handled directly by protocol number in summarize.hpp instead.
+namespace wireframe::net {
+
+struct IcmpHeader {
+ std::uint8_t type;
+ std::uint8_t code;
+ std::optional<std::uint16_t> identifier; // echo request/reply only
+ std::optional<std::uint16_t> sequence; // echo request/reply only
+};
+
+inline std::optional<IcmpHeader> parse_icmpv4(std::span<const unsigned char> bytes) {
+ if (bytes.size() < 4) return std::nullopt;
+
+ IcmpHeader header{};
+ header.type = bytes[0];
+ header.code = bytes[1];
+ if ((header.type == 8 || header.type == 0) && bytes.size() >= 8) { // echo request/reply
+ header.identifier = read_be16(bytes, 4);
+ header.sequence = read_be16(bytes, 6);
+ }
+ return header;
+}
+
+inline std::string icmpv4_type_name(std::uint8_t type) {
+ switch (type) {
+ case 0: return "Echo Reply";
+ case 3: return "Destination Unreachable";
+ case 4: return "Source Quench";
+ case 5: return "Redirect";
+ case 8: return "Echo Request";
+ case 11: return "Time Exceeded";
+ case 12: return "Parameter Problem";
+ case 13: return "Timestamp Request";
+ case 14: return "Timestamp Reply";
+ default: return "type=" + std::to_string(type);
+ }
+}
+
+inline std::optional<IcmpHeader> parse_icmpv6(std::span<const unsigned char> bytes) {
+ if (bytes.size() < 4) return std::nullopt;
+
+ IcmpHeader header{};
+ header.type = bytes[0];
+ header.code = bytes[1];
+ if ((header.type == 128 || header.type == 129) && bytes.size() >= 8) { // echo request/reply
+ header.identifier = read_be16(bytes, 4);
+ header.sequence = read_be16(bytes, 6);
+ }
+ return header;
+}
+
+inline std::string icmpv6_type_name(std::uint8_t type) {
+ switch (type) {
+ case 1: return "Destination Unreachable";
+ case 2: return "Packet Too Big";
+ case 3: return "Time Exceeded";
+ case 4: return "Parameter Problem";
+ case 128: return "Echo Request";
+ case 129: return "Echo Reply";
+ case 133: return "Router Solicitation";
+ case 134: return "Router Advertisement";
+ case 135: return "Neighbor Solicitation";
+ case 136: return "Neighbor Advertisement";
+ case 137: return "Redirect";
+ default: return "type=" + std::to_string(type);
+ }
+}
+
+} // namespace wireframe::net
diff --git a/include/wireframe/net/tcp_reassembly.hpp b/include/wireframe/net/tcp_reassembly.hpp
new file mode 100644
index 0000000..90824a4
--- /dev/null
+++ b/include/wireframe/net/tcp_reassembly.hpp
@@ -0,0 +1,125 @@
+#pragma once
+
+#include <cstdint>
+#include <map>
+#include <optional>
+#include <span>
+#include <tuple>
+#include <vector>
+
+#include "wireframe/net/ipv4.hpp"
+
+// Minimal, in-order-only TCP stream reassembly: tracks each flow's two
+// directions separately, accumulating payload bytes as segments arrive
+// exactly in sequence order. Out-of-order segments and retransmissions
+// are dropped rather than buffered for later reordering - a real
+// limitation, but a reasonable one for a learning-focused reassembler
+// capturing directly on an endpoint (this project's demonstrated use
+// all session: lo, wlp1s0, tailscale0), where segments mostly do
+// arrive in order. A capture point far from either endpoint (e.g. a
+// middlebox) would need real out-of-order buffering this doesn't do.
+//
+// The point: HTTP's dissector (wireframe/l7/http.hpp) only ever sees
+// one segment at a time, so a request/response split across TCP
+// segments - a Host: header landing in the second packet of a
+// request, say - is invisible to it. Feeding the *reassembled* stream
+// back through the same parse_http() lets it see what single-segment
+// dissection structurally can't.
+namespace wireframe::net {
+
+struct FlowKey {
+ Ipv4Address ip_a;
+ std::uint16_t port_a;
+ Ipv4Address ip_b;
+ std::uint16_t port_b;
+
+ bool operator<(const FlowKey& other) const {
+ return std::tie(ip_a.bytes, port_a, ip_b.bytes, port_b) <
+ std::tie(other.ip_a.bytes, other.port_a, other.ip_b.bytes, other.port_b);
+ }
+};
+
+// Canonicalizes a (src, dst) pair into a direction-independent
+// FlowKey - both directions of the same connection map to the same
+// key - plus whether this segment's source was the "a" side.
+inline std::pair<FlowKey, bool> canonicalize_flow(const Ipv4Address& src_ip,
+ std::uint16_t src_port,
+ const Ipv4Address& dst_ip,
+ std::uint16_t dst_port) {
+ bool src_is_a = std::tie(src_ip.bytes, src_port) < std::tie(dst_ip.bytes, dst_port);
+ FlowKey key = src_is_a ? FlowKey{src_ip, src_port, dst_ip, dst_port}
+ : FlowKey{dst_ip, dst_port, src_ip, src_port};
+ return {key, src_is_a};
+}
+
+struct DirectionState {
+ bool syn_seen = false;
+ std::uint32_t next_seq = 0;
+ std::vector<unsigned char> buffer;
+};
+
+struct FlowState {
+ DirectionState a_to_b;
+ DirectionState b_to_a;
+};
+
+class TcpReassembler {
+public:
+ explicit TcpReassembler(std::size_t max_buffer_per_direction = 65536,
+ std::size_t max_flows = 4096)
+ : max_buffer_(max_buffer_per_direction), max_flows_(max_flows) {}
+
+ // Feeds one TCP segment in. Returns a snapshot of the *sender's*
+ // accumulated stream so far if this segment extended it
+ // contiguously in order; nullopt if the segment was out of order,
+ // a retransmission, a control segment with no payload, or the flow
+ // table was full and this would be a brand new flow. Returned by
+ // value rather than by reference: the buffer this points at can
+ // grow/move on the next call, and bounded copies (max 64 KiB by
+ // default) are cheap enough that this isn't worth the lifetime risk.
+ std::optional<std::vector<unsigned char>> process_segment(
+ const Ipv4Address& src_ip, std::uint16_t src_port, const Ipv4Address& dst_ip,
+ std::uint16_t dst_port, std::uint32_t seq, std::uint8_t flags,
+ std::span<const unsigned char> payload) {
+ auto [key, src_is_a] = canonicalize_flow(src_ip, src_port, dst_ip, dst_port);
+
+ auto it = flows_.find(key);
+ if (it == flows_.end()) {
+ if (flows_.size() >= max_flows_) return std::nullopt; // table full: drop new flows
+ it = flows_.emplace(key, FlowState{}).first;
+ }
+ DirectionState& dir = src_is_a ? it->second.a_to_b : it->second.b_to_a;
+
+ constexpr std::uint8_t kSyn = 0x02;
+ if (flags & kSyn) {
+ dir.syn_seen = true;
+ dir.next_seq = seq + 1; // the SYN itself consumes one sequence number
+ return std::nullopt;
+ }
+
+ // seq != dir.next_seq covers both out-of-order segments and
+ // retransmissions (a retransmit repeats a seq already below
+ // next_seq) - unsigned wraparound makes plain equality correct
+ // even across a sequence-number wrap, no need for RFC 1982
+ // serial-number comparison for an exact-match check like this.
+ if (!dir.syn_seen || payload.empty() || seq != dir.next_seq) {
+ return std::nullopt;
+ }
+
+ if (dir.buffer.size() + payload.size() <= max_buffer_) {
+ dir.buffer.insert(dir.buffer.end(), payload.begin(), payload.end());
+ }
+ dir.next_seq = seq + static_cast<std::uint32_t>(payload.size());
+
+ return dir.buffer;
+ }
+
+ std::size_t flow_count() const { return flows_.size(); }
+
+private:
+ std::map<FlowKey, FlowState> flows_;
+ std::size_t max_buffer_;
+ std::size_t max_flows_;
+};
+
+} // namespace wireframe::net