diff options
Diffstat (limited to 'include/wireframe')
| -rw-r--r-- | include/wireframe/capture_session.hpp | 11 | ||||
| -rw-r--r-- | include/wireframe/net/checksum.hpp | 91 | ||||
| -rw-r--r-- | include/wireframe/net/icmp.hpp | 84 | ||||
| -rw-r--r-- | include/wireframe/net/tcp_reassembly.hpp | 125 | ||||
| -rw-r--r-- | include/wireframe/privileges.hpp | 87 | ||||
| -rw-r--r-- | include/wireframe/summarize.hpp | 19 |
6 files changed, 416 insertions, 1 deletions
diff --git a/include/wireframe/capture_session.hpp b/include/wireframe/capture_session.hpp index 50764a8..2e3b05a 100644 --- a/include/wireframe/capture_session.hpp +++ b/include/wireframe/capture_session.hpp @@ -14,6 +14,7 @@ #include "wireframe/filter.hpp" #include "wireframe/pcapng/reader.hpp" #include "wireframe/pcapng/writer.hpp" +#include "wireframe/privileges.hpp" // Device-open -> datalink-validate -> filter/pcapng-setup -> signal-hook // pipeline, shared by every frontend (CLI, TUI, GUI). Centralized so a @@ -99,6 +100,16 @@ public: return std::string("pcap_open_live failed: ") + errbuf; } + // Everything CAP_NET_RAW/root was needed for is done: the + // handle is open. Drop immediately, before the datalink check + // or -w's file is even created - the latter is also why this + // runs this early rather than at the very end of open(), since + // it means a -w output file gets created as the real user, not + // root, and doesn't need a manual chown to read back afterward. + if (auto err = drop_privileges_if_root()) { + return "failed to drop root privileges after opening the capture handle: " + *err; + } + datalink_ = pcap_datalink(handle_); if (!is_supported_datalink(datalink_)) { return std::string("unsupported datalink type on ") + device_ + ": " + diff --git a/include/wireframe/net/checksum.hpp b/include/wireframe/net/checksum.hpp new file mode 100644 index 0000000..97e5254 --- /dev/null +++ b/include/wireframe/net/checksum.hpp @@ -0,0 +1,91 @@ +#pragma once + +#include <cstdint> +#include <span> +#include <vector> + +#include "wireframe/net/ipv4.hpp" + +// RFC 1071 Internet checksum, and the IPv4/TCP/UDP verification built +// on it. Not wired into summarize_packet(): on loopback, and for many +// packets captured right as they leave the local machine, the +// transmitted checksum is legitimately 0x0000 or garbage - modern +// NICs compute it in hardware ("checksum offload") only once the frame +// actually reaches them, which is *after* most capture points see it. +// Flagging that as "BAD" by default would be noise, not signal, on +// exactly the interfaces this project has been tested against all +// session (lo, tailscale0). Wireshark makes this opt-in for the same +// reason; so does this (CLI's -c flag calls these directly). +namespace wireframe::net { + +// One's-complement sum of 16-bit big-endian words, folded back into 16 +// bits, then complemented. Used identically by IPv4's header checksum +// and, over a pseudo-header + segment instead of a plain header, by +// TCP/UDP. +inline std::uint16_t internet_checksum(std::span<const unsigned char> data) { + std::uint32_t sum = 0; + std::size_t i = 0; + for (; i + 1 < data.size(); i += 2) { + sum += (static_cast<std::uint32_t>(data[i]) << 8) | data[i + 1]; + } + if (i < data.size()) { + sum += static_cast<std::uint32_t>(data[i]) << 8; // odd trailing byte: high half only + } + while (sum >> 16) { + sum = (sum & 0xFFFFu) + (sum >> 16); + } + return static_cast<std::uint16_t>(~sum & 0xFFFFu); +} + +// `header_bytes` must be exactly the IPv4 header as it appeared on the +// wire (IHL*4 bytes, options included, checksum field included as its +// real transmitted value - not zeroed). Summing a header that already +// contains its own valid checksum comes out to exactly 0; that's the +// verification, no need for a mutable copy with the field zeroed out. +inline bool verify_ipv4_checksum(std::span<const unsigned char> header_bytes) { + return internet_checksum(header_bytes) == 0; +} + +enum class ChecksumResult { kValid, kInvalid, kNotPresent }; + +namespace detail { + +inline std::vector<unsigned char> build_ipv4_pseudo_header(const Ipv4Address& src, + const Ipv4Address& dst, + std::uint8_t protocol, + std::span<const unsigned char> segment) { + std::vector<unsigned char> buf; + buf.reserve(12 + segment.size()); + buf.insert(buf.end(), src.bytes.begin(), src.bytes.end()); + buf.insert(buf.end(), dst.bytes.begin(), dst.bytes.end()); + buf.push_back(0); + buf.push_back(protocol); + std::uint16_t len = static_cast<std::uint16_t>(segment.size()); + buf.push_back(static_cast<unsigned char>(len >> 8)); + buf.push_back(static_cast<unsigned char>(len & 0xFF)); + buf.insert(buf.end(), segment.begin(), segment.end()); + return buf; +} + +} // namespace detail + +// TCP's checksum is mandatory - always kValid or kInvalid. +inline ChecksumResult verify_tcp_checksum_ipv4(const Ipv4Address& src, const Ipv4Address& dst, + std::span<const unsigned char> tcp_segment) { + auto buf = detail::build_ipv4_pseudo_header(src, dst, kProtoTcp, tcp_segment); + return internet_checksum(buf) == 0 ? ChecksumResult::kValid : ChecksumResult::kInvalid; +} + +// UDP's checksum is optional over IPv4 (RFC 768): a transmitted value +// of exactly 0x0000 means "no checksum was computed", not "checksum is +// zero" - that's kNotPresent, not a failure. +inline ChecksumResult verify_udp_checksum_ipv4(const Ipv4Address& src, const Ipv4Address& dst, + std::span<const unsigned char> udp_datagram) { + if (udp_datagram.size() >= 8 && udp_datagram[6] == 0 && udp_datagram[7] == 0) { + return ChecksumResult::kNotPresent; + } + auto buf = detail::build_ipv4_pseudo_header(src, dst, kProtoUdp, udp_datagram); + return internet_checksum(buf) == 0 ? ChecksumResult::kValid : ChecksumResult::kInvalid; +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/icmp.hpp b/include/wireframe/net/icmp.hpp new file mode 100644 index 0000000..af83916 --- /dev/null +++ b/include/wireframe/net/icmp.hpp @@ -0,0 +1,84 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> +#include <string> + +#include "wireframe/byteio.hpp" + +// ICMPv4 (RFC 792) and ICMPv6 (RFC 4443) share the same first-4-byte +// shape (Type, Code, Checksum) but a completely different type +// namespace - the same numeric type means something different in each +// - so they get separate parse functions and separate type-name +// tables, sharing only the header struct shape. Neither protocol has +// ports, so this doesn't fit L7Registry's port-keyed dispatch at all; +// it's handled directly by protocol number in summarize.hpp instead. +namespace wireframe::net { + +struct IcmpHeader { + std::uint8_t type; + std::uint8_t code; + std::optional<std::uint16_t> identifier; // echo request/reply only + std::optional<std::uint16_t> sequence; // echo request/reply only +}; + +inline std::optional<IcmpHeader> parse_icmpv4(std::span<const unsigned char> bytes) { + if (bytes.size() < 4) return std::nullopt; + + IcmpHeader header{}; + header.type = bytes[0]; + header.code = bytes[1]; + if ((header.type == 8 || header.type == 0) && bytes.size() >= 8) { // echo request/reply + header.identifier = read_be16(bytes, 4); + header.sequence = read_be16(bytes, 6); + } + return header; +} + +inline std::string icmpv4_type_name(std::uint8_t type) { + switch (type) { + case 0: return "Echo Reply"; + case 3: return "Destination Unreachable"; + case 4: return "Source Quench"; + case 5: return "Redirect"; + case 8: return "Echo Request"; + case 11: return "Time Exceeded"; + case 12: return "Parameter Problem"; + case 13: return "Timestamp Request"; + case 14: return "Timestamp Reply"; + default: return "type=" + std::to_string(type); + } +} + +inline std::optional<IcmpHeader> parse_icmpv6(std::span<const unsigned char> bytes) { + if (bytes.size() < 4) return std::nullopt; + + IcmpHeader header{}; + header.type = bytes[0]; + header.code = bytes[1]; + if ((header.type == 128 || header.type == 129) && bytes.size() >= 8) { // echo request/reply + header.identifier = read_be16(bytes, 4); + header.sequence = read_be16(bytes, 6); + } + return header; +} + +inline std::string icmpv6_type_name(std::uint8_t type) { + switch (type) { + case 1: return "Destination Unreachable"; + case 2: return "Packet Too Big"; + case 3: return "Time Exceeded"; + case 4: return "Parameter Problem"; + case 128: return "Echo Request"; + case 129: return "Echo Reply"; + case 133: return "Router Solicitation"; + case 134: return "Router Advertisement"; + case 135: return "Neighbor Solicitation"; + case 136: return "Neighbor Advertisement"; + case 137: return "Redirect"; + default: return "type=" + std::to_string(type); + } +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/tcp_reassembly.hpp b/include/wireframe/net/tcp_reassembly.hpp new file mode 100644 index 0000000..90824a4 --- /dev/null +++ b/include/wireframe/net/tcp_reassembly.hpp @@ -0,0 +1,125 @@ +#pragma once + +#include <cstdint> +#include <map> +#include <optional> +#include <span> +#include <tuple> +#include <vector> + +#include "wireframe/net/ipv4.hpp" + +// Minimal, in-order-only TCP stream reassembly: tracks each flow's two +// directions separately, accumulating payload bytes as segments arrive +// exactly in sequence order. Out-of-order segments and retransmissions +// are dropped rather than buffered for later reordering - a real +// limitation, but a reasonable one for a learning-focused reassembler +// capturing directly on an endpoint (this project's demonstrated use +// all session: lo, wlp1s0, tailscale0), where segments mostly do +// arrive in order. A capture point far from either endpoint (e.g. a +// middlebox) would need real out-of-order buffering this doesn't do. +// +// The point: HTTP's dissector (wireframe/l7/http.hpp) only ever sees +// one segment at a time, so a request/response split across TCP +// segments - a Host: header landing in the second packet of a +// request, say - is invisible to it. Feeding the *reassembled* stream +// back through the same parse_http() lets it see what single-segment +// dissection structurally can't. +namespace wireframe::net { + +struct FlowKey { + Ipv4Address ip_a; + std::uint16_t port_a; + Ipv4Address ip_b; + std::uint16_t port_b; + + bool operator<(const FlowKey& other) const { + return std::tie(ip_a.bytes, port_a, ip_b.bytes, port_b) < + std::tie(other.ip_a.bytes, other.port_a, other.ip_b.bytes, other.port_b); + } +}; + +// Canonicalizes a (src, dst) pair into a direction-independent +// FlowKey - both directions of the same connection map to the same +// key - plus whether this segment's source was the "a" side. +inline std::pair<FlowKey, bool> canonicalize_flow(const Ipv4Address& src_ip, + std::uint16_t src_port, + const Ipv4Address& dst_ip, + std::uint16_t dst_port) { + bool src_is_a = std::tie(src_ip.bytes, src_port) < std::tie(dst_ip.bytes, dst_port); + FlowKey key = src_is_a ? FlowKey{src_ip, src_port, dst_ip, dst_port} + : FlowKey{dst_ip, dst_port, src_ip, src_port}; + return {key, src_is_a}; +} + +struct DirectionState { + bool syn_seen = false; + std::uint32_t next_seq = 0; + std::vector<unsigned char> buffer; +}; + +struct FlowState { + DirectionState a_to_b; + DirectionState b_to_a; +}; + +class TcpReassembler { +public: + explicit TcpReassembler(std::size_t max_buffer_per_direction = 65536, + std::size_t max_flows = 4096) + : max_buffer_(max_buffer_per_direction), max_flows_(max_flows) {} + + // Feeds one TCP segment in. Returns a snapshot of the *sender's* + // accumulated stream so far if this segment extended it + // contiguously in order; nullopt if the segment was out of order, + // a retransmission, a control segment with no payload, or the flow + // table was full and this would be a brand new flow. Returned by + // value rather than by reference: the buffer this points at can + // grow/move on the next call, and bounded copies (max 64 KiB by + // default) are cheap enough that this isn't worth the lifetime risk. + std::optional<std::vector<unsigned char>> process_segment( + const Ipv4Address& src_ip, std::uint16_t src_port, const Ipv4Address& dst_ip, + std::uint16_t dst_port, std::uint32_t seq, std::uint8_t flags, + std::span<const unsigned char> payload) { + auto [key, src_is_a] = canonicalize_flow(src_ip, src_port, dst_ip, dst_port); + + auto it = flows_.find(key); + if (it == flows_.end()) { + if (flows_.size() >= max_flows_) return std::nullopt; // table full: drop new flows + it = flows_.emplace(key, FlowState{}).first; + } + DirectionState& dir = src_is_a ? it->second.a_to_b : it->second.b_to_a; + + constexpr std::uint8_t kSyn = 0x02; + if (flags & kSyn) { + dir.syn_seen = true; + dir.next_seq = seq + 1; // the SYN itself consumes one sequence number + return std::nullopt; + } + + // seq != dir.next_seq covers both out-of-order segments and + // retransmissions (a retransmit repeats a seq already below + // next_seq) - unsigned wraparound makes plain equality correct + // even across a sequence-number wrap, no need for RFC 1982 + // serial-number comparison for an exact-match check like this. + if (!dir.syn_seen || payload.empty() || seq != dir.next_seq) { + return std::nullopt; + } + + if (dir.buffer.size() + payload.size() <= max_buffer_) { + dir.buffer.insert(dir.buffer.end(), payload.begin(), payload.end()); + } + dir.next_seq = seq + static_cast<std::uint32_t>(payload.size()); + + return dir.buffer; + } + + std::size_t flow_count() const { return flows_.size(); } + +private: + std::map<FlowKey, FlowState> flows_; + std::size_t max_buffer_; + std::size_t max_flows_; +}; + +} // namespace wireframe::net diff --git a/include/wireframe/privileges.hpp b/include/wireframe/privileges.hpp new file mode 100644 index 0000000..69df725 --- /dev/null +++ b/include/wireframe/privileges.hpp @@ -0,0 +1,87 @@ +#pragma once + +#ifndef _WIN32 +#include <grp.h> +#include <unistd.h> +#endif + +#include <cerrno> +#include <cstdlib> +#include <cstring> +#include <optional> +#include <string> + +// After pcap_open_live() succeeds, the process has gotten everything +// CAP_NET_RAW exists for - running the rest of the program (decoding +// untrusted packet bytes, an interactive TUI/GUI event loop) as root +// from that point on is unnecessary exposure, and PLAN.md says as much +// directly: "Drop privileges immediately after opening the capture +// handle; use CAP_NET_RAW via file capabilities instead of running as +// root." +// +// The recommended path doesn't need this file at all: run +// `sudo setcap cap_net_raw+ep <binary>` once, then invoke the binary +// directly, unprivileged, forever after - CAP_NET_RAW alone is enough +// for pcap_open_live(), no root required at any point. This exists for +// the case someone still runs the binary via sudo (out of habit, or +// because setcap isn't available/permitted in some environments): drop +// straight back to the invoking user immediately, so the rest of the +// process's lifetime - including any -w output file, which then ends +// up owned by that user instead of root - runs unprivileged either way. +namespace wireframe { + +// Drops from root to the user who actually invoked the program, via +// sudo's SUDO_UID/SUDO_GID (which sudo always sets). A no-op if not +// currently root, or if SUDO_UID isn't set (e.g. a genuine root login, +// not sudo - there's no "real" user to drop to in that case). +// +// setuid() to a nonzero UID also clears the process's Linux capability +// sets as a kernel-level side effect, so this covers both "running as +// root via sudo" and "root's own CAP_NET_RAW" the same way, without a +// separate libcap dependency. +// +// Returns an error message on failure. The drop is safety-critical: a +// failure here should be treated as fatal by the caller, not silently +// ignored while the process keeps running as root. +inline std::optional<std::string> drop_privileges_if_root() { +#ifdef _WIN32 + return std::nullopt; // no POSIX privilege model to drop from +#else + if (geteuid() != 0) return std::nullopt; // already unprivileged + + const char* sudo_uid = std::getenv("SUDO_UID"); + const char* sudo_gid = std::getenv("SUDO_GID"); + if (sudo_uid == nullptr || sudo_gid == nullptr) { + return std::nullopt; // no safe target to drop to + } + + uid_t target_uid = static_cast<uid_t>(std::strtoul(sudo_uid, nullptr, 10)); + gid_t target_gid = static_cast<gid_t>(std::strtoul(sudo_gid, nullptr, 10)); + + // Order matters: groups and GID need root to change, so they must + // be dropped before UID - once UID is gone, so is the privilege + // to change the others. + if (setgroups(1, &target_gid) == -1) { + return "setgroups failed: " + std::string(std::strerror(errno)); + } + if (setgid(target_gid) == -1) { + return "setgid failed: " + std::string(std::strerror(errno)); + } + if (setuid(target_uid) == -1) { + return "setuid failed: " + std::string(std::strerror(errno)); + } + + // Defense in depth (standard advice from setuid-privilege-drop + // write-ups): confirm root can't be reclaimed. If the saved-UID + // was somehow left at 0, this would succeed and silently undo the + // drop - so a *successful* setuid(0) here means something is + // wrong, and is treated as the failure case. + if (setuid(0) != -1) { + return "failed to permanently drop root (setuid(0) unexpectedly succeeded)"; + } + + return std::nullopt; +#endif +} + +} // namespace wireframe diff --git a/include/wireframe/summarize.hpp b/include/wireframe/summarize.hpp index 59aa621..e7e9ae3 100644 --- a/include/wireframe/summarize.hpp +++ b/include/wireframe/summarize.hpp @@ -13,6 +13,7 @@ #include "wireframe/l7/http.hpp" #include "wireframe/l7/tls.hpp" #include "wireframe/net/ethernet.hpp" +#include "wireframe/net/icmp.hpp" #include "wireframe/net/ipv4.hpp" #include "wireframe/net/ipv6.hpp" #include "wireframe/net/tcp.hpp" @@ -122,8 +123,24 @@ inline std::string summarize_transport_and_above(const IpInfo& info) { out += " | " + *l7; } } + } else if (info.proto == net::kProtoIcmp) { + if (auto icmp = net::parse_icmpv4(info.payload)) { + out += " | ICMP " + net::icmpv4_type_name(icmp->type); + if (icmp->identifier) { + out += " id=" + std::to_string(*icmp->identifier) + + " seq=" + std::to_string(*icmp->sequence); + } + } } else if (info.proto == net::kNextHeaderIcmpv6) { - out += " | ICMPv6"; + if (auto icmp = net::parse_icmpv6(info.payload)) { + out += " | ICMPv6 " + net::icmpv6_type_name(icmp->type); + if (icmp->identifier) { + out += " id=" + std::to_string(*icmp->identifier) + + " seq=" + std::to_string(*icmp->sequence); + } + } else { + out += " | ICMPv6"; // truncated: at least say what it is + } } return out; } |