diff options
Diffstat (limited to 'include')
| -rw-r--r-- | include/wireframe/byteio.hpp | 22 | ||||
| -rw-r--r-- | include/wireframe/capture_queue.hpp | 98 | ||||
| -rw-r--r-- | include/wireframe/capture_session.hpp | 301 | ||||
| -rw-r--r-- | include/wireframe/filter.hpp | 36 | ||||
| -rw-r--r-- | include/wireframe/l7/dissector.hpp | 45 | ||||
| -rw-r--r-- | include/wireframe/l7/dns.hpp | 108 | ||||
| -rw-r--r-- | include/wireframe/l7/http.hpp | 113 | ||||
| -rw-r--r-- | include/wireframe/l7/tls.hpp | 139 | ||||
| -rw-r--r-- | include/wireframe/net/ethernet.hpp | 44 | ||||
| -rw-r--r-- | include/wireframe/net/ipv4.hpp | 56 | ||||
| -rw-r--r-- | include/wireframe/net/ipv6.hpp | 173 | ||||
| -rw-r--r-- | include/wireframe/net/tcp.hpp | 54 | ||||
| -rw-r--r-- | include/wireframe/net/udp.hpp | 35 | ||||
| -rw-r--r-- | include/wireframe/pcapng/reader.hpp | 123 | ||||
| -rw-r--r-- | include/wireframe/pcapng/writer.hpp | 94 | ||||
| -rw-r--r-- | include/wireframe/search.hpp | 26 | ||||
| -rw-r--r-- | include/wireframe/summarize.hpp | 248 |
17 files changed, 1715 insertions, 0 deletions
diff --git a/include/wireframe/byteio.hpp b/include/wireframe/byteio.hpp new file mode 100644 index 0000000..c37c29e --- /dev/null +++ b/include/wireframe/byteio.hpp @@ -0,0 +1,22 @@ +#pragma once + +#include <cstdint> +#include <span> + +// Manual big-endian reads instead of reinterpret_cast onto a packed +// struct: network buffers from pcap aren't guaranteed aligned for +// multi-byte integer types, so casting would be undefined behavior. +namespace wireframe { + +inline std::uint16_t read_be16(std::span<const unsigned char> bytes, std::size_t offset) { + return static_cast<std::uint16_t>((bytes[offset] << 8) | bytes[offset + 1]); +} + +inline std::uint32_t read_be32(std::span<const unsigned char> bytes, std::size_t offset) { + return (static_cast<std::uint32_t>(bytes[offset]) << 24) | + (static_cast<std::uint32_t>(bytes[offset + 1]) << 16) | + (static_cast<std::uint32_t>(bytes[offset + 2]) << 8) | + static_cast<std::uint32_t>(bytes[offset + 3]); +} + +} // namespace wireframe diff --git a/include/wireframe/capture_queue.hpp b/include/wireframe/capture_queue.hpp new file mode 100644 index 0000000..14794ba --- /dev/null +++ b/include/wireframe/capture_queue.hpp @@ -0,0 +1,98 @@ +#pragma once + +#include <condition_variable> +#include <cstdint> +#include <mutex> +#include <optional> +#include <queue> +#include <vector> + +// Bounded queue between the capture thread and the render/analysis +// thread (PLAN.md's architecture sketch). Owns a copy of each packet's +// bytes since the buffer libpcap hands the callback is only valid for +// the duration of that call. +namespace wireframe { + +struct CapturedPacket { + std::uint32_t ts_sec; + std::uint32_t ts_usec; + std::uint32_t original_len; + std::vector<unsigned char> data; // caplen bytes +}; + +// Single-producer / single-consumer. Two producer-side push variants +// for two different producers with different constraints: a live +// capture thread can't be allowed to stall (PLAN.md is explicit that a +// traffic spike should drop packets, not block), but a replay-from-file +// producer has no such real-time pressure, and dropping from a fixed +// historical record would defeat the point of "faithfully replaying +// what was captured" - so it blocks for room instead. +class CaptureQueue { +public: + explicit CaptureQueue(std::size_t capacity) : capacity_(capacity) {} + + // Never blocks: drops the packet and counts it if the queue is full. + bool try_push(CapturedPacket&& packet) { + { + std::lock_guard<std::mutex> lock(mutex_); + if (queue_.size() >= capacity_) { + ++dropped_; + return false; + } + queue_.push(std::move(packet)); + } + cv_.notify_all(); + return true; + } + + // Blocks until there's room, then pushes. Returns false without + // pushing if stop() is called while waiting - the consumer side is + // going away, so nothing will ever pop it. + bool push(CapturedPacket&& packet) { + { + std::unique_lock<std::mutex> lock(mutex_); + cv_.wait(lock, [this] { return queue_.size() < capacity_ || stopped_; }); + if (stopped_) return false; + queue_.push(std::move(packet)); + } + cv_.notify_all(); + return true; + } + + // Blocks until a packet is available. Returns nullopt only once + // stop() has been called and the queue has fully drained - so a + // consumer loop on pop() processes everything queued before the + // capture side stopped, rather than discarding it. + std::optional<CapturedPacket> pop() { + std::unique_lock<std::mutex> lock(mutex_); + cv_.wait(lock, [this] { return !queue_.empty() || stopped_; }); + if (queue_.empty()) return std::nullopt; + CapturedPacket packet = std::move(queue_.front()); + queue_.pop(); + cv_.notify_all(); // wake a push() blocked on room, if any + return packet; + } + + void stop() { + { + std::lock_guard<std::mutex> lock(mutex_); + stopped_ = true; + } + cv_.notify_all(); + } + + std::uint64_t dropped() const { + std::lock_guard<std::mutex> lock(mutex_); + return dropped_; + } + +private: + mutable std::mutex mutex_; + std::condition_variable cv_; + std::queue<CapturedPacket> queue_; + std::size_t capacity_; + bool stopped_ = false; + std::uint64_t dropped_ = 0; +}; + +} // namespace wireframe diff --git a/include/wireframe/capture_session.hpp b/include/wireframe/capture_session.hpp new file mode 100644 index 0000000..50764a8 --- /dev/null +++ b/include/wireframe/capture_session.hpp @@ -0,0 +1,301 @@ +#pragma once + +#include <pcap.h> + +#include <atomic> +#include <csignal> +#include <cstdio> +#include <cstring> +#include <optional> +#include <string> +#include <thread> + +#include "wireframe/capture_queue.hpp" +#include "wireframe/filter.hpp" +#include "wireframe/pcapng/reader.hpp" +#include "wireframe/pcapng/writer.hpp" + +// Device-open -> datalink-validate -> filter/pcapng-setup -> signal-hook +// pipeline, shared by every frontend (CLI, TUI, GUI). Centralized so a +// new frontend can't silently skip a step the others rely on - e.g. +// the DLT_RAW/DLT_EN10MB check that summarize_packet() depends on, or +// the pcap_breakloop() shutdown hook that keeps a -w pcapng file from +// being truncated on Ctrl-C (see main.cpp's history: both were real +// bugs before this was centralized). +// +// Also covers replay mode (-r <file>): reading a previously-saved +// pcapng file back through the exact same queue/render/search pipeline +// as a live capture, so every frontend gets it for free rather than +// needing a second code path. The render/consumer side only ever talks +// to a CaptureQueue - it has no way to tell whether packets are +// arriving from a live pcap_loop or being read back from disk. +namespace wireframe { + +namespace detail { +inline pcap_t* g_capture_handle = nullptr; +inline std::atomic<bool>* g_replay_stop_flag = nullptr; +inline void handle_stop_signal(int) { + if (g_capture_handle != nullptr) pcap_breakloop(g_capture_handle); + if (g_replay_stop_flag != nullptr) g_replay_stop_flag->store(true); +} +} // namespace detail + +struct CaptureSessionOptions { + std::string device; // empty = pick the first device via pcap_findalldevs + std::optional<std::string> filter_expr; + std::optional<std::string> pcapng_output_path; + std::optional<std::string> replay_input_path; // -r: read from this pcapng file, not a live device +}; + +inline bool is_supported_datalink(int datalink) { + return datalink == DLT_EN10MB || datalink == DLT_RAW; +} + +// Kernel/NIC-level counters, distinct from CaptureQueue::dropped(): +// the queue can only count packets libpcap already handed to our +// callback. A traffic spike can drop packets in the kernel's capture +// buffer before that ever happens - invisible without this. Not +// meaningful in replay mode (stats() returns nullopt there). +struct CaptureStats { + unsigned int received; // ps_recv + unsigned int dropped; // ps_drop: kernel buffer had no room + unsigned int if_dropped; // ps_ifdrop: dropped by the interface/driver +}; + +class CaptureSession { +public: + ~CaptureSession() { close(); } + + CaptureSession() = default; + CaptureSession(const CaptureSession&) = delete; + CaptureSession& operator=(const CaptureSession&) = delete; + + // Returns an error message on failure. The session remains safe to + // destroy (or close()) regardless of how far setup got. + std::optional<std::string> open(const CaptureSessionOptions& options) { + if (options.replay_input_path) { + if (options.filter_expr) { + return std::string( + "-f (capture filter) isn't supported with -r (replay); use -g to filter " + "what's displayed instead"); + } + return open_replay(*options.replay_input_path, options.pcapng_output_path); + } + + char errbuf[PCAP_ERRBUF_SIZE]; + + if (options.device.empty()) { + if (pcap_findalldevs(&all_devices_, errbuf) == -1 || all_devices_ == nullptr) { + return std::string("no capture device found: ") + errbuf; + } + device_ = all_devices_->name; + } else { + device_ = options.device; + } + + handle_ = pcap_open_live(device_.c_str(), /*snaplen=*/65535, /*promisc=*/0, + /*to_ms=*/1000, errbuf); + if (handle_ == nullptr) { + return std::string("pcap_open_live failed: ") + errbuf; + } + + datalink_ = pcap_datalink(handle_); + if (!is_supported_datalink(datalink_)) { + return std::string("unsupported datalink type on ") + device_ + ": " + + pcap_datalink_val_to_name(datalink_) + " (" + + pcap_datalink_val_to_description(datalink_) + ")"; + } + + if (options.filter_expr) { + bpf_program program{}; + if (auto err = compile_filter(handle_, *options.filter_expr, &program)) { + return "invalid filter '" + *options.filter_expr + "': " + *err; + } + if (pcap_setfilter(handle_, &program) == -1) { + std::string err = std::string("pcap_setfilter failed: ") + pcap_geterr(handle_); + pcap_freecode(&program); + return err; + } + pcap_freecode(&program); // bytecode is copied into the kernel by pcap_setfilter + } + + if (options.pcapng_output_path) { + if (auto err = open_pcapng_writer(*options.pcapng_output_path)) return err; + } + + return std::nullopt; + } + + // pcap_loop() blocks in a read/poll waiting for the next packet, so + // a plain "stop requested" flag wouldn't unblock it promptly. + // pcap_breakloop() is documented as signal-safe and is what + // actually interrupts that wait. Replay mode has no handle to + // breakloop, so it's interrupted via g_replay_stop_flag instead -- + // both are armed here so one signal handler covers either mode. + void install_signal_handlers() { + detail::g_capture_handle = handle_; + detail::g_replay_stop_flag = &replay_stop_requested_; + std::signal(SIGINT, detail::handle_stop_signal); + std::signal(SIGTERM, detail::handle_stop_signal); + } + + void request_stop() { + if (handle_ != nullptr) pcap_breakloop(handle_); + replay_stop_requested_.store(true); + } + + // True once a stop has been explicitly requested - via + // request_stop() or an external SIGINT/SIGTERM (the signal handler + // sets the same flag). Lets a frontend tell "the producer stopped + // because someone asked it to" apart from "the producer ran out of + // data on its own" (replay reaching end-of-file), which call for + // different UI behavior: the former should close the window, the + // latter should leave it open so what's already loaded can still be + // browsed. + bool stop_requested() const { return replay_stop_requested_.load(); } + + // Must be called before close()/the destructor - pcap_stats() + // needs a still-open handle. Safe to call after request_stop(), + // since breakloop only stops pcap_loop(), it doesn't close handle_. + // Always nullopt in replay mode (handle_ is never set there). + std::optional<CaptureStats> stats() const { + if (handle_ == nullptr) return std::nullopt; + pcap_stat stat{}; + if (pcap_stats(handle_, &stat) == -1) return std::nullopt; + return CaptureStats{stat.ps_recv, stat.ps_drop, stat.ps_ifdrop}; + } + + // Capture-thread side: copy each packet into the queue and return + // immediately. No decoding, printing, or file I/O here - that's + // every frontend's own consumer-side job. + // + // Live mode drops on backpressure (try_push, via capture_callback) + // since a traffic spike can't be paused. Replay mode blocks instead + // (push): a file has no real-time pressure forcing a drop, and + // dropping from what's supposed to be a faithful replay of a fixed + // historical record would defeat the point of replaying it. + std::thread start_capture_thread(CaptureQueue& queue) { + if (is_replay_) { + return std::thread([this, &queue] { + queue.push(to_captured_packet(std::move(*first_replay_packet_))); + while (!replay_stop_requested_.load()) { + auto record = replay_reader_->next_packet(); + if (!record) break; + if (!queue.push(to_captured_packet(std::move(*record)))) break; + } + queue.stop(); + }); + } + return std::thread([this, &queue] { + pcap_loop(handle_, /*count=*/-1, capture_callback, + reinterpret_cast<unsigned char*>(&queue)); + queue.stop(); + }); + } + + void close() { + if (handle_ != nullptr) { + pcap_close(handle_); + handle_ = nullptr; + } + if (all_devices_ != nullptr) { + pcap_freealldevs(all_devices_); + all_devices_ = nullptr; + } + if (pcapng_file_ != nullptr) { + std::fclose(pcapng_file_); + pcapng_file_ = nullptr; + } + if (replay_file_ != nullptr) { + std::fclose(replay_file_); + replay_file_ = nullptr; + } + } + + pcap_t* handle() const { return handle_; } + const std::string& device() const { return device_; } + int datalink() const { return datalink_; } + bool is_replay() const { return is_replay_; } + pcapng::Writer* pcapng_writer() { return pcapng_writer_ ? &*pcapng_writer_ : nullptr; } + +private: + static void capture_callback(unsigned char* user, const pcap_pkthdr* header, + const unsigned char* raw) { + auto* queue = reinterpret_cast<CaptureQueue*>(user); + CapturedPacket packet; + packet.ts_sec = static_cast<std::uint32_t>(header->ts.tv_sec); + packet.ts_usec = static_cast<std::uint32_t>(header->ts.tv_usec); + packet.original_len = header->len; + packet.data.assign(raw, raw + header->caplen); + queue->try_push(std::move(packet)); + } + + static CapturedPacket to_captured_packet(pcapng::PacketRecord&& record) { + CapturedPacket packet; + packet.ts_sec = static_cast<std::uint32_t>(record.timestamp_us / 1'000'000ULL); + packet.ts_usec = static_cast<std::uint32_t>(record.timestamp_us % 1'000'000ULL); + packet.original_len = record.original_len; + packet.data = std::move(record.data); + return packet; + } + + std::optional<std::string> open_pcapng_writer(const std::string& path) { + pcapng_file_ = std::fopen(path.c_str(), "wb"); + if (pcapng_file_ == nullptr) { + return "failed to open " + path + " for writing: " + std::strerror(errno); + } + pcapng_writer_.emplace(pcapng_file_); + pcapng_writer_->write_section_header(); + pcapng_writer_->write_interface_description(65535, + static_cast<std::uint16_t>(datalink_)); + return std::nullopt; + } + + std::optional<std::string> open_replay(const std::string& path, + const std::optional<std::string>& pcapng_output_path) { + replay_file_ = std::fopen(path.c_str(), "rb"); + if (replay_file_ == nullptr) { + return "failed to open " + path + " for reading: " + std::strerror(errno); + } + + replay_reader_.emplace(replay_file_); + // Reading the first packet is also what makes the reader consume + // the SHB/IDB blocks that precede it, which is what populates + // link_type() below - there's no separate "just read the + // header" step, so the packet itself is kept, not discarded. + first_replay_packet_ = replay_reader_->next_packet(); + if (!first_replay_packet_) { + return "no packets found in " + path + " (empty, or not a valid pcapng file)"; + } + + auto link_type = replay_reader_->link_type(); + if (!link_type || !is_supported_datalink(static_cast<int>(*link_type))) { + return "unsupported or missing link type in " + path; + } + + datalink_ = static_cast<int>(*link_type); + device_ = path; + is_replay_ = true; + + if (pcapng_output_path) { + if (auto err = open_pcapng_writer(*pcapng_output_path)) return err; + } + + return std::nullopt; + } + + pcap_t* handle_ = nullptr; + pcap_if_t* all_devices_ = nullptr; + std::string device_; + int datalink_ = 0; + std::FILE* pcapng_file_ = nullptr; + std::optional<pcapng::Writer> pcapng_writer_; + + bool is_replay_ = false; + std::FILE* replay_file_ = nullptr; + std::optional<pcapng::Reader> replay_reader_; + std::optional<pcapng::PacketRecord> first_replay_packet_; + std::atomic<bool> replay_stop_requested_{false}; +}; + +} // namespace wireframe diff --git a/include/wireframe/filter.hpp b/include/wireframe/filter.hpp new file mode 100644 index 0000000..49fa4ab --- /dev/null +++ b/include/wireframe/filter.hpp @@ -0,0 +1,36 @@ +#pragma once + +#include <pcap.h> + +#include <optional> +#include <string> + +// Thin wrapper around libpcap's BPF filter compiler. tcpdump-style +// filter syntax ("tcp port 80", "host 10.0.0.1 and not icmp") already +// has a correct, well-tested parser and compiler in libpcap itself -- +// hand-rolling a second one would be a large, separate project with no +// bearing on this one's actual goal (the C++ memory model), so this +// wraps the existing implementation instead of reinventing it. +namespace wireframe { + +// Compiles `expression` against `handle`'s linktype/snaplen into +// `out`. `handle` can be a real, already-open capture handle, or a +// throwaway one from pcap_open_dead() - pcap_compile() only needs the +// handle to know the linktype and to report errors via pcap_geterr(), +// it doesn't require an active capture. That's what makes this +// testable without root or a real interface. +// +// Returns nullopt on success (with `out` filled in and owned by the +// caller - pcap_freecode(out) once it's no longer needed, including +// after a successful pcap_setfilter()). Returns pcap's error message +// on failure, and leaves `out` unmodified. +inline std::optional<std::string> compile_filter(pcap_t* handle, const std::string& expression, + bpf_program* out) { + if (pcap_compile(handle, out, expression.c_str(), /*optimize=*/1, PCAP_NETMASK_UNKNOWN) == + -1) { + return std::string(pcap_geterr(handle)); + } + return std::nullopt; +} + +} // namespace wireframe diff --git a/include/wireframe/l7/dissector.hpp b/include/wireframe/l7/dissector.hpp new file mode 100644 index 0000000..9b2cc32 --- /dev/null +++ b/include/wireframe/l7/dissector.hpp @@ -0,0 +1,45 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> +#include <string> +#include <vector> + +// Small interface/vtable for L7 dissectors (PLAN.md's architecture +// sketch), so protocols can be registered and added incrementally +// without touching the L2-L4 decode path or main.cpp's dispatch logic. +namespace wireframe::net { + +class L7Dissector { +public: + virtual ~L7Dissector() = default; + + // The transport port this dissector claims (e.g. 53 for DNS). A + // single fixed port is enough for the protocols in scope so far; + // dissectors needing a port range or heuristic sniffing can widen + // this later without changing the registry's shape. + virtual std::uint16_t port() const = 0; + + // A one-line summary of the payload, or nullopt if it doesn't look + // like this protocol (e.g. truncated/malformed). + virtual std::optional<std::string> summarize(std::span<const unsigned char> payload) const = 0; +}; + +class L7Registry { +public: + void add(const L7Dissector* dissector) { dissectors_.push_back(dissector); } + + std::optional<std::string> dissect(std::uint16_t port, + std::span<const unsigned char> payload) const { + for (const auto* dissector : dissectors_) { + if (dissector->port() == port) return dissector->summarize(payload); + } + return std::nullopt; + } + +private: + std::vector<const L7Dissector*> dissectors_; +}; + +} // namespace wireframe::net diff --git a/include/wireframe/l7/dns.hpp b/include/wireframe/l7/dns.hpp new file mode 100644 index 0000000..5c1ab36 --- /dev/null +++ b/include/wireframe/l7/dns.hpp @@ -0,0 +1,108 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> +#include <string> +#include <utility> + +#include "wireframe/byteio.hpp" +#include "wireframe/l7/dissector.hpp" + +// Hand-rolled DNS message parsing: header + the first question record. +// Answer/authority/additional records aren't decoded (not needed for a +// one-line summary), so name-compression pointers there are never +// followed - a pointer in the question section itself is rejected +// rather than chased, keeping this a pure forward scan with no risk of +// a pointer loop. +namespace wireframe::net { + +inline constexpr std::uint16_t kDnsPort = 53; + +struct DnsHeader { + std::uint16_t id; + bool is_response; + std::uint8_t opcode; + std::uint8_t rcode; + std::uint16_t qdcount; + std::uint16_t ancount; +}; + +struct DnsQuestion { + std::string name; + std::uint16_t qtype; +}; + +struct DnsMessage { + DnsHeader header; + std::optional<DnsQuestion> question; // first question only +}; + +// Reads a (possibly multi-label) dotted name starting at offset. +// Returns the name and the offset just past it, or nullopt on +// truncation or a compression pointer (0xC0 prefix - valid in +// answer/authority records, not supported here). +inline std::optional<std::pair<std::string, std::size_t>> read_dns_name( + std::span<const unsigned char> bytes, std::size_t offset) { + std::string name; + while (true) { + if (offset >= bytes.size()) return std::nullopt; + std::uint8_t len = bytes[offset]; + if (len == 0) { + ++offset; + break; + } + if ((len & 0xC0) == 0xC0) return std::nullopt; // compression pointer: unsupported + ++offset; + if (offset + len > bytes.size()) return std::nullopt; + if (!name.empty()) name += '.'; + for (std::uint8_t i = 0; i < len; ++i) name += static_cast<char>(bytes[offset + i]); + offset += len; + } + return std::make_pair(std::move(name), offset); +} + +inline std::optional<DnsMessage> parse_dns(std::span<const unsigned char> bytes) { + if (bytes.size() < 12) return std::nullopt; + + DnsHeader header{}; + header.id = read_be16(bytes, 0); + std::uint16_t flags = read_be16(bytes, 2); + header.is_response = (flags & 0x8000) != 0; + header.opcode = static_cast<std::uint8_t>((flags >> 11) & 0x0F); + header.rcode = static_cast<std::uint8_t>(flags & 0x0F); + header.qdcount = read_be16(bytes, 4); + header.ancount = read_be16(bytes, 6); + + DnsMessage msg{header, std::nullopt}; + if (header.qdcount >= 1) { + if (auto result = read_dns_name(bytes, 12)) { + auto& [name, next_offset] = *result; + if (next_offset + 4 <= bytes.size()) { + msg.question = DnsQuestion{std::move(name), read_be16(bytes, next_offset)}; + } + } + } + return msg; +} + +class DnsDissector : public L7Dissector { +public: + std::uint16_t port() const override { return kDnsPort; } + + std::optional<std::string> summarize(std::span<const unsigned char> payload) const override { + auto msg = parse_dns(payload); + if (!msg) return std::nullopt; + + std::string out = "DNS "; + out += msg->header.is_response ? "response" : "query"; + out += " id=" + std::to_string(msg->header.id); + if (msg->header.is_response) out += " ancount=" + std::to_string(msg->header.ancount); + if (msg->question) { + out += " " + msg->question->name + " type=" + std::to_string(msg->question->qtype); + } + return out; + } +}; + +} // namespace wireframe::net diff --git a/include/wireframe/l7/http.hpp b/include/wireframe/l7/http.hpp new file mode 100644 index 0000000..4780b23 --- /dev/null +++ b/include/wireframe/l7/http.hpp @@ -0,0 +1,113 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> +#include <string> +#include <string_view> + +#include "wireframe/l7/dissector.hpp" + +// Best-effort, single-segment HTTP/1.x request/status-line parsing (plus +// the Host: header for requests). No TCP stream reassembly, so a +// message split across multiple packets is only partially visible here +// - the same scope DNS already has (single UDP datagram, no +// reassembly). Good enough for a one-line summary, not a full dissector. +namespace wireframe::net { + +inline constexpr std::uint16_t kHttpPort = 80; + +struct HttpMessage { + bool is_request; + std::string method_or_version; // request: method (GET); response: "HTTP/1.1" + std::string target_or_status; // request: target path; response: status code + std::optional<std::string> host; // request only, from a Host: header if present +}; + +inline std::optional<HttpMessage> parse_http(std::span<const unsigned char> payload) { + std::string_view text(reinterpret_cast<const char*>(payload.data()), payload.size()); + + std::size_t line_end = text.find("\r\n"); + std::size_t term_len = 2; + if (line_end == std::string_view::npos) { + line_end = text.find('\n'); + term_len = 1; + if (line_end == std::string_view::npos) return std::nullopt; + } + std::string_view first_line = text.substr(0, line_end); + + std::size_t sp1 = first_line.find(' '); + if (sp1 == std::string_view::npos) return std::nullopt; + std::size_t sp2 = first_line.find(' ', sp1 + 1); + if (sp2 == std::string_view::npos) return std::nullopt; + + std::string_view field1 = first_line.substr(0, sp1); + std::string_view field2 = first_line.substr(sp1 + 1, sp2 - sp1 - 1); + + HttpMessage msg; + + if (field1.substr(0, 5) == "HTTP/") { + msg.is_request = false; + msg.method_or_version = std::string(field1); + msg.target_or_status = std::string(field2); + return msg; + } + + static constexpr std::string_view kMethods[] = {"GET", "POST", "PUT", "DELETE", + "HEAD", "OPTIONS", "PATCH", "CONNECT", + "TRACE"}; + bool known_method = false; + for (auto method : kMethods) { + if (field1 == method) { + known_method = true; + break; + } + } + if (!known_method) return std::nullopt; + + msg.is_request = true; + msg.method_or_version = std::string(field1); + msg.target_or_status = std::string(field2); + + // Best-effort Host: header scan, bounded by whatever this one + // packet contains and terminated at the first blank line (end of + // headers) or the end of the payload - never loops past text.size(). + std::size_t pos = line_end + term_len; + while (pos < text.size()) { + std::size_t next_end = text.find("\r\n", pos); + std::size_t header_len = (next_end == std::string_view::npos) ? text.size() - pos + : next_end - pos; + std::string_view header_line = text.substr(pos, header_len); + if (header_line.empty()) break; // blank line: end of headers + + if (header_line.size() > 5 && + (header_line.substr(0, 5) == "Host:" || header_line.substr(0, 5) == "host:")) { + std::size_t value_start = 5; + while (value_start < header_line.size() && header_line[value_start] == ' ') { + ++value_start; + } + msg.host = std::string(header_line.substr(value_start)); + } + + if (next_end == std::string_view::npos) break; + pos = next_end + 2; + } + + return msg; +} + +class HttpDissector : public L7Dissector { +public: + std::uint16_t port() const override { return kHttpPort; } + + std::optional<std::string> summarize(std::span<const unsigned char> payload) const override { + auto msg = parse_http(payload); + if (!msg) return std::nullopt; + + std::string out = "HTTP " + msg->method_or_version + " " + msg->target_or_status; + if (msg->host) out += " Host: " + *msg->host; + return out; + } +}; + +} // namespace wireframe::net diff --git a/include/wireframe/l7/tls.hpp b/include/wireframe/l7/tls.hpp new file mode 100644 index 0000000..1c6dc57 --- /dev/null +++ b/include/wireframe/l7/tls.hpp @@ -0,0 +1,139 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> +#include <string> + +#include "wireframe/byteio.hpp" +#include "wireframe/l7/dissector.hpp" + +// TLS ClientHello -> SNI extension parsing. Most web traffic is TLS +// today, so HTTP alone covers a shrinking fraction of it - SNI is what +// makes a packet analyzer useful against that traffic without +// decrypting anything: the server name is sent in cleartext in the +// ClientHello, before any encryption starts, in every TLS version this +// parses (the ClientHello/extension wire format hasn't changed across +// versions - only what happens after it has). +// +// Same scope as the other L7 dissectors: single-segment, best-effort. +// A ClientHello padded across multiple TCP segments (large cookie/PSK +// extensions, unusual but possible) is only partially visible here. +// Every length field is bounds-checked against what's actually left in +// the buffer before use - this is exactly the kind of nested, +// attacker-influenced TLV structure the project's decoders are meant +// to get right. +namespace wireframe::net { + +inline constexpr std::uint16_t kTlsPort = 443; +inline constexpr std::uint8_t kTlsContentTypeHandshake = 0x16; +inline constexpr std::uint8_t kTlsHandshakeTypeClientHello = 0x01; +inline constexpr std::uint16_t kTlsExtensionServerName = 0x0000; + +struct TlsClientHello { + std::optional<std::string> server_name; // SNI, if the extension was present and well-formed +}; + +inline std::optional<TlsClientHello> parse_tls_client_hello(std::span<const unsigned char> bytes) { + // Record header: ContentType(1) ProtocolVersion(2) Length(2) + if (bytes.size() < 5) return std::nullopt; + if (bytes[0] != kTlsContentTypeHandshake) return std::nullopt; + std::uint16_t record_len = read_be16(bytes, 3); + if (bytes.size() < static_cast<std::size_t>(5) + record_len) return std::nullopt; + + std::span<const unsigned char> handshake = bytes.subspan(5); + + // Handshake header: HandshakeType(1) Length(3, 24-bit BE) + if (handshake.size() < 4) return std::nullopt; + if (handshake[0] != kTlsHandshakeTypeClientHello) return std::nullopt; + std::uint32_t hs_len = (static_cast<std::uint32_t>(handshake[1]) << 16) | + (static_cast<std::uint32_t>(handshake[2]) << 8) | + static_cast<std::uint32_t>(handshake[3]); + + std::span<const unsigned char> body = handshake.subspan(4); + if (body.size() < hs_len) return std::nullopt; + body = body.first(hs_len); // never read past the declared handshake body + + std::size_t offset = 0; + + // client_version(2) + random(32) + if (body.size() < offset + 34) return std::nullopt; + offset += 34; + + // legacy_session_id: length(1) + data + if (body.size() < offset + 1) return std::nullopt; + std::uint8_t session_id_len = body[offset]; + offset += 1; + if (body.size() < offset + session_id_len) return std::nullopt; + offset += session_id_len; + + // cipher_suites: length(2) + data + if (body.size() < offset + 2) return std::nullopt; + std::uint16_t cipher_suites_len = read_be16(body, offset); + offset += 2; + if (body.size() < static_cast<std::size_t>(offset) + cipher_suites_len) return std::nullopt; + offset += cipher_suites_len; + + // legacy_compression_methods: length(1) + data + if (body.size() < offset + 1) return std::nullopt; + std::uint8_t compression_len = body[offset]; + offset += 1; + if (body.size() < offset + compression_len) return std::nullopt; + offset += compression_len; + + TlsClientHello hello; + if (offset == body.size()) return hello; // no extensions block: no SNI, still a valid hello + + // extensions: length(2) + data + if (body.size() < offset + 2) return std::nullopt; + std::uint16_t extensions_len = read_be16(body, offset); + offset += 2; + if (body.size() < static_cast<std::size_t>(offset) + extensions_len) return std::nullopt; + std::size_t extensions_end = offset + extensions_len; + + while (offset + 4 <= extensions_end) { + std::uint16_t ext_type = read_be16(body, offset); + std::uint16_t ext_len = read_be16(body, offset + 2); + std::size_t ext_data_start = offset + 4; + std::size_t ext_data_end = ext_data_start + ext_len; + if (ext_data_end > extensions_end) break; // malformed: stop, keep what we have + + if (ext_type == kTlsExtensionServerName && ext_len >= 2) { + // ServerNameList: list_len(2) + entries; only the first + // entry is used, matching every real client's behavior of + // sending exactly one host_name entry. + std::uint16_t list_len = read_be16(body, ext_data_start); + std::size_t list_start = ext_data_start + 2; + std::size_t list_end = list_start + list_len; + if (list_end <= ext_data_end && list_start + 3 <= list_end) { + std::uint8_t name_type = body[list_start]; + std::uint16_t name_len = read_be16(body, list_start + 1); + std::size_t name_start = list_start + 3; + if (name_type == 0 && name_start + name_len <= list_end) { + hello.server_name = std::string( + reinterpret_cast<const char*>(body.data() + name_start), name_len); + } + } + } + + offset = ext_data_end; + } + + return hello; +} + +class TlsSniDissector : public L7Dissector { +public: + std::uint16_t port() const override { return kTlsPort; } + + std::optional<std::string> summarize(std::span<const unsigned char> payload) const override { + auto hello = parse_tls_client_hello(payload); + if (!hello) return std::nullopt; + + std::string out = "TLS ClientHello"; + if (hello->server_name) out += " SNI=" + *hello->server_name; + return out; + } +}; + +} // namespace wireframe::net diff --git a/include/wireframe/net/ethernet.hpp b/include/wireframe/net/ethernet.hpp new file mode 100644 index 0000000..2da4cc8 --- /dev/null +++ b/include/wireframe/net/ethernet.hpp @@ -0,0 +1,44 @@ +#pragma once + +#include <algorithm> +#include <array> +#include <cstdint> +#include <optional> +#include <span> + +#include "wireframe/byteio.hpp" + +namespace wireframe::net { + +inline constexpr std::size_t kEthernetHeaderLen = 14; +inline constexpr std::uint16_t kEthertypeIPv4 = 0x0800; +inline constexpr std::uint16_t kEthertypeIPv6 = 0x86DD; +inline constexpr std::uint16_t kEthertypeArp = 0x0806; + +struct MacAddress { + std::array<unsigned char, 6> bytes; +}; + +struct EthernetHeader { + MacAddress dst; + MacAddress src; + std::uint16_t ethertype; +}; + +struct EthernetFrame { + EthernetHeader header; + std::span<const unsigned char> payload; +}; + +inline std::optional<EthernetFrame> parse_ethernet(std::span<const unsigned char> bytes) { + if (bytes.size() < kEthernetHeaderLen) return std::nullopt; + + EthernetHeader header{}; + std::copy_n(bytes.begin(), 6, header.dst.bytes.begin()); + std::copy_n(bytes.begin() + 6, 6, header.src.bytes.begin()); + header.ethertype = read_be16(bytes, 12); + + return EthernetFrame{header, bytes.subspan(kEthernetHeaderLen)}; +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/ipv4.hpp b/include/wireframe/net/ipv4.hpp new file mode 100644 index 0000000..f53b4f2 --- /dev/null +++ b/include/wireframe/net/ipv4.hpp @@ -0,0 +1,56 @@ +#pragma once + +#include <algorithm> +#include <array> +#include <cstdint> +#include <optional> +#include <span> + +#include "wireframe/byteio.hpp" + +namespace wireframe::net { + +inline constexpr std::uint8_t kProtoIcmp = 1; +inline constexpr std::uint8_t kProtoTcp = 6; +inline constexpr std::uint8_t kProtoUdp = 17; + +struct Ipv4Address { + std::array<unsigned char, 4> bytes; +}; + +struct Ipv4Header { + std::uint8_t version; + std::uint8_t ihl; // header length in 32-bit words + std::uint16_t total_length; + std::uint8_t ttl; + std::uint8_t protocol; + Ipv4Address src; + Ipv4Address dst; +}; + +struct Ipv4Packet { + Ipv4Header header; + std::span<const unsigned char> payload; +}; + +inline std::optional<Ipv4Packet> parse_ipv4(std::span<const unsigned char> bytes) { + if (bytes.size() < 20) return std::nullopt; + + std::uint8_t version = static_cast<std::uint8_t>(bytes[0] >> 4); + std::uint8_t ihl = bytes[0] & 0x0F; + std::size_t header_len = static_cast<std::size_t>(ihl) * 4; + if (version != 4 || header_len < 20 || bytes.size() < header_len) return std::nullopt; + + Ipv4Header header{}; + header.version = version; + header.ihl = ihl; + header.total_length = read_be16(bytes, 2); + header.ttl = bytes[8]; + header.protocol = bytes[9]; + std::copy_n(bytes.begin() + 12, 4, header.src.bytes.begin()); + std::copy_n(bytes.begin() + 16, 4, header.dst.bytes.begin()); + + return Ipv4Packet{header, bytes.subspan(header_len)}; +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/ipv6.hpp b/include/wireframe/net/ipv6.hpp new file mode 100644 index 0000000..4b6b28a --- /dev/null +++ b/include/wireframe/net/ipv6.hpp @@ -0,0 +1,173 @@ +#pragma once + +#include <algorithm> +#include <array> +#include <cstdint> +#include <cstdio> +#include <optional> +#include <span> +#include <string> + +#include "wireframe/byteio.hpp" + +namespace wireframe::net { + +inline constexpr std::size_t kIpv6HeaderLen = 40; +inline constexpr std::uint8_t kNextHeaderHopByHop = 0; +inline constexpr std::uint8_t kNextHeaderRouting = 43; +inline constexpr std::uint8_t kNextHeaderFragment = 44; +inline constexpr std::uint8_t kNextHeaderEsp = 50; +inline constexpr std::uint8_t kNextHeaderAh = 51; +inline constexpr std::uint8_t kNextHeaderIcmpv6 = 58; +inline constexpr std::uint8_t kNextHeaderDestOptions = 60; + +struct Ipv6Address { + std::array<unsigned char, 16> bytes; +}; + +struct Ipv6Header { + std::uint8_t version; + std::uint8_t traffic_class; + std::uint32_t flow_label; + std::uint16_t payload_length; + std::uint8_t next_header; // transport protocol, or an extension header type + std::uint8_t hop_limit; + Ipv6Address src; + Ipv6Address dst; +}; + +struct Ipv6Packet { + Ipv6Header header; + std::span<const unsigned char> payload; +}; + +// Only the fixed 40-byte header is decoded here - header.next_header +// may name an extension header rather than a transport protocol. +// walk_ipv6_extension_headers() (below) resolves that; parse_ipv6() +// itself stays a direct, unconditional decode of exactly the fixed +// header, nothing more. +inline std::optional<Ipv6Packet> parse_ipv6(std::span<const unsigned char> bytes) { + if (bytes.size() < kIpv6HeaderLen) return std::nullopt; + + std::uint8_t version = static_cast<std::uint8_t>(bytes[0] >> 4); + if (version != 6) return std::nullopt; + + Ipv6Header header{}; + header.version = version; + std::uint32_t first_word = read_be32(bytes, 0); + header.traffic_class = static_cast<std::uint8_t>((first_word >> 20) & 0xFF); + header.flow_label = first_word & 0x000FFFFF; + header.payload_length = read_be16(bytes, 4); + header.next_header = bytes[6]; + header.hop_limit = bytes[7]; + std::copy_n(bytes.begin() + 8, 16, header.src.bytes.begin()); + std::copy_n(bytes.begin() + 24, 16, header.dst.bytes.begin()); + + return Ipv6Packet{header, bytes.subspan(kIpv6HeaderLen)}; +} + +struct Ipv6ExtensionWalkResult { + std::uint8_t final_next_header; // a transport protocol, or an extension type we stopped at + std::span<const unsigned char> payload; // bytes after every extension header walked + bool stopped_at_esp; // true if ESP was hit - see walk_ipv6_extension_headers() +}; + +// Walks Hop-by-Hop, Routing, Destination Options, Fragment, and AH +// extension headers to find the real transport protocol underneath +// them, so e.g. TCP/UDP wrapped in a Hop-by-Hop options header is still +// decoded instead of silently stopping at "next_header=0". Each header +// carries its own length, so this never needs to understand a header +// type's *meaning* to skip over it correctly - only Hop-by-Hop/ +// Routing/Dest-Options (length in 8-byte units from a trailing byte), +// Fragment (fixed 8 bytes), and AH (length in 4-byte units, RFC 4302) +// have different encodings, all handled explicitly below. +// +// ESP is a hard stop, not a bug: its own next-header field lives in a +// trailer *after* the encrypted payload, at an offset this code has no +// way to know without decrypting first. Reported as stopped_at_esp +// rather than guessed at. +// +// Bounded to a handful of iterations as defense in depth against a +// hostile/corrupt chain - not strictly needed for termination (every +// header is at least 8 bytes, so payload.size() strictly decreases +// each iteration and the loop can't actually run forever), but a +// pathological chain of many tiny headers would otherwise still cost +// real work for no legitimate reason. +inline Ipv6ExtensionWalkResult walk_ipv6_extension_headers(std::uint8_t next_header, + std::span<const unsigned char> payload) { + constexpr int kMaxExtensionHeaders = 8; + + for (int i = 0; i < kMaxExtensionHeaders; ++i) { + if (next_header == kNextHeaderEsp) { + return {next_header, payload, /*stopped_at_esp=*/true}; + } + + std::size_t ext_len; + if (next_header == kNextHeaderFragment) { + if (payload.size() < 8) return {next_header, payload, false}; + ext_len = 8; + } else if (next_header == kNextHeaderAh) { + if (payload.size() < 2) return {next_header, payload, false}; + ext_len = (static_cast<std::size_t>(payload[1]) + 2) * 4; + } else if (next_header == kNextHeaderHopByHop || next_header == kNextHeaderRouting || + next_header == kNextHeaderDestOptions) { + if (payload.size() < 2) return {next_header, payload, false}; + ext_len = (static_cast<std::size_t>(payload[1]) + 1) * 8; + } else { + break; // TCP/UDP/ICMPv6/anything else we don't chain through: stop here + } + + if (payload.size() < ext_len) return {next_header, payload, false}; // truncated: stop + + std::uint8_t this_next_header = payload[0]; + payload = payload.subspan(ext_len); + next_header = this_next_header; + } + + return {next_header, payload, false}; +} + +// RFC 5952 canonical text form: lowercase hex, and the longest run of +// two-or-more consecutive zero groups (leftmost wins a tie) collapsed to +// "::". A lone zero group is left as "0", not compressed, per 5952 4.2.2. +inline std::string ipv6_to_string(const Ipv6Address& addr) { + std::array<std::uint16_t, 8> groups{}; + for (std::size_t i = 0; i < 8; ++i) { + groups[i] = static_cast<std::uint16_t>((addr.bytes[i * 2] << 8) | addr.bytes[i * 2 + 1]); + } + + int best_start = -1; + int best_len = 0; + int cur_start = -1; + int cur_len = 0; + for (int i = 0; i < 8; ++i) { + if (groups[i] == 0) { + if (cur_start < 0) cur_start = i; + ++cur_len; + if (cur_len > best_len) { + best_start = cur_start; + best_len = cur_len; + } + } else { + cur_start = -1; + cur_len = 0; + } + } + if (best_len < 2) best_start = -1; // don't compress a lone zero group + + std::string out; + char buf[6]; + for (int i = 0; i < 8; ++i) { + if (i == best_start) { + out += "::"; + i += best_len - 1; // the for-loop's ++i advances past the run + continue; + } + if (!out.empty() && out.back() != ':') out += ':'; + std::snprintf(buf, sizeof(buf), "%x", groups[i]); + out += buf; + } + return out; +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/tcp.hpp b/include/wireframe/net/tcp.hpp new file mode 100644 index 0000000..f691a7f --- /dev/null +++ b/include/wireframe/net/tcp.hpp @@ -0,0 +1,54 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> + +#include "wireframe/byteio.hpp" + +namespace wireframe::net { + +// Lower 6 bits of the flags byte: URG ACK PSH RST SYN FIN. CWR/ECE (the +// top 2 bits) are masked off - not needed for now. +inline constexpr std::uint8_t kTcpFin = 0x01; +inline constexpr std::uint8_t kTcpSyn = 0x02; +inline constexpr std::uint8_t kTcpRst = 0x04; +inline constexpr std::uint8_t kTcpPsh = 0x08; +inline constexpr std::uint8_t kTcpAck = 0x10; +inline constexpr std::uint8_t kTcpUrg = 0x20; + +struct TcpHeader { + std::uint16_t src_port; + std::uint16_t dst_port; + std::uint32_t seq; + std::uint32_t ack; + std::uint8_t data_offset; // header length in 32-bit words + std::uint8_t flags; + std::uint16_t window; +}; + +struct TcpSegment { + TcpHeader header; + std::span<const unsigned char> payload; +}; + +inline std::optional<TcpSegment> parse_tcp(std::span<const unsigned char> bytes) { + if (bytes.size() < 20) return std::nullopt; + + std::uint8_t data_offset = static_cast<std::uint8_t>(bytes[12] >> 4); + std::size_t header_len = static_cast<std::size_t>(data_offset) * 4; + if (header_len < 20 || bytes.size() < header_len) return std::nullopt; + + TcpHeader header{}; + header.src_port = read_be16(bytes, 0); + header.dst_port = read_be16(bytes, 2); + header.seq = read_be32(bytes, 4); + header.ack = read_be32(bytes, 8); + header.data_offset = data_offset; + header.flags = bytes[13] & 0x3F; + header.window = read_be16(bytes, 14); + + return TcpSegment{header, bytes.subspan(header_len)}; +} + +} // namespace wireframe::net diff --git a/include/wireframe/net/udp.hpp b/include/wireframe/net/udp.hpp new file mode 100644 index 0000000..07664c2 --- /dev/null +++ b/include/wireframe/net/udp.hpp @@ -0,0 +1,35 @@ +#pragma once + +#include <cstdint> +#include <optional> +#include <span> + +#include "wireframe/byteio.hpp" + +namespace wireframe::net { + +inline constexpr std::size_t kUdpHeaderLen = 8; + +struct UdpHeader { + std::uint16_t src_port; + std::uint16_t dst_port; + std::uint16_t length; +}; + +struct UdpDatagram { + UdpHeader header; + std::span<const unsigned char> payload; +}; + +inline std::optional<UdpDatagram> parse_udp(std::span<const unsigned char> bytes) { + if (bytes.size() < kUdpHeaderLen) return std::nullopt; + + UdpHeader header{}; + header.src_port = read_be16(bytes, 0); + header.dst_port = read_be16(bytes, 2); + header.length = read_be16(bytes, 4); + + return UdpDatagram{header, bytes.subspan(kUdpHeaderLen)}; +} + +} // namespace wireframe::net diff --git a/include/wireframe/pcapng/reader.hpp b/include/wireframe/pcapng/reader.hpp new file mode 100644 index 0000000..d01b431 --- /dev/null +++ b/include/wireframe/pcapng/reader.hpp @@ -0,0 +1,123 @@ +#pragma once + +#include <array> +#include <cstdint> +#include <cstdio> +#include <optional> +#include <span> +#include <vector> + +// Minimal pcapng reader, paired with writer.hpp: reads Enhanced Packet +// Blocks sequentially, skipping the Section Header Block, Interface +// Description Block, and any other block type transparently. +// +// Assumes little-endian block encoding (checked against the Section +// Header Block's byte-order magic, not just assumed) since that's what +// writer.hpp emits and what pcapng writers on this class of hardware +// (tcpdump, dumpcap) produce. A big-endian file is out of scope - this +// pairs with our own writer, not general pcapng interop. +namespace wireframe::pcapng { + +struct PacketRecord { + std::uint32_t interface_id; + std::uint64_t timestamp_us; + std::uint32_t original_len; + std::vector<unsigned char> data; +}; + +class Reader { +public: + explicit Reader(std::FILE* file) : file_(file) {} + + // Returns the next packet, or nullopt once the file is exhausted or + // a malformed/unsupported block is hit - treated as end of stream + // rather than a hard error, to keep this reader small. + std::optional<PacketRecord> next_packet() { + for (;;) { + std::array<std::uint8_t, 4> field{}; + if (std::fread(field.data(), 1, 4, file_) != 4) return std::nullopt; + std::uint32_t type = get_u32(field); + + if (std::fread(field.data(), 1, 4, file_) != 4) return std::nullopt; + std::uint32_t total_len = get_u32(field); + if (total_len < 12) return std::nullopt; + + std::size_t body_len = total_len - 12; + // total_len is an untrusted 32-bit value straight from the + // file; without a cap, a corrupted/hostile file can claim + // a multi-gigabyte block and OOM the process on the + // allocation below before a single byte is even read to + // check whether the file actually contains that much data + // (found by fuzzing fuzz_pcapng_reader.cpp - real crash, + // not theoretical). Bounded well above any block our own + // writer produces (packets capped at a 65535 snaplen; this + // reader is explicitly scoped to pair with that writer, + // not arbitrary pcapng interop). + if (body_len > kMaxBlockBodyLen) return std::nullopt; + std::vector<std::uint8_t> body(body_len); + if (body_len > 0 && std::fread(body.data(), 1, body_len, file_) != body_len) { + return std::nullopt; + } + + if (std::fread(field.data(), 1, 4, file_) != 4) return std::nullopt; + if (get_u32(field) != total_len) return std::nullopt; // corrupt trailer + + if (type == kBlockTypeShb) { + if (body_len < 4 || get_u32({body.data(), 4}) != kByteOrderMagic) { + return std::nullopt; // not little-endian, or malformed + } + continue; + } + if (type == kBlockTypeIdb) { + // LinkType is the first 2 bytes of the IDB body (see + // writer.hpp's write_interface_description). Only the + // first IDB is captured - correct for a file our own + // writer produced, which only ever writes one + // interface, matching this reader's documented scope. + if (!link_type_ && body_len >= 2) { + link_type_ = static_cast<std::uint16_t>(body[0] | (body[1] << 8)); + } + continue; + } + if (type != kBlockTypeEpb) continue; // anything else: skip + + if (body_len < 20) return std::nullopt; + + PacketRecord record; + record.interface_id = get_u32({body.data() + 0, 4}); + std::uint32_t ts_high = get_u32({body.data() + 4, 4}); + std::uint32_t ts_low = get_u32({body.data() + 8, 4}); + record.timestamp_us = (static_cast<std::uint64_t>(ts_high) << 32) | ts_low; + std::uint32_t caplen = get_u32({body.data() + 12, 4}); + record.original_len = get_u32({body.data() + 16, 4}); + + if (body_len < 20 + caplen) return std::nullopt; + record.data.assign(body.begin() + 20, body.begin() + 20 + caplen); + return record; + } + } + + // The interface's link type, learned from the Interface + // Description Block once next_packet() has read past it (which + // happens before it ever returns the first EPB, so this is + // populated by the time the first successful next_packet() call + // returns). nullopt if no IDB has been seen yet. + std::optional<std::uint16_t> link_type() const { return link_type_; } + +private: + static std::uint32_t get_u32(std::span<const std::uint8_t> b) { + return static_cast<std::uint32_t>(b[0]) | (static_cast<std::uint32_t>(b[1]) << 8) | + (static_cast<std::uint32_t>(b[2]) << 16) | (static_cast<std::uint32_t>(b[3]) << 24); + } + + static constexpr std::uint32_t kBlockTypeShb = 0x0A0D0D0A; + static constexpr std::uint32_t kBlockTypeIdb = 0x00000001; + static constexpr std::uint32_t kBlockTypeEpb = 0x00000006; + static constexpr std::uint32_t kByteOrderMagic = 0x1A2B3C4D; + static constexpr std::size_t kMaxBlockBodyLen = 1 << 20; // 1 MiB + + std::FILE* file_; + std::optional<std::uint16_t> link_type_; +}; + +} // namespace wireframe::pcapng diff --git a/include/wireframe/pcapng/writer.hpp b/include/wireframe/pcapng/writer.hpp new file mode 100644 index 0000000..18f6022 --- /dev/null +++ b/include/wireframe/pcapng/writer.hpp @@ -0,0 +1,94 @@ +#pragma once + +#include <algorithm> +#include <cstdint> +#include <cstdio> +#include <span> +#include <vector> + +// Minimal pcapng writer: one Section Header Block, one Interface +// Description Block, then an Enhanced Packet Block per captured packet. +// Per-block Options are skipped entirely - they're optional in the +// spec, and a block with none simply omits that section, so this stays +// a valid, Wireshark-readable file without needing to hand-encode TLVs. +// +// Multi-byte fields are written little-endian by hand (matching the +// 0x1A2B3C4D byte-order magic below) rather than via struct-casting, +// for the same alignment/UB reasons as the src/wireframe/net decoders. +namespace wireframe::pcapng { + +inline constexpr std::uint32_t kBlockTypeShb = 0x0A0D0D0A; +inline constexpr std::uint32_t kBlockTypeIdb = 0x00000001; +inline constexpr std::uint32_t kBlockTypeEpb = 0x00000006; +inline constexpr std::uint32_t kByteOrderMagic = 0x1A2B3C4D; +inline constexpr std::uint16_t kLinkTypeEthernet = 1; + +class Writer { +public: + explicit Writer(std::FILE* file) : file_(file) {} + + void write_section_header() { + std::uint8_t body[16]; + put_u32(body + 0, kByteOrderMagic); + put_u16(body + 4, 1); // major version + put_u16(body + 6, 0); // minor version + put_u64(body + 8, 0xFFFFFFFFFFFFFFFFULL); // section length: unknown + write_block(kBlockTypeShb, {body, sizeof(body)}); + } + + void write_interface_description(std::uint32_t snaplen, std::uint16_t link_type) { + std::uint8_t body[8]; + put_u16(body + 0, link_type); + put_u16(body + 2, 0); // reserved + put_u32(body + 4, snaplen); + write_block(kBlockTypeIdb, {body, sizeof(body)}); + } + + void write_packet(std::uint32_t interface_id, std::uint32_t ts_sec, std::uint32_t ts_usec, + std::span<const unsigned char> data, std::uint32_t original_len) { + std::uint64_t ts_us = static_cast<std::uint64_t>(ts_sec) * 1'000'000ULL + ts_usec; + std::uint32_t ts_high = static_cast<std::uint32_t>(ts_us >> 32); + std::uint32_t ts_low = static_cast<std::uint32_t>(ts_us & 0xFFFFFFFFULL); + + std::size_t padded_len = (data.size() + 3) & ~std::size_t(3); + std::vector<std::uint8_t> body(20 + padded_len, 0); // tail is padding, stays zero + put_u32(body.data() + 0, interface_id); + put_u32(body.data() + 4, ts_high); + put_u32(body.data() + 8, ts_low); + put_u32(body.data() + 12, static_cast<std::uint32_t>(data.size())); + put_u32(body.data() + 16, original_len); + std::copy(data.begin(), data.end(), body.begin() + 20); + + write_block(kBlockTypeEpb, body); + } + +private: + static void put_u16(std::uint8_t* p, std::uint16_t v) { + p[0] = static_cast<std::uint8_t>(v & 0xFF); + p[1] = static_cast<std::uint8_t>((v >> 8) & 0xFF); + } + + static void put_u32(std::uint8_t* p, std::uint32_t v) { + for (int i = 0; i < 4; ++i) p[i] = static_cast<std::uint8_t>((v >> (8 * i)) & 0xFF); + } + + static void put_u64(std::uint8_t* p, std::uint64_t v) { + for (int i = 0; i < 8; ++i) p[i] = static_cast<std::uint8_t>((v >> (8 * i)) & 0xFF); + } + + void write_block(std::uint32_t type, std::span<const std::uint8_t> body) { + std::uint32_t total_len = static_cast<std::uint32_t>(8 + body.size() + 4); + std::uint8_t type_buf[4]; + std::uint8_t len_buf[4]; + put_u32(type_buf, type); + put_u32(len_buf, total_len); + std::fwrite(type_buf, 1, 4, file_); + std::fwrite(len_buf, 1, 4, file_); + std::fwrite(body.data(), 1, body.size(), file_); + std::fwrite(len_buf, 1, 4, file_); + } + + std::FILE* file_; +}; + +} // namespace wireframe::pcapng diff --git a/include/wireframe/search.hpp b/include/wireframe/search.hpp new file mode 100644 index 0000000..4cb9203 --- /dev/null +++ b/include/wireframe/search.hpp @@ -0,0 +1,26 @@ +#pragma once + +#include <algorithm> +#include <cctype> +#include <string> + +// A display filter, distinct from -f's capture filter (wireframe/filter.hpp): +// -f decides what's captured - and, combined with -w, what's written to +// disk. This decides what's shown, without touching either. Same +// distinction Wireshark draws between a capture filter and a display +// filter, just without the display filter's expression language - a +// plain case-insensitive substring match over the packet's summary line +// is enough for "find the packets mentioning this host/port", which is +// the actual use case. +namespace wireframe { + +inline bool matches_search(const std::string& haystack, const std::string& needle) { + if (needle.empty()) return true; + auto it = std::search(haystack.begin(), haystack.end(), needle.begin(), needle.end(), + [](unsigned char a, unsigned char b) { + return std::tolower(a) == std::tolower(b); + }); + return it != haystack.end(); +} + +} // namespace wireframe diff --git a/include/wireframe/summarize.hpp b/include/wireframe/summarize.hpp new file mode 100644 index 0000000..59aa621 --- /dev/null +++ b/include/wireframe/summarize.hpp @@ -0,0 +1,248 @@ +#pragma once + +#include <cstdio> +#include <optional> +#include <span> +#include <string> +#include <vector> + +#include <pcap.h> + +#include "wireframe/l7/dissector.hpp" +#include "wireframe/l7/dns.hpp" +#include "wireframe/l7/http.hpp" +#include "wireframe/l7/tls.hpp" +#include "wireframe/net/ethernet.hpp" +#include "wireframe/net/ipv4.hpp" +#include "wireframe/net/ipv6.hpp" +#include "wireframe/net/tcp.hpp" +#include "wireframe/net/udp.hpp" + +// Packet -> human-readable summary. Shared by every frontend (plain +// CLI, TUI, GUI) so they can't drift apart on what a given packet +// decodes to - one source of truth, not three copies to keep in sync. +namespace wireframe { + +inline std::string mac_to_string(const net::MacAddress& mac) { + char buf[18]; + std::snprintf(buf, sizeof(buf), "%02x:%02x:%02x:%02x:%02x:%02x", mac.bytes[0], mac.bytes[1], + mac.bytes[2], mac.bytes[3], mac.bytes[4], mac.bytes[5]); + return buf; +} + +inline std::string ipv4_to_string(const net::Ipv4Address& ip) { + char buf[16]; + std::snprintf(buf, sizeof(buf), "%u.%u.%u.%u", ip.bytes[0], ip.bytes[1], ip.bytes[2], + ip.bytes[3]); + return buf; +} + +inline std::string tcp_flags_to_string(std::uint8_t flags) { + using namespace net; + std::string out; + if (flags & kTcpSyn) out += 'S'; + if (flags & kTcpAck) out += 'A'; + if (flags & kTcpFin) out += 'F'; + if (flags & kTcpRst) out += 'R'; + if (flags & kTcpPsh) out += 'P'; + if (flags & kTcpUrg) out += 'U'; + return out.empty() ? "-" : out; +} + +// Registered once. DNS (UDP) was the first L7 dissector, proving the +// interface (wireframe/l7/dissector.hpp) is enough to add a protocol +// without touching the L2-L4 decode path; HTTP (TCP) is the second, +// and the first to actually exercise L7Registry's TCP-payload path -- +// DNS alone never did, since it only ever runs over UDP port 53. TLS +// (also TCP, port 443) covers what HTTP increasingly can't: most web +// traffic today is encrypted, and SNI is the one piece of a TLS +// handshake still readable without decrypting anything. +inline const net::L7Registry& l7_registry() { + static const net::DnsDissector dns_dissector; + static const net::HttpDissector http_dissector; + static const net::TlsSniDissector tls_dissector; + static const net::L7Registry registry = [] { + net::L7Registry r; + r.add(&dns_dissector); + r.add(&http_dissector); + r.add(&tls_dissector); + return r; + }(); + return registry; +} + +// Tries the destination port first (the common case: a client talking +// to a well-known server port), then the source port (a server's +// reply, coming from that same well-known port). +inline std::optional<std::string> l7_summarize(std::span<const unsigned char> payload, + std::uint16_t src_port, std::uint16_t dst_port) { + if (auto summary = l7_registry().dissect(dst_port, payload)) return summary; + return l7_registry().dissect(src_port, payload); +} + +// IPv4 and IPv6 headers carry different fields (ttl vs. hop_limit, +// 4-byte vs. 16-byte addresses), but everything above the IP layer -- +// TCP/UDP decode plus the L7 lookup - is identical once normalized to +// this. Keeping that dispatch in one place means TCP/UDP/L7 formatting +// can't drift between the two IP versions. +struct IpInfo { + const char* label; // "IPv4" or "IPv6" + std::string src_str; + std::string dst_str; + std::uint8_t ttl_or_hop_limit; + std::uint8_t proto; + std::span<const unsigned char> payload; +}; + +inline std::string summarize_transport_and_above(const IpInfo& info) { + char ip_buf[160]; + std::snprintf(ip_buf, sizeof(ip_buf), " | %s %s -> %s ttl=%u proto=%u", info.label, + info.src_str.c_str(), info.dst_str.c_str(), info.ttl_or_hop_limit, info.proto); + std::string out = ip_buf; + + if (info.proto == net::kProtoTcp) { + if (auto tcp = net::parse_tcp(info.payload)) { + char tcp_buf[128]; + std::snprintf(tcp_buf, sizeof(tcp_buf), " | TCP %u -> %u [%s] seq=%u ack=%u win=%u", + tcp->header.src_port, tcp->header.dst_port, + tcp_flags_to_string(tcp->header.flags).c_str(), tcp->header.seq, + tcp->header.ack, tcp->header.window); + out += tcp_buf; + if (auto l7 = l7_summarize(tcp->payload, tcp->header.src_port, tcp->header.dst_port)) { + out += " | " + *l7; + } + } + } else if (info.proto == net::kProtoUdp) { + if (auto udp = net::parse_udp(info.payload)) { + char udp_buf[64]; + std::snprintf(udp_buf, sizeof(udp_buf), " | UDP %u -> %u len=%u", + udp->header.src_port, udp->header.dst_port, udp->header.length); + out += udp_buf; + if (auto l7 = l7_summarize(udp->payload, udp->header.src_port, udp->header.dst_port)) { + out += " | " + *l7; + } + } + } else if (info.proto == net::kNextHeaderIcmpv6) { + out += " | ICMPv6"; + } + return out; +} + +// `datalink` is the interface's actual pcap_datalink() type, not an +// assumption: tunnel/VPN interfaces (tailscale0, wireguard, plain +// tun/tap) hand libpcap raw IP with no link-layer header at all +// (DLT_RAW), unlike a real NIC or even `lo` (both DLT_EN10MB on +// Linux). Treating raw IP bytes as an Ethernet frame silently produces +// garbage MACs and ethertypes - verified by actually capturing on +// tailscale0 before this branch existed. +// +// IP version is read from the packet itself (the first nibble), not +// inferred from ethertype/datalink: DLT_RAW has no ethertype to key +// off at all, and even on Ethernet this keeps IPv4/IPv6 dispatch in +// one place. +inline std::string summarize_packet(std::span<const unsigned char> bytes, int datalink) { + std::span<const unsigned char> ip_bytes; + std::string out; + + if (datalink == DLT_RAW) { + out = "RAW"; + ip_bytes = bytes; + } else { + auto eth = net::parse_ethernet(bytes); + if (!eth) { + char buf[64]; + std::snprintf(buf, sizeof(buf), "[%zu bytes] truncated ethernet frame", bytes.size()); + return buf; + } + + out = "ETH " + mac_to_string(eth->header.src) + " -> " + mac_to_string(eth->header.dst); + char eth_buf[32]; + std::snprintf(eth_buf, sizeof(eth_buf), " ethertype=0x%04x", eth->header.ethertype); + out += eth_buf; + + if (eth->header.ethertype != net::kEthertypeIPv4 && + eth->header.ethertype != net::kEthertypeIPv6) { + return out; + } + ip_bytes = eth->payload; + } + + if (ip_bytes.empty()) { + out += " | IP (empty payload)"; + return out; + } + std::uint8_t version = static_cast<std::uint8_t>(ip_bytes[0] >> 4); + + if (version == 4) { + auto ip = net::parse_ipv4(ip_bytes); + if (!ip) { + out += " | IPv4 (truncated)"; + return out; + } + out += summarize_transport_and_above({"IPv4", ipv4_to_string(ip->header.src), + ipv4_to_string(ip->header.dst), ip->header.ttl, + ip->header.protocol, ip->payload}); + } else if (version == 6) { + auto ip6 = net::parse_ipv6(ip_bytes); + if (!ip6) { + out += " | IPv6 (truncated)"; + return out; + } + std::string src_str = net::ipv6_to_string(ip6->header.src); + std::string dst_str = net::ipv6_to_string(ip6->header.dst); + + // next_header may name an extension header (Hop-by-Hop, + // Routing, Dest Options, Fragment, AH) rather than the actual + // transport protocol; walk through those to find it. + auto walked = net::walk_ipv6_extension_headers(ip6->header.next_header, ip6->payload); + if (walked.stopped_at_esp) { + char buf[160]; + std::snprintf(buf, sizeof(buf), " | IPv6 %s -> %s ttl=%u proto=%u | ESP (encrypted)", + src_str.c_str(), dst_str.c_str(), ip6->header.hop_limit, + net::kNextHeaderEsp); + out += buf; + } else { + out += summarize_transport_and_above({"IPv6", src_str, dst_str, ip6->header.hop_limit, + walked.final_next_header, walked.payload}); + } + } else { + out += " | IP version " + std::to_string(version) + " (unsupported)"; + } + return out; +} + +// One formatted line per 16 bytes: offset, hex, ASCII gutter. Returned +// as lines rather than printed so both the CLI's -x output and a GUI +// details pane can use the same formatting. +inline std::vector<std::string> hex_dump_lines(std::span<const unsigned char> bytes) { + std::vector<std::string> lines; + for (std::size_t offset = 0; offset < bytes.size(); offset += 16) { + char offset_buf[32]; + std::snprintf(offset_buf, sizeof(offset_buf), "%06zx ", offset); + std::string line = offset_buf; + + std::size_t line_len = std::min<std::size_t>(16, bytes.size() - offset); + for (std::size_t i = 0; i < 16; ++i) { + if (i < line_len) { + char byte_buf[4]; + std::snprintf(byte_buf, sizeof(byte_buf), "%02x ", bytes[offset + i]); + line += byte_buf; + } else { + line += " "; + } + if (i == 7) line += ' '; + } + + line += " |"; + for (std::size_t i = 0; i < line_len; ++i) { + unsigned char c = bytes[offset + i]; + line += (c >= 0x20 && c < 0x7f) ? static_cast<char>(c) : '.'; + } + line += '|'; + + lines.push_back(std::move(line)); + } + return lines; +} + +} // namespace wireframe |