1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
|
// AF_PACKET + PACKET_RX_RING: capture without libpcap's internal buffer
// copy, using a memory-mapped ring buffer shared directly with the
// kernel. This demonstrates PLAN.md's originally-listed alternative
// capture backend ("libpcap, or raw AF_PACKET with an mmap'd ring
// buffer to skip libpcap's copies") as a focused, standalone artifact.
//
// Deliberately NOT wired into CaptureSession/the main pipeline: doing
// that would mean reimplementing filtering (SO_ATTACH_FILTER instead
// of pcap_setfilter), kernel stats (raw sockopts instead of
// pcap_stats), and datalink detection (ARPHRD_* mapping instead of
// pcap_datalink) at every one of CaptureSession's already-tested call
// sites - real risk to working, verified functionality for a
// copy-avoidance benefit modern libpcap on Linux already gets much of
// internally. What this file actually explores - a std::span reading
// packet bytes directly out of kernel-shared mapped memory, with zero
// copies between the NIC and this process at all - is a more direct
// exploration of this project's actual point (the C++ memory model)
// than anything routed through libpcap's own abstraction, and doesn't
// need to touch the rest of the tool to demonstrate that.
//
// Linux-only: AF_PACKET is a Linux-specific socket family, unlike the
// portable libpcap path the rest of this project uses.
#include <linux/if_ether.h>
#include <linux/if_packet.h>
#include <net/if.h>
#include <poll.h>
#include <sys/mman.h>
#include <sys/socket.h>
#include <unistd.h>
#include <atomic>
#include <cerrno>
#include <csignal>
#include <cstdio>
#include <cstring>
#include <span>
#include "packeteer/privileges.hpp"
#include "packeteer/summarize.hpp"
namespace {
// TPACKET_V2: a simpler one-frame-per-slot layout than TPACKET_V3's
// block-batching, still genuinely mmap'd and zero-copy. The right
// complexity level for demonstrating the technique clearly, not for
// maximizing throughput.
constexpr std::size_t kFrameSize = 2048; // room for a max-size Ethernet frame + header + padding
constexpr std::size_t kFramesPerBlock = 2;
constexpr std::size_t kBlockSize = kFrameSize * kFramesPerBlock; // must be a page-size multiple
constexpr std::size_t kBlockCount = 64;
constexpr std::size_t kFrameCount = kFramesPerBlock * kBlockCount;
std::atomic<bool> g_stop{false};
void handle_stop_signal(int) { g_stop.store(true); }
} // namespace
int main(int argc, char** argv) {
if (argc < 2) {
std::fprintf(stderr, "usage: %s <interface>\n", argv[0]);
return 1;
}
const char* ifname = argv[1];
long page_size = sysconf(_SC_PAGESIZE);
if (page_size <= 0 || kBlockSize % static_cast<std::size_t>(page_size) != 0) {
std::fprintf(stderr,
"kBlockSize (%zu) isn't a multiple of this system's page size (%ld) - "
"TPACKET_V2 requires it to be\n",
kBlockSize, page_size);
return 1;
}
int sock = socket(AF_PACKET, SOCK_RAW, htons(ETH_P_ALL));
if (sock == -1) {
std::fprintf(stderr, "socket(AF_PACKET) failed: %s\n", std::strerror(errno));
return 1;
}
int version = TPACKET_V2;
if (setsockopt(sock, SOL_PACKET, PACKET_VERSION, &version, sizeof(version)) == -1) {
std::fprintf(stderr, "setsockopt(PACKET_VERSION) failed: %s\n", std::strerror(errno));
close(sock);
return 1;
}
tpacket_req req{};
req.tp_block_size = kBlockSize;
req.tp_block_nr = kBlockCount;
req.tp_frame_size = kFrameSize;
req.tp_frame_nr = kFrameCount;
if (setsockopt(sock, SOL_PACKET, PACKET_RX_RING, &req, sizeof(req)) == -1) {
std::fprintf(stderr, "setsockopt(PACKET_RX_RING) failed: %s\n", std::strerror(errno));
close(sock);
return 1;
}
std::size_t ring_size = req.tp_block_size * req.tp_block_nr;
// This mapping *is* the ring buffer: the kernel writes captured
// frames into these same pages, and every packet read below is a
// pointer straight into this mapping - no read()/recv() call, no
// buffer of our own, no copy of the packet data at any point
// between the NIC and summarize_packet() seeing it.
void* ring = mmap(nullptr, ring_size, PROT_READ | PROT_WRITE, MAP_SHARED, sock, 0);
if (ring == MAP_FAILED) {
std::fprintf(stderr, "mmap failed: %s\n", std::strerror(errno));
close(sock);
return 1;
}
unsigned int ifindex = if_nametoindex(ifname);
if (ifindex == 0) {
std::fprintf(stderr, "if_nametoindex(%s) failed: %s\n", ifname, std::strerror(errno));
munmap(ring, ring_size);
close(sock);
return 1;
}
sockaddr_ll addr{};
addr.sll_family = AF_PACKET;
addr.sll_protocol = htons(ETH_P_ALL);
addr.sll_ifindex = static_cast<int>(ifindex);
if (bind(sock, reinterpret_cast<sockaddr*>(&addr), sizeof(addr)) == -1) {
std::fprintf(stderr, "bind failed: %s\n", std::strerror(errno));
munmap(ring, ring_size);
close(sock);
return 1;
}
// Everything CAP_NET_RAW was needed for is done: socket created,
// ring mapped, bound to the interface. Same drop-after-open
// principle as CaptureSession (packeteer/privileges.hpp).
if (auto err = packeteer::drop_privileges_if_root()) {
std::fprintf(stderr, "failed to drop privileges: %s\n", err->c_str());
munmap(ring, ring_size);
close(sock);
return 1;
}
std::signal(SIGINT, handle_stop_signal);
std::signal(SIGTERM, handle_stop_signal);
std::printf(
"capturing on %s via AF_PACKET/mmap ring buffer (%zu frames x %zu bytes, ctrl-c to "
"stop)\n",
ifname, kFrameCount, kFrameSize);
std::size_t frame_index = 0;
while (!g_stop.load()) {
// The status byte at the start of each slot is how the kernel
// and this process hand a frame back and forth without ever
// copying the packet itself: TP_STATUS_KERNEL means "not
// written yet, keep waiting"; the kernel flips it once a
// packet lands, and only then are these bytes safe to read.
auto* header = reinterpret_cast<tpacket2_hdr*>(static_cast<unsigned char*>(ring) +
frame_index * kFrameSize);
if (header->tp_status == TP_STATUS_KERNEL) {
pollfd pfd{};
pfd.fd = sock;
pfd.events = POLLIN;
poll(&pfd, 1, /*timeout_ms=*/200); // bounded so g_stop is still checked promptly
continue;
}
// tp_mac is the offset from the start of this header to the
// start of the actual frame data - still inside the same
// mmap'd page, never copied elsewhere.
const auto* packet_start =
reinterpret_cast<const unsigned char*>(header) + header->tp_mac;
std::span<const unsigned char> bytes(packet_start, header->tp_snaplen);
std::printf("%s\n", packeteer::summarize_packet(bytes, DLT_EN10MB).c_str());
std::fflush(stdout);
// Hand the slot back to the kernel so it can reuse it for a
// future packet - the mirror image of the status flip above.
header->tp_status = TP_STATUS_KERNEL;
frame_index = (frame_index + 1) % kFrameCount;
}
munmap(ring, ring_size);
close(sock);
return 0;
}
|