mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: 성병찬 <tjdqudcks0424@naver.com>
To: "Pablo Neira Ayuso" <pablo@netfilter.org>
Cc: <fw@strlen.de>, <phil@nwl.cc>, <netfilter-devel@vger.kernel.org>,
	<netdev@vger.kernel.org>, <linux-kernel@vger.kernel.org>
Subject: Re: [BUG] netfilter: IPv6 conntrack fragment reassembly truncates header offset
Date: Tue, 29 Sep 2026 18:14:02 +0900	[thread overview]
Message-ID: <2aafeb116ccba6fbd3fa457d53d17be2@cweb006.nm> (raw)
In-Reply-To: <art-XRsR09GIvc4D@chamomile>

[-- Attachment #1: Type: text/plain, Size: 1330 bytes --]


Thanks for the information. I confirmed that Jérémy Jean's existing
patch is identical to the fix I tested. Please treat my patch as
superseded by that patch.

I have attached my self-contained reproducer as repro-v2.c. It only
uses IPv6 loopback (::1) and does not send traffic outside the test
system.

Build:

  cc -std=c11 -O2 -Wall -Wextra -Werror \
    -o repro-v2 repro-v2.c

The test requires root or CAP_NET_RAW. I ran it in a QEMU guest booted
with:

  nf_conntrack.enable_hooks=1

Run:

  ./repro-v2 1

Expected result on the unmodified kernel:

  iteration=1 control_received=yes
  iteration=1 boundary_received=no
  RESULT: differential observed (control delivered, boundary not delivered)

The reproducer exits with status 0 when the bug is reproduced.

Expected result with Jérémy Jean's patch applied:

  iteration=1 control_received=yes
  iteration=1 boundary_received=yes
  RESULT: control passed but expected boundary drop was not observed

In this case it exits with status 2 because the bug is no longer
reproduced.

I reproduced the unmodified result twice and tested the fixed result
twice on Linux v7.2.8.

The SHA-256 of the attached source is:

  511dd29a2c41d0c734bb1369a6ab538a85f2498d92c137802c6775240ca17903

You may add:

Tested-by: 성병찬 <tjdqudcks0424@naver.com>

Regards,
Sung Byeongchan

[-- Attachment #2: repro-v2.c --]
[-- Type: application/octet-stream, Size: 15352 bytes --]

#define _GNU_SOURCE

#include <arpa/inet.h>
#include <errno.h>
#include <netinet/in.h>
#include <poll.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/socket.h>
#include <time.h>
#include <unistd.h>

/*
 * NFA-02 differential reproducer, version 2.
 *
 * Packets are injected as complete IPv6 packets through
 * AF_INET6/SOCK_RAW/IPV6_HDRINCL to ::1.  No link-layer header is supplied by
 * userspace.  Linux rawv6_send_hdrinc() sends these skbs through
 * NF_INET_LOCAL_OUT before loopback delivery, so netfilter reassembly happens
 * before the receive-side IPv6 extension-header parser initializes its skb
 * offsets.
 *
 * All offsets below are relative to IPv6 byte zero.
 *
 * control:
 *   IPv6 [0,40), Hop-by-Hop [40,248), Destination Options [248,256),
 *   Fragment [256,264), then 8 bytes of fragment data.
 *   The predecessor Next Header byte is exactly offset 248.
 *
 * boundary:
 *   IPv6 [0,40), Hop-by-Hop [40,256), Destination Options [256,264),
 *   Fragment [264,272), then 8 bytes of fragment data.
 *   The predecessor Next Header byte is exactly offset 256.
 *
 * Each UDP datagram is split into two fragments.  Fragment zero contains the
 * complete 8-byte UDP header and has M=1.  Fragment one has byte offset 8,
 * M=0, and contains the complete 8-byte UDP payload.
 */

enum {
	IPV6_LEN = 40,
	DEST_LEN = 8,
	FRAG_LEN = 8,
	UDP_LEN = 8,
	APP_PAYLOAD_LEN = 8,
	UDP_DATAGRAM_LEN = UDP_LEN + APP_PAYLOAD_LEN,
	CONTROL_HBH_LEN = 208,
	BOUNDARY_HBH_LEN = 216,
	CONTROL_PREDECESSOR_NHOFF = 248,
	BOUNDARY_PREDECESSOR_NHOFF = 256,
	CONTROL_FRAGMENT_OFFSET = 256,
	BOUNDARY_FRAGMENT_OFFSET = 264,
	MAX_PACKET_LEN = 512,
	DEST_PORT = 45555,
	MAX_ITERATIONS = 10,
	RECEIVE_TIMEOUT_MS = 1500,
	V6_NH_HOP = 0,
	V6_NH_UDP = 17,
	V6_NH_FRAGMENT = 44,
	V6_NH_DEST = 60,
	FRAGMENT_M_FLAG = 1,
	UNKNOWN_SKIP_OPTION = 0x1e,
	PADN_OPTION = 1,
};

static void write_be16(unsigned char *p, uint16_t value)
{
	p[0] = (unsigned char)(value >> 8);
	p[1] = (unsigned char)value;
}

static void write_be32(unsigned char *p, uint32_t value)
{
	p[0] = (unsigned char)(value >> 24);
	p[1] = (unsigned char)(value >> 16);
	p[2] = (unsigned char)(value >> 8);
	p[3] = (unsigned char)value;
}

static uint16_t read_be16(const unsigned char *p)
{
	return (uint16_t)(((uint16_t)p[0] << 8) | p[1]);
}

static uint32_t checksum_add(uint32_t sum, const unsigned char *data,
			     size_t len)
{
	while (len >= 2) {
		sum += ((uint32_t)data[0] << 8) | data[1];
		data += 2;
		len -= 2;
	}
	if (len != 0)
		sum += (uint32_t)data[0] << 8;
	return sum;
}

static uint16_t checksum_finish(uint32_t sum)
{
	uint16_t result;

	while (sum >> 16)
		sum = (sum & 0xffffU) + (sum >> 16);
	result = (uint16_t)~sum;
	return result == 0 ? 0xffffU : result;
}

static uint16_t udp6_checksum(const unsigned char *udp,
			      const unsigned char *payload)
{
	unsigned char loopback[16] = { 0 };
	uint32_t sum = 0;

	loopback[15] = 1;
	sum = checksum_add(sum, loopback, sizeof(loopback));
	sum = checksum_add(sum, loopback, sizeof(loopback));
	/* 32-bit UDP length contributes 0x0000 and 0x0010. */
	sum += UDP_DATAGRAM_LEN;
	/* Three zero bytes and Next Header 17 contribute 0x0000 and 0x0011. */
	sum += V6_NH_UDP;
	sum = checksum_add(sum, udp, UDP_LEN);
	sum = checksum_add(sum, payload, APP_PAYLOAD_LEN);
	return checksum_finish(sum);
}

static int build_udp(unsigned char udp[UDP_LEN], uint16_t source_port,
		     const unsigned char payload[APP_PAYLOAD_LEN])
{
	uint16_t checksum;

	memset(udp, 0, UDP_LEN);
	write_be16(udp, source_port);
	write_be16(udp + 2, DEST_PORT);
	write_be16(udp + 4, UDP_DATAGRAM_LEN);
	checksum = udp6_checksum(udp, payload);
	write_be16(udp + 6, checksum);
	return 0;
}

static int validate_udp(const unsigned char udp[UDP_LEN], uint16_t source_port,
			const unsigned char payload[APP_PAYLOAD_LEN])
{
	unsigned char check[UDP_LEN];
	uint16_t expected;
	uint16_t actual;

	if (read_be16(udp) != source_port ||
	    read_be16(udp + 2) != DEST_PORT ||
	    read_be16(udp + 4) != UDP_DATAGRAM_LEN)
		return -1;
	actual = read_be16(udp + 6);
	if (actual == 0)
		return -1;
	memcpy(check, udp, sizeof(check));
	check[6] = 0;
	check[7] = 0;
	expected = udp6_checksum(check, payload);
	return actual == expected ? 0 : -1;
}

static int fill_hop_by_hop(unsigned char *hbh, size_t hbh_len)
{
	size_t unknown_total;
	size_t unknown_data;
	size_t pad_offset;

	if (hbh_len != CONTROL_HBH_LEN && hbh_len != BOUNDARY_HBH_LEN)
		return -1;

	memset(hbh, 0, hbh_len);
	hbh[0] = V6_NH_DEST;
	hbh[1] = (unsigned char)(hbh_len / 8 - 1);

	/*
	 * One unknown option with action bits 00 (skip) plus one six-byte PadN.
	 * control: 200-byte unknown TLV + 6-byte PadN = 206 option bytes.
	 * boundary: 208-byte unknown TLV + 6-byte PadN = 214 option bytes.
	 */
	unknown_total = hbh_len - 8;
	unknown_data = unknown_total - 2;
	if (unknown_data > UINT8_MAX)
		return -1;
	hbh[2] = UNKNOWN_SKIP_OPTION;
	hbh[3] = (unsigned char)unknown_data;

	pad_offset = 2 + unknown_total;
	hbh[pad_offset] = PADN_OPTION;
	hbh[pad_offset + 1] = 4;
	return 0;
}

static void fill_destination_options(unsigned char dest[DEST_LEN])
{
	memset(dest, 0, DEST_LEN);
	dest[0] = V6_NH_FRAGMENT;
	dest[1] = 0;
	dest[2] = PADN_OPTION;
	dest[3] = 4;
}

static int build_fragment(unsigned char packet[MAX_PACKET_LEN],
			  size_t *packet_len, size_t hbh_len,
			  uint32_t identification, uint16_t fragment_field,
			  const unsigned char fragment_data[8])
{
	unsigned char *ip6 = packet;
	unsigned char *hbh = ip6 + IPV6_LEN;
	unsigned char *dest = hbh + hbh_len;
	unsigned char *frag = dest + DEST_LEN;
	unsigned char *data = frag + FRAG_LEN;
	size_t payload_len = hbh_len + DEST_LEN + FRAG_LEN + 8;
	size_t total_len = IPV6_LEN + payload_len;

	if (total_len > MAX_PACKET_LEN || payload_len > UINT16_MAX)
		return -1;
	memset(packet, 0, MAX_PACKET_LEN);

	/* IPv6: version 6, explicit payload length, HBH, hop limit 64, ::1 -> ::1. */
	ip6[0] = 0x60;
	write_be16(ip6 + 4, (uint16_t)payload_len);
	ip6[6] = V6_NH_HOP;
	ip6[7] = 64;
	ip6[8 + 15] = 1;
	ip6[24 + 15] = 1;

	if (fill_hop_by_hop(hbh, hbh_len) < 0)
		return -1;
	fill_destination_options(dest);

	frag[0] = V6_NH_UDP;
	frag[1] = 0;
	write_be16(frag + 2, fragment_field);
	write_be32(frag + 4, identification);
	memcpy(data, fragment_data, 8);

	*packet_len = total_len;
	return 0;
}

static int validate_fragment(const unsigned char *packet, size_t packet_len,
			     size_t hbh_len, uint16_t fragment_field,
			     const unsigned char fragment_data[8])
{
	const unsigned char *hbh = packet + IPV6_LEN;
	const unsigned char *dest = hbh + hbh_len;
	const unsigned char *frag = dest + DEST_LEN;
	const unsigned char *data = frag + FRAG_LEN;
	size_t expected_len = IPV6_LEN + hbh_len + DEST_LEN + FRAG_LEN + 8;

	if (packet_len != expected_len || packet[0] != 0x60 ||
	    read_be16(packet + 4) != packet_len - IPV6_LEN ||
	    packet[6] != V6_NH_HOP || packet[7] != 64)
		return -1;
	if (packet[8 + 15] != 1 || packet[24 + 15] != 1)
		return -1;
	if (hbh[0] != V6_NH_DEST || hbh[1] != hbh_len / 8 - 1)
		return -1;
	if (dest[0] != V6_NH_FRAGMENT || dest[1] != 0 ||
	    dest[2] != PADN_OPTION || dest[3] != 4)
		return -1;
	if (frag[0] != V6_NH_UDP || frag[1] != 0 ||
	    read_be16(frag + 2) != fragment_field)
		return -1;
	return memcmp(data, fragment_data, 8) == 0 ? 0 : -1;
}

static int send_one_packet(int fd, const unsigned char *packet,
			   size_t packet_len)
{
	struct sockaddr_in6 destination;
	ssize_t sent;

	memset(&destination, 0, sizeof(destination));
	destination.sin6_family = AF_INET6;
	destination.sin6_addr = in6addr_loopback;

	sent = sendto(fd, packet, packet_len, 0,
		      (const struct sockaddr *)&destination, sizeof(destination));
	if (sent < 0) {
		perror("sendto(AF_INET6/SOCK_RAW/IPV6_HDRINCL)");
		return -1;
	}
	if ((size_t)sent != packet_len) {
		fprintf(stderr, "short raw IPv6 send: %zd of %zu bytes\n",
			sent, packet_len);
		return -1;
	}
	return 0;
}

static int send_case(int fd, const char *name, size_t hbh_len,
		     uint32_t identification, uint16_t source_port,
		     const unsigned char payload[APP_PAYLOAD_LEN])
{
	unsigned char first[MAX_PACKET_LEN];
	unsigned char second[MAX_PACKET_LEN];
	unsigned char udp[UDP_LEN];
	size_t first_len;
	size_t second_len;

	if (build_udp(udp, source_port, payload) < 0 ||
	    validate_udp(udp, source_port, payload) < 0) {
		fprintf(stderr, "%s: UDP header/checksum self-check failed\n", name);
		return -1;
	}
	if (build_fragment(first, &first_len, hbh_len, identification,
			   FRAGMENT_M_FLAG, udp) < 0 ||
	    validate_fragment(first, first_len, hbh_len, FRAGMENT_M_FLAG,
			      udp) < 0) {
		fprintf(stderr, "%s: first-fragment self-check failed\n", name);
		return -1;
	}
	/* Wire value 8 means byte offset 8 because the low three bits are flags. */
	if (build_fragment(second, &second_len, hbh_len, identification,
			   8, payload) < 0 ||
	    validate_fragment(second, second_len, hbh_len, 8, payload) < 0) {
		fprintf(stderr, "%s: second-fragment self-check failed\n", name);
		return -1;
	}

	printf("%s: first_len=%zu second_len=%zu payload_len=%u "
	       "udp_checksum=0x%04x fragment_fields=0x0001,0x0008\n",
	       name, first_len, second_len, read_be16(first + 4),
	       read_be16(udp + 6));
	if (send_one_packet(fd, first, first_len) < 0)
		return -1;
	if (send_one_packet(fd, second, second_len) < 0)
		return -1;
	return 0;
}

static int monotonic_ms(uint64_t *value)
{
	struct timespec ts;

	if (clock_gettime(CLOCK_MONOTONIC, &ts) < 0) {
		perror("clock_gettime");
		return -1;
	}
	*value = (uint64_t)ts.tv_sec * 1000U +
		 (uint64_t)ts.tv_nsec / 1000000U;
	return 0;
}

static int drain_receiver(int fd)
{
	unsigned char buffer[256];

	for (;;) {
		ssize_t received = recv(fd, buffer, sizeof(buffer), 0);

		if (received >= 0)
			continue;
		if (errno == EAGAIN || errno == EWOULDBLOCK)
			return 0;
		perror("recv(drain)");
		return -1;
	}
}

/* Return 1 for a matching datagram, 0 for timeout, and -1 for syscall error. */
static int wait_for_payload(int fd,
			    const unsigned char expected[APP_PAYLOAD_LEN])
{
	uint64_t start;
	uint64_t now;
	struct pollfd pfd = { .fd = fd, .events = POLLIN };

	if (monotonic_ms(&start) < 0)
		return -1;
	for (;;) {
		unsigned char buffer[256];
		ssize_t received;
		int remaining;
		int ready;

		if (monotonic_ms(&now) < 0)
			return -1;
		if (now - start >= RECEIVE_TIMEOUT_MS)
			return 0;
		remaining = RECEIVE_TIMEOUT_MS - (int)(now - start);
		pfd.revents = 0;
		ready = poll(&pfd, 1, remaining);
		if (ready < 0) {
			perror("poll");
			return -1;
		}
		if (ready == 0)
			return 0;
		if ((pfd.revents & POLLIN) == 0) {
			fprintf(stderr, "unexpected poll revents: 0x%x\n", pfd.revents);
			return -1;
		}

		received = recv(fd, buffer, sizeof(buffer), 0);
		if (received < 0) {
			perror("recv");
			return -1;
		}
		if (received == APP_PAYLOAD_LEN &&
		    memcmp(buffer, expected, APP_PAYLOAD_LEN) == 0)
			return 1;
		fprintf(stderr, "ignored unrelated UDP datagram of %zd bytes\n",
			received);
	}
}

static int parse_iterations(const char *text, int *iterations)
{
	char *end = NULL;
	long value;

	errno = 0;
	value = strtol(text, &end, 10);
	if (errno != 0 || end == text || *end != '\0' ||
	    value < 1 || value > MAX_ITERATIONS)
		return -1;
	*iterations = (int)value;
	return 0;
}

static int close_checked(int fd, const char *name)
{
	if (close(fd) < 0) {
		fprintf(stderr, "close(%s): %s\n", name, strerror(errno));
		return -1;
	}
	return 0;
}

int main(int argc, char **argv)
{
	static const unsigned char control_payload[APP_PAYLOAD_LEN] =
		{ 'N', 'F', 'A', '2', 'V', '2', 'C', '!' };
	static const unsigned char boundary_payload[APP_PAYLOAD_LEN] =
		{ 'N', 'F', 'A', '2', 'V', '2', 'B', '!' };
	static const char loopback_device[] = "lo";
	struct sockaddr_in6 receive_address;
	int hdrincl = 1;
	int iterations = 1;
	int receiver = -1;
	int raw = -1;
	int result = EXIT_SUCCESS;
	int i;

	if (argc > 2 || (argc == 2 && parse_iterations(argv[1], &iterations) < 0)) {
		fprintf(stderr, "usage: %s [iterations:1-%d]\n", argv[0],
			MAX_ITERATIONS);
		return EXIT_FAILURE;
	}

	receiver = socket(AF_INET6, SOCK_DGRAM | SOCK_CLOEXEC | SOCK_NONBLOCK,
			  IPPROTO_UDP);
	if (receiver < 0) {
		perror("socket(AF_INET6, SOCK_DGRAM)");
		return EXIT_FAILURE;
	}
	memset(&receive_address, 0, sizeof(receive_address));
	receive_address.sin6_family = AF_INET6;
	receive_address.sin6_port = htons(DEST_PORT);
	receive_address.sin6_addr = in6addr_loopback;
	if (bind(receiver, (const struct sockaddr *)&receive_address,
		 sizeof(receive_address)) < 0) {
		perror("bind(UDP ::1)");
		result = EXIT_FAILURE;
		goto out;
	}

	raw = socket(AF_INET6, SOCK_RAW | SOCK_CLOEXEC, IPPROTO_RAW);
	if (raw < 0) {
		perror("socket(AF_INET6, SOCK_RAW, IPPROTO_RAW); CAP_NET_RAW required");
		result = EXIT_FAILURE;
		goto out;
	}
	if (setsockopt(raw, IPPROTO_IPV6, IPV6_HDRINCL, &hdrincl,
		       sizeof(hdrincl)) < 0) {
		perror("setsockopt(IPV6_HDRINCL)");
		result = EXIT_FAILURE;
		goto out;
	}
	if (setsockopt(raw, SOL_SOCKET, SO_BINDTODEVICE, loopback_device,
		       sizeof(loopback_device)) < 0) {
		perror("setsockopt(SO_BINDTODEVICE=lo)");
		result = EXIT_FAILURE;
		goto out;
	}

	printf("injection=AF_INET6/SOCK_RAW/IPPROTO_RAW/IPV6_HDRINCL "
	       "device=lo destination=[::1]:%d iterations=%d\n",
	       DEST_PORT, iterations);
	printf("control: hbh=208 predecessor_nhoff=%d fragment_offset=%d\n",
	       CONTROL_PREDECESSOR_NHOFF, CONTROL_FRAGMENT_OFFSET);
	printf("boundary: hbh=216 predecessor_nhoff=%d fragment_offset=%d "
	       "stored_u8=0\n",
	       BOUNDARY_PREDECESSOR_NHOFF, BOUNDARY_FRAGMENT_OFFSET);

	for (i = 0; i < iterations; i++) {
		uint32_t base_id = 0x4e460400U + (uint32_t)i * 2U;
		int observed;

		if (drain_receiver(receiver) < 0) {
			result = EXIT_FAILURE;
			goto out;
		}
		if (send_case(raw, "control", CONTROL_HBH_LEN, base_id,
			      (uint16_t)(40000 + i * 2), control_payload) < 0) {
			result = EXIT_FAILURE;
			goto out;
		}
		observed = wait_for_payload(receiver, control_payload);
		if (observed < 0) {
			result = EXIT_FAILURE;
			goto out;
		}
		printf("iteration=%d control_received=%s\n", i + 1,
		       observed ? "yes" : "no");
		if (!observed) {
			printf("iteration=%d boundary_received=not-tested\n", i + 1);
			printf("RESULT: control failed; boundary was not sent and is not interpretable\n");
			result = 3;
			goto out;
		}

		if (drain_receiver(receiver) < 0) {
			result = EXIT_FAILURE;
			goto out;
		}
		if (send_case(raw, "boundary", BOUNDARY_HBH_LEN, base_id + 1,
			      (uint16_t)(40001 + i * 2), boundary_payload) < 0) {
			result = EXIT_FAILURE;
			goto out;
		}
		observed = wait_for_payload(receiver, boundary_payload);
		if (observed < 0) {
			result = EXIT_FAILURE;
			goto out;
		}
		printf("iteration=%d boundary_received=%s\n", i + 1,
		       observed ? "yes" : "no");
		if (observed)
			result = 2;
	}

	if (result == EXIT_SUCCESS)
		printf("RESULT: differential observed (control delivered, boundary not delivered)\n");
	else if (result == 2)
		printf("RESULT: control passed but expected boundary drop was not observed\n");

out:
	if (raw >= 0 && close_checked(raw, "raw IPv6") < 0)
		result = EXIT_FAILURE;
	if (receiver >= 0 && close_checked(receiver, "UDP receiver") < 0)
		result = EXIT_FAILURE;
	return result;
}

      reply	other threads:[~2026-09-29  9:34 UTC|newest]

Thread overview: 2+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
     [not found] <31da96413a406da2116f44df2db84f@cweb009.nm>
2026-09-29  9:01 ` Pablo Neira Ayuso
2026-09-29  9:14   ` 성병찬 [this message]

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=2aafeb116ccba6fbd3fa457d53d17be2@cweb006.nm \
    --to=tjdqudcks0424@naver.com \
    --cc=fw@strlen.de \
    --cc=linux-kernel@vger.kernel.org \
    --cc=netdev@vger.kernel.org \
    --cc=netfilter-devel@vger.kernel.org \
    --cc=pablo@netfilter.org \
    --cc=phil@nwl.cc \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®