mirror of https://lore.kernel.org/lkml/
 help / color / mirror / Atom feed
From: Jia Jia <physicalmtea@gmail.com>
To: stefanha@redhat.com, sgarzare@redhat.com, netdev@vger.kernel.org,
	virtualization@lists.linux.dev, kvm@vger.kernel.org
Cc: mst@redhat.com, jasowangio@gmail.com, eperezma@redhat.com,
	xuanzhuo@linux.alibaba.com, davem@davemloft.net,
	edumazet@kernel.org, kuba@kernel.org, pabeni@redhat.com,
	horms@kernel.org, linux-kernel@vger.kernel.org,
	bpf@vger.kernel.org, Jia Jia <physicalmtea@gmail.com>
Subject: [PATCH net-next v2 4/5] vsock/virtio: coalesce RX write-space notifications in lock batches
Date: Sat, 10 Oct 2026 22:22:46 +0800	[thread overview]
Message-ID: <20261010142247.99223-5-physicalmtea@gmail.com> (raw)
In-Reply-To: <20261010142247.99223-1-physicalmtea@gmail.com>

Each received packet updates peer credit and calls sk_write_space() when
send space is available. A writer cannot use the newly advertised credit
until the socket lock is released, so repeated callbacks within one lock
batch cannot let it make progress sooner.

For the callback installed by sock_init_data(), record one pending
write-space notification and deliver it through the saved default callback
before release_sock(). If the callback has been replaced by batch finish,
invoke the replacement as well. This keeps the native writer wakeup from
being lost across a sockmap callback change while still notifying sockmap.

Save the initial sock_def_write_space() callback when the AF_VSOCK socket
is created because it is not visible to virtio_transport_common when built
as a module. The fast path uses READ_ONCE() and adds no callback lock.

Set the batch socket before processing its first packet so a batch ending
on that packet cannot lose the notification.

Packets outside the eligible STREAM/RW batch path retain per-packet
notification behavior. The 64-packet and 64K limits cap the packets whose
notifications can be coalesced.

Signed-off-by: Jia Jia <physicalmtea@gmail.com>
---
 include/linux/virtio_vsock.h            |  1 +
 include/net/af_vsock.h                  |  2 ++
 net/vmw_vsock/af_vsock.c                |  1 +
 net/vmw_vsock/virtio_transport_common.c | 42 ++++++++++++++++++++++---
 4 files changed, 41 insertions(+), 5 deletions(-)

diff --git a/include/linux/virtio_vsock.h b/include/linux/virtio_vsock.h
index d6528681e052..3a58120d9078 100644
--- a/include/linux/virtio_vsock.h
+++ b/include/linux/virtio_vsock.h
@@ -290,6 +290,7 @@ struct virtio_transport_rx_batch {
 	struct net *net;
 	struct sockaddr_vm src;
 	struct sockaddr_vm dst;
+	bool write_space_pending;
 };
 
 void virtio_transport_recv_pkt_batch(struct virtio_transport *t,
diff --git a/include/net/af_vsock.h b/include/net/af_vsock.h
index 5549298c1ec6..9d8ae62209ed 100644
--- a/include/net/af_vsock.h
+++ b/include/net/af_vsock.h
@@ -63,6 +63,8 @@ struct vsock_sock {
 	u32 peer_shutdown;
 	bool sent_request;
 	bool ignore_connecting_rst;
+	/* Initial callback, used to identify replacements. */
+	void (*default_write_space)(struct sock *sk);
 
 	/* Protected by lock_sock(sk) */
 	u64 buffer_size;
diff --git a/net/vmw_vsock/af_vsock.c b/net/vmw_vsock/af_vsock.c
index 44cee7451955..fe7f49da6c4b 100644
--- a/net/vmw_vsock/af_vsock.c
+++ b/net/vmw_vsock/af_vsock.c
@@ -958,6 +958,7 @@ static struct sock *__vsock_create(struct net *net,
 		sk->sk_type = type;
 
 	vsk = vsock_sk(sk);
+	vsk->default_write_space = sk->sk_write_space;
 	vsock_addr_init(&vsk->local_addr, VMADDR_CID_ANY, VMADDR_PORT_ANY);
 	vsock_addr_init(&vsk->remote_addr, VMADDR_CID_ANY, VMADDR_PORT_ANY);
 
diff --git a/net/vmw_vsock/virtio_transport_common.c b/net/vmw_vsock/virtio_transport_common.c
index 73e5c7dfe8b4..f0398ae2200d 100644
--- a/net/vmw_vsock/virtio_transport_common.c
+++ b/net/vmw_vsock/virtio_transport_common.c
@@ -1835,6 +1835,7 @@ struct virtio_transport_rx_pkt_ctx {
 	const struct sockaddr_vm *src;
 	const struct sockaddr_vm *dst;
 	bool *batchable;
+	struct virtio_transport_rx_batch *batch;
 };
 
 static bool
@@ -1863,6 +1864,7 @@ virtio_transport_recv_pkt_locked(struct virtio_transport *t,
 	const struct sockaddr_vm *src = ctx->src;
 	const struct sockaddr_vm *dst = ctx->dst;
 	struct vsock_sock *vsk = vsock_sk(sk);
+	void (*write_space)(struct sock *sk);
 	struct net *net = ctx->net;
 	bool space_available;
 
@@ -1885,8 +1887,17 @@ virtio_transport_recv_pkt_locked(struct virtio_transport *t,
 	if (vsk->local_addr.svm_cid != VMADDR_CID_ANY)
 		vsk->local_addr.svm_cid = dst->svm_cid;
 
-	if (space_available)
-		sk->sk_write_space(sk);
+	if (space_available) {
+		write_space = READ_ONCE(sk->sk_write_space);
+		if (ctx->batch &&
+		    write_space == vsk->default_write_space &&
+		    virtio_transport_recv_pkt_batchable(t, sk)) {
+			ctx->batch->write_space_pending = true;
+		} else {
+			/* Use the callback seen for this packet. */
+			write_space(sk);
+		}
+	}
 
 	switch (sk->sk_state) {
 	case TCP_LISTEN:
@@ -1972,16 +1983,28 @@ EXPORT_SYMBOL_GPL(virtio_transport_recv_pkt);
  */
 void virtio_transport_rx_batch_finish(struct virtio_transport_rx_batch *batch)
 {
+	bool write_space_pending = batch->write_space_pending;
+	void (*write_space)(struct sock *sk);
 	struct sock *sk = batch->sk;
+	struct vsock_sock *vsk;
 
 	batch->sk = NULL;
 	batch->pkts = 0;
 	batch->bytes = 0;
 	batch->net = NULL;
+	batch->write_space_pending = false;
 
 	if (!sk)
 		return;
 
+	if (write_space_pending) {
+		vsk = vsock_sk(sk);
+		vsk->default_write_space(sk);
+		write_space = READ_ONCE(sk->sk_write_space);
+		if (write_space != vsk->default_write_space)
+			write_space(sk);
+	}
+
 	release_sock(sk);
 	sock_put(sk);
 }
@@ -2027,6 +2050,7 @@ void virtio_transport_recv_pkt_batch(struct virtio_transport *t,
 				.src = &src,
 				.dst = &dst,
 				.batchable = &batchable,
+				.batch = batch,
 			};
 			free_pkt = virtio_transport_recv_pkt_locked(t, skb, sk, &ctx);
 
@@ -2063,11 +2087,15 @@ void virtio_transport_recv_pkt_batch(struct virtio_transport *t,
 	start_batch = virtio_transport_recv_pkt_batchable(t, sk);
 	read_unlock_bh(&sk->sk_callback_lock);
 
+	if (start_batch)
+		batch->sk = sk;
+
 	ctx = (struct virtio_transport_rx_pkt_ctx) {
 		.net = net,
 		.src = &src,
 		.dst = &dst,
 		.batchable = start_batch ? &batchable : NULL,
+		.batch = start_batch ? batch : NULL,
 	};
 	free_pkt = virtio_transport_recv_pkt_locked(t, skb, sk, &ctx);
 	if (start_batch && batchable) {
@@ -2075,12 +2103,16 @@ void virtio_transport_recv_pkt_batch(struct virtio_transport *t,
 		batch->net = net;
 		batch->src = src;
 		batch->dst = dst;
-		batch->sk = sk;
 		return;
 	}
 
-	release_sock(sk);
-	sock_put(sk);
+	if (start_batch) {
+		virtio_transport_rx_batch_finish(batch);
+	} else {
+		release_sock(sk);
+		sock_put(sk);
+	}
+
 	if (free_pkt)
 		kfree_skb(skb);
 }
-- 
2.34.1


  parent reply	other threads:[~2026-10-10 14:23 UTC|newest]

Thread overview: 9+ messages / expand[flat|nested]  mbox.gz  Atom feed  top
2026-10-10 14:22 [PATCH net-next v2 0/5] vsock/virtio: reduce RX per-packet socket overhead Jia Jia
2026-10-10 14:22 ` [PATCH net-next v2 1/5] vsock/virtio: split socket lookup from locked RX processing Jia Jia
2026-10-11 14:25   ` netdev-bot+sashiko
2026-10-10 14:22 ` [PATCH net-next v2 2/5] vsock/virtio: amortize RX socket locking for stream packets Jia Jia
2026-10-11 14:25   ` netdev-bot+sashiko
2026-10-10 14:22 ` [PATCH net-next v2 3/5] vsock/virtio: reuse same-flow socket lookup in RX batches Jia Jia
2026-10-10 14:22 ` Jia Jia [this message]
2026-10-10 14:22 ` [PATCH net-next v2 5/5] vsock/virtio: defer RX readable notifications until batch unlock Jia Jia
2026-10-11 14:25   ` netdev-bot+sashiko

Reply instructions:

You may reply publicly to this message via plain-text email
using any one of the following methods:

* Save the following mbox file, import it into your mail client,
  and reply-to-all from there: mbox

  Avoid top-posting and favor interleaved quoting:
  https://en.wikipedia.org/wiki/Posting_style#Interleaved_style

* Reply using the --to, --cc, and --in-reply-to
  switches of git-send-email(1):

  git send-email \
    --in-reply-to=20261010142247.99223-5-physicalmtea@gmail.com \
    --to=physicalmtea@gmail.com \
    --cc=bpf@vger.kernel.org \
    --cc=davem@davemloft.net \
    --cc=edumazet@kernel.org \
    --cc=eperezma@redhat.com \
    --cc=horms@kernel.org \
    --cc=jasowangio@gmail.com \
    --cc=kuba@kernel.org \
    --cc=kvm@vger.kernel.org \
    --cc=linux-kernel@vger.kernel.org \
    --cc=mst@redhat.com \
    --cc=netdev@vger.kernel.org \
    --cc=pabeni@redhat.com \
    --cc=sgarzare@redhat.com \
    --cc=stefanha@redhat.com \
    --cc=virtualization@lists.linux.dev \
    --cc=xuanzhuo@linux.alibaba.com \
    /path/to/YOUR_REPLY

  https://kernel.org/pub/software/scm/git/docs/git-send-email.html

* If your mail client supports setting the In-Reply-To header
  via mailto: links, try the mailto: link
Be sure your reply has a Subject: header at the top and a blank line before the message body.
This is a public inbox, see mirroring instructions
for how to clone and mirror all data and code used for this inbox

all inboxes | Powered by JetHome®