2017-11-01 10:09:13 -04:00
|
|
|
/* SPDX-License-Identifier: GPL-2.0 WITH Linux-syscall-note */
|
2012-10-09 04:47:14 -04:00
|
|
|
/*
|
|
|
|
* This file is subject to the terms and conditions of the GNU General Public
|
|
|
|
* License. See the file "COPYING" in the main directory of this archive
|
|
|
|
* for more details.
|
|
|
|
*
|
|
|
|
* Copyright (C) 1997, 1999, 2000, 2001 Ralf Baechle
|
|
|
|
* Copyright (C) 2000, 2001 Silicon Graphics, Inc.
|
|
|
|
*/
|
|
|
|
#ifndef _UAPI_ASM_SOCKET_H
|
|
|
|
#define _UAPI_ASM_SOCKET_H
|
|
|
|
|
2019-03-11 11:38:17 -04:00
|
|
|
#include <linux/posix_types.h>
|
2012-10-09 04:47:14 -04:00
|
|
|
#include <asm/sockios.h>
|
|
|
|
|
|
|
|
/*
|
|
|
|
* For setsockopt(2)
|
|
|
|
*
|
|
|
|
* This defines are ABI conformant as far as Linux supports these ...
|
|
|
|
*/
|
|
|
|
#define SOL_SOCKET 0xffff
|
|
|
|
|
|
|
|
#define SO_DEBUG 0x0001 /* Record debugging information. */
|
|
|
|
#define SO_REUSEADDR 0x0004 /* Allow reuse of local addresses. */
|
|
|
|
#define SO_KEEPALIVE 0x0008 /* Keep connections alive and send
|
|
|
|
SIGPIPE when they die. */
|
|
|
|
#define SO_DONTROUTE 0x0010 /* Don't do local routing. */
|
|
|
|
#define SO_BROADCAST 0x0020 /* Allow transmission of
|
2013-01-22 06:59:30 -05:00
|
|
|
broadcast messages. */
|
2012-10-09 04:47:14 -04:00
|
|
|
#define SO_LINGER 0x0080 /* Block on close of a reliable
|
|
|
|
socket to transmit pending data. */
|
|
|
|
#define SO_OOBINLINE 0x0100 /* Receive out-of-band data in-band. */
|
2013-01-22 04:49:50 -05:00
|
|
|
#define SO_REUSEPORT 0x0200 /* Allow local address and port reuse. */
|
2012-10-09 04:47:14 -04:00
|
|
|
|
|
|
|
#define SO_TYPE 0x1008 /* Compatible name for SO_STYLE. */
|
2013-01-22 06:59:30 -05:00
|
|
|
#define SO_STYLE SO_TYPE /* Synonym */
|
2012-10-09 04:47:14 -04:00
|
|
|
#define SO_ERROR 0x1007 /* get error status and clear */
|
|
|
|
#define SO_SNDBUF 0x1001 /* Send buffer size. */
|
|
|
|
#define SO_RCVBUF 0x1002 /* Receive buffer. */
|
|
|
|
#define SO_SNDLOWAT 0x1003 /* send low-water mark */
|
|
|
|
#define SO_RCVLOWAT 0x1004 /* receive low-water mark */
|
2019-02-02 10:34:53 -05:00
|
|
|
#define SO_SNDTIMEO_OLD 0x1005 /* send timeout */
|
|
|
|
#define SO_RCVTIMEO_OLD 0x1006 /* receive timeout */
|
2012-10-09 04:47:14 -04:00
|
|
|
#define SO_ACCEPTCONN 0x1009
|
|
|
|
#define SO_PROTOCOL 0x1028 /* protocol type */
|
|
|
|
#define SO_DOMAIN 0x1029 /* domain/socket family */
|
|
|
|
|
|
|
|
/* linux-specific, might as well be the same as on i386 */
|
|
|
|
#define SO_NO_CHECK 11
|
|
|
|
#define SO_PRIORITY 12
|
|
|
|
#define SO_BSDCOMPAT 14
|
|
|
|
|
|
|
|
#define SO_PASSCRED 17
|
|
|
|
#define SO_PEERCRED 18
|
|
|
|
|
|
|
|
/* Security levels - as per NRL IPv6 - don't actually do anything */
|
|
|
|
#define SO_SECURITY_AUTHENTICATION 22
|
|
|
|
#define SO_SECURITY_ENCRYPTION_TRANSPORT 23
|
|
|
|
#define SO_SECURITY_ENCRYPTION_NETWORK 24
|
|
|
|
|
|
|
|
#define SO_BINDTODEVICE 25
|
|
|
|
|
|
|
|
/* Socket filtering */
|
2013-01-22 06:59:30 -05:00
|
|
|
#define SO_ATTACH_FILTER 26
|
|
|
|
#define SO_DETACH_FILTER 27
|
sk-filter: Add ability to get socket filter program (v2)
The SO_ATTACH_FILTER option is set only. I propose to add the get
ability by using SO_ATTACH_FILTER in getsockopt. To be less
irritating to eyes the SO_GET_FILTER alias to it is declared. This
ability is required by checkpoint-restore project to be able to
save full state of a socket.
There are two issues with getting filter back.
First, kernel modifies the sock_filter->code on filter load, thus in
order to return the filter element back to user we have to decode it
into user-visible constants. Fortunately the modification in question
is interconvertible.
Second, the BPF_S_ALU_DIV_K code modifies the command argument k to
speed up the run-time division by doing kernel_k = reciprocal(user_k).
Bad news is that different user_k may result in same kernel_k, so we
can't get the original user_k back. Good news is that we don't have
to do it. What we need to is calculate a user2_k so, that
reciprocal(user2_k) == reciprocal(user_k) == kernel_k
i.e. if it's re-loaded back the compiled again value will be exactly
the same as it was. That said, the user2_k can be calculated like this
user2_k = reciprocal(kernel_k)
with an exception, that if kernel_k == 0, then user2_k == 1.
The optlen argument is treated like this -- when zero, kernel returns
the amount of sock_fprog elements in filter, otherwise it should be
large enough for the sock_fprog array.
changes since v1:
* Declared SO_GET_FILTER in all arch headers
* Added decode of vlan-tag codes
Signed-off-by: Pavel Emelyanov <xemul@parallels.com>
Signed-off-by: David S. Miller <davem@davemloft.net>
2012-10-31 22:01:48 -04:00
|
|
|
#define SO_GET_FILTER SO_ATTACH_FILTER
|
2012-10-09 04:47:14 -04:00
|
|
|
|
2013-01-22 06:59:30 -05:00
|
|
|
#define SO_PEERNAME 28
|
2012-10-09 04:47:14 -04:00
|
|
|
|
|
|
|
#define SO_PEERSEC 30
|
|
|
|
#define SO_SNDBUFFORCE 31
|
|
|
|
#define SO_RCVBUFFORCE 33
|
|
|
|
#define SO_PASSSEC 34
|
|
|
|
|
|
|
|
#define SO_MARK 36
|
|
|
|
|
2013-01-22 06:59:30 -05:00
|
|
|
#define SO_RXQ_OVFL 40
|
2012-10-09 04:47:14 -04:00
|
|
|
|
|
|
|
#define SO_WIFI_STATUS 41
|
|
|
|
#define SCM_WIFI_STATUS SO_WIFI_STATUS
|
|
|
|
#define SO_PEEK_OFF 42
|
|
|
|
|
|
|
|
/* Instruct lower device to use last 4-bytes of skb data as FCS */
|
|
|
|
#define SO_NOFCS 43
|
|
|
|
|
2013-01-16 16:55:49 -05:00
|
|
|
#define SO_LOCK_FILTER 44
|
2012-10-09 04:47:14 -04:00
|
|
|
|
2013-03-28 07:19:25 -04:00
|
|
|
#define SO_SELECT_ERR_QUEUE 45
|
|
|
|
|
2013-07-10 10:13:36 -04:00
|
|
|
#define SO_BUSY_POLL 46
|
2013-06-14 09:33:57 -04:00
|
|
|
|
2013-09-24 11:20:52 -04:00
|
|
|
#define SO_MAX_PACING_RATE 47
|
|
|
|
|
2014-01-17 11:09:45 -05:00
|
|
|
#define SO_BPF_EXTENSIONS 48
|
|
|
|
|
net: introduce SO_INCOMING_CPU
Alternative to RPS/RFS is to use hardware support for multiple
queues.
Then split a set of million of sockets into worker threads, each
one using epoll() to manage events on its own socket pool.
Ideally, we want one thread per RX/TX queue/cpu, but we have no way to
know after accept() or connect() on which queue/cpu a socket is managed.
We normally use one cpu per RX queue (IRQ smp_affinity being properly
set), so remembering on socket structure which cpu delivered last packet
is enough to solve the problem.
After accept(), connect(), or even file descriptor passing around
processes, applications can use :
int cpu;
socklen_t len = sizeof(cpu);
getsockopt(fd, SOL_SOCKET, SO_INCOMING_CPU, &cpu, &len);
And use this information to put the socket into the right silo
for optimal performance, as all networking stack should run
on the appropriate cpu, without need to send IPI (RPS/RFS).
Signed-off-by: Eric Dumazet <edumazet@google.com>
Signed-off-by: David S. Miller <davem@davemloft.net>
2014-11-11 08:54:28 -05:00
|
|
|
#define SO_INCOMING_CPU 49
|
|
|
|
|
2014-12-01 18:06:35 -05:00
|
|
|
#define SO_ATTACH_BPF 50
|
|
|
|
#define SO_DETACH_BPF SO_DETACH_FILTER
|
|
|
|
|
2016-01-04 17:41:47 -05:00
|
|
|
#define SO_ATTACH_REUSEPORT_CBPF 51
|
|
|
|
#define SO_ATTACH_REUSEPORT_EBPF 52
|
|
|
|
|
2016-02-24 13:02:52 -05:00
|
|
|
#define SO_CNX_ADVICE 53
|
|
|
|
|
2016-11-28 02:07:18 -05:00
|
|
|
#define SCM_TIMESTAMPING_OPT_STATS 54
|
|
|
|
|
2017-03-20 15:22:03 -04:00
|
|
|
#define SO_MEMINFO 55
|
|
|
|
|
2017-03-24 13:08:36 -04:00
|
|
|
#define SO_INCOMING_NAPI_ID 56
|
2017-03-20 15:22:03 -04:00
|
|
|
|
2017-04-05 22:00:55 -04:00
|
|
|
#define SO_COOKIE 57
|
|
|
|
|
2017-05-21 23:13:37 -04:00
|
|
|
#define SCM_TIMESTAMPING_PKTINFO 58
|
|
|
|
|
net: introduce SO_PEERGROUPS getsockopt
This adds the new getsockopt(2) option SO_PEERGROUPS on SOL_SOCKET to
retrieve the auxiliary groups of the remote peer. It is designed to
naturally extend SO_PEERCRED. That is, the underlying data is from the
same credentials. Regarding its syntax, it is based on SO_PEERSEC. That
is, if the provided buffer is too small, ERANGE is returned and @optlen
is updated. Otherwise, the information is copied, @optlen is set to the
actual size, and 0 is returned.
While SO_PEERCRED (and thus `struct ucred') already returns the primary
group, it lacks the auxiliary group vector. However, nearly all access
controls (including kernel side VFS and SYSVIPC, but also user-space
polkit, DBus, ...) consider the entire set of groups, rather than just
the primary group. But this is currently not possible with pure
SO_PEERCRED. Instead, user-space has to work around this and query the
system database for the auxiliary groups of a UID retrieved via
SO_PEERCRED.
Unfortunately, there is no race-free way to query the auxiliary groups
of the PID/UID retrieved via SO_PEERCRED. Hence, the current user-space
solution is to use getgrouplist(3p), which itself falls back to NSS and
whatever is configured in nsswitch.conf(3). This effectively checks
which groups we *would* assign to the user if it logged in *now*. On
normal systems it is as easy as reading /etc/group, but with NSS it can
resort to quering network databases (eg., LDAP), using IPC or network
communication.
Long story short: Whenever we want to use auxiliary groups for access
checks on IPC, we need further IPC to talk to the user/group databases,
rather than just relying on SO_PEERCRED and the incoming socket. This
is unfortunate, and might even result in dead-locks if the database
query uses the same IPC as the original request.
So far, those recursions / dead-locks have been avoided by using
primitive IPC for all crucial NSS modules. However, we want to avoid
re-inventing the wheel for each NSS module that might be involved in
user/group queries. Hence, we would preferably make DBus (and other IPC
that supports access-management based on groups) work without resorting
to the user/group database. This new SO_PEERGROUPS ioctl would allow us
to make dbus-daemon work without ever calling into NSS.
Cc: Michal Sekletar <msekleta@redhat.com>
Cc: Simon McVittie <simon.mcvittie@collabora.co.uk>
Reviewed-by: Tom Gundersen <teg@jklm.no>
Signed-off-by: David Herrmann <dh.herrmann@gmail.com>
Signed-off-by: David S. Miller <davem@davemloft.net>
2017-06-21 04:47:15 -04:00
|
|
|
#define SO_PEERGROUPS 59
|
|
|
|
|
2017-08-03 16:29:40 -04:00
|
|
|
#define SO_ZEROCOPY 60
|
|
|
|
|
2018-07-03 18:42:48 -04:00
|
|
|
#define SO_TXTIME 61
|
|
|
|
#define SCM_TXTIME SO_TXTIME
|
|
|
|
|
net: introduce SO_BINDTOIFINDEX sockopt
This introduces a new generic SOL_SOCKET-level socket option called
SO_BINDTOIFINDEX. It behaves similar to SO_BINDTODEVICE, but takes a
network interface index as argument, rather than the network interface
name.
User-space often refers to network-interfaces via their index, but has
to temporarily resolve it to a name for a call into SO_BINDTODEVICE.
This might pose problems when the network-device is renamed
asynchronously by other parts of the system. When this happens, the
SO_BINDTODEVICE might either fail, or worse, it might bind to the wrong
device.
In most cases user-space only ever operates on devices which they
either manage themselves, or otherwise have a guarantee that the device
name will not change (e.g., devices that are UP cannot be renamed).
However, particularly in libraries this guarantee is non-obvious and it
would be nice if that race-condition would simply not exist. It would
make it easier for those libraries to operate even in situations where
the device-name might change under the hood.
A real use-case that we recently hit is trying to start the network
stack early in the initrd but make it survive into the real system.
Existing distributions rename network-interfaces during the transition
from initrd into the real system. This, obviously, cannot affect
devices that are up and running (unless you also consider moving them
between network-namespaces). However, the network manager now has to
make sure its management engine for dormant devices will not run in
parallel to these renames. Particularly, when you offload operations
like DHCP into separate processes, these might setup their sockets
early, and thus have to resolve the device-name possibly running into
this race-condition.
By avoiding a call to resolve the device-name, we no longer depend on
the name and can run network setup of dormant devices in parallel to
the transition off the initrd. The SO_BINDTOIFINDEX ioctl plugs this
race.
Reviewed-by: Tom Gundersen <teg@jklm.no>
Signed-off-by: David Herrmann <dh.herrmann@gmail.com>
Acked-by: Willem de Bruijn <willemb@google.com>
Signed-off-by: David S. Miller <davem@davemloft.net>
2019-01-15 08:42:14 -05:00
|
|
|
#define SO_BINDTOIFINDEX 62
|
|
|
|
|
2019-02-02 10:34:46 -05:00
|
|
|
#define SO_TIMESTAMP_OLD 29
|
|
|
|
#define SO_TIMESTAMPNS_OLD 35
|
|
|
|
#define SO_TIMESTAMPING_OLD 37
|
|
|
|
|
2019-02-02 10:34:50 -05:00
|
|
|
#define SO_TIMESTAMP_NEW 63
|
|
|
|
#define SO_TIMESTAMPNS_NEW 64
|
2019-02-02 10:34:51 -05:00
|
|
|
#define SO_TIMESTAMPING_NEW 65
|
2019-02-02 10:34:50 -05:00
|
|
|
|
2019-02-02 10:34:54 -05:00
|
|
|
#define SO_RCVTIMEO_NEW 66
|
|
|
|
#define SO_SNDTIMEO_NEW 67
|
|
|
|
|
2019-06-13 18:00:01 -04:00
|
|
|
#define SO_DETACH_REUSEPORT_BPF 68
|
|
|
|
|
2019-02-02 10:34:46 -05:00
|
|
|
#if !defined(__KERNEL__)
|
|
|
|
|
2019-02-02 10:34:50 -05:00
|
|
|
#if __BITS_PER_LONG == 64
|
|
|
|
#define SO_TIMESTAMP SO_TIMESTAMP_OLD
|
|
|
|
#define SO_TIMESTAMPNS SO_TIMESTAMPNS_OLD
|
2019-02-02 10:34:51 -05:00
|
|
|
#define SO_TIMESTAMPING SO_TIMESTAMPING_OLD
|
2019-02-02 10:34:54 -05:00
|
|
|
|
|
|
|
#define SO_RCVTIMEO SO_RCVTIMEO_OLD
|
|
|
|
#define SO_SNDTIMEO SO_SNDTIMEO_OLD
|
2019-02-02 10:34:50 -05:00
|
|
|
#else
|
|
|
|
#define SO_TIMESTAMP (sizeof(time_t) == sizeof(__kernel_long_t) ? SO_TIMESTAMP_OLD : SO_TIMESTAMP_NEW)
|
|
|
|
#define SO_TIMESTAMPNS (sizeof(time_t) == sizeof(__kernel_long_t) ? SO_TIMESTAMPNS_OLD : SO_TIMESTAMPNS_NEW)
|
2019-02-02 10:34:51 -05:00
|
|
|
#define SO_TIMESTAMPING (sizeof(time_t) == sizeof(__kernel_long_t) ? SO_TIMESTAMPING_OLD : SO_TIMESTAMPING_NEW)
|
2019-02-02 10:34:54 -05:00
|
|
|
|
|
|
|
#define SO_RCVTIMEO (sizeof(time_t) == sizeof(__kernel_long_t) ? SO_RCVTIMEO_OLD : SO_RCVTIMEO_NEW)
|
|
|
|
#define SO_SNDTIMEO (sizeof(time_t) == sizeof(__kernel_long_t) ? SO_SNDTIMEO_OLD : SO_SNDTIMEO_NEW)
|
2019-02-02 10:34:50 -05:00
|
|
|
#endif
|
|
|
|
|
2019-02-02 10:34:46 -05:00
|
|
|
#define SCM_TIMESTAMP SO_TIMESTAMP
|
|
|
|
#define SCM_TIMESTAMPNS SO_TIMESTAMPNS
|
|
|
|
#define SCM_TIMESTAMPING SO_TIMESTAMPING
|
|
|
|
|
|
|
|
#endif
|
|
|
|
|
2012-10-09 04:47:14 -04:00
|
|
|
#endif /* _UAPI_ASM_SOCKET_H */
|