inet_sock.h 12 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453
  1. /* SPDX-License-Identifier: GPL-2.0-or-later */
  2. /*
  3. * INET An implementation of the TCP/IP protocol suite for the LINUX
  4. * operating system. INET is implemented using the BSD Socket
  5. * interface as the means of communication with the user level.
  6. *
  7. * Definitions for inet_sock
  8. *
  9. * Authors: Many, reorganised here by
  10. * Arnaldo Carvalho de Melo <acme@mandriva.com>
  11. */
  12. #ifndef _INET_SOCK_H
  13. #define _INET_SOCK_H
  14. #include <linux/bitops.h>
  15. #include <linux/string.h>
  16. #include <linux/types.h>
  17. #include <linux/jhash.h>
  18. #include <linux/netdevice.h>
  19. #include <net/flow.h>
  20. #include <net/inet_dscp.h>
  21. #include <net/sock.h>
  22. #include <net/request_sock.h>
  23. #include <net/netns/hash.h>
  24. #include <net/tcp_states.h>
  25. #include <net/l3mdev.h>
  26. #define IP_OPTIONS_DATA_FIXED_SIZE 40
  27. /** struct ip_options - IP Options
  28. *
  29. * @faddr - Saved first hop address
  30. * @nexthop - Saved nexthop address in LSRR and SSRR
  31. * @is_strictroute - Strict source route
  32. * @srr_is_hit - Packet destination addr was our one
  33. * @is_changed - IP checksum more not valid
  34. * @rr_needaddr - Need to record addr of outgoing dev
  35. * @ts_needtime - Need to record timestamp
  36. * @ts_needaddr - Need to record addr of outgoing dev
  37. */
  38. struct ip_options {
  39. __be32 faddr;
  40. __be32 nexthop;
  41. unsigned char optlen;
  42. unsigned char srr;
  43. unsigned char rr;
  44. unsigned char ts;
  45. unsigned char is_strictroute:1,
  46. srr_is_hit:1,
  47. is_changed:1,
  48. rr_needaddr:1,
  49. ts_needtime:1,
  50. ts_needaddr:1;
  51. unsigned char router_alert;
  52. unsigned char cipso;
  53. unsigned char __pad2;
  54. unsigned char __data[];
  55. };
  56. struct ip_options_rcu {
  57. struct rcu_head rcu;
  58. /* Must be last as it ends in a flexible-array member. */
  59. struct ip_options opt;
  60. };
  61. struct inet_request_sock {
  62. struct request_sock req;
  63. #define ir_loc_addr req.__req_common.skc_rcv_saddr
  64. #define ir_rmt_addr req.__req_common.skc_daddr
  65. #define ir_num req.__req_common.skc_num
  66. #define ir_rmt_port req.__req_common.skc_dport
  67. #define ir_v6_rmt_addr req.__req_common.skc_v6_daddr
  68. #define ir_v6_loc_addr req.__req_common.skc_v6_rcv_saddr
  69. #define ir_iif req.__req_common.skc_bound_dev_if
  70. #define ir_cookie req.__req_common.skc_cookie
  71. #define ireq_net req.__req_common.skc_net
  72. #define ireq_state req.__req_common.skc_state
  73. #define ireq_family req.__req_common.skc_family
  74. u16 snd_wscale : 4,
  75. rcv_wscale : 4,
  76. tstamp_ok : 1,
  77. sack_ok : 1,
  78. wscale_ok : 1,
  79. ecn_ok : 1,
  80. acked : 1,
  81. no_srccheck: 1,
  82. smc_ok : 1;
  83. u32 ir_mark;
  84. union {
  85. struct ip_options_rcu __rcu *ireq_opt;
  86. #if IS_ENABLED(CONFIG_IPV6)
  87. struct {
  88. struct ipv6_txoptions *ipv6_opt;
  89. struct sk_buff *pktopts;
  90. };
  91. #endif
  92. };
  93. };
  94. #define inet_rsk(ptr) container_of_const(ptr, struct inet_request_sock, req)
  95. static inline u32 inet_request_mark(const struct sock *sk, struct sk_buff *skb)
  96. {
  97. u32 mark = READ_ONCE(sk->sk_mark);
  98. if (!mark && READ_ONCE(sock_net(sk)->ipv4.sysctl_tcp_fwmark_accept))
  99. return skb->mark;
  100. return mark;
  101. }
  102. static inline int inet_request_bound_dev_if(const struct sock *sk,
  103. struct sk_buff *skb)
  104. {
  105. int bound_dev_if = READ_ONCE(sk->sk_bound_dev_if);
  106. #ifdef CONFIG_NET_L3_MASTER_DEV
  107. struct net *net = sock_net(sk);
  108. if (!bound_dev_if && READ_ONCE(net->ipv4.sysctl_tcp_l3mdev_accept))
  109. return l3mdev_master_ifindex_by_index(net, skb->skb_iif);
  110. #endif
  111. return bound_dev_if;
  112. }
  113. static inline int inet_sk_bound_l3mdev(const struct sock *sk)
  114. {
  115. #ifdef CONFIG_NET_L3_MASTER_DEV
  116. struct net *net = sock_net(sk);
  117. if (!READ_ONCE(net->ipv4.sysctl_tcp_l3mdev_accept))
  118. return l3mdev_master_ifindex_by_index(net,
  119. sk->sk_bound_dev_if);
  120. #endif
  121. return 0;
  122. }
  123. static inline bool inet_bound_dev_eq(bool l3mdev_accept, int bound_dev_if,
  124. int dif, int sdif)
  125. {
  126. if (!bound_dev_if)
  127. return !sdif || l3mdev_accept;
  128. return bound_dev_if == dif || bound_dev_if == sdif;
  129. }
  130. static inline bool inet_sk_bound_dev_eq(const struct net *net,
  131. int bound_dev_if,
  132. int dif, int sdif)
  133. {
  134. #if IS_ENABLED(CONFIG_NET_L3_MASTER_DEV)
  135. return inet_bound_dev_eq(!!READ_ONCE(net->ipv4.sysctl_tcp_l3mdev_accept),
  136. bound_dev_if, dif, sdif);
  137. #else
  138. return inet_bound_dev_eq(true, bound_dev_if, dif, sdif);
  139. #endif
  140. }
  141. struct inet6_cork {
  142. struct ipv6_txoptions *opt;
  143. u8 hop_limit;
  144. u8 tclass;
  145. u8 dontfrag:1;
  146. };
  147. struct inet_cork {
  148. unsigned int flags;
  149. __be32 addr;
  150. struct ip_options *opt;
  151. unsigned int fragsize;
  152. int length; /* Total length of all frames */
  153. struct dst_entry *dst;
  154. u8 tx_flags;
  155. __u8 ttl;
  156. __s16 tos;
  157. u32 priority;
  158. __u16 gso_size;
  159. u32 ts_opt_id;
  160. u64 transmit_time;
  161. u32 mark;
  162. };
  163. struct inet_cork_full {
  164. struct inet_cork base;
  165. struct flowi fl;
  166. #if IS_ENABLED(CONFIG_IPV6)
  167. struct inet6_cork base6;
  168. #endif
  169. };
  170. struct ip_mc_socklist;
  171. struct ipv6_pinfo;
  172. struct rtable;
  173. /** struct inet_sock - representation of INET sockets
  174. *
  175. * @sk - ancestor class
  176. * @pinet6 - pointer to IPv6 control block
  177. * @inet_daddr - Foreign IPv4 addr
  178. * @inet_rcv_saddr - Bound local IPv4 addr
  179. * @inet_dport - Destination port
  180. * @inet_num - Local port
  181. * @inet_flags - various atomic flags
  182. * @inet_saddr - Sending source
  183. * @uc_ttl - Unicast TTL
  184. * @inet_sport - Source port
  185. * @inet_id - ID counter for DF pkts
  186. * @tos - TOS
  187. * @mc_ttl - Multicasting TTL
  188. * @uc_index - Unicast outgoing device index
  189. * @mc_index - Multicast device index
  190. * @mc_list - Group array
  191. * @cork - info to build ip hdr on each ip frag while socket is corked
  192. */
  193. struct inet_sock {
  194. /* sk and pinet6 has to be the first two members of inet_sock */
  195. struct sock sk;
  196. #if IS_ENABLED(CONFIG_IPV6)
  197. struct ipv6_pinfo *pinet6;
  198. struct ipv6_fl_socklist __rcu *ipv6_fl_list;
  199. #endif
  200. /* Socket demultiplex comparisons on incoming packets. */
  201. #define inet_daddr sk.__sk_common.skc_daddr
  202. #define inet_rcv_saddr sk.__sk_common.skc_rcv_saddr
  203. #define inet_dport sk.__sk_common.skc_dport
  204. #define inet_num sk.__sk_common.skc_num
  205. unsigned long inet_flags;
  206. __be32 inet_saddr;
  207. __s16 uc_ttl;
  208. __be16 inet_sport;
  209. struct ip_options_rcu __rcu *inet_opt;
  210. atomic_t inet_id;
  211. __u8 tos;
  212. __u8 min_ttl;
  213. __u8 mc_ttl;
  214. __u8 pmtudisc;
  215. __u8 rcv_tos;
  216. __u8 convert_csum;
  217. int uc_index;
  218. int mc_index;
  219. __be32 mc_addr;
  220. u32 local_port_range; /* high << 16 | low */
  221. struct ip_mc_socklist __rcu *mc_list;
  222. struct inet_cork_full cork;
  223. };
  224. #define IPCORK_OPT 1 /* ip-options has been held in ipcork.opt */
  225. #define IPCORK_TS_OPT_ID 2 /* ts_opt_id field is valid, overriding sk_tskey */
  226. enum {
  227. INET_FLAGS_PKTINFO = 0,
  228. INET_FLAGS_TTL = 1,
  229. INET_FLAGS_TOS = 2,
  230. INET_FLAGS_RECVOPTS = 3,
  231. INET_FLAGS_RETOPTS = 4,
  232. INET_FLAGS_PASSSEC = 5,
  233. INET_FLAGS_ORIGDSTADDR = 6,
  234. INET_FLAGS_CHECKSUM = 7,
  235. INET_FLAGS_RECVFRAGSIZE = 8,
  236. INET_FLAGS_RECVERR = 9,
  237. INET_FLAGS_RECVERR_RFC4884 = 10,
  238. INET_FLAGS_FREEBIND = 11,
  239. INET_FLAGS_HDRINCL = 12,
  240. INET_FLAGS_MC_LOOP = 13,
  241. INET_FLAGS_MC_ALL = 14,
  242. INET_FLAGS_TRANSPARENT = 15,
  243. INET_FLAGS_IS_ICSK = 16,
  244. INET_FLAGS_NODEFRAG = 17,
  245. INET_FLAGS_BIND_ADDRESS_NO_PORT = 18,
  246. INET_FLAGS_DEFER_CONNECT = 19,
  247. INET_FLAGS_MC6_LOOP = 20,
  248. INET_FLAGS_RECVERR6_RFC4884 = 21,
  249. INET_FLAGS_MC6_ALL = 22,
  250. INET_FLAGS_AUTOFLOWLABEL_SET = 23,
  251. INET_FLAGS_AUTOFLOWLABEL = 24,
  252. INET_FLAGS_DONTFRAG = 25,
  253. INET_FLAGS_RECVERR6 = 26,
  254. INET_FLAGS_REPFLOW = 27,
  255. INET_FLAGS_RTALERT_ISOLATE = 28,
  256. INET_FLAGS_SNDFLOW = 29,
  257. INET_FLAGS_RTALERT = 30,
  258. };
  259. /* cmsg flags for inet */
  260. #define IP_CMSG_PKTINFO BIT(INET_FLAGS_PKTINFO)
  261. #define IP_CMSG_TTL BIT(INET_FLAGS_TTL)
  262. #define IP_CMSG_TOS BIT(INET_FLAGS_TOS)
  263. #define IP_CMSG_RECVOPTS BIT(INET_FLAGS_RECVOPTS)
  264. #define IP_CMSG_RETOPTS BIT(INET_FLAGS_RETOPTS)
  265. #define IP_CMSG_PASSSEC BIT(INET_FLAGS_PASSSEC)
  266. #define IP_CMSG_ORIGDSTADDR BIT(INET_FLAGS_ORIGDSTADDR)
  267. #define IP_CMSG_CHECKSUM BIT(INET_FLAGS_CHECKSUM)
  268. #define IP_CMSG_RECVFRAGSIZE BIT(INET_FLAGS_RECVFRAGSIZE)
  269. #define IP_CMSG_ALL (IP_CMSG_PKTINFO | IP_CMSG_TTL | \
  270. IP_CMSG_TOS | IP_CMSG_RECVOPTS | \
  271. IP_CMSG_RETOPTS | IP_CMSG_PASSSEC | \
  272. IP_CMSG_ORIGDSTADDR | IP_CMSG_CHECKSUM | \
  273. IP_CMSG_RECVFRAGSIZE)
  274. static inline unsigned long inet_cmsg_flags(const struct inet_sock *inet)
  275. {
  276. return READ_ONCE(inet->inet_flags) & IP_CMSG_ALL;
  277. }
  278. static inline dscp_t inet_sk_dscp(const struct inet_sock *inet)
  279. {
  280. return inet_dsfield_to_dscp(READ_ONCE(inet->tos));
  281. }
  282. #define inet_test_bit(nr, sk) \
  283. test_bit(INET_FLAGS_##nr, &inet_sk(sk)->inet_flags)
  284. #define inet_set_bit(nr, sk) \
  285. set_bit(INET_FLAGS_##nr, &inet_sk(sk)->inet_flags)
  286. #define inet_clear_bit(nr, sk) \
  287. clear_bit(INET_FLAGS_##nr, &inet_sk(sk)->inet_flags)
  288. #define inet_assign_bit(nr, sk, val) \
  289. assign_bit(INET_FLAGS_##nr, &inet_sk(sk)->inet_flags, val)
  290. /**
  291. * sk_to_full_sk - Access to a full socket
  292. * @sk: pointer to a socket
  293. *
  294. * SYNACK messages might be attached to request sockets.
  295. * Some places want to reach the listener in this case.
  296. */
  297. static inline struct sock *sk_to_full_sk(struct sock *sk)
  298. {
  299. #ifdef CONFIG_INET
  300. if (sk && READ_ONCE(sk->sk_state) == TCP_NEW_SYN_RECV)
  301. sk = inet_reqsk(sk)->rsk_listener;
  302. if (sk && READ_ONCE(sk->sk_state) == TCP_TIME_WAIT)
  303. sk = NULL;
  304. #endif
  305. return sk;
  306. }
  307. /* sk_to_full_sk() variant with a const argument */
  308. static inline const struct sock *sk_const_to_full_sk(const struct sock *sk)
  309. {
  310. #ifdef CONFIG_INET
  311. if (sk && READ_ONCE(sk->sk_state) == TCP_NEW_SYN_RECV)
  312. sk = ((const struct request_sock *)sk)->rsk_listener;
  313. if (sk && READ_ONCE(sk->sk_state) == TCP_TIME_WAIT)
  314. sk = NULL;
  315. #endif
  316. return sk;
  317. }
  318. static inline struct sock *skb_to_full_sk(const struct sk_buff *skb)
  319. {
  320. return sk_to_full_sk(skb->sk);
  321. }
  322. #define inet_sk(ptr) container_of_const(ptr, struct inet_sock, sk)
  323. int inet_sk_rebuild_header(struct sock *sk);
  324. /**
  325. * inet_sk_state_load - read sk->sk_state for lockless contexts
  326. * @sk: socket pointer
  327. *
  328. * Paired with inet_sk_state_store(). Used in places we don't hold socket lock:
  329. * tcp_diag_get_info(), tcp_get_info(), tcp_poll(), get_tcp4_sock() ...
  330. */
  331. static inline int inet_sk_state_load(const struct sock *sk)
  332. {
  333. /* state change might impact lockless readers. */
  334. return smp_load_acquire(&sk->sk_state);
  335. }
  336. /**
  337. * inet_sk_state_store - update sk->sk_state
  338. * @sk: socket pointer
  339. * @newstate: new state
  340. *
  341. * Paired with inet_sk_state_load(). Should be used in contexts where
  342. * state change might impact lockless readers.
  343. */
  344. void inet_sk_state_store(struct sock *sk, int newstate);
  345. void inet_sk_set_state(struct sock *sk, int state);
  346. static inline unsigned int __inet_ehashfn(const __be32 laddr,
  347. const __u16 lport,
  348. const __be32 faddr,
  349. const __be16 fport,
  350. u32 initval)
  351. {
  352. return jhash_3words((__force __u32) laddr,
  353. (__force __u32) faddr,
  354. ((__u32) lport) << 16 | (__force __u32)fport,
  355. initval);
  356. }
  357. struct request_sock *inet_reqsk_alloc(const struct request_sock_ops *ops,
  358. struct sock *sk_listener,
  359. bool attach_listener);
  360. static inline __u8 inet_sk_flowi_flags(const struct sock *sk)
  361. {
  362. __u8 flags = 0;
  363. if (inet_test_bit(TRANSPARENT, sk) || inet_test_bit(HDRINCL, sk))
  364. flags |= FLOWI_FLAG_ANYSRC;
  365. return flags;
  366. }
  367. static inline void inet_inc_convert_csum(struct sock *sk)
  368. {
  369. inet_sk(sk)->convert_csum++;
  370. }
  371. static inline void inet_dec_convert_csum(struct sock *sk)
  372. {
  373. if (inet_sk(sk)->convert_csum > 0)
  374. inet_sk(sk)->convert_csum--;
  375. }
  376. static inline bool inet_get_convert_csum(struct sock *sk)
  377. {
  378. return !!inet_sk(sk)->convert_csum;
  379. }
  380. static inline bool inet_can_nonlocal_bind(struct net *net,
  381. struct inet_sock *inet)
  382. {
  383. return READ_ONCE(net->ipv4.sysctl_ip_nonlocal_bind) ||
  384. test_bit(INET_FLAGS_FREEBIND, &inet->inet_flags) ||
  385. test_bit(INET_FLAGS_TRANSPARENT, &inet->inet_flags);
  386. }
  387. static inline bool inet_addr_valid_or_nonlocal(struct net *net,
  388. struct inet_sock *inet,
  389. __be32 addr,
  390. int addr_type)
  391. {
  392. return inet_can_nonlocal_bind(net, inet) ||
  393. addr == htonl(INADDR_ANY) ||
  394. addr_type == RTN_LOCAL ||
  395. addr_type == RTN_MULTICAST ||
  396. addr_type == RTN_BROADCAST;
  397. }
  398. #endif /* _INET_SOCK_H */